feat: render Markdown, mathematics and Mermaid diagrams server-side
This commit is contained in:
@@ -0,0 +1,222 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "bytes"
|
||||
|
||||
// The GFM extended autolinks: bare www., http://, https:// and ftp://
|
||||
// addresses, and bare email addresses, recognised without angle brackets.
|
||||
// The rules follow the GFM specification literally.
|
||||
|
||||
// isAutolinkStartByte reports whether the previous character allows an
|
||||
// extended www or URL autolink to begin here: beginning of the inline
|
||||
// source, whitespace, or one of the delimiting characters.
|
||||
func isAutolinkStartByte(prev byte) bool {
|
||||
return prev == 0 || isSpaceTab(prev) || prev == '\n' ||
|
||||
prev == '*' || prev == '_' || prev == '~' || prev == '('
|
||||
}
|
||||
|
||||
// isEmailLocalByte reports whether c may appear in the local part of an
|
||||
// extended email autolink.
|
||||
func isEmailLocalByte(c byte) bool {
|
||||
return isAlnum(c) || c == '.' || c == '-' || c == '_' || c == '+'
|
||||
}
|
||||
|
||||
// isEmailDomainByte reports whether c may appear in a domain segment of an
|
||||
// extended email autolink.
|
||||
func isEmailDomainByte(c byte) bool {
|
||||
return isAlnum(c) || c == '-' || c == '_'
|
||||
}
|
||||
|
||||
// isEmailPrevByte reports whether the character before a candidate would
|
||||
// continue a longer email address, which disqualifies the candidate: the
|
||||
// local part must be maximal.
|
||||
func isEmailPrevByte(prev byte) bool {
|
||||
return isAlnum(prev) || prev == '.' || prev == '-' || prev == '_' ||
|
||||
prev == '+' || prev == '@'
|
||||
}
|
||||
|
||||
// scanExtendedAutolink recognises an extended autolink at i, returning
|
||||
// the length, the link text and the destination. The prev byte is the
|
||||
// character before i; a www or URL candidate may only begin where the
|
||||
// GFM spec allows it, and an email candidate must start a maximal local
|
||||
// part.
|
||||
func scanExtendedAutolink(src []byte, i int, prev byte) (n int, text, dest string, ok bool) {
|
||||
if isAutolinkStartByte(prev) {
|
||||
if n, end, ok := scanWWWOrURL(src, i); ok {
|
||||
text := string(src[i:end])
|
||||
dest := text
|
||||
if src[i] == 'w' {
|
||||
dest = "http://" + text
|
||||
}
|
||||
return n, text, dest, true
|
||||
}
|
||||
}
|
||||
if !isEmailPrevByte(prev) {
|
||||
if n, end, ok := scanEmailAutolink(src, i); ok {
|
||||
addr := string(src[i:end])
|
||||
return n, addr, "mailto:" + addr, true
|
||||
}
|
||||
}
|
||||
return 0, "", "", false
|
||||
}
|
||||
|
||||
// scanWWWOrURL recognises a bare www address or one with an explicit
|
||||
// http, https or ftp scheme. It returns the length and the end offset.
|
||||
func scanWWWOrURL(src []byte, i int) (int, int, bool) {
|
||||
var schemeLen int
|
||||
switch {
|
||||
case bytes.HasPrefix(src[i:], []byte("www.")):
|
||||
schemeLen = 4
|
||||
case bytes.HasPrefix(src[i:], []byte("http://")):
|
||||
schemeLen = 7
|
||||
case bytes.HasPrefix(src[i:], []byte("https://")):
|
||||
schemeLen = 8
|
||||
case bytes.HasPrefix(src[i:], []byte("ftp://")):
|
||||
schemeLen = 6
|
||||
default:
|
||||
return 0, 0, false
|
||||
}
|
||||
domainEnd, ok := scanValidDomain(src, i+schemeLen)
|
||||
if !ok {
|
||||
return 0, 0, false
|
||||
}
|
||||
// Zero or more non-space, non-< characters follow the domain.
|
||||
end := domainEnd
|
||||
for end < len(src) && src[end] != ' ' && src[end] != '\t' &&
|
||||
src[end] != '\n' && src[end] != '<' {
|
||||
end++
|
||||
}
|
||||
end = validateAutolinkPath(src, i, end)
|
||||
if end <= i+schemeLen {
|
||||
return 0, 0, false
|
||||
}
|
||||
return end - i, end, true
|
||||
}
|
||||
|
||||
// scanValidDomain reads a GFM valid domain at i: one or more segments of
|
||||
// alphanumerics, underscores and hyphens separated by periods, with at
|
||||
// least one period, and no underscore in the last two segments. It
|
||||
// returns the offset after the domain.
|
||||
func scanValidDomain(src []byte, i int) (int, bool) {
|
||||
var starts, ends []int
|
||||
j := i
|
||||
for {
|
||||
start := j
|
||||
for j < len(src) && (isAlnum(src[j]) || src[j] == '-' || src[j] == '_') {
|
||||
j++
|
||||
}
|
||||
if j == start {
|
||||
return 0, false
|
||||
}
|
||||
starts = append(starts, start)
|
||||
ends = append(ends, j)
|
||||
// A period only continues the domain when a segment follows it;
|
||||
// a trailing period belongs to whatever comes after.
|
||||
if j+1 < len(src) && src[j] == '.' &&
|
||||
(isAlnum(src[j+1]) || src[j+1] == '-' || src[j+1] == '_') {
|
||||
j++
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
if len(ends) < 2 {
|
||||
return 0, false
|
||||
}
|
||||
for k := len(ends) - 2; k < len(ends); k++ {
|
||||
for c := starts[k]; c < ends[k]; c++ {
|
||||
if src[c] == '_' {
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
}
|
||||
return j, true
|
||||
}
|
||||
|
||||
// validateAutolinkPath applies the extended autolink path validation to
|
||||
// the candidate src[begin:end]: trailing punctuation is trimmed, an
|
||||
// unbalanced closing parenthesis is trimmed, and a trailing entity
|
||||
// reference is excluded.
|
||||
func validateAutolinkPath(src []byte, begin, end int) int {
|
||||
for end > begin {
|
||||
switch src[end-1] {
|
||||
case '?', '!', '.', ',', ':', '*', '_', '~':
|
||||
end--
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
for end > begin && src[end-1] == ')' {
|
||||
opens, closes := 0, 0
|
||||
for k := begin; k < end; k++ {
|
||||
switch src[k] {
|
||||
case '(':
|
||||
opens++
|
||||
case ')':
|
||||
closes++
|
||||
}
|
||||
}
|
||||
if closes <= opens {
|
||||
break
|
||||
}
|
||||
end--
|
||||
}
|
||||
if end > begin && src[end-1] == ';' {
|
||||
if amp := bytes.LastIndexByte(src[begin:end], '&'); amp >= 0 {
|
||||
entity := src[begin+amp+1 : end-1]
|
||||
valid := len(entity) > 0
|
||||
for _, c := range entity {
|
||||
if !isAlnum(c) {
|
||||
valid = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if valid {
|
||||
end = begin + amp
|
||||
}
|
||||
}
|
||||
}
|
||||
return end
|
||||
}
|
||||
|
||||
// scanEmailAutolink recognises an extended email autolink at i: a local
|
||||
// part of alphanumerics and .-_+, an @, and a domain of alphanumerics,
|
||||
// hyphens and underscores separated by at least one period, whose last
|
||||
// character is not a hyphen or underscore. A trailing period is not part
|
||||
// of the address.
|
||||
func scanEmailAutolink(src []byte, i int) (int, int, bool) {
|
||||
j := i
|
||||
for j < len(src) && isEmailLocalByte(src[j]) {
|
||||
j++
|
||||
}
|
||||
if j == i || j >= len(src) || src[j] != '@' {
|
||||
return 0, 0, false
|
||||
}
|
||||
j++
|
||||
segments := 0
|
||||
for {
|
||||
start := j
|
||||
for j < len(src) && isEmailDomainByte(src[j]) {
|
||||
j++
|
||||
}
|
||||
if j == start {
|
||||
return 0, 0, false
|
||||
}
|
||||
segments++
|
||||
// A period only continues the domain when a segment follows it;
|
||||
// a trailing period is not part of the address.
|
||||
if j+1 < len(src) && src[j] == '.' && isEmailDomainByte(src[j+1]) {
|
||||
j++
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
if segments < 2 {
|
||||
return 0, 0, false
|
||||
}
|
||||
if last := src[j-1]; last == '-' || last == '_' {
|
||||
return 0, 0, false
|
||||
}
|
||||
return j - i, j, true
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "testing"
|
||||
|
||||
// blockCorpus is the project's own hand-written corpus. Every case states
|
||||
// an input and the exact HTML the engine produces for it.
|
||||
var blockCorpus = []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
// ATX headings.
|
||||
{"atx level 1", "# foo", "<h1>foo</h1>\n"},
|
||||
{"atx level 6", "###### foo", "<h6>foo</h6>\n"},
|
||||
{"atx seven hashes", "####### foo", "<p>####### foo</p>\n"},
|
||||
{"atx needs a space", "#foo", "<p>#foo</p>\n"},
|
||||
{"atx closing sequence", "### foo ###", "<h3>foo</h3>\n"},
|
||||
{"atx closing sequence mid content", "### foo ### ###", "<h3>foo ###</h3>\n"},
|
||||
{"atx hash without space stays", "# foo#", "<h1>foo#</h1>\n"},
|
||||
{"atx indented three", " # foo", "<h1>foo</h1>\n"},
|
||||
{"atx indented four is code", " # foo", "<pre><code># foo\n</code></pre>\n"},
|
||||
{"atx bare hash", "#", "<h1></h1>\n"},
|
||||
{"atx empty content", "###", "<h3></h3>\n"},
|
||||
|
||||
// Setext headings.
|
||||
{"setext level 1", "Foo\n===", "<h1>Foo</h1>\n"},
|
||||
{"setext level 2", "Foo\n---", "<h2>Foo</h2>\n"},
|
||||
{"setext multi line", "Foo\nBar\n---", "<h2>Foo\nBar</h2>\n"},
|
||||
{"setext takes whole paragraph", "Foo\nbar\n---", "<h2>Foo\nbar</h2>\n"},
|
||||
{"setext indented underline", "Foo\n ===", "<h1>Foo</h1>\n"},
|
||||
{"setext dashes with spaces stay thematic", "Foo\n- - -", "<p>Foo</p>\n<hr />\n"},
|
||||
{"setext cannot be lazy", "> Foo\n===", "<blockquote>\n<p>Foo\n===</p>\n</blockquote>\n"},
|
||||
{"setext inside quote", "> foo\n> ---", "<blockquote>\n<h2>foo</h2>\n</blockquote>\n"},
|
||||
|
||||
// Thematic breaks.
|
||||
{"thematic stars", "***", "<hr />\n"},
|
||||
{"thematic underscores", "___", "<hr />\n"},
|
||||
{"thematic spaced dashes", "- - -", "<hr />\n"},
|
||||
{"thematic interrupts paragraph", "foo\n***", "<p>foo</p>\n<hr />\n"},
|
||||
{"thematic long", "---------------------------------------", "<hr />\n"},
|
||||
{"thematic trailing spaces", "*** ", "<hr />\n"},
|
||||
{"thematic indented three", " ***", "<hr />\n"},
|
||||
|
||||
// Indented code blocks.
|
||||
{"indented code", " code", "<pre><code>code\n</code></pre>\n"},
|
||||
{"indented code two lines", " code\n more", "<pre><code>code\nmore\n</code></pre>\n"},
|
||||
{"indented code keeps internal blank", " a\n\n b", "<pre><code>a\n\nb\n</code></pre>\n"},
|
||||
{"indented code strips trailing blanks", " a\n\n\n", "<pre><code>a\n</code></pre>\n"},
|
||||
{"tab indents code", "\tcode", "<pre><code>code\n</code></pre>\n"},
|
||||
{"indented code cannot interrupt paragraph", "foo\n bar", "<p>foo\nbar</p>\n"},
|
||||
|
||||
// Fenced code blocks.
|
||||
{"fenced backticks", "```\ncode\n```", "<pre><code>code\n</code></pre>\n"},
|
||||
{"fenced tildes", "~~~\ncode\n~~~", "<pre><code>code\n</code></pre>\n"},
|
||||
{"fenced info class", "```ruby\nx = 1\n```", "<pre><code class=\"language-ruby\">x = 1\n</code></pre>\n"},
|
||||
{"fenced longer close", "````\n```\n````", "<pre><code>```\n</code></pre>\n"},
|
||||
{"fenced short close is content", "````\ncode\n```", "<pre><code>code\n```\n</code></pre>\n"},
|
||||
{"fenced indent stripping", " ```\n foo\nbar\n```", "<pre><code>foo\nbar\n</code></pre>\n"},
|
||||
{"fenced empty", "```\n```", "<pre><code></code></pre>\n"},
|
||||
{"fenced unclosed", "```\ncode", "<pre><code>code\n</code></pre>\n"},
|
||||
{"tilde info may hold backticks", "~~~ js `x`\ncode\n~~~", "<pre><code class=\"language-js\">code\n</code></pre>\n"},
|
||||
{"fenced blank lines kept", "```\na\n\nb\n```", "<pre><code>a\n\nb\n</code></pre>\n"},
|
||||
|
||||
// HTML blocks.
|
||||
{"html type six", "<div>\nfoo\n</div>", "<div>\nfoo\n</div>\n"},
|
||||
{"html comment", "<!-- comment\n-->", "<!-- comment\n-->\n"},
|
||||
{"html processing instruction", "<?php\necho 1;\n?>", "<?php\necho 1;\n?>\n"},
|
||||
{"html declaration ends at bracket", "<!DOCTYPE html>\nfoo", "<!DOCTYPE html>\n<p>foo</p>\n"},
|
||||
{"html cdata", "<![CDATA[\nfoo\n]]>", "<![CDATA[\nfoo\n]]>\n"},
|
||||
{"html script block", "<script>\nvar x = 1;\n</script>", "<script>\nvar x = 1;\n</script>\n"},
|
||||
{"html blank ends six", "<div>\nfoo\n\nbar", "<div>\nfoo\n<p>bar</p>\n"},
|
||||
{"html six interrupts paragraph", "Foo\n<div>", "<p>Foo</p>\n<div>\n"},
|
||||
{"html textarea", "<textarea>\nfoo\n</textarea>", "<textarea>\nfoo\n</textarea>\n"},
|
||||
{"html seven complete tag", "<a href=\"x\">\nfoo", "<a href=\"x\">\nfoo\n"},
|
||||
|
||||
// Block quotes.
|
||||
{"quote basic", "> foo", "<blockquote>\n<p>foo</p>\n</blockquote>\n"},
|
||||
{"quote two lines", "> foo\n> bar", "<blockquote>\n<p>foo\nbar</p>\n</blockquote>\n"},
|
||||
{"quote without space", ">foo", "<blockquote>\n<p>foo</p>\n</blockquote>\n"},
|
||||
{"quote lazy", "> foo\nbar", "<blockquote>\n<p>foo\nbar</p>\n</blockquote>\n"},
|
||||
{"quote blank closes paragraph", "> foo\n\nbar", "<blockquote>\n<p>foo</p>\n</blockquote>\n<p>bar</p>\n"},
|
||||
{"quote empty", ">", "<blockquote>\n</blockquote>\n"},
|
||||
{"quote nested", "> > foo", "<blockquote>\n<blockquote>\n<p>foo</p>\n</blockquote>\n</blockquote>\n"},
|
||||
{"quote heading", "> # foo", "<blockquote>\n<h1>foo</h1>\n</blockquote>\n"},
|
||||
{"quote list", "> - foo", "<blockquote>\n<ul>\n<li>foo</li>\n</ul>\n</blockquote>\n"},
|
||||
{"quote thematic interrupt", "> foo\n---", "<blockquote>\n<p>foo</p>\n</blockquote>\n<hr />\n"},
|
||||
{"quote blank line inside", "> foo\n>\n> bar", "<blockquote>\n<p>foo</p>\n<p>bar</p>\n</blockquote>\n"},
|
||||
{"quote blank line between", "> a\n\n> b", "<blockquote>\n<p>a</p>\n</blockquote>\n<blockquote>\n<p>b</p>\n</blockquote>\n"},
|
||||
{"quote tab content", ">\tfoo", "<blockquote>\n<p>foo</p>\n</blockquote>\n"},
|
||||
|
||||
// Lists.
|
||||
{"list bullet", "- foo", "<ul>\n<li>foo</li>\n</ul>\n"},
|
||||
{"list star", "* foo", "<ul>\n<li>foo</li>\n</ul>\n"},
|
||||
{"list plus", "+ foo", "<ul>\n<li>foo</li>\n</ul>\n"},
|
||||
{"list tight items", "- foo\n- bar", "<ul>\n<li>foo</li>\n<li>bar</li>\n</ul>\n"},
|
||||
{"list loose items", "- foo\n\n- bar", "<ul>\n<li>\n<p>foo</p>\n</li>\n<li>\n<p>bar</p>\n</li>\n</ul>\n"},
|
||||
{"list blank splits items loose", "- foo\n- bar\n\n- baz", "<ul>\n<li>\n<p>foo</p>\n</li>\n<li>\n<p>bar</p>\n</li>\n<li>\n<p>baz</p>\n</li>\n</ul>\n"},
|
||||
{"list bullet change splits", "* foo\n+ bar", "<ul>\n<li>foo</li>\n</ul>\n<ul>\n<li>bar</li>\n</ul>\n"},
|
||||
{"ordered list", "1. foo\n2. bar", "<ol>\n<li>foo</li>\n<li>bar</li>\n</ol>\n"},
|
||||
{"ordered start", "3. foo", "<ol start=\"3\">\n<li>foo</li>\n</ol>\n"},
|
||||
{"ordered paren delimiter", "1) foo", "<ol>\n<li>foo</li>\n</ol>\n"},
|
||||
{"ordered delimiter change splits", "1. foo\n1) bar", "<ol>\n<li>foo</li>\n</ol>\n<ol>\n<li>bar</li>\n</ol>\n"},
|
||||
{"list item two paragraphs loose", "- foo\n\n bar", "<ul>\n<li>\n<p>foo</p>\n<p>bar</p>\n</li>\n</ul>\n"},
|
||||
{"list nested", "- foo\n - bar", "<ul>\n<li>foo\n<ul>\n<li>bar</li>\n</ul>\n</li>\n</ul>\n"},
|
||||
{"list item continuation", "- foo\n bar", "<ul>\n<li>foo\nbar</li>\n</ul>\n"},
|
||||
{"list item lazy", "- foo\nbar", "<ul>\n<li>foo\nbar</li>\n</ul>\n"},
|
||||
{"list lazy carries on", "- foo\n bar\ncar", "<ul>\n<li>foo\nbar\ncar</li>\n</ul>\n"},
|
||||
{"ordered nine digits", "123456789. foo", "<ol start=\"123456789\">\n<li>foo</li>\n</ol>\n"},
|
||||
{"ordered ten digits not a list", "1234567890. foo", "<p>1234567890. foo</p>\n"},
|
||||
{"empty items", "- foo\n-\n- bar", "<ul>\n<li>foo</li>\n<li></li>\n<li>bar</li>\n</ul>\n"},
|
||||
{"list interrupts paragraph", "foo\n- bar", "<p>foo</p>\n<ul>\n<li>bar</li>\n</ul>\n"},
|
||||
{"ordered two cannot interrupt", "foo\n2. bar", "<p>foo\n2. bar</p>\n"},
|
||||
{"single dash setext under paragraph", "foo\n-", "<h2>foo</h2>\n"},
|
||||
{"bullet blank cannot interrupt", "foo\n+", "<p>foo\n+</p>\n"},
|
||||
{"list marker five spaces makes code", "- indented code", "<ul>\n<li>\n<pre><code>indented code\n</code></pre>\n</li>\n</ul>\n"},
|
||||
{"item fence on marker line", "- ```\n foo\n ```", "<ul>\n<li>\n<pre><code>foo\n</code></pre>\n</li>\n</ul>\n"},
|
||||
{"item fence after paragraph", "- foo\n ```\n bar\n ```", "<ul>\n<li>foo\n<pre><code>bar\n</code></pre>\n</li>\n</ul>\n"},
|
||||
{"item paragraph indented content", "1. foo\n bar", "<ol>\n<li>foo\nbar</li>\n</ol>\n"},
|
||||
{"ordered paragraph two lines", "1. A paragraph\n with two lines.", "<ol>\n<li>A paragraph\nwith two lines.</li>\n</ol>\n"},
|
||||
{"second item less indented", "- a\n - b", "<ul>\n<li>a</li>\n<li>b</li>\n</ul>\n"},
|
||||
{"lazy into quoted item", "> - foo\nbar", "<blockquote>\n<ul>\n<li>foo\nbar</li>\n</ul>\n</blockquote>\n"},
|
||||
{"loose ordered with blank", "1. a\n\n 2. b", "<ol>\n<li>\n<p>a</p>\n</li>\n<li>\n<p>b</p>\n</li>\n</ol>\n"},
|
||||
{"list item second paragraph after blank", "- a\n- b\n\n c", "<ul>\n<li>\n<p>a</p>\n</li>\n<li>\n<p>b</p>\n<p>c</p>\n</li>\n</ul>\n"},
|
||||
|
||||
// Paragraphs.
|
||||
{"paragraph single", "foo", "<p>foo</p>\n"},
|
||||
{"paragraph lines", "foo\nbar", "<p>foo\nbar</p>\n"},
|
||||
{"paragraph leading spaces", " foo", "<p>foo</p>\n"},
|
||||
{"paragraph escaping", "a < b & c", "<p>a < b & c</p>\n"},
|
||||
{"paragraph quotes escape", "say \"hi\"", "<p>say "hi"</p>\n"},
|
||||
{"blank lines separate", "foo\n\nbar", "<p>foo</p>\n<p>bar</p>\n"},
|
||||
{"leading blanks ignored", "\n\nfoo", "<p>foo</p>\n"},
|
||||
{"crlf normalised", "foo\r\nbar\r\n", "<p>foo\nbar</p>\n"},
|
||||
|
||||
// Link reference definitions.
|
||||
{"reference definition alone", "[foo]: /url", ""},
|
||||
{"reference definition with title", "[foo]: /url \"title\"", ""},
|
||||
{"reference definition then text", "[foo]: /url\nused", "<p>used</p>\n"},
|
||||
{"reference definition on two lines", "[foo]:\n/url", ""},
|
||||
{"reference definition on three lines", "[foo]:\n/url\n\"title\"", ""},
|
||||
{"junk after destination", "[foo]: /url junk", "<p>[foo]: /url junk</p>\n"},
|
||||
{"two definitions", "[foo]: /a\n[bar]: /b", ""},
|
||||
{"definition indented", " [foo]: /url", ""},
|
||||
|
||||
// Documents.
|
||||
{"empty input", "", ""},
|
||||
{"only blanks", "\n\n\n", ""},
|
||||
{"composite document", "# Title\n\nIntro text here.\n\n- one\n- two\n\n```go\nfmt.Println(1)\n```\n\n> quoted\n",
|
||||
"<h1>Title</h1>\n<p>Intro text here.</p>\n<ul>\n<li>one</li>\n<li>two</li>\n</ul>\n" +
|
||||
"<pre><code class=\"language-go\">fmt.Println(1)\n</code></pre>\n<blockquote>\n<p>quoted</p>\n</blockquote>\n"},
|
||||
}
|
||||
|
||||
func TestBlockCorpus(t *testing.T) {
|
||||
for _, tc := range blockCorpus {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestFootnotes(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
{"basic footnote", "Here is a reference.[^1]\n\n[^1]: Here is the note.",
|
||||
"<p>Here is a reference." +
|
||||
`<sup class="footnote-ref"><a href="#fn-1" id="fnref-1" data-footnote-ref>1</a></sup>` +
|
||||
"</p>\n" +
|
||||
"<section class=\"footnotes\" data-footnotes>\n<ol>\n" +
|
||||
"<li id=\"fn-1\">\n" +
|
||||
"<p>Here is the note." +
|
||||
` <a href="#fnref-1" class="data-footnote-backref" aria-label="Back to reference 1">` + "↩" + `</a>` +
|
||||
"</p>\n" +
|
||||
"</li>\n</ol>\n</section>\n"},
|
||||
{"unreferenced definition renders nothing", "[^1]: the note", ""},
|
||||
{"undefined reference stays text", "Text[^missing]", "<p>Text[^missing]</p>\n"},
|
||||
{"numbered by reference order", "[^b] and [^a]\n\n[^a]: A\n[^b]: B",
|
||||
"<p>" +
|
||||
`<sup class="footnote-ref"><a href="#fn-1" id="fnref-1" data-footnote-ref>1</a></sup>` + " and " +
|
||||
`<sup class="footnote-ref"><a href="#fn-2" id="fnref-2" data-footnote-ref>2</a></sup>` +
|
||||
"</p>\n" +
|
||||
"<section class=\"footnotes\" data-footnotes>\n<ol>\n" +
|
||||
"<li id=\"fn-1\">\n<p>B" +
|
||||
` <a href="#fnref-1" class="data-footnote-backref" aria-label="Back to reference 1">` + "↩" + `</a>` +
|
||||
"</p>\n</li>\n" +
|
||||
"<li id=\"fn-2\">\n<p>A" +
|
||||
` <a href="#fnref-2" class="data-footnote-backref" aria-label="Back to reference 2">` + "↩" + `</a>` +
|
||||
"</p>\n</li>\n" +
|
||||
"</ol>\n</section>\n"},
|
||||
{"repeated reference ids", "[^a] again [^a]\n\n[^a]: A",
|
||||
"<p>" +
|
||||
`<sup class="footnote-ref"><a href="#fn-1" id="fnref-1" data-footnote-ref>1</a></sup>` + " again " +
|
||||
`<sup class="footnote-ref"><a href="#fn-1" id="fnref-1-2" data-footnote-ref>1</a></sup>` +
|
||||
"</p>\n" +
|
||||
"<section class=\"footnotes\" data-footnotes>\n<ol>\n" +
|
||||
"<li id=\"fn-1\">\n<p>A" +
|
||||
` <a href="#fnref-1" class="data-footnote-backref" aria-label="Back to reference 1">` + "↩" + `</a>` +
|
||||
"</p>\n</li>\n</ol>\n</section>\n"},
|
||||
{"multi block definition", "Ref.[^1]\n\n[^1]: first\n\n second",
|
||||
"<p>Ref." +
|
||||
`<sup class="footnote-ref"><a href="#fn-1" id="fnref-1" data-footnote-ref>1</a></sup>` +
|
||||
"</p>\n" +
|
||||
"<section class=\"footnotes\" data-footnotes>\n<ol>\n" +
|
||||
"<li id=\"fn-1\">\n<p>first</p>\n<p>second" +
|
||||
` <a href="#fnref-1" class="data-footnote-backref" aria-label="Back to reference 1">` + "↩" + `</a>` +
|
||||
"</p>\n</li>\n</ol>\n</section>\n"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionLists(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
{"definition basic", "Term\n: Definition", "<dl>\n<dt>Term</dt>\n<dd>Definition</dd>\n</dl>\n"},
|
||||
{"definition two entries", "Term 1\n: Def 1\n\nTerm 2\n: Def 2",
|
||||
"<dl>\n<dt>Term 1</dt>\n<dd>\n<p>Def 1</p>\n</dd>\n<dt>Term 2</dt>\n<dd>\n<p>Def 2</p>\n</dd>\n</dl>\n"},
|
||||
{"definition two definitions", "Term\n: Def a\n: Def b",
|
||||
"<dl>\n<dt>Term</dt>\n<dd>Def a</dd>\n<dd>Def b</dd>\n</dl>\n"},
|
||||
{"definition multiline terms", "Term 1\nTerm 2\n: Def",
|
||||
"<dl>\n<dt>Term 1</dt>\n<dt>Term 2</dt>\n<dd>Def</dd>\n</dl>\n"},
|
||||
{"definition second paragraph", "Term\n: Def\n\n more",
|
||||
"<dl>\n<dt>Term</dt>\n<dd>\n<p>Def</p>\n<p>more</p>\n</dd>\n</dl>\n"},
|
||||
{"definition inline content", "Term\n: A *bold* claim",
|
||||
"<dl>\n<dt>Term</dt>\n<dd>A <em>bold</em> claim</dd>\n</dl>\n"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "testing"
|
||||
|
||||
// gfmCorpus holds the cases of the GitHub Flavored Markdown extensions:
|
||||
// tables, strikethrough and task lists.
|
||||
var gfmCorpus = []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
// Tables, in the multi-line shape the GFM specification prints.
|
||||
{"table basic", "| foo | bar |\n| --- | --- |\n| baz | bim |",
|
||||
"<table>\n<thead>\n<tr>\n<th>foo</th>\n<th>bar</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>baz</td>\n<td>bim</td>\n</tr>\n</tbody>\n</table>\n"},
|
||||
{"table without boundary pipes", "foo | bar\n--- | ---\nbaz | bim",
|
||||
"<table>\n<thead>\n<tr>\n<th>foo</th>\n<th>bar</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>baz</td>\n<td>bim</td>\n</tr>\n</tbody>\n</table>\n"},
|
||||
{"table alignment", "| a | b | c | d |\n| :- | :-: | -: | - |",
|
||||
"<table>\n<thead>\n<tr>\n<th align=\"left\">a</th>\n<th align=\"center\">b</th>\n<th align=\"right\">c</th>\n<th>d</th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"table without body", "| abc | def |\n| --- | --- |",
|
||||
"<table>\n<thead>\n<tr>\n<th>abc</th>\n<th>def</th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"table excess cell ignored", "| a | b |\n| --- | --- |\n| c | d | e |",
|
||||
"<table>\n<thead>\n<tr>\n<th>a</th>\n<th>b</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>c</td>\n<td>d</td>\n</tr>\n</tbody>\n</table>\n"},
|
||||
{"table missing cell empty", "| a | b |\n| --- | --- |\n| c |",
|
||||
"<table>\n<thead>\n<tr>\n<th>a</th>\n<th>b</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>c</td>\n<td></td>\n</tr>\n</tbody>\n</table>\n"},
|
||||
{"table count mismatch stays paragraph", "| a | b |\n| --- |\n| c |",
|
||||
"<p>| a | b |\n| --- |\n| c |</p>\n"},
|
||||
{"table escaped pipe", "| a \\| b | c |\n| --- | --- |",
|
||||
"<table>\n<thead>\n<tr>\n<th>a | b</th>\n<th>c</th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"table cells hold inline", "| *a* | `b` |\n| --- | --- |",
|
||||
"<table>\n<thead>\n<tr>\n<th><em>a</em></th>\n<th><code>b</code></th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"table code span escaped pipe", "| f\\|oo |\n| --- |\n| b `\\|` az |",
|
||||
"<table>\n<thead>\n<tr>\n<th>f|oo</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>b <code>|</code> az</td>\n</tr>\n</tbody>\n</table>\n"},
|
||||
{"table breaks at blank", "| a |\n| --- |\n| b |\n\nparagraph",
|
||||
"<table>\n<thead>\n<tr>\n<th>a</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>b</td>\n</tr>\n</tbody>\n</table>\n<p>paragraph</p>\n"},
|
||||
{"table interrupted by heading", "| a |\n| --- |\n| b |\n# h",
|
||||
"<table>\n<thead>\n<tr>\n<th>a</th>\n</tr>\n</thead>\n<tbody>\n<tr>\n<td>b</td>\n</tr>\n</tbody>\n</table>\n<h1>h</h1>\n"},
|
||||
{"table interrupts paragraph", "a | b\n--- | ---",
|
||||
"<table>\n<thead>\n<tr>\n<th>a</th>\n<th>b</th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"table splits paragraph", "foo\na | b\n--- | ---",
|
||||
"<p>foo</p>\n<table>\n<thead>\n<tr>\n<th>a</th>\n<th>b</th>\n</tr>\n</thead>\n</table>\n"},
|
||||
{"setext without pipes stays", "abc\n---", "<h2>abc</h2>\n"},
|
||||
|
||||
// Strikethrough.
|
||||
{"strikethrough double tilde", "~~foo~~", "<p><del>foo</del></p>\n"},
|
||||
{"strikethrough single tilde", "~foo~", "<p><del>foo</del></p>\n"},
|
||||
{"strikethrough unmatched pair", "~~foo~", "<p>~<del>foo</del></p>\n"},
|
||||
{"strikethrough intraword", "foo~~bar~~baz", "<p>foo<del>bar</del>baz</p>\n"},
|
||||
{"tilde spaced stays literal", "a ~ b", "<p>a ~ b</p>\n"},
|
||||
|
||||
// Task lists.
|
||||
{"task checked", "- [x] foo", "<ul>\n<li><input checked=\"\" disabled=\"\" type=\"checkbox\"> foo</li>\n</ul>\n"},
|
||||
{"task unchecked", "- [ ] foo", "<ul>\n<li><input disabled=\"\" type=\"checkbox\"> foo</li>\n</ul>\n"},
|
||||
{"task uppercase x", "- [X] foo", "<ul>\n<li><input checked=\"\" disabled=\"\" type=\"checkbox\"> foo</li>\n</ul>\n"},
|
||||
{"task needs a space", "- [x]foo", "<ul>\n<li>[x]foo</li>\n</ul>\n"},
|
||||
{"task list of two", "- [x] foo\n- [ ] bar",
|
||||
"<ul>\n<li><input checked=\"\" disabled=\"\" type=\"checkbox\"> foo</li>\n<li><input disabled=\"\" type=\"checkbox\"> bar</li>\n</ul>\n"},
|
||||
{"task nested", "- [x] foo\n - [ ] bar",
|
||||
"<ul>\n<li><input checked=\"\" disabled=\"\" type=\"checkbox\"> foo\n<ul>\n<li><input disabled=\"\" type=\"checkbox\"> bar</li>\n</ul>\n</li>\n</ul>\n"},
|
||||
{"task in loose list", "- [x] foo\n\n- [ ] bar",
|
||||
"<ul>\n<li>\n<p><input checked=\"\" disabled=\"\" type=\"checkbox\"> foo</p>\n</li>\n<li>\n<p><input disabled=\"\" type=\"checkbox\"> bar</p>\n</li>\n</ul>\n"},
|
||||
}
|
||||
|
||||
func TestGFMCorpus(t *testing.T) {
|
||||
for _, tc := range gfmCorpus {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,742 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
type inlineKind uint8
|
||||
|
||||
const (
|
||||
inlText inlineKind = iota
|
||||
inlCode
|
||||
inlRawHTML
|
||||
inlEmph
|
||||
inlStrong
|
||||
inlStrikethrough
|
||||
inlLink
|
||||
inlImage
|
||||
inlBreak
|
||||
inlFootnoteRef
|
||||
)
|
||||
|
||||
// inline is one node of the inline tree. While parsing, the top level
|
||||
// nodes form a doubly-linked chain; when emphasis or a link takes a range,
|
||||
// the range becomes the children slice of the wrapping node.
|
||||
type inline struct {
|
||||
kind inlineKind
|
||||
literal string // text, code content, raw HTML, soft break
|
||||
dest string
|
||||
title string
|
||||
hasTitle bool
|
||||
num int // footnote reference: ordinal
|
||||
occurrence int // footnote reference: which reference to that ordinal
|
||||
children []*inline
|
||||
prev *inline
|
||||
next *inline
|
||||
}
|
||||
|
||||
// footnoteTracker numbers footnote references in document order: a
|
||||
// definition receives its ordinal at its first reference, and later
|
||||
// references to the same definition count their occurrences.
|
||||
type footnoteTracker struct {
|
||||
defs map[string]*Node
|
||||
ordinals map[string]int
|
||||
seen map[string]int
|
||||
order []*Node
|
||||
}
|
||||
|
||||
func newFootnoteTracker(defs []*Node) *footnoteTracker {
|
||||
m := make(map[string]*Node, len(defs))
|
||||
for _, d := range defs {
|
||||
if _, ok := m[d.label]; !ok {
|
||||
m[d.label] = d
|
||||
}
|
||||
}
|
||||
return &footnoteTracker{defs: m, ordinals: map[string]int{}, seen: map[string]int{}}
|
||||
}
|
||||
|
||||
// reference records one reference to the labelled footnote.
|
||||
func (t *footnoteTracker) reference(label string) (ordinal, occurrence int, ok bool) {
|
||||
if _, defined := t.defs[label]; !defined {
|
||||
return 0, 0, false
|
||||
}
|
||||
t.seen[label]++
|
||||
if t.seen[label] == 1 {
|
||||
t.ordinals[label] = len(t.order) + 1
|
||||
t.order = append(t.order, t.defs[label])
|
||||
}
|
||||
return t.ordinals[label], t.seen[label], true
|
||||
}
|
||||
|
||||
// delimiter is one entry of the delimiter stack: a run of asterisks or
|
||||
// underscores waiting to be matched, or an open link or image bracket.
|
||||
type delimiter struct {
|
||||
node *inline
|
||||
char byte
|
||||
isBracket bool
|
||||
image bool
|
||||
active bool
|
||||
length int
|
||||
origLen int
|
||||
canOpen bool
|
||||
canClose bool
|
||||
srcPos int // bracket: index just after the opening literal
|
||||
prev *delimiter
|
||||
next *delimiter
|
||||
}
|
||||
|
||||
// parseInlines parses the inline content of a paragraph or heading against
|
||||
// the document's link reference definitions and footnotes.
|
||||
func parseInlines(content []byte, refs map[string]reference, footnotes *footnoteTracker) []*inline {
|
||||
p := &inlineParser{src: bytes.TrimRight(content, " \t"), refs: refs, footnotes: footnotes}
|
||||
p.parse()
|
||||
return p.chain()
|
||||
}
|
||||
|
||||
type inlineParser struct {
|
||||
src []byte
|
||||
pos int
|
||||
refs map[string]reference
|
||||
footnotes *footnoteTracker
|
||||
|
||||
first *inline
|
||||
last *inline
|
||||
firstDelim *delimiter
|
||||
lastDelim *delimiter
|
||||
}
|
||||
|
||||
func (p *inlineParser) chain() []*inline {
|
||||
var nodes []*inline
|
||||
for n := p.first; n != nil; n = n.next {
|
||||
nodes = append(nodes, n)
|
||||
}
|
||||
return nodes
|
||||
}
|
||||
|
||||
func (p *inlineParser) push(n *inline) *inline {
|
||||
n.prev = p.last
|
||||
n.next = nil
|
||||
if p.last != nil {
|
||||
p.last.next = n
|
||||
} else {
|
||||
p.first = n
|
||||
}
|
||||
p.last = n
|
||||
return n
|
||||
}
|
||||
|
||||
func (p *inlineParser) removeNode(n *inline) {
|
||||
if n.prev != nil {
|
||||
n.prev.next = n.next
|
||||
} else {
|
||||
p.first = n.next
|
||||
}
|
||||
if n.next != nil {
|
||||
n.next.prev = n.prev
|
||||
} else {
|
||||
p.last = n.prev
|
||||
}
|
||||
}
|
||||
|
||||
func (p *inlineParser) pushDelim(d *delimiter) {
|
||||
d.prev = p.lastDelim
|
||||
d.next = nil
|
||||
if p.lastDelim != nil {
|
||||
p.lastDelim.next = d
|
||||
} else {
|
||||
p.firstDelim = d
|
||||
}
|
||||
p.lastDelim = d
|
||||
}
|
||||
|
||||
func (p *inlineParser) removeDelim(d *delimiter) {
|
||||
if d.prev != nil {
|
||||
d.prev.next = d.next
|
||||
} else {
|
||||
p.firstDelim = d.next
|
||||
}
|
||||
if d.next != nil {
|
||||
d.next.prev = d.prev
|
||||
} else {
|
||||
p.lastDelim = d.prev
|
||||
}
|
||||
}
|
||||
|
||||
// lastBracket returns the most recent open bracket on the stack.
|
||||
func (p *inlineParser) lastBracket() *delimiter {
|
||||
for d := p.lastDelim; d != nil; d = d.prev {
|
||||
if d.isBracket {
|
||||
return d
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *inlineParser) parse() {
|
||||
for p.pos < len(p.src) {
|
||||
switch c := p.src[p.pos]; c {
|
||||
case '\n':
|
||||
p.handleNewline()
|
||||
case '\\':
|
||||
p.handleBackslash()
|
||||
case '`':
|
||||
p.handleBackticks()
|
||||
case '<':
|
||||
p.handleLessThan()
|
||||
case '&':
|
||||
p.handleAmpersand()
|
||||
case '*', '_', '~':
|
||||
p.handleDelimiterRun(c)
|
||||
case '[':
|
||||
if p.tryFootnoteRef() {
|
||||
continue
|
||||
}
|
||||
p.pushBracket(false)
|
||||
case '!':
|
||||
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '[' {
|
||||
p.pushBracket(true)
|
||||
} else {
|
||||
p.push(&inline{kind: inlText, literal: "!"})
|
||||
p.pos++
|
||||
}
|
||||
case ']':
|
||||
p.handleCloseBracket()
|
||||
default:
|
||||
p.textRun()
|
||||
}
|
||||
}
|
||||
p.processEmphasis(nil)
|
||||
}
|
||||
|
||||
// handleNewline ends the line: the whitespace before the line ending is
|
||||
// stripped, and two or more spaces make the break hard.
|
||||
func (p *inlineParser) handleNewline() {
|
||||
spaces := 0
|
||||
if p.last != nil && p.last.kind == inlText {
|
||||
lit := p.last.literal
|
||||
end := len(lit)
|
||||
for end > 0 && isSpaceTab(lit[end-1]) {
|
||||
if lit[end-1] == ' ' {
|
||||
spaces++
|
||||
}
|
||||
end--
|
||||
}
|
||||
if end == 0 {
|
||||
p.removeNode(p.last)
|
||||
} else {
|
||||
p.last.literal = lit[:end]
|
||||
}
|
||||
}
|
||||
if spaces >= 2 {
|
||||
p.push(&inline{kind: inlBreak})
|
||||
} else {
|
||||
p.push(&inline{kind: inlText, literal: "\n"})
|
||||
}
|
||||
p.pos++
|
||||
}
|
||||
|
||||
func (p *inlineParser) handleBackslash() {
|
||||
if p.pos+1 < len(p.src) {
|
||||
next := p.src[p.pos+1]
|
||||
switch {
|
||||
case next == '\n':
|
||||
p.push(&inline{kind: inlBreak})
|
||||
p.pos += 2
|
||||
return
|
||||
case isASCIIPunct(next):
|
||||
p.push(&inline{kind: inlText, literal: string(next)})
|
||||
p.pos += 2
|
||||
return
|
||||
}
|
||||
}
|
||||
p.push(&inline{kind: inlText, literal: `\`})
|
||||
p.pos++
|
||||
}
|
||||
|
||||
// handleBackticks scans for a closing backtick string of equal length and
|
||||
// emits the code span between them, or the literal opening run.
|
||||
func (p *inlineParser) handleBackticks() {
|
||||
openLen := fenceRun(p.src[p.pos:], '`')
|
||||
i := p.pos + openLen
|
||||
for i < len(p.src) {
|
||||
if p.src[i] == '`' {
|
||||
l := fenceRun(p.src[i:], '`')
|
||||
if l == openLen {
|
||||
p.push(&inline{kind: inlCode, literal: codeSpanContent(p.src[p.pos+openLen : i])})
|
||||
p.pos = i + l
|
||||
return
|
||||
}
|
||||
i += l
|
||||
continue
|
||||
}
|
||||
i++
|
||||
}
|
||||
p.push(&inline{kind: inlText, literal: string(p.src[p.pos : p.pos+openLen])})
|
||||
p.pos += openLen
|
||||
}
|
||||
|
||||
// codeSpanContent converts line endings to spaces and strips the one-space
|
||||
// margin a span carries at both ends when it is not all spaces.
|
||||
func codeSpanContent(c []byte) string {
|
||||
c = bytes.ReplaceAll(c, []byte("\n"), []byte(" "))
|
||||
if len(c) >= 2 && c[0] == ' ' && c[len(c)-1] == ' ' {
|
||||
allSpaces := true
|
||||
for _, b := range c {
|
||||
if b != ' ' {
|
||||
allSpaces = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if !allSpaces {
|
||||
c = c[1 : len(c)-1]
|
||||
}
|
||||
}
|
||||
return string(c)
|
||||
}
|
||||
|
||||
func (p *inlineParser) handleLessThan() {
|
||||
if text, dest, n, ok := scanAutolink(p.src[p.pos:]); ok {
|
||||
p.push(&inline{kind: inlLink, dest: dest, children: []*inline{{kind: inlText, literal: text}}})
|
||||
p.pos += n
|
||||
return
|
||||
}
|
||||
if n := scanRawHTML(p.src[p.pos:]); n > 0 {
|
||||
p.push(&inline{kind: inlRawHTML, literal: string(p.src[p.pos : p.pos+n])})
|
||||
p.pos += n
|
||||
return
|
||||
}
|
||||
p.push(&inline{kind: inlText, literal: "<"})
|
||||
p.pos++
|
||||
}
|
||||
|
||||
func (p *inlineParser) handleAmpersand() {
|
||||
if s, n, ok := scanEntity(p.src, p.pos); ok {
|
||||
p.push(&inline{kind: inlText, literal: s})
|
||||
p.pos += n
|
||||
return
|
||||
}
|
||||
p.push(&inline{kind: inlText, literal: "&"})
|
||||
p.pos++
|
||||
}
|
||||
|
||||
// handleDelimiterRun records a run of asterisks or underscores and whether
|
||||
// it may open or close emphasis under the flanking rules.
|
||||
func (p *inlineParser) handleDelimiterRun(char byte) {
|
||||
start := p.pos
|
||||
end := start + fenceRun(p.src[start:], char)
|
||||
beforeWS, beforePunct := classifyRune(runeBefore(p.src, start))
|
||||
afterWS, afterPunct := classifyRune(runeAfter(p.src, end))
|
||||
left := !afterWS && (!afterPunct || beforeWS || beforePunct)
|
||||
right := !beforeWS && (!beforePunct || afterWS || afterPunct)
|
||||
d := &delimiter{char: char, length: end - start, origLen: end - start}
|
||||
if char == '_' {
|
||||
d.canOpen = left && (!right || beforePunct)
|
||||
d.canClose = right && (!left || afterPunct)
|
||||
} else {
|
||||
// asterisks and tildes flank the same way
|
||||
d.canOpen, d.canClose = left, right
|
||||
}
|
||||
d.node = p.push(&inline{kind: inlText, literal: string(p.src[start:end])})
|
||||
p.pushDelim(d)
|
||||
p.pos = end
|
||||
}
|
||||
|
||||
// tryFootnoteRef consumes a reference to a defined footnote and emits its
|
||||
// marker. A reference to an undefined footnote stays bracket text.
|
||||
func (p *inlineParser) tryFootnoteRef() bool {
|
||||
if p.footnotes == nil {
|
||||
return false
|
||||
}
|
||||
label, n, ok := scanFootnoteLabel(p.src[p.pos:])
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
num, occurrence, ok := p.footnotes.reference(normaliseLabel(label))
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
p.push(&inline{kind: inlFootnoteRef, num: num, occurrence: occurrence})
|
||||
p.pos += n
|
||||
return true
|
||||
}
|
||||
|
||||
func (p *inlineParser) pushBracket(image bool) {
|
||||
lit := "["
|
||||
if image {
|
||||
lit = "!["
|
||||
}
|
||||
n := p.push(&inline{kind: inlText, literal: lit})
|
||||
p.pushDelim(&delimiter{
|
||||
node: n, char: '[', isBracket: true, image: image, active: true,
|
||||
srcPos: p.pos + len(lit),
|
||||
})
|
||||
p.pos += len(lit)
|
||||
}
|
||||
|
||||
// textRun consumes the run of ordinary characters up to the next special
|
||||
// one, emitting extended autolinks and plain text in the order they come.
|
||||
func (p *inlineParser) textRun() {
|
||||
start := p.pos
|
||||
for p.pos < len(p.src) {
|
||||
c := p.src[p.pos]
|
||||
if isInlineSpecial(c) {
|
||||
break
|
||||
}
|
||||
var prev byte
|
||||
if p.pos > 0 {
|
||||
prev = p.src[p.pos-1]
|
||||
}
|
||||
if c == 'w' || c == 'h' || c == 'f' ||
|
||||
(!isEmailPrevByte(prev) && isEmailLocalByte(c)) {
|
||||
if n, text, dest, ok := scanExtendedAutolink(p.src, p.pos, prev); ok {
|
||||
if p.pos > start {
|
||||
p.push(&inline{kind: inlText, literal: string(p.src[start:p.pos])})
|
||||
}
|
||||
p.push(&inline{
|
||||
kind: inlLink,
|
||||
dest: dest,
|
||||
children: []*inline{{kind: inlText, literal: text}},
|
||||
})
|
||||
p.pos += n
|
||||
start = p.pos
|
||||
continue
|
||||
}
|
||||
}
|
||||
p.pos++
|
||||
}
|
||||
if p.pos > start {
|
||||
p.push(&inline{kind: inlText, literal: string(p.src[start:p.pos])})
|
||||
}
|
||||
}
|
||||
|
||||
func isInlineSpecial(c byte) bool {
|
||||
switch c {
|
||||
case '\n', '\\', '`', '<', '&', '*', '_', '~', '[', ']', '!':
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// processEmphasis matches delimiter runs between the stack bottom and the
|
||||
// top into emphasis and strong nodes, following the reference algorithm:
|
||||
// closers walk forward, openers are searched backwards, a matching pair
|
||||
// may not both be intraword delimiters whose run lengths add up against
|
||||
// the rule of three, and the matched run lengths shrink from the inner
|
||||
// sides.
|
||||
func (p *inlineParser) processEmphasis(bottom *delimiter) {
|
||||
var closer *delimiter
|
||||
if bottom == nil {
|
||||
closer = p.firstDelim
|
||||
} else {
|
||||
closer = bottom.next
|
||||
}
|
||||
for closer != nil {
|
||||
if closer.isBracket || !closer.canClose {
|
||||
closer = closer.next
|
||||
continue
|
||||
}
|
||||
opener, found := p.findOpener(closer, bottom)
|
||||
if !found {
|
||||
if !closer.canOpen {
|
||||
p.removeDelim(closer)
|
||||
}
|
||||
closer = closer.next
|
||||
continue
|
||||
}
|
||||
use := 1
|
||||
kind := inlEmph
|
||||
switch closer.char {
|
||||
case '~':
|
||||
if closer.length >= 2 && opener.length >= 2 {
|
||||
use = 2
|
||||
}
|
||||
kind = inlStrikethrough
|
||||
default:
|
||||
if closer.length >= 2 && opener.length >= 2 {
|
||||
use = 2
|
||||
kind = inlStrong
|
||||
}
|
||||
}
|
||||
var children []*inline
|
||||
for n := opener.node.next; n != closer.node; n = n.next {
|
||||
children = append(children, n)
|
||||
}
|
||||
node := &inline{kind: kind, children: children}
|
||||
opener.node.next = node
|
||||
node.prev = opener.node
|
||||
node.next = closer.node
|
||||
closer.node.prev = node
|
||||
opener.node.literal = opener.node.literal[:len(opener.node.literal)-use]
|
||||
closer.node.literal = closer.node.literal[use:]
|
||||
opener.length -= use
|
||||
closer.length -= use
|
||||
for d := closer.prev; d != nil && d != opener; {
|
||||
prev := d.prev
|
||||
p.removeDelim(d)
|
||||
d = prev
|
||||
}
|
||||
if opener.length == 0 {
|
||||
p.removeNode(opener.node)
|
||||
p.removeDelim(opener)
|
||||
}
|
||||
if closer.length == 0 {
|
||||
next := closer.next
|
||||
p.removeNode(closer.node)
|
||||
p.removeDelim(closer)
|
||||
closer = next
|
||||
}
|
||||
}
|
||||
for p.lastDelim != nil && p.lastDelim != bottom {
|
||||
p.removeDelim(p.lastDelim)
|
||||
}
|
||||
}
|
||||
|
||||
// findOpener searches backwards from the closer for a run of the same
|
||||
// character that may open, honouring the rule of three.
|
||||
func (p *inlineParser) findOpener(closer, bottom *delimiter) (*delimiter, bool) {
|
||||
for opener := closer.prev; opener != nil && opener != bottom; opener = opener.prev {
|
||||
if opener.isBracket || opener.char != closer.char || !opener.canOpen {
|
||||
continue
|
||||
}
|
||||
oddMatch := closer.char != '~' &&
|
||||
(closer.canOpen || opener.canClose) &&
|
||||
(opener.origLen+closer.origLen)%3 == 0 &&
|
||||
!(opener.origLen%3 == 0 && closer.origLen%3 == 0)
|
||||
if !oddMatch {
|
||||
return opener, true
|
||||
}
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
// handleCloseBracket tries to close the most recent open bracket as an
|
||||
// inline link or image, as a reference link with an explicit, collapsed or
|
||||
// empty label, and leaves the bracket as literal text otherwise.
|
||||
func (p *inlineParser) handleCloseBracket() {
|
||||
opener := p.lastBracket()
|
||||
if opener == nil {
|
||||
p.push(&inline{kind: inlText, literal: "]"})
|
||||
p.pos++
|
||||
return
|
||||
}
|
||||
if !opener.active {
|
||||
p.removeDelim(opener)
|
||||
p.push(&inline{kind: inlText, literal: "]"})
|
||||
p.pos++
|
||||
return
|
||||
}
|
||||
closerIdx := p.pos
|
||||
p.pos++
|
||||
|
||||
var dest, title string
|
||||
var hasTitle bool
|
||||
matched := false
|
||||
|
||||
if p.pos < len(p.src) && p.src[p.pos] == '(' {
|
||||
save := p.pos
|
||||
p.pos++
|
||||
if d, t, ht, ok := p.scanInlineSpec(); ok {
|
||||
dest, title, hasTitle, matched = d, t, ht, true
|
||||
} else {
|
||||
p.pos = save
|
||||
}
|
||||
}
|
||||
if !matched {
|
||||
save := p.pos
|
||||
label, ok := p.referenceLabel(opener, closerIdx)
|
||||
if ok && len(bytes.TrimSpace(label)) > 0 && validLabel(label) {
|
||||
if ref, exists := p.refs[normaliseLabel(string(label))]; exists {
|
||||
dest, title, hasTitle, matched = ref.destination, ref.title, ref.hasTitle, true
|
||||
}
|
||||
}
|
||||
if !matched {
|
||||
p.pos = save
|
||||
}
|
||||
}
|
||||
if !matched {
|
||||
p.removeDelim(opener)
|
||||
p.push(&inline{kind: inlText, literal: "]"})
|
||||
return
|
||||
}
|
||||
|
||||
p.processEmphasis(opener)
|
||||
var children []*inline
|
||||
for n := opener.node.next; n != nil; n = n.next {
|
||||
children = append(children, n)
|
||||
}
|
||||
kind := inlLink
|
||||
if opener.image {
|
||||
kind = inlImage
|
||||
}
|
||||
node := &inline{kind: kind, dest: dest, title: title, hasTitle: hasTitle, children: children}
|
||||
if opener.node.prev != nil {
|
||||
opener.node.prev.next = node
|
||||
} else {
|
||||
p.first = node
|
||||
}
|
||||
node.prev = opener.node.prev
|
||||
node.next = nil
|
||||
p.last = node
|
||||
p.removeDelim(opener)
|
||||
if !opener.image {
|
||||
// Links may not nest in links; image brackets stay open.
|
||||
for d := opener.prev; d != nil; d = d.prev {
|
||||
if d.isBracket && !d.image {
|
||||
d.active = false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// referenceLabel reads the label of a reference link after the closing
|
||||
// bracket: an explicit label in brackets, the collapsed empty brackets, or
|
||||
// the shortcut label taken from the link text itself.
|
||||
func (p *inlineParser) referenceLabel(opener *delimiter, closerIdx int) ([]byte, bool) {
|
||||
if p.pos < len(p.src) && p.src[p.pos] == '[' {
|
||||
if p.pos+1 < len(p.src) && p.src[p.pos+1] == ']' {
|
||||
p.pos += 2
|
||||
return p.src[opener.srcPos:closerIdx], true
|
||||
}
|
||||
j := p.pos + 1
|
||||
for j < len(p.src) {
|
||||
if p.src[j] == '\\' && j+1 < len(p.src) {
|
||||
j += 2
|
||||
continue
|
||||
}
|
||||
if p.src[j] == ']' {
|
||||
break
|
||||
}
|
||||
j++
|
||||
}
|
||||
if j < len(p.src) {
|
||||
label := p.src[p.pos+1 : j]
|
||||
p.pos = j + 1
|
||||
return label, true
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
return p.src[opener.srcPos:closerIdx], true
|
||||
}
|
||||
|
||||
// scanInlineSpec parses the destination and optional title of an inline
|
||||
// link, starting after the opening parenthesis.
|
||||
func (p *inlineParser) scanInlineSpec() (string, string, bool, bool) {
|
||||
p.skipWhitespace()
|
||||
if p.pos < len(p.src) && p.src[p.pos] == ')' {
|
||||
p.pos++
|
||||
return "", "", false, true
|
||||
}
|
||||
d, n, ok := scanDestination(p.src[p.pos:])
|
||||
if !ok {
|
||||
return "", "", false, false
|
||||
}
|
||||
dest := unescapeText(string(d))
|
||||
p.pos += n
|
||||
p.skipWhitespace()
|
||||
if p.pos < len(p.src) {
|
||||
switch c := p.src[p.pos]; c {
|
||||
case '"', '\'', '(':
|
||||
title, ok := p.scanInlineTitle(c)
|
||||
if !ok {
|
||||
return "", "", false, false
|
||||
}
|
||||
p.skipWhitespace()
|
||||
if p.pos < len(p.src) && p.src[p.pos] == ')' {
|
||||
p.pos++
|
||||
return dest, unescapeText(title), true, true
|
||||
}
|
||||
return "", "", false, false
|
||||
}
|
||||
}
|
||||
if p.pos < len(p.src) && p.src[p.pos] == ')' {
|
||||
p.pos++
|
||||
return dest, "", false, true
|
||||
}
|
||||
return "", "", false, false
|
||||
}
|
||||
|
||||
// scanInlineTitle scans a title through its closing quote, line endings
|
||||
// included, leaving the position past the closing quote.
|
||||
func (p *inlineParser) scanInlineTitle(open byte) (string, bool) {
|
||||
closer := open
|
||||
if open == '(' {
|
||||
closer = ')'
|
||||
}
|
||||
i := p.pos + 1
|
||||
for i < len(p.src) {
|
||||
c := p.src[i]
|
||||
if c == '\\' && i+1 < len(p.src) {
|
||||
i += 2
|
||||
continue
|
||||
}
|
||||
if c == closer {
|
||||
title := string(p.src[p.pos+1 : i])
|
||||
p.pos = i + 1
|
||||
return title, true
|
||||
}
|
||||
i++
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
func (p *inlineParser) skipWhitespace() {
|
||||
for p.pos < len(p.src) {
|
||||
switch p.src[p.pos] {
|
||||
case ' ', '\t', '\n':
|
||||
p.pos++
|
||||
default:
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// runeBefore returns the rune that ends just before the position, or a
|
||||
// null rune at the start, which counts as whitespace.
|
||||
func runeBefore(src []byte, pos int) rune {
|
||||
if pos == 0 {
|
||||
return 0
|
||||
}
|
||||
r, _ := utf8.DecodeLastRune(src[:pos])
|
||||
return r
|
||||
}
|
||||
|
||||
// runeAfter returns the rune that starts at the position, or a null rune
|
||||
// at the end, which counts as whitespace.
|
||||
func runeAfter(src []byte, pos int) rune {
|
||||
if pos >= len(src) {
|
||||
return 0
|
||||
}
|
||||
r, _ := utf8.DecodeRune(src[pos:])
|
||||
return r
|
||||
}
|
||||
|
||||
// classifyRune reports whether the rune is whitespace and whether it is
|
||||
// punctuation, under the CommonMark definitions: Unicode whitespace, and
|
||||
// ASCII or Unicode punctuation or symbol characters.
|
||||
func classifyRune(r rune) (space, punct bool) {
|
||||
if r == 0 {
|
||||
return true, false
|
||||
}
|
||||
if r < utf8.RuneSelf {
|
||||
space = r == ' ' || r == '\t' || r == '\n' || r == '\v' || r == '\f' || r == '\r'
|
||||
punct = isASCIIPunct(byte(r))
|
||||
return space, punct
|
||||
}
|
||||
return unicode.IsSpace(r), unicode.IsPunct(r) || unicode.IsSymbol(r)
|
||||
}
|
||||
|
||||
// isASCIIPunct reports whether c is an ASCII punctuation character.
|
||||
func isASCIIPunct(c byte) bool {
|
||||
switch c {
|
||||
case '!', '"', '#', '$', '%', '&', '\'', '(', ')', '*', '+', ',', '-',
|
||||
'.', '/', ':', ';', '<', '=', '>', '?', '@', '[', '\\', ']', '^',
|
||||
'_', '`', '{', '|', '}', '~':
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "testing"
|
||||
|
||||
// inlineCorpus is the project's own hand-written corpus for the inline
|
||||
// layer, on top of a paragraph unless the case states otherwise.
|
||||
var inlineCorpus = []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
// Emphasis.
|
||||
{"em asterisk", "*foo*", "<p><em>foo</em></p>\n"},
|
||||
{"strong asterisk", "**foo**", "<p><strong>foo</strong></p>\n"},
|
||||
{"em and strong", "***foo***", "<p><em><strong>foo</strong></em></p>\n"},
|
||||
{"strong with inner em", "**foo *bar* baz**", "<p><strong>foo <em>bar</em> baz</strong></p>\n"},
|
||||
{"intraword asterisk", "foo*bar*baz", "<p>foo<em>bar</em>baz</p>\n"},
|
||||
{"intraword underscore stays", "foo_bar_baz", "<p>foo_bar_baz</p>\n"},
|
||||
{"em underscore with spaces", "_foo bar_", "<p><em>foo bar</em></p>\n"},
|
||||
{"underscore cannot open after space", "_ foo_", "<p>_ foo_</p>\n"},
|
||||
{"literal asterisk when spaced", "a * foo *", "<p>a * foo *</p>\n"},
|
||||
{"escaped asterisk", "\\*not em\\*", "<p>*not em*</p>\n"},
|
||||
{"escaped backslash then em", "\\\\*foo*", "<p>\\<em>foo</em></p>\n"},
|
||||
{"lone closer stays", "a *", "<p>a *</p>\n"},
|
||||
{"em inside word boundaries", "a*b*c", "<p>a<em>b</em>c</p>\n"},
|
||||
{"em holding strong", "*foo**bar**baz*", "<p><em>foo<strong>bar</strong>baz</em></p>\n"},
|
||||
{"strong inside em with text", "***foo** bar*", "<p><em><strong>foo</strong> bar</em></p>\n"},
|
||||
{"em inside strong at end", "**foo *bar***", "<p><strong>foo <em>bar</em></strong></p>\n"},
|
||||
{"unmatched inner run stays", "*foo**bar*", "<p><em>foo**bar</em></p>\n"},
|
||||
{"intraword digits", "5*6*78", "<p>5<em>6</em>78</p>\n"},
|
||||
{"underscore opens before punctuation", "_(bar)_", "<p><em>(bar)</em></p>\n"},
|
||||
{"intraword underscore before punctuation stays", "foo_(bar)_", "<p>foo_(bar)_</p>\n"},
|
||||
|
||||
// Code spans.
|
||||
{"code span", "`foo`", "<p><code>foo</code></p>\n"},
|
||||
{"code span strips one space margin", "` foo `", "<p><code>foo</code></p>\n"},
|
||||
{"code span keeps margin with doubles", "`` foo ``", "<p><code> foo </code></p>\n"},
|
||||
{"code span double backticks", "``foo ` bar``", "<p><code>foo ` bar</code></p>\n"},
|
||||
{"code span has no escapes", "`foo\\`bar`", "<p><code>foo\\</code>bar`</p>\n"},
|
||||
{"code span unmatched", "foo ` bar", "<p>foo ` bar</p>\n"},
|
||||
{"code span escapes markup", "`*em*`", "<p><code>*em*</code></p>\n"},
|
||||
|
||||
// Links and images.
|
||||
{"inline link", "[foo](/uri)", "<p><a href=\"/uri\">foo</a></p>\n"},
|
||||
{"inline link with title", "[foo](/uri \"title\")", "<p><a href=\"/uri\" title=\"title\">foo</a></p>\n"},
|
||||
{"inline link single quoted title", "[foo](/uri 'title')", "<p><a href=\"/uri\" title=\"title\">foo</a></p>\n"},
|
||||
{"inline link empty destination", "[foo]()", "<p><a href=\"\">foo</a></p>\n"},
|
||||
{"angle destination with space", "[foo](<my uri>)", "<p><a href=\"my%20uri\">foo</a></p>\n"},
|
||||
{"trailing paren stays text", "[foo](bar))", "<p><a href=\"bar\">foo</a>)</p>\n"},
|
||||
{"em inside link", "[*foo*](/uri)", "<p><a href=\"/uri\"><em>foo</em></a></p>\n"},
|
||||
{"image with title", "", "<p><img src=\"/url\" alt=\"foo\" title=\"title\" /></p>\n"},
|
||||
{"image inside link", "[](page)", "<p><a href=\"page\"><img src=\"img\" alt=\"alt\" /></a></p>\n"},
|
||||
{"no nested links", "[a [b](x)](y)", "<p>[a <a href=\"x\">b</a>](y)</p>\n"},
|
||||
{"undefined reference stays", "[foo]", "<p>[foo]</p>\n"},
|
||||
{"link destination escaped ampersand", "[a](/url?a=1&b=2)", "<p><a href=\"/url?a=1&b=2\">a</a></p>\n"},
|
||||
|
||||
// Autolinks.
|
||||
{"uri autolink", "<http://example.com>", "<p><a href=\"http://example.com\">http://example.com</a></p>\n"},
|
||||
{"email autolink", "<foo@bar.example.com>", "<p><a href=\"mailto:foo@bar.example.com\">foo@bar.example.com</a></p>\n"},
|
||||
{"not an autolink", "<3>", "<p><3></p>\n"},
|
||||
|
||||
// Extended autolinks (GFM).
|
||||
{"extended www", "Visit www.commonmark.org for more.",
|
||||
"<p>Visit <a href=\"http://www.commonmark.org\">www.commonmark.org</a> for more.</p>\n"},
|
||||
{"extended www with path", "go to www.commonmark.org/help today",
|
||||
"<p>go to <a href=\"http://www.commonmark.org/help\">www.commonmark.org/help</a> today</p>\n"},
|
||||
{"extended trailing punctuation", "see www.example.com.",
|
||||
"<p>see <a href=\"http://www.example.com\">www.example.com</a>.</p>\n"},
|
||||
{"extended paren balance", "www.example.com/query?q=(a+b)))",
|
||||
"<p><a href=\"http://www.example.com/query?q=(a+b)\">www.example.com/query?q=(a+b)</a>))</p>\n"},
|
||||
{"extended entity suffix", "www.example.com?q=x&hl;",
|
||||
"<p><a href=\"http://www.example.com?q=x\">www.example.com?q=x</a>&hl;</p>\n"},
|
||||
{"extended https", "open https://example.com/page",
|
||||
"<p>open <a href=\"https://example.com/page\">https://example.com/page</a></p>\n"},
|
||||
{"extended ftp", "ftp://files.example.org/pub",
|
||||
"<p><a href=\"ftp://files.example.org/pub\">ftp://files.example.org/pub</a></p>\n"},
|
||||
{"extended email", "write to a.b-c_d@example.com soon",
|
||||
"<p>write to <a href=\"mailto:a.b-c_d@example.com\">a.b-c_d@example.com</a> soon</p>\n"},
|
||||
{"extended email trailing dot", "mail me at user@example.net.",
|
||||
"<p>mail me at <a href=\"mailto:user@example.net\">user@example.net</a>.</p>\n"},
|
||||
{"plus before at only", "hello@mail+xyz.example is not, but hello+xyz@mail.example is",
|
||||
"<p>hello@mail+xyz.example is not, but <a href=\"mailto:hello+xyz@mail.example\">hello+xyz@mail.example</a> is</p>\n"},
|
||||
{"underscore banned in last segments", "www.un_der_score.org stays text",
|
||||
"<p>www.un_der_score.org stays text</p>\n"},
|
||||
{"no autolink mid word", "awww.example.com stays text",
|
||||
"<p>awww.example.com stays text</p>\n"},
|
||||
|
||||
// Raw inline HTML and entities.
|
||||
{"raw inline html", "a <b>c</b> d", "<p>a <b>c</b> d</p>\n"},
|
||||
{"html comment inline", "a <!-- c --> b", "<p>a <!-- c --> b</p>\n"},
|
||||
{"entity ampersand", "AT&T", "<p>AT&T</p>\n"},
|
||||
{"entity numeric", "#", "<p>#</p>\n"},
|
||||
{"entity hex", """, "<p>"</p>\n"},
|
||||
{"bare ampersand", "AT&T", "<p>AT&T</p>\n"},
|
||||
{"not an entity", "&x;", "<p>&x;</p>\n"},
|
||||
{"less than escaped", "a < b", "<p>a < b</p>\n"},
|
||||
|
||||
// Breaks.
|
||||
{"soft break", "foo\nbar", "<p>foo\nbar</p>\n"},
|
||||
{"hard break spaces", "foo \nbar", "<p>foo<br />\nbar</p>\n"},
|
||||
{"hard break backslash", "foo\\\nbar", "<p>foo<br />\nbar</p>\n"},
|
||||
{"one trailing space is soft", "foo \nbar", "<p>foo\nbar</p>\n"},
|
||||
{"trailing spaces at end dropped", "foo ", "<p>foo</p>\n"},
|
||||
}
|
||||
|
||||
func TestInlineCorpus(t *testing.T) {
|
||||
for _, tc := range inlineCorpus {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Reference links need definitions from earlier blocks, so these cases
|
||||
// carry multi-block inputs.
|
||||
func TestReferenceLinks(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
input string
|
||||
want string
|
||||
}{
|
||||
{"explicit reference", "[foo][bar]\n\n[bar]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
||||
{"collapsed reference", "[foo][]\n\n[foo]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
||||
{"shortcut reference", "[foo]\n\n[foo]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
||||
{"reference with title", "[foo]\n\n[foo]: /url \"the title\"", "<p><a href=\"/url\" title=\"the title\">foo</a></p>\n"},
|
||||
{"reference label case folded", "[Foo]\n\n[foo]: /url", "<p><a href=\"/url\">Foo</a></p>\n"},
|
||||
{"image reference", "![foo]\n\n[foo]: /url", "<p><img src=\"/url\" alt=\"foo\" /></p>\n"},
|
||||
{"shortcut takes whole text", "[foo *bar*]\n\n[foo *bar*]: /url", "<p><a href=\"/url\">foo <em>bar</em></a></p>\n"},
|
||||
{"inline beats reference", "[foo](/inline)\n\n[foo]: /ref", "<p><a href=\"/inline\">foo</a></p>\n"},
|
||||
{"link in heading", "# [foo](/uri)", "<h1><a href=\"/uri\">foo</a></h1>\n"},
|
||||
{"code span in heading", "## a `b` c", "<h2>a <code>b</code> c</h2>\n"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := string(RenderHTML([]byte(tc.input)))
|
||||
if got != tc.want {
|
||||
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Package markdown renders Markdown to HTML with an engine of its own,
|
||||
// built on the standard library alone. Parse builds the tree of blocks
|
||||
// and RenderHTML serialises it; the grammar of the block structure follows
|
||||
// CommonMark. Inline content is escaped plain text.
|
||||
package markdown
|
||||
|
||||
// RenderHTML parses source and renders it to HTML. The same input always
|
||||
// produces byte-identical output.
|
||||
func RenderHTML(source []byte) []byte {
|
||||
return RenderHTMLNode(Parse(source))
|
||||
}
|
||||
|
||||
// RenderHTMLNode renders a parsed document tree to HTML.
|
||||
func RenderHTMLNode(doc *Node) []byte {
|
||||
r := &renderer{refs: doc.refs, footnotes: newFootnoteTracker(doc.footnotes)}
|
||||
r.blocks(doc.children)
|
||||
r.footnoteSection()
|
||||
return r.out
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
// nodeKind identifies a block node of the document tree.
|
||||
type nodeKind uint8
|
||||
|
||||
const (
|
||||
kindDocument nodeKind = iota
|
||||
kindParagraph
|
||||
kindHeading
|
||||
kindCodeBlock
|
||||
kindHTMLBlock
|
||||
kindBlockquote
|
||||
kindList
|
||||
kindListItem
|
||||
kindThematicBreak
|
||||
kindTable
|
||||
kindFootnoteDef
|
||||
kindDefList
|
||||
kindDefTerm
|
||||
kindDefItem
|
||||
)
|
||||
|
||||
// listKind distinguishes bullet lists from ordered lists.
|
||||
type listKind uint8
|
||||
|
||||
const (
|
||||
bulletList listKind = iota
|
||||
orderedList
|
||||
)
|
||||
|
||||
// Column alignment of a table, carried per column.
|
||||
const (
|
||||
alignNone uint8 = iota
|
||||
alignLeft
|
||||
alignCentre
|
||||
alignRight
|
||||
)
|
||||
|
||||
// Node is one block of the parsed document. The zero value is a document
|
||||
// root; every other kind is created by the parser.
|
||||
type Node struct {
|
||||
kind nodeKind
|
||||
parent *Node
|
||||
children []*Node
|
||||
|
||||
// content holds the raw text of a leaf block: paragraph or heading
|
||||
// inline text, code source, or the lines of an HTML block.
|
||||
content []byte
|
||||
|
||||
level int // heading level, 1 to 6
|
||||
|
||||
fenced bool // code block opened by a fence
|
||||
fenceChar byte // fence character, '`' or '~'
|
||||
fenceLength int // length of the opening fence
|
||||
fenceOffset int // columns of indentation before the opening fence
|
||||
info string
|
||||
|
||||
htmlType int // HTML block start condition, 1 to 7
|
||||
|
||||
listKind listKind
|
||||
bulletChar byte // bullet list: the marker character
|
||||
delimiter byte // ordered list: '.' or ')'
|
||||
start int // ordered list: number of the first item
|
||||
tight bool
|
||||
markerOffset int // item: indentation of the marker inside its container
|
||||
padding int // item: columns from the marker start to the content
|
||||
|
||||
// refs collects the link reference definitions of the document; it is
|
||||
// carried by the root node only.
|
||||
refs map[string]reference
|
||||
|
||||
// footnotes collects the footnote definitions of the document, in
|
||||
// document order; it is carried by the root node only. The renderer
|
||||
// orders the rendered section by reference.
|
||||
footnotes []*Node
|
||||
|
||||
// label names a footnote definition.
|
||||
label string
|
||||
|
||||
// table columns and rows, cells as raw inline content.
|
||||
align []uint8
|
||||
header [][]byte
|
||||
rows [][][]byte
|
||||
|
||||
task bool // item: the first paragraph begins with a task marker
|
||||
taskDone bool
|
||||
|
||||
startLine int
|
||||
|
||||
lastLineBlank bool
|
||||
lastLineChecked bool
|
||||
finalised bool
|
||||
}
|
||||
|
||||
// reference is one link reference definition of the document.
|
||||
type reference struct {
|
||||
destination string
|
||||
title string
|
||||
hasTitle bool
|
||||
}
|
||||
|
||||
// canContain reports whether parent accepts child blocks of the given kind.
|
||||
func canContain(parent, child nodeKind) bool {
|
||||
switch parent {
|
||||
case kindDocument, kindBlockquote, kindListItem, kindDefItem, kindFootnoteDef:
|
||||
return child != kindDocument
|
||||
case kindList:
|
||||
return child == kindListItem
|
||||
case kindDefList:
|
||||
return child == kindDefTerm || child == kindDefItem || child == kindParagraph
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,817 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import "bytes"
|
||||
|
||||
const (
|
||||
tabStop = 4
|
||||
codeIndent = 4
|
||||
)
|
||||
|
||||
// Parse parses Markdown source into a tree of block nodes. The source is
|
||||
// trusted input: no sanitisation is applied, because whether the output may
|
||||
// reach an audience is the consumer's policy.
|
||||
func Parse(source []byte) *Node {
|
||||
p := &parser{doc: &Node{kind: kindDocument, refs: map[string]reference{}}}
|
||||
p.tip = p.doc
|
||||
for _, line := range splitLines(normalise(source)) {
|
||||
p.processLine(line)
|
||||
}
|
||||
p.closeUnmatched(p.doc)
|
||||
p.finalise(p.doc)
|
||||
return p.doc
|
||||
}
|
||||
|
||||
// parser holds the block parsing state. The offset, column, indent and
|
||||
// blank fields describe the current line from the current offset onward,
|
||||
// which moves as container prefixes are consumed; a tab may end up
|
||||
// partially consumed, in which case offset rests on the tab and column
|
||||
// counts only the consumed part of it.
|
||||
type parser struct {
|
||||
doc *Node
|
||||
tip *Node
|
||||
|
||||
line []byte
|
||||
lineNo int
|
||||
|
||||
offset int
|
||||
column int
|
||||
|
||||
firstNonspace int
|
||||
firstNonspaceColumn int
|
||||
indent int
|
||||
blank bool
|
||||
|
||||
partiallyConsumedTab bool
|
||||
|
||||
// suppressBlankMark keeps the blank-line bookkeeping away from a line
|
||||
// the parser consumed entirely, such as a closing code fence.
|
||||
suppressBlankMark bool
|
||||
}
|
||||
|
||||
func (p *parser) processLine(line []byte) {
|
||||
p.line = line
|
||||
p.lineNo++
|
||||
p.offset = 0
|
||||
p.column = 0
|
||||
p.partiallyConsumedTab = false
|
||||
p.findFirstNonspace()
|
||||
|
||||
lastMatched := p.checkOpenBlocks()
|
||||
container, opened := p.openNewBlocks(lastMatched)
|
||||
p.addText(container, opened)
|
||||
|
||||
switch p.tip.kind {
|
||||
case kindHeading, kindThematicBreak:
|
||||
p.finalise(p.tip)
|
||||
}
|
||||
}
|
||||
|
||||
// checkOpenBlocks matches the line against the open block chain, from the
|
||||
// document down to the tip, consuming the prefix of every block that
|
||||
// continues. It returns the deepest matched block; the blocks below it stay
|
||||
// open until addText closes or lazily continues them.
|
||||
func (p *parser) checkOpenBlocks() *Node {
|
||||
var chain []*Node
|
||||
for n := p.tip; n != nil; n = n.parent {
|
||||
chain = append(chain, n)
|
||||
}
|
||||
for i := len(chain) - 2; i >= 0; i-- {
|
||||
p.findFirstNonspace()
|
||||
if !p.continueBlock(chain[i]) {
|
||||
return chain[i+1]
|
||||
}
|
||||
}
|
||||
return chain[0]
|
||||
}
|
||||
|
||||
// continueBlock reports whether the open block continues on the current
|
||||
// line, consuming its prefix when it does.
|
||||
func (p *parser) continueBlock(n *Node) bool {
|
||||
switch n.kind {
|
||||
case kindBlockquote:
|
||||
if p.blank || p.indent > 3 {
|
||||
return false
|
||||
}
|
||||
if p.line[p.firstNonspace] != '>' {
|
||||
return false
|
||||
}
|
||||
p.advanceOffset(p.firstNonspace+1-p.offset, false)
|
||||
if p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) {
|
||||
p.advanceOffset(1, true)
|
||||
}
|
||||
return true
|
||||
case kindListItem:
|
||||
if p.blank {
|
||||
// A blank line ends an item that never took content.
|
||||
if len(n.children) == 0 {
|
||||
return false
|
||||
}
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
return true
|
||||
}
|
||||
if p.indent >= n.markerOffset+n.padding {
|
||||
p.advanceOffset(n.markerOffset+n.padding, true)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
case kindCodeBlock:
|
||||
if !n.fenced {
|
||||
switch {
|
||||
case p.indent >= codeIndent:
|
||||
p.advanceOffset(codeIndent, true)
|
||||
return true
|
||||
case p.blank:
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
if !p.blank && p.indent <= 3 && p.line[p.firstNonspace] == n.fenceChar {
|
||||
length := fenceRun(p.line[p.firstNonspace:], n.fenceChar)
|
||||
if length >= n.fenceLength && allSpaceTab(p.line[p.firstNonspace+length:]) {
|
||||
// A closing fence ends the block; the rest of the line is
|
||||
// nothing but whitespace.
|
||||
p.advanceOffset(len(p.line)-p.offset, false)
|
||||
p.finalise(n)
|
||||
p.suppressBlankMark = true
|
||||
return false
|
||||
}
|
||||
}
|
||||
// A content line gives up to the opening fence's indentation.
|
||||
for i := n.fenceOffset; i > 0 && p.offset < len(p.line) && isSpaceTab(p.line[p.offset]); i-- {
|
||||
p.advanceOffset(1, true)
|
||||
}
|
||||
return true
|
||||
case kindHTMLBlock:
|
||||
// The tag-based kinds end at a blank line; the raw kinds run to
|
||||
// their closing condition.
|
||||
if n.htmlType == 6 || n.htmlType == 7 {
|
||||
return !p.blank
|
||||
}
|
||||
return true
|
||||
case kindParagraph:
|
||||
return !p.blank
|
||||
case kindTable:
|
||||
return !p.blank
|
||||
case kindFootnoteDef:
|
||||
if p.blank {
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
return true
|
||||
}
|
||||
if p.indent >= codeIndent {
|
||||
p.advanceOffset(codeIndent, true)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
case kindDefItem:
|
||||
if p.blank {
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
return true
|
||||
}
|
||||
if p.indent >= n.markerOffset+n.padding {
|
||||
p.advanceOffset(n.markerOffset+n.padding, true)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// openNewBlocks starts new blocks on the line, beginning at the last
|
||||
// matched container and opening containers until a leaf takes over. It
|
||||
// returns the container the remaining text belongs to and whether anything
|
||||
// was opened or changed on the line.
|
||||
func (p *parser) openNewBlocks(container *Node) (*Node, bool) {
|
||||
opened := false
|
||||
for {
|
||||
switch container.kind {
|
||||
case kindCodeBlock, kindHTMLBlock:
|
||||
return container, opened
|
||||
}
|
||||
p.findFirstNonspace()
|
||||
if p.blank {
|
||||
return container, opened
|
||||
}
|
||||
indented := p.indent >= codeIndent
|
||||
|
||||
if !indented && p.line[p.firstNonspace] == '>' {
|
||||
p.advanceOffset(p.firstNonspace+1-p.offset, false)
|
||||
if p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) {
|
||||
p.advanceOffset(1, true)
|
||||
}
|
||||
container = p.addChild(container, kindBlockquote)
|
||||
opened = true
|
||||
continue
|
||||
}
|
||||
if !indented {
|
||||
if level, ok := scanATX(p.line[p.firstNonspace:]); ok {
|
||||
h := p.addChild(container, kindHeading)
|
||||
h.level = level
|
||||
p.advanceOffset(p.firstNonspace+level-p.offset, false)
|
||||
return h, true
|
||||
}
|
||||
}
|
||||
if !indented {
|
||||
if char, length, ok := scanOpenFence(p.line[p.firstNonspace:]); ok {
|
||||
code := p.addChild(container, kindCodeBlock)
|
||||
code.fenced = true
|
||||
code.fenceChar = char
|
||||
code.fenceLength = length
|
||||
code.fenceOffset = p.indent
|
||||
p.advanceOffset(p.firstNonspace+length-p.offset, false)
|
||||
return code, true
|
||||
}
|
||||
}
|
||||
if !indented {
|
||||
if t := scanHTMLBlockStart(p.line[p.firstNonspace:], container.kind == kindParagraph); t > 0 {
|
||||
h := p.addChild(container, kindHTMLBlock)
|
||||
h.htmlType = t
|
||||
return h, true
|
||||
}
|
||||
}
|
||||
if !indented && container.kind == kindParagraph {
|
||||
if aligns, ok := scanTableDelimiter(p.line[p.firstNonspace:]); ok {
|
||||
if table := p.tryOpenTable(container, aligns); table != nil {
|
||||
p.advanceOffset(len(p.line)-p.offset, false)
|
||||
return table, true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !indented {
|
||||
if label, markerLen, ok := scanFootnoteDefStart(p.line[p.firstNonspace:]); ok {
|
||||
def := p.addChild(container, kindFootnoteDef)
|
||||
def.label = normaliseLabel(label)
|
||||
def.padding = codeIndent
|
||||
p.advanceOffset(p.firstNonspace+markerLen-p.offset, false)
|
||||
container = def
|
||||
opened = true
|
||||
continue
|
||||
}
|
||||
}
|
||||
if !indented {
|
||||
if container.kind == kindParagraph || container.kind == kindDefList {
|
||||
if scanDefMarker(p.line[p.firstNonspace:]) {
|
||||
if item := p.tryOpenDefItem(container); item != nil {
|
||||
container = item
|
||||
opened = true
|
||||
continue
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if !indented && container.kind == kindParagraph {
|
||||
if level, ok := scanSetext(p.line[p.firstNonspace:]); ok {
|
||||
// Reference definitions leave the paragraph first; the
|
||||
// heading forms only over what remains of it, and a
|
||||
// paragraph the definitions emptied turns the underline
|
||||
// back into plain text.
|
||||
p.extractReferences(container)
|
||||
if len(container.content) == 0 {
|
||||
parent := container.parent
|
||||
parent.children = parent.children[:len(parent.children)-1]
|
||||
p.tip = parent
|
||||
return parent, true
|
||||
}
|
||||
container.kind = kindHeading
|
||||
container.level = level
|
||||
p.advanceOffset(len(p.line)-p.offset, false)
|
||||
return container, true
|
||||
}
|
||||
}
|
||||
if !indented && isThematicBreak(p.line[p.firstNonspace:]) {
|
||||
p.addChild(container, kindThematicBreak)
|
||||
p.advanceOffset(len(p.line)-p.offset, false)
|
||||
return p.tip, true
|
||||
}
|
||||
if !indented {
|
||||
if data, markerLen, ok := p.parseListMarker(container.kind == kindParagraph); ok {
|
||||
if container.kind != kindList || !listsMatch(container, data) {
|
||||
l := p.addChild(container, kindList)
|
||||
l.listKind = data.listKind
|
||||
l.bulletChar = data.bulletChar
|
||||
l.delimiter = data.delimiter
|
||||
l.start = data.start
|
||||
container = l
|
||||
}
|
||||
item := p.addChild(container, kindListItem)
|
||||
item.markerOffset = p.indent
|
||||
p.advanceOffset(p.firstNonspace+markerLen-p.offset, false)
|
||||
saveOffset, saveColumn, saveTab := p.offset, p.column, p.partiallyConsumedTab
|
||||
for p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) {
|
||||
p.advanceOffset(1, true)
|
||||
}
|
||||
cols := p.column - saveColumn
|
||||
blankItem := p.offset >= len(p.line)
|
||||
padding := markerLen + cols
|
||||
if blankItem || cols >= 5 || cols < 1 {
|
||||
padding = markerLen + 1
|
||||
}
|
||||
p.offset, p.column, p.partiallyConsumedTab = saveOffset, saveColumn, saveTab
|
||||
item.padding = padding
|
||||
p.advanceOffset(padding-markerLen, true)
|
||||
container = item
|
||||
opened = true
|
||||
continue
|
||||
}
|
||||
}
|
||||
return container, opened
|
||||
}
|
||||
}
|
||||
|
||||
// listData carries what a list marker says about the list it belongs to.
|
||||
type listData struct {
|
||||
listKind listKind
|
||||
bulletChar byte
|
||||
delimiter byte
|
||||
start int
|
||||
}
|
||||
|
||||
// listsMatch reports whether a new marker continues the given list.
|
||||
func listsMatch(l *Node, d listData) bool {
|
||||
if l.listKind != d.listKind {
|
||||
return false
|
||||
}
|
||||
if d.listKind == bulletList {
|
||||
return l.bulletChar == d.bulletChar
|
||||
}
|
||||
return l.delimiter == d.delimiter
|
||||
}
|
||||
|
||||
// parseListMarker recognises a list marker at the first non-space
|
||||
// character. A marker interrupting a paragraph must carry content, and an
|
||||
// ordered one must number 1.
|
||||
func (p *parser) parseListMarker(interrupts bool) (listData, int, bool) {
|
||||
s := p.line[p.firstNonspace:]
|
||||
if len(s) == 0 {
|
||||
return listData{}, 0, false
|
||||
}
|
||||
var d listData
|
||||
i := 0
|
||||
switch c := s[0]; {
|
||||
case c == '-' || c == '+' || c == '*':
|
||||
d.listKind = bulletList
|
||||
d.bulletChar = c
|
||||
i = 1
|
||||
case c >= '0' && c <= '9':
|
||||
start := 0
|
||||
for i < len(s) && i < 9 && s[i] >= '0' && s[i] <= '9' {
|
||||
start = start*10 + int(s[i]-'0')
|
||||
i++
|
||||
}
|
||||
if i >= len(s) || (s[i] != '.' && s[i] != ')') {
|
||||
return listData{}, 0, false
|
||||
}
|
||||
if interrupts && start != 1 {
|
||||
return listData{}, 0, false
|
||||
}
|
||||
d.listKind = orderedList
|
||||
d.delimiter = s[i]
|
||||
d.start = start
|
||||
i++
|
||||
default:
|
||||
return listData{}, 0, false
|
||||
}
|
||||
if i < len(s) && !isSpaceTab(s[i]) {
|
||||
return listData{}, 0, false
|
||||
}
|
||||
if interrupts {
|
||||
j := i
|
||||
for j < len(s) && isSpaceTab(s[j]) {
|
||||
j++
|
||||
}
|
||||
if j >= len(s) {
|
||||
return listData{}, 0, false
|
||||
}
|
||||
}
|
||||
return d, i, true
|
||||
}
|
||||
|
||||
// tryOpenTable turns the last line of an open paragraph into the header of
|
||||
// a table whose delimiter row is on the current line. The paragraph keeps
|
||||
// its earlier lines, or disappears when the header was all of it. The
|
||||
// table opens only when the header and the delimiter row agree on the
|
||||
// number of columns.
|
||||
func (p *parser) tryOpenTable(para *Node, aligns []uint8) *Node {
|
||||
content := para.content
|
||||
if len(content) == 0 || content[len(content)-1] != '\n' {
|
||||
return nil
|
||||
}
|
||||
lastNewline := bytes.LastIndexByte(content[:len(content)-1], '\n') + 1
|
||||
headerLine := content[lastNewline : len(content)-1]
|
||||
cells := splitTableRow(headerLine)
|
||||
if len(cells) != len(aligns) {
|
||||
return nil
|
||||
}
|
||||
parent := para.parent
|
||||
if lastNewline > 0 {
|
||||
para.content = content[:lastNewline]
|
||||
p.finalise(para)
|
||||
} else {
|
||||
parent.children = parent.children[:len(parent.children)-1]
|
||||
p.tip = parent
|
||||
}
|
||||
table := p.addChild(parent, kindTable)
|
||||
table.align = aligns
|
||||
table.header = cells
|
||||
return table
|
||||
}
|
||||
|
||||
// tryOpenDefItem turns the line's definition marker into a definition
|
||||
// inside a definition list. When the container is a paragraph, the
|
||||
// paragraph's lines become the terms of a new entry; when it is a
|
||||
// definition list, the marker adds another definition to the entry. The
|
||||
// returned item is the container for the definition's content, with the
|
||||
// position placed at the content.
|
||||
func (p *parser) tryOpenDefItem(container *Node) *Node {
|
||||
var dl *Node
|
||||
if container.kind == kindParagraph {
|
||||
dl = p.defListFromTerms(container)
|
||||
if dl == nil {
|
||||
return nil
|
||||
}
|
||||
} else {
|
||||
dl = container
|
||||
}
|
||||
item := p.addChild(dl, kindDefItem)
|
||||
item.markerOffset = p.indent
|
||||
p.advanceOffset(p.firstNonspace+1-p.offset, false)
|
||||
saveOffset, saveColumn, saveTab := p.offset, p.column, p.partiallyConsumedTab
|
||||
for p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) {
|
||||
p.advanceOffset(1, true)
|
||||
}
|
||||
cols := p.column - saveColumn
|
||||
blankItem := p.offset >= len(p.line)
|
||||
padding := 1 + cols
|
||||
if blankItem || cols >= 5 || cols < 1 {
|
||||
padding = 2
|
||||
}
|
||||
p.offset, p.column, p.partiallyConsumedTab = saveOffset, saveColumn, saveTab
|
||||
item.padding = padding
|
||||
p.advanceOffset(padding-1, true)
|
||||
return item
|
||||
}
|
||||
|
||||
// defListFromTerms turns a paragraph of terms into definition terms of a
|
||||
// definition list, continuing the list when one is already open beside the
|
||||
// paragraph.
|
||||
func (p *parser) defListFromTerms(para *Node) *Node {
|
||||
if len(para.content) == 0 || para.content[len(para.content)-1] != '\n' {
|
||||
return nil
|
||||
}
|
||||
lines := splitLines(para.content)
|
||||
if len(lines) == 0 {
|
||||
return nil
|
||||
}
|
||||
parent := para.parent
|
||||
parent.children = parent.children[:len(parent.children)-1]
|
||||
p.tip = parent
|
||||
var dl *Node
|
||||
switch {
|
||||
case parent.kind == kindDefList:
|
||||
dl = parent
|
||||
case len(parent.children) > 0 && parent.children[len(parent.children)-1].kind == kindDefList:
|
||||
dl = parent.children[len(parent.children)-1]
|
||||
default:
|
||||
dl = p.addChild(parent, kindDefList)
|
||||
}
|
||||
for _, line := range lines {
|
||||
dt := p.addChild(dl, kindDefTerm)
|
||||
dt.content = line
|
||||
}
|
||||
return dl
|
||||
}
|
||||
|
||||
// addText places the remaining text of the line: into the open paragraph
|
||||
// when it continues, lazily or not, or into the container that accepts
|
||||
// lines, or into a new block otherwise. It also records which blocks the
|
||||
// blank line terminates, which decides list tightness.
|
||||
func (p *parser) addText(container *Node, opened bool) {
|
||||
p.findFirstNonspace()
|
||||
|
||||
if p.suppressBlankMark {
|
||||
p.suppressBlankMark = false
|
||||
return
|
||||
}
|
||||
|
||||
if p.blank && len(container.children) > 0 {
|
||||
container.children[len(container.children)-1].lastLineBlank = true
|
||||
}
|
||||
lastBlank := p.blank &&
|
||||
container.kind != kindBlockquote &&
|
||||
container.kind != kindHeading &&
|
||||
container.kind != kindThematicBreak &&
|
||||
!(container.kind == kindCodeBlock && container.fenced) &&
|
||||
!(container.kind == kindListItem && len(container.children) == 0 && container.startLine == p.lineNo)
|
||||
container.lastLineBlank = lastBlank
|
||||
for n := container.parent; n != nil; n = n.parent {
|
||||
n.lastLineBlank = false
|
||||
}
|
||||
|
||||
maybeLazy := p.tip.kind == kindParagraph
|
||||
if maybeLazy && !opened && !p.blank {
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
p.appendLine(p.tip, true)
|
||||
return
|
||||
}
|
||||
|
||||
p.closeUnmatched(container)
|
||||
|
||||
switch {
|
||||
case container.kind == kindCodeBlock:
|
||||
p.appendLine(container, true)
|
||||
case container.kind == kindHTMLBlock:
|
||||
p.appendLine(container, true)
|
||||
if htmlBlockEnds(container.htmlType, p.line[p.offset:]) {
|
||||
p.finalise(container)
|
||||
}
|
||||
case p.blank:
|
||||
// A blank line adds no content.
|
||||
case container.kind == kindTable:
|
||||
row := splitTableRow(p.line[p.offset:])
|
||||
for len(row) < len(container.header) {
|
||||
row = append(row, nil)
|
||||
}
|
||||
container.rows = append(container.rows, row[:len(container.header)])
|
||||
case container.kind == kindParagraph:
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
p.appendLine(container, true)
|
||||
case container.kind == kindHeading:
|
||||
container.content = append(container.content, chopClosingHashes(p.line[p.firstNonspace:])...)
|
||||
default:
|
||||
if p.indent >= codeIndent && !maybeLazy {
|
||||
code := p.addChild(container, kindCodeBlock)
|
||||
p.advanceOffset(codeIndent, true)
|
||||
p.appendLine(code, true)
|
||||
} else {
|
||||
if container.kind == kindListItem && len(container.children) == 0 && container.startLine == p.lineNo {
|
||||
if checked, n, ok := scanTaskMarker(p.line[p.firstNonspace:]); ok {
|
||||
p.advanceOffset(p.firstNonspace+n-p.offset, false)
|
||||
container.task = true
|
||||
container.taskDone = checked
|
||||
p.findFirstNonspace()
|
||||
}
|
||||
}
|
||||
para := p.addChild(container, kindParagraph)
|
||||
p.advanceOffset(p.firstNonspace-p.offset, false)
|
||||
p.appendLine(para, true)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// addChild attaches a new block below parent, closing open blocks that
|
||||
// cannot contain it, and makes it the tip.
|
||||
func (p *parser) addChild(parent *Node, kind nodeKind) *Node {
|
||||
for !canContain(parent.kind, kind) {
|
||||
p.finalise(parent)
|
||||
parent = parent.parent
|
||||
}
|
||||
n := &Node{kind: kind, parent: parent, startLine: p.lineNo}
|
||||
parent.children = append(parent.children, n)
|
||||
p.tip = n
|
||||
return n
|
||||
}
|
||||
|
||||
// closeUnmatched closes every open block below the given container.
|
||||
func (p *parser) closeUnmatched(container *Node) {
|
||||
for p.tip != container {
|
||||
p.finalise(p.tip)
|
||||
}
|
||||
}
|
||||
|
||||
// finalise closes a block and its children, trimming content and computing
|
||||
// derived data such as list tightness.
|
||||
func (p *parser) finalise(n *Node) {
|
||||
if n.finalised {
|
||||
return
|
||||
}
|
||||
n.finalised = true
|
||||
for _, c := range n.children {
|
||||
p.finalise(c)
|
||||
}
|
||||
switch n.kind {
|
||||
case kindParagraph:
|
||||
p.extractReferences(n)
|
||||
if len(n.content) == 0 && n.parent != nil {
|
||||
n.parent.children = n.parent.children[:len(n.parent.children)-1]
|
||||
}
|
||||
case kindHeading:
|
||||
if bytes.HasSuffix(n.content, []byte("\n")) {
|
||||
n.content = n.content[:len(n.content)-1]
|
||||
}
|
||||
case kindCodeBlock:
|
||||
if n.fenced {
|
||||
if i := bytes.IndexByte(n.content, '\n'); i >= 0 {
|
||||
n.info = unescapeText(string(bytes.TrimSpace(n.content[:i])))
|
||||
n.content = n.content[i+1:]
|
||||
} else {
|
||||
n.info = unescapeText(string(bytes.TrimSpace(n.content)))
|
||||
n.content = nil
|
||||
}
|
||||
} else {
|
||||
n.content = trimTrailingBlankLines(n.content)
|
||||
}
|
||||
case kindList, kindDefList:
|
||||
n.tight = !blocksAreLoose(n.children)
|
||||
case kindDocument:
|
||||
p.gatherFootnotes(n)
|
||||
}
|
||||
if p.tip == n {
|
||||
p.tip = n.parent
|
||||
}
|
||||
}
|
||||
|
||||
// trimTrailingBlankLines removes the trailing blank lines of an indented
|
||||
// code block. Every content line carries its newline, so the result of a
|
||||
// non-empty block ends with exactly one.
|
||||
func trimTrailingBlankLines(c []byte) []byte {
|
||||
for len(c) > 0 {
|
||||
end := len(c) - 1 // the final newline
|
||||
start := bytes.LastIndexByte(c[:end], '\n') + 1
|
||||
if !allSpaceTab(c[start:end]) {
|
||||
return c
|
||||
}
|
||||
c = c[:start]
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// blocksAreLoose reports whether any two sibling blocks, or any two blocks
|
||||
// of one container-like block, are separated by a blank line.
|
||||
func blocksAreLoose(nodes []*Node) bool {
|
||||
for i, n := range nodes {
|
||||
if endsWithBlank(n) && i+1 < len(nodes) {
|
||||
return true
|
||||
}
|
||||
switch n.kind {
|
||||
case kindListItem, kindDefItem:
|
||||
for j, child := range n.children {
|
||||
lastItem := i+1 == len(nodes)
|
||||
lastChild := j+1 == len(n.children)
|
||||
if endsWithBlank(child) && (!lastItem || !lastChild) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// endsWithBlank reports whether the block, or the last block inside a
|
||||
// container chain, was followed by a blank line.
|
||||
func endsWithBlank(n *Node) bool {
|
||||
if n.lastLineChecked {
|
||||
return n.lastLineBlank
|
||||
}
|
||||
n.lastLineChecked = true
|
||||
switch n.kind {
|
||||
case kindList, kindListItem, kindDefList, kindDefItem, kindFootnoteDef:
|
||||
if len(n.children) > 0 {
|
||||
return endsWithBlank(n.children[len(n.children)-1])
|
||||
}
|
||||
}
|
||||
return n.lastLineBlank
|
||||
}
|
||||
|
||||
// gatherFootnotes lifts every footnote definition out of the tree, keeping
|
||||
// the first definition of a label.
|
||||
func (p *parser) gatherFootnotes(doc *Node) {
|
||||
seen := map[string]bool{}
|
||||
var defs []*Node
|
||||
var walk func(n *Node)
|
||||
walk = func(n *Node) {
|
||||
keep := n.children[:0]
|
||||
for _, c := range n.children {
|
||||
if c.kind == kindFootnoteDef {
|
||||
if !seen[c.label] {
|
||||
seen[c.label] = true
|
||||
defs = append(defs, c)
|
||||
}
|
||||
continue
|
||||
}
|
||||
walk(c)
|
||||
keep = append(keep, c)
|
||||
}
|
||||
n.children = keep
|
||||
}
|
||||
walk(doc)
|
||||
doc.footnotes = defs
|
||||
}
|
||||
|
||||
// appendLine appends the remaining line to the block's content. A tab
|
||||
// partially consumed while skipping indentation becomes the spaces it
|
||||
// stood for.
|
||||
func (p *parser) appendLine(n *Node, newline bool) {
|
||||
if p.partiallyConsumedTab {
|
||||
p.offset++
|
||||
for i := tabStop - p.column%tabStop; i > 0; i-- {
|
||||
n.content = append(n.content, ' ')
|
||||
}
|
||||
}
|
||||
n.content = append(n.content, p.line[p.offset:]...)
|
||||
if newline {
|
||||
n.content = append(n.content, '\n')
|
||||
}
|
||||
}
|
||||
|
||||
// advanceOffset moves into the line by count bytes, or columns when
|
||||
// columns is set, expanding tabs to tab stops and leaving a tab partially
|
||||
// consumed when the count stops inside it.
|
||||
func (p *parser) advanceOffset(count int, columns bool) {
|
||||
for count > 0 && p.offset < len(p.line) {
|
||||
c := p.line[p.offset]
|
||||
if c != '\t' {
|
||||
p.partiallyConsumedTab = false
|
||||
p.offset++
|
||||
p.column++
|
||||
count--
|
||||
continue
|
||||
}
|
||||
toTab := tabStop - p.column%tabStop
|
||||
if columns {
|
||||
p.partiallyConsumedTab = toTab > count
|
||||
steps := min(count, toTab)
|
||||
p.column += steps
|
||||
if !p.partiallyConsumedTab {
|
||||
p.offset++
|
||||
}
|
||||
count -= steps
|
||||
} else {
|
||||
p.partiallyConsumedTab = false
|
||||
p.offset++
|
||||
p.column += toTab
|
||||
count--
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// findFirstNonspace locates the first non-space character from the offset
|
||||
// onward, computing the column of that character and the indentation of
|
||||
// the line relative to the offset. The line is blank when the first
|
||||
// non-space character does not exist.
|
||||
func (p *parser) findFirstNonspace() {
|
||||
toTab := tabStop - p.column%tabStop
|
||||
p.firstNonspace = p.offset
|
||||
p.firstNonspaceColumn = p.column
|
||||
for p.firstNonspace < len(p.line) {
|
||||
c := p.line[p.firstNonspace]
|
||||
switch c {
|
||||
case ' ':
|
||||
p.firstNonspace++
|
||||
p.firstNonspaceColumn++
|
||||
toTab--
|
||||
if toTab == 0 {
|
||||
toTab = tabStop
|
||||
}
|
||||
case '\t':
|
||||
p.firstNonspace++
|
||||
p.firstNonspaceColumn += toTab
|
||||
toTab = tabStop
|
||||
default:
|
||||
p.indent = p.firstNonspaceColumn - p.column
|
||||
p.blank = false
|
||||
return
|
||||
}
|
||||
}
|
||||
p.indent = p.firstNonspaceColumn - p.column
|
||||
p.blank = true
|
||||
}
|
||||
|
||||
// normalise prepares source for parsing: line endings become newlines and
|
||||
// a null byte becomes the replacement character.
|
||||
func normalise(source []byte) []byte {
|
||||
out := make([]byte, 0, len(source))
|
||||
for i := 0; i < len(source); i++ {
|
||||
switch c := source[i]; c {
|
||||
case '\r':
|
||||
if i+1 < len(source) && source[i+1] == '\n' {
|
||||
i++
|
||||
}
|
||||
out = append(out, '\n')
|
||||
case 0:
|
||||
out = append(out, '\xef', '\xbf', '\xbd')
|
||||
default:
|
||||
out = append(out, c)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// splitLines splits normalised source into lines without their newlines.
|
||||
// A trailing newline produces no empty final line.
|
||||
func splitLines(source []byte) [][]byte {
|
||||
var lines [][]byte
|
||||
start := 0
|
||||
for i, c := range source {
|
||||
if c == '\n' {
|
||||
lines = append(lines, source[start:i])
|
||||
start = i + 1
|
||||
}
|
||||
}
|
||||
if start < len(source) {
|
||||
lines = append(lines, source[start:])
|
||||
}
|
||||
return lines
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestDeterministicOutput(t *testing.T) {
|
||||
source := []byte("# Title\n\nText with < and &.\n\n- one\n - nested\n\n> quoted\n\n" +
|
||||
"```go\nx := 1\n```\n\n[ref]: /url \"title\"\n")
|
||||
first := RenderHTML(source)
|
||||
for i := range 10 {
|
||||
if next := RenderHTML(source); !bytes.Equal(first, next) {
|
||||
t.Fatalf("render %d differs:\nfirst: %q\nnext: %q", i, first, next)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestReferenceNormalisation(t *testing.T) {
|
||||
doc := Parse([]byte("[Foo Bar]: /first\n[FOO\t bar]: /second\n"))
|
||||
if len(doc.refs) != 1 {
|
||||
t.Fatalf("got %d references, want 1", len(doc.refs))
|
||||
}
|
||||
ref, ok := doc.refs["foo bar"]
|
||||
if !ok {
|
||||
t.Fatalf("no reference under the normalised label \"foo bar\"")
|
||||
}
|
||||
if ref.destination != "/first" {
|
||||
t.Errorf("destination = %q, want \"/first\": the first definition wins", ref.destination)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReferenceTitleKept(t *testing.T) {
|
||||
doc := Parse([]byte("[foo]: /url 'the title'\n"))
|
||||
ref, ok := doc.refs["foo"]
|
||||
if !ok {
|
||||
t.Fatal("no reference recorded")
|
||||
}
|
||||
if !ref.hasTitle || ref.title != "the title" {
|
||||
t.Errorf("title = %q with hasTitle %v, want \"the title\" with hasTitle", ref.title, ref.hasTitle)
|
||||
}
|
||||
}
|
||||
|
||||
func TestJunkAfterTitleIsNotADefinition(t *testing.T) {
|
||||
doc := Parse([]byte("[foo]: /url \"title\" ok\n"))
|
||||
if len(doc.refs) != 0 {
|
||||
t.Errorf("got %d references, want 0", len(doc.refs))
|
||||
}
|
||||
if len(doc.children) != 1 || doc.children[0].kind != kindParagraph {
|
||||
t.Fatalf("the line should stay a paragraph")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBacktickInfoStringRejectsFence(t *testing.T) {
|
||||
doc := Parse([]byte("``` aaa ```\n"))
|
||||
if len(doc.children) != 1 || doc.children[0].kind != kindParagraph {
|
||||
t.Fatalf("the line should stay a paragraph, got %d children", len(doc.children))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseEmptyDocument(t *testing.T) {
|
||||
doc := Parse(nil)
|
||||
if doc.kind != kindDocument || len(doc.children) != 0 {
|
||||
t.Errorf("Parse(nil) should give an empty document")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,253 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// extractReferences strips leading link reference definitions from a
|
||||
// closed paragraph and records them in the document. The first definition
|
||||
// of a label wins. What remains of the paragraph keeps a single trailing
|
||||
// newline removed.
|
||||
func (p *parser) extractReferences(n *Node) {
|
||||
for {
|
||||
def, consumed, ok := parseReference(n.content)
|
||||
if !ok || consumed <= 0 {
|
||||
break
|
||||
}
|
||||
if _, exists := p.doc.refs[def.label]; !exists {
|
||||
p.doc.refs[def.label] = def.reference
|
||||
}
|
||||
n.content = n.content[consumed:]
|
||||
}
|
||||
if bytes.HasSuffix(n.content, []byte("\n")) {
|
||||
n.content = n.content[:len(n.content)-1]
|
||||
}
|
||||
}
|
||||
|
||||
// refDef is a parsed link reference definition: the normalised label under
|
||||
// which it is recorded and the reference it defines.
|
||||
type refDef struct {
|
||||
label string
|
||||
reference
|
||||
}
|
||||
|
||||
// parseReference parses one link reference definition from the start of
|
||||
// the paragraph content, possibly spanning lines. It returns the
|
||||
// definition, the number of bytes consumed and whether a definition was
|
||||
// there at all.
|
||||
func parseReference(c []byte) (refDef, int, bool) {
|
||||
// The label: brackets around one to 999 characters, no unescaped
|
||||
// bracket inside.
|
||||
if len(c) == 0 || c[0] != '[' {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
i := 1
|
||||
labelEnd := -1
|
||||
for i < len(c) {
|
||||
ch := c[i]
|
||||
if ch == '\\' && i+1 < len(c) {
|
||||
i += 2
|
||||
continue
|
||||
}
|
||||
if ch == ']' {
|
||||
labelEnd = i
|
||||
break
|
||||
}
|
||||
i++
|
||||
}
|
||||
if labelEnd < 0 {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
label := c[1:labelEnd]
|
||||
if len(label) < 1 || len(label) > 999 || len(bytes.TrimSpace(label)) == 0 || !validLabel(label) {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
i = labelEnd + 1
|
||||
if i >= len(c) || c[i] != ':' {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
i++
|
||||
|
||||
// Up to one line ending may sit between the colon and the destination.
|
||||
i, _, ok := skipSpaceOneNewline(c, i)
|
||||
if !ok {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
dest, n, ok := scanDestination(c[i:])
|
||||
if !ok {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
i += n
|
||||
|
||||
// A definition without a title needs the rest of the destination's
|
||||
// line to be blank.
|
||||
titlelessEnd := -1
|
||||
if lineEnd := bytes.IndexByte(c[i:], '\n'); lineEnd < 0 {
|
||||
if allSpaceTab(c[i:]) {
|
||||
titlelessEnd = len(c)
|
||||
}
|
||||
} else if allSpaceTab(c[i : i+lineEnd]) {
|
||||
titlelessEnd = i + lineEnd + 1
|
||||
}
|
||||
|
||||
// A title, when present, sits after at least one character of
|
||||
// whitespace, with at most one line ending between it and the
|
||||
// destination, and nothing but whitespace may follow it.
|
||||
if j, skipped, ok := skipSpaceOneNewline(c, i); ok && skipped > 0 && j < len(c) && (c[j] == '"' || c[j] == '\'' || c[j] == '(') {
|
||||
open := c[j]
|
||||
closer := open
|
||||
if open == '(' {
|
||||
closer = ')'
|
||||
}
|
||||
k := j + 1
|
||||
for k < len(c) {
|
||||
ch := c[k]
|
||||
if ch == '\\' && k+1 < len(c) {
|
||||
k += 2
|
||||
continue
|
||||
}
|
||||
if open == '(' && ch == '(' {
|
||||
break
|
||||
}
|
||||
if ch == closer {
|
||||
after := k + 1
|
||||
for after < len(c) && isSpaceTab(c[after]) {
|
||||
after++
|
||||
}
|
||||
if after >= len(c) {
|
||||
return def(label, dest, c[j+1:k], len(c))
|
||||
}
|
||||
if c[after] == '\n' {
|
||||
return def(label, dest, c[j+1:k], after+1)
|
||||
}
|
||||
break
|
||||
}
|
||||
k++
|
||||
}
|
||||
}
|
||||
|
||||
if titlelessEnd < 0 {
|
||||
return refDef{}, 0, false
|
||||
}
|
||||
return def(label, dest, nil, titlelessEnd)
|
||||
}
|
||||
|
||||
// def builds the result of a parsed definition.
|
||||
func def(label, dest, title []byte, consumed int) (refDef, int, bool) {
|
||||
r := reference{destination: unescapeText(string(dest))}
|
||||
if title != nil {
|
||||
r.title = unescapeText(string(title))
|
||||
r.hasTitle = true
|
||||
}
|
||||
return refDef{label: normaliseLabel(string(label)), reference: r}, consumed, true
|
||||
}
|
||||
|
||||
// skipSpaceOneNewline skips spaces, tabs and at most one newline, stopping
|
||||
// at the first other character or the end. It reports how much it skipped
|
||||
// and fails on a second line ending.
|
||||
func skipSpaceOneNewline(c []byte, i int) (int, int, bool) {
|
||||
start := i
|
||||
newlines := 0
|
||||
for i < len(c) {
|
||||
switch c[i] {
|
||||
case ' ', '\t':
|
||||
i++
|
||||
case '\n':
|
||||
newlines++
|
||||
if newlines > 1 {
|
||||
return i, i - start, false
|
||||
}
|
||||
i++
|
||||
default:
|
||||
return i, i - start, true
|
||||
}
|
||||
}
|
||||
return i, i - start, true
|
||||
}
|
||||
|
||||
// scanDestination parses a link destination: a run in angle brackets with
|
||||
// no line ending inside, or a bare run without whitespace in which
|
||||
// parentheses stay balanced.
|
||||
func scanDestination(c []byte) ([]byte, int, bool) {
|
||||
if len(c) > 0 && c[0] == '<' {
|
||||
i := 1
|
||||
for i < len(c) {
|
||||
ch := c[i]
|
||||
if ch == '\\' && i+1 < len(c) {
|
||||
i += 2
|
||||
continue
|
||||
}
|
||||
if ch == '>' {
|
||||
return c[1:i], i + 1, true
|
||||
}
|
||||
if ch == '<' || ch == '\n' {
|
||||
return nil, 0, false
|
||||
}
|
||||
i++
|
||||
}
|
||||
return nil, 0, false
|
||||
}
|
||||
i := 0
|
||||
depth := 0
|
||||
for i < len(c) {
|
||||
ch := c[i]
|
||||
if ch == '\\' && i+1 < len(c) {
|
||||
i += 2
|
||||
continue
|
||||
}
|
||||
if ch == '(' {
|
||||
depth++
|
||||
i++
|
||||
continue
|
||||
}
|
||||
if ch == ')' {
|
||||
if depth == 0 {
|
||||
break
|
||||
}
|
||||
depth--
|
||||
i++
|
||||
continue
|
||||
}
|
||||
if isSpaceTab(ch) || ch == '\n' {
|
||||
break
|
||||
}
|
||||
i++
|
||||
}
|
||||
if depth != 0 || i == 0 {
|
||||
return nil, 0, false
|
||||
}
|
||||
return c[:i], i, true
|
||||
}
|
||||
|
||||
// validLabel reports whether the raw label text carries no unescaped open
|
||||
// bracket, which a link label may not contain.
|
||||
func validLabel(label []byte) bool {
|
||||
for i := 0; i < len(label); i++ {
|
||||
switch label[i] {
|
||||
case '\\':
|
||||
i++
|
||||
case '[':
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// labelFolder carries the case fold pairs the standard library's lower
|
||||
// casing does not perform, which the Unicode case fold CommonMark names
|
||||
// does.
|
||||
var labelFolder = strings.NewReplacer(
|
||||
"ß", "ss", "ff", "ff", "fi", "fi", "fl", "fl",
|
||||
"ffi", "ffi", "ffl", "ffl", "ſt", "st", "st", "st",
|
||||
)
|
||||
|
||||
// normaliseLabel brings a link label to the form definitions and uses are
|
||||
// compared under: surrounding and repeated whitespace collapsed to single
|
||||
// spaces, then case folded.
|
||||
func normaliseLabel(s string) string {
|
||||
return labelFolder.Replace(strings.ToLower(strings.Join(strings.Fields(s), " ")))
|
||||
}
|
||||
@@ -0,0 +1,336 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// renderer serialises the block tree. Every block ends its output with a
|
||||
// newline; a paragraph inside a tight list item is the one exception, where
|
||||
// the text stands without its own wrapper.
|
||||
type renderer struct {
|
||||
out []byte
|
||||
refs map[string]reference
|
||||
footnotes *footnoteTracker
|
||||
}
|
||||
|
||||
func (r *renderer) puts(s string) {
|
||||
r.out = append(r.out, s...)
|
||||
}
|
||||
|
||||
func (r *renderer) blocks(nodes []*Node) {
|
||||
for _, n := range nodes {
|
||||
r.block(n)
|
||||
}
|
||||
}
|
||||
|
||||
func (r *renderer) block(n *Node) {
|
||||
switch n.kind {
|
||||
case kindParagraph:
|
||||
task := taskInput(n)
|
||||
if tightParent(n) {
|
||||
r.puts(task)
|
||||
r.inlineContent(n.content)
|
||||
if !lastChild(n) {
|
||||
r.puts("\n")
|
||||
}
|
||||
return
|
||||
}
|
||||
r.puts("<p>")
|
||||
r.puts(task)
|
||||
r.inlineContent(n.content)
|
||||
r.puts("</p>\n")
|
||||
case kindHeading:
|
||||
level := strconv.Itoa(n.level)
|
||||
r.puts("<h" + level + ">")
|
||||
r.inlineContent(n.content)
|
||||
r.puts("</h" + level + ">\n")
|
||||
case kindCodeBlock:
|
||||
r.puts("<pre><code")
|
||||
if word := infoWord(n.info); word != "" {
|
||||
r.puts(` class="language-` + escapeHTML(word) + `"`)
|
||||
}
|
||||
r.puts(">")
|
||||
r.puts(escapeHTML(string(n.content)))
|
||||
r.puts("</code></pre>\n")
|
||||
case kindHTMLBlock:
|
||||
r.out = append(r.out, n.content...)
|
||||
case kindBlockquote:
|
||||
r.puts("<blockquote>\n")
|
||||
r.blocks(n.children)
|
||||
r.puts("</blockquote>\n")
|
||||
case kindList:
|
||||
switch n.listKind {
|
||||
case bulletList:
|
||||
r.puts("<ul>\n")
|
||||
r.blocks(n.children)
|
||||
r.puts("</ul>\n")
|
||||
case orderedList:
|
||||
if n.start != 1 {
|
||||
r.puts(`<ol start="` + strconv.Itoa(n.start) + `">` + "\n")
|
||||
} else {
|
||||
r.puts("<ol>\n")
|
||||
}
|
||||
r.blocks(n.children)
|
||||
r.puts("</ol>\n")
|
||||
}
|
||||
case kindListItem:
|
||||
r.itemLike("li", n)
|
||||
case kindDefItem:
|
||||
r.itemLike("dd", n)
|
||||
case kindDefTerm:
|
||||
r.puts("<dt>")
|
||||
r.inlineContent(n.content)
|
||||
r.puts("</dt>\n")
|
||||
case kindDefList:
|
||||
r.puts("<dl>\n")
|
||||
r.blocks(n.children)
|
||||
r.puts("</dl>\n")
|
||||
case kindThematicBreak:
|
||||
r.puts("<hr />\n")
|
||||
case kindTable:
|
||||
r.puts("<table>\n<thead>\n<tr>\n")
|
||||
for i, cell := range n.header {
|
||||
r.puts("<th")
|
||||
r.alignAttr(n.align[i])
|
||||
r.puts(">")
|
||||
r.inlineContent(cell)
|
||||
r.puts("</th>\n")
|
||||
}
|
||||
r.puts("</tr>\n</thead>\n")
|
||||
if len(n.rows) > 0 {
|
||||
r.puts("<tbody>\n")
|
||||
for _, row := range n.rows {
|
||||
r.puts("<tr>\n")
|
||||
for i, cell := range row {
|
||||
r.puts("<td")
|
||||
r.alignAttr(n.align[i])
|
||||
r.puts(">")
|
||||
r.inlineContent(cell)
|
||||
r.puts("</td>\n")
|
||||
}
|
||||
r.puts("</tr>\n")
|
||||
}
|
||||
r.puts("</tbody>\n")
|
||||
}
|
||||
r.puts("</table>\n")
|
||||
}
|
||||
}
|
||||
|
||||
// itemLike renders a container of a list-shaped structure: an item of a
|
||||
// list or a definition of a definition list. An empty container never
|
||||
// gains a newline, whatever the tightness.
|
||||
func (r *renderer) itemLike(tag string, n *Node) {
|
||||
firstIsParagraph := len(n.children) > 0 && n.children[0].kind == kindParagraph
|
||||
r.puts("<" + tag + ">")
|
||||
if len(n.children) > 0 && (!n.parent.tight || !firstIsParagraph) {
|
||||
r.puts("\n")
|
||||
}
|
||||
r.blocks(n.children)
|
||||
r.puts("</" + tag + ">\n")
|
||||
}
|
||||
|
||||
// taskInput renders the checkbox of a task list item, at the head of the
|
||||
// item's first paragraph.
|
||||
func taskInput(n *Node) string {
|
||||
if n.parent == nil || n.parent.kind != kindListItem || !n.parent.task {
|
||||
return ""
|
||||
}
|
||||
siblings := n.parent.children
|
||||
if siblings[0] != n {
|
||||
return ""
|
||||
}
|
||||
if n.parent.taskDone {
|
||||
return `<input checked="" disabled="" type="checkbox"> `
|
||||
}
|
||||
return `<input disabled="" type="checkbox"> `
|
||||
}
|
||||
|
||||
// alignAttr writes the alignment attribute of a table column. The
|
||||
// attribute value keeps the spelling the HTML vocabulary defines.
|
||||
func (r *renderer) alignAttr(a uint8) {
|
||||
switch a {
|
||||
case alignLeft:
|
||||
r.puts(` align="left"`)
|
||||
case alignCentre:
|
||||
r.puts(` align="center"`)
|
||||
case alignRight:
|
||||
r.puts(` align="right"`)
|
||||
}
|
||||
}
|
||||
|
||||
// inlineContent parses and renders the inline content of a leaf block.
|
||||
func (r *renderer) inlineContent(content []byte) {
|
||||
for _, n := range parseInlines(content, r.refs, r.footnotes) {
|
||||
r.renderInline(n)
|
||||
}
|
||||
}
|
||||
|
||||
func (r *renderer) renderInline(n *inline) {
|
||||
switch n.kind {
|
||||
case inlText:
|
||||
r.puts(escapeHTML(n.literal))
|
||||
case inlCode:
|
||||
r.puts("<code>")
|
||||
r.puts(escapeHTML(n.literal))
|
||||
r.puts("</code>")
|
||||
case inlRawHTML:
|
||||
r.puts(n.literal)
|
||||
case inlEmph:
|
||||
r.puts("<em>")
|
||||
r.inlineNodes(n.children)
|
||||
r.puts("</em>")
|
||||
case inlStrong:
|
||||
r.puts("<strong>")
|
||||
r.inlineNodes(n.children)
|
||||
r.puts("</strong>")
|
||||
case inlStrikethrough:
|
||||
r.puts("<del>")
|
||||
r.inlineNodes(n.children)
|
||||
r.puts("</del>")
|
||||
case inlLink:
|
||||
r.puts(`<a href="` + escapeURL(n.dest) + `"`)
|
||||
if n.hasTitle {
|
||||
r.puts(` title="` + escapeHTML(n.title) + `"`)
|
||||
}
|
||||
r.puts(">")
|
||||
r.inlineNodes(n.children)
|
||||
r.puts("</a>")
|
||||
case inlImage:
|
||||
r.puts(`<img src="` + escapeURL(n.dest) + `" alt="` + escapeHTML(plainText(n.children)) + `"`)
|
||||
if n.hasTitle {
|
||||
r.puts(` title="` + escapeHTML(n.title) + `"`)
|
||||
}
|
||||
r.puts(" />")
|
||||
case inlBreak:
|
||||
r.puts("<br />\n")
|
||||
case inlFootnoteRef:
|
||||
id := "fnref-" + strconv.Itoa(n.num)
|
||||
if n.occurrence > 1 {
|
||||
id += "-" + strconv.Itoa(n.occurrence)
|
||||
}
|
||||
num := strconv.Itoa(n.num)
|
||||
r.puts(`<sup class="footnote-ref"><a href="#fn-` + num + `" id="` + id + `" data-footnote-ref>` + num + `</a></sup>`)
|
||||
}
|
||||
}
|
||||
|
||||
// footnoteSection renders the definitions of every referenced footnote, in
|
||||
// the order of their first reference. The back reference lands at the end
|
||||
// of the definition's last paragraph.
|
||||
func (r *renderer) footnoteSection() {
|
||||
if len(r.footnotes.order) == 0 {
|
||||
return
|
||||
}
|
||||
r.puts("<section class=\"footnotes\" data-footnotes>\n<ol>\n")
|
||||
for i, def := range r.footnotes.order {
|
||||
num := strconv.Itoa(i + 1)
|
||||
r.puts(`<li id="fn-` + num + `">` + "\n")
|
||||
backref := ` <a href="#fnref-` + num + `" class="data-footnote-backref" aria-label="Back to reference ` + num + `">` + "↩" + `</a>`
|
||||
for j, c := range def.children {
|
||||
if c.kind == kindParagraph && j+1 == len(def.children) {
|
||||
r.puts("<p>")
|
||||
r.inlineContent(c.content)
|
||||
r.puts(backref)
|
||||
r.puts("</p>\n")
|
||||
continue
|
||||
}
|
||||
r.block(c)
|
||||
}
|
||||
r.puts("</li>\n")
|
||||
}
|
||||
r.puts("</ol>\n</section>\n")
|
||||
}
|
||||
|
||||
func (r *renderer) inlineNodes(nodes []*inline) {
|
||||
for _, n := range nodes {
|
||||
r.renderInline(n)
|
||||
}
|
||||
}
|
||||
|
||||
// plainText renders inline nodes without markup, for the alt text of an
|
||||
// image.
|
||||
func plainText(nodes []*inline) string {
|
||||
var b strings.Builder
|
||||
for _, n := range nodes {
|
||||
switch n.kind {
|
||||
case inlText, inlCode, inlRawHTML:
|
||||
b.WriteString(n.literal)
|
||||
case inlBreak:
|
||||
b.WriteByte('\n')
|
||||
default:
|
||||
b.WriteString(plainText(n.children))
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// urlSafe marks the bytes that stay literal in an escaped destination.
|
||||
const urlSafe = "!#$%()*+,-./:;=?@_~$"
|
||||
|
||||
// escapeURL escapes a link destination for an href or src attribute: the
|
||||
// ampersand and the apostrophe become entities, the bytes outside the safe
|
||||
// set become percent escapes.
|
||||
func escapeURL(s string) string {
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); i++ {
|
||||
c := s[i]
|
||||
switch {
|
||||
case c == '&':
|
||||
b.WriteString("&")
|
||||
case c == '\'':
|
||||
b.WriteString("'")
|
||||
case isAlnum(c) || strings.IndexByte(urlSafe, c) >= 0:
|
||||
b.WriteByte(c)
|
||||
default:
|
||||
b.WriteByte('%')
|
||||
b.WriteByte("0123456789ABCDEF"[c>>4])
|
||||
b.WriteByte("0123456789ABCDEF"[c&0x0f])
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// tightParent reports whether the node is a paragraph directly inside an
|
||||
// item of a tight list or a definition of a tight definition list.
|
||||
func tightParent(n *Node) bool {
|
||||
if n.parent == nil || n.parent.parent == nil {
|
||||
return false
|
||||
}
|
||||
switch {
|
||||
case n.parent.kind == kindListItem && n.parent.parent.kind == kindList:
|
||||
return n.parent.parent.tight
|
||||
case n.parent.kind == kindDefItem && n.parent.parent.kind == kindDefList:
|
||||
return n.parent.parent.tight
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// lastChild reports whether the node is the last child of its parent.
|
||||
func lastChild(n *Node) bool {
|
||||
siblings := n.parent.children
|
||||
return siblings[len(siblings)-1] == n
|
||||
}
|
||||
|
||||
// infoWord returns the first word of a code block's info string.
|
||||
func infoWord(info string) string {
|
||||
fields := strings.Fields(info)
|
||||
if len(fields) == 0 {
|
||||
return ""
|
||||
}
|
||||
return fields[0]
|
||||
}
|
||||
|
||||
var htmlEscaper = strings.NewReplacer(
|
||||
"&", "&",
|
||||
"<", "<",
|
||||
">", ">",
|
||||
`"`, """,
|
||||
)
|
||||
|
||||
// escapeHTML escapes plain text for HTML output.
|
||||
func escapeHTML(s string) string {
|
||||
return htmlEscaper.Replace(s)
|
||||
}
|
||||
@@ -0,0 +1,752 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package markdown
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"html"
|
||||
"strings"
|
||||
)
|
||||
|
||||
func isSpaceTab(c byte) bool { return c == ' ' || c == '\t' }
|
||||
|
||||
// isTagSpace marks the whitespace an inline HTML tag may contain between
|
||||
// its parts, line endings included.
|
||||
func isTagSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' }
|
||||
|
||||
func isAlpha(c byte) bool {
|
||||
return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z'
|
||||
}
|
||||
|
||||
func isAlnum(c byte) bool {
|
||||
return isAlpha(c) || c >= '0' && c <= '9'
|
||||
}
|
||||
|
||||
// allSpaceTab reports whether s is empty or holds only spaces and tabs.
|
||||
func allSpaceTab(s []byte) bool {
|
||||
for _, c := range s {
|
||||
if !isSpaceTab(c) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// scanATX recognises an ATX heading opener: one to six hashes followed by
|
||||
// a space, a tab or the end of the line. It returns the heading level.
|
||||
func scanATX(s []byte) (int, bool) {
|
||||
n := 0
|
||||
for n < len(s) && s[n] == '#' {
|
||||
n++
|
||||
}
|
||||
if n == 0 || n > 6 {
|
||||
return 0, false
|
||||
}
|
||||
if n < len(s) && !isSpaceTab(s[n]) {
|
||||
return 0, false
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// chopClosingHashes removes an ATX heading's closing sequence of hashes
|
||||
// together with the whitespace around it.
|
||||
func chopClosingHashes(s []byte) []byte {
|
||||
end := len(s)
|
||||
for end > 0 && isSpaceTab(s[end-1]) {
|
||||
end--
|
||||
}
|
||||
h := end
|
||||
for h > 0 && s[h-1] == '#' {
|
||||
h--
|
||||
}
|
||||
if h == end {
|
||||
return s[:end]
|
||||
}
|
||||
if h > 0 && !isSpaceTab(s[h-1]) {
|
||||
return s[:end]
|
||||
}
|
||||
for h > 0 && isSpaceTab(s[h-1]) {
|
||||
h--
|
||||
}
|
||||
return s[:h]
|
||||
}
|
||||
|
||||
// scanSetext recognises a setext heading underline: a run of equals or
|
||||
// dashes followed by nothing but whitespace. It returns the heading level.
|
||||
func scanSetext(s []byte) (int, bool) {
|
||||
if len(s) == 0 {
|
||||
return 0, false
|
||||
}
|
||||
c := s[0]
|
||||
if c != '=' && c != '-' {
|
||||
return 0, false
|
||||
}
|
||||
i := 0
|
||||
for i < len(s) && s[i] == c {
|
||||
i++
|
||||
}
|
||||
if !allSpaceTab(s[i:]) {
|
||||
return 0, false
|
||||
}
|
||||
if c == '=' {
|
||||
return 1, true
|
||||
}
|
||||
return 2, true
|
||||
}
|
||||
|
||||
// isThematicBreak reports whether the line is a thematic break: three or
|
||||
// more matching dashes, asterisks or underscores with optional whitespace
|
||||
// between them.
|
||||
func isThematicBreak(s []byte) bool {
|
||||
if len(s) == 0 {
|
||||
return false
|
||||
}
|
||||
c := s[0]
|
||||
if c != '-' && c != '*' && c != '_' {
|
||||
return false
|
||||
}
|
||||
count := 0
|
||||
for _, ch := range s {
|
||||
switch ch {
|
||||
case c:
|
||||
count++
|
||||
case ' ', '\t':
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return count >= 3
|
||||
}
|
||||
|
||||
// fenceRun counts the leading run of the fence character.
|
||||
func fenceRun(s []byte, c byte) int {
|
||||
n := 0
|
||||
for n < len(s) && s[n] == c {
|
||||
n++
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// scanOpenFence recognises a code fence opener: three or more backticks or
|
||||
// tildes. The info string of a backtick fence may not contain a backtick,
|
||||
// so such a line is not a fence at all.
|
||||
func scanOpenFence(s []byte) (byte, int, bool) {
|
||||
if len(s) == 0 {
|
||||
return 0, 0, false
|
||||
}
|
||||
c := s[0]
|
||||
if c != '`' && c != '~' {
|
||||
return 0, 0, false
|
||||
}
|
||||
n := fenceRun(s, c)
|
||||
if n < 3 {
|
||||
return 0, 0, false
|
||||
}
|
||||
if c == '`' && bytes.IndexByte(s[n:], '`') >= 0 {
|
||||
return 0, 0, false
|
||||
}
|
||||
return c, n, true
|
||||
}
|
||||
|
||||
// blockTagNames are the tag names whose open or closing tag starts an HTML
|
||||
// block of the sixth kind.
|
||||
var blockTagNames = map[string]bool{
|
||||
"address": true, "article": true, "aside": true, "base": true,
|
||||
"basefont": true, "blockquote": true, "body": true, "caption": true,
|
||||
"center": true, "col": true, "colgroup": true, "dd": true,
|
||||
"details": true, "dialog": true, "dir": true, "div": true,
|
||||
"dl": true, "dt": true, "fieldset": true, "figcaption": true,
|
||||
"figure": true, "footer": true, "form": true, "frame": true,
|
||||
"frameset": true, "h1": true, "h2": true, "h3": true, "h4": true,
|
||||
"h5": true, "h6": true, "head": true, "header": true, "hr": true,
|
||||
"html": true, "iframe": true, "legend": true, "li": true,
|
||||
"link": true, "main": true, "menu": true, "menuitem": true,
|
||||
"nav": true, "noframes": true, "ol": true, "optgroup": true,
|
||||
"option": true, "p": true, "param": true, "search": true,
|
||||
"section": true, "source": true, "summary": true, "table": true,
|
||||
"tbody": true, "td": true, "tfoot": true, "th": true, "thead": true,
|
||||
"title": true, "tr": true, "track": true, "ul": true,
|
||||
}
|
||||
|
||||
// scanHTMLBlockStart recognises an HTML block opener and returns its start
|
||||
// condition, 1 to 7, or 0. The seventh condition, a complete tag alone on
|
||||
// the line, may not interrupt a paragraph.
|
||||
func scanHTMLBlockStart(s []byte, inParagraph bool) int {
|
||||
if len(s) == 0 || s[0] != '<' {
|
||||
return 0
|
||||
}
|
||||
r := s[1:]
|
||||
for _, name := range [...]string{"script", "pre", "style", "textarea"} {
|
||||
if len(r) >= len(name) && bytes.EqualFold(r[:len(name)], []byte(name)) {
|
||||
after := r[len(name):]
|
||||
if len(after) == 0 || after[0] == ' ' || after[0] == '\t' || after[0] == '>' {
|
||||
return 1
|
||||
}
|
||||
}
|
||||
}
|
||||
if bytes.HasPrefix(r, []byte("!--")) {
|
||||
return 2
|
||||
}
|
||||
if len(r) > 0 && r[0] == '?' {
|
||||
return 3
|
||||
}
|
||||
if bytes.HasPrefix(r, []byte("![CDATA[")) {
|
||||
return 5
|
||||
}
|
||||
if len(r) > 1 && r[0] == '!' && isAlpha(r[1]) {
|
||||
return 4
|
||||
}
|
||||
if typeSixStart(r) {
|
||||
return 6
|
||||
}
|
||||
if !inParagraph && completeTag(r) {
|
||||
return 7
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// typeSixStart reports whether r, the line after '<', opens an HTML block
|
||||
// of the sixth kind: an optional slash, a known tag name and a boundary.
|
||||
func typeSixStart(r []byte) bool {
|
||||
i := 0
|
||||
if i < len(r) && r[i] == '/' {
|
||||
i++
|
||||
}
|
||||
start := i
|
||||
for i < len(r) && isAlpha(r[i]) {
|
||||
i++
|
||||
}
|
||||
if i == start || !blockTagNames[strings.ToLower(string(r[start:i]))] {
|
||||
return false
|
||||
}
|
||||
rest := r[i:]
|
||||
if len(rest) == 0 || isSpaceTab(rest[0]) || rest[0] == '>' {
|
||||
return true
|
||||
}
|
||||
return len(rest) >= 2 && rest[0] == '/' && rest[1] == '>'
|
||||
}
|
||||
|
||||
// completeTag reports whether r is a complete open or closing tag followed
|
||||
// by nothing but whitespace, per the HTML grammar CommonMark quotes.
|
||||
func completeTag(r []byte) bool {
|
||||
n := tagLength(r)
|
||||
return n > 0 && allSpaceTab(r[n:])
|
||||
}
|
||||
|
||||
// tagLength returns the length of the open or closing tag at the start of
|
||||
// r, including the final '>', or 0 when r does not begin with one.
|
||||
func tagLength(r []byte) int {
|
||||
i := 0
|
||||
closing := false
|
||||
if i < len(r) && r[i] == '/' {
|
||||
closing = true
|
||||
i++
|
||||
}
|
||||
if i >= len(r) || !isAlpha(r[i]) {
|
||||
return 0
|
||||
}
|
||||
for i < len(r) && (isAlnum(r[i]) || r[i] == '-') {
|
||||
i++
|
||||
}
|
||||
if closing {
|
||||
for i < len(r) && isTagSpace(r[i]) {
|
||||
i++
|
||||
}
|
||||
if i < len(r) && r[i] == '>' {
|
||||
return i + 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
for {
|
||||
j := i
|
||||
for j < len(r) && isTagSpace(r[j]) {
|
||||
j++
|
||||
}
|
||||
if j < len(r) && r[j] == '>' {
|
||||
return j + 1
|
||||
}
|
||||
if j+1 < len(r) && r[j] == '/' && r[j+1] == '>' {
|
||||
return j + 2
|
||||
}
|
||||
if j == i || j >= len(r) {
|
||||
return 0
|
||||
}
|
||||
i = j
|
||||
if i >= len(r) || !(isAlpha(r[i]) || r[i] == '_' || r[i] == ':') {
|
||||
return 0
|
||||
}
|
||||
for i < len(r) && (isAlnum(r[i]) || r[i] == '_' || r[i] == ':' || r[i] == '.' || r[i] == '-') {
|
||||
i++
|
||||
}
|
||||
k := i
|
||||
for k < len(r) && isTagSpace(r[k]) {
|
||||
k++
|
||||
}
|
||||
if k < len(r) && r[k] == '=' {
|
||||
k++
|
||||
for k < len(r) && isTagSpace(r[k]) {
|
||||
k++
|
||||
}
|
||||
if k >= len(r) {
|
||||
return 0
|
||||
}
|
||||
switch r[k] {
|
||||
case '"', '\'':
|
||||
q := r[k]
|
||||
k++
|
||||
for k < len(r) && r[k] != q {
|
||||
k++
|
||||
}
|
||||
if k >= len(r) {
|
||||
return 0
|
||||
}
|
||||
k++
|
||||
case '<', '>', '`', '=':
|
||||
return 0
|
||||
default:
|
||||
start := k
|
||||
for k < len(r) && !isTagSpace(r[k]) && r[k] != '"' && r[k] != '\'' && r[k] != '=' && r[k] != '<' && r[k] != '>' && r[k] != '`' {
|
||||
k++
|
||||
}
|
||||
if k == start {
|
||||
return 0
|
||||
}
|
||||
}
|
||||
i = k
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// scanRawHTML returns the length of the raw HTML construct at the start of
|
||||
// s: a comment, a processing instruction, a declaration, a CDATA section,
|
||||
// or an open or closing tag.
|
||||
func scanRawHTML(s []byte) int {
|
||||
if len(s) == 0 || s[0] != '<' {
|
||||
return 0
|
||||
}
|
||||
if bytes.HasPrefix(s, []byte("<!--")) {
|
||||
return scanHTMLComment(s)
|
||||
}
|
||||
if len(s) > 1 && s[1] == '?' {
|
||||
if i := bytes.Index(s, []byte("?>")); i >= 0 {
|
||||
return i + 2
|
||||
}
|
||||
return 0
|
||||
}
|
||||
if bytes.HasPrefix(s, []byte("<![CDATA[")) {
|
||||
if i := bytes.Index(s, []byte("]]>")); i >= 0 {
|
||||
return i + 3
|
||||
}
|
||||
return 0
|
||||
}
|
||||
if len(s) > 2 && s[1] == '!' && isAlpha(s[2]) {
|
||||
if i := bytes.IndexByte(s, '>'); i >= 0 {
|
||||
return i + 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
if n := tagLength(s[1:]); n > 0 {
|
||||
return n + 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// scanHTMLComment returns the length of the HTML comment at the start of
|
||||
// s, which runs from <!-- to the first -->. The text may not start with
|
||||
// '>' or '->' and may not end with '-'; the empty spellings are accepted.
|
||||
func scanHTMLComment(s []byte) int {
|
||||
if len(s) >= 5 && s[4] == '>' {
|
||||
return 5
|
||||
}
|
||||
if len(s) >= 6 && s[4] == '-' && s[5] == '>' {
|
||||
return 6
|
||||
}
|
||||
for i := 4; i < len(s); i++ {
|
||||
if !bytes.HasPrefix(s[i:], []byte("-->")) {
|
||||
continue
|
||||
}
|
||||
text := s[4:i]
|
||||
if len(text) == 0 {
|
||||
return i + 3
|
||||
}
|
||||
if text[len(text)-1] == '-' {
|
||||
return 0
|
||||
}
|
||||
if text[0] == '>' || (len(text) > 1 && text[0] == '-' && text[1] == '>') {
|
||||
return 0
|
||||
}
|
||||
return i + 3
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// scanAutolink recognises a URI autolink or an email autolink at the start
|
||||
// of s, returning its text, its destination and its length.
|
||||
func scanAutolink(s []byte) (text, dest string, n int, ok bool) {
|
||||
if len(s) == 0 || s[0] != '<' {
|
||||
return "", "", 0, false
|
||||
}
|
||||
// An email autolink: local part, one at sign, and a domain of labels.
|
||||
i := 1
|
||||
local := i
|
||||
for i < len(s) && isEmailByte(s[i]) {
|
||||
i++
|
||||
}
|
||||
if i > local && i < len(s) && s[i] == '@' {
|
||||
if end, domOK := scanEmailDomain(s, i+1); domOK && end < len(s) && s[end] == '>' {
|
||||
addr := string(s[1:end])
|
||||
return addr, "mailto:" + addr, end + 1, true
|
||||
}
|
||||
}
|
||||
// A URI autolink: scheme, colon, and a destination without whitespace
|
||||
// or angle brackets.
|
||||
i = 1
|
||||
schemeEnd := -1
|
||||
if i < len(s) && isAlpha(s[i]) {
|
||||
i++
|
||||
for i < len(s) && i <= 32 && (isAlnum(s[i]) || s[i] == '+' || s[i] == '-' || s[i] == '.') {
|
||||
i++
|
||||
}
|
||||
if i < len(s) && s[i] == ':' && i >= 3 {
|
||||
schemeEnd = i
|
||||
}
|
||||
}
|
||||
if schemeEnd < 0 {
|
||||
return "", "", 0, false
|
||||
}
|
||||
i = schemeEnd + 1
|
||||
for i < len(s) && s[i] != '>' {
|
||||
if s[i] <= ' ' || s[i] == '<' || s[i] == '>' {
|
||||
return "", "", 0, false
|
||||
}
|
||||
i++
|
||||
}
|
||||
if i >= len(s) || i == schemeEnd+1 {
|
||||
return "", "", 0, false
|
||||
}
|
||||
uri := string(s[1:i])
|
||||
return uri, uri, i + 1, true
|
||||
}
|
||||
|
||||
func isEmailByte(c byte) bool {
|
||||
return isAlnum(c) || strings.IndexByte(".!#$%&'*+/=?^_`{|}~-", c) >= 0
|
||||
}
|
||||
|
||||
// scanEmailDomain scans a domain of dot-separated labels, where a label
|
||||
// starts and ends with an alphanumeric, may hold dashes inside, and is at
|
||||
// most 63 bytes long.
|
||||
func scanEmailDomain(s []byte, i int) (int, bool) {
|
||||
end := 0
|
||||
for {
|
||||
if i >= len(s) || !isAlnum(s[i]) {
|
||||
return 0, false
|
||||
}
|
||||
start := i
|
||||
i++
|
||||
for i < len(s) && (isAlnum(s[i]) || s[i] == '-') {
|
||||
i++
|
||||
}
|
||||
for i > start+1 && s[i-1] == '-' {
|
||||
i--
|
||||
}
|
||||
if i-start > 63 {
|
||||
return 0, false
|
||||
}
|
||||
end = i
|
||||
if i < len(s) && s[i] == '.' {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
return end, true
|
||||
}
|
||||
}
|
||||
|
||||
// scanEntity recognises an HTML entity at pos and returns its decoded
|
||||
// text, the position after it, and whether one was there. Numeric
|
||||
// references follow CommonMark strictly: a malformed or out-of-range
|
||||
// number is no entity at all, while null and surrogate code points decode
|
||||
// to the replacement character.
|
||||
func scanEntity(src []byte, pos int) (string, int, bool) {
|
||||
return scanEntityAt(string(src[pos:]))
|
||||
}
|
||||
|
||||
func scanEntityAt(s string) (string, int, bool) {
|
||||
if len(s) < 3 || s[0] != '&' {
|
||||
return "", 0, false
|
||||
}
|
||||
if s[1] == '#' {
|
||||
i := 2
|
||||
base := 10
|
||||
if i < len(s) && (s[i] == 'x' || s[i] == 'X') {
|
||||
base = 16
|
||||
i++
|
||||
}
|
||||
start := i
|
||||
value := 0
|
||||
for i < len(s) {
|
||||
d := digitValue(s[i])
|
||||
if d < 0 || d >= base {
|
||||
break
|
||||
}
|
||||
value = value*base + d
|
||||
if value > 0x10FFFF {
|
||||
return "", 0, false
|
||||
}
|
||||
i++
|
||||
}
|
||||
if i == start || i >= len(s) || s[i] != ';' {
|
||||
return "", 0, false
|
||||
}
|
||||
i++
|
||||
r := rune(value)
|
||||
if r == 0 || (r >= 0xD800 && r <= 0xDFFF) {
|
||||
r = 0xFFFD
|
||||
}
|
||||
return string(r), i, true
|
||||
}
|
||||
limit := min(len(s), 33)
|
||||
for i := 1; i < limit; i++ {
|
||||
c := s[i]
|
||||
if c == ';' {
|
||||
slice := s[:i+1]
|
||||
if decoded := html.UnescapeString(slice); decoded != slice {
|
||||
return decoded, i + 1, true
|
||||
}
|
||||
return "", 0, false
|
||||
}
|
||||
if !isAlnum(c) {
|
||||
return "", 0, false
|
||||
}
|
||||
}
|
||||
return "", 0, false
|
||||
}
|
||||
|
||||
func digitValue(c byte) int {
|
||||
switch {
|
||||
case c >= '0' && c <= '9':
|
||||
return int(c - '0')
|
||||
case c >= 'a' && c <= 'f':
|
||||
return int(c-'a') + 10
|
||||
case c >= 'A' && c <= 'F':
|
||||
return int(c-'A') + 10
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// unescapeText resolves backslash escapes and entities, as link
|
||||
// destinations, titles and code info strings are read.
|
||||
func unescapeText(s string) string {
|
||||
if !strings.ContainsAny(s, "\\&") {
|
||||
return s
|
||||
}
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); {
|
||||
c := s[i]
|
||||
if c == '\\' && i+1 < len(s) && isASCIIPunct(s[i+1]) {
|
||||
b.WriteByte(s[i+1])
|
||||
i += 2
|
||||
continue
|
||||
}
|
||||
if c == '&' {
|
||||
if decoded, n, ok := scanEntityAt(s[i:]); ok {
|
||||
b.WriteString(decoded)
|
||||
i += n
|
||||
continue
|
||||
}
|
||||
}
|
||||
b.WriteByte(c)
|
||||
i++
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// htmlBlockEnds reports whether the line ends an HTML block of the given
|
||||
// start condition. The sixth and seventh conditions end on a blank line,
|
||||
// which the parser handles without this check.
|
||||
func htmlBlockEnds(t int, line []byte) bool {
|
||||
switch t {
|
||||
case 1:
|
||||
lower := bytes.ToLower(line)
|
||||
return bytes.Contains(lower, []byte("</script>")) ||
|
||||
bytes.Contains(lower, []byte("</pre>")) ||
|
||||
bytes.Contains(lower, []byte("</style>")) ||
|
||||
bytes.Contains(lower, []byte("</textarea>"))
|
||||
case 2:
|
||||
return bytes.Contains(line, []byte("-->"))
|
||||
case 3:
|
||||
return bytes.Contains(line, []byte("?>"))
|
||||
case 4:
|
||||
return bytes.Contains(line, []byte(">"))
|
||||
case 5:
|
||||
return bytes.Contains(line, []byte("]]>"))
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// scanTableDelimiter recognises a table delimiter row: at least one pipe,
|
||||
// and cells of dashes with optional flanking colons. It returns the
|
||||
// alignment of every column, which also gives the column count.
|
||||
func scanTableDelimiter(line []byte) ([]uint8, bool) {
|
||||
hasPipe := false
|
||||
for i := 0; i < len(line); i++ {
|
||||
if line[i] == '\\' {
|
||||
i++
|
||||
continue
|
||||
}
|
||||
if line[i] == '|' {
|
||||
hasPipe = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasPipe {
|
||||
return nil, false
|
||||
}
|
||||
cells := splitTableRow(line)
|
||||
if len(cells) == 0 {
|
||||
return nil, false
|
||||
}
|
||||
aligns := make([]uint8, len(cells))
|
||||
for i, cell := range cells {
|
||||
a, ok := parseAlignCell(cell)
|
||||
if !ok {
|
||||
return nil, false
|
||||
}
|
||||
aligns[i] = a
|
||||
}
|
||||
return aligns, true
|
||||
}
|
||||
|
||||
// parseAlignCell reads one delimiter cell: dashes with an optional leading
|
||||
// and trailing colon.
|
||||
func parseAlignCell(cell []byte) (uint8, bool) {
|
||||
i := 0
|
||||
left := false
|
||||
if i < len(cell) && cell[i] == ':' {
|
||||
left = true
|
||||
i++
|
||||
}
|
||||
dashes := 0
|
||||
for i < len(cell) && cell[i] == '-' {
|
||||
dashes++
|
||||
i++
|
||||
}
|
||||
right := false
|
||||
if i < len(cell) && cell[i] == ':' {
|
||||
right = true
|
||||
i++
|
||||
}
|
||||
if dashes == 0 || i != len(cell) {
|
||||
return alignNone, false
|
||||
}
|
||||
switch {
|
||||
case left && right:
|
||||
return alignCentre, true
|
||||
case left:
|
||||
return alignLeft, true
|
||||
case right:
|
||||
return alignRight, true
|
||||
}
|
||||
return alignNone, true
|
||||
}
|
||||
|
||||
// splitTableRow splits a row into trimmed cells on unescaped pipes. The
|
||||
// empty cells produced by leading and trailing boundary pipes are dropped.
|
||||
// An escaped pipe resolves to a plain pipe here, before the inline parser
|
||||
// runs, so a code span in a cell never shows the backslash.
|
||||
func splitTableRow(line []byte) [][]byte {
|
||||
trimmed := bytes.TrimSpace(line)
|
||||
var cells [][]byte
|
||||
var cur []byte
|
||||
flush := func() {
|
||||
cells = append(cells, bytes.TrimSpace(cur))
|
||||
cur = nil
|
||||
}
|
||||
for i := 0; i < len(trimmed); {
|
||||
switch c := trimmed[i]; {
|
||||
case c == '\\' && i+1 < len(trimmed) && trimmed[i+1] == '|':
|
||||
cur = append(cur, '|')
|
||||
i += 2
|
||||
case c == '|':
|
||||
flush()
|
||||
i++
|
||||
default:
|
||||
cur = append(cur, c)
|
||||
i++
|
||||
}
|
||||
}
|
||||
flush()
|
||||
if len(cells) > 1 && len(cells[0]) == 0 {
|
||||
cells = cells[1:]
|
||||
}
|
||||
if len(cells) > 1 && len(cells[len(cells)-1]) == 0 {
|
||||
cells = cells[:len(cells)-1]
|
||||
}
|
||||
return cells
|
||||
}
|
||||
|
||||
// scanTaskMarker recognises a task list item marker: brackets around a
|
||||
// space or an x, followed by a space and content.
|
||||
func scanTaskMarker(s []byte) (checked bool, n int, ok bool) {
|
||||
if len(s) < 4 || s[0] != '[' || s[2] != ']' || !isSpaceTab(s[3]) {
|
||||
return false, 0, false
|
||||
}
|
||||
switch s[1] {
|
||||
case ' ':
|
||||
case 'x', 'X':
|
||||
checked = true
|
||||
default:
|
||||
return false, 0, false
|
||||
}
|
||||
for i := 3; i < len(s); i++ {
|
||||
if !isSpaceTab(s[i]) {
|
||||
return checked, 4, true
|
||||
}
|
||||
}
|
||||
return false, 0, false
|
||||
}
|
||||
|
||||
// scanFootnoteLabel reads the bracketed label of a footnote, the "[^label]"
|
||||
// spelling, returning the label and the length of the whole bracket. The
|
||||
// label holds no whitespace and no brackets.
|
||||
func scanFootnoteLabel(s []byte) (string, int, bool) {
|
||||
if len(s) < 4 || s[0] != '[' || s[1] != '^' {
|
||||
return "", 0, false
|
||||
}
|
||||
i := 2
|
||||
for i < len(s) {
|
||||
switch s[i] {
|
||||
case ']':
|
||||
if i == 2 {
|
||||
return "", 0, false
|
||||
}
|
||||
return string(s[2:i]), i + 1, true
|
||||
case '[', ' ', '\t', '\n':
|
||||
return "", 0, false
|
||||
}
|
||||
i++
|
||||
}
|
||||
return "", 0, false
|
||||
}
|
||||
|
||||
// scanFootnoteDefStart recognises a footnote definition opener, the
|
||||
// "[^label]:" spelling, returning the label and the length of the marker.
|
||||
func scanFootnoteDefStart(s []byte) (string, int, bool) {
|
||||
label, n, ok := scanFootnoteLabel(s)
|
||||
if !ok {
|
||||
return "", 0, false
|
||||
}
|
||||
if n >= len(s) || s[n] != ':' {
|
||||
return "", 0, false
|
||||
}
|
||||
if n+1 < len(s) && !isSpaceTab(s[n+1]) {
|
||||
return "", 0, false
|
||||
}
|
||||
return label, n + 1, true
|
||||
}
|
||||
|
||||
// scanDefMarker recognises a definition list marker: a colon followed by a
|
||||
// space or the end of the line.
|
||||
func scanDefMarker(s []byte) bool {
|
||||
return len(s) > 0 && s[0] == ':' && (len(s) == 1 || isSpaceTab(s[1]))
|
||||
}
|
||||
Reference in New Issue
Block a user