148 lines
7.9 KiB
Go
148 lines
7.9 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package markdown
|
|
|
|
import "testing"
|
|
|
|
// inlineCorpus is the project's own hand-written corpus for the inline
|
|
// layer, on top of a paragraph unless the case states otherwise.
|
|
var inlineCorpus = []struct {
|
|
name string
|
|
input string
|
|
want string
|
|
}{
|
|
// Emphasis.
|
|
{"em asterisk", "*foo*", "<p><em>foo</em></p>\n"},
|
|
{"strong asterisk", "**foo**", "<p><strong>foo</strong></p>\n"},
|
|
{"em and strong", "***foo***", "<p><em><strong>foo</strong></em></p>\n"},
|
|
{"strong with inner em", "**foo *bar* baz**", "<p><strong>foo <em>bar</em> baz</strong></p>\n"},
|
|
{"intraword asterisk", "foo*bar*baz", "<p>foo<em>bar</em>baz</p>\n"},
|
|
{"intraword underscore stays", "foo_bar_baz", "<p>foo_bar_baz</p>\n"},
|
|
{"em underscore with spaces", "_foo bar_", "<p><em>foo bar</em></p>\n"},
|
|
{"underscore cannot open after space", "_ foo_", "<p>_ foo_</p>\n"},
|
|
{"literal asterisk when spaced", "a * foo *", "<p>a * foo *</p>\n"},
|
|
{"escaped asterisk", "\\*not em\\*", "<p>*not em*</p>\n"},
|
|
{"escaped backslash then em", "\\\\*foo*", "<p>\\<em>foo</em></p>\n"},
|
|
{"lone closer stays", "a *", "<p>a *</p>\n"},
|
|
{"em inside word boundaries", "a*b*c", "<p>a<em>b</em>c</p>\n"},
|
|
{"em holding strong", "*foo**bar**baz*", "<p><em>foo<strong>bar</strong>baz</em></p>\n"},
|
|
{"strong inside em with text", "***foo** bar*", "<p><em><strong>foo</strong> bar</em></p>\n"},
|
|
{"em inside strong at end", "**foo *bar***", "<p><strong>foo <em>bar</em></strong></p>\n"},
|
|
{"unmatched inner run stays", "*foo**bar*", "<p><em>foo**bar</em></p>\n"},
|
|
{"intraword digits", "5*6*78", "<p>5<em>6</em>78</p>\n"},
|
|
{"underscore opens before punctuation", "_(bar)_", "<p><em>(bar)</em></p>\n"},
|
|
{"intraword underscore before punctuation stays", "foo_(bar)_", "<p>foo_(bar)_</p>\n"},
|
|
|
|
// Code spans.
|
|
{"code span", "`foo`", "<p><code>foo</code></p>\n"},
|
|
{"code span strips one space margin", "` foo `", "<p><code>foo</code></p>\n"},
|
|
{"code span keeps margin with doubles", "`` foo ``", "<p><code> foo </code></p>\n"},
|
|
{"code span double backticks", "``foo ` bar``", "<p><code>foo ` bar</code></p>\n"},
|
|
{"code span has no escapes", "`foo\\`bar`", "<p><code>foo\\</code>bar`</p>\n"},
|
|
{"code span unmatched", "foo ` bar", "<p>foo ` bar</p>\n"},
|
|
{"code span escapes markup", "`*em*`", "<p><code>*em*</code></p>\n"},
|
|
|
|
// Links and images.
|
|
{"inline link", "[foo](/uri)", "<p><a href=\"/uri\">foo</a></p>\n"},
|
|
{"inline link with title", "[foo](/uri \"title\")", "<p><a href=\"/uri\" title=\"title\">foo</a></p>\n"},
|
|
{"inline link single quoted title", "[foo](/uri 'title')", "<p><a href=\"/uri\" title=\"title\">foo</a></p>\n"},
|
|
{"inline link empty destination", "[foo]()", "<p><a href=\"\">foo</a></p>\n"},
|
|
{"angle destination with space", "[foo](<my uri>)", "<p><a href=\"my%20uri\">foo</a></p>\n"},
|
|
{"trailing paren stays text", "[foo](bar))", "<p><a href=\"bar\">foo</a>)</p>\n"},
|
|
{"em inside link", "[*foo*](/uri)", "<p><a href=\"/uri\"><em>foo</em></a></p>\n"},
|
|
{"image with title", "", "<p><img src=\"/url\" alt=\"foo\" title=\"title\" /></p>\n"},
|
|
{"image inside link", "[](page)", "<p><a href=\"page\"><img src=\"img\" alt=\"alt\" /></a></p>\n"},
|
|
{"no nested links", "[a [b](x)](y)", "<p>[a <a href=\"x\">b</a>](y)</p>\n"},
|
|
{"undefined reference stays", "[foo]", "<p>[foo]</p>\n"},
|
|
{"link destination escaped ampersand", "[a](/url?a=1&b=2)", "<p><a href=\"/url?a=1&b=2\">a</a></p>\n"},
|
|
|
|
// Autolinks.
|
|
{"uri autolink", "<http://example.com>", "<p><a href=\"http://example.com\">http://example.com</a></p>\n"},
|
|
{"email autolink", "<foo@bar.example.com>", "<p><a href=\"mailto:foo@bar.example.com\">foo@bar.example.com</a></p>\n"},
|
|
{"not an autolink", "<3>", "<p><3></p>\n"},
|
|
|
|
// Extended autolinks (GFM).
|
|
{"extended www", "Visit www.commonmark.org for more.",
|
|
"<p>Visit <a href=\"http://www.commonmark.org\">www.commonmark.org</a> for more.</p>\n"},
|
|
{"extended www with path", "go to www.commonmark.org/help today",
|
|
"<p>go to <a href=\"http://www.commonmark.org/help\">www.commonmark.org/help</a> today</p>\n"},
|
|
{"extended trailing punctuation", "see www.example.com.",
|
|
"<p>see <a href=\"http://www.example.com\">www.example.com</a>.</p>\n"},
|
|
{"extended paren balance", "www.example.com/query?q=(a+b)))",
|
|
"<p><a href=\"http://www.example.com/query?q=(a+b)\">www.example.com/query?q=(a+b)</a>))</p>\n"},
|
|
{"extended entity suffix", "www.example.com?q=x&hl;",
|
|
"<p><a href=\"http://www.example.com?q=x\">www.example.com?q=x</a>&hl;</p>\n"},
|
|
{"extended https", "open https://example.com/page",
|
|
"<p>open <a href=\"https://example.com/page\">https://example.com/page</a></p>\n"},
|
|
{"extended ftp", "ftp://files.example.org/pub",
|
|
"<p><a href=\"ftp://files.example.org/pub\">ftp://files.example.org/pub</a></p>\n"},
|
|
{"extended email", "write to a.b-c_d@example.com soon",
|
|
"<p>write to <a href=\"mailto:a.b-c_d@example.com\">a.b-c_d@example.com</a> soon</p>\n"},
|
|
{"extended email trailing dot", "mail me at user@example.net.",
|
|
"<p>mail me at <a href=\"mailto:user@example.net\">user@example.net</a>.</p>\n"},
|
|
{"plus before at only", "hello@mail+xyz.example is not, but hello+xyz@mail.example is",
|
|
"<p>hello@mail+xyz.example is not, but <a href=\"mailto:hello+xyz@mail.example\">hello+xyz@mail.example</a> is</p>\n"},
|
|
{"underscore banned in last segments", "www.un_der_score.org stays text",
|
|
"<p>www.un_der_score.org stays text</p>\n"},
|
|
{"no autolink mid word", "awww.example.com stays text",
|
|
"<p>awww.example.com stays text</p>\n"},
|
|
|
|
// Raw inline HTML and entities.
|
|
{"raw inline html", "a <b>c</b> d", "<p>a <b>c</b> d</p>\n"},
|
|
{"html comment inline", "a <!-- c --> b", "<p>a <!-- c --> b</p>\n"},
|
|
{"entity ampersand", "AT&T", "<p>AT&T</p>\n"},
|
|
{"entity numeric", "#", "<p>#</p>\n"},
|
|
{"entity hex", """, "<p>"</p>\n"},
|
|
{"bare ampersand", "AT&T", "<p>AT&T</p>\n"},
|
|
{"not an entity", "&x;", "<p>&x;</p>\n"},
|
|
{"less than escaped", "a < b", "<p>a < b</p>\n"},
|
|
|
|
// Breaks.
|
|
{"soft break", "foo\nbar", "<p>foo\nbar</p>\n"},
|
|
{"hard break spaces", "foo \nbar", "<p>foo<br />\nbar</p>\n"},
|
|
{"hard break backslash", "foo\\\nbar", "<p>foo<br />\nbar</p>\n"},
|
|
{"one trailing space is soft", "foo \nbar", "<p>foo\nbar</p>\n"},
|
|
{"trailing spaces at end dropped", "foo ", "<p>foo</p>\n"},
|
|
}
|
|
|
|
func TestInlineCorpus(t *testing.T) {
|
|
for _, tc := range inlineCorpus {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
got := string(RenderHTML([]byte(tc.input)))
|
|
if got != tc.want {
|
|
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// Reference links need definitions from earlier blocks, so these cases
|
|
// carry multi-block inputs.
|
|
func TestReferenceLinks(t *testing.T) {
|
|
cases := []struct {
|
|
name string
|
|
input string
|
|
want string
|
|
}{
|
|
{"explicit reference", "[foo][bar]\n\n[bar]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
|
{"collapsed reference", "[foo][]\n\n[foo]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
|
{"shortcut reference", "[foo]\n\n[foo]: /url", "<p><a href=\"/url\">foo</a></p>\n"},
|
|
{"reference with title", "[foo]\n\n[foo]: /url \"the title\"", "<p><a href=\"/url\" title=\"the title\">foo</a></p>\n"},
|
|
{"reference label case folded", "[Foo]\n\n[foo]: /url", "<p><a href=\"/url\">Foo</a></p>\n"},
|
|
{"image reference", "![foo]\n\n[foo]: /url", "<p><img src=\"/url\" alt=\"foo\" /></p>\n"},
|
|
{"shortcut takes whole text", "[foo *bar*]\n\n[foo *bar*]: /url", "<p><a href=\"/url\">foo <em>bar</em></a></p>\n"},
|
|
{"inline beats reference", "[foo](/inline)\n\n[foo]: /ref", "<p><a href=\"/inline\">foo</a></p>\n"},
|
|
{"link in heading", "# [foo](/uri)", "<h1><a href=\"/uri\">foo</a></h1>\n"},
|
|
{"code span in heading", "## a `b` c", "<h2>a <code>b</code> c</h2>\n"},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
got := string(RenderHTML([]byte(tc.input)))
|
|
if got != tc.want {
|
|
t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|