Files

753 lines
17 KiB
Go
Raw Permalink Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package markdown
import (
"bytes"
"html"
"strings"
)
func isSpaceTab(c byte) bool { return c == ' ' || c == '\t' }
// isTagSpace marks the whitespace an inline HTML tag may contain between
// its parts, line endings included.
func isTagSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' }
func isAlpha(c byte) bool {
return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z'
}
func isAlnum(c byte) bool {
return isAlpha(c) || c >= '0' && c <= '9'
}
// allSpaceTab reports whether s is empty or holds only spaces and tabs.
func allSpaceTab(s []byte) bool {
for _, c := range s {
if !isSpaceTab(c) {
return false
}
}
return true
}
// scanATX recognises an ATX heading opener: one to six hashes followed by
// a space, a tab or the end of the line. It returns the heading level.
func scanATX(s []byte) (int, bool) {
n := 0
for n < len(s) && s[n] == '#' {
n++
}
if n == 0 || n > 6 {
return 0, false
}
if n < len(s) && !isSpaceTab(s[n]) {
return 0, false
}
return n, true
}
// chopClosingHashes removes an ATX heading's closing sequence of hashes
// together with the whitespace around it.
func chopClosingHashes(s []byte) []byte {
end := len(s)
for end > 0 && isSpaceTab(s[end-1]) {
end--
}
h := end
for h > 0 && s[h-1] == '#' {
h--
}
if h == end {
return s[:end]
}
if h > 0 && !isSpaceTab(s[h-1]) {
return s[:end]
}
for h > 0 && isSpaceTab(s[h-1]) {
h--
}
return s[:h]
}
// scanSetext recognises a setext heading underline: a run of equals or
// dashes followed by nothing but whitespace. It returns the heading level.
func scanSetext(s []byte) (int, bool) {
if len(s) == 0 {
return 0, false
}
c := s[0]
if c != '=' && c != '-' {
return 0, false
}
i := 0
for i < len(s) && s[i] == c {
i++
}
if !allSpaceTab(s[i:]) {
return 0, false
}
if c == '=' {
return 1, true
}
return 2, true
}
// isThematicBreak reports whether the line is a thematic break: three or
// more matching dashes, asterisks or underscores with optional whitespace
// between them.
func isThematicBreak(s []byte) bool {
if len(s) == 0 {
return false
}
c := s[0]
if c != '-' && c != '*' && c != '_' {
return false
}
count := 0
for _, ch := range s {
switch ch {
case c:
count++
case ' ', '\t':
default:
return false
}
}
return count >= 3
}
// fenceRun counts the leading run of the fence character.
func fenceRun(s []byte, c byte) int {
n := 0
for n < len(s) && s[n] == c {
n++
}
return n
}
// scanOpenFence recognises a code fence opener: three or more backticks or
// tildes. The info string of a backtick fence may not contain a backtick,
// so such a line is not a fence at all.
func scanOpenFence(s []byte) (byte, int, bool) {
if len(s) == 0 {
return 0, 0, false
}
c := s[0]
if c != '`' && c != '~' {
return 0, 0, false
}
n := fenceRun(s, c)
if n < 3 {
return 0, 0, false
}
if c == '`' && bytes.IndexByte(s[n:], '`') >= 0 {
return 0, 0, false
}
return c, n, true
}
// blockTagNames are the tag names whose open or closing tag starts an HTML
// block of the sixth kind.
var blockTagNames = map[string]bool{
"address": true, "article": true, "aside": true, "base": true,
"basefont": true, "blockquote": true, "body": true, "caption": true,
"center": true, "col": true, "colgroup": true, "dd": true,
"details": true, "dialog": true, "dir": true, "div": true,
"dl": true, "dt": true, "fieldset": true, "figcaption": true,
"figure": true, "footer": true, "form": true, "frame": true,
"frameset": true, "h1": true, "h2": true, "h3": true, "h4": true,
"h5": true, "h6": true, "head": true, "header": true, "hr": true,
"html": true, "iframe": true, "legend": true, "li": true,
"link": true, "main": true, "menu": true, "menuitem": true,
"nav": true, "noframes": true, "ol": true, "optgroup": true,
"option": true, "p": true, "param": true, "search": true,
"section": true, "source": true, "summary": true, "table": true,
"tbody": true, "td": true, "tfoot": true, "th": true, "thead": true,
"title": true, "tr": true, "track": true, "ul": true,
}
// scanHTMLBlockStart recognises an HTML block opener and returns its start
// condition, 1 to 7, or 0. The seventh condition, a complete tag alone on
// the line, may not interrupt a paragraph.
func scanHTMLBlockStart(s []byte, inParagraph bool) int {
if len(s) == 0 || s[0] != '<' {
return 0
}
r := s[1:]
for _, name := range [...]string{"script", "pre", "style", "textarea"} {
if len(r) >= len(name) && bytes.EqualFold(r[:len(name)], []byte(name)) {
after := r[len(name):]
if len(after) == 0 || after[0] == ' ' || after[0] == '\t' || after[0] == '>' {
return 1
}
}
}
if bytes.HasPrefix(r, []byte("!--")) {
return 2
}
if len(r) > 0 && r[0] == '?' {
return 3
}
if bytes.HasPrefix(r, []byte("![CDATA[")) {
return 5
}
if len(r) > 1 && r[0] == '!' && isAlpha(r[1]) {
return 4
}
if typeSixStart(r) {
return 6
}
if !inParagraph && completeTag(r) {
return 7
}
return 0
}
// typeSixStart reports whether r, the line after '<', opens an HTML block
// of the sixth kind: an optional slash, a known tag name and a boundary.
func typeSixStart(r []byte) bool {
i := 0
if i < len(r) && r[i] == '/' {
i++
}
start := i
for i < len(r) && isAlpha(r[i]) {
i++
}
if i == start || !blockTagNames[strings.ToLower(string(r[start:i]))] {
return false
}
rest := r[i:]
if len(rest) == 0 || isSpaceTab(rest[0]) || rest[0] == '>' {
return true
}
return len(rest) >= 2 && rest[0] == '/' && rest[1] == '>'
}
// completeTag reports whether r is a complete open or closing tag followed
// by nothing but whitespace, per the HTML grammar CommonMark quotes.
func completeTag(r []byte) bool {
n := tagLength(r)
return n > 0 && allSpaceTab(r[n:])
}
// tagLength returns the length of the open or closing tag at the start of
// r, including the final '>', or 0 when r does not begin with one.
func tagLength(r []byte) int {
i := 0
closing := false
if i < len(r) && r[i] == '/' {
closing = true
i++
}
if i >= len(r) || !isAlpha(r[i]) {
return 0
}
for i < len(r) && (isAlnum(r[i]) || r[i] == '-') {
i++
}
if closing {
for i < len(r) && isTagSpace(r[i]) {
i++
}
if i < len(r) && r[i] == '>' {
return i + 1
}
return 0
}
for {
j := i
for j < len(r) && isTagSpace(r[j]) {
j++
}
if j < len(r) && r[j] == '>' {
return j + 1
}
if j+1 < len(r) && r[j] == '/' && r[j+1] == '>' {
return j + 2
}
if j == i || j >= len(r) {
return 0
}
i = j
if i >= len(r) || !(isAlpha(r[i]) || r[i] == '_' || r[i] == ':') {
return 0
}
for i < len(r) && (isAlnum(r[i]) || r[i] == '_' || r[i] == ':' || r[i] == '.' || r[i] == '-') {
i++
}
k := i
for k < len(r) && isTagSpace(r[k]) {
k++
}
if k < len(r) && r[k] == '=' {
k++
for k < len(r) && isTagSpace(r[k]) {
k++
}
if k >= len(r) {
return 0
}
switch r[k] {
case '"', '\'':
q := r[k]
k++
for k < len(r) && r[k] != q {
k++
}
if k >= len(r) {
return 0
}
k++
case '<', '>', '`', '=':
return 0
default:
start := k
for k < len(r) && !isTagSpace(r[k]) && r[k] != '"' && r[k] != '\'' && r[k] != '=' && r[k] != '<' && r[k] != '>' && r[k] != '`' {
k++
}
if k == start {
return 0
}
}
i = k
}
}
}
// scanRawHTML returns the length of the raw HTML construct at the start of
// s: a comment, a processing instruction, a declaration, a CDATA section,
// or an open or closing tag.
func scanRawHTML(s []byte) int {
if len(s) == 0 || s[0] != '<' {
return 0
}
if bytes.HasPrefix(s, []byte("<!--")) {
return scanHTMLComment(s)
}
if len(s) > 1 && s[1] == '?' {
if i := bytes.Index(s, []byte("?>")); i >= 0 {
return i + 2
}
return 0
}
if bytes.HasPrefix(s, []byte("<![CDATA[")) {
if i := bytes.Index(s, []byte("]]>")); i >= 0 {
return i + 3
}
return 0
}
if len(s) > 2 && s[1] == '!' && isAlpha(s[2]) {
if i := bytes.IndexByte(s, '>'); i >= 0 {
return i + 1
}
return 0
}
if n := tagLength(s[1:]); n > 0 {
return n + 1
}
return 0
}
// scanHTMLComment returns the length of the HTML comment at the start of
// s, which runs from <!-- to the first -->. The text may not start with
// '>' or '->' and may not end with '-'; the empty spellings are accepted.
func scanHTMLComment(s []byte) int {
if len(s) >= 5 && s[4] == '>' {
return 5
}
if len(s) >= 6 && s[4] == '-' && s[5] == '>' {
return 6
}
for i := 4; i < len(s); i++ {
if !bytes.HasPrefix(s[i:], []byte("-->")) {
continue
}
text := s[4:i]
if len(text) == 0 {
return i + 3
}
if text[len(text)-1] == '-' {
return 0
}
if text[0] == '>' || (len(text) > 1 && text[0] == '-' && text[1] == '>') {
return 0
}
return i + 3
}
return 0
}
// scanAutolink recognises a URI autolink or an email autolink at the start
// of s, returning its text, its destination and its length.
func scanAutolink(s []byte) (text, dest string, n int, ok bool) {
if len(s) == 0 || s[0] != '<' {
return "", "", 0, false
}
// An email autolink: local part, one at sign, and a domain of labels.
i := 1
local := i
for i < len(s) && isEmailByte(s[i]) {
i++
}
if i > local && i < len(s) && s[i] == '@' {
if end, domOK := scanEmailDomain(s, i+1); domOK && end < len(s) && s[end] == '>' {
addr := string(s[1:end])
return addr, "mailto:" + addr, end + 1, true
}
}
// A URI autolink: scheme, colon, and a destination without whitespace
// or angle brackets.
i = 1
schemeEnd := -1
if i < len(s) && isAlpha(s[i]) {
i++
for i < len(s) && i <= 32 && (isAlnum(s[i]) || s[i] == '+' || s[i] == '-' || s[i] == '.') {
i++
}
if i < len(s) && s[i] == ':' && i >= 3 {
schemeEnd = i
}
}
if schemeEnd < 0 {
return "", "", 0, false
}
i = schemeEnd + 1
for i < len(s) && s[i] != '>' {
if s[i] <= ' ' || s[i] == '<' || s[i] == '>' {
return "", "", 0, false
}
i++
}
if i >= len(s) || i == schemeEnd+1 {
return "", "", 0, false
}
uri := string(s[1:i])
return uri, uri, i + 1, true
}
func isEmailByte(c byte) bool {
return isAlnum(c) || strings.IndexByte(".!#$%&'*+/=?^_`{|}~-", c) >= 0
}
// scanEmailDomain scans a domain of dot-separated labels, where a label
// starts and ends with an alphanumeric, may hold dashes inside, and is at
// most 63 bytes long.
func scanEmailDomain(s []byte, i int) (int, bool) {
end := 0
for {
if i >= len(s) || !isAlnum(s[i]) {
return 0, false
}
start := i
i++
for i < len(s) && (isAlnum(s[i]) || s[i] == '-') {
i++
}
for i > start+1 && s[i-1] == '-' {
i--
}
if i-start > 63 {
return 0, false
}
end = i
if i < len(s) && s[i] == '.' {
i++
continue
}
return end, true
}
}
// scanEntity recognises an HTML entity at pos and returns its decoded
// text, the position after it, and whether one was there. Numeric
// references follow CommonMark strictly: a malformed or out-of-range
// number is no entity at all, while null and surrogate code points decode
// to the replacement character.
func scanEntity(src []byte, pos int) (string, int, bool) {
return scanEntityAt(string(src[pos:]))
}
func scanEntityAt(s string) (string, int, bool) {
if len(s) < 3 || s[0] != '&' {
return "", 0, false
}
if s[1] == '#' {
i := 2
base := 10
if i < len(s) && (s[i] == 'x' || s[i] == 'X') {
base = 16
i++
}
start := i
value := 0
for i < len(s) {
d := digitValue(s[i])
if d < 0 || d >= base {
break
}
value = value*base + d
if value > 0x10FFFF {
return "", 0, false
}
i++
}
if i == start || i >= len(s) || s[i] != ';' {
return "", 0, false
}
i++
r := rune(value)
if r == 0 || (r >= 0xD800 && r <= 0xDFFF) {
r = 0xFFFD
}
return string(r), i, true
}
limit := min(len(s), 33)
for i := 1; i < limit; i++ {
c := s[i]
if c == ';' {
slice := s[:i+1]
if decoded := html.UnescapeString(slice); decoded != slice {
return decoded, i + 1, true
}
return "", 0, false
}
if !isAlnum(c) {
return "", 0, false
}
}
return "", 0, false
}
func digitValue(c byte) int {
switch {
case c >= '0' && c <= '9':
return int(c - '0')
case c >= 'a' && c <= 'f':
return int(c-'a') + 10
case c >= 'A' && c <= 'F':
return int(c-'A') + 10
}
return -1
}
// unescapeText resolves backslash escapes and entities, as link
// destinations, titles and code info strings are read.
func unescapeText(s string) string {
if !strings.ContainsAny(s, "\\&") {
return s
}
var b strings.Builder
for i := 0; i < len(s); {
c := s[i]
if c == '\\' && i+1 < len(s) && isASCIIPunct(s[i+1]) {
b.WriteByte(s[i+1])
i += 2
continue
}
if c == '&' {
if decoded, n, ok := scanEntityAt(s[i:]); ok {
b.WriteString(decoded)
i += n
continue
}
}
b.WriteByte(c)
i++
}
return b.String()
}
// htmlBlockEnds reports whether the line ends an HTML block of the given
// start condition. The sixth and seventh conditions end on a blank line,
// which the parser handles without this check.
func htmlBlockEnds(t int, line []byte) bool {
switch t {
case 1:
lower := bytes.ToLower(line)
return bytes.Contains(lower, []byte("</script>")) ||
bytes.Contains(lower, []byte("</pre>")) ||
bytes.Contains(lower, []byte("</style>")) ||
bytes.Contains(lower, []byte("</textarea>"))
case 2:
return bytes.Contains(line, []byte("-->"))
case 3:
return bytes.Contains(line, []byte("?>"))
case 4:
return bytes.Contains(line, []byte(">"))
case 5:
return bytes.Contains(line, []byte("]]>"))
}
return false
}
// scanTableDelimiter recognises a table delimiter row: at least one pipe,
// and cells of dashes with optional flanking colons. It returns the
// alignment of every column, which also gives the column count.
func scanTableDelimiter(line []byte) ([]uint8, bool) {
hasPipe := false
for i := 0; i < len(line); i++ {
if line[i] == '\\' {
i++
continue
}
if line[i] == '|' {
hasPipe = true
break
}
}
if !hasPipe {
return nil, false
}
cells := splitTableRow(line)
if len(cells) == 0 {
return nil, false
}
aligns := make([]uint8, len(cells))
for i, cell := range cells {
a, ok := parseAlignCell(cell)
if !ok {
return nil, false
}
aligns[i] = a
}
return aligns, true
}
// parseAlignCell reads one delimiter cell: dashes with an optional leading
// and trailing colon.
func parseAlignCell(cell []byte) (uint8, bool) {
i := 0
left := false
if i < len(cell) && cell[i] == ':' {
left = true
i++
}
dashes := 0
for i < len(cell) && cell[i] == '-' {
dashes++
i++
}
right := false
if i < len(cell) && cell[i] == ':' {
right = true
i++
}
if dashes == 0 || i != len(cell) {
return alignNone, false
}
switch {
case left && right:
return alignCentre, true
case left:
return alignLeft, true
case right:
return alignRight, true
}
return alignNone, true
}
// splitTableRow splits a row into trimmed cells on unescaped pipes. The
// empty cells produced by leading and trailing boundary pipes are dropped.
// An escaped pipe resolves to a plain pipe here, before the inline parser
// runs, so a code span in a cell never shows the backslash.
func splitTableRow(line []byte) [][]byte {
trimmed := bytes.TrimSpace(line)
var cells [][]byte
var cur []byte
flush := func() {
cells = append(cells, bytes.TrimSpace(cur))
cur = nil
}
for i := 0; i < len(trimmed); {
switch c := trimmed[i]; {
case c == '\\' && i+1 < len(trimmed) && trimmed[i+1] == '|':
cur = append(cur, '|')
i += 2
case c == '|':
flush()
i++
default:
cur = append(cur, c)
i++
}
}
flush()
if len(cells) > 1 && len(cells[0]) == 0 {
cells = cells[1:]
}
if len(cells) > 1 && len(cells[len(cells)-1]) == 0 {
cells = cells[:len(cells)-1]
}
return cells
}
// scanTaskMarker recognises a task list item marker: brackets around a
// space or an x, followed by a space and content.
func scanTaskMarker(s []byte) (checked bool, n int, ok bool) {
if len(s) < 4 || s[0] != '[' || s[2] != ']' || !isSpaceTab(s[3]) {
return false, 0, false
}
switch s[1] {
case ' ':
case 'x', 'X':
checked = true
default:
return false, 0, false
}
for i := 3; i < len(s); i++ {
if !isSpaceTab(s[i]) {
return checked, 4, true
}
}
return false, 0, false
}
// scanFootnoteLabel reads the bracketed label of a footnote, the "[^label]"
// spelling, returning the label and the length of the whole bracket. The
// label holds no whitespace and no brackets.
func scanFootnoteLabel(s []byte) (string, int, bool) {
if len(s) < 4 || s[0] != '[' || s[1] != '^' {
return "", 0, false
}
i := 2
for i < len(s) {
switch s[i] {
case ']':
if i == 2 {
return "", 0, false
}
return string(s[2:i]), i + 1, true
case '[', ' ', '\t', '\n':
return "", 0, false
}
i++
}
return "", 0, false
}
// scanFootnoteDefStart recognises a footnote definition opener, the
// "[^label]:" spelling, returning the label and the length of the marker.
func scanFootnoteDefStart(s []byte) (string, int, bool) {
label, n, ok := scanFootnoteLabel(s)
if !ok {
return "", 0, false
}
if n >= len(s) || s[n] != ':' {
return "", 0, false
}
if n+1 < len(s) && !isSpaceTab(s[n+1]) {
return "", 0, false
}
return label, n + 1, true
}
// scanDefMarker recognises a definition list marker: a colon followed by a
// space or the end of the line.
func scanDefMarker(s []byte) bool {
return len(s) > 0 && s[0] == ':' && (len(s) == 1 || isSpaceTab(s[1]))
}