// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package markdown import ( "bytes" "html" "strings" ) func isSpaceTab(c byte) bool { return c == ' ' || c == '\t' } // isTagSpace marks the whitespace an inline HTML tag may contain between // its parts, line endings included. func isTagSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' } func isAlpha(c byte) bool { return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' } func isAlnum(c byte) bool { return isAlpha(c) || c >= '0' && c <= '9' } // allSpaceTab reports whether s is empty or holds only spaces and tabs. func allSpaceTab(s []byte) bool { for _, c := range s { if !isSpaceTab(c) { return false } } return true } // scanATX recognises an ATX heading opener: one to six hashes followed by // a space, a tab or the end of the line. It returns the heading level. func scanATX(s []byte) (int, bool) { n := 0 for n < len(s) && s[n] == '#' { n++ } if n == 0 || n > 6 { return 0, false } if n < len(s) && !isSpaceTab(s[n]) { return 0, false } return n, true } // chopClosingHashes removes an ATX heading's closing sequence of hashes // together with the whitespace around it. func chopClosingHashes(s []byte) []byte { end := len(s) for end > 0 && isSpaceTab(s[end-1]) { end-- } h := end for h > 0 && s[h-1] == '#' { h-- } if h == end { return s[:end] } if h > 0 && !isSpaceTab(s[h-1]) { return s[:end] } for h > 0 && isSpaceTab(s[h-1]) { h-- } return s[:h] } // scanSetext recognises a setext heading underline: a run of equals or // dashes followed by nothing but whitespace. It returns the heading level. func scanSetext(s []byte) (int, bool) { if len(s) == 0 { return 0, false } c := s[0] if c != '=' && c != '-' { return 0, false } i := 0 for i < len(s) && s[i] == c { i++ } if !allSpaceTab(s[i:]) { return 0, false } if c == '=' { return 1, true } return 2, true } // isThematicBreak reports whether the line is a thematic break: three or // more matching dashes, asterisks or underscores with optional whitespace // between them. func isThematicBreak(s []byte) bool { if len(s) == 0 { return false } c := s[0] if c != '-' && c != '*' && c != '_' { return false } count := 0 for _, ch := range s { switch ch { case c: count++ case ' ', '\t': default: return false } } return count >= 3 } // fenceRun counts the leading run of the fence character. func fenceRun(s []byte, c byte) int { n := 0 for n < len(s) && s[n] == c { n++ } return n } // scanOpenFence recognises a code fence opener: three or more backticks or // tildes. The info string of a backtick fence may not contain a backtick, // so such a line is not a fence at all. func scanOpenFence(s []byte) (byte, int, bool) { if len(s) == 0 { return 0, 0, false } c := s[0] if c != '`' && c != '~' { return 0, 0, false } n := fenceRun(s, c) if n < 3 { return 0, 0, false } if c == '`' && bytes.IndexByte(s[n:], '`') >= 0 { return 0, 0, false } return c, n, true } // blockTagNames are the tag names whose open or closing tag starts an HTML // block of the sixth kind. var blockTagNames = map[string]bool{ "address": true, "article": true, "aside": true, "base": true, "basefont": true, "blockquote": true, "body": true, "caption": true, "center": true, "col": true, "colgroup": true, "dd": true, "details": true, "dialog": true, "dir": true, "div": true, "dl": true, "dt": true, "fieldset": true, "figcaption": true, "figure": true, "footer": true, "form": true, "frame": true, "frameset": true, "h1": true, "h2": true, "h3": true, "h4": true, "h5": true, "h6": true, "head": true, "header": true, "hr": true, "html": true, "iframe": true, "legend": true, "li": true, "link": true, "main": true, "menu": true, "menuitem": true, "nav": true, "noframes": true, "ol": true, "optgroup": true, "option": true, "p": true, "param": true, "search": true, "section": true, "source": true, "summary": true, "table": true, "tbody": true, "td": true, "tfoot": true, "th": true, "thead": true, "title": true, "tr": true, "track": true, "ul": true, } // scanHTMLBlockStart recognises an HTML block opener and returns its start // condition, 1 to 7, or 0. The seventh condition, a complete tag alone on // the line, may not interrupt a paragraph. func scanHTMLBlockStart(s []byte, inParagraph bool) int { if len(s) == 0 || s[0] != '<' { return 0 } r := s[1:] for _, name := range [...]string{"script", "pre", "style", "textarea"} { if len(r) >= len(name) && bytes.EqualFold(r[:len(name)], []byte(name)) { after := r[len(name):] if len(after) == 0 || after[0] == ' ' || after[0] == '\t' || after[0] == '>' { return 1 } } } if bytes.HasPrefix(r, []byte("!--")) { return 2 } if len(r) > 0 && r[0] == '?' { return 3 } if bytes.HasPrefix(r, []byte("![CDATA[")) { return 5 } if len(r) > 1 && r[0] == '!' && isAlpha(r[1]) { return 4 } if typeSixStart(r) { return 6 } if !inParagraph && completeTag(r) { return 7 } return 0 } // typeSixStart reports whether r, the line after '<', opens an HTML block // of the sixth kind: an optional slash, a known tag name and a boundary. func typeSixStart(r []byte) bool { i := 0 if i < len(r) && r[i] == '/' { i++ } start := i for i < len(r) && isAlpha(r[i]) { i++ } if i == start || !blockTagNames[strings.ToLower(string(r[start:i]))] { return false } rest := r[i:] if len(rest) == 0 || isSpaceTab(rest[0]) || rest[0] == '>' { return true } return len(rest) >= 2 && rest[0] == '/' && rest[1] == '>' } // completeTag reports whether r is a complete open or closing tag followed // by nothing but whitespace, per the HTML grammar CommonMark quotes. func completeTag(r []byte) bool { n := tagLength(r) return n > 0 && allSpaceTab(r[n:]) } // tagLength returns the length of the open or closing tag at the start of // r, including the final '>', or 0 when r does not begin with one. func tagLength(r []byte) int { i := 0 closing := false if i < len(r) && r[i] == '/' { closing = true i++ } if i >= len(r) || !isAlpha(r[i]) { return 0 } for i < len(r) && (isAlnum(r[i]) || r[i] == '-') { i++ } if closing { for i < len(r) && isTagSpace(r[i]) { i++ } if i < len(r) && r[i] == '>' { return i + 1 } return 0 } for { j := i for j < len(r) && isTagSpace(r[j]) { j++ } if j < len(r) && r[j] == '>' { return j + 1 } if j+1 < len(r) && r[j] == '/' && r[j+1] == '>' { return j + 2 } if j == i || j >= len(r) { return 0 } i = j if i >= len(r) || !(isAlpha(r[i]) || r[i] == '_' || r[i] == ':') { return 0 } for i < len(r) && (isAlnum(r[i]) || r[i] == '_' || r[i] == ':' || r[i] == '.' || r[i] == '-') { i++ } k := i for k < len(r) && isTagSpace(r[k]) { k++ } if k < len(r) && r[k] == '=' { k++ for k < len(r) && isTagSpace(r[k]) { k++ } if k >= len(r) { return 0 } switch r[k] { case '"', '\'': q := r[k] k++ for k < len(r) && r[k] != q { k++ } if k >= len(r) { return 0 } k++ case '<', '>', '`', '=': return 0 default: start := k for k < len(r) && !isTagSpace(r[k]) && r[k] != '"' && r[k] != '\'' && r[k] != '=' && r[k] != '<' && r[k] != '>' && r[k] != '`' { k++ } if k == start { return 0 } } i = k } } } // scanRawHTML returns the length of the raw HTML construct at the start of // s: a comment, a processing instruction, a declaration, a CDATA section, // or an open or closing tag. func scanRawHTML(s []byte) int { if len(s) == 0 || s[0] != '<' { return 0 } if bytes.HasPrefix(s, []byte(". The text may not start with // '>' or '->' and may not end with '-'; the empty spellings are accepted. func scanHTMLComment(s []byte) int { if len(s) >= 5 && s[4] == '>' { return 5 } if len(s) >= 6 && s[4] == '-' && s[5] == '>' { return 6 } for i := 4; i < len(s); i++ { if !bytes.HasPrefix(s[i:], []byte("-->")) { continue } text := s[4:i] if len(text) == 0 { return i + 3 } if text[len(text)-1] == '-' { return 0 } if text[0] == '>' || (len(text) > 1 && text[0] == '-' && text[1] == '>') { return 0 } return i + 3 } return 0 } // scanAutolink recognises a URI autolink or an email autolink at the start // of s, returning its text, its destination and its length. func scanAutolink(s []byte) (text, dest string, n int, ok bool) { if len(s) == 0 || s[0] != '<' { return "", "", 0, false } // An email autolink: local part, one at sign, and a domain of labels. i := 1 local := i for i < len(s) && isEmailByte(s[i]) { i++ } if i > local && i < len(s) && s[i] == '@' { if end, domOK := scanEmailDomain(s, i+1); domOK && end < len(s) && s[end] == '>' { addr := string(s[1:end]) return addr, "mailto:" + addr, end + 1, true } } // A URI autolink: scheme, colon, and a destination without whitespace // or angle brackets. i = 1 schemeEnd := -1 if i < len(s) && isAlpha(s[i]) { i++ for i < len(s) && i <= 32 && (isAlnum(s[i]) || s[i] == '+' || s[i] == '-' || s[i] == '.') { i++ } if i < len(s) && s[i] == ':' && i >= 3 { schemeEnd = i } } if schemeEnd < 0 { return "", "", 0, false } i = schemeEnd + 1 for i < len(s) && s[i] != '>' { if s[i] <= ' ' || s[i] == '<' || s[i] == '>' { return "", "", 0, false } i++ } if i >= len(s) || i == schemeEnd+1 { return "", "", 0, false } uri := string(s[1:i]) return uri, uri, i + 1, true } func isEmailByte(c byte) bool { return isAlnum(c) || strings.IndexByte(".!#$%&'*+/=?^_`{|}~-", c) >= 0 } // scanEmailDomain scans a domain of dot-separated labels, where a label // starts and ends with an alphanumeric, may hold dashes inside, and is at // most 63 bytes long. func scanEmailDomain(s []byte, i int) (int, bool) { end := 0 for { if i >= len(s) || !isAlnum(s[i]) { return 0, false } start := i i++ for i < len(s) && (isAlnum(s[i]) || s[i] == '-') { i++ } for i > start+1 && s[i-1] == '-' { i-- } if i-start > 63 { return 0, false } end = i if i < len(s) && s[i] == '.' { i++ continue } return end, true } } // scanEntity recognises an HTML entity at pos and returns its decoded // text, the position after it, and whether one was there. Numeric // references follow CommonMark strictly: a malformed or out-of-range // number is no entity at all, while null and surrogate code points decode // to the replacement character. func scanEntity(src []byte, pos int) (string, int, bool) { return scanEntityAt(string(src[pos:])) } func scanEntityAt(s string) (string, int, bool) { if len(s) < 3 || s[0] != '&' { return "", 0, false } if s[1] == '#' { i := 2 base := 10 if i < len(s) && (s[i] == 'x' || s[i] == 'X') { base = 16 i++ } start := i value := 0 for i < len(s) { d := digitValue(s[i]) if d < 0 || d >= base { break } value = value*base + d if value > 0x10FFFF { return "", 0, false } i++ } if i == start || i >= len(s) || s[i] != ';' { return "", 0, false } i++ r := rune(value) if r == 0 || (r >= 0xD800 && r <= 0xDFFF) { r = 0xFFFD } return string(r), i, true } limit := min(len(s), 33) for i := 1; i < limit; i++ { c := s[i] if c == ';' { slice := s[:i+1] if decoded := html.UnescapeString(slice); decoded != slice { return decoded, i + 1, true } return "", 0, false } if !isAlnum(c) { return "", 0, false } } return "", 0, false } func digitValue(c byte) int { switch { case c >= '0' && c <= '9': return int(c - '0') case c >= 'a' && c <= 'f': return int(c-'a') + 10 case c >= 'A' && c <= 'F': return int(c-'A') + 10 } return -1 } // unescapeText resolves backslash escapes and entities, as link // destinations, titles and code info strings are read. func unescapeText(s string) string { if !strings.ContainsAny(s, "\\&") { return s } var b strings.Builder for i := 0; i < len(s); { c := s[i] if c == '\\' && i+1 < len(s) && isASCIIPunct(s[i+1]) { b.WriteByte(s[i+1]) i += 2 continue } if c == '&' { if decoded, n, ok := scanEntityAt(s[i:]); ok { b.WriteString(decoded) i += n continue } } b.WriteByte(c) i++ } return b.String() } // htmlBlockEnds reports whether the line ends an HTML block of the given // start condition. The sixth and seventh conditions end on a blank line, // which the parser handles without this check. func htmlBlockEnds(t int, line []byte) bool { switch t { case 1: lower := bytes.ToLower(line) return bytes.Contains(lower, []byte("")) || bytes.Contains(lower, []byte("")) || bytes.Contains(lower, []byte("")) || bytes.Contains(lower, []byte("")) case 2: return bytes.Contains(line, []byte("-->")) case 3: return bytes.Contains(line, []byte("?>")) case 4: return bytes.Contains(line, []byte(">")) case 5: return bytes.Contains(line, []byte("]]>")) } return false } // scanTableDelimiter recognises a table delimiter row: at least one pipe, // and cells of dashes with optional flanking colons. It returns the // alignment of every column, which also gives the column count. func scanTableDelimiter(line []byte) ([]uint8, bool) { hasPipe := false for i := 0; i < len(line); i++ { if line[i] == '\\' { i++ continue } if line[i] == '|' { hasPipe = true break } } if !hasPipe { return nil, false } cells := splitTableRow(line) if len(cells) == 0 { return nil, false } aligns := make([]uint8, len(cells)) for i, cell := range cells { a, ok := parseAlignCell(cell) if !ok { return nil, false } aligns[i] = a } return aligns, true } // parseAlignCell reads one delimiter cell: dashes with an optional leading // and trailing colon. func parseAlignCell(cell []byte) (uint8, bool) { i := 0 left := false if i < len(cell) && cell[i] == ':' { left = true i++ } dashes := 0 for i < len(cell) && cell[i] == '-' { dashes++ i++ } right := false if i < len(cell) && cell[i] == ':' { right = true i++ } if dashes == 0 || i != len(cell) { return alignNone, false } switch { case left && right: return alignCentre, true case left: return alignLeft, true case right: return alignRight, true } return alignNone, true } // splitTableRow splits a row into trimmed cells on unescaped pipes. The // empty cells produced by leading and trailing boundary pipes are dropped. // An escaped pipe resolves to a plain pipe here, before the inline parser // runs, so a code span in a cell never shows the backslash. func splitTableRow(line []byte) [][]byte { trimmed := bytes.TrimSpace(line) var cells [][]byte var cur []byte flush := func() { cells = append(cells, bytes.TrimSpace(cur)) cur = nil } for i := 0; i < len(trimmed); { switch c := trimmed[i]; { case c == '\\' && i+1 < len(trimmed) && trimmed[i+1] == '|': cur = append(cur, '|') i += 2 case c == '|': flush() i++ default: cur = append(cur, c) i++ } } flush() if len(cells) > 1 && len(cells[0]) == 0 { cells = cells[1:] } if len(cells) > 1 && len(cells[len(cells)-1]) == 0 { cells = cells[:len(cells)-1] } return cells } // scanTaskMarker recognises a task list item marker: brackets around a // space or an x, followed by a space and content. func scanTaskMarker(s []byte) (checked bool, n int, ok bool) { if len(s) < 4 || s[0] != '[' || s[2] != ']' || !isSpaceTab(s[3]) { return false, 0, false } switch s[1] { case ' ': case 'x', 'X': checked = true default: return false, 0, false } for i := 3; i < len(s); i++ { if !isSpaceTab(s[i]) { return checked, 4, true } } return false, 0, false } // scanFootnoteLabel reads the bracketed label of a footnote, the "[^label]" // spelling, returning the label and the length of the whole bracket. The // label holds no whitespace and no brackets. func scanFootnoteLabel(s []byte) (string, int, bool) { if len(s) < 4 || s[0] != '[' || s[1] != '^' { return "", 0, false } i := 2 for i < len(s) { switch s[i] { case ']': if i == 2 { return "", 0, false } return string(s[2:i]), i + 1, true case '[', ' ', '\t', '\n': return "", 0, false } i++ } return "", 0, false } // scanFootnoteDefStart recognises a footnote definition opener, the // "[^label]:" spelling, returning the label and the length of the marker. func scanFootnoteDefStart(s []byte) (string, int, bool) { label, n, ok := scanFootnoteLabel(s) if !ok { return "", 0, false } if n >= len(s) || s[n] != ':' { return "", 0, false } if n+1 < len(s) && !isSpaceTab(s[n+1]) { return "", 0, false } return label, n + 1, true } // scanDefMarker recognises a definition list marker: a colon followed by a // space or the end of the line. func scanDefMarker(s []byte) bool { return len(s) > 0 && s[0] == ':' && (len(s) == 1 || isSpaceTab(s[1])) }