Files
scriptorium/internal/markdown/autolink.go
T

223 lines
6.0 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package markdown
import "bytes"
// The GFM extended autolinks: bare www., http://, https:// and ftp://
// addresses, and bare email addresses, recognised without angle brackets.
// The rules follow the GFM specification literally.
// isAutolinkStartByte reports whether the previous character allows an
// extended www or URL autolink to begin here: beginning of the inline
// source, whitespace, or one of the delimiting characters.
func isAutolinkStartByte(prev byte) bool {
return prev == 0 || isSpaceTab(prev) || prev == '\n' ||
prev == '*' || prev == '_' || prev == '~' || prev == '('
}
// isEmailLocalByte reports whether c may appear in the local part of an
// extended email autolink.
func isEmailLocalByte(c byte) bool {
return isAlnum(c) || c == '.' || c == '-' || c == '_' || c == '+'
}
// isEmailDomainByte reports whether c may appear in a domain segment of an
// extended email autolink.
func isEmailDomainByte(c byte) bool {
return isAlnum(c) || c == '-' || c == '_'
}
// isEmailPrevByte reports whether the character before a candidate would
// continue a longer email address, which disqualifies the candidate: the
// local part must be maximal.
func isEmailPrevByte(prev byte) bool {
return isAlnum(prev) || prev == '.' || prev == '-' || prev == '_' ||
prev == '+' || prev == '@'
}
// scanExtendedAutolink recognises an extended autolink at i, returning
// the length, the link text and the destination. The prev byte is the
// character before i; a www or URL candidate may only begin where the
// GFM spec allows it, and an email candidate must start a maximal local
// part.
func scanExtendedAutolink(src []byte, i int, prev byte) (n int, text, dest string, ok bool) {
if isAutolinkStartByte(prev) {
if n, end, ok := scanWWWOrURL(src, i); ok {
text := string(src[i:end])
dest := text
if src[i] == 'w' {
dest = "http://" + text
}
return n, text, dest, true
}
}
if !isEmailPrevByte(prev) {
if n, end, ok := scanEmailAutolink(src, i); ok {
addr := string(src[i:end])
return n, addr, "mailto:" + addr, true
}
}
return 0, "", "", false
}
// scanWWWOrURL recognises a bare www address or one with an explicit
// http, https or ftp scheme. It returns the length and the end offset.
func scanWWWOrURL(src []byte, i int) (int, int, bool) {
var schemeLen int
switch {
case bytes.HasPrefix(src[i:], []byte("www.")):
schemeLen = 4
case bytes.HasPrefix(src[i:], []byte("http://")):
schemeLen = 7
case bytes.HasPrefix(src[i:], []byte("https://")):
schemeLen = 8
case bytes.HasPrefix(src[i:], []byte("ftp://")):
schemeLen = 6
default:
return 0, 0, false
}
domainEnd, ok := scanValidDomain(src, i+schemeLen)
if !ok {
return 0, 0, false
}
// Zero or more non-space, non-< characters follow the domain.
end := domainEnd
for end < len(src) && src[end] != ' ' && src[end] != '\t' &&
src[end] != '\n' && src[end] != '<' {
end++
}
end = validateAutolinkPath(src, i, end)
if end <= i+schemeLen {
return 0, 0, false
}
return end - i, end, true
}
// scanValidDomain reads a GFM valid domain at i: one or more segments of
// alphanumerics, underscores and hyphens separated by periods, with at
// least one period, and no underscore in the last two segments. It
// returns the offset after the domain.
func scanValidDomain(src []byte, i int) (int, bool) {
var starts, ends []int
j := i
for {
start := j
for j < len(src) && (isAlnum(src[j]) || src[j] == '-' || src[j] == '_') {
j++
}
if j == start {
return 0, false
}
starts = append(starts, start)
ends = append(ends, j)
// A period only continues the domain when a segment follows it;
// a trailing period belongs to whatever comes after.
if j+1 < len(src) && src[j] == '.' &&
(isAlnum(src[j+1]) || src[j+1] == '-' || src[j+1] == '_') {
j++
continue
}
break
}
if len(ends) < 2 {
return 0, false
}
for k := len(ends) - 2; k < len(ends); k++ {
for c := starts[k]; c < ends[k]; c++ {
if src[c] == '_' {
return 0, false
}
}
}
return j, true
}
// validateAutolinkPath applies the extended autolink path validation to
// the candidate src[begin:end]: trailing punctuation is trimmed, an
// unbalanced closing parenthesis is trimmed, and a trailing entity
// reference is excluded.
func validateAutolinkPath(src []byte, begin, end int) int {
for end > begin {
switch src[end-1] {
case '?', '!', '.', ',', ':', '*', '_', '~':
end--
continue
}
break
}
for end > begin && src[end-1] == ')' {
opens, closes := 0, 0
for k := begin; k < end; k++ {
switch src[k] {
case '(':
opens++
case ')':
closes++
}
}
if closes <= opens {
break
}
end--
}
if end > begin && src[end-1] == ';' {
if amp := bytes.LastIndexByte(src[begin:end], '&'); amp >= 0 {
entity := src[begin+amp+1 : end-1]
valid := len(entity) > 0
for _, c := range entity {
if !isAlnum(c) {
valid = false
break
}
}
if valid {
end = begin + amp
}
}
}
return end
}
// scanEmailAutolink recognises an extended email autolink at i: a local
// part of alphanumerics and .-_+, an @, and a domain of alphanumerics,
// hyphens and underscores separated by at least one period, whose last
// character is not a hyphen or underscore. A trailing period is not part
// of the address.
func scanEmailAutolink(src []byte, i int) (int, int, bool) {
j := i
for j < len(src) && isEmailLocalByte(src[j]) {
j++
}
if j == i || j >= len(src) || src[j] != '@' {
return 0, 0, false
}
j++
segments := 0
for {
start := j
for j < len(src) && isEmailDomainByte(src[j]) {
j++
}
if j == start {
return 0, 0, false
}
segments++
// A period only continues the domain when a segment follows it;
// a trailing period is not part of the address.
if j+1 < len(src) && src[j] == '.' && isEmailDomainByte(src[j+1]) {
j++
continue
}
break
}
if segments < 2 {
return 0, 0, false
}
if last := src[j-1]; last == '-' || last == '_' {
return 0, 0, false
}
return j - i, j, true
}