// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0 // Package biblio turns the structured `refs` frontmatter of a post into // a rendered reference list, machine-readable citations, and the // resolver links a scholar expects. It is a leaf: it reads plain // metadata values and writes HTML and JSON shapes, importing no other // domain package. // // The shape is measured on the author's own volumes: numbered `[n]` // entries, inline `[n]` citations, and `doi:` and `arXiv:` identifiers // embedded in free text. The engine keeps that convention and makes it // live: each entry gets a target the inline citations point at, and // every identifier becomes a link to its resolver. package biblio import ( stdhtml "html" "regexp" "strconv" "strings" ) // Author is one cited author: a name, and the author's ORCID when the // reference records one. type Author struct { Name string `json:"name"` ORCID string `json:"orcid,omitempty"` } // Entry is one reference. A hand-written entry that does not break down // into the structured fields keeps its verbatim text in Raw, which the // renderer prints as given, with any identifier inside it linked: a // reference is never dropped or rewritten into something the author did // not write. type Entry struct { Num int `json:"num"` Authors []Author `json:"authors,omitempty"` Title string `json:"title,omitempty"` Venue string `json:"venue,omitempty"` Year string `json:"year,omitempty"` Volume string `json:"volume,omitempty"` Pages string `json:"pages,omitempty"` DOI string `json:"doi,omitempty"` ArXiv string `json:"arxiv,omitempty"` URL string `json:"url,omitempty"` Raw string `json:"raw,omitempty"` // Internal is the same-instance link: the API URL of the post whose // DOI this entry cites, set by the caller that knows the instance // (post.RefsLinked annotates each post's entries). Empty when the cited // work is not published here, or is the citing post itself. Internal string `json:"internal,omitempty"` } // href is the address the rendered link points at: the internal post // when the entry cites a work published in this instance, otherwise // the external resolver. func (e Entry) href() string { if e.Internal != "" { return e.Internal } return e.Link() } // Marker is the token an author places where the reference list should // be rendered; a body without it gets the list appended. const Marker = "[[refs]]" var ( markerRe = regexp.MustCompile(`(?s)

\s*\Q` + Marker + `\E\s*

`) doiTokenRe = regexp.MustCompile(`(?i)\bdoi:\s*(10\.[0-9]{4,9}/[^\s]+)`) arxivTokenRe = regexp.MustCompile(`(?i)\barXiv:\s*([A-Za-z0-9][A-Za-z0-9./:-]*)`) yearRe = regexp.MustCompile(`\b(1[89][0-9]{2}|20[0-9]{2})\b`) inlineRefRe = regexp.MustCompile(`\[(\d{1,3})\]`) ) // Parse reads the refs array out of flattened frontmatter tables. Each // element may carry authors, title, venue, year, volume, pages, doi, // arxiv, url or raw; authors is a list of strings or of {name, orcid} // tables. Entries without an explicit number are numbered by position. func Parse(refs []map[string]any) []Entry { var out []Entry for i, raw := range refs { entry := Entry{ Num: numOr(raw["num"], i+1), Title: strings.TrimSpace(str(raw["title"])), Venue: strings.TrimSpace(str(raw["venue"])), Year: strings.TrimSpace(str(raw["year"])), Volume: strings.TrimSpace(str(raw["volume"])), Pages: strings.TrimSpace(str(raw["pages"])), DOI: cleanDOI(str(raw["doi"])), ArXiv: cleanArxiv(str(raw["arxiv"])), URL: strings.TrimSpace(str(raw["url"])), Raw: strings.TrimSpace(str(raw["raw"])), } entry.Authors = parseAuthors(raw["authors"]) entry.enrichFromText() out = append(out, entry) } return out } func parseAuthors(v any) []Author { list, ok := v.([]any) if !ok { return nil } var out []Author for _, item := range list { switch value := item.(type) { case string: if name := strings.TrimSpace(value); name != "" { out = append(out, Author{Name: name}) } case map[string]any: author := Author{ Name: strings.TrimSpace(str(value["name"])), ORCID: strings.TrimSpace(str(value["orcid"])), } if author.Name != "" { out = append(out, author) } } } return out } func numOr(v any, fallback int) int { switch n := v.(type) { case int64: return int(n) case int: return n case float64: return int(n) case string: if parsed, err := strconv.Atoi(strings.TrimSpace(n)); err == nil { return parsed } } return fallback } func str(v any) string { s, _ := v.(string) return s } // cleanDOI accepts a bare identifier, a doi: prefix or a resolver URL // and returns the bare form. func cleanDOI(s string) string { s = strings.TrimSpace(s) if s == "" { return "" } if m := doiTokenRe.FindStringSubmatch(s); m != nil { return strings.TrimRight(m[1], ".,;)") } s = strings.TrimPrefix(strings.TrimPrefix(s, "https://doi.org/"), "doi:") return strings.TrimRight(strings.TrimSpace(s), ".,;)") } func cleanArxiv(s string) string { s = strings.TrimSpace(s) if s == "" { return "" } if m := arxivTokenRe.FindStringSubmatch(s); m != nil { s = m[1] } s = strings.TrimRight(strings.TrimPrefix(strings.TrimPrefix(s, "arXiv:"), "https://arxiv.org/abs/"), ".,;)") if !strings.HasPrefix(s, "arXiv:") && s != "" { // Already a bare id; accept anything identifier-shaped. if strings.ContainsAny(s, " \t<\"") { return "" } return s } return "" } // stripTokens removes the identifier tokens from the entry's own text, // used once an entry is structured: the renderer puts the identifier at // the end as a link, and it must not also sit in the title as prose. func (e *Entry) stripTokens() { if e.Title == "" && len(e.Authors) == 0 { return // a raw entry prints its tokens inline as links instead } e.Title = tidyTokens(e.Title) e.Venue = tidyTokens(e.Venue) } func tidyTokens(s string) string { s = doiTokenRe.ReplaceAllString(s, "") s = arxivTokenRe.ReplaceAllString(s, "") s = strings.ReplaceAll(s, ", ,", ",") s = strings.TrimSpace(s) s = strings.TrimRight(s, ",;") return strings.TrimSpace(s) } // enrichFromText recovers identifiers and a year a hand-written entry // kept in its free text, so a migrated volume gains live links without // every field being split by hand. func (e *Entry) enrichFromText() { joined := strings.Join([]string{e.Title, e.Venue, e.Raw}, " ") if e.DOI == "" { if m := doiTokenRe.FindStringSubmatch(joined); m != nil { e.DOI = strings.TrimRight(m[1], ".,;)") e.stripTokens() } } if e.ArXiv == "" { if m := arxivTokenRe.FindStringSubmatch(joined); m != nil { e.ArXiv = m[1] e.stripTokens() } } // The year is a structured field only when the entry is structured: // a raw entry already prints its year inside its own text. if e.Year == "" && (e.Title != "" || len(e.Authors) > 0) { if m := yearRe.FindStringSubmatch(joined); m != nil { e.Year = m[1] } } } // Link returns the resolver URL for this entry: DOI first, then arXiv, // then a plain URL. Empty when the entry carries no identifier. func (e Entry) Link() string { switch { case e.DOI != "": return "https://doi.org/" + e.DOI case e.ArXiv != "": return "https://arxiv.org/abs/" + e.ArXiv case e.URL != "": return e.URL default: return "" } } // ListHTML renders the numbered reference list. Each entry is anchored // so an inline citation can point at it. func ListHTML(entries []Entry) string { if len(entries) == 0 { return "" } var b strings.Builder b.WriteString(`
` + "\n
    \n") for _, e := range entries { b.WriteString(`
  1. `) b.WriteString(e.line()) b.WriteString("
  2. \n") } b.WriteString("
\n
\n") return b.String() } // line renders one entry: authors, title, the venue tail, and the // identifier link. A purely raw entry prints its verbatim text with // embedded identifiers made clickable, and nothing else: the author's // own wording already carries the year and venue. func (e Entry) line() string { if e.Raw != "" && e.Title == "" && len(e.Authors) == 0 { return e.identifierLinks() } var b strings.Builder if len(e.Authors) > 0 { names := make([]string, 0, len(e.Authors)) for _, a := range e.Authors { name := stdhtml.EscapeString(a.Name) if a.ORCID != "" { name += " " + link("https://orcid.org/"+a.ORCID, "orcid:"+a.ORCID) } names = append(names, name) } // An author string that already ends in a period ("Riess, A. G. // a kol.") must not gain a second one at the segment boundary. authorText := strings.Join(names, ", ") b.WriteString(authorText) if !strings.HasSuffix(authorText, ".") { b.WriteString(".") } b.WriteString(" ") } if e.Title != "" { b.WriteString(stdhtml.EscapeString(e.Title) + ".") } tail := make([]string, 0, 4) if e.Venue != "" { tail = append(tail, stdhtml.EscapeString(e.Venue)) } if e.Volume != "" { tail = append(tail, "vol. "+stdhtml.EscapeString(e.Volume)) } if e.Pages != "" { tail = append(tail, "pp. "+stdhtml.EscapeString(e.Pages)) } if e.Year != "" { tail = append(tail, stdhtml.EscapeString(e.Year)) } if len(tail) > 0 { if e.Title != "" { b.WriteString(" ") } b.WriteString(strings.Join(tail, ", ")) b.WriteString(".") } if url := e.href(); url != "" { b.WriteString(" " + link(url, label(e))) } return b.String() } // label names the identifier link by whichever resolver it uses. func label(e Entry) string { switch { case e.DOI != "": return "doi:" + e.DOI case e.ArXiv != "": return "arXiv:" + e.ArXiv default: return "url" } } // identifierLinks prints a raw entry with each doi:/arXiv: token it // embeds replaced by a live link, keeping the author's own wording // around it untouched. A token equal to the entry's cited DOI takes the // internal link when the entry has one; every other token keeps its // resolver. func (e Entry) identifierLinks() string { raw, doi, arxiv := e.Raw, e.DOI, e.ArXiv html := stdhtml.EscapeString(raw) if doi != "" { html = doiTokenRe.ReplaceAllStringFunc(html, func(match string) string { id := strings.TrimRight(doiTokenRe.FindStringSubmatch(match)[1], ".,;)") trail := match[len("doi:"):] trail = trail[strings.Index(trail, id)+len(id):] href := "https://doi.org/" + id if e.Internal != "" && id == doi { // The slug comes from the store, so it is escaped like // every other href: a hand-edited file whose slug carries // a quote must not open an attribute here. href = stdhtml.EscapeString(e.Internal) } return `doi:` + id + `` + trail }) } if arxiv != "" { html = arxivTokenRe.ReplaceAllStringFunc(html, func(match string) string { id := arxivTokenRe.FindStringSubmatch(match)[1] body := "arXiv:" + id trail := match[len(body):] return `arXiv:` + id + `` + trail }) } return html } // Citations renders the entries as schema.org citation objects for the // post's JSON-LD block, so a machine reading the article also reads the // works it cites, with each author and identifier resolved. func Citations(entries []Entry) []map[string]any { var out []map[string]any for _, e := range entries { name := e.Title if name == "" { name = e.Raw } if name == "" { continue } c := map[string]any{"@type": "ScholarlyArticle", "position": e.Num, "name": name} if len(e.Authors) > 0 { persons := make([]map[string]any, 0, len(e.Authors)) for _, a := range e.Authors { person := map[string]any{"@type": "Person", "name": a.Name} if a.ORCID != "" { person["identifier"] = "https://orcid.org/" + a.ORCID } persons = append(persons, person) } c["author"] = persons } if e.Venue != "" { c["isPartOf"] = map[string]any{"@type": "PublicationJournal", "name": e.Venue} } if e.Year != "" { c["datePublished"] = e.Year } if url := e.Link(); url != "" { c["identifier"] = url } if e.Internal != "" { c["url"] = e.Internal } out = append(out, c) } return out } func link(href, text string) string { return `` + stdhtml.EscapeString(text) + "" } // Place splices the list into the rendered body: at the marker // paragraph when the author wrote one, otherwise appended. func Place(bodyHTML string, entries []Entry) string { list := ListHTML(entries) if list == "" { return bodyHTML } if markerRe.MatchString(bodyHTML) { return markerRe.ReplaceAllString(bodyHTML, list) } return strings.TrimRight(bodyHTML, "\n") + "\n" + list } // LinkCitations rewrites inline `[n]` markers in the rendered HTML into // links to the numbered entries. It walks the markup and touches text // only: tags and attribute values are copied verbatim, and the inside // of ,
 and  elements is left alone, so code that holds
// bracketed numbers and the reference list itself keep their text.
func LinkCitations(html string, entries []Entry) string {
	if len(entries) == 0 {
		return html
	}
	highest := 0
	for _, e := range entries {
		if e.Num > highest {
			highest = e.Num
		}
	}
	var b strings.Builder
	b.Grow(len(html) + 2*len(html)/8)
	i := 0
	skipping := "" // non-empty while inside a protected element
	for i < len(html) {
		if html[i] != '<' {
			j := i
			for j < len(html) && html[j] != '<' {
				j++
			}
			b.WriteString(citeText(html[i:j], highest, skipping != ""))
			i = j
			continue
		}
		end := strings.IndexByte(html[i:], '>')
		if end < 0 {
			b.WriteString(html[i:])
			break
		}
		tag := html[i : i+end+1]
		b.WriteString(tag)
		switch {
		case skipping != "":
			// Inside a protected element: wait for its close.
			closing := ""
			if pos := strings.Index(strings.ToLower(tag), closing); pos >= 0 {
				skipping = ""
			}
		case strings.HasPrefix(strings.ToLower(tag), "") {
				kind := "a"
				switch {
				case strings.HasPrefix(strings.ToLower(tag), " highest {
			return tok
		}
		return `` + tok + ""
	})
}