Initial commit
Test / test (push) Successful in 7m5s
Release / gates (push) Successful in 7m28s
Release / build (amd64, freebsd) (push) Successful in 2m52s
Release / build (amd64, linux) (push) Successful in 2m46s
Release / build (arm64, freebsd) (push) Successful in 2m22s
Release / build (arm64, linux) (push) Successful in 2m38s
Release / build (loong64, linux) (push) Successful in 2m7s
Release / build (riscv64, linux) (push) Successful in 2m17s
Release / release (push) Successful in 1m0s

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-29 10:03:32 +02:00
commit f8ed33df83
206 changed files with 44165 additions and 0 deletions
+497
View File
@@ -0,0 +1,497 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
// Package biblio turns the structured `refs` frontmatter of a post into
// a rendered reference list, machine-readable citations, and the
// resolver links a scholar expects. It is a leaf: it reads plain
// metadata values and writes HTML and JSON shapes, importing no other
// domain package.
//
// The shape is measured on the author's own volumes: numbered `[n]`
// entries, inline `[n]` citations, and `doi:` and `arXiv:` identifiers
// embedded in free text. The engine keeps that convention and makes it
// live: each entry gets a target the inline citations point at, and
// every identifier becomes a link to its resolver.
package biblio
import (
stdhtml "html"
"regexp"
"strconv"
"strings"
)
// Author is one cited author: a name, and the author's ORCID when the
// reference records one.
type Author struct {
Name string `json:"name"`
ORCID string `json:"orcid,omitempty"`
}
// Entry is one reference. A hand-written entry that does not break down
// into the structured fields keeps its verbatim text in Raw, which the
// renderer prints as given, with any identifier inside it linked: a
// reference is never dropped or rewritten into something the author did
// not write.
type Entry struct {
Num int `json:"num"`
Authors []Author `json:"authors,omitempty"`
Title string `json:"title,omitempty"`
Venue string `json:"venue,omitempty"`
Year string `json:"year,omitempty"`
Volume string `json:"volume,omitempty"`
Pages string `json:"pages,omitempty"`
DOI string `json:"doi,omitempty"`
ArXiv string `json:"arxiv,omitempty"`
URL string `json:"url,omitempty"`
Raw string `json:"raw,omitempty"`
// Internal is the same-instance link: the API URL of the post whose
// DOI this entry cites, set by the caller that knows the instance
// (post.RefsLinked annotates each post's entries). Empty when the cited
// work is not published here, or is the citing post itself.
Internal string `json:"internal,omitempty"`
}
// href is the address the rendered link points at: the internal post
// when the entry cites a work published in this instance, otherwise
// the external resolver.
func (e Entry) href() string {
if e.Internal != "" {
return e.Internal
}
return e.Link()
}
// Marker is the token an author places where the reference list should
// be rendered; a body without it gets the list appended.
const Marker = "[[refs]]"
var (
markerRe = regexp.MustCompile(`(?s)<p>\s*\Q` + Marker + `\E\s*</p>`)
doiTokenRe = regexp.MustCompile(`(?i)\bdoi:\s*(10\.[0-9]{4,9}/[^\s]+)`)
arxivTokenRe = regexp.MustCompile(`(?i)\barXiv:\s*([A-Za-z0-9][A-Za-z0-9./:-]*)`)
yearRe = regexp.MustCompile(`\b(1[89][0-9]{2}|20[0-9]{2})\b`)
inlineRefRe = regexp.MustCompile(`\[(\d{1,3})\]`)
)
// Parse reads the refs array out of flattened frontmatter tables. Each
// element may carry authors, title, venue, year, volume, pages, doi,
// arxiv, url or raw; authors is a list of strings or of {name, orcid}
// tables. Entries without an explicit number are numbered by position.
func Parse(refs []map[string]any) []Entry {
var out []Entry
for i, raw := range refs {
entry := Entry{
Num: numOr(raw["num"], i+1),
Title: strings.TrimSpace(str(raw["title"])),
Venue: strings.TrimSpace(str(raw["venue"])),
Year: strings.TrimSpace(str(raw["year"])),
Volume: strings.TrimSpace(str(raw["volume"])),
Pages: strings.TrimSpace(str(raw["pages"])),
DOI: cleanDOI(str(raw["doi"])),
ArXiv: cleanArxiv(str(raw["arxiv"])),
URL: strings.TrimSpace(str(raw["url"])),
Raw: strings.TrimSpace(str(raw["raw"])),
}
entry.Authors = parseAuthors(raw["authors"])
entry.enrichFromText()
out = append(out, entry)
}
return out
}
func parseAuthors(v any) []Author {
list, ok := v.([]any)
if !ok {
return nil
}
var out []Author
for _, item := range list {
switch value := item.(type) {
case string:
if name := strings.TrimSpace(value); name != "" {
out = append(out, Author{Name: name})
}
case map[string]any:
author := Author{
Name: strings.TrimSpace(str(value["name"])),
ORCID: strings.TrimSpace(str(value["orcid"])),
}
if author.Name != "" {
out = append(out, author)
}
}
}
return out
}
func numOr(v any, fallback int) int {
switch n := v.(type) {
case int64:
return int(n)
case int:
return n
case float64:
return int(n)
case string:
if parsed, err := strconv.Atoi(strings.TrimSpace(n)); err == nil {
return parsed
}
}
return fallback
}
func str(v any) string {
s, _ := v.(string)
return s
}
// cleanDOI accepts a bare identifier, a doi: prefix or a resolver URL
// and returns the bare form.
func cleanDOI(s string) string {
s = strings.TrimSpace(s)
if s == "" {
return ""
}
if m := doiTokenRe.FindStringSubmatch(s); m != nil {
return strings.TrimRight(m[1], ".,;)")
}
s = strings.TrimPrefix(strings.TrimPrefix(s, "https://doi.org/"), "doi:")
return strings.TrimRight(strings.TrimSpace(s), ".,;)")
}
func cleanArxiv(s string) string {
s = strings.TrimSpace(s)
if s == "" {
return ""
}
if m := arxivTokenRe.FindStringSubmatch(s); m != nil {
s = m[1]
}
s = strings.TrimRight(strings.TrimPrefix(strings.TrimPrefix(s, "arXiv:"), "https://arxiv.org/abs/"), ".,;)")
if !strings.HasPrefix(s, "arXiv:") && s != "" {
// Already a bare id; accept anything identifier-shaped.
if strings.ContainsAny(s, " \t<\"") {
return ""
}
return s
}
return ""
}
// stripTokens removes the identifier tokens from the entry's own text,
// used once an entry is structured: the renderer puts the identifier at
// the end as a link, and it must not also sit in the title as prose.
func (e *Entry) stripTokens() {
if e.Title == "" && len(e.Authors) == 0 {
return // a raw entry prints its tokens inline as links instead
}
e.Title = tidyTokens(e.Title)
e.Venue = tidyTokens(e.Venue)
}
func tidyTokens(s string) string {
s = doiTokenRe.ReplaceAllString(s, "")
s = arxivTokenRe.ReplaceAllString(s, "")
s = strings.ReplaceAll(s, ", ,", ",")
s = strings.TrimSpace(s)
s = strings.TrimRight(s, ",;")
return strings.TrimSpace(s)
}
// enrichFromText recovers identifiers and a year a hand-written entry
// kept in its free text, so a migrated volume gains live links without
// every field being split by hand.
func (e *Entry) enrichFromText() {
joined := strings.Join([]string{e.Title, e.Venue, e.Raw}, " ")
if e.DOI == "" {
if m := doiTokenRe.FindStringSubmatch(joined); m != nil {
e.DOI = strings.TrimRight(m[1], ".,;)")
e.stripTokens()
}
}
if e.ArXiv == "" {
if m := arxivTokenRe.FindStringSubmatch(joined); m != nil {
e.ArXiv = m[1]
e.stripTokens()
}
}
// The year is a structured field only when the entry is structured:
// a raw entry already prints its year inside its own text.
if e.Year == "" && (e.Title != "" || len(e.Authors) > 0) {
if m := yearRe.FindStringSubmatch(joined); m != nil {
e.Year = m[1]
}
}
}
// Link returns the resolver URL for this entry: DOI first, then arXiv,
// then a plain URL. Empty when the entry carries no identifier.
func (e Entry) Link() string {
switch {
case e.DOI != "":
return "https://doi.org/" + e.DOI
case e.ArXiv != "":
return "https://arxiv.org/abs/" + e.ArXiv
case e.URL != "":
return e.URL
default:
return ""
}
}
// ListHTML renders the numbered reference list. Each entry is anchored
// so an inline citation can point at it.
func ListHTML(entries []Entry) string {
if len(entries) == 0 {
return ""
}
var b strings.Builder
b.WriteString(`<section class="refs" id="references">` + "\n<ol>\n")
for _, e := range entries {
b.WriteString(`<li id="ref-` + strconv.Itoa(e.Num) + `">`)
b.WriteString(e.line())
b.WriteString("</li>\n")
}
b.WriteString("</ol>\n</section>\n")
return b.String()
}
// line renders one entry: authors, title, the venue tail, and the
// identifier link. A purely raw entry prints its verbatim text with
// embedded identifiers made clickable, and nothing else: the author's
// own wording already carries the year and venue.
func (e Entry) line() string {
if e.Raw != "" && e.Title == "" && len(e.Authors) == 0 {
return e.identifierLinks()
}
var b strings.Builder
if len(e.Authors) > 0 {
names := make([]string, 0, len(e.Authors))
for _, a := range e.Authors {
name := stdhtml.EscapeString(a.Name)
if a.ORCID != "" {
name += " " + link("https://orcid.org/"+a.ORCID, "orcid:"+a.ORCID)
}
names = append(names, name)
}
// An author string that already ends in a period ("Riess, A. G.
// a kol.") must not gain a second one at the segment boundary.
authorText := strings.Join(names, ", ")
b.WriteString(authorText)
if !strings.HasSuffix(authorText, ".") {
b.WriteString(".")
}
b.WriteString(" ")
}
if e.Title != "" {
b.WriteString(stdhtml.EscapeString(e.Title) + ".")
}
tail := make([]string, 0, 4)
if e.Venue != "" {
tail = append(tail, stdhtml.EscapeString(e.Venue))
}
if e.Volume != "" {
tail = append(tail, "vol. "+stdhtml.EscapeString(e.Volume))
}
if e.Pages != "" {
tail = append(tail, "pp. "+stdhtml.EscapeString(e.Pages))
}
if e.Year != "" {
tail = append(tail, stdhtml.EscapeString(e.Year))
}
if len(tail) > 0 {
if e.Title != "" {
b.WriteString(" ")
}
b.WriteString(strings.Join(tail, ", "))
b.WriteString(".")
}
if url := e.href(); url != "" {
b.WriteString(" " + link(url, label(e)))
}
return b.String()
}
// label names the identifier link by whichever resolver it uses.
func label(e Entry) string {
switch {
case e.DOI != "":
return "doi:" + e.DOI
case e.ArXiv != "":
return "arXiv:" + e.ArXiv
default:
return "url"
}
}
// identifierLinks prints a raw entry with each doi:/arXiv: token it
// embeds replaced by a live link, keeping the author's own wording
// around it untouched. A token equal to the entry's cited DOI takes the
// internal link when the entry has one; every other token keeps its
// resolver.
func (e Entry) identifierLinks() string {
raw, doi, arxiv := e.Raw, e.DOI, e.ArXiv
html := stdhtml.EscapeString(raw)
if doi != "" {
html = doiTokenRe.ReplaceAllStringFunc(html, func(match string) string {
id := strings.TrimRight(doiTokenRe.FindStringSubmatch(match)[1], ".,;)")
trail := match[len("doi:"):]
trail = trail[strings.Index(trail, id)+len(id):]
href := "https://doi.org/" + id
if e.Internal != "" && id == doi {
// The slug comes from the store, so it is escaped like
// every other href: a hand-edited file whose slug carries
// a quote must not open an attribute here.
href = stdhtml.EscapeString(e.Internal)
}
return `<a class="refs-link" href="` + href + `">doi:` + id + `</a>` + trail
})
}
if arxiv != "" {
html = arxivTokenRe.ReplaceAllStringFunc(html, func(match string) string {
id := arxivTokenRe.FindStringSubmatch(match)[1]
body := "arXiv:" + id
trail := match[len(body):]
return `<a class="refs-link" href="https://arxiv.org/abs/` + id + `">arXiv:` + id + `</a>` + trail
})
}
return html
}
// Citations renders the entries as schema.org citation objects for the
// post's JSON-LD block, so a machine reading the article also reads the
// works it cites, with each author and identifier resolved.
func Citations(entries []Entry) []map[string]any {
var out []map[string]any
for _, e := range entries {
name := e.Title
if name == "" {
name = e.Raw
}
if name == "" {
continue
}
c := map[string]any{"@type": "ScholarlyArticle", "position": e.Num, "name": name}
if len(e.Authors) > 0 {
persons := make([]map[string]any, 0, len(e.Authors))
for _, a := range e.Authors {
person := map[string]any{"@type": "Person", "name": a.Name}
if a.ORCID != "" {
person["identifier"] = "https://orcid.org/" + a.ORCID
}
persons = append(persons, person)
}
c["author"] = persons
}
if e.Venue != "" {
c["isPartOf"] = map[string]any{"@type": "PublicationJournal", "name": e.Venue}
}
if e.Year != "" {
c["datePublished"] = e.Year
}
if url := e.Link(); url != "" {
c["identifier"] = url
}
if e.Internal != "" {
c["url"] = e.Internal
}
out = append(out, c)
}
return out
}
func link(href, text string) string {
return `<a class="refs-link" href="` + stdhtml.EscapeString(href) + `">` +
stdhtml.EscapeString(text) + "</a>"
}
// Place splices the list into the rendered body: at the marker
// paragraph when the author wrote one, otherwise appended.
func Place(bodyHTML string, entries []Entry) string {
list := ListHTML(entries)
if list == "" {
return bodyHTML
}
if markerRe.MatchString(bodyHTML) {
return markerRe.ReplaceAllString(bodyHTML, list)
}
return strings.TrimRight(bodyHTML, "\n") + "\n" + list
}
// LinkCitations rewrites inline `[n]` markers in the rendered HTML into
// links to the numbered entries. It walks the markup and touches text
// only: tags and attribute values are copied verbatim, and the inside
// of <code>, <pre> and <a> elements is left alone, so code that holds
// bracketed numbers and the reference list itself keep their text.
func LinkCitations(html string, entries []Entry) string {
if len(entries) == 0 {
return html
}
highest := 0
for _, e := range entries {
if e.Num > highest {
highest = e.Num
}
}
var b strings.Builder
b.Grow(len(html) + 2*len(html)/8)
i := 0
skipping := "" // non-empty while inside a protected element
for i < len(html) {
if html[i] != '<' {
j := i
for j < len(html) && html[j] != '<' {
j++
}
b.WriteString(citeText(html[i:j], highest, skipping != ""))
i = j
continue
}
end := strings.IndexByte(html[i:], '>')
if end < 0 {
b.WriteString(html[i:])
break
}
tag := html[i : i+end+1]
b.WriteString(tag)
switch {
case skipping != "":
// Inside a protected element: wait for its close.
closing := "</" + skipping + ">"
if pos := strings.Index(strings.ToLower(tag), closing); pos >= 0 {
skipping = ""
}
case strings.HasPrefix(strings.ToLower(tag), "<pre"),
strings.HasPrefix(strings.ToLower(tag), "<code"),
strings.HasPrefix(strings.ToLower(tag), "<a"):
if !strings.HasSuffix(tag, "/>") {
kind := "a"
switch {
case strings.HasPrefix(strings.ToLower(tag), "<pre"):
kind = "pre"
case strings.HasPrefix(strings.ToLower(tag), "<code"):
kind = "code"
}
skipping = kind
}
}
i += end + 1
}
return b.String()
}
// citeText replaces the bracketed numbers within one text run when
// linking is allowed there.
func citeText(text string, highest int, protected bool) string {
if protected || highest == 0 {
return text
}
return inlineRefRe.ReplaceAllStringFunc(text, func(tok string) string {
n, err := strconv.Atoi(strings.Trim(tok, "[]"))
if err != nil || n < 1 || n > highest {
return tok
}
return `<a class="ref-cite" href="#ref-` + strconv.Itoa(n) + `">` + tok + "</a>"
})
}