Files
volumen/internal/payloads/payloads.go
T

799 lines
23 KiB
Go
Raw Normal View History

2026-09-18 12:03:35 +02:00
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
// Package payloads builds the public API responses: filtering,
// pagination, tag clouds, series, and post validation helpers.
package payloads
import (
json "encoding/json/v2"
"fmt"
"log/slog"
"maps"
"regexp"
"slices"
"strconv"
"strings"
"time"
"unicode/utf8"
"sourcedock.dev/petrbalvin/interpres/v2"
"sourcedock.dev/petrbalvin/volumen/internal/fediverse"
"sourcedock.dev/petrbalvin/volumen/internal/frontmatter"
"sourcedock.dev/petrbalvin/volumen/internal/identifiers"
"sourcedock.dev/petrbalvin/volumen/internal/markdown"
"sourcedock.dev/petrbalvin/volumen/internal/post"
)
var (
// SlugRegex accepts lowercase slugs with inner dots, dashes and
// underscores.
SlugRegex = regexp.MustCompile(`^[a-z0-9](?:[a-z0-9._-]*[a-z0-9])?$`)
// LangRegex accepts simple language codes such as cs or pt-BR.
LangRegex = regexp.MustCompile(`^[A-Za-z0-9_-]+$`)
)
// MaxSlugLength caps slug length.
const MaxSlugLength = 200
// MaxPageSize is the largest page the list endpoints return, and the
// documented limit.
const MaxPageSize = 100
// maxSeriesOrder is the sort key for a post that carries no
// series_order: it sorts after every explicit order.
const maxSeriesOrder = 1 << 30
// Source is what the payload builders read from a content store: the
// whole set for a listing, and one post by slug for a validation rule.
type Source interface {
All() []*post.Post
Find(slug, lang string) *post.Post
}
// PublishedPosts returns all published posts, newest first. A post that
// is a draft or still scheduled is withheld.
func PublishedPosts(s Source) []*post.Post {
var visible []*post.Post
for _, p := range s.All() {
if p.Published() {
visible = append(visible, p)
}
}
slices.SortStableFunc(visible, func(a, b *post.Post) int {
return strings.Compare(b.DateString(), a.DateString())
})
return visible
}
// FilterPosts filters published posts by lang, tag and query.
func FilterPosts(posts []*post.Post, lang, tag, query string) []*post.Post {
if lang != "" {
var out []*post.Post
for _, p := range posts {
if p.Lang() == lang || p.AllLangs() {
out = append(out, p)
}
}
posts = out
}
if tag != "" {
var out []*post.Post
for _, p := range posts {
if slices.Contains(p.Tags(), tag) {
out = append(out, p)
}
}
posts = out
}
query = strings.ToLower(strings.TrimSpace(query))
if query != "" {
scored := make([]scoredPost, 0, len(posts))
for _, p := range posts {
if score := searchScore(p, query); score > 0 {
scored = append(scored, scoredPost{post: p, score: score})
}
}
// Relevance first: the strongest match leads, and posts that
// score the same keep the date order they arrived in.
slices.SortStableFunc(scored, func(a, b scoredPost) int {
return b.score - a.score
})
posts = make([]*post.Post, len(scored))
for i, s := range scored {
posts[i] = s.post
}
}
return posts
}
// scoredPost pairs a post with the relevance it scored against the
// running query.
type scoredPost struct {
post *post.Post
score int
}
// searchScore ranks one post against the lowered query: a match higher
// in the post weighs more, so a title hit outranks a passing mention in
// the body. Zero means the post does not match at all and leaves the
// result list.
func searchScore(p *post.Post, query string) int {
score := 0
title := strings.ToLower(p.Title())
if strings.Contains(title, query) {
score += 100
if strings.HasPrefix(title, query) {
score += 50
}
}
for _, tag := range p.Tags() {
tag = strings.ToLower(tag)
switch {
case tag == query:
score += 40
case strings.Contains(tag, query):
score += 15
}
}
if strings.Contains(strings.ToLower(p.Excerpt()), query) {
score += 20
}
if hits := strings.Count(strings.ToLower(p.Body), query); hits > 0 {
score += 10 + min(hits-1, 4)*2
}
return score
}
// PostsPayload builds the paginated payload for /api/volumen/posts.
// With a cursor the list starts after that slug and page numbers are
// ignored; otherwise the list is page-numbered. The two shapes differ
// in which fields they carry, so they are distinct types. A cursor that
// names no post yields an empty page rather than silently rewinding to
// the first one, which would send a paging client posts it already has.
func PostsPayload(s Source, lang, tag, query string, page, limit int, cursor string) any {
posts := FilterPosts(PublishedPosts(s), lang, tag, query)
limit = min(max(limit, 1), MaxPageSize)
total := len(posts)
offset := 0
if cursor != "" {
offset = cursorOffset(posts, cursor)
} else {
page = max(page, 1)
offset = (page - 1) * limit
}
end := min(offset+limit, total)
var pagePosts []*post.Post
if offset < total {
pagePosts = posts[offset:end]
}
summaries := make([]Summary, 0, len(pagePosts))
for _, p := range pagePosts {
summaries = append(summaries, BuildSummary(p))
}
base := PostListBase{PageSize: limit, Total: total, Posts: summaries}
var nextCursor *string
if len(pagePosts) > 0 && offset+limit < total {
last := pagePosts[len(pagePosts)-1]
value := last.Slug()
if slugShared(posts, last.Slug()) {
// Two posts share the slug (translations do): name the exact
// post, or the next page resumes after the first variant and
// serves the second one twice.
value += "." + last.Lang()
}
nextCursor = &value
}
if cursor != "" {
return CursorList{PostListBase: base, NextCursor: nextCursor}
}
return PageList{
PostListBase: base,
Page: max(page, 1),
HasNext: offset+limit < total,
HasPrev: max(page, 1) > 1,
}
}
// cursorOffset resolves a cursor to the index the next page starts
// after. A cursor is "slug", or "slug.lang" when the slug is shared by
// translations: the composite form is matched first so an ambiguous
// slug cannot resume the walk at the wrong variant. A cursor that names
// no post yields the end, an empty page, rather than rewinding.
func cursorOffset(posts []*post.Post, cursor string) int {
for i, p := range posts {
if p.Slug()+"."+p.Lang() == cursor {
return i + 1
}
}
for i, p := range posts {
if p.Slug() == cursor {
return i + 1
}
}
return len(posts)
}
// slugShared reports whether more than one post in the list carries the
// slug.
func slugShared(posts []*post.Post, slug string) bool {
count := 0
for _, p := range posts {
if p.Slug() == slug {
count++
if count > 1 {
return true
}
}
}
return false
}
// IsEmpty reports whether a paginated payload carries no posts. A
// payload of an unknown type is not treated as empty, so a future shape
// cannot accidentally turn into a 404.
func IsEmpty(payload any) bool {
switch list := payload.(type) {
case PageList:
return len(list.Posts) == 0
case CursorList:
return len(list.Posts) == 0
}
return false
}
// BuildTagCounts counts the tags of the given posts, sorted by
// frequency then name. The caller chooses the post set, so the admin
// dashboard counts its own view (drafts included) and the API counts the
// published posts.
func BuildTagCounts(posts []*post.Post) []CountedName {
counts := map[string]int{}
for _, p := range posts {
for _, tag := range p.Tags() {
counts[tag]++
}
}
return orderedCounts(counts)
}
// BuildSeriesList lists series with their post counts, most posts
// first.
func BuildSeriesList(s Source) []CountedName {
counts := map[string]int{}
for _, p := range PublishedPosts(s) {
if p.Series() != "" {
counts[p.Series()]++
}
}
return orderedCounts(counts)
}
func orderedCounts(counts map[string]int) []CountedName {
// slices.SortedFunc takes the keys as an iterator, so the list is
// built and ordered in one step.
names := slices.SortedFunc(maps.Keys(counts), func(a, b string) int {
if counts[a] != counts[b] {
return counts[b] - counts[a]
}
return strings.Compare(a, b)
})
out := make([]CountedName, 0, len(names))
for _, name := range names {
out = append(out, CountedName{Name: name, Count: counts[name]})
}
return out
}
// SeriesPosts returns the published posts of one series, ordered by
// series_order then date then slug.
func SeriesPosts(s Source, name string) []*post.Post {
var posts []*post.Post
for _, p := range PublishedPosts(s) {
if p.Series() == name {
posts = append(posts, p)
}
}
slices.SortStableFunc(posts, func(a, b *post.Post) int {
orderA, okA := a.SeriesOrder()
if !okA {
orderA = maxSeriesOrder
}
orderB, okB := b.SeriesOrder()
if !okB {
orderB = maxSeriesOrder
}
if orderA != orderB {
return orderA - orderB
}
if c := strings.Compare(a.DateString(), b.DateString()); c != 0 {
return c
}
return strings.Compare(a.Slug(), b.Slug())
})
return posts
}
// Presence returns a stripped string, or "" when empty.
func Presence(value any) string {
if value == nil {
return ""
}
return strings.TrimSpace(fmt.Sprintf("%v", value))
}
// ParseDate parses an ISO 8601 date string.
func ParseDate(value any) (time.Time, bool) {
text := Presence(value)
if text == "" {
return time.Time{}, false
}
t, err := time.Parse("2006-01-02", text)
if err != nil {
return time.Time{}, false
}
return t, true
}
// ParseInt parses an integer from form data.
func ParseInt(value any) (int, bool) {
text := Presence(value)
if text == "" {
return 0, false
}
n, err := strconv.Atoi(text)
if err != nil {
return 0, false
}
return n, true
}
// ParseTags parses a comma-separated tag string.
func ParseTags(value any) []string {
raw := Presence(value)
if raw == "" {
return nil
}
var out []string
for tag := range strings.SplitSeq(raw, ",") {
if trimmed := strings.TrimSpace(tag); trimmed != "" {
out = append(out, trimmed)
}
}
return out
}
// PostFromParams builds a post from admin form data, carrying the
// existing post's path when editing. Metadata keys the form does not
// manage (aliases, translations, custom fields) are inherited from the
// existing post so an editor save never drops them. A date field the
// form carries but cannot parse is reported as a ValidationError on the
// built post: an unparseable value dropping the key would silently
// publish a scheduled post. The caller renders the returned post back
// into the form either way.
func PostFromParams(form map[string]string, existing *post.Post) (*post.Post, error) {
badDateField := ""
for _, field := range []string{"date", "publish_at"} {
if Presence(form[field]) != "" {
if _, ok := ParseDate(form[field]); !ok {
badDateField = field
}
}
}
badUTF8Field := ""
for _, field := range append(slices.Clone(metadataOrder), "body") {
if value, present := form[field]; present && !utf8.ValidString(value) {
badUTF8Field = field
break
}
}
m := frontmatter.NewMeta()
if existing != nil {
// The clone carries the comments and the nested shapes of the
// stored file: the form only rewrites its own fields, and every
// key it does not name round-trips untouched.
m = existing.Metadata.Clone()
}
cleaned := cleanMetadata(baseMetadata(form, existing))
for _, key := range metadataOrder {
if key == badDateField {
// The form value was rejected, so the field keeps its
// inherited value: the post goes back into the editor with
// the schedule it had, not with the bad input written into
// the metadata nor with the schedule silently dropped.
continue
}
value, present := cleaned[key]
if !present {
// Cleared in the form: drop any inherited value too.
m.Delete(key)
continue
}
m.Set(key, value)
}
p := post.New(m, form["body"])
if existing != nil {
p.Path = existing.Path
}
if badUTF8Field != "" {
// Saving would silently replace the invalid bytes with U+FFFD,
// so the form is rejected instead: the author sees the field that
// carries them and keeps control over the text.
return p, invalid(fmt.Sprintf("%s must be valid UTF-8 text.", badUTF8Field))
}
if badDateField != "" {
return p, invalid(fmt.Sprintf("%s must be an ISO 8601 date.", badDateField))
}
// The references arrive as one JSON field from the editor. An absent
// field means the form never carried one, and the stored list
// round-trips untouched; an empty list is an explicit deletion.
tables, present, err := refsFromForm(form)
if err != nil {
return p, err
}
if present {
if len(tables) == 0 {
m.Delete("refs")
} else {
m.Set("refs", tables)
}
}
return p, nil
}
var metadataOrder = []string{
"title", "slug", "lang", "author", "fediverse_creator",
"doi", "orcid",
"date", "publish_at", "tags", "excerpt", "cover", "cover_alt",
"cover_caption", "series", "series_order", "draft", "all_langs",
}
func baseMetadata(form map[string]string, existing *post.Post) map[string]any {
slug := Presence(form["slug"])
if slug == "" && existing != nil {
slug = existing.Slug()
}
meta := map[string]any{
"title": Presence(form["title"]),
"slug": slug,
"lang": Presence(form["lang"]),
"author": Presence(form["author"]),
"fediverse_creator": Presence(form["fediverse_creator"]),
"doi": identifiers.NormalizeDOI(form["doi"]),
"orcid": identifiers.NormalizeORCID(form["orcid"]),
"tags": ParseTags(form["tags"]),
"excerpt": Presence(form["excerpt"]),
"cover": Presence(form["cover"]),
"cover_alt": Presence(form["cover_alt"]),
"cover_caption": Presence(form["cover_caption"]),
"series": Presence(form["series"]),
}
if d, ok := ParseDate(form["date"]); ok {
meta["date"] = interpres.LocalDate{Time: d}
}
if d, ok := ParseDate(form["publish_at"]); ok {
meta["publish_at"] = interpres.LocalDate{Time: d}
}
if n, ok := ParseInt(form["series_order"]); ok {
meta["series_order"] = int64(n)
}
if form["draft"] == "on" {
meta["draft"] = true
}
if form["all_langs"] == "on" {
meta["all_langs"] = true
}
return meta
}
// cleanMetadata drops nil, empty and false entries, keeping the
// original key order.
func cleanMetadata(meta map[string]any) map[string]any {
out := make(map[string]any, len(meta))
for _, key := range metadataOrder {
value, ok := meta[key]
if !ok || value == nil {
continue
}
switch v := value.(type) {
case string:
if v == "" {
continue
}
case []string:
if len(v) == 0 {
continue
}
case bool:
if !v {
continue
}
}
out[key] = value
}
return out
}
// ValidationError is a user-facing validation failure. The message is
// shown verbatim in the API error envelope and in the admin forms.
type ValidationError struct {
Message string
}
func (e *ValidationError) Error() string { return e.Message }
func invalid(message string) error { return &ValidationError{Message: message} }
// The bounds of one reference list written from the editor. They exist so
// a runaway payload dies at a named rule rather than at the body limit.
const (
maxRefEntries = 500
maxRefField = 4000
maxRefAuthors = 200
)
// refsFromForm decodes the editor's reference list. The second return
// value reports whether the form carried the field at all: absent means
// the stored list survives untouched, present means the decoded list (or
// its deletion) is the author's explicit choice. A row that carries
// nothing at all is dropped rather than saved as an empty table. The
// tables come back as []map[string]any, the shape the frontmatter writer
// renders as [[refs]] blocks rather than one inline array.
func refsFromForm(form map[string]string) ([]map[string]any, bool, error) {
encoded, present := form["refs"]
if !present {
return nil, false, nil
}
if strings.TrimSpace(encoded) == "" {
return nil, true, nil
}
var entries []map[string]any
if err := json.Unmarshal([]byte(encoded), &entries); err != nil {
return nil, true, invalid("References must be a list of entries.")
}
if len(entries) > maxRefEntries {
return nil, true, invalid(fmt.Sprintf("References must hold at most %d entries.", maxRefEntries))
}
out := make([]map[string]any, 0, len(entries))
for i, entry := range entries {
table, err := cleanRefEntry(entry, i+1)
if err != nil {
return nil, true, err
}
if table != nil {
out = append(out, table)
}
}
return out, true, nil
}
// cleanRefEntry validates one reference table and returns it in the
// canonical shape the frontmatter writer accepts: empty strings dropped,
// identifiers normalised, authors as strings or name tables. A row whose
// every field is blank returns nil, meaning it is dropped.
func cleanRefEntry(entry map[string]any, position int) (map[string]any, error) {
which := fmt.Sprintf("Reference %d", position)
clip := func(value any, field string) (string, error) {
var text string
switch v := value.(type) {
case string:
text = v
case float64:
// A year or volume the client sent as a JSON number still
// belongs in the table, as the string the model stores.
if v == float64(int64(v)) {
text = strconv.FormatInt(int64(v), 10)
} else {
text = strconv.FormatFloat(v, 'f', -1, 64)
}
}
text = strings.TrimSpace(text)
if len(text) > maxRefField {
return "", invalid(fmt.Sprintf("%s: %s must be at most %d characters.", which, field, maxRefField))
}
return text, nil
}
out := map[string]any{}
for _, field := range []string{"raw", "title", "venue", "year", "volume", "pages", "arxiv", "url"} {
text, err := clip(entry[field], field)
if err != nil {
return nil, err
}
if text != "" {
out[field] = text
}
}
if doi, _ := entry["doi"].(string); strings.TrimSpace(doi) != "" {
doi = identifiers.NormalizeDOI(doi)
if !identifiers.ValidDOI(doi) {
return nil, invalid(fmt.Sprintf("%s: DOI must look like 10.xxxx/suffix.", which))
}
out["doi"] = doi
}
if arxiv, ok := out["arxiv"]; ok {
id := strings.TrimPrefix(strings.TrimPrefix(arxiv.(string), "https://arxiv.org/abs/"), "arXiv:")
if id == "" || strings.ContainsAny(id, " \t\"'<>") {
return nil, invalid(fmt.Sprintf("%s: arXiv must be the bare identifier, e.g. 2401.12345.", which))
}
out["arxiv"] = id
}
if urlField, ok := out["url"]; ok {
if !strings.HasPrefix(urlField.(string), "http://") && !strings.HasPrefix(urlField.(string), "https://") {
return nil, invalid(fmt.Sprintf("%s: URL must be an http(s) address.", which))
}
}
authors, err := cleanRefAuthors(entry["authors"], which)
if err != nil {
return nil, err
}
if len(authors) > 0 {
out["authors"] = authors
}
// An explicit number survives a round trip, so a hand-numbered list
// keeps its numbering through the editor.
switch n := entry["num"].(type) {
case float64:
if n == float64(int(n)) && n >= 1 && n <= 9999 {
out["num"] = int64(n)
}
case string:
if parsed, err := strconv.Atoi(strings.TrimSpace(n)); err == nil && parsed >= 1 && parsed <= 9999 {
out["num"] = int64(parsed)
}
}
// A row with only identifiers and no citation of its own would render
// as an empty entry; an entirely blank row is simply dropped.
if _, cited := out["raw"]; !cited {
if _, ok := out["title"]; !ok {
if _, hasAuthors := out["authors"]; !hasAuthors {
if len(out) == 0 {
return nil, nil
}
return nil, invalid(fmt.Sprintf("%s needs the citation itself: the verbatim line, the title, or the authors.", which))
}
}
}
return out, nil
}
// cleanRefAuthors validates the author list of one reference: plain name
// strings, or name tables with an optional ORCID checked to its digit.
func cleanRefAuthors(value any, which string) ([]any, error) {
list, ok := value.([]any)
if !ok {
return nil, nil
}
if len(list) > maxRefAuthors {
return nil, invalid(fmt.Sprintf("%s: at most %d authors per reference.", which, maxRefAuthors))
}
out := make([]any, 0, len(list))
for _, item := range list {
switch author := item.(type) {
case string:
if name := strings.TrimSpace(author); name != "" {
out = append(out, name)
}
case map[string]any:
name, _ := author["name"].(string)
name = strings.TrimSpace(name)
if name == "" {
continue
}
orcid, _ := author["orcid"].(string)
orcid = identifiers.NormalizeORCID(strings.TrimSpace(orcid))
if orcid != "" && !identifiers.ValidORCID(orcid) {
return nil, invalid(fmt.Sprintf("%s: ORCID must look like 0000-0002-1825-0097.", which))
}
if orcid != "" {
out = append(out, map[string]any{"name": name, "orcid": orcid})
} else {
out = append(out, name)
}
}
}
return out, nil
}
// CreationError validates a post before saving; nil means valid.
func CreationError(p *post.Post, s Source, existing *post.Post) error {
if Presence(p.Slug()) == "" {
return invalid("Slug is required.")
}
slug := p.Slug()
if len([]rune(slug)) > MaxSlugLength {
return invalid(fmt.Sprintf("Slug must be at most %d characters.", MaxSlugLength))
}
if !SlugRegex.MatchString(slug) {
return invalid("Invalid slug.")
}
if lang := p.Lang(); lang != "" && !LangRegex.MatchString(lang) {
return invalid("Invalid language.")
}
if len(p.Body) > markdown.MaxBodyLength {
return invalid(fmt.Sprintf("Body must be at most %d bytes.", markdown.MaxBodyLength))
}
if found := s.Find(slug, ""); found != nil &&
(existing == nil || found.Path != existing.Path) {
return invalid("A post with that slug already exists.")
}
if err := FediverseCreatorError(p); err != nil {
return err
}
if err := DOIError(p); err != nil {
return err
}
return ORCIDError(p)
}
// FediverseCreatorError validates the optional fediverse handle.
func FediverseCreatorError(p *post.Post) error {
value, ok := p.Metadata.Get("fediverse_creator")
if !ok || value == nil {
return nil
}
if fediverse.Valid(fmt.Sprintf("%v", value)) {
return nil
}
return invalid("Fediverse creator must look like @user@host.")
}
// DOIError validates the optional DOI. The stored value is the bare
// form; a doi.org URL or doi: prefix normalises away before the rule.
func DOIError(p *post.Post) error {
value, ok := p.Metadata.Get("doi")
if !ok || value == nil {
return nil
}
text := fmt.Sprintf("%v", value)
if text == "" || identifiers.ValidDOI(text) {
return nil
}
return invalid("DOI must look like 10.xxxx/suffix.")
}
// ORCIDError validates the optional ORCID iD, check digit included.
func ORCIDError(p *post.Post) error {
value, ok := p.Metadata.Get("orcid")
if !ok || value == nil {
return nil
}
text := fmt.Sprintf("%v", value)
if text == "" || identifiers.ValidORCID(text) {
return nil
}
return invalid("ORCID must look like 0000-0002-1825-0097.")
}
// Repository is what the payload builders write through.
type Repository interface {
Source
Save(p *post.Post) (*post.Post, error)
Delete(slug, lang string) (*post.Post, bool, error)
}
// SavePost persists a post through the repository. When the slug changed
// since existing, the file moves to the path its new slug implies and
// the old file is soft-deleted, so a rename is one operation with one
// undo and one revision archive. The admin and the API both save through
// here, so the on-disk result never depends on which one asked.
func SavePost(st Repository, p, existing *post.Post) (*post.Post, error) {
renamed := existing != nil && existing.Path != "" && p.Slug() != "" && p.Slug() != existing.Slug()
if renamed {
p.Path = ""
}
saved, err := st.Save(p)
if err != nil {
return nil, err
}
if renamed {
if _, _, err := st.Delete(existing.Slug(), existing.Lang()); err != nil {
slog.Warn("payloads: could not archive the post under its old slug",
"slug", existing.Slug(), "error", err)
}
}
return saved, nil
}