// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0 // Package payloads builds the public API responses: filtering, // pagination, tag clouds, series, and post validation helpers. package payloads import ( json "encoding/json/v2" "fmt" "log/slog" "maps" "regexp" "slices" "strconv" "strings" "time" "unicode/utf8" "sourcedock.dev/petrbalvin/interpres/v2" "sourcedock.dev/petrbalvin/volumen/internal/fediverse" "sourcedock.dev/petrbalvin/volumen/internal/frontmatter" "sourcedock.dev/petrbalvin/volumen/internal/identifiers" "sourcedock.dev/petrbalvin/volumen/internal/markdown" "sourcedock.dev/petrbalvin/volumen/internal/post" ) var ( // SlugRegex accepts lowercase slugs with inner dots, dashes and // underscores. SlugRegex = regexp.MustCompile(`^[a-z0-9](?:[a-z0-9._-]*[a-z0-9])?$`) // LangRegex accepts simple language codes such as cs or pt-BR. LangRegex = regexp.MustCompile(`^[A-Za-z0-9_-]+$`) ) // MaxSlugLength caps slug length. const MaxSlugLength = 200 // MaxPageSize is the largest page the list endpoints return, and the // documented limit. const MaxPageSize = 100 // maxSeriesOrder is the sort key for a post that carries no // series_order: it sorts after every explicit order. const maxSeriesOrder = 1 << 30 // Source is what the payload builders read from a content store: the // whole set for a listing, and one post by slug for a validation rule. type Source interface { All() []*post.Post Find(slug, lang string) *post.Post } // PublishedPosts returns all published posts, newest first. A post that // is a draft or still scheduled is withheld. func PublishedPosts(s Source) []*post.Post { var visible []*post.Post for _, p := range s.All() { if p.Published() { visible = append(visible, p) } } slices.SortStableFunc(visible, func(a, b *post.Post) int { return strings.Compare(b.DateString(), a.DateString()) }) return visible } // FilterPosts filters published posts by lang, tag and query. func FilterPosts(posts []*post.Post, lang, tag, query string) []*post.Post { if lang != "" { var out []*post.Post for _, p := range posts { if p.Lang() == lang || p.AllLangs() { out = append(out, p) } } posts = out } if tag != "" { var out []*post.Post for _, p := range posts { if slices.Contains(p.Tags(), tag) { out = append(out, p) } } posts = out } query = strings.ToLower(strings.TrimSpace(query)) if query != "" { scored := make([]scoredPost, 0, len(posts)) for _, p := range posts { if score := searchScore(p, query); score > 0 { scored = append(scored, scoredPost{post: p, score: score}) } } // Relevance first: the strongest match leads, and posts that // score the same keep the date order they arrived in. slices.SortStableFunc(scored, func(a, b scoredPost) int { return b.score - a.score }) posts = make([]*post.Post, len(scored)) for i, s := range scored { posts[i] = s.post } } return posts } // scoredPost pairs a post with the relevance it scored against the // running query. type scoredPost struct { post *post.Post score int } // searchScore ranks one post against the lowered query: a match higher // in the post weighs more, so a title hit outranks a passing mention in // the body. Zero means the post does not match at all and leaves the // result list. func searchScore(p *post.Post, query string) int { score := 0 title := strings.ToLower(p.Title()) if strings.Contains(title, query) { score += 100 if strings.HasPrefix(title, query) { score += 50 } } for _, tag := range p.Tags() { tag = strings.ToLower(tag) switch { case tag == query: score += 40 case strings.Contains(tag, query): score += 15 } } if strings.Contains(strings.ToLower(p.Excerpt()), query) { score += 20 } if hits := strings.Count(strings.ToLower(p.Body), query); hits > 0 { score += 10 + min(hits-1, 4)*2 } return score } // PostsPayload builds the paginated payload for /api/volumen/posts. // With a cursor the list starts after that slug and page numbers are // ignored; otherwise the list is page-numbered. The two shapes differ // in which fields they carry, so they are distinct types. A cursor that // names no post yields an empty page rather than silently rewinding to // the first one, which would send a paging client posts it already has. func PostsPayload(s Source, lang, tag, query string, page, limit int, cursor string) any { posts := FilterPosts(PublishedPosts(s), lang, tag, query) limit = min(max(limit, 1), MaxPageSize) total := len(posts) offset := 0 if cursor != "" { offset = cursorOffset(posts, cursor) } else { page = max(page, 1) offset = (page - 1) * limit } end := min(offset+limit, total) var pagePosts []*post.Post if offset < total { pagePosts = posts[offset:end] } summaries := make([]Summary, 0, len(pagePosts)) for _, p := range pagePosts { summaries = append(summaries, BuildSummary(p)) } base := PostListBase{PageSize: limit, Total: total, Posts: summaries} var nextCursor *string if len(pagePosts) > 0 && offset+limit < total { last := pagePosts[len(pagePosts)-1] value := last.Slug() if slugShared(posts, last.Slug()) { // Two posts share the slug (translations do): name the exact // post, or the next page resumes after the first variant and // serves the second one twice. value += "." + last.Lang() } nextCursor = &value } if cursor != "" { return CursorList{PostListBase: base, NextCursor: nextCursor} } return PageList{ PostListBase: base, Page: max(page, 1), HasNext: offset+limit < total, HasPrev: max(page, 1) > 1, } } // cursorOffset resolves a cursor to the index the next page starts // after. A cursor is "slug", or "slug.lang" when the slug is shared by // translations: the composite form is matched first so an ambiguous // slug cannot resume the walk at the wrong variant. A cursor that names // no post yields the end, an empty page, rather than rewinding. func cursorOffset(posts []*post.Post, cursor string) int { for i, p := range posts { if p.Slug()+"."+p.Lang() == cursor { return i + 1 } } for i, p := range posts { if p.Slug() == cursor { return i + 1 } } return len(posts) } // slugShared reports whether more than one post in the list carries the // slug. func slugShared(posts []*post.Post, slug string) bool { count := 0 for _, p := range posts { if p.Slug() == slug { count++ if count > 1 { return true } } } return false } // IsEmpty reports whether a paginated payload carries no posts. A // payload of an unknown type is not treated as empty, so a future shape // cannot accidentally turn into a 404. func IsEmpty(payload any) bool { switch list := payload.(type) { case PageList: return len(list.Posts) == 0 case CursorList: return len(list.Posts) == 0 } return false } // BuildTagCounts counts the tags of the given posts, sorted by // frequency then name. The caller chooses the post set, so the admin // dashboard counts its own view (drafts included) and the API counts the // published posts. func BuildTagCounts(posts []*post.Post) []CountedName { counts := map[string]int{} for _, p := range posts { for _, tag := range p.Tags() { counts[tag]++ } } return orderedCounts(counts) } // BuildSeriesList lists series with their post counts, most posts // first. func BuildSeriesList(s Source) []CountedName { counts := map[string]int{} for _, p := range PublishedPosts(s) { if p.Series() != "" { counts[p.Series()]++ } } return orderedCounts(counts) } func orderedCounts(counts map[string]int) []CountedName { // slices.SortedFunc takes the keys as an iterator, so the list is // built and ordered in one step. names := slices.SortedFunc(maps.Keys(counts), func(a, b string) int { if counts[a] != counts[b] { return counts[b] - counts[a] } return strings.Compare(a, b) }) out := make([]CountedName, 0, len(names)) for _, name := range names { out = append(out, CountedName{Name: name, Count: counts[name]}) } return out } // SeriesPosts returns the published posts of one series, ordered by // series_order then date then slug. func SeriesPosts(s Source, name string) []*post.Post { var posts []*post.Post for _, p := range PublishedPosts(s) { if p.Series() == name { posts = append(posts, p) } } slices.SortStableFunc(posts, func(a, b *post.Post) int { orderA, okA := a.SeriesOrder() if !okA { orderA = maxSeriesOrder } orderB, okB := b.SeriesOrder() if !okB { orderB = maxSeriesOrder } if orderA != orderB { return orderA - orderB } if c := strings.Compare(a.DateString(), b.DateString()); c != 0 { return c } return strings.Compare(a.Slug(), b.Slug()) }) return posts } // Presence returns a stripped string, or "" when empty. func Presence(value any) string { if value == nil { return "" } return strings.TrimSpace(fmt.Sprintf("%v", value)) } // ParseDate parses an ISO 8601 date string. func ParseDate(value any) (time.Time, bool) { text := Presence(value) if text == "" { return time.Time{}, false } t, err := time.Parse("2006-01-02", text) if err != nil { return time.Time{}, false } return t, true } // ParseInt parses an integer from form data. func ParseInt(value any) (int, bool) { text := Presence(value) if text == "" { return 0, false } n, err := strconv.Atoi(text) if err != nil { return 0, false } return n, true } // ParseTags parses a comma-separated tag string. func ParseTags(value any) []string { raw := Presence(value) if raw == "" { return nil } var out []string for tag := range strings.SplitSeq(raw, ",") { if trimmed := strings.TrimSpace(tag); trimmed != "" { out = append(out, trimmed) } } return out } // PostFromParams builds a post from admin form data, carrying the // existing post's path when editing. Metadata keys the form does not // manage (aliases, translations, custom fields) are inherited from the // existing post so an editor save never drops them. A date field the // form carries but cannot parse is reported as a ValidationError on the // built post: an unparseable value dropping the key would silently // publish a scheduled post. The caller renders the returned post back // into the form either way. func PostFromParams(form map[string]string, existing *post.Post) (*post.Post, error) { badDateField := "" for _, field := range []string{"date", "publish_at"} { if Presence(form[field]) != "" { if _, ok := ParseDate(form[field]); !ok { badDateField = field } } } badUTF8Field := "" for _, field := range append(slices.Clone(metadataOrder), "body") { if value, present := form[field]; present && !utf8.ValidString(value) { badUTF8Field = field break } } m := frontmatter.NewMeta() if existing != nil { // The clone carries the comments and the nested shapes of the // stored file: the form only rewrites its own fields, and every // key it does not name round-trips untouched. m = existing.Metadata.Clone() } cleaned := cleanMetadata(baseMetadata(form, existing)) for _, key := range metadataOrder { if key == badDateField { // The form value was rejected, so the field keeps its // inherited value: the post goes back into the editor with // the schedule it had, not with the bad input written into // the metadata nor with the schedule silently dropped. continue } value, present := cleaned[key] if !present { // Cleared in the form: drop any inherited value too. m.Delete(key) continue } m.Set(key, value) } p := post.New(m, form["body"]) if existing != nil { p.Path = existing.Path } if badUTF8Field != "" { // Saving would silently replace the invalid bytes with U+FFFD, // so the form is rejected instead: the author sees the field that // carries them and keeps control over the text. return p, invalid(fmt.Sprintf("%s must be valid UTF-8 text.", badUTF8Field)) } if badDateField != "" { return p, invalid(fmt.Sprintf("%s must be an ISO 8601 date.", badDateField)) } // The references arrive as one JSON field from the editor. An absent // field means the form never carried one, and the stored list // round-trips untouched; an empty list is an explicit deletion. tables, present, err := refsFromForm(form) if err != nil { return p, err } if present { if len(tables) == 0 { m.Delete("refs") } else { m.Set("refs", tables) } } return p, nil } var metadataOrder = []string{ "title", "slug", "lang", "author", "fediverse_creator", "doi", "orcid", "date", "publish_at", "tags", "excerpt", "cover", "cover_alt", "cover_caption", "series", "series_order", "draft", "all_langs", } func baseMetadata(form map[string]string, existing *post.Post) map[string]any { slug := Presence(form["slug"]) if slug == "" && existing != nil { slug = existing.Slug() } meta := map[string]any{ "title": Presence(form["title"]), "slug": slug, "lang": Presence(form["lang"]), "author": Presence(form["author"]), "fediverse_creator": Presence(form["fediverse_creator"]), "doi": identifiers.NormalizeDOI(form["doi"]), "orcid": identifiers.NormalizeORCID(form["orcid"]), "tags": ParseTags(form["tags"]), "excerpt": Presence(form["excerpt"]), "cover": Presence(form["cover"]), "cover_alt": Presence(form["cover_alt"]), "cover_caption": Presence(form["cover_caption"]), "series": Presence(form["series"]), } if d, ok := ParseDate(form["date"]); ok { meta["date"] = interpres.LocalDate{Time: d} } if d, ok := ParseDate(form["publish_at"]); ok { meta["publish_at"] = interpres.LocalDate{Time: d} } if n, ok := ParseInt(form["series_order"]); ok { meta["series_order"] = int64(n) } if form["draft"] == "on" { meta["draft"] = true } if form["all_langs"] == "on" { meta["all_langs"] = true } return meta } // cleanMetadata drops nil, empty and false entries, keeping the // original key order. func cleanMetadata(meta map[string]any) map[string]any { out := make(map[string]any, len(meta)) for _, key := range metadataOrder { value, ok := meta[key] if !ok || value == nil { continue } switch v := value.(type) { case string: if v == "" { continue } case []string: if len(v) == 0 { continue } case bool: if !v { continue } } out[key] = value } return out } // ValidationError is a user-facing validation failure. The message is // shown verbatim in the API error envelope and in the admin forms. type ValidationError struct { Message string } func (e *ValidationError) Error() string { return e.Message } func invalid(message string) error { return &ValidationError{Message: message} } // The bounds of one reference list written from the editor. They exist so // a runaway payload dies at a named rule rather than at the body limit. const ( maxRefEntries = 500 maxRefField = 4000 maxRefAuthors = 200 ) // refsFromForm decodes the editor's reference list. The second return // value reports whether the form carried the field at all: absent means // the stored list survives untouched, present means the decoded list (or // its deletion) is the author's explicit choice. A row that carries // nothing at all is dropped rather than saved as an empty table. The // tables come back as []map[string]any, the shape the frontmatter writer // renders as [[refs]] blocks rather than one inline array. func refsFromForm(form map[string]string) ([]map[string]any, bool, error) { encoded, present := form["refs"] if !present { return nil, false, nil } if strings.TrimSpace(encoded) == "" { return nil, true, nil } var entries []map[string]any if err := json.Unmarshal([]byte(encoded), &entries); err != nil { return nil, true, invalid("References must be a list of entries.") } if len(entries) > maxRefEntries { return nil, true, invalid(fmt.Sprintf("References must hold at most %d entries.", maxRefEntries)) } out := make([]map[string]any, 0, len(entries)) for i, entry := range entries { table, err := cleanRefEntry(entry, i+1) if err != nil { return nil, true, err } if table != nil { out = append(out, table) } } return out, true, nil } // cleanRefEntry validates one reference table and returns it in the // canonical shape the frontmatter writer accepts: empty strings dropped, // identifiers normalised, authors as strings or name tables. A row whose // every field is blank returns nil, meaning it is dropped. func cleanRefEntry(entry map[string]any, position int) (map[string]any, error) { which := fmt.Sprintf("Reference %d", position) clip := func(value any, field string) (string, error) { var text string switch v := value.(type) { case string: text = v case float64: // A year or volume the client sent as a JSON number still // belongs in the table, as the string the model stores. if v == float64(int64(v)) { text = strconv.FormatInt(int64(v), 10) } else { text = strconv.FormatFloat(v, 'f', -1, 64) } } text = strings.TrimSpace(text) if len(text) > maxRefField { return "", invalid(fmt.Sprintf("%s: %s must be at most %d characters.", which, field, maxRefField)) } return text, nil } out := map[string]any{} for _, field := range []string{"raw", "title", "venue", "year", "volume", "pages", "arxiv", "url"} { text, err := clip(entry[field], field) if err != nil { return nil, err } if text != "" { out[field] = text } } if doi, _ := entry["doi"].(string); strings.TrimSpace(doi) != "" { doi = identifiers.NormalizeDOI(doi) if !identifiers.ValidDOI(doi) { return nil, invalid(fmt.Sprintf("%s: DOI must look like 10.xxxx/suffix.", which)) } out["doi"] = doi } if arxiv, ok := out["arxiv"]; ok { id := strings.TrimPrefix(strings.TrimPrefix(arxiv.(string), "https://arxiv.org/abs/"), "arXiv:") if id == "" || strings.ContainsAny(id, " \t\"'<>") { return nil, invalid(fmt.Sprintf("%s: arXiv must be the bare identifier, e.g. 2401.12345.", which)) } out["arxiv"] = id } if urlField, ok := out["url"]; ok { if !strings.HasPrefix(urlField.(string), "http://") && !strings.HasPrefix(urlField.(string), "https://") { return nil, invalid(fmt.Sprintf("%s: URL must be an http(s) address.", which)) } } authors, err := cleanRefAuthors(entry["authors"], which) if err != nil { return nil, err } if len(authors) > 0 { out["authors"] = authors } // An explicit number survives a round trip, so a hand-numbered list // keeps its numbering through the editor. switch n := entry["num"].(type) { case float64: if n == float64(int(n)) && n >= 1 && n <= 9999 { out["num"] = int64(n) } case string: if parsed, err := strconv.Atoi(strings.TrimSpace(n)); err == nil && parsed >= 1 && parsed <= 9999 { out["num"] = int64(parsed) } } // A row with only identifiers and no citation of its own would render // as an empty entry; an entirely blank row is simply dropped. if _, cited := out["raw"]; !cited { if _, ok := out["title"]; !ok { if _, hasAuthors := out["authors"]; !hasAuthors { if len(out) == 0 { return nil, nil } return nil, invalid(fmt.Sprintf("%s needs the citation itself: the verbatim line, the title, or the authors.", which)) } } } return out, nil } // cleanRefAuthors validates the author list of one reference: plain name // strings, or name tables with an optional ORCID checked to its digit. func cleanRefAuthors(value any, which string) ([]any, error) { list, ok := value.([]any) if !ok { return nil, nil } if len(list) > maxRefAuthors { return nil, invalid(fmt.Sprintf("%s: at most %d authors per reference.", which, maxRefAuthors)) } out := make([]any, 0, len(list)) for _, item := range list { switch author := item.(type) { case string: if name := strings.TrimSpace(author); name != "" { out = append(out, name) } case map[string]any: name, _ := author["name"].(string) name = strings.TrimSpace(name) if name == "" { continue } orcid, _ := author["orcid"].(string) orcid = identifiers.NormalizeORCID(strings.TrimSpace(orcid)) if orcid != "" && !identifiers.ValidORCID(orcid) { return nil, invalid(fmt.Sprintf("%s: ORCID must look like 0000-0002-1825-0097.", which)) } if orcid != "" { out = append(out, map[string]any{"name": name, "orcid": orcid}) } else { out = append(out, name) } } } return out, nil } // CreationError validates a post before saving; nil means valid. func CreationError(p *post.Post, s Source, existing *post.Post) error { if Presence(p.Slug()) == "" { return invalid("Slug is required.") } slug := p.Slug() if len([]rune(slug)) > MaxSlugLength { return invalid(fmt.Sprintf("Slug must be at most %d characters.", MaxSlugLength)) } if !SlugRegex.MatchString(slug) { return invalid("Invalid slug.") } if lang := p.Lang(); lang != "" && !LangRegex.MatchString(lang) { return invalid("Invalid language.") } if len(p.Body) > markdown.MaxBodyLength { return invalid(fmt.Sprintf("Body must be at most %d bytes.", markdown.MaxBodyLength)) } if found := s.Find(slug, ""); found != nil && (existing == nil || found.Path != existing.Path) { return invalid("A post with that slug already exists.") } if err := FediverseCreatorError(p); err != nil { return err } if err := DOIError(p); err != nil { return err } return ORCIDError(p) } // FediverseCreatorError validates the optional fediverse handle. func FediverseCreatorError(p *post.Post) error { value, ok := p.Metadata.Get("fediverse_creator") if !ok || value == nil { return nil } if fediverse.Valid(fmt.Sprintf("%v", value)) { return nil } return invalid("Fediverse creator must look like @user@host.") } // DOIError validates the optional DOI. The stored value is the bare // form; a doi.org URL or doi: prefix normalises away before the rule. func DOIError(p *post.Post) error { value, ok := p.Metadata.Get("doi") if !ok || value == nil { return nil } text := fmt.Sprintf("%v", value) if text == "" || identifiers.ValidDOI(text) { return nil } return invalid("DOI must look like 10.xxxx/suffix.") } // ORCIDError validates the optional ORCID iD, check digit included. func ORCIDError(p *post.Post) error { value, ok := p.Metadata.Get("orcid") if !ok || value == nil { return nil } text := fmt.Sprintf("%v", value) if text == "" || identifiers.ValidORCID(text) { return nil } return invalid("ORCID must look like 0000-0002-1825-0097.") } // Repository is what the payload builders write through. type Repository interface { Source Save(p *post.Post) (*post.Post, error) Delete(slug, lang string) (*post.Post, bool, error) } // SavePost persists a post through the repository. When the slug changed // since existing, the file moves to the path its new slug implies and // the old file is soft-deleted, so a rename is one operation with one // undo and one revision archive. The admin and the API both save through // here, so the on-disk result never depends on which one asked. func SavePost(st Repository, p, existing *post.Post) (*post.Post, error) { renamed := existing != nil && existing.Path != "" && p.Slug() != "" && p.Slug() != existing.Slug() if renamed { p.Path = "" } saved, err := st.Save(p) if err != nil { return nil, err } if renamed { if _, _, err := st.Delete(existing.Slug(), existing.Lang()); err != nil { slog.Warn("payloads: could not archive the post under its old slug", "slug", existing.Slug(), "error", err) } } return saved, nil }