799 lines
23 KiB
Go
799 lines
23 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
|
||
|
|
|
||
|
|
// Package payloads builds the public API responses: filtering,
|
||
|
|
// pagination, tag clouds, series, and post validation helpers.
|
||
|
|
package payloads
|
||
|
|
|
||
|
|
import (
|
||
|
|
json "encoding/json/v2"
|
||
|
|
"fmt"
|
||
|
|
"log/slog"
|
||
|
|
"maps"
|
||
|
|
"regexp"
|
||
|
|
"slices"
|
||
|
|
"strconv"
|
||
|
|
"strings"
|
||
|
|
"time"
|
||
|
|
"unicode/utf8"
|
||
|
|
|
||
|
|
"sourcedock.dev/petrbalvin/interpres/v2"
|
||
|
|
"sourcedock.dev/petrbalvin/volumen/internal/fediverse"
|
||
|
|
"sourcedock.dev/petrbalvin/volumen/internal/frontmatter"
|
||
|
|
"sourcedock.dev/petrbalvin/volumen/internal/identifiers"
|
||
|
|
"sourcedock.dev/petrbalvin/volumen/internal/markdown"
|
||
|
|
"sourcedock.dev/petrbalvin/volumen/internal/post"
|
||
|
|
)
|
||
|
|
|
||
|
|
var (
|
||
|
|
// SlugRegex accepts lowercase slugs with inner dots, dashes and
|
||
|
|
// underscores.
|
||
|
|
SlugRegex = regexp.MustCompile(`^[a-z0-9](?:[a-z0-9._-]*[a-z0-9])?$`)
|
||
|
|
// LangRegex accepts simple language codes such as cs or pt-BR.
|
||
|
|
LangRegex = regexp.MustCompile(`^[A-Za-z0-9_-]+$`)
|
||
|
|
)
|
||
|
|
|
||
|
|
// MaxSlugLength caps slug length.
|
||
|
|
const MaxSlugLength = 200
|
||
|
|
|
||
|
|
// MaxPageSize is the largest page the list endpoints return, and the
|
||
|
|
// documented limit.
|
||
|
|
const MaxPageSize = 100
|
||
|
|
|
||
|
|
// maxSeriesOrder is the sort key for a post that carries no
|
||
|
|
// series_order: it sorts after every explicit order.
|
||
|
|
const maxSeriesOrder = 1 << 30
|
||
|
|
|
||
|
|
// Source is what the payload builders read from a content store: the
|
||
|
|
// whole set for a listing, and one post by slug for a validation rule.
|
||
|
|
type Source interface {
|
||
|
|
All() []*post.Post
|
||
|
|
Find(slug, lang string) *post.Post
|
||
|
|
}
|
||
|
|
|
||
|
|
// PublishedPosts returns all published posts, newest first. A post that
|
||
|
|
// is a draft or still scheduled is withheld.
|
||
|
|
func PublishedPosts(s Source) []*post.Post {
|
||
|
|
var visible []*post.Post
|
||
|
|
for _, p := range s.All() {
|
||
|
|
if p.Published() {
|
||
|
|
visible = append(visible, p)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
slices.SortStableFunc(visible, func(a, b *post.Post) int {
|
||
|
|
return strings.Compare(b.DateString(), a.DateString())
|
||
|
|
})
|
||
|
|
return visible
|
||
|
|
}
|
||
|
|
|
||
|
|
// FilterPosts filters published posts by lang, tag and query.
|
||
|
|
func FilterPosts(posts []*post.Post, lang, tag, query string) []*post.Post {
|
||
|
|
if lang != "" {
|
||
|
|
var out []*post.Post
|
||
|
|
for _, p := range posts {
|
||
|
|
if p.Lang() == lang || p.AllLangs() {
|
||
|
|
out = append(out, p)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
posts = out
|
||
|
|
}
|
||
|
|
if tag != "" {
|
||
|
|
var out []*post.Post
|
||
|
|
for _, p := range posts {
|
||
|
|
if slices.Contains(p.Tags(), tag) {
|
||
|
|
out = append(out, p)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
posts = out
|
||
|
|
}
|
||
|
|
query = strings.ToLower(strings.TrimSpace(query))
|
||
|
|
if query != "" {
|
||
|
|
scored := make([]scoredPost, 0, len(posts))
|
||
|
|
for _, p := range posts {
|
||
|
|
if score := searchScore(p, query); score > 0 {
|
||
|
|
scored = append(scored, scoredPost{post: p, score: score})
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Relevance first: the strongest match leads, and posts that
|
||
|
|
// score the same keep the date order they arrived in.
|
||
|
|
slices.SortStableFunc(scored, func(a, b scoredPost) int {
|
||
|
|
return b.score - a.score
|
||
|
|
})
|
||
|
|
posts = make([]*post.Post, len(scored))
|
||
|
|
for i, s := range scored {
|
||
|
|
posts[i] = s.post
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return posts
|
||
|
|
}
|
||
|
|
|
||
|
|
// scoredPost pairs a post with the relevance it scored against the
|
||
|
|
// running query.
|
||
|
|
type scoredPost struct {
|
||
|
|
post *post.Post
|
||
|
|
score int
|
||
|
|
}
|
||
|
|
|
||
|
|
// searchScore ranks one post against the lowered query: a match higher
|
||
|
|
// in the post weighs more, so a title hit outranks a passing mention in
|
||
|
|
// the body. Zero means the post does not match at all and leaves the
|
||
|
|
// result list.
|
||
|
|
func searchScore(p *post.Post, query string) int {
|
||
|
|
score := 0
|
||
|
|
title := strings.ToLower(p.Title())
|
||
|
|
if strings.Contains(title, query) {
|
||
|
|
score += 100
|
||
|
|
if strings.HasPrefix(title, query) {
|
||
|
|
score += 50
|
||
|
|
}
|
||
|
|
}
|
||
|
|
for _, tag := range p.Tags() {
|
||
|
|
tag = strings.ToLower(tag)
|
||
|
|
switch {
|
||
|
|
case tag == query:
|
||
|
|
score += 40
|
||
|
|
case strings.Contains(tag, query):
|
||
|
|
score += 15
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if strings.Contains(strings.ToLower(p.Excerpt()), query) {
|
||
|
|
score += 20
|
||
|
|
}
|
||
|
|
if hits := strings.Count(strings.ToLower(p.Body), query); hits > 0 {
|
||
|
|
score += 10 + min(hits-1, 4)*2
|
||
|
|
}
|
||
|
|
return score
|
||
|
|
}
|
||
|
|
|
||
|
|
// PostsPayload builds the paginated payload for /api/volumen/posts.
|
||
|
|
// With a cursor the list starts after that slug and page numbers are
|
||
|
|
// ignored; otherwise the list is page-numbered. The two shapes differ
|
||
|
|
// in which fields they carry, so they are distinct types. A cursor that
|
||
|
|
// names no post yields an empty page rather than silently rewinding to
|
||
|
|
// the first one, which would send a paging client posts it already has.
|
||
|
|
func PostsPayload(s Source, lang, tag, query string, page, limit int, cursor string) any {
|
||
|
|
posts := FilterPosts(PublishedPosts(s), lang, tag, query)
|
||
|
|
limit = min(max(limit, 1), MaxPageSize)
|
||
|
|
|
||
|
|
total := len(posts)
|
||
|
|
offset := 0
|
||
|
|
if cursor != "" {
|
||
|
|
offset = cursorOffset(posts, cursor)
|
||
|
|
} else {
|
||
|
|
page = max(page, 1)
|
||
|
|
offset = (page - 1) * limit
|
||
|
|
}
|
||
|
|
|
||
|
|
end := min(offset+limit, total)
|
||
|
|
var pagePosts []*post.Post
|
||
|
|
if offset < total {
|
||
|
|
pagePosts = posts[offset:end]
|
||
|
|
}
|
||
|
|
|
||
|
|
summaries := make([]Summary, 0, len(pagePosts))
|
||
|
|
for _, p := range pagePosts {
|
||
|
|
summaries = append(summaries, BuildSummary(p))
|
||
|
|
}
|
||
|
|
base := PostListBase{PageSize: limit, Total: total, Posts: summaries}
|
||
|
|
|
||
|
|
var nextCursor *string
|
||
|
|
if len(pagePosts) > 0 && offset+limit < total {
|
||
|
|
last := pagePosts[len(pagePosts)-1]
|
||
|
|
value := last.Slug()
|
||
|
|
if slugShared(posts, last.Slug()) {
|
||
|
|
// Two posts share the slug (translations do): name the exact
|
||
|
|
// post, or the next page resumes after the first variant and
|
||
|
|
// serves the second one twice.
|
||
|
|
value += "." + last.Lang()
|
||
|
|
}
|
||
|
|
nextCursor = &value
|
||
|
|
}
|
||
|
|
if cursor != "" {
|
||
|
|
return CursorList{PostListBase: base, NextCursor: nextCursor}
|
||
|
|
}
|
||
|
|
return PageList{
|
||
|
|
PostListBase: base,
|
||
|
|
Page: max(page, 1),
|
||
|
|
HasNext: offset+limit < total,
|
||
|
|
HasPrev: max(page, 1) > 1,
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
// cursorOffset resolves a cursor to the index the next page starts
|
||
|
|
// after. A cursor is "slug", or "slug.lang" when the slug is shared by
|
||
|
|
// translations: the composite form is matched first so an ambiguous
|
||
|
|
// slug cannot resume the walk at the wrong variant. A cursor that names
|
||
|
|
// no post yields the end, an empty page, rather than rewinding.
|
||
|
|
func cursorOffset(posts []*post.Post, cursor string) int {
|
||
|
|
for i, p := range posts {
|
||
|
|
if p.Slug()+"."+p.Lang() == cursor {
|
||
|
|
return i + 1
|
||
|
|
}
|
||
|
|
}
|
||
|
|
for i, p := range posts {
|
||
|
|
if p.Slug() == cursor {
|
||
|
|
return i + 1
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return len(posts)
|
||
|
|
}
|
||
|
|
|
||
|
|
// slugShared reports whether more than one post in the list carries the
|
||
|
|
// slug.
|
||
|
|
func slugShared(posts []*post.Post, slug string) bool {
|
||
|
|
count := 0
|
||
|
|
for _, p := range posts {
|
||
|
|
if p.Slug() == slug {
|
||
|
|
count++
|
||
|
|
if count > 1 {
|
||
|
|
return true
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return false
|
||
|
|
}
|
||
|
|
|
||
|
|
// IsEmpty reports whether a paginated payload carries no posts. A
|
||
|
|
// payload of an unknown type is not treated as empty, so a future shape
|
||
|
|
// cannot accidentally turn into a 404.
|
||
|
|
func IsEmpty(payload any) bool {
|
||
|
|
switch list := payload.(type) {
|
||
|
|
case PageList:
|
||
|
|
return len(list.Posts) == 0
|
||
|
|
case CursorList:
|
||
|
|
return len(list.Posts) == 0
|
||
|
|
}
|
||
|
|
return false
|
||
|
|
}
|
||
|
|
|
||
|
|
// BuildTagCounts counts the tags of the given posts, sorted by
|
||
|
|
// frequency then name. The caller chooses the post set, so the admin
|
||
|
|
// dashboard counts its own view (drafts included) and the API counts the
|
||
|
|
// published posts.
|
||
|
|
func BuildTagCounts(posts []*post.Post) []CountedName {
|
||
|
|
counts := map[string]int{}
|
||
|
|
for _, p := range posts {
|
||
|
|
for _, tag := range p.Tags() {
|
||
|
|
counts[tag]++
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return orderedCounts(counts)
|
||
|
|
}
|
||
|
|
|
||
|
|
// BuildSeriesList lists series with their post counts, most posts
|
||
|
|
// first.
|
||
|
|
func BuildSeriesList(s Source) []CountedName {
|
||
|
|
counts := map[string]int{}
|
||
|
|
for _, p := range PublishedPosts(s) {
|
||
|
|
if p.Series() != "" {
|
||
|
|
counts[p.Series()]++
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return orderedCounts(counts)
|
||
|
|
}
|
||
|
|
|
||
|
|
func orderedCounts(counts map[string]int) []CountedName {
|
||
|
|
// slices.SortedFunc takes the keys as an iterator, so the list is
|
||
|
|
// built and ordered in one step.
|
||
|
|
names := slices.SortedFunc(maps.Keys(counts), func(a, b string) int {
|
||
|
|
if counts[a] != counts[b] {
|
||
|
|
return counts[b] - counts[a]
|
||
|
|
}
|
||
|
|
return strings.Compare(a, b)
|
||
|
|
})
|
||
|
|
out := make([]CountedName, 0, len(names))
|
||
|
|
for _, name := range names {
|
||
|
|
out = append(out, CountedName{Name: name, Count: counts[name]})
|
||
|
|
}
|
||
|
|
return out
|
||
|
|
}
|
||
|
|
|
||
|
|
// SeriesPosts returns the published posts of one series, ordered by
|
||
|
|
// series_order then date then slug.
|
||
|
|
func SeriesPosts(s Source, name string) []*post.Post {
|
||
|
|
var posts []*post.Post
|
||
|
|
for _, p := range PublishedPosts(s) {
|
||
|
|
if p.Series() == name {
|
||
|
|
posts = append(posts, p)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
slices.SortStableFunc(posts, func(a, b *post.Post) int {
|
||
|
|
orderA, okA := a.SeriesOrder()
|
||
|
|
if !okA {
|
||
|
|
orderA = maxSeriesOrder
|
||
|
|
}
|
||
|
|
orderB, okB := b.SeriesOrder()
|
||
|
|
if !okB {
|
||
|
|
orderB = maxSeriesOrder
|
||
|
|
}
|
||
|
|
if orderA != orderB {
|
||
|
|
return orderA - orderB
|
||
|
|
}
|
||
|
|
if c := strings.Compare(a.DateString(), b.DateString()); c != 0 {
|
||
|
|
return c
|
||
|
|
}
|
||
|
|
return strings.Compare(a.Slug(), b.Slug())
|
||
|
|
})
|
||
|
|
return posts
|
||
|
|
}
|
||
|
|
|
||
|
|
// Presence returns a stripped string, or "" when empty.
|
||
|
|
func Presence(value any) string {
|
||
|
|
if value == nil {
|
||
|
|
return ""
|
||
|
|
}
|
||
|
|
return strings.TrimSpace(fmt.Sprintf("%v", value))
|
||
|
|
}
|
||
|
|
|
||
|
|
// ParseDate parses an ISO 8601 date string.
|
||
|
|
func ParseDate(value any) (time.Time, bool) {
|
||
|
|
text := Presence(value)
|
||
|
|
if text == "" {
|
||
|
|
return time.Time{}, false
|
||
|
|
}
|
||
|
|
t, err := time.Parse("2006-01-02", text)
|
||
|
|
if err != nil {
|
||
|
|
return time.Time{}, false
|
||
|
|
}
|
||
|
|
return t, true
|
||
|
|
}
|
||
|
|
|
||
|
|
// ParseInt parses an integer from form data.
|
||
|
|
func ParseInt(value any) (int, bool) {
|
||
|
|
text := Presence(value)
|
||
|
|
if text == "" {
|
||
|
|
return 0, false
|
||
|
|
}
|
||
|
|
n, err := strconv.Atoi(text)
|
||
|
|
if err != nil {
|
||
|
|
return 0, false
|
||
|
|
}
|
||
|
|
return n, true
|
||
|
|
}
|
||
|
|
|
||
|
|
// ParseTags parses a comma-separated tag string.
|
||
|
|
func ParseTags(value any) []string {
|
||
|
|
raw := Presence(value)
|
||
|
|
if raw == "" {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
var out []string
|
||
|
|
for tag := range strings.SplitSeq(raw, ",") {
|
||
|
|
if trimmed := strings.TrimSpace(tag); trimmed != "" {
|
||
|
|
out = append(out, trimmed)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return out
|
||
|
|
}
|
||
|
|
|
||
|
|
// PostFromParams builds a post from admin form data, carrying the
|
||
|
|
// existing post's path when editing. Metadata keys the form does not
|
||
|
|
// manage (aliases, translations, custom fields) are inherited from the
|
||
|
|
// existing post so an editor save never drops them. A date field the
|
||
|
|
// form carries but cannot parse is reported as a ValidationError on the
|
||
|
|
// built post: an unparseable value dropping the key would silently
|
||
|
|
// publish a scheduled post. The caller renders the returned post back
|
||
|
|
// into the form either way.
|
||
|
|
func PostFromParams(form map[string]string, existing *post.Post) (*post.Post, error) {
|
||
|
|
badDateField := ""
|
||
|
|
for _, field := range []string{"date", "publish_at"} {
|
||
|
|
if Presence(form[field]) != "" {
|
||
|
|
if _, ok := ParseDate(form[field]); !ok {
|
||
|
|
badDateField = field
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
badUTF8Field := ""
|
||
|
|
for _, field := range append(slices.Clone(metadataOrder), "body") {
|
||
|
|
if value, present := form[field]; present && !utf8.ValidString(value) {
|
||
|
|
badUTF8Field = field
|
||
|
|
break
|
||
|
|
}
|
||
|
|
}
|
||
|
|
m := frontmatter.NewMeta()
|
||
|
|
if existing != nil {
|
||
|
|
// The clone carries the comments and the nested shapes of the
|
||
|
|
// stored file: the form only rewrites its own fields, and every
|
||
|
|
// key it does not name round-trips untouched.
|
||
|
|
m = existing.Metadata.Clone()
|
||
|
|
}
|
||
|
|
cleaned := cleanMetadata(baseMetadata(form, existing))
|
||
|
|
for _, key := range metadataOrder {
|
||
|
|
if key == badDateField {
|
||
|
|
// The form value was rejected, so the field keeps its
|
||
|
|
// inherited value: the post goes back into the editor with
|
||
|
|
// the schedule it had, not with the bad input written into
|
||
|
|
// the metadata nor with the schedule silently dropped.
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
value, present := cleaned[key]
|
||
|
|
if !present {
|
||
|
|
// Cleared in the form: drop any inherited value too.
|
||
|
|
m.Delete(key)
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
m.Set(key, value)
|
||
|
|
}
|
||
|
|
p := post.New(m, form["body"])
|
||
|
|
if existing != nil {
|
||
|
|
p.Path = existing.Path
|
||
|
|
}
|
||
|
|
if badUTF8Field != "" {
|
||
|
|
// Saving would silently replace the invalid bytes with U+FFFD,
|
||
|
|
// so the form is rejected instead: the author sees the field that
|
||
|
|
// carries them and keeps control over the text.
|
||
|
|
return p, invalid(fmt.Sprintf("%s must be valid UTF-8 text.", badUTF8Field))
|
||
|
|
}
|
||
|
|
if badDateField != "" {
|
||
|
|
return p, invalid(fmt.Sprintf("%s must be an ISO 8601 date.", badDateField))
|
||
|
|
}
|
||
|
|
// The references arrive as one JSON field from the editor. An absent
|
||
|
|
// field means the form never carried one, and the stored list
|
||
|
|
// round-trips untouched; an empty list is an explicit deletion.
|
||
|
|
tables, present, err := refsFromForm(form)
|
||
|
|
if err != nil {
|
||
|
|
return p, err
|
||
|
|
}
|
||
|
|
if present {
|
||
|
|
if len(tables) == 0 {
|
||
|
|
m.Delete("refs")
|
||
|
|
} else {
|
||
|
|
m.Set("refs", tables)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return p, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
var metadataOrder = []string{
|
||
|
|
"title", "slug", "lang", "author", "fediverse_creator",
|
||
|
|
"doi", "orcid",
|
||
|
|
"date", "publish_at", "tags", "excerpt", "cover", "cover_alt",
|
||
|
|
"cover_caption", "series", "series_order", "draft", "all_langs",
|
||
|
|
}
|
||
|
|
|
||
|
|
func baseMetadata(form map[string]string, existing *post.Post) map[string]any {
|
||
|
|
slug := Presence(form["slug"])
|
||
|
|
if slug == "" && existing != nil {
|
||
|
|
slug = existing.Slug()
|
||
|
|
}
|
||
|
|
meta := map[string]any{
|
||
|
|
"title": Presence(form["title"]),
|
||
|
|
"slug": slug,
|
||
|
|
"lang": Presence(form["lang"]),
|
||
|
|
"author": Presence(form["author"]),
|
||
|
|
"fediverse_creator": Presence(form["fediverse_creator"]),
|
||
|
|
"doi": identifiers.NormalizeDOI(form["doi"]),
|
||
|
|
"orcid": identifiers.NormalizeORCID(form["orcid"]),
|
||
|
|
"tags": ParseTags(form["tags"]),
|
||
|
|
"excerpt": Presence(form["excerpt"]),
|
||
|
|
"cover": Presence(form["cover"]),
|
||
|
|
"cover_alt": Presence(form["cover_alt"]),
|
||
|
|
"cover_caption": Presence(form["cover_caption"]),
|
||
|
|
"series": Presence(form["series"]),
|
||
|
|
}
|
||
|
|
if d, ok := ParseDate(form["date"]); ok {
|
||
|
|
meta["date"] = interpres.LocalDate{Time: d}
|
||
|
|
}
|
||
|
|
if d, ok := ParseDate(form["publish_at"]); ok {
|
||
|
|
meta["publish_at"] = interpres.LocalDate{Time: d}
|
||
|
|
}
|
||
|
|
if n, ok := ParseInt(form["series_order"]); ok {
|
||
|
|
meta["series_order"] = int64(n)
|
||
|
|
}
|
||
|
|
if form["draft"] == "on" {
|
||
|
|
meta["draft"] = true
|
||
|
|
}
|
||
|
|
if form["all_langs"] == "on" {
|
||
|
|
meta["all_langs"] = true
|
||
|
|
}
|
||
|
|
return meta
|
||
|
|
}
|
||
|
|
|
||
|
|
// cleanMetadata drops nil, empty and false entries, keeping the
|
||
|
|
// original key order.
|
||
|
|
func cleanMetadata(meta map[string]any) map[string]any {
|
||
|
|
out := make(map[string]any, len(meta))
|
||
|
|
for _, key := range metadataOrder {
|
||
|
|
value, ok := meta[key]
|
||
|
|
if !ok || value == nil {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
switch v := value.(type) {
|
||
|
|
case string:
|
||
|
|
if v == "" {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
case []string:
|
||
|
|
if len(v) == 0 {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
case bool:
|
||
|
|
if !v {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
}
|
||
|
|
out[key] = value
|
||
|
|
}
|
||
|
|
return out
|
||
|
|
}
|
||
|
|
|
||
|
|
// ValidationError is a user-facing validation failure. The message is
|
||
|
|
// shown verbatim in the API error envelope and in the admin forms.
|
||
|
|
type ValidationError struct {
|
||
|
|
Message string
|
||
|
|
}
|
||
|
|
|
||
|
|
func (e *ValidationError) Error() string { return e.Message }
|
||
|
|
|
||
|
|
func invalid(message string) error { return &ValidationError{Message: message} }
|
||
|
|
|
||
|
|
// The bounds of one reference list written from the editor. They exist so
|
||
|
|
// a runaway payload dies at a named rule rather than at the body limit.
|
||
|
|
const (
|
||
|
|
maxRefEntries = 500
|
||
|
|
maxRefField = 4000
|
||
|
|
maxRefAuthors = 200
|
||
|
|
)
|
||
|
|
|
||
|
|
// refsFromForm decodes the editor's reference list. The second return
|
||
|
|
// value reports whether the form carried the field at all: absent means
|
||
|
|
// the stored list survives untouched, present means the decoded list (or
|
||
|
|
// its deletion) is the author's explicit choice. A row that carries
|
||
|
|
// nothing at all is dropped rather than saved as an empty table. The
|
||
|
|
// tables come back as []map[string]any, the shape the frontmatter writer
|
||
|
|
// renders as [[refs]] blocks rather than one inline array.
|
||
|
|
func refsFromForm(form map[string]string) ([]map[string]any, bool, error) {
|
||
|
|
encoded, present := form["refs"]
|
||
|
|
if !present {
|
||
|
|
return nil, false, nil
|
||
|
|
}
|
||
|
|
if strings.TrimSpace(encoded) == "" {
|
||
|
|
return nil, true, nil
|
||
|
|
}
|
||
|
|
var entries []map[string]any
|
||
|
|
if err := json.Unmarshal([]byte(encoded), &entries); err != nil {
|
||
|
|
return nil, true, invalid("References must be a list of entries.")
|
||
|
|
}
|
||
|
|
if len(entries) > maxRefEntries {
|
||
|
|
return nil, true, invalid(fmt.Sprintf("References must hold at most %d entries.", maxRefEntries))
|
||
|
|
}
|
||
|
|
out := make([]map[string]any, 0, len(entries))
|
||
|
|
for i, entry := range entries {
|
||
|
|
table, err := cleanRefEntry(entry, i+1)
|
||
|
|
if err != nil {
|
||
|
|
return nil, true, err
|
||
|
|
}
|
||
|
|
if table != nil {
|
||
|
|
out = append(out, table)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return out, true, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// cleanRefEntry validates one reference table and returns it in the
|
||
|
|
// canonical shape the frontmatter writer accepts: empty strings dropped,
|
||
|
|
// identifiers normalised, authors as strings or name tables. A row whose
|
||
|
|
// every field is blank returns nil, meaning it is dropped.
|
||
|
|
func cleanRefEntry(entry map[string]any, position int) (map[string]any, error) {
|
||
|
|
which := fmt.Sprintf("Reference %d", position)
|
||
|
|
clip := func(value any, field string) (string, error) {
|
||
|
|
var text string
|
||
|
|
switch v := value.(type) {
|
||
|
|
case string:
|
||
|
|
text = v
|
||
|
|
case float64:
|
||
|
|
// A year or volume the client sent as a JSON number still
|
||
|
|
// belongs in the table, as the string the model stores.
|
||
|
|
if v == float64(int64(v)) {
|
||
|
|
text = strconv.FormatInt(int64(v), 10)
|
||
|
|
} else {
|
||
|
|
text = strconv.FormatFloat(v, 'f', -1, 64)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
text = strings.TrimSpace(text)
|
||
|
|
if len(text) > maxRefField {
|
||
|
|
return "", invalid(fmt.Sprintf("%s: %s must be at most %d characters.", which, field, maxRefField))
|
||
|
|
}
|
||
|
|
return text, nil
|
||
|
|
}
|
||
|
|
out := map[string]any{}
|
||
|
|
for _, field := range []string{"raw", "title", "venue", "year", "volume", "pages", "arxiv", "url"} {
|
||
|
|
text, err := clip(entry[field], field)
|
||
|
|
if err != nil {
|
||
|
|
return nil, err
|
||
|
|
}
|
||
|
|
if text != "" {
|
||
|
|
out[field] = text
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if doi, _ := entry["doi"].(string); strings.TrimSpace(doi) != "" {
|
||
|
|
doi = identifiers.NormalizeDOI(doi)
|
||
|
|
if !identifiers.ValidDOI(doi) {
|
||
|
|
return nil, invalid(fmt.Sprintf("%s: DOI must look like 10.xxxx/suffix.", which))
|
||
|
|
}
|
||
|
|
out["doi"] = doi
|
||
|
|
}
|
||
|
|
if arxiv, ok := out["arxiv"]; ok {
|
||
|
|
id := strings.TrimPrefix(strings.TrimPrefix(arxiv.(string), "https://arxiv.org/abs/"), "arXiv:")
|
||
|
|
if id == "" || strings.ContainsAny(id, " \t\"'<>") {
|
||
|
|
return nil, invalid(fmt.Sprintf("%s: arXiv must be the bare identifier, e.g. 2401.12345.", which))
|
||
|
|
}
|
||
|
|
out["arxiv"] = id
|
||
|
|
}
|
||
|
|
if urlField, ok := out["url"]; ok {
|
||
|
|
if !strings.HasPrefix(urlField.(string), "http://") && !strings.HasPrefix(urlField.(string), "https://") {
|
||
|
|
return nil, invalid(fmt.Sprintf("%s: URL must be an http(s) address.", which))
|
||
|
|
}
|
||
|
|
}
|
||
|
|
authors, err := cleanRefAuthors(entry["authors"], which)
|
||
|
|
if err != nil {
|
||
|
|
return nil, err
|
||
|
|
}
|
||
|
|
if len(authors) > 0 {
|
||
|
|
out["authors"] = authors
|
||
|
|
}
|
||
|
|
// An explicit number survives a round trip, so a hand-numbered list
|
||
|
|
// keeps its numbering through the editor.
|
||
|
|
switch n := entry["num"].(type) {
|
||
|
|
case float64:
|
||
|
|
if n == float64(int(n)) && n >= 1 && n <= 9999 {
|
||
|
|
out["num"] = int64(n)
|
||
|
|
}
|
||
|
|
case string:
|
||
|
|
if parsed, err := strconv.Atoi(strings.TrimSpace(n)); err == nil && parsed >= 1 && parsed <= 9999 {
|
||
|
|
out["num"] = int64(parsed)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// A row with only identifiers and no citation of its own would render
|
||
|
|
// as an empty entry; an entirely blank row is simply dropped.
|
||
|
|
if _, cited := out["raw"]; !cited {
|
||
|
|
if _, ok := out["title"]; !ok {
|
||
|
|
if _, hasAuthors := out["authors"]; !hasAuthors {
|
||
|
|
if len(out) == 0 {
|
||
|
|
return nil, nil
|
||
|
|
}
|
||
|
|
return nil, invalid(fmt.Sprintf("%s needs the citation itself: the verbatim line, the title, or the authors.", which))
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return out, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// cleanRefAuthors validates the author list of one reference: plain name
|
||
|
|
// strings, or name tables with an optional ORCID checked to its digit.
|
||
|
|
func cleanRefAuthors(value any, which string) ([]any, error) {
|
||
|
|
list, ok := value.([]any)
|
||
|
|
if !ok {
|
||
|
|
return nil, nil
|
||
|
|
}
|
||
|
|
if len(list) > maxRefAuthors {
|
||
|
|
return nil, invalid(fmt.Sprintf("%s: at most %d authors per reference.", which, maxRefAuthors))
|
||
|
|
}
|
||
|
|
out := make([]any, 0, len(list))
|
||
|
|
for _, item := range list {
|
||
|
|
switch author := item.(type) {
|
||
|
|
case string:
|
||
|
|
if name := strings.TrimSpace(author); name != "" {
|
||
|
|
out = append(out, name)
|
||
|
|
}
|
||
|
|
case map[string]any:
|
||
|
|
name, _ := author["name"].(string)
|
||
|
|
name = strings.TrimSpace(name)
|
||
|
|
if name == "" {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
orcid, _ := author["orcid"].(string)
|
||
|
|
orcid = identifiers.NormalizeORCID(strings.TrimSpace(orcid))
|
||
|
|
if orcid != "" && !identifiers.ValidORCID(orcid) {
|
||
|
|
return nil, invalid(fmt.Sprintf("%s: ORCID must look like 0000-0002-1825-0097.", which))
|
||
|
|
}
|
||
|
|
if orcid != "" {
|
||
|
|
out = append(out, map[string]any{"name": name, "orcid": orcid})
|
||
|
|
} else {
|
||
|
|
out = append(out, name)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return out, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// CreationError validates a post before saving; nil means valid.
|
||
|
|
func CreationError(p *post.Post, s Source, existing *post.Post) error {
|
||
|
|
if Presence(p.Slug()) == "" {
|
||
|
|
return invalid("Slug is required.")
|
||
|
|
}
|
||
|
|
slug := p.Slug()
|
||
|
|
if len([]rune(slug)) > MaxSlugLength {
|
||
|
|
return invalid(fmt.Sprintf("Slug must be at most %d characters.", MaxSlugLength))
|
||
|
|
}
|
||
|
|
if !SlugRegex.MatchString(slug) {
|
||
|
|
return invalid("Invalid slug.")
|
||
|
|
}
|
||
|
|
if lang := p.Lang(); lang != "" && !LangRegex.MatchString(lang) {
|
||
|
|
return invalid("Invalid language.")
|
||
|
|
}
|
||
|
|
if len(p.Body) > markdown.MaxBodyLength {
|
||
|
|
return invalid(fmt.Sprintf("Body must be at most %d bytes.", markdown.MaxBodyLength))
|
||
|
|
}
|
||
|
|
if found := s.Find(slug, ""); found != nil &&
|
||
|
|
(existing == nil || found.Path != existing.Path) {
|
||
|
|
return invalid("A post with that slug already exists.")
|
||
|
|
}
|
||
|
|
if err := FediverseCreatorError(p); err != nil {
|
||
|
|
return err
|
||
|
|
}
|
||
|
|
if err := DOIError(p); err != nil {
|
||
|
|
return err
|
||
|
|
}
|
||
|
|
return ORCIDError(p)
|
||
|
|
}
|
||
|
|
|
||
|
|
// FediverseCreatorError validates the optional fediverse handle.
|
||
|
|
func FediverseCreatorError(p *post.Post) error {
|
||
|
|
value, ok := p.Metadata.Get("fediverse_creator")
|
||
|
|
if !ok || value == nil {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
if fediverse.Valid(fmt.Sprintf("%v", value)) {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
return invalid("Fediverse creator must look like @user@host.")
|
||
|
|
}
|
||
|
|
|
||
|
|
// DOIError validates the optional DOI. The stored value is the bare
|
||
|
|
// form; a doi.org URL or doi: prefix normalises away before the rule.
|
||
|
|
func DOIError(p *post.Post) error {
|
||
|
|
value, ok := p.Metadata.Get("doi")
|
||
|
|
if !ok || value == nil {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
text := fmt.Sprintf("%v", value)
|
||
|
|
if text == "" || identifiers.ValidDOI(text) {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
return invalid("DOI must look like 10.xxxx/suffix.")
|
||
|
|
}
|
||
|
|
|
||
|
|
// ORCIDError validates the optional ORCID iD, check digit included.
|
||
|
|
func ORCIDError(p *post.Post) error {
|
||
|
|
value, ok := p.Metadata.Get("orcid")
|
||
|
|
if !ok || value == nil {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
text := fmt.Sprintf("%v", value)
|
||
|
|
if text == "" || identifiers.ValidORCID(text) {
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
return invalid("ORCID must look like 0000-0002-1825-0097.")
|
||
|
|
}
|
||
|
|
|
||
|
|
// Repository is what the payload builders write through.
|
||
|
|
type Repository interface {
|
||
|
|
Source
|
||
|
|
Save(p *post.Post) (*post.Post, error)
|
||
|
|
Delete(slug, lang string) (*post.Post, bool, error)
|
||
|
|
}
|
||
|
|
|
||
|
|
// SavePost persists a post through the repository. When the slug changed
|
||
|
|
// since existing, the file moves to the path its new slug implies and
|
||
|
|
// the old file is soft-deleted, so a rename is one operation with one
|
||
|
|
// undo and one revision archive. The admin and the API both save through
|
||
|
|
// here, so the on-disk result never depends on which one asked.
|
||
|
|
func SavePost(st Repository, p, existing *post.Post) (*post.Post, error) {
|
||
|
|
renamed := existing != nil && existing.Path != "" && p.Slug() != "" && p.Slug() != existing.Slug()
|
||
|
|
if renamed {
|
||
|
|
p.Path = ""
|
||
|
|
}
|
||
|
|
saved, err := st.Save(p)
|
||
|
|
if err != nil {
|
||
|
|
return nil, err
|
||
|
|
}
|
||
|
|
if renamed {
|
||
|
|
if _, _, err := st.Delete(existing.Slug(), existing.Lang()); err != nil {
|
||
|
|
slog.Warn("payloads: could not archive the post under its old slug",
|
||
|
|
"slug", existing.Slug(), "error", err)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return saved, nil
|
||
|
|
}
|