Files

511 lines
14 KiB
Go
Raw Permalink Normal View History

2026-09-18 12:03:35 +02:00
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
// Package store reads and writes Markdown posts and media in a content
// directory: an mtime-snapshot cache, atomic writes, per-target locks,
// revision archives, and rename-based soft deletes.
package store
import (
"errors"
"fmt"
"io/fs"
"log/slog"
"os"
"path"
"path/filepath"
"regexp"
"slices"
"strings"
"sync"
"sourcedock.dev/petrbalvin/volumen/internal/identifiers"
"sourcedock.dev/petrbalvin/volumen/internal/post"
)
const revisionsDirname = ".revisions"
// maxPostFileBytes bounds one content file the cache will read. A saved
// post cannot come near it (the body is capped at 1 MiB before it is
// ever written); the bound exists so an abandoned file dropped into the
// content directory cannot be pulled into memory whole on every scan.
const maxPostFileBytes = 8 << 20
var safeSlugRe = regexp.MustCompile(`[^a-zA-Z0-9._-]`)
// MediaDirName is the uploads directory inside the content directory.
const MediaDirName = "media"
// Store manages posts on disk in a flat or language-subdivided
// directory.
type Store struct {
ContentDir string
defaultLang string
followSymlinks bool
revisionLimit int
mu sync.Mutex
cached []*post.Post
byPath map[string]*post.Post
snapshot []snapshotEntry
unreadable []Unreadable
haveCache bool
slugIndex map[string][]string
aliasIndex map[string]string
locksGuard sync.Mutex
saveLocks map[string]*lockEntry
// rootOnce guards root, a handle confined to the content directory.
// Every path the store touches outside the post walk is opened
// through it, so a name that would escape the tree through ".." or a
// symlink is refused by the kernel rather than checked for in Go.
rootOnce sync.Once
root *os.Root
rootErr error
}
// openRoot returns the handle confined to the content directory, opening
// it once per store.
func (s *Store) openRoot() (*os.Root, error) {
s.rootOnce.Do(func() {
if err := os.MkdirAll(s.ContentDir, 0o755); err != nil {
s.rootErr = fmt.Errorf("create the content directory: %w", err)
return
}
s.root, s.rootErr = os.OpenRoot(s.ContentDir)
})
return s.root, s.rootErr
}
// readDirIn lists a directory through the root, and returns nil when it
// does not exist.
func readDirIn(root *os.Root, dir string) []os.DirEntry {
handle, err := root.Open(dir)
if err != nil {
return nil
}
defer handle.Close()
entries, err := handle.ReadDir(-1)
if err != nil {
return nil
}
return entries
}
// statIn reports one entry of a directory listed through the root.
func statIn(root *os.Root, dir, name string) (os.FileInfo, bool) {
info, err := root.Stat(path.Join(dir, name))
if err != nil {
return nil, false
}
return info, true
}
// lockEntry is a per-target write lock plus the number of callers
// holding or waiting for it, so an entry can be dropped once it is idle.
type lockEntry struct {
mu sync.Mutex
refs int
}
type snapshotEntry struct {
path string
mtime int64
size int64
}
// Options configure a Store.
type Options struct {
// ContentDir is the directory holding the posts, and the media and
// revision directories inside it.
ContentDir string
// DefaultLang is the language of a post that has none and that does
// not sit in a language subdirectory.
DefaultLang string
// RevisionLimit is how many previous versions of a post to keep; zero
// or less disables archiving, which also makes a delete permanent.
RevisionLimit int
// FollowSymlinks reads a symlinked post file instead of skipping it.
FollowSymlinks bool
}
// New opens a store over Options.ContentDir. The path is resolved through
// any symlinks, so a content directory reached by a symlinked path walks
// and compares consistently.
func New(opts Options) *Store {
abs, err := filepath.Abs(opts.ContentDir)
if err != nil {
abs = opts.ContentDir
}
if resolved, err := filepath.EvalSymlinks(abs); err == nil {
abs = resolved
}
return &Store{
ContentDir: abs,
defaultLang: opts.DefaultLang,
followSymlinks: opts.FollowSymlinks,
revisionLimit: opts.RevisionLimit,
byPath: map[string]*post.Post{},
slugIndex: map[string][]string{},
aliasIndex: map[string]string{},
saveLocks: map[string]*lockEntry{},
}
}
// CleanupStaleTombstones lives in revision.go: a stale tombstone is a
// hazard only once the server serves the tree, so the read-only commands
// do not trigger it.
// InvalidateCache drops the cached posts so the next read re-scans the
// disk.
func (s *Store) InvalidateCache() {
s.mu.Lock()
defer s.mu.Unlock()
s.cached = nil
s.byPath = map[string]*post.Post{}
s.unreadable = nil
s.snapshot = nil
s.slugIndex = map[string][]string{}
s.aliasIndex = map[string]string{}
s.haveCache = false
}
// All returns every loadable post, using the mtime snapshot cache.
func (s *Store) All() []*post.Post {
s.mu.Lock()
defer s.mu.Unlock()
s.refresh()
return slices.Clone(s.cached)
}
// refresh rebuilds the cache when the (path, mtime, size) snapshot has
// changed, and does nothing on a hit. The caller must hold s.mu.
func (s *Store) refresh() {
snapshot := s.buildSnapshot()
if s.haveCache && slices.Equal(snapshot, s.snapshot) {
return
}
s.snapshot = snapshot
posts := make([]*post.Post, 0, len(snapshot))
byPath := make(map[string]*post.Post, len(snapshot))
var unreadable []Unreadable
for _, entry := range snapshot {
p, err := s.loadPost(entry.path)
if err != nil {
slog.Warn("store: skipping post", "path", entry.path, "error", err)
unreadable = append(unreadable, Unreadable{Path: entry.path, Error: err.Error()})
continue
}
posts = append(posts, p)
byPath[entry.path] = p
}
s.cached = posts
s.byPath = byPath
s.unreadable = unreadable
s.buildIndex()
s.buildLinkIndex()
s.haveCache = true
}
// buildLinkIndex maps every valid frontmatter DOI to the slug of the
// post published under it, and hands the map to the freshly loaded
// posts, so a reference citing one of those DOIs links to its post
// inside the instance instead of leaving for the resolver. The first
// post published under a DOI wins; the map is shared read-only.
func (s *Store) buildLinkIndex() {
index := map[string]string{}
for _, p := range s.cached {
doi := identifiers.NormalizeDOI(p.DOI())
if !identifiers.ValidDOI(doi) || p.Slug() == "" {
continue
}
key := strings.ToLower(doi)
if _, seen := index[key]; !seen {
index[key] = p.Slug()
}
}
for _, p := range s.cached {
p.SetLinkIndex(index)
}
}
// Unreadable is a content file the store could not load.
type Unreadable struct {
Path string
Error string
}
// Unreadable lists the content files the current scan could not read or
// parse. Every read path skips them, so a diagnostic has to ask for
// them explicitly.
func (s *Store) Unreadable() []Unreadable {
s.mu.Lock()
defer s.mu.Unlock()
s.refresh()
return slices.Clone(s.unreadable)
}
// Find returns the post with the given slug, optionally filtered by
// language (posts flagged all_langs match any language). The returned
// post is shared with the cache: clone it before mutating.
func (s *Store) Find(slug, lang string) *post.Post {
s.mu.Lock()
defer s.mu.Unlock()
s.refresh()
for _, path := range s.slugIndex[slug] {
if p := s.byPath[path]; p != nil && p.Slug() == slug && languageMatches(p, lang) {
return p
}
}
for _, p := range s.cached {
if p.Slug() == slug && languageMatches(p, lang) {
return p
}
}
return nil
}
func languageMatches(p *post.Post, lang string) bool {
return lang == "" || p.Lang() == lang || p.AllLangs()
}
// ResolveAlias returns the canonical slug for an alias, or "".
func (s *Store) ResolveAlias(alias string) string {
s.mu.Lock()
defer s.mu.Unlock()
s.refresh()
return s.aliasIndex[alias]
}
// CacheKey returns a string that changes whenever the content directory
// changes, for a caller that memoises rendering of the whole post set.
// It is derived from the same (path, mtime, size) snapshot the read
// cache uses, so an edit that leaves the slug and date alone still
// changes it.
func (s *Store) CacheKey() string {
s.mu.Lock()
defer s.mu.Unlock()
s.refresh()
var b strings.Builder
for _, entry := range s.snapshot {
fmt.Fprintf(&b, "%s|%d|%d;", entry.path, entry.mtime, entry.size)
}
return b.String()
}
// Save writes the post atomically, archiving the previous version as a
// revision first. It returns the post with Path set, and it sets Path on
// the argument it is given, so a caller passing a post obtained from the
// cache must clone it first.
func (s *Store) Save(p *post.Post) (*post.Post, error) {
target := p.Path
if target == "" {
var err error
target, err = s.defaultPathFor(p)
if err != nil {
return nil, err
}
}
root, err := s.openRoot()
if err != nil {
return nil, err
}
name, err := filepath.Rel(s.ContentDir, target)
if err != nil {
return nil, fmt.Errorf("locate %s: %w", target, err)
}
if err := root.MkdirAll(path.Dir(name), 0o755); err != nil {
return nil, fmt.Errorf("create post directory: %w", err)
}
unlock := s.targetLock(target)
defer unlock()
slug := p.Slug()
if slug == "" {
slug = strings.TrimSuffix(filepath.Base(target), ".md")
}
s.archiveRevision(name, slug)
content, err := p.ToFile()
if err != nil {
return nil, err
}
if err := atomicWriteIn(root, name, []byte(content)); err != nil {
return nil, err
}
p.Path = target
s.InvalidateCache()
return p, nil
}
// Delete removes the post. With revisions enabled the file is moved into
// the revision archive as a tombstone and undoable is true; with a
// non-positive revision limit it is removed outright and undoable is
// false. A nil post with a nil error means no such post.
func (s *Store) Delete(slug, lang string) (p *post.Post, undoable bool, err error) {
p = s.Find(slug, lang)
if p == nil || p.Path == "" {
return p, false, nil
}
unlock := s.targetLock(p.Path)
defer unlock()
if err := s.writeTombstone(p); err != nil {
// The file vanished between Find and the lock: the delete the
// caller asked for has already happened, so it is "no such post"
// rather than a failure.
if errors.Is(err, fs.ErrNotExist) {
return nil, false, nil
}
return nil, false, err
}
s.InvalidateCache()
return p, s.revisionLimit > 0, nil
}
// mdFiles lists the .md files in the content directory, relative to it.
func (s *Store) mdFiles() []string {
var results []string
root := s.ContentDir
_ = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
if err != nil {
// A walk that cannot read a subtree still lists the rest, but
// an unreadable root must not pass silently as an empty site.
slog.Error("store: cannot walk the content directory", "path", path, "error", err)
return nil
}
if d.IsDir() {
name := d.Name()
if path != root && (name == MediaDirName || name == revisionsDirname) {
return filepath.SkipDir
}
return nil
}
if !strings.HasSuffix(d.Name(), ".md") {
return nil
}
if !s.followSymlinks {
if info, err := d.Info(); err == nil && info.Mode()&os.ModeSymlink != 0 {
return nil
}
}
results = append(results, path)
return nil
})
return results
}
func (s *Store) buildSnapshot() []snapshotEntry {
files := s.mdFiles()
rows := make([]snapshotEntry, 0, len(files))
for _, f := range files {
info, err := os.Stat(f)
if err != nil {
continue
}
rows = append(rows, snapshotEntry{path: f, mtime: info.ModTime().UnixNano(), size: info.Size()})
}
slices.SortFunc(rows, func(a, b snapshotEntry) int { return strings.Compare(a.path, b.path) })
return rows
}
func (s *Store) buildIndex() {
s.slugIndex = map[string][]string{}
s.aliasIndex = map[string]string{}
for _, p := range s.cached {
if p.Path == "" || p.Slug() == "" {
continue
}
s.slugIndex[p.Slug()] = append(s.slugIndex[p.Slug()], p.Path)
for _, alias := range p.Aliases() {
s.aliasIndex[alias] = p.Slug()
}
}
}
func (s *Store) loadPost(path string) (*post.Post, error) {
if info, err := os.Stat(path); err == nil && info.Size() > maxPostFileBytes {
return nil, fmt.Errorf("file exceeds %d bytes and is not a post this engine could render", maxPostFileBytes)
}
content, err := os.ReadFile(path)
if err != nil {
return nil, err
}
parsed, err := post.Parse(string(content))
if err != nil {
return nil, err
}
p := parsed
// Both values are derived here and held outside the metadata, so a
// later save writes the file back as it was read.
slug := p.Slug()
if slug == "" {
slug = strings.TrimSuffix(filepath.Base(path), filepath.Ext(path))
}
lang := p.Lang()
if lang == "" {
lang = s.langFromPath(path)
if lang == "" {
lang = s.defaultLang
}
}
p.SetFileLocation(slug, lang)
p.Path = path
return p, nil
}
func (s *Store) langFromPath(path string) string {
parent := filepath.Dir(absPath(path))
if parent == s.ContentDir {
return ""
}
return filepath.Base(parent)
}
// defaultNameFor returns the path of a post relative to the content
// directory: the language subdirectory when the post has one, then its
// slug.
func (s *Store) defaultNameFor(p *post.Post) (string, error) {
name := p.Slug() + ".md"
if p.Lang() != "" && p.Lang() != s.defaultLang {
name = path.Join(p.Lang(), name)
}
if !within(s.ContentDir, filepath.Join(s.ContentDir, name)) {
return "", fmt.Errorf("slug escapes content directory")
}
return name, nil
}
// defaultPathFor returns the absolute path a post is written to.
func (s *Store) defaultPathFor(p *post.Post) (string, error) {
name, err := s.defaultNameFor(p)
if err != nil {
return "", err
}
return filepath.Join(s.ContentDir, name), nil
}
func (s *Store) targetLock(target string) func() {
s.locksGuard.Lock()
entry, ok := s.saveLocks[target]
if !ok {
entry = &lockEntry{}
s.saveLocks[target] = entry
}
entry.refs++
s.locksGuard.Unlock()
entry.mu.Lock()
return func() {
entry.mu.Unlock()
s.locksGuard.Lock()
entry.refs--
if entry.refs == 0 {
delete(s.saveLocks, target)
}
s.locksGuard.Unlock()
}
}