// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0 // Package store reads and writes Markdown posts and media in a content // directory: an mtime-snapshot cache, atomic writes, per-target locks, // revision archives, and rename-based soft deletes. package store import ( "errors" "fmt" "io/fs" "log/slog" "os" "path" "path/filepath" "regexp" "slices" "strings" "sync" "sourcedock.dev/petrbalvin/volumen/internal/identifiers" "sourcedock.dev/petrbalvin/volumen/internal/post" ) const revisionsDirname = ".revisions" // maxPostFileBytes bounds one content file the cache will read. A saved // post cannot come near it (the body is capped at 1 MiB before it is // ever written); the bound exists so an abandoned file dropped into the // content directory cannot be pulled into memory whole on every scan. const maxPostFileBytes = 8 << 20 var safeSlugRe = regexp.MustCompile(`[^a-zA-Z0-9._-]`) // MediaDirName is the uploads directory inside the content directory. const MediaDirName = "media" // Store manages posts on disk in a flat or language-subdivided // directory. type Store struct { ContentDir string defaultLang string followSymlinks bool revisionLimit int mu sync.Mutex cached []*post.Post byPath map[string]*post.Post snapshot []snapshotEntry unreadable []Unreadable haveCache bool slugIndex map[string][]string aliasIndex map[string]string locksGuard sync.Mutex saveLocks map[string]*lockEntry // rootOnce guards root, a handle confined to the content directory. // Every path the store touches outside the post walk is opened // through it, so a name that would escape the tree through ".." or a // symlink is refused by the kernel rather than checked for in Go. rootOnce sync.Once root *os.Root rootErr error } // openRoot returns the handle confined to the content directory, opening // it once per store. func (s *Store) openRoot() (*os.Root, error) { s.rootOnce.Do(func() { if err := os.MkdirAll(s.ContentDir, 0o755); err != nil { s.rootErr = fmt.Errorf("create the content directory: %w", err) return } s.root, s.rootErr = os.OpenRoot(s.ContentDir) }) return s.root, s.rootErr } // readDirIn lists a directory through the root, and returns nil when it // does not exist. func readDirIn(root *os.Root, dir string) []os.DirEntry { handle, err := root.Open(dir) if err != nil { return nil } defer handle.Close() entries, err := handle.ReadDir(-1) if err != nil { return nil } return entries } // statIn reports one entry of a directory listed through the root. func statIn(root *os.Root, dir, name string) (os.FileInfo, bool) { info, err := root.Stat(path.Join(dir, name)) if err != nil { return nil, false } return info, true } // lockEntry is a per-target write lock plus the number of callers // holding or waiting for it, so an entry can be dropped once it is idle. type lockEntry struct { mu sync.Mutex refs int } type snapshotEntry struct { path string mtime int64 size int64 } // Options configure a Store. type Options struct { // ContentDir is the directory holding the posts, and the media and // revision directories inside it. ContentDir string // DefaultLang is the language of a post that has none and that does // not sit in a language subdirectory. DefaultLang string // RevisionLimit is how many previous versions of a post to keep; zero // or less disables archiving, which also makes a delete permanent. RevisionLimit int // FollowSymlinks reads a symlinked post file instead of skipping it. FollowSymlinks bool } // New opens a store over Options.ContentDir. The path is resolved through // any symlinks, so a content directory reached by a symlinked path walks // and compares consistently. func New(opts Options) *Store { abs, err := filepath.Abs(opts.ContentDir) if err != nil { abs = opts.ContentDir } if resolved, err := filepath.EvalSymlinks(abs); err == nil { abs = resolved } return &Store{ ContentDir: abs, defaultLang: opts.DefaultLang, followSymlinks: opts.FollowSymlinks, revisionLimit: opts.RevisionLimit, byPath: map[string]*post.Post{}, slugIndex: map[string][]string{}, aliasIndex: map[string]string{}, saveLocks: map[string]*lockEntry{}, } } // CleanupStaleTombstones lives in revision.go: a stale tombstone is a // hazard only once the server serves the tree, so the read-only commands // do not trigger it. // InvalidateCache drops the cached posts so the next read re-scans the // disk. func (s *Store) InvalidateCache() { s.mu.Lock() defer s.mu.Unlock() s.cached = nil s.byPath = map[string]*post.Post{} s.unreadable = nil s.snapshot = nil s.slugIndex = map[string][]string{} s.aliasIndex = map[string]string{} s.haveCache = false } // All returns every loadable post, using the mtime snapshot cache. func (s *Store) All() []*post.Post { s.mu.Lock() defer s.mu.Unlock() s.refresh() return slices.Clone(s.cached) } // refresh rebuilds the cache when the (path, mtime, size) snapshot has // changed, and does nothing on a hit. The caller must hold s.mu. func (s *Store) refresh() { snapshot := s.buildSnapshot() if s.haveCache && slices.Equal(snapshot, s.snapshot) { return } s.snapshot = snapshot posts := make([]*post.Post, 0, len(snapshot)) byPath := make(map[string]*post.Post, len(snapshot)) var unreadable []Unreadable for _, entry := range snapshot { p, err := s.loadPost(entry.path) if err != nil { slog.Warn("store: skipping post", "path", entry.path, "error", err) unreadable = append(unreadable, Unreadable{Path: entry.path, Error: err.Error()}) continue } posts = append(posts, p) byPath[entry.path] = p } s.cached = posts s.byPath = byPath s.unreadable = unreadable s.buildIndex() s.buildLinkIndex() s.haveCache = true } // buildLinkIndex maps every valid frontmatter DOI to the slug of the // post published under it, and hands the map to the freshly loaded // posts, so a reference citing one of those DOIs links to its post // inside the instance instead of leaving for the resolver. The first // post published under a DOI wins; the map is shared read-only. func (s *Store) buildLinkIndex() { index := map[string]string{} for _, p := range s.cached { doi := identifiers.NormalizeDOI(p.DOI()) if !identifiers.ValidDOI(doi) || p.Slug() == "" { continue } key := strings.ToLower(doi) if _, seen := index[key]; !seen { index[key] = p.Slug() } } for _, p := range s.cached { p.SetLinkIndex(index) } } // Unreadable is a content file the store could not load. type Unreadable struct { Path string Error string } // Unreadable lists the content files the current scan could not read or // parse. Every read path skips them, so a diagnostic has to ask for // them explicitly. func (s *Store) Unreadable() []Unreadable { s.mu.Lock() defer s.mu.Unlock() s.refresh() return slices.Clone(s.unreadable) } // Find returns the post with the given slug, optionally filtered by // language (posts flagged all_langs match any language). The returned // post is shared with the cache: clone it before mutating. func (s *Store) Find(slug, lang string) *post.Post { s.mu.Lock() defer s.mu.Unlock() s.refresh() for _, path := range s.slugIndex[slug] { if p := s.byPath[path]; p != nil && p.Slug() == slug && languageMatches(p, lang) { return p } } for _, p := range s.cached { if p.Slug() == slug && languageMatches(p, lang) { return p } } return nil } func languageMatches(p *post.Post, lang string) bool { return lang == "" || p.Lang() == lang || p.AllLangs() } // ResolveAlias returns the canonical slug for an alias, or "". func (s *Store) ResolveAlias(alias string) string { s.mu.Lock() defer s.mu.Unlock() s.refresh() return s.aliasIndex[alias] } // CacheKey returns a string that changes whenever the content directory // changes, for a caller that memoises rendering of the whole post set. // It is derived from the same (path, mtime, size) snapshot the read // cache uses, so an edit that leaves the slug and date alone still // changes it. func (s *Store) CacheKey() string { s.mu.Lock() defer s.mu.Unlock() s.refresh() var b strings.Builder for _, entry := range s.snapshot { fmt.Fprintf(&b, "%s|%d|%d;", entry.path, entry.mtime, entry.size) } return b.String() } // Save writes the post atomically, archiving the previous version as a // revision first. It returns the post with Path set, and it sets Path on // the argument it is given, so a caller passing a post obtained from the // cache must clone it first. func (s *Store) Save(p *post.Post) (*post.Post, error) { target := p.Path if target == "" { var err error target, err = s.defaultPathFor(p) if err != nil { return nil, err } } root, err := s.openRoot() if err != nil { return nil, err } name, err := filepath.Rel(s.ContentDir, target) if err != nil { return nil, fmt.Errorf("locate %s: %w", target, err) } if err := root.MkdirAll(path.Dir(name), 0o755); err != nil { return nil, fmt.Errorf("create post directory: %w", err) } unlock := s.targetLock(target) defer unlock() slug := p.Slug() if slug == "" { slug = strings.TrimSuffix(filepath.Base(target), ".md") } s.archiveRevision(name, slug) content, err := p.ToFile() if err != nil { return nil, err } if err := atomicWriteIn(root, name, []byte(content)); err != nil { return nil, err } p.Path = target s.InvalidateCache() return p, nil } // Delete removes the post. With revisions enabled the file is moved into // the revision archive as a tombstone and undoable is true; with a // non-positive revision limit it is removed outright and undoable is // false. A nil post with a nil error means no such post. func (s *Store) Delete(slug, lang string) (p *post.Post, undoable bool, err error) { p = s.Find(slug, lang) if p == nil || p.Path == "" { return p, false, nil } unlock := s.targetLock(p.Path) defer unlock() if err := s.writeTombstone(p); err != nil { // The file vanished between Find and the lock: the delete the // caller asked for has already happened, so it is "no such post" // rather than a failure. if errors.Is(err, fs.ErrNotExist) { return nil, false, nil } return nil, false, err } s.InvalidateCache() return p, s.revisionLimit > 0, nil } // mdFiles lists the .md files in the content directory, relative to it. func (s *Store) mdFiles() []string { var results []string root := s.ContentDir _ = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error { if err != nil { // A walk that cannot read a subtree still lists the rest, but // an unreadable root must not pass silently as an empty site. slog.Error("store: cannot walk the content directory", "path", path, "error", err) return nil } if d.IsDir() { name := d.Name() if path != root && (name == MediaDirName || name == revisionsDirname) { return filepath.SkipDir } return nil } if !strings.HasSuffix(d.Name(), ".md") { return nil } if !s.followSymlinks { if info, err := d.Info(); err == nil && info.Mode()&os.ModeSymlink != 0 { return nil } } results = append(results, path) return nil }) return results } func (s *Store) buildSnapshot() []snapshotEntry { files := s.mdFiles() rows := make([]snapshotEntry, 0, len(files)) for _, f := range files { info, err := os.Stat(f) if err != nil { continue } rows = append(rows, snapshotEntry{path: f, mtime: info.ModTime().UnixNano(), size: info.Size()}) } slices.SortFunc(rows, func(a, b snapshotEntry) int { return strings.Compare(a.path, b.path) }) return rows } func (s *Store) buildIndex() { s.slugIndex = map[string][]string{} s.aliasIndex = map[string]string{} for _, p := range s.cached { if p.Path == "" || p.Slug() == "" { continue } s.slugIndex[p.Slug()] = append(s.slugIndex[p.Slug()], p.Path) for _, alias := range p.Aliases() { s.aliasIndex[alias] = p.Slug() } } } func (s *Store) loadPost(path string) (*post.Post, error) { if info, err := os.Stat(path); err == nil && info.Size() > maxPostFileBytes { return nil, fmt.Errorf("file exceeds %d bytes and is not a post this engine could render", maxPostFileBytes) } content, err := os.ReadFile(path) if err != nil { return nil, err } parsed, err := post.Parse(string(content)) if err != nil { return nil, err } p := parsed // Both values are derived here and held outside the metadata, so a // later save writes the file back as it was read. slug := p.Slug() if slug == "" { slug = strings.TrimSuffix(filepath.Base(path), filepath.Ext(path)) } lang := p.Lang() if lang == "" { lang = s.langFromPath(path) if lang == "" { lang = s.defaultLang } } p.SetFileLocation(slug, lang) p.Path = path return p, nil } func (s *Store) langFromPath(path string) string { parent := filepath.Dir(absPath(path)) if parent == s.ContentDir { return "" } return filepath.Base(parent) } // defaultNameFor returns the path of a post relative to the content // directory: the language subdirectory when the post has one, then its // slug. func (s *Store) defaultNameFor(p *post.Post) (string, error) { name := p.Slug() + ".md" if p.Lang() != "" && p.Lang() != s.defaultLang { name = path.Join(p.Lang(), name) } if !within(s.ContentDir, filepath.Join(s.ContentDir, name)) { return "", fmt.Errorf("slug escapes content directory") } return name, nil } // defaultPathFor returns the absolute path a post is written to. func (s *Store) defaultPathFor(p *post.Post) (string, error) { name, err := s.defaultNameFor(p) if err != nil { return "", err } return filepath.Join(s.ContentDir, name), nil } func (s *Store) targetLock(target string) func() { s.locksGuard.Lock() entry, ok := s.saveLocks[target] if !ok { entry = &lockEntry{} s.saveLocks[target] = entry } entry.refs++ s.locksGuard.Unlock() entry.mu.Lock() return func() { entry.mu.Unlock() s.locksGuard.Lock() entry.refs-- if entry.refs == 0 { delete(s.saveLocks, target) } s.locksGuard.Unlock() } }