// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package nfsfs import ( "errors" "fmt" "io/fs" "os" "syscall" ) // fdCacheLimit bounds the descriptor cache. The cache holds at most this // many idle descriptors across both access classes; descriptors checked // out by running operations sit above the bound for their lifetime. The // bound keeps a busy server under the process file descriptor ceiling: // without it, one descriptor per file ever touched would grow without end. const fdCacheLimit = 512 // An fdKey identifies one cached descriptor: the path it was opened // through and the access class. The path is part of the key, not just the // device and inode, so a file removed and recreated under a recycled inode // number never inherits the old descriptor: the new object resolves to a // fresh open, and the retired one fails its next identity check. type fdKey struct { path string wr bool } // An fdEntry is one cached descriptor. refs counts the operations holding // it right now; the entry leaves the cache, and its descriptor closes, // when it is retired or evicted and the last reference lets go. type fdEntry struct { f *os.File refs int dead bool use uint64 } // dataFD hands out an open descriptor for the regular file the handle // names: from the cache when one is held, from a fresh verified open // otherwise. The identity is reverified on every use, a cache hit // included, twice over: the registered path must still Lstat to the // device, inode and kind the handle encodes, and the descriptor itself // must still carry that identity with a link count above zero. A file // removed while a descriptor of it sits in the cache therefore answers // stale exactly as it does without the cache, and a name swapped for a // symlink is never served through the cached descriptor. // // The returned release function must be called: it returns the descriptor // to the cache, or closes it when the descriptor was retired, evicted or // excluded from the cache while the lease was out. func (l *Local) dataFD(h Handle, wr bool) (*os.File, func(), error) { kind, dev, ino, path, err := l.resolve(h) if err != nil { return nil, nil, err } if kind != typeFile { return nil, nil, ErrIsDir } key := fdKey{path: path, wr: wr} if cached := l.fdCheckout(key); cached != nil { // The descriptor is reverified against the path and against its // own stat: a link count of zero means the cached descriptor is // holding an unlinked inode, whatever the path names now. fi, err := os.Lstat(path) fst, ferr := cached.Stat() switch { case err == nil && ferr == nil && sameFile(fi, kind, dev, ino) && sameFile(fst, kind, dev, ino) && nlinkOf(fst) > 0: return cached, func() { l.fdRelease(key, cached) }, nil case err == nil: // The path no longer names the inode, or the descriptor // serves an unlinked one: retire and answer stale. l.fdRetire(key) return nil, nil, ErrStale default: l.fdRetire(key) return nil, nil, revalidateStatErr(err) } } flag := os.O_RDONLY if wr { flag = os.O_WRONLY } f, _, err := l.openVerified(h, flag) if err != nil { if em := l.fdRelieve(err); em { // The open starved on descriptors; the cache gave its idle // ones up. One retry is entitled to succeed now. f, _, err = l.openVerified(h, flag) } if err != nil { return nil, nil, err } } l.fdInsert(key, f) return f, func() { l.fdRelease(key, f) }, nil } // nlinkOf reports the link count a stat carried, zero when the platform // data is missing. func nlinkOf(fi os.FileInfo) uint64 { if st, ok := fi.Sys().(*syscall.Stat_t); ok { return uint64(st.Nlink) } return 0 } // revalidateStatErr maps the errors of the revalidating Lstat, the same // mapping revalidate applies. func revalidateStatErr(err error) error { switch { case errors.Is(err, fs.ErrNotExist), errors.Is(err, syscall.ENOTDIR): return ErrStale case errors.Is(err, fs.ErrPermission): return ErrPermission default: return fmt.Errorf("%w: %v", ErrIO, err) } } // fdCheckout hands the cached descriptor of key out to one operation and // marks the entry busy, or reports nil when nothing usable is cached. func (l *Local) fdCheckout(key fdKey) *os.File { l.fdMu.Lock() defer l.fdMu.Unlock() e := l.fds[key] if e == nil || e.dead { return nil } e.refs++ l.fdUse++ e.use = l.fdUse return e.f } // fdInsert admits a freshly opened, already verified descriptor into the // cache with one reference held. When a retired entry under the same key // is still draining its outstanding leases, the new descriptor bypasses // the cache and closes on release instead. func (l *Local) fdInsert(key fdKey, f *os.File) { l.fdMu.Lock() if old := l.fds[key]; old != nil { // Only a drained entry leaves the map, so anything here is a // retired one waiting for its leases; this descriptor stays out. l.fdMu.Unlock() return } l.fds[key] = &fdEntry{f: f, refs: 1, use: l.fdUse + 1} l.fdUse++ closed := l.fdEvictLocked() l.fdMu.Unlock() for _, idle := range closed { idle.Close() } } // fdRelease ends one lease. A live entry takes the descriptor back; a // retired, evicted or bypassed one closes it, at the last release. func (l *Local) fdRelease(key fdKey, f *os.File) { l.fdMu.Lock() e := l.fds[key] if e == nil || e.f != f { l.fdMu.Unlock() f.Close() return } e.refs-- if e.refs > 0 { l.fdMu.Unlock() return } if e.dead { delete(l.fds, key) l.fdMu.Unlock() f.Close() return } l.fdMu.Unlock() } // fdRetire marks the cached descriptor of key dead: it is never handed // out again, and it closes when its outstanding leases release. func (l *Local) fdRetire(key fdKey) { l.fdMu.Lock() if e := l.fds[key]; e != nil { e.dead = true } l.fdMu.Unlock() } // fdEvictLocked picks idle descriptors until the cache fits the bound and // returns them for the caller to close outside the lock. Entries with // outstanding leases are untouchable; a cache full of busy entries // temporarily exceeds the bound by exactly the number of running // operations. func (l *Local) fdEvictLocked() []*os.File { var closed []*os.File for len(l.fds) > fdCacheLimit { var victim *fdEntry var victimKey fdKey for key, e := range l.fds { if e.dead || e.refs > 0 { continue } if victim == nil || e.use < victim.use { victim, victimKey = e, key } } if victim == nil { return closed } delete(l.fds, victimKey) closed = append(closed, victim.f) } return closed } // fdRelieve answers whether err is an open refused for descriptor // exhaustion and, when it is, retires every idle descriptor so one retry // can run. The cache must never be the reason a server runs out of file // descriptors. func (l *Local) fdRelieve(err error) bool { if !errors.Is(err, syscall.EMFILE) && !errors.Is(err, syscall.ENFILE) { return false } l.fdMu.Lock() var closed []*os.File for key, e := range l.fds { if e.refs > 0 { e.dead = true continue } delete(l.fds, key) closed = append(closed, e.f) } l.fdMu.Unlock() for _, f := range closed { f.Close() } return true }