// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package nfsfs import ( "encoding/binary" "encoding/json" "errors" "fmt" "io" "io/fs" "net" "os" "path/filepath" "slices" "strconv" "strings" "sync" "syscall" "time" ) // A Local serves one local directory tree over dev and ino based file // handles. A handle encodes the device and inode number; the mapping from // that pair to a path is held in memory and persisted on demand, so a // handle from before a restart resolves when the mapping is loaded back. // Every use revalidates the mapping: the path must still name the device, // inode and kind the handle encodes, and no resolution follows a final // symlink, so a name swapped for a link is stale rather than an escape. type Local struct { root string mu sync.RWMutex paths map[fileID]string persistPath string // The descriptor cache of fdcache.go: idle descriptors of regular // files, bounded by fdCacheLimit, every use reverified against the // registered path and the descriptor's own identity. fdMu sync.Mutex fds map[fdKey]*fdEntry fdUse uint64 // The listing cache of ReadDir: the sorted names of the directories // being paged, bounded by dirCacheMax, valid while the directory's // modification time matches. dirMu sync.Mutex dirs map[fileID]*cachedDir dirUse uint64 } // A fileID identifies one inode on one device: the key of the handle to // path mapping. type fileID struct { dev uint64 ino uint64 } // handle layout: magic byte, version byte, type byte, dev, ino. const ( handleMagic = 0x4e // handleVersion names the handle layout. Version two keys the mapping // by device and inode and revalidates on use; handles of version one // carry no device to check against and answer stale. handleVersion = 2 handleSize = 3 + 8 + 8 ) // handle type bytes, mirroring the file kinds the protocol distinguishes. const ( typeDir = 1 typeFile = 2 typeOther = 3 ) // NewLocal returns a Local serving root. The path must be an existing // directory. func NewLocal(root string) (*Local, error) { abs, err := filepath.Abs(root) if err != nil { return nil, fmt.Errorf("%w: %v", ErrIO, err) } st, err := os.Lstat(abs) if err != nil { return nil, fmt.Errorf("%w: %v", ErrNoEnt, err) } if !st.IsDir() { return nil, fmt.Errorf("%w: %s is not a directory", ErrNotDir, abs) } l := &Local{root: abs, paths: make(map[fileID]string), fds: make(map[fdKey]*fdEntry), dirs: make(map[fileID]*cachedDir)} if _, _, err := l.link(abs); err != nil { return nil, err } return l, nil } // stat converts an os.FileInfo plus its raw stat into an Info. func stat(fi os.FileInfo) Info { info := Info{ Size: fi.Size(), Mode: fi.Mode(), ModTime: fi.ModTime(), Nlink: 1, } if st, ok := fi.Sys().(*syscall.Stat_t); ok { info.Dev = uint64(st.Dev) info.Ino = uint64(st.Ino) info.Nlink = uint64(st.Nlink) info.UID = st.Uid info.GID = st.Gid } return info } // kindOf maps a file mode onto the handle type byte. func kindOf(mode fs.FileMode) byte { switch { case mode.IsDir(): return typeDir case mode.IsRegular(): return typeFile default: return typeOther } } // sameFile reports whether fi names the device, inode and kind a handle // encodes. func sameFile(fi os.FileInfo, kind byte, dev, ino uint64) bool { st, ok := fi.Sys().(*syscall.Stat_t) if !ok { return false } return uint64(st.Dev) == dev && uint64(st.Ino) == ino && kindOf(fi.Mode()) == kind } // SetPersistPath aims the handle mapping persistence at a file inside // dir. Every registered handle is saved through it, and the mapping is // written once right away. func (l *Local) SetPersistPath(dir string) { l.mu.Lock() l.persistPath = filepath.Join(dir, "handles.json") l.mu.Unlock() l.save() } // persistTarget answers the persistence file path, read under the lock. func (l *Local) persistTarget() string { l.mu.RLock() defer l.mu.RUnlock() return l.persistPath } // save writes the mapping file when persistence is armed. The snapshot is // taken under the read lock; the writing runs outside it. func (l *Local) save() { target := l.persistTarget() if target == "" { return } l.mu.RLock() out := make(map[string]string, len(l.paths)) for id, p := range l.paths { out[persistKey(id)] = p } l.mu.RUnlock() data, err := json.Marshal(out) if err != nil { return } tmp := target + ".tmp" if err := os.WriteFile(tmp, data, 0o600); err != nil { return } _ = os.Rename(tmp, target) } // persistKey renders the map key of one registered pair, the device and // inode numbers in hex joined by a colon. func persistKey(id fileID) string { return strconv.FormatUint(id.dev, 16) + ":" + strconv.FormatUint(id.ino, 16) } // link records the path under its device and inode number and returns its // handle and attributes. func (l *Local) link(path string) (Handle, Info, error) { fi, err := os.Lstat(path) if err != nil { if errors.Is(err, syscall.ENOTDIR) { return nil, Info{}, ErrNotDir } if errors.Is(err, fs.ErrPermission) { return nil, Info{}, ErrPermission } return nil, Info{}, fmt.Errorf("%w: %v", ErrNoEnt, err) } info := stat(fi) if info.Ino == 0 { return nil, Info{}, fmt.Errorf("%w: %s has no inode number", ErrIO, path) } dirty := false l.mu.Lock() id := fileID{dev: info.Dev, ino: info.Ino} if prev, ok := l.paths[id]; !ok || prev != path { l.paths[id] = path dirty = true } l.mu.Unlock() // The mapping is the recovery state of the handles: it is written the // moment it changes, so a restart never loses a handle it served. An // unchanged registration writes nothing, which keeps a large READDIR // from rewriting the same file once per entry. if dirty && l.persistTarget() != "" { l.save() } return encodeHandle(info, path), info, nil } // encodeHandle builds the opaque handle for an already linked path. func encodeHandle(info Info, path string) Handle { h := make(Handle, handleSize) h[0] = handleMagic h[1] = handleVersion h[2] = kindOf(info.Mode) binary.BigEndian.PutUint64(h[3:11], info.Dev) binary.BigEndian.PutUint64(h[11:19], info.Ino) _ = path return h } // decode parses a handle and returns its kind byte, device and inode. A // handle of another magic, length or version is stale. func decode(h Handle) (byte, uint64, uint64, error) { if len(h) != handleSize || h[0] != handleMagic || h[1] != handleVersion { return 0, 0, 0, ErrStale } return h[2], binary.BigEndian.Uint64(h[3:11]), binary.BigEndian.Uint64(h[11:19]), nil } // resolve decodes a handle and reports its kind byte and the registered // path of the device and inode it names. A handle the process never // issued, or one from before a restart, resolves to ErrStale. func (l *Local) resolve(h Handle) (byte, uint64, uint64, string, error) { kind, dev, ino, err := decode(h) if err != nil { return 0, 0, 0, "", err } l.mu.RLock() path, ok := l.paths[fileID{dev: dev, ino: ino}] l.mu.RUnlock() if !ok { return 0, 0, 0, "", ErrStale } return kind, dev, ino, path, nil } // revalidate resolves a handle and confirms through Lstat that the // registered path still names its device, inode and kind. Lstat never // follows a final symlink, so a name swapped for a link is stale. func (l *Local) revalidate(h Handle) (byte, string, os.FileInfo, error) { kind, dev, ino, path, err := l.resolve(h) if err != nil { return 0, "", nil, err } fi, err := os.Lstat(path) if err != nil { return 0, "", nil, revalidateStatErr(err) } if !sameFile(fi, kind, dev, ino) { return 0, "", nil, ErrStale } return kind, path, fi, nil } // dirOf resolves a handle that must name a directory, revalidated against // the registered path. func (l *Local) dirOf(h Handle) (string, error) { kind, path, _, err := l.revalidate(h) if err != nil { return "", err } if kind != typeDir { return "", ErrNotDir } return path, nil } // openVerified resolves a handle to an open descriptor, never following a // final symlink, and requires the descriptor to name the device, inode // and kind the handle encodes. Anything else is stale. func (l *Local) openVerified(h Handle, flag int) (*os.File, os.FileInfo, error) { kind, dev, ino, path, err := l.resolve(h) if err != nil { return nil, nil, err } f, err := os.OpenFile(path, flag|syscall.O_NOFOLLOW, 0) if err != nil { switch { case errors.Is(err, fs.ErrNotExist), errors.Is(err, syscall.ENOTDIR), symlinkRefused(err): return nil, nil, ErrStale case errors.Is(err, syscall.EISDIR): return nil, nil, ErrIsDir case errors.Is(err, fs.ErrPermission): return nil, nil, ErrPermission default: return nil, nil, fmt.Errorf("%w: %v", ErrIO, err) } } fi, err := f.Stat() if err != nil { f.Close() return nil, nil, fmt.Errorf("%w: %v", ErrIO, err) } if !sameFile(fi, kind, dev, ino) { f.Close() return nil, nil, ErrStale } return f, fi, nil } // Root returns the handle of the export root. func (l *Local) Root() (Handle, error) { h, _, err := l.link(l.root) return h, err } // Lookup resolves name under the parent handle. func (l *Local) Lookup(parent Handle, name string) (Handle, Info, error) { if err := ValidName(name); err != nil { return nil, Info{}, err } parentPath, err := l.dirOf(parent) if err != nil { return nil, Info{}, err } child := filepath.Join(parentPath, name) h, info, err := l.link(child) if err != nil { return nil, Info{}, err } return h, info, nil } // Getattr reports the attributes of a handle. func (l *Local) Getattr(h Handle) (Info, error) { _, _, fi, err := l.revalidate(h) if err != nil { return Info{}, err } return stat(fi), nil } // dirCacheMax bounds the directory listing cache: the sorted names of // the directories a client pages through, held until their modification // time moves. One bounded map, entries evicted least recently used. const dirCacheMax = 64 // A cachedDir is the sorted name order of one directory, keyed by the // directory's file identity, valid while its verifier matches. type cachedDir struct { names []string verifier [8]byte use uint64 } // ReadDir lists the directory from the given cookie. The cookie is the // one-based position in the sorted name order, and the verifier is the // directory's modification time, so a listing that raced a change is // detected by the caller. // // The sorted order is cached per directory and revalidated against the // verifier on every page: paging a large directory costs the page alone, // not a full re listing and re sort, and any change to the directory // moves the verifier and forces a fresh listing. func (l *Local) ReadDir(h Handle, cookie uint64, count int) (DirPage, error) { path, err := l.dirOf(h) if err != nil { return DirPage{}, err } root, err2 := filepath.Abs(path) if err2 != nil { return DirPage{}, fmt.Errorf("%w: %v", ErrIO, err2) } info, err2 := os.Lstat(root) if err2 != nil { return DirPage{}, fmt.Errorf("%w: %v", ErrIO, err2) } var verifier [8]byte binary.BigEndian.PutUint64(verifier[:], uint64(info.ModTime().UnixNano())) var id fileID haveID := false if st, ok := info.Sys().(*syscall.Stat_t); ok { id = fileID{dev: uint64(st.Dev), ino: uint64(st.Ino)} haveID = true } names, err2 := l.cachedNames(id, haveID, verifier, func() ([]string, error) { entries, err := os.ReadDir(path) if err != nil { if errors.Is(err, syscall.ENOTDIR) { return nil, ErrNotDir } return nil, fmt.Errorf("%w: %v", ErrIO, err) } sortNames(entries) names := make([]string, len(entries)) for i, e := range entries { names[i] = e.Name() } return names, nil }) if err2 != nil { return DirPage{}, err2 } var page DirPage page.Verifier = verifier for i, name := range names { c := uint64(i) + 1 if c <= cookie { continue } if count > 0 && len(page.Entries) >= count { return page, nil } child := filepath.Join(root, name) h, info, err := l.link(child) if err != nil { // A file removed between ReadDir and Lstat is skipped, not an // error for the whole listing. continue } page.Entries = append(page.Entries, Entry{Cookie: c, Name: name, Handle: h, Info: info}) } page.EOF = true return page, nil } // cachedNames answers the sorted names of a directory: from the cache // while the verifier matches, from the fill function otherwise. A // directory without a raw stat bypasses the cache, since its identity // would collide with the next one. func (l *Local) cachedNames(id fileID, haveID bool, verifier [8]byte, fill func() ([]string, error)) ([]string, error) { if haveID { l.dirMu.Lock() if cd := l.dirs[id]; cd != nil && cd.verifier == verifier { l.dirUse++ cd.use = l.dirUse l.dirMu.Unlock() return cd.names, nil } l.dirMu.Unlock() } names, err := fill() if err != nil { return nil, err } if haveID { l.dirMu.Lock() l.dirUse++ l.dirs[id] = &cachedDir{names: names, verifier: verifier, use: l.dirUse} for len(l.dirs) > dirCacheMax { var victim fileID var oldest uint64 first := true for key, cd := range l.dirs { if first || cd.use < oldest { victim, oldest, first = key, cd.use, false } } delete(l.dirs, victim) } l.dirMu.Unlock() } return names, nil } // sortNames orders a directory listing by name, the order the cookies are // defined against. func sortNames(entries []os.DirEntry) { slices.SortFunc(entries, func(a, b os.DirEntry) int { return strings.Compare(a.Name(), b.Name()) }) } // Read reads up to count bytes at the offset from a regular file. The // descriptor comes from the cache or a fresh verified open, and the // identity is revalidated before anything is read. func (l *Local) Read(h Handle, off int64, count int) ([]byte, error) { f, release, err := l.dataFD(h, false) if err != nil { return nil, err } defer release() buf := make([]byte, count) n, err := f.ReadAt(buf, off) if err != nil && !errors.Is(err, io.EOF) { return nil, fmt.Errorf("%w: %v", ErrIO, err) } return buf[:n], nil } // Access evaluates the requested mask bits for the credential, using the // classic owner, group and other selection over the permission bits. The // superuser is granted everything. func (l *Local) Access(h Handle, mask uint32, uid, gid uint32, groups []uint32) (uint32, error) { if uid == 0 { return mask, nil } info, err := l.Getattr(h) if err != nil { return 0, err } var mode fs.FileMode switch { case uid == info.UID: mode = info.Mode.Perm() >> 6 case gid == info.GID || containsGID(groups, info.GID): mode = info.Mode.Perm() >> 3 default: mode = info.Mode.Perm() } var granted uint32 for bit, want := range map[uint32]fs.FileMode{ AccessRead: 0o4, AccessLookup: 0o1, AccessModify: 0o2, AccessExtend: 0o2, AccessDelete: 0o2, AccessExec: 0o1, } { if mask&bit != 0 && mode&want != 0 { granted |= bit } } return granted, nil } func containsGID(groups []uint32, gid uint32) bool { return slices.Contains(groups, gid) } // modeBits renders a mode the way the raw create and chmod calls receive // it: the low nine permission bits plus the setuid, setgid and sticky // bits, wherever the mode carries them. func modeBits(m fs.FileMode) uint32 { bits := uint32(m & (os.ModePerm | 0o7000)) if m&os.ModeSetuid != 0 { bits |= 0o4000 } if m&os.ModeSetgid != 0 { bits |= 0o2000 } if m&os.ModeSticky != 0 { bits |= 0o1000 } return bits } // fileMode renders twelve raw mode bits as the FileMode the chmod family // receives. func fileMode(bits uint32) fs.FileMode { m := fs.FileMode(bits & 0o777) if bits&0o4000 != 0 { m |= os.ModeSetuid } if bits&0o2000 != 0 { m |= os.ModeSetgid } if bits&0o1000 != 0 { m |= os.ModeSticky } return m } // Create makes the object the spec describes under the parent handle. An // existing target is an error for every kind; the permission bits are // applied exactly, all twelve of them, with a chmod after the creation, so // the daemon's umask never distorts what the client asked for. func (l *Local) Create(parent Handle, name string, spec CreateSpec) (Handle, Info, error) { if err := ValidName(name); err != nil { return nil, Info{}, err } parentPath, err := l.dirOf(parent) if err != nil { return nil, Info{}, err } path := filepath.Join(parentPath, name) if _, err := os.Lstat(path); err == nil { return nil, Info{}, ErrExist } else if !errors.Is(err, fs.ErrNotExist) { return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err) } perm := modeBits(spec.Perm) made := false switch spec.Kind { case KindDir: if err := os.Mkdir(path, fileMode(perm)); err != nil { return nil, Info{}, wrapCreateErr(err) } _ = os.Chmod(path, fileMode(perm)) made = true case KindLnk: if err := os.Symlink(spec.LinkData, path); err != nil { return nil, Info{}, wrapCreateErr(err) } case KindFifo: if err := syscall.Mkfifo(path, perm); err != nil { return nil, Info{}, wrapCreateErr(err) } _ = os.Chmod(path, fileMode(perm)) made = true case KindSock: if err := bindUnixSocket(path); err != nil { return nil, Info{}, wrapCreateErr(err) } _ = os.Chmod(path, fileMode(perm)) made = true case KindBlk, KindChr: // A device node needs CAP_MKNOD on Linux; without it the failure // is a permission problem and says so. if err := mknod(path, spec, perm); err != nil { return nil, Info{}, wrapCreateErr(err) } _ = os.Chmod(path, fileMode(perm)) made = true default: return nil, Info{}, ErrBadName } if made { // The daemon hands every object it makes over to the owner the // client named; a symlink carries no access check of its own, so // it alone keeps the daemon's identity. _ = os.Chown(path, int(spec.Owner.UID), int(spec.Owner.GID)) } h, info, err := l.link(path) if err != nil { return nil, Info{}, err } return h, info, nil } // Write writes all of data at the offset of a regular file. The // descriptor comes from the cache or a fresh verified open, and the // identity is revalidated before anything is written. func (l *Local) Write(h Handle, off int64, data []byte) (int, error) { f, release, err := l.dataFD(h, true) if err != nil { return 0, err } defer release() n, err := f.WriteAt(data, off) if err != nil && !errors.Is(err, io.EOF) { if errors.Is(err, syscall.ENOSPC) { return int(n), ErrNoSpace } return int(n), fmt.Errorf("%w: %v", ErrIO, err) } return n, nil } // wrapCreateErr maps the errors of the create system calls. func wrapCreateErr(err error) error { switch { case err == nil: return nil case errors.Is(err, fs.ErrExist): return ErrExist case errors.Is(err, fs.ErrNotExist): // A create whose parent directory is missing: the client learns // the name it walked to is gone, not that the disk failed. return ErrNoEnt case errors.Is(err, fs.ErrPermission): return ErrPermission case errors.Is(err, fs.ErrInvalid): return ErrBadName case errors.Is(err, syscall.EISDIR): return ErrIsDir case errors.Is(err, syscall.ENOTDIR): return ErrNotDir case errors.Is(err, syscall.ENOSPC): return ErrNoSpace default: return fmt.Errorf("%w: %v", ErrIO, err) } } // bindUnixSocket creates a unix domain socket file at path. The listener // is closed at once; the file it bound remains. func bindUnixSocket(path string) error { ln, err := net.Listen("unix", path) if err != nil { return wrapCreateErr(err) } if u, ok := ln.(*net.UnixListener); ok { u.SetUnlinkOnClose(false) } return ln.Close() } // Remove takes the named entry out of the directory. An empty directory is // removed like anything else; a directory that still holds entries is // ErrNotEmpty. A name that cannot be examined for permission reasons is a // permission error, not a missing one. func (l *Local) Remove(dir Handle, name string) error { if err := ValidName(name); err != nil { return err } dirPath, err := l.dirOf(dir) if err != nil { return err } path := filepath.Join(dirPath, name) fi, err := os.Lstat(path) if err != nil { switch { case errors.Is(err, fs.ErrNotExist): return ErrNoEnt case errors.Is(err, fs.ErrPermission), errors.Is(err, syscall.EPERM): return ErrPermission default: return fmt.Errorf("%w: %v", ErrIO, err) } } if err := os.Remove(path); err != nil { if errors.Is(err, syscall.ENOTEMPTY) { return ErrNotEmpty } if errors.Is(err, fs.ErrPermission) { return ErrPermission } return fmt.Errorf("%w: %v", ErrIO, err) } if id := stat(fi); id.Ino != 0 { l.mu.Lock() if l.paths[fileID{dev: id.Dev, ino: id.Ino}] == path { delete(l.paths, fileID{dev: id.Dev, ino: id.Ino}) } l.mu.Unlock() } return nil } // Rename moves oldName from oldDir to newName in newDir, replacing an // existing plain target the way POSIX rename does. The moved subtree is // re-registered under its new paths, so the handles of the object and of // its descendants keep resolving after the move. func (l *Local) Rename(oldDir Handle, oldName string, newDir Handle, newName string) error { if err := ValidName(oldName); err != nil { return err } if err := ValidName(newName); err != nil { return err } oldDirPath, err := l.dirOf(oldDir) if err != nil { return err } newDirPath, err := l.dirOf(newDir) if err != nil { return err } oldPath := filepath.Join(oldDirPath, oldName) newPath := filepath.Join(newDirPath, newName) if oldPath == newPath { return nil } if _, err := os.Lstat(oldPath); err != nil { if errors.Is(err, fs.ErrNotExist) { return ErrNoEnt } return fmt.Errorf("%w: %v", ErrIO, err) } if err := os.Rename(oldPath, newPath); err != nil { switch { case errors.Is(err, fs.ErrNotExist): return ErrNoEnt case errors.Is(err, syscall.ENOTEMPTY): return ErrNotEmpty case errors.Is(err, syscall.EINVAL): return ErrInval case errors.Is(err, fs.ErrPermission): return ErrPermission default: return fmt.Errorf("%w: %v", ErrIO, err) } } // Register the moved object, and when it is a directory, every // descendant under its new path, so the handles already issued by // earlier READDIRs and LOOKUPs keep working. h, info, err := l.link(newPath) if err != nil { return err } _ = h if info.IsDir() { // Every descendant moved too: re-register them under their new // paths, skipping entries that vanish while the walk runs. _ = filepath.WalkDir(newPath, func(p string, d fs.DirEntry, werr error) error { if werr != nil || p == newPath { return nil } _, _, _ = l.link(p) return nil }) } return nil } // Setattr applies the named changes to a file, in the order size, mode, // owner, times. The registered path is revalidated first, so a name // swapped for another inode, a symlink included, is stale before anything // is applied. A change the backend cannot apply fails the whole call. func (l *Local) Setattr(h Handle, s SetAttrs) error { _, path, fi, err := l.revalidate(h) if err != nil { return err } if s.Size != nil { if fi.IsDir() { return ErrIsDir } if serr := os.Truncate(path, *s.Size); serr != nil { return wrapWriteErr(serr) } } if s.Mode != nil { if serr := os.Chmod(path, fileMode(*s.Mode&0o7777)); serr != nil { return wrapWriteErr(serr) } } if s.UID != nil || s.GID != nil { uid, gid := -1, -1 if s.UID != nil { uid = int(*s.UID) } if s.GID != nil { gid = int(*s.GID) } if serr := os.Chown(path, uid, gid); serr != nil { return wrapWriteErr(serr) } } if s.Atime != nil || s.Mtime != nil { // Chtimes wants both times: whatever the client left out keeps the // value the file carries now. atime, mtime := time.Now(), time.Now() if s.Atime == nil || s.Mtime == nil { if fi, serr := os.Lstat(path); serr == nil { atime, mtime = fi.ModTime(), fi.ModTime() } } if s.Atime != nil { atime = resolveTime(*s.Atime) } if s.Mtime != nil { mtime = resolveTime(*s.Mtime) } if serr := os.Chtimes(path, atime, mtime); serr != nil { return wrapWriteErr(serr) } } return nil } // resolveTime turns a settime4 into the time it names. func resolveTime(t TimeSet) time.Time { if t.Now { return time.Now() } return t.Time } // wrapWriteErr maps the errors of the mutating system calls. func wrapWriteErr(err error) error { switch { case err == nil: return nil case errors.Is(err, fs.ErrPermission): return ErrPermission case errors.Is(err, fs.ErrNotExist): return ErrStale case errors.Is(err, syscall.ENOSPC): return ErrNoSpace case errors.Is(err, syscall.EINVAL): return ErrInval default: return fmt.Errorf("%w: %v", ErrIO, err) } } // Link makes newName in dir a hard link to the target file. Directories // are refused: the protocol reserves hard links for regular files and the // kernel refuses the rest. func (l *Local) Link(target Handle, dir Handle, name string) (Handle, Info, error) { if err := ValidName(name); err != nil { return nil, Info{}, err } targetKind, targetPath, _, err := l.revalidate(target) if err != nil { return nil, Info{}, err } if targetKind == typeDir { return nil, Info{}, ErrIsDir } dirPath, err := l.dirOf(dir) if err != nil { return nil, Info{}, err } newPath := filepath.Join(dirPath, name) if _, err := os.Lstat(newPath); err == nil { return nil, Info{}, ErrExist } else if !errors.Is(err, fs.ErrNotExist) { return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err) } if err := os.Link(targetPath, newPath); err != nil { if errors.Is(err, fs.ErrExist) { return nil, Info{}, ErrExist } if errors.Is(err, fs.ErrPermission) { return nil, Info{}, ErrPermission } if errors.Is(err, syscall.EPERM) { // The kernel refuses hard links to directories. return nil, Info{}, ErrIsDir } return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err) } return l.link(newPath) } // ReadLink reports the target of a symlink. Anything else is refused: // the protocol answers NFS4ERR_INVAL for a READLINK on a non link. func (l *Local) ReadLink(h Handle) (string, error) { kind, path, fi, err := l.revalidate(h) if err != nil { return "", err } if kind != typeOther || fi.Mode()&os.ModeSymlink == 0 { return "", ErrNotLnk } target, err := os.Readlink(path) if err != nil { return "", fmt.Errorf("%w: %v", ErrIO, err) } return target, nil } // Sync flushes the dirty data of a regular file, or of the directory // itself, to stable storage. The stateless backend writes synchronously, // so this is the belt to the braces of the FILE_SYNC answer. A regular // file syncs through the cached descriptor, a directory through a fresh // verified open; both revalidate the identity first. func (l *Local) Sync(h Handle) error { kind, _, _, _, err := l.resolve(h) if err != nil { return err } if kind != typeFile { f, _, err := l.openVerified(h, os.O_RDONLY) if err != nil { return err } defer f.Close() if err := f.Sync(); err != nil { return wrapWriteErr(err) } return nil } f, release, err := l.dataFD(h, true) if err != nil { return err } defer release() if err := f.Sync(); err != nil { return wrapWriteErr(err) } return nil } // Parent resolves the directory that holds h and the component name of h // under it. The export root has no parent name and is refused. func (l *Local) Parent(h Handle) (Handle, string, error) { _, path, _, err := l.revalidate(h) if err != nil { return nil, "", err } dir := filepath.Dir(path) if dir == path || path == l.root { // The export root has no parent name, and a parent outside the // export must never resolve. return nil, "", ErrInval } ph, _, err := l.link(dir) if err != nil { return nil, "", err } return ph, filepath.Base(path), nil } // Open opens the regular file name under dir for writing. It is the // backend half of the OPEN operation: the create, guarded and truncate // decisions belong to the caller, which reads them from the protocol. The // truncate of an existing file runs through a revalidated descriptor, so a // name swapped for a symlink under the call is never followed. A file this // call creates carries owner. func (l *Local) Open(dir Handle, name string, create, guarded, truncate bool, perm fs.FileMode, owner Owner) (Handle, Info, bool, error) { if err := ValidName(name); err != nil { return nil, Info{}, false, err } dirPath, err := l.dirOf(dir) if err != nil { return nil, Info{}, false, err } path := filepath.Join(dirPath, name) fi, err := os.Lstat(path) switch { case err == nil: if !fi.Mode().IsRegular() { return nil, Info{}, false, ErrIsDir } // A guarded create refuses an existing name outright, the // create mode GUARDED and EXCLUSIVE4_1 of RFC 8881 section // 18.16. if create && guarded { return nil, Info{}, false, ErrExist } if create && truncate { f, terr := os.OpenFile(path, os.O_WRONLY|syscall.O_NOFOLLOW, 0) if terr != nil { if symlinkRefused(terr) { return nil, Info{}, false, ErrStale } return nil, Info{}, false, wrapWriteErr(terr) } tfi, serr := f.Stat() id := stat(fi) if serr != nil || !sameFile(tfi, kindOf(fi.Mode()), id.Dev, id.Ino) { f.Close() if serr != nil { return nil, Info{}, false, wrapWriteErr(serr) } return nil, Info{}, false, ErrStale } terr = f.Truncate(0) f.Close() if terr != nil { return nil, Info{}, false, wrapWriteErr(terr) } } case errors.Is(err, fs.ErrNotExist): if !create { return nil, Info{}, false, ErrNoEnt } f, ferr := os.OpenFile(path, os.O_CREATE|os.O_EXCL|os.O_WRONLY, fileMode(modeBits(perm))) if ferr != nil { return nil, Info{}, false, wrapCreateErr(ferr) } f.Close() // The permission bits are applied exactly, the way CREATE does: // the kernel distorted them by the daemon's umask at the open, // and the client asked for the bits, not for the umask. if serr := os.Chmod(path, fileMode(modeBits(perm))); serr != nil { return nil, Info{}, false, wrapWriteErr(serr) } // The daemon's own identity owns what it makes; a client of // another owner expects its object to carry its owner, so the // fresh file is handed over at once. A chown the daemon cannot // make leaves the file in place rather than undoing the create. _ = os.Chown(path, int(owner.UID), int(owner.GID)) default: return nil, Info{}, false, wrapWriteErr(err) } h, info, lerr := l.link(path) if lerr != nil { return nil, Info{}, false, lerr } return h, info, fi == nil, nil } // PersistHandles writes the dev, ino to path mapping into dir, so a // restarted server can resolve the handles it issued before. The mapping // is the recovery state of the backend: without it every pre restart // handle is stale, whatever the grace window says. func (l *Local) PersistHandles(dir string) error { l.mu.RLock() out := make(map[string]string, len(l.paths)) for id, p := range l.paths { out[persistKey(id)] = p } l.mu.RUnlock() data, err := json.Marshal(out) if err != nil { return err } return os.WriteFile(filepath.Join(dir, "handles.json"), data, 0o600) } // inside reports whether the cleaned path stays inside the export root. func (l *Local) inside(p string) bool { c := filepath.Clean(p) return c == l.root || strings.HasPrefix(c, l.root+string(os.PathSeparator)) } // LoadPersistedHandles reads a previously persisted dev, ino to path // mapping back into the store. Entries of the older ino only format are // dropped, and so is any entry whose path does not stay inside the export // root: the file is recovery state, never a source of export boundaries. func (l *Local) LoadPersistedHandles(dir string) error { data, err := os.ReadFile(filepath.Join(dir, "handles.json")) if err != nil { if errors.Is(err, os.ErrNotExist) { return nil } return err } var out map[string]string if err := json.Unmarshal(data, &out); err != nil { return err } l.mu.Lock() defer l.mu.Unlock() for key, p := range out { devS, inoS, ok := strings.Cut(key, ":") if !ok { continue } dev, derr := strconv.ParseUint(devS, 16, 64) if derr != nil { continue } ino, ierr := strconv.ParseUint(inoS, 16, 64) if ierr != nil { continue } if !l.inside(p) { continue } id := fileID{dev: dev, ino: ino} if _, exists := l.paths[id]; !exists { l.paths[id] = p } } return nil }