Files
nfs/internal/nfsfs/local.go
T

1149 lines
32 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"encoding/binary"
"encoding/json"
"errors"
"fmt"
"io"
"io/fs"
"net"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"sync"
"syscall"
"time"
)
// A Local serves one local directory tree over dev and ino based file
// handles. A handle encodes the device and inode number; the mapping from
// that pair to a path is held in memory and persisted on demand, so a
// handle from before a restart resolves when the mapping is loaded back.
// Every use revalidates the mapping: the path must still name the device,
// inode and kind the handle encodes, and no resolution follows a final
// symlink, so a name swapped for a link is stale rather than an escape.
type Local struct {
root string
mu sync.RWMutex
paths map[fileID]string
persistPath string
// The descriptor cache of fdcache.go: idle descriptors of regular
// files, bounded by fdCacheLimit, every use reverified against the
// registered path and the descriptor's own identity.
fdMu sync.Mutex
fds map[fdKey]*fdEntry
fdUse uint64
// The listing cache of ReadDir: the sorted names of the directories
// being paged, bounded by dirCacheMax, valid while the directory's
// modification time matches.
dirMu sync.Mutex
dirs map[fileID]*cachedDir
dirUse uint64
}
// A fileID identifies one inode on one device: the key of the handle to
// path mapping.
type fileID struct {
dev uint64
ino uint64
}
// handle layout: magic byte, version byte, type byte, dev, ino.
const (
handleMagic = 0x4e
// handleVersion names the handle layout. Version two keys the mapping
// by device and inode and revalidates on use; handles of version one
// carry no device to check against and answer stale.
handleVersion = 2
handleSize = 3 + 8 + 8
)
// handle type bytes, mirroring the file kinds the protocol distinguishes.
const (
typeDir = 1
typeFile = 2
typeOther = 3
)
// NewLocal returns a Local serving root. The path must be an existing
// directory.
func NewLocal(root string) (*Local, error) {
abs, err := filepath.Abs(root)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrIO, err)
}
st, err := os.Lstat(abs)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrNoEnt, err)
}
if !st.IsDir() {
return nil, fmt.Errorf("%w: %s is not a directory", ErrNotDir, abs)
}
l := &Local{root: abs, paths: make(map[fileID]string),
fds: make(map[fdKey]*fdEntry), dirs: make(map[fileID]*cachedDir)}
if _, _, err := l.link(abs); err != nil {
return nil, err
}
return l, nil
}
// stat converts an os.FileInfo plus its raw stat into an Info.
func stat(fi os.FileInfo) Info {
info := Info{
Size: fi.Size(),
Mode: fi.Mode(),
ModTime: fi.ModTime(),
Nlink: 1,
}
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
info.Dev = uint64(st.Dev)
info.Ino = uint64(st.Ino)
info.Nlink = uint64(st.Nlink)
info.UID = st.Uid
info.GID = st.Gid
}
return info
}
// kindOf maps a file mode onto the handle type byte.
func kindOf(mode fs.FileMode) byte {
switch {
case mode.IsDir():
return typeDir
case mode.IsRegular():
return typeFile
default:
return typeOther
}
}
// sameFile reports whether fi names the device, inode and kind a handle
// encodes.
func sameFile(fi os.FileInfo, kind byte, dev, ino uint64) bool {
st, ok := fi.Sys().(*syscall.Stat_t)
if !ok {
return false
}
return uint64(st.Dev) == dev && uint64(st.Ino) == ino && kindOf(fi.Mode()) == kind
}
// SetPersistPath aims the handle mapping persistence at a file inside
// dir. Every registered handle is saved through it, and the mapping is
// written once right away.
func (l *Local) SetPersistPath(dir string) {
l.mu.Lock()
l.persistPath = filepath.Join(dir, "handles.json")
l.mu.Unlock()
l.save()
}
// persistTarget answers the persistence file path, read under the lock.
func (l *Local) persistTarget() string {
l.mu.RLock()
defer l.mu.RUnlock()
return l.persistPath
}
// save writes the mapping file when persistence is armed. The snapshot is
// taken under the read lock; the writing runs outside it.
func (l *Local) save() {
target := l.persistTarget()
if target == "" {
return
}
l.mu.RLock()
out := make(map[string]string, len(l.paths))
for id, p := range l.paths {
out[persistKey(id)] = p
}
l.mu.RUnlock()
data, err := json.Marshal(out)
if err != nil {
return
}
tmp := target + ".tmp"
if err := os.WriteFile(tmp, data, 0o600); err != nil {
return
}
_ = os.Rename(tmp, target)
}
// persistKey renders the map key of one registered pair, the device and
// inode numbers in hex joined by a colon.
func persistKey(id fileID) string {
return strconv.FormatUint(id.dev, 16) + ":" + strconv.FormatUint(id.ino, 16)
}
// link records the path under its device and inode number and returns its
// handle and attributes.
func (l *Local) link(path string) (Handle, Info, error) {
fi, err := os.Lstat(path)
if err != nil {
if errors.Is(err, syscall.ENOTDIR) {
return nil, Info{}, ErrNotDir
}
if errors.Is(err, fs.ErrPermission) {
return nil, Info{}, ErrPermission
}
return nil, Info{}, fmt.Errorf("%w: %v", ErrNoEnt, err)
}
info := stat(fi)
if info.Ino == 0 {
return nil, Info{}, fmt.Errorf("%w: %s has no inode number", ErrIO, path)
}
dirty := false
l.mu.Lock()
id := fileID{dev: info.Dev, ino: info.Ino}
if prev, ok := l.paths[id]; !ok || prev != path {
l.paths[id] = path
dirty = true
}
l.mu.Unlock()
// The mapping is the recovery state of the handles: it is written the
// moment it changes, so a restart never loses a handle it served. An
// unchanged registration writes nothing, which keeps a large READDIR
// from rewriting the same file once per entry.
if dirty && l.persistTarget() != "" {
l.save()
}
return encodeHandle(info, path), info, nil
}
// encodeHandle builds the opaque handle for an already linked path.
func encodeHandle(info Info, path string) Handle {
h := make(Handle, handleSize)
h[0] = handleMagic
h[1] = handleVersion
h[2] = kindOf(info.Mode)
binary.BigEndian.PutUint64(h[3:11], info.Dev)
binary.BigEndian.PutUint64(h[11:19], info.Ino)
_ = path
return h
}
// decode parses a handle and returns its kind byte, device and inode. A
// handle of another magic, length or version is stale.
func decode(h Handle) (byte, uint64, uint64, error) {
if len(h) != handleSize || h[0] != handleMagic || h[1] != handleVersion {
return 0, 0, 0, ErrStale
}
return h[2], binary.BigEndian.Uint64(h[3:11]), binary.BigEndian.Uint64(h[11:19]), nil
}
// resolve decodes a handle and reports its kind byte and the registered
// path of the device and inode it names. A handle the process never
// issued, or one from before a restart, resolves to ErrStale.
func (l *Local) resolve(h Handle) (byte, uint64, uint64, string, error) {
kind, dev, ino, err := decode(h)
if err != nil {
return 0, 0, 0, "", err
}
l.mu.RLock()
path, ok := l.paths[fileID{dev: dev, ino: ino}]
l.mu.RUnlock()
if !ok {
return 0, 0, 0, "", ErrStale
}
return kind, dev, ino, path, nil
}
// revalidate resolves a handle and confirms through Lstat that the
// registered path still names its device, inode and kind. Lstat never
// follows a final symlink, so a name swapped for a link is stale.
func (l *Local) revalidate(h Handle) (byte, string, os.FileInfo, error) {
kind, dev, ino, path, err := l.resolve(h)
if err != nil {
return 0, "", nil, err
}
fi, err := os.Lstat(path)
if err != nil {
return 0, "", nil, revalidateStatErr(err)
}
if !sameFile(fi, kind, dev, ino) {
return 0, "", nil, ErrStale
}
return kind, path, fi, nil
}
// dirOf resolves a handle that must name a directory, revalidated against
// the registered path.
func (l *Local) dirOf(h Handle) (string, error) {
kind, path, _, err := l.revalidate(h)
if err != nil {
return "", err
}
if kind != typeDir {
return "", ErrNotDir
}
return path, nil
}
// openVerified resolves a handle to an open descriptor, never following a
// final symlink, and requires the descriptor to name the device, inode
// and kind the handle encodes. Anything else is stale.
func (l *Local) openVerified(h Handle, flag int) (*os.File, os.FileInfo, error) {
kind, dev, ino, path, err := l.resolve(h)
if err != nil {
return nil, nil, err
}
f, err := os.OpenFile(path, flag|syscall.O_NOFOLLOW, 0)
if err != nil {
switch {
case errors.Is(err, fs.ErrNotExist), errors.Is(err, syscall.ENOTDIR),
symlinkRefused(err):
return nil, nil, ErrStale
case errors.Is(err, syscall.EISDIR):
return nil, nil, ErrIsDir
case errors.Is(err, fs.ErrPermission):
return nil, nil, ErrPermission
default:
return nil, nil, fmt.Errorf("%w: %v", ErrIO, err)
}
}
fi, err := f.Stat()
if err != nil {
f.Close()
return nil, nil, fmt.Errorf("%w: %v", ErrIO, err)
}
if !sameFile(fi, kind, dev, ino) {
f.Close()
return nil, nil, ErrStale
}
return f, fi, nil
}
// Root returns the handle of the export root.
func (l *Local) Root() (Handle, error) {
h, _, err := l.link(l.root)
return h, err
}
// Lookup resolves name under the parent handle.
func (l *Local) Lookup(parent Handle, name string) (Handle, Info, error) {
if err := ValidName(name); err != nil {
return nil, Info{}, err
}
parentPath, err := l.dirOf(parent)
if err != nil {
return nil, Info{}, err
}
child := filepath.Join(parentPath, name)
h, info, err := l.link(child)
if err != nil {
return nil, Info{}, err
}
return h, info, nil
}
// Getattr reports the attributes of a handle.
func (l *Local) Getattr(h Handle) (Info, error) {
_, _, fi, err := l.revalidate(h)
if err != nil {
return Info{}, err
}
return stat(fi), nil
}
// dirCacheMax bounds the directory listing cache: the sorted names of
// the directories a client pages through, held until their modification
// time moves. One bounded map, entries evicted least recently used.
const dirCacheMax = 64
// A cachedDir is the sorted name order of one directory, keyed by the
// directory's file identity, valid while its verifier matches.
type cachedDir struct {
names []string
verifier [8]byte
use uint64
}
// ReadDir lists the directory from the given cookie. The cookie is the
// one-based position in the sorted name order, and the verifier is the
// directory's modification time, so a listing that raced a change is
// detected by the caller.
//
// The sorted order is cached per directory and revalidated against the
// verifier on every page: paging a large directory costs the page alone,
// not a full re listing and re sort, and any change to the directory
// moves the verifier and forces a fresh listing.
func (l *Local) ReadDir(h Handle, cookie uint64, count int) (DirPage, error) {
path, err := l.dirOf(h)
if err != nil {
return DirPage{}, err
}
root, err2 := filepath.Abs(path)
if err2 != nil {
return DirPage{}, fmt.Errorf("%w: %v", ErrIO, err2)
}
info, err2 := os.Lstat(root)
if err2 != nil {
return DirPage{}, fmt.Errorf("%w: %v", ErrIO, err2)
}
var verifier [8]byte
binary.BigEndian.PutUint64(verifier[:], uint64(info.ModTime().UnixNano()))
var id fileID
haveID := false
if st, ok := info.Sys().(*syscall.Stat_t); ok {
id = fileID{dev: uint64(st.Dev), ino: uint64(st.Ino)}
haveID = true
}
names, err2 := l.cachedNames(id, haveID, verifier, func() ([]string, error) {
entries, err := os.ReadDir(path)
if err != nil {
if errors.Is(err, syscall.ENOTDIR) {
return nil, ErrNotDir
}
return nil, fmt.Errorf("%w: %v", ErrIO, err)
}
sortNames(entries)
names := make([]string, len(entries))
for i, e := range entries {
names[i] = e.Name()
}
return names, nil
})
if err2 != nil {
return DirPage{}, err2
}
var page DirPage
page.Verifier = verifier
for i, name := range names {
c := uint64(i) + 1
if c <= cookie {
continue
}
if count > 0 && len(page.Entries) >= count {
return page, nil
}
child := filepath.Join(root, name)
h, info, err := l.link(child)
if err != nil {
// A file removed between ReadDir and Lstat is skipped, not an
// error for the whole listing.
continue
}
page.Entries = append(page.Entries, Entry{Cookie: c, Name: name, Handle: h, Info: info})
}
page.EOF = true
return page, nil
}
// cachedNames answers the sorted names of a directory: from the cache
// while the verifier matches, from the fill function otherwise. A
// directory without a raw stat bypasses the cache, since its identity
// would collide with the next one.
func (l *Local) cachedNames(id fileID, haveID bool, verifier [8]byte, fill func() ([]string, error)) ([]string, error) {
if haveID {
l.dirMu.Lock()
if cd := l.dirs[id]; cd != nil && cd.verifier == verifier {
l.dirUse++
cd.use = l.dirUse
l.dirMu.Unlock()
return cd.names, nil
}
l.dirMu.Unlock()
}
names, err := fill()
if err != nil {
return nil, err
}
if haveID {
l.dirMu.Lock()
l.dirUse++
l.dirs[id] = &cachedDir{names: names, verifier: verifier, use: l.dirUse}
for len(l.dirs) > dirCacheMax {
var victim fileID
var oldest uint64
first := true
for key, cd := range l.dirs {
if first || cd.use < oldest {
victim, oldest, first = key, cd.use, false
}
}
delete(l.dirs, victim)
}
l.dirMu.Unlock()
}
return names, nil
}
// sortNames orders a directory listing by name, the order the cookies are
// defined against.
func sortNames(entries []os.DirEntry) {
slices.SortFunc(entries, func(a, b os.DirEntry) int {
return strings.Compare(a.Name(), b.Name())
})
}
// Read reads up to count bytes at the offset from a regular file. The
// descriptor comes from the cache or a fresh verified open, and the
// identity is revalidated before anything is read.
func (l *Local) Read(h Handle, off int64, count int) ([]byte, error) {
f, release, err := l.dataFD(h, false)
if err != nil {
return nil, err
}
defer release()
buf := make([]byte, count)
n, err := f.ReadAt(buf, off)
if err != nil && !errors.Is(err, io.EOF) {
return nil, fmt.Errorf("%w: %v", ErrIO, err)
}
return buf[:n], nil
}
// Access evaluates the requested mask bits for the credential, using the
// classic owner, group and other selection over the permission bits. The
// superuser is granted everything.
func (l *Local) Access(h Handle, mask uint32, uid, gid uint32, groups []uint32) (uint32, error) {
if uid == 0 {
return mask, nil
}
info, err := l.Getattr(h)
if err != nil {
return 0, err
}
var mode fs.FileMode
switch {
case uid == info.UID:
mode = info.Mode.Perm() >> 6
case gid == info.GID || containsGID(groups, info.GID):
mode = info.Mode.Perm() >> 3
default:
mode = info.Mode.Perm()
}
var granted uint32
for bit, want := range map[uint32]fs.FileMode{
AccessRead: 0o4,
AccessLookup: 0o1,
AccessModify: 0o2,
AccessExtend: 0o2,
AccessDelete: 0o2,
AccessExec: 0o1,
} {
if mask&bit != 0 && mode&want != 0 {
granted |= bit
}
}
return granted, nil
}
func containsGID(groups []uint32, gid uint32) bool {
return slices.Contains(groups, gid)
}
// modeBits renders a mode the way the raw create and chmod calls receive
// it: the low nine permission bits plus the setuid, setgid and sticky
// bits, wherever the mode carries them.
func modeBits(m fs.FileMode) uint32 {
bits := uint32(m & (os.ModePerm | 0o7000))
if m&os.ModeSetuid != 0 {
bits |= 0o4000
}
if m&os.ModeSetgid != 0 {
bits |= 0o2000
}
if m&os.ModeSticky != 0 {
bits |= 0o1000
}
return bits
}
// fileMode renders twelve raw mode bits as the FileMode the chmod family
// receives.
func fileMode(bits uint32) fs.FileMode {
m := fs.FileMode(bits & 0o777)
if bits&0o4000 != 0 {
m |= os.ModeSetuid
}
if bits&0o2000 != 0 {
m |= os.ModeSetgid
}
if bits&0o1000 != 0 {
m |= os.ModeSticky
}
return m
}
// Create makes the object the spec describes under the parent handle. An
// existing target is an error for every kind; the permission bits are
// applied exactly, all twelve of them, with a chmod after the creation, so
// the daemon's umask never distorts what the client asked for.
func (l *Local) Create(parent Handle, name string, spec CreateSpec) (Handle, Info, error) {
if err := ValidName(name); err != nil {
return nil, Info{}, err
}
parentPath, err := l.dirOf(parent)
if err != nil {
return nil, Info{}, err
}
path := filepath.Join(parentPath, name)
if _, err := os.Lstat(path); err == nil {
return nil, Info{}, ErrExist
} else if !errors.Is(err, fs.ErrNotExist) {
return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err)
}
perm := modeBits(spec.Perm)
made := false
switch spec.Kind {
case KindDir:
if err := os.Mkdir(path, fileMode(perm)); err != nil {
return nil, Info{}, wrapCreateErr(err)
}
_ = os.Chmod(path, fileMode(perm))
made = true
case KindLnk:
if err := os.Symlink(spec.LinkData, path); err != nil {
return nil, Info{}, wrapCreateErr(err)
}
case KindFifo:
if err := syscall.Mkfifo(path, perm); err != nil {
return nil, Info{}, wrapCreateErr(err)
}
_ = os.Chmod(path, fileMode(perm))
made = true
case KindSock:
if err := bindUnixSocket(path); err != nil {
return nil, Info{}, wrapCreateErr(err)
}
_ = os.Chmod(path, fileMode(perm))
made = true
case KindBlk, KindChr:
// A device node needs CAP_MKNOD on Linux; without it the failure
// is a permission problem and says so.
if err := mknod(path, spec, perm); err != nil {
return nil, Info{}, wrapCreateErr(err)
}
_ = os.Chmod(path, fileMode(perm))
made = true
default:
return nil, Info{}, ErrBadName
}
if made {
// The daemon hands every object it makes over to the owner the
// client named; a symlink carries no access check of its own, so
// it alone keeps the daemon's identity.
_ = os.Chown(path, int(spec.Owner.UID), int(spec.Owner.GID))
}
h, info, err := l.link(path)
if err != nil {
return nil, Info{}, err
}
return h, info, nil
}
// Write writes all of data at the offset of a regular file. The
// descriptor comes from the cache or a fresh verified open, and the
// identity is revalidated before anything is written.
func (l *Local) Write(h Handle, off int64, data []byte) (int, error) {
f, release, err := l.dataFD(h, true)
if err != nil {
return 0, err
}
defer release()
n, err := f.WriteAt(data, off)
if err != nil && !errors.Is(err, io.EOF) {
if errors.Is(err, syscall.ENOSPC) {
return int(n), ErrNoSpace
}
return int(n), fmt.Errorf("%w: %v", ErrIO, err)
}
return n, nil
}
// wrapCreateErr maps the errors of the create system calls.
func wrapCreateErr(err error) error {
switch {
case err == nil:
return nil
case errors.Is(err, fs.ErrExist):
return ErrExist
case errors.Is(err, fs.ErrNotExist):
// A create whose parent directory is missing: the client learns
// the name it walked to is gone, not that the disk failed.
return ErrNoEnt
case errors.Is(err, fs.ErrPermission):
return ErrPermission
case errors.Is(err, fs.ErrInvalid):
return ErrBadName
case errors.Is(err, syscall.EISDIR):
return ErrIsDir
case errors.Is(err, syscall.ENOTDIR):
return ErrNotDir
case errors.Is(err, syscall.ENOSPC):
return ErrNoSpace
default:
return fmt.Errorf("%w: %v", ErrIO, err)
}
}
// bindUnixSocket creates a unix domain socket file at path. The listener
// is closed at once; the file it bound remains.
func bindUnixSocket(path string) error {
ln, err := net.Listen("unix", path)
if err != nil {
return wrapCreateErr(err)
}
if u, ok := ln.(*net.UnixListener); ok {
u.SetUnlinkOnClose(false)
}
return ln.Close()
}
// Remove takes the named entry out of the directory. An empty directory is
// removed like anything else; a directory that still holds entries is
// ErrNotEmpty. A name that cannot be examined for permission reasons is a
// permission error, not a missing one.
func (l *Local) Remove(dir Handle, name string) error {
if err := ValidName(name); err != nil {
return err
}
dirPath, err := l.dirOf(dir)
if err != nil {
return err
}
path := filepath.Join(dirPath, name)
fi, err := os.Lstat(path)
if err != nil {
switch {
case errors.Is(err, fs.ErrNotExist):
return ErrNoEnt
case errors.Is(err, fs.ErrPermission), errors.Is(err, syscall.EPERM):
return ErrPermission
default:
return fmt.Errorf("%w: %v", ErrIO, err)
}
}
if err := os.Remove(path); err != nil {
if errors.Is(err, syscall.ENOTEMPTY) {
return ErrNotEmpty
}
if errors.Is(err, fs.ErrPermission) {
return ErrPermission
}
return fmt.Errorf("%w: %v", ErrIO, err)
}
if id := stat(fi); id.Ino != 0 {
l.mu.Lock()
if l.paths[fileID{dev: id.Dev, ino: id.Ino}] == path {
delete(l.paths, fileID{dev: id.Dev, ino: id.Ino})
}
l.mu.Unlock()
}
return nil
}
// Rename moves oldName from oldDir to newName in newDir, replacing an
// existing plain target the way POSIX rename does. The moved subtree is
// re-registered under its new paths, so the handles of the object and of
// its descendants keep resolving after the move.
func (l *Local) Rename(oldDir Handle, oldName string, newDir Handle, newName string) error {
if err := ValidName(oldName); err != nil {
return err
}
if err := ValidName(newName); err != nil {
return err
}
oldDirPath, err := l.dirOf(oldDir)
if err != nil {
return err
}
newDirPath, err := l.dirOf(newDir)
if err != nil {
return err
}
oldPath := filepath.Join(oldDirPath, oldName)
newPath := filepath.Join(newDirPath, newName)
if oldPath == newPath {
return nil
}
if _, err := os.Lstat(oldPath); err != nil {
if errors.Is(err, fs.ErrNotExist) {
return ErrNoEnt
}
return fmt.Errorf("%w: %v", ErrIO, err)
}
if err := os.Rename(oldPath, newPath); err != nil {
switch {
case errors.Is(err, fs.ErrNotExist):
return ErrNoEnt
case errors.Is(err, syscall.ENOTEMPTY):
return ErrNotEmpty
case errors.Is(err, syscall.EINVAL):
return ErrInval
case errors.Is(err, fs.ErrPermission):
return ErrPermission
default:
return fmt.Errorf("%w: %v", ErrIO, err)
}
}
// Register the moved object, and when it is a directory, every
// descendant under its new path, so the handles already issued by
// earlier READDIRs and LOOKUPs keep working.
h, info, err := l.link(newPath)
if err != nil {
return err
}
_ = h
if info.IsDir() {
// Every descendant moved too: re-register them under their new
// paths, skipping entries that vanish while the walk runs.
_ = filepath.WalkDir(newPath, func(p string, d fs.DirEntry, werr error) error {
if werr != nil || p == newPath {
return nil
}
_, _, _ = l.link(p)
return nil
})
}
return nil
}
// Setattr applies the named changes to a file, in the order size, mode,
// owner, times. The registered path is revalidated first, so a name
// swapped for another inode, a symlink included, is stale before anything
// is applied. A change the backend cannot apply fails the whole call.
func (l *Local) Setattr(h Handle, s SetAttrs) error {
_, path, fi, err := l.revalidate(h)
if err != nil {
return err
}
if s.Size != nil {
if fi.IsDir() {
return ErrIsDir
}
if serr := os.Truncate(path, *s.Size); serr != nil {
return wrapWriteErr(serr)
}
}
if s.Mode != nil {
if serr := os.Chmod(path, fileMode(*s.Mode&0o7777)); serr != nil {
return wrapWriteErr(serr)
}
}
if s.UID != nil || s.GID != nil {
uid, gid := -1, -1
if s.UID != nil {
uid = int(*s.UID)
}
if s.GID != nil {
gid = int(*s.GID)
}
if serr := os.Chown(path, uid, gid); serr != nil {
return wrapWriteErr(serr)
}
}
if s.Atime != nil || s.Mtime != nil {
// Chtimes wants both times: whatever the client left out keeps the
// value the file carries now.
atime, mtime := time.Now(), time.Now()
if s.Atime == nil || s.Mtime == nil {
if fi, serr := os.Lstat(path); serr == nil {
atime, mtime = fi.ModTime(), fi.ModTime()
}
}
if s.Atime != nil {
atime = resolveTime(*s.Atime)
}
if s.Mtime != nil {
mtime = resolveTime(*s.Mtime)
}
if serr := os.Chtimes(path, atime, mtime); serr != nil {
return wrapWriteErr(serr)
}
}
return nil
}
// resolveTime turns a settime4 into the time it names.
func resolveTime(t TimeSet) time.Time {
if t.Now {
return time.Now()
}
return t.Time
}
// wrapWriteErr maps the errors of the mutating system calls.
func wrapWriteErr(err error) error {
switch {
case err == nil:
return nil
case errors.Is(err, fs.ErrPermission):
return ErrPermission
case errors.Is(err, fs.ErrNotExist):
return ErrStale
case errors.Is(err, syscall.ENOSPC):
return ErrNoSpace
case errors.Is(err, syscall.EINVAL):
return ErrInval
default:
return fmt.Errorf("%w: %v", ErrIO, err)
}
}
// Link makes newName in dir a hard link to the target file. Directories
// are refused: the protocol reserves hard links for regular files and the
// kernel refuses the rest.
func (l *Local) Link(target Handle, dir Handle, name string) (Handle, Info, error) {
if err := ValidName(name); err != nil {
return nil, Info{}, err
}
targetKind, targetPath, _, err := l.revalidate(target)
if err != nil {
return nil, Info{}, err
}
if targetKind == typeDir {
return nil, Info{}, ErrIsDir
}
dirPath, err := l.dirOf(dir)
if err != nil {
return nil, Info{}, err
}
newPath := filepath.Join(dirPath, name)
if _, err := os.Lstat(newPath); err == nil {
return nil, Info{}, ErrExist
} else if !errors.Is(err, fs.ErrNotExist) {
return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err)
}
if err := os.Link(targetPath, newPath); err != nil {
if errors.Is(err, fs.ErrExist) {
return nil, Info{}, ErrExist
}
if errors.Is(err, fs.ErrPermission) {
return nil, Info{}, ErrPermission
}
if errors.Is(err, syscall.EPERM) {
// The kernel refuses hard links to directories.
return nil, Info{}, ErrIsDir
}
return nil, Info{}, fmt.Errorf("%w: %v", ErrIO, err)
}
return l.link(newPath)
}
// ReadLink reports the target of a symlink. Anything else is refused:
// the protocol answers NFS4ERR_INVAL for a READLINK on a non link.
func (l *Local) ReadLink(h Handle) (string, error) {
kind, path, fi, err := l.revalidate(h)
if err != nil {
return "", err
}
if kind != typeOther || fi.Mode()&os.ModeSymlink == 0 {
return "", ErrNotLnk
}
target, err := os.Readlink(path)
if err != nil {
return "", fmt.Errorf("%w: %v", ErrIO, err)
}
return target, nil
}
// Sync flushes the dirty data of a regular file, or of the directory
// itself, to stable storage. The stateless backend writes synchronously,
// so this is the belt to the braces of the FILE_SYNC answer. A regular
// file syncs through the cached descriptor, a directory through a fresh
// verified open; both revalidate the identity first.
func (l *Local) Sync(h Handle) error {
kind, _, _, _, err := l.resolve(h)
if err != nil {
return err
}
if kind != typeFile {
f, _, err := l.openVerified(h, os.O_RDONLY)
if err != nil {
return err
}
defer f.Close()
if err := f.Sync(); err != nil {
return wrapWriteErr(err)
}
return nil
}
f, release, err := l.dataFD(h, true)
if err != nil {
return err
}
defer release()
if err := f.Sync(); err != nil {
return wrapWriteErr(err)
}
return nil
}
// Parent resolves the directory that holds h and the component name of h
// under it. The export root has no parent name and is refused.
func (l *Local) Parent(h Handle) (Handle, string, error) {
_, path, _, err := l.revalidate(h)
if err != nil {
return nil, "", err
}
dir := filepath.Dir(path)
if dir == path || path == l.root {
// The export root has no parent name, and a parent outside the
// export must never resolve.
return nil, "", ErrInval
}
ph, _, err := l.link(dir)
if err != nil {
return nil, "", err
}
return ph, filepath.Base(path), nil
}
// Open opens the regular file name under dir for writing. It is the
// backend half of the OPEN operation: the create, guarded and truncate
// decisions belong to the caller, which reads them from the protocol. The
// truncate of an existing file runs through a revalidated descriptor, so a
// name swapped for a symlink under the call is never followed. A file this
// call creates carries owner.
func (l *Local) Open(dir Handle, name string, create, guarded, truncate bool, perm fs.FileMode, owner Owner) (Handle, Info, bool, error) {
if err := ValidName(name); err != nil {
return nil, Info{}, false, err
}
dirPath, err := l.dirOf(dir)
if err != nil {
return nil, Info{}, false, err
}
path := filepath.Join(dirPath, name)
fi, err := os.Lstat(path)
switch {
case err == nil:
if !fi.Mode().IsRegular() {
return nil, Info{}, false, ErrIsDir
}
// A guarded create refuses an existing name outright, the
// create mode GUARDED and EXCLUSIVE4_1 of RFC 8881 section
// 18.16.
if create && guarded {
return nil, Info{}, false, ErrExist
}
if create && truncate {
f, terr := os.OpenFile(path, os.O_WRONLY|syscall.O_NOFOLLOW, 0)
if terr != nil {
if symlinkRefused(terr) {
return nil, Info{}, false, ErrStale
}
return nil, Info{}, false, wrapWriteErr(terr)
}
tfi, serr := f.Stat()
id := stat(fi)
if serr != nil || !sameFile(tfi, kindOf(fi.Mode()), id.Dev, id.Ino) {
f.Close()
if serr != nil {
return nil, Info{}, false, wrapWriteErr(serr)
}
return nil, Info{}, false, ErrStale
}
terr = f.Truncate(0)
f.Close()
if terr != nil {
return nil, Info{}, false, wrapWriteErr(terr)
}
}
case errors.Is(err, fs.ErrNotExist):
if !create {
return nil, Info{}, false, ErrNoEnt
}
f, ferr := os.OpenFile(path, os.O_CREATE|os.O_EXCL|os.O_WRONLY, fileMode(modeBits(perm)))
if ferr != nil {
return nil, Info{}, false, wrapCreateErr(ferr)
}
f.Close()
// The permission bits are applied exactly, the way CREATE does:
// the kernel distorted them by the daemon's umask at the open,
// and the client asked for the bits, not for the umask.
if serr := os.Chmod(path, fileMode(modeBits(perm))); serr != nil {
return nil, Info{}, false, wrapWriteErr(serr)
}
// The daemon's own identity owns what it makes; a client of
// another owner expects its object to carry its owner, so the
// fresh file is handed over at once. A chown the daemon cannot
// make leaves the file in place rather than undoing the create.
_ = os.Chown(path, int(owner.UID), int(owner.GID))
default:
return nil, Info{}, false, wrapWriteErr(err)
}
h, info, lerr := l.link(path)
if lerr != nil {
return nil, Info{}, false, lerr
}
return h, info, fi == nil, nil
}
// PersistHandles writes the dev, ino to path mapping into dir, so a
// restarted server can resolve the handles it issued before. The mapping
// is the recovery state of the backend: without it every pre restart
// handle is stale, whatever the grace window says.
func (l *Local) PersistHandles(dir string) error {
l.mu.RLock()
out := make(map[string]string, len(l.paths))
for id, p := range l.paths {
out[persistKey(id)] = p
}
l.mu.RUnlock()
data, err := json.Marshal(out)
if err != nil {
return err
}
return os.WriteFile(filepath.Join(dir, "handles.json"), data, 0o600)
}
// inside reports whether the cleaned path stays inside the export root.
func (l *Local) inside(p string) bool {
c := filepath.Clean(p)
return c == l.root || strings.HasPrefix(c, l.root+string(os.PathSeparator))
}
// LoadPersistedHandles reads a previously persisted dev, ino to path
// mapping back into the store. Entries of the older ino only format are
// dropped, and so is any entry whose path does not stay inside the export
// root: the file is recovery state, never a source of export boundaries.
func (l *Local) LoadPersistedHandles(dir string) error {
data, err := os.ReadFile(filepath.Join(dir, "handles.json"))
if err != nil {
if errors.Is(err, os.ErrNotExist) {
return nil
}
return err
}
var out map[string]string
if err := json.Unmarshal(data, &out); err != nil {
return err
}
l.mu.Lock()
defer l.mu.Unlock()
for key, p := range out {
devS, inoS, ok := strings.Cut(key, ":")
if !ok {
continue
}
dev, derr := strconv.ParseUint(devS, 16, 64)
if derr != nil {
continue
}
ino, ierr := strconv.ParseUint(inoS, 16, 64)
if ierr != nil {
continue
}
if !l.inside(p) {
continue
}
id := fileID{dev: dev, ino: ino}
if _, exists := l.paths[id]; !exists {
l.paths[id] = p
}
}
return nil
}