feat: full NFSv4.2 server and client in pure Go
Test / test (push) Successful in 2m4s
Release / gates (push) Successful in 2m5s
Release / build (amd64, freebsd) (push) Successful in 1m27s
Release / build (amd64, linux) (push) Successful in 1m22s
Release / build (amd64, netbsd) (push) Successful in 1m19s
Release / build (amd64, openbsd) (push) Successful in 1m20s
Release / build (arm64, darwin) (push) Successful in 1m21s
Release / build (arm64, freebsd) (push) Successful in 1m26s
Release / build (arm64, linux) (push) Successful in 1m25s
Release / build (arm64, netbsd) (push) Successful in 1m31s
Release / build (arm64, openbsd) (push) Successful in 1m27s
Release / build (loong64, linux) (push) Successful in 1m37s
Release / build (riscv64, linux) (push) Successful in 1m21s
Release / release (push) Successful in 40s
Test / test (push) Successful in 2m4s
Release / gates (push) Successful in 2m5s
Release / build (amd64, freebsd) (push) Successful in 1m27s
Release / build (amd64, linux) (push) Successful in 1m22s
Release / build (amd64, netbsd) (push) Successful in 1m19s
Release / build (amd64, openbsd) (push) Successful in 1m20s
Release / build (arm64, darwin) (push) Successful in 1m21s
Release / build (arm64, freebsd) (push) Successful in 1m26s
Release / build (arm64, linux) (push) Successful in 1m25s
Release / build (arm64, netbsd) (push) Successful in 1m31s
Release / build (arm64, openbsd) (push) Successful in 1m27s
Release / build (loong64, linux) (push) Successful in 1m37s
Release / build (riscv64, linux) (push) Successful in 1m21s
Release / release (push) Successful in 40s
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -0,0 +1,112 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"runtime"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// The reflink ioctl of linux/fs.h, FICLONERANGE, and the syscall number
|
||||
// of copy_file_range(2), which the standard library does not export on
|
||||
// linux. FICLONERANGE is _IOW(0x94, 13, struct file_clone_range) with a
|
||||
// 32 byte struct: (1<<30)|(32<<16)|(0x94<<8)|13. The copy_file_range
|
||||
// numbers follow the kernel's syscall tables: 319 on amd64 and 286 on
|
||||
// the asm generic table of arm64, riscv64 and loong64. An architecture
|
||||
// outside the table answers unshareable, and the caller keeps its
|
||||
// userspace path.
|
||||
const ioctlFICLONERANGE = 0x4020940D
|
||||
|
||||
// fileCloneRange mirrors struct file_clone_range of linux/fs.h, the
|
||||
// argument of FICLONERANGE.
|
||||
type fileCloneRange struct {
|
||||
srcFD int64
|
||||
srcOffset uint64
|
||||
srcLength uint64
|
||||
destOffset uint64
|
||||
}
|
||||
|
||||
// CloneRange makes the destination carry the source's bytes through the
|
||||
// reflink of the filesystem, XFS, btrfs and ZFS among them. The
|
||||
// descriptor cache hands out both ends, so a range cloned through cached
|
||||
// descriptors stays as verified as any other read or write.
|
||||
func (l *Local) CloneRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error {
|
||||
sf, releaseSrc, err := l.dataFD(src, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer releaseSrc()
|
||||
df, releaseDst, err := l.dataFD(dst, true)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer releaseDst()
|
||||
cr := fileCloneRange{
|
||||
srcFD: int64(sf.Fd()),
|
||||
srcOffset: uint64(srcOff),
|
||||
srcLength: uint64(length),
|
||||
destOffset: uint64(dstOff),
|
||||
}
|
||||
// SAFETY: ioctl takes the address of exactly the 32 byte struct the
|
||||
// FICLONERANGE command name carries; the kernel reads it and writes
|
||||
// nothing through it.
|
||||
if _, _, errno := syscall.Syscall(syscall.SYS_IOCTL, df.Fd(), ioctlFICLONERANGE,
|
||||
uintptr(unsafe.Pointer(&cr))); errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// CopyRange copies the bytes through copy_file_range(2), in chunks until
|
||||
// the length is served. The syscall answers how much moved; a short move
|
||||
// on the first call means the kernel refused for these files and the
|
||||
// error travels to the caller's fallback.
|
||||
func (l *Local) CopyRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error {
|
||||
sf, releaseSrc, err := l.dataFD(src, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer releaseSrc()
|
||||
df, releaseDst, err := l.dataFD(dst, true)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer releaseDst()
|
||||
number := copyFileRangeSyscall()
|
||||
if number == 0 {
|
||||
return syscall.ENOSYS
|
||||
}
|
||||
var inOff, outOff int64 = srcOff, dstOff
|
||||
for length > 0 {
|
||||
chunk := min(length, 8<<20)
|
||||
// SAFETY: the syscall copies from and to the file offsets behind
|
||||
// the two pointers and updates them; both live across the call.
|
||||
n, _, errno := syscall.Syscall6(number, sf.Fd(), uintptr(unsafe.Pointer(&inOff)),
|
||||
df.Fd(), uintptr(unsafe.Pointer(&outOff)), uintptr(chunk), 0)
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
if n == 0 {
|
||||
return syscall.EINVAL
|
||||
}
|
||||
length -= int64(n)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// copyFileRangeSyscall answers the syscall number of copy_file_range on
|
||||
// the architectures this project builds for linux, zero elsewhere.
|
||||
func copyFileRangeSyscall() uintptr {
|
||||
switch runtime.GOARCH {
|
||||
case "amd64":
|
||||
return 319
|
||||
case "arm64", "riscv64", "loong64":
|
||||
return 286
|
||||
default:
|
||||
return 0
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"syscall"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// cloneTestFS builds a backend over a fresh directory holding a source
|
||||
// file of the given content.
|
||||
func cloneTestFS(t *testing.T, content []byte) (*Local, Handle, Handle, string) {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(root, "src"), content, 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(root, "dst"), make([]byte, len(content)), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatalf("Root: %v", err)
|
||||
}
|
||||
src, _, err := l.Lookup(rh, "src")
|
||||
if err != nil {
|
||||
t.Fatalf("Lookup src: %v", err)
|
||||
}
|
||||
dst, _, err := l.Lookup(rh, "dst")
|
||||
if err != nil {
|
||||
t.Fatalf("Lookup dst: %v", err)
|
||||
}
|
||||
return l, src, dst, filepath.Join(root, "dst")
|
||||
}
|
||||
|
||||
// TestCopyRangeKernel drives the kernel copy and verifies the bytes
|
||||
// landed. Filesystems that refuse the syscall for their files skip the
|
||||
// test: the handler falls back to its userspace path either way.
|
||||
func TestCopyRangeKernel(t *testing.T) {
|
||||
content := make([]byte, 1<<20)
|
||||
for i := range content {
|
||||
content[i] = byte(i * 3)
|
||||
}
|
||||
l, src, dst, dstPath := cloneTestFS(t, content)
|
||||
if err := l.CopyRange(src, 0, dst, 0, int64(len(content))); err != nil {
|
||||
if errors.Is(err, syscall.ENOSYS) || errors.Is(err, syscall.EOPNOTSUPP) ||
|
||||
errors.Is(err, syscall.EXDEV) || errors.Is(err, syscall.EINVAL) {
|
||||
t.Skipf("the filesystem refuses copy_file_range: %v", err)
|
||||
}
|
||||
t.Fatalf("CopyRange: %v", err)
|
||||
}
|
||||
got, err := os.ReadFile(dstPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile: %v", err)
|
||||
}
|
||||
if !bytes.Equal(got, content) {
|
||||
t.Fatalf("the copied bytes differ: %d vs %d", len(got), len(content))
|
||||
}
|
||||
}
|
||||
|
||||
// TestCloneRangeKernel drives the reflink where the filesystem provides
|
||||
// one, and verifies the clone reads back as the source.
|
||||
func TestCloneRangeKernel(t *testing.T) {
|
||||
content := make([]byte, 1<<20)
|
||||
for i := range content {
|
||||
content[i] = byte(i * 7)
|
||||
}
|
||||
l, src, dst, dstPath := cloneTestFS(t, content)
|
||||
err := l.CloneRange(src, 0, dst, 0, int64(len(content)))
|
||||
if err != nil {
|
||||
if errors.Is(err, syscall.EOPNOTSUPP) || errors.Is(err, syscall.EXDEV) ||
|
||||
errors.Is(err, syscall.EINVAL) || errors.Is(err, syscall.EBADF) {
|
||||
t.Skipf("the filesystem refuses FICLONERANGE: %v", err)
|
||||
}
|
||||
t.Fatalf("CloneRange: %v", err)
|
||||
}
|
||||
got, err := os.ReadFile(dstPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile: %v", err)
|
||||
}
|
||||
if !bytes.Equal(got, content) {
|
||||
t.Fatalf("the cloned bytes differ: %d vs %d", len(got), len(content))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"syscall"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The create and write error wrappers map the errno families the
|
||||
// protocol knows: a missing parent is NoEnt, a permission problem is
|
||||
// Permission, a collision is Exist, and the rest stay raw.
|
||||
func TestWrapErrFamilies(t *testing.T) {
|
||||
// The wrapper level: each mapped errno.
|
||||
cases := []struct {
|
||||
err error
|
||||
want error
|
||||
}{
|
||||
{nil, nil},
|
||||
{&os.PathError{Op: "open", Err: syscall.EACCES}, ErrPermission},
|
||||
{&os.PathError{Op: "open", Err: syscall.EPERM}, ErrPermission},
|
||||
{&os.PathError{Op: "open", Err: syscall.EEXIST}, ErrExist},
|
||||
{&os.PathError{Op: "open", Err: syscall.EISDIR}, ErrIsDir},
|
||||
{&os.PathError{Op: "open", Err: syscall.ENOTDIR}, ErrNotDir},
|
||||
{&os.PathError{Op: "write", Err: syscall.ENOSPC}, ErrNoSpace},
|
||||
{&os.PathError{Op: "write", Err: syscall.EIO}, ErrIO},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := wrapCreateErr(c.err); !errors.Is(got, c.want) {
|
||||
t.Fatalf("create wrap of %v: %v, want %v", c.err, got, c.want)
|
||||
}
|
||||
}
|
||||
// The write wrapper answers a vanished path with stale, not noent.
|
||||
if got := wrapWriteErr(&os.PathError{Op: "open", Err: syscall.ENOENT}); !errors.Is(got, ErrStale) {
|
||||
t.Fatalf("write wrap of a vanished path: %v, want stale", got)
|
||||
}
|
||||
// An errno outside the families lands in the io error family with
|
||||
// its cause kept in the text.
|
||||
if got := wrapCreateErr(&os.PathError{Op: "open", Err: syscall.EDQUOT}); !errors.Is(got, ErrIO) {
|
||||
t.Fatalf("unmapped errno: %v, want the io error family", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A hard link to a missing target and one under a file both answer
|
||||
// their own sentinels.
|
||||
func TestLinkErrors(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, _, err := l.Link(nfsfsMissingHandle(), rh, "x"); err == nil {
|
||||
t.Fatal("a link to a missing target succeeded")
|
||||
}
|
||||
target := filepath.Join(root, "t.txt")
|
||||
if err := os.WriteFile(target, []byte("t"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
th, _, err := l.Lookup(rh, "t.txt")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, _, err := l.Link(th, nfsfsMissingHandle(), "x"); err == nil {
|
||||
t.Fatal("a link under a missing directory succeeded")
|
||||
}
|
||||
}
|
||||
|
||||
// nfsfsMissingHandle builds a handle that resolves to nothing.
|
||||
func nfsfsMissingHandle() Handle {
|
||||
return Handle("nfs\x02\x00\x00\x00stale-handle-bytes")
|
||||
}
|
||||
@@ -0,0 +1,244 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"os"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// fdCacheLimit bounds the descriptor cache. The cache holds at most this
|
||||
// many idle descriptors across both access classes; descriptors checked
|
||||
// out by running operations sit above the bound for their lifetime. The
|
||||
// bound keeps a busy server under the process file descriptor ceiling:
|
||||
// without it, one descriptor per file ever touched would grow without end.
|
||||
const fdCacheLimit = 512
|
||||
|
||||
// An fdKey identifies one cached descriptor: the path it was opened
|
||||
// through and the access class. The path is part of the key, not just the
|
||||
// device and inode, so a file removed and recreated under a recycled inode
|
||||
// number never inherits the old descriptor: the new object resolves to a
|
||||
// fresh open, and the retired one fails its next identity check.
|
||||
type fdKey struct {
|
||||
path string
|
||||
wr bool
|
||||
}
|
||||
|
||||
// An fdEntry is one cached descriptor. refs counts the operations holding
|
||||
// it right now; the entry leaves the cache, and its descriptor closes,
|
||||
// when it is retired or evicted and the last reference lets go.
|
||||
type fdEntry struct {
|
||||
f *os.File
|
||||
refs int
|
||||
dead bool
|
||||
use uint64
|
||||
}
|
||||
|
||||
// dataFD hands out an open descriptor for the regular file the handle
|
||||
// names: from the cache when one is held, from a fresh verified open
|
||||
// otherwise. The identity is reverified on every use, a cache hit
|
||||
// included, twice over: the registered path must still Lstat to the
|
||||
// device, inode and kind the handle encodes, and the descriptor itself
|
||||
// must still carry that identity with a link count above zero. A file
|
||||
// removed while a descriptor of it sits in the cache therefore answers
|
||||
// stale exactly as it does without the cache, and a name swapped for a
|
||||
// symlink is never served through the cached descriptor.
|
||||
//
|
||||
// The returned release function must be called: it returns the descriptor
|
||||
// to the cache, or closes it when the descriptor was retired, evicted or
|
||||
// excluded from the cache while the lease was out.
|
||||
func (l *Local) dataFD(h Handle, wr bool) (*os.File, func(), error) {
|
||||
kind, dev, ino, path, err := l.resolve(h)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if kind != typeFile {
|
||||
return nil, nil, ErrIsDir
|
||||
}
|
||||
key := fdKey{path: path, wr: wr}
|
||||
if cached := l.fdCheckout(key); cached != nil {
|
||||
// The descriptor is reverified against the path and against its
|
||||
// own stat: a link count of zero means the cached descriptor is
|
||||
// holding an unlinked inode, whatever the path names now.
|
||||
fi, err := os.Lstat(path)
|
||||
fst, ferr := cached.Stat()
|
||||
switch {
|
||||
case err == nil && ferr == nil && sameFile(fi, kind, dev, ino) &&
|
||||
sameFile(fst, kind, dev, ino) && nlinkOf(fst) > 0:
|
||||
return cached, func() { l.fdRelease(key, cached) }, nil
|
||||
case err == nil:
|
||||
// The path no longer names the inode, or the descriptor
|
||||
// serves an unlinked one: retire and answer stale.
|
||||
l.fdRetire(key)
|
||||
return nil, nil, ErrStale
|
||||
default:
|
||||
l.fdRetire(key)
|
||||
return nil, nil, revalidateStatErr(err)
|
||||
}
|
||||
}
|
||||
flag := os.O_RDONLY
|
||||
if wr {
|
||||
flag = os.O_WRONLY
|
||||
}
|
||||
f, _, err := l.openVerified(h, flag)
|
||||
if err != nil {
|
||||
if em := l.fdRelieve(err); em {
|
||||
// The open starved on descriptors; the cache gave its idle
|
||||
// ones up. One retry is entitled to succeed now.
|
||||
f, _, err = l.openVerified(h, flag)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
l.fdInsert(key, f)
|
||||
return f, func() { l.fdRelease(key, f) }, nil
|
||||
}
|
||||
|
||||
// nlinkOf reports the link count a stat carried, zero when the platform
|
||||
// data is missing.
|
||||
func nlinkOf(fi os.FileInfo) uint64 {
|
||||
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
||||
return uint64(st.Nlink)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// revalidateStatErr maps the errors of the revalidating Lstat, the same
|
||||
// mapping revalidate applies.
|
||||
func revalidateStatErr(err error) error {
|
||||
switch {
|
||||
case errors.Is(err, fs.ErrNotExist), errors.Is(err, syscall.ENOTDIR):
|
||||
return ErrStale
|
||||
case errors.Is(err, fs.ErrPermission):
|
||||
return ErrPermission
|
||||
default:
|
||||
return fmt.Errorf("%w: %v", ErrIO, err)
|
||||
}
|
||||
}
|
||||
|
||||
// fdCheckout hands the cached descriptor of key out to one operation and
|
||||
// marks the entry busy, or reports nil when nothing usable is cached.
|
||||
func (l *Local) fdCheckout(key fdKey) *os.File {
|
||||
l.fdMu.Lock()
|
||||
defer l.fdMu.Unlock()
|
||||
e := l.fds[key]
|
||||
if e == nil || e.dead {
|
||||
return nil
|
||||
}
|
||||
e.refs++
|
||||
l.fdUse++
|
||||
e.use = l.fdUse
|
||||
return e.f
|
||||
}
|
||||
|
||||
// fdInsert admits a freshly opened, already verified descriptor into the
|
||||
// cache with one reference held. When a retired entry under the same key
|
||||
// is still draining its outstanding leases, the new descriptor bypasses
|
||||
// the cache and closes on release instead.
|
||||
func (l *Local) fdInsert(key fdKey, f *os.File) {
|
||||
l.fdMu.Lock()
|
||||
if old := l.fds[key]; old != nil {
|
||||
// Only a drained entry leaves the map, so anything here is a
|
||||
// retired one waiting for its leases; this descriptor stays out.
|
||||
l.fdMu.Unlock()
|
||||
return
|
||||
}
|
||||
l.fds[key] = &fdEntry{f: f, refs: 1, use: l.fdUse + 1}
|
||||
l.fdUse++
|
||||
closed := l.fdEvictLocked()
|
||||
l.fdMu.Unlock()
|
||||
for _, idle := range closed {
|
||||
idle.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// fdRelease ends one lease. A live entry takes the descriptor back; a
|
||||
// retired, evicted or bypassed one closes it, at the last release.
|
||||
func (l *Local) fdRelease(key fdKey, f *os.File) {
|
||||
l.fdMu.Lock()
|
||||
e := l.fds[key]
|
||||
if e == nil || e.f != f {
|
||||
l.fdMu.Unlock()
|
||||
f.Close()
|
||||
return
|
||||
}
|
||||
e.refs--
|
||||
if e.refs > 0 {
|
||||
l.fdMu.Unlock()
|
||||
return
|
||||
}
|
||||
if e.dead {
|
||||
delete(l.fds, key)
|
||||
l.fdMu.Unlock()
|
||||
f.Close()
|
||||
return
|
||||
}
|
||||
l.fdMu.Unlock()
|
||||
}
|
||||
|
||||
// fdRetire marks the cached descriptor of key dead: it is never handed
|
||||
// out again, and it closes when its outstanding leases release.
|
||||
func (l *Local) fdRetire(key fdKey) {
|
||||
l.fdMu.Lock()
|
||||
if e := l.fds[key]; e != nil {
|
||||
e.dead = true
|
||||
}
|
||||
l.fdMu.Unlock()
|
||||
}
|
||||
|
||||
// fdEvictLocked picks idle descriptors until the cache fits the bound and
|
||||
// returns them for the caller to close outside the lock. Entries with
|
||||
// outstanding leases are untouchable; a cache full of busy entries
|
||||
// temporarily exceeds the bound by exactly the number of running
|
||||
// operations.
|
||||
func (l *Local) fdEvictLocked() []*os.File {
|
||||
var closed []*os.File
|
||||
for len(l.fds) > fdCacheLimit {
|
||||
var victim *fdEntry
|
||||
var victimKey fdKey
|
||||
for key, e := range l.fds {
|
||||
if e.dead || e.refs > 0 {
|
||||
continue
|
||||
}
|
||||
if victim == nil || e.use < victim.use {
|
||||
victim, victimKey = e, key
|
||||
}
|
||||
}
|
||||
if victim == nil {
|
||||
return closed
|
||||
}
|
||||
delete(l.fds, victimKey)
|
||||
closed = append(closed, victim.f)
|
||||
}
|
||||
return closed
|
||||
}
|
||||
|
||||
// fdRelieve answers whether err is an open refused for descriptor
|
||||
// exhaustion and, when it is, retires every idle descriptor so one retry
|
||||
// can run. The cache must never be the reason a server runs out of file
|
||||
// descriptors.
|
||||
func (l *Local) fdRelieve(err error) bool {
|
||||
if !errors.Is(err, syscall.EMFILE) && !errors.Is(err, syscall.ENFILE) {
|
||||
return false
|
||||
}
|
||||
l.fdMu.Lock()
|
||||
var closed []*os.File
|
||||
for key, e := range l.fds {
|
||||
if e.refs > 0 {
|
||||
e.dead = true
|
||||
continue
|
||||
}
|
||||
delete(l.fds, key)
|
||||
closed = append(closed, e.f)
|
||||
}
|
||||
l.fdMu.Unlock()
|
||||
for _, f := range closed {
|
||||
f.Close()
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -0,0 +1,263 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Package nfsfs defines the virtual filesystem the NFS server serves, and
|
||||
// provides a backend over a local directory.
|
||||
//
|
||||
// The interface carries exactly what the protocol layer needs and nothing
|
||||
// more: handles that the backend itself interprets, the attributes each
|
||||
// GETATTR turns into an fattr4, the data operations LOOKUP, READDIR and
|
||||
// READ, and the Writer half that WRITE, CREATE and the other state
|
||||
// changing operations reach.
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io/fs"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A Handle is an opaque file handle. The backend defines its layout; the
|
||||
// protocol layer treats it as bytes.
|
||||
type Handle []byte
|
||||
|
||||
// Sentinels the dispatcher maps onto NFS4ERR statuses. Use errors.Is.
|
||||
var (
|
||||
ErrStale = errors.New("nfsfs: unknown file handle")
|
||||
ErrNoEnt = errors.New("nfsfs: no such file or directory")
|
||||
ErrNotDir = errors.New("nfsfs: not a directory")
|
||||
ErrIsDir = errors.New("nfsfs: is a directory")
|
||||
ErrNameTooLong = errors.New("nfsfs: name too long")
|
||||
ErrBadName = errors.New("nfsfs: invalid name")
|
||||
ErrPermission = errors.New("nfsfs: permission denied")
|
||||
ErrIO = errors.New("nfsfs: io error")
|
||||
ErrExist = errors.New("nfsfs: file exists")
|
||||
ErrNoSpace = errors.New("nfsfs: no space left")
|
||||
ErrNotEmpty = errors.New("nfsfs: directory not empty")
|
||||
ErrInval = errors.New("nfsfs: invalid argument")
|
||||
ErrNotLnk = errors.New("nfsfs: not a symlink")
|
||||
)
|
||||
|
||||
// Access mask bits, RFC 8881 section 15.2.2. The same values the protocol
|
||||
// layer speaks.
|
||||
const (
|
||||
AccessRead = 1 << 0
|
||||
AccessLookup = 1 << 1
|
||||
AccessModify = 1 << 2
|
||||
AccessExtend = 1 << 3
|
||||
AccessDelete = 1 << 4
|
||||
AccessExec = 1 << 5
|
||||
)
|
||||
|
||||
// MaxName is the name length limit the backends enforce.
|
||||
const MaxName = 255
|
||||
|
||||
// An Info carries the file attributes the backends report.
|
||||
type Info struct {
|
||||
Size int64
|
||||
Mode fs.FileMode
|
||||
ModTime time.Time
|
||||
Dev uint64
|
||||
Ino uint64
|
||||
Nlink uint64
|
||||
UID uint32
|
||||
GID uint32
|
||||
}
|
||||
|
||||
// IsDir reports whether the file is a directory.
|
||||
func (i Info) IsDir() bool { return i.Mode.IsDir() }
|
||||
|
||||
// An Entry is one READDIR row: the cookie the client resumes from, the
|
||||
// name, its handle and its attributes.
|
||||
type Entry struct {
|
||||
Cookie uint64
|
||||
Name string
|
||||
Handle Handle
|
||||
Info Info
|
||||
}
|
||||
|
||||
// A DirPage is one READDIR result page: the entries the cookie asked for,
|
||||
// the verifier of the directory order, and whether the listing is complete.
|
||||
type DirPage struct {
|
||||
Entries []Entry
|
||||
Verifier [8]byte
|
||||
EOF bool
|
||||
}
|
||||
|
||||
// An FS is the virtual filesystem the server serves. Implementations must
|
||||
// be safe for concurrent use.
|
||||
type FS interface {
|
||||
// Root returns the handle of the export root.
|
||||
Root() (Handle, error)
|
||||
// Lookup resolves name under the parent handle.
|
||||
Lookup(parent Handle, name string) (Handle, Info, error)
|
||||
// Getattr reports the attributes of a handle.
|
||||
Getattr(h Handle) (Info, error)
|
||||
// ReadLink reports the target of a symlink. A handle that names
|
||||
// anything else is an error.
|
||||
ReadLink(h Handle) (string, error)
|
||||
// Parent resolves the directory that holds h and the component name
|
||||
// of h under it. The root of the export has no parent name.
|
||||
Parent(h Handle) (Handle, string, error)
|
||||
// ReadDir lists the directory from the given cookie, returning at most
|
||||
// count entries, or all of them when count is zero or less.
|
||||
ReadDir(h Handle, cookie uint64, count int) (DirPage, error)
|
||||
// Read reads up to count bytes at the offset. A short result means end
|
||||
// of file was reached.
|
||||
Read(h Handle, off int64, count int) ([]byte, error)
|
||||
// Access evaluates the requested mask bits for the credential and
|
||||
// returns the bits granted.
|
||||
Access(h Handle, mask uint32, uid, gid uint32, groups []uint32) (uint32, error)
|
||||
}
|
||||
|
||||
// ValidName reports whether name can appear in a LOOKUP. Names carrying a
|
||||
// separator or a control byte never belong to the client, because no
|
||||
// backend resolves them.
|
||||
func ValidName(name string) error {
|
||||
switch {
|
||||
case name == "":
|
||||
return ErrBadName
|
||||
case name == "." || name == "..":
|
||||
return ErrBadName
|
||||
case len(name) > MaxName:
|
||||
return ErrNameTooLong
|
||||
}
|
||||
for i := range len(name) {
|
||||
if name[i] == '/' || name[i] == 0 {
|
||||
return ErrBadName
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// The object kinds a CREATE may carry, the same values the protocol's
|
||||
// createtype4 uses. A regular file is not among them: in NFSv4 regular
|
||||
// files are created by OPEN.
|
||||
const (
|
||||
KindDir = 2
|
||||
KindLnk = 5
|
||||
KindSock = 6
|
||||
KindFifo = 7
|
||||
KindBlk = 3
|
||||
KindChr = 4
|
||||
)
|
||||
|
||||
// An Owner names the unix owner and group an object carries after its
|
||||
// creation. The server runs under its own identity, so a backend that
|
||||
// serves clients of several owners applies these values when a client
|
||||
// makes a new object.
|
||||
type Owner struct {
|
||||
UID uint32
|
||||
GID uint32
|
||||
}
|
||||
|
||||
// A CreateSpec describes one object a CREATE makes.
|
||||
type CreateSpec struct {
|
||||
Kind uint32
|
||||
Perm fs.FileMode // permission bits, applied exactly, umask aside
|
||||
LinkData string // the target of a symlink
|
||||
Major uint32 // device numbers of a character or block device
|
||||
Minor uint32
|
||||
Owner Owner // the owner a fresh object carries
|
||||
}
|
||||
|
||||
// A Writer is the mutating half of an FS. A backend that serves reads only
|
||||
// does not implement it, and the dispatcher answers NFS4ERR_ROFS.
|
||||
type Writer interface {
|
||||
// Create makes the object the spec describes under the parent handle.
|
||||
// An existing target is an error for every kind.
|
||||
Create(parent Handle, name string, spec CreateSpec) (Handle, Info, error)
|
||||
// Write writes all of data at the offset of a regular file and
|
||||
// returns how many bytes landed.
|
||||
Write(h Handle, off int64, data []byte) (int, error)
|
||||
// Remove takes the named entry out of the directory. Removing a
|
||||
// directory that is not empty is an error.
|
||||
Remove(dir Handle, name string) error
|
||||
// Rename moves oldName from the oldDir directory to newName in the
|
||||
// newDir directory, replacing an existing plain target the way POSIX
|
||||
// rename does. The handles of the moved object and of its descendants
|
||||
// keep working after the move.
|
||||
Rename(oldDir Handle, oldName string, newDir Handle, newName string) error
|
||||
// Setattr applies the named changes to a file. Changes that the
|
||||
// backend cannot apply make the whole call fail.
|
||||
Setattr(h Handle, s SetAttrs) error
|
||||
// Link makes newName in dir a hard link to the target file.
|
||||
Link(target Handle, dir Handle, name string) (Handle, Info, error)
|
||||
// Sync flushes the file's dirty data to stable storage.
|
||||
Sync(h Handle) error
|
||||
// Open opens the regular file name under dir for writing. When create
|
||||
// is set a missing file is made with perm and carried by owner; when
|
||||
// guarded is set an existing name answers ErrExist instead of
|
||||
// opening, which the GUARDED and EXCLUSIVE4_1 create modes require;
|
||||
// when truncate is set an existing file is cut to zero first. The
|
||||
// boolean reports whether the file was created by this call.
|
||||
Open(dir Handle, name string, create, guarded, truncate bool, perm fs.FileMode, owner Owner) (h Handle, info Info, created bool, err error)
|
||||
}
|
||||
|
||||
// A TimeSet is one time attribute of a SETATTR: either the server's
|
||||
// current time or the time the client names.
|
||||
type TimeSet struct {
|
||||
Now bool
|
||||
Time time.Time
|
||||
}
|
||||
|
||||
// SetAttrs carries the changes a SETATTR names. A nil field is a change
|
||||
// the client did not ask for.
|
||||
type SetAttrs struct {
|
||||
Mode *uint32
|
||||
Size *int64
|
||||
UID *uint32
|
||||
GID *uint32
|
||||
Atime *TimeSet
|
||||
Mtime *TimeSet
|
||||
}
|
||||
|
||||
// XattrFS is the optional extended attribute half of a backend. A backend
|
||||
// that does not implement it answers NOT_SUPP to the xattr family.
|
||||
type XattrFS interface {
|
||||
// GetXattr reads one named attribute of the object.
|
||||
GetXattr(h Handle, name string, max int) ([]byte, error)
|
||||
// SetXattr writes one named attribute; the mode follows the
|
||||
// SETXATTR4mode4 enum of RFC 8276.
|
||||
SetXattr(h Handle, name string, value []byte, mode uint32) error
|
||||
// ListXattr names the attributes of the object.
|
||||
ListXattr(h Handle, max int) ([]string, error)
|
||||
// RemoveXattr deletes one named attribute.
|
||||
RemoveXattr(h Handle, name string) error
|
||||
}
|
||||
|
||||
// ErrNoXattr marks a missing attribute, NFS4ERR_NOXATTR on the wire.
|
||||
var ErrNoXattr = errors.New("nfsfs: no such attribute")
|
||||
|
||||
// ErrXattrNotSupp marks a backend that carries no extended attributes at
|
||||
// all, NFS4ERR_NOT_SUPP on the wire.
|
||||
var ErrXattrNotSupp = errors.New("nfsfs: extended attributes are not supported")
|
||||
|
||||
// ErrBeyondEOF marks a SEEK that starts past the end of the file,
|
||||
// NFS4ERR_NXIO on the wire per RFC 7862 section 15.11.
|
||||
var ErrBeyondEOF = errors.New("nfsfs: seek past the end")
|
||||
|
||||
// ErrNoSparse marks a backend whose platform carries no space
|
||||
// reservation or hole seeking calls, NFS4ERR_NOT_SUPP on the wire.
|
||||
var ErrNoSparse = errors.New("nfsfs: sparse file operations are not supported")
|
||||
|
||||
// The SETXATTR create modes of RFC 8276, mirrored from the wire
|
||||
// protocol. They are platform independent: a backend that carries no
|
||||
// extended attributes answers NOT_SUPP regardless of the mode.
|
||||
const (
|
||||
XattrModeCreate = 1
|
||||
XattrModeReplace = 2
|
||||
)
|
||||
|
||||
// A RangeCloner is the optional half that clones or copies a byte range
|
||||
// between two regular files inside the kernel: the bytes never travel
|
||||
// through userspace. A backend that does not carry it, or a filesystem
|
||||
// that refuses the call for one pair of files, leaves the caller its
|
||||
// userspace path.
|
||||
type RangeCloner interface {
|
||||
// CloneRange makes dstOff carry the same bytes as srcOff through a
|
||||
// reflink where the filesystem provides one.
|
||||
CloneRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error
|
||||
// CopyRange copies length bytes through the kernel's copy syscall.
|
||||
CopyRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,168 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
const benchChunk = 64 << 10
|
||||
|
||||
// benchRoot builds a backend over a fresh directory with one file of the
|
||||
// given size, filled with a repeating pattern, and returns the backend
|
||||
// and the file handle. Setup runs once, outside the measured region.
|
||||
func benchRoot(b *testing.B, size int) (*Local, Handle) {
|
||||
b.Helper()
|
||||
root := b.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
b.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
p := filepath.Join(root, "file")
|
||||
buf := make([]byte, 1<<20)
|
||||
for i := range buf {
|
||||
buf[i] = byte(i)
|
||||
}
|
||||
f, err := os.Create(p)
|
||||
if err != nil {
|
||||
b.Fatalf("Create: %v", err)
|
||||
}
|
||||
for written := 0; written < size; written += len(buf) {
|
||||
if _, err := f.Write(buf); err != nil {
|
||||
b.Fatalf("Write: %v", err)
|
||||
}
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
b.Fatalf("Close: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
b.Fatalf("Root: %v", err)
|
||||
}
|
||||
h, _, err := l.Lookup(rh, "file")
|
||||
if err != nil {
|
||||
b.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
return l, h
|
||||
}
|
||||
|
||||
// BenchmarkRead64K reads 64 KiB at a time from a 64 MiB file, cycling
|
||||
// through the offsets so every read touches pages the previous read left.
|
||||
func BenchmarkRead64K(b *testing.B) {
|
||||
l, h := benchRoot(b, 64<<20)
|
||||
off := int64(0)
|
||||
b.SetBytes(benchChunk)
|
||||
b.ResetTimer()
|
||||
for b.Loop() {
|
||||
if _, err := l.Read(h, off, benchChunk); err != nil {
|
||||
b.Fatalf("Read: %v", err)
|
||||
}
|
||||
off += benchChunk
|
||||
if off > 64<<20-benchChunk {
|
||||
off = 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkWrite64K writes 64 KiB at a time over a preallocated 64 MiB
|
||||
// file, cycling through the offsets, so no read has to grow the file.
|
||||
func BenchmarkWrite64K(b *testing.B) {
|
||||
l, h := benchRoot(b, 64<<20)
|
||||
buf := make([]byte, benchChunk)
|
||||
off := int64(0)
|
||||
b.SetBytes(benchChunk)
|
||||
b.ResetTimer()
|
||||
for b.Loop() {
|
||||
if _, err := l.Write(h, off, buf); err != nil {
|
||||
b.Fatalf("Write: %v", err)
|
||||
}
|
||||
off += benchChunk
|
||||
if off > 64<<20-benchChunk {
|
||||
off = 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkGetattr reports the attributes of one file.
|
||||
func BenchmarkGetattr(b *testing.B) {
|
||||
l, h := benchRoot(b, 1<<20)
|
||||
b.ResetTimer()
|
||||
for b.Loop() {
|
||||
if _, err := l.Getattr(h); err != nil {
|
||||
b.Fatalf("Getattr: %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkLookup resolves one name under the export root of a directory
|
||||
// holding a hundred files.
|
||||
func BenchmarkLookup(b *testing.B) {
|
||||
root := b.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
b.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
b.Fatalf("Root: %v", err)
|
||||
}
|
||||
for i := range 100 {
|
||||
name := fmt.Sprintf("f%d", i)
|
||||
if err := os.WriteFile(filepath.Join(root, name), []byte("x"), 0o644); err != nil {
|
||||
b.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
}
|
||||
b.ResetTimer()
|
||||
i := 0
|
||||
for b.Loop() {
|
||||
if _, _, err := l.Lookup(rh, fmt.Sprintf("f%d", i%100)); err != nil {
|
||||
b.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
i++
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkReadDirPage64 pages a 10 000 entry directory 64 entries at a
|
||||
// time after the first page: with the listing cache the cost of a page
|
||||
// is the page, and the benchmark holds that to the measurement.
|
||||
func BenchmarkReadDirPage64(b *testing.B) {
|
||||
root := b.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
b.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
dir := filepath.Join(root, "big")
|
||||
if err := os.Mkdir(dir, 0o755); err != nil {
|
||||
b.Fatalf("Mkdir: %v", err)
|
||||
}
|
||||
for i := range 10000 {
|
||||
if err := os.WriteFile(filepath.Join(dir, fmt.Sprintf("f%04d", i)), []byte("x"), 0o644); err != nil {
|
||||
b.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
b.Fatalf("Root: %v", err)
|
||||
}
|
||||
h, _, err := l.Lookup(rh, "big")
|
||||
if err != nil {
|
||||
b.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
if _, err := l.ReadDir(h, 0, 64); err != nil {
|
||||
b.Fatalf("first page: %v", err)
|
||||
}
|
||||
b.ResetTimer()
|
||||
for b.Loop() {
|
||||
page, err := l.ReadDir(h, 0, 64)
|
||||
if err != nil {
|
||||
b.Fatalf("ReadDir: %v", err)
|
||||
}
|
||||
if len(page.Entries) != 64 || page.EOF {
|
||||
b.Fatalf("page: %d entries, eof %v", len(page.Entries), page.EOF)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// openFDs counts the descriptors the process holds through /proc. The
|
||||
// count is a lower bound of live descriptors and good enough to prove a
|
||||
// cache does not leak: a thousand operations over one file must not add a
|
||||
// thousand descriptors.
|
||||
func openFDs(t *testing.T) int {
|
||||
t.Helper()
|
||||
entries, err := os.ReadDir("/proc/self/fd")
|
||||
if err != nil {
|
||||
t.Skipf("/proc/self/fd is unavailable: %v", err)
|
||||
}
|
||||
return len(entries)
|
||||
}
|
||||
|
||||
// TestCacheNoDescriptorLeak drives more operations over one file than the
|
||||
// cache can hold and requires the process descriptor count to stay flat.
|
||||
func TestCacheNoDescriptorLeak(t *testing.T) {
|
||||
l, h, _ := cacheTestRoot(t, "leak")
|
||||
before := openFDs(t)
|
||||
for range 1000 {
|
||||
if _, err := l.Read(h, 0, 4); err != nil {
|
||||
t.Fatalf("Read: %v", err)
|
||||
}
|
||||
}
|
||||
after := openFDs(t)
|
||||
if after-before > 8 {
|
||||
t.Fatalf("descriptors grew from %d to %d over 1000 reads", before, after)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheBound holds when many distinct files flow through: after
|
||||
// touching twice the bound, at most the bound of cache entries may remain.
|
||||
func TestCacheBound(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatalf("Root: %v", err)
|
||||
}
|
||||
for i := range 2 * fdCacheLimit {
|
||||
name := fmt.Sprintf("f%d", i)
|
||||
if err := os.WriteFile(filepath.Join(root, name), []byte("x"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
h, _, err := l.Lookup(rh, name)
|
||||
if err != nil {
|
||||
t.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
if _, err := l.Read(h, 0, 1); err != nil {
|
||||
t.Fatalf("Read %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
l.fdMu.Lock()
|
||||
n := len(l.fds)
|
||||
l.fdMu.Unlock()
|
||||
if n > fdCacheLimit {
|
||||
t.Fatalf("cache holds %d entries, bound is %d", n, fdCacheLimit)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// cacheTestRoot builds a backend over a fresh directory holding one file
|
||||
// and returns the backend, the file handle and the path.
|
||||
func cacheTestRoot(t *testing.T, content string) (*Local, Handle, string) {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
p := filepath.Join(root, "file")
|
||||
if err := os.WriteFile(p, []byte(content), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatalf("Root: %v", err)
|
||||
}
|
||||
h, _, err := l.Lookup(rh, "file")
|
||||
if err != nil {
|
||||
t.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
return l, h, p
|
||||
}
|
||||
|
||||
// TestCacheStaleAfterRemove covers the identity contract on a cache
|
||||
// hit: a file removed while a descriptor of it is cached answers stale,
|
||||
// exactly as it does without the cache.
|
||||
func TestCacheStaleAfterRemove(t *testing.T) {
|
||||
l, h, p := cacheTestRoot(t, "hello")
|
||||
if _, err := l.Read(h, 0, 5); err != nil {
|
||||
t.Fatalf("warm read: %v", err)
|
||||
}
|
||||
if err := os.Remove(p); err != nil {
|
||||
t.Fatalf("Remove: %v", err)
|
||||
}
|
||||
if _, err := l.Read(h, 0, 5); !errors.Is(err, ErrStale) {
|
||||
t.Fatalf("read after remove: %v, want ErrStale", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheStaleAfterReplace covers a name swapped for another inode: the
|
||||
// cached descriptor of the old inode must never serve through it.
|
||||
func TestCacheStaleAfterReplace(t *testing.T) {
|
||||
l, h, p := cacheTestRoot(t, "old")
|
||||
if _, err := l.Read(h, 0, 3); err != nil {
|
||||
t.Fatalf("warm read: %v", err)
|
||||
}
|
||||
if err := os.Remove(p); err != nil {
|
||||
t.Fatalf("Remove: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(p, []byte("new"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
if _, err := l.Read(h, 0, 3); !errors.Is(err, ErrStale) {
|
||||
t.Fatalf("read after replace: %v, want ErrStale", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheInodeReuse covers the recycled inode number at the same path:
|
||||
// a cached descriptor of the unlinked old inode must never serve the
|
||||
// identity the recycled file now carries. The map entry is forged onto
|
||||
// the new file, which is exactly the state a reuse produces.
|
||||
func TestCacheInodeReuse(t *testing.T) {
|
||||
l, h, p := cacheTestRoot(t, "stale data")
|
||||
if _, err := l.Read(h, 0, 4); err != nil {
|
||||
t.Fatalf("warm read: %v", err)
|
||||
}
|
||||
// Remove the file behind the backend's back and recreate a fresh one
|
||||
// at the same path, then point the old handle's identity at it the
|
||||
// way a recycled inode number would.
|
||||
fi, err := os.Lstat(p)
|
||||
if err != nil {
|
||||
t.Fatalf("Lstat: %v", err)
|
||||
}
|
||||
oldID := fileID{dev: stat(fi).Dev, ino: stat(fi).Ino}
|
||||
if err := os.Remove(p); err != nil {
|
||||
t.Fatalf("Remove: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(p, []byte("fresh"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
fi2, err := os.Lstat(p)
|
||||
if err != nil {
|
||||
t.Fatalf("Lstat: %v", err)
|
||||
}
|
||||
newID := fileID{dev: stat(fi2).Dev, ino: stat(fi2).Ino}
|
||||
l.mu.Lock()
|
||||
l.paths[newID] = p
|
||||
if newID == oldID {
|
||||
// The filesystem handed back the very same inode; the scenario
|
||||
// holds without forging anything.
|
||||
l.paths[oldID] = p
|
||||
} else {
|
||||
delete(l.paths, oldID)
|
||||
}
|
||||
l.mu.Unlock()
|
||||
// Whatever the inode numbers did, a fresh read must serve the fresh
|
||||
// content, never the unlinked inode's bytes.
|
||||
got, err := l.Read(h, 0, 5)
|
||||
if err != nil {
|
||||
// A stale answer is also safe: the identity broke and the cache
|
||||
// refused. Serving the old bytes is the only failure.
|
||||
if !errors.Is(err, ErrStale) {
|
||||
t.Fatalf("read after reuse: %v", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
if string(got) != "fresh" {
|
||||
t.Fatalf("read after reuse: %q, want the fresh content", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheWriteThrough covers that writes land and that a follow up read
|
||||
// of the same cached file sees them.
|
||||
func TestCacheWriteThrough(t *testing.T) {
|
||||
l, h, p := cacheTestRoot(t, "0123456789")
|
||||
if _, err := l.Read(h, 0, 10); err != nil {
|
||||
t.Fatalf("warm read: %v", err)
|
||||
}
|
||||
if n, err := l.Write(h, 2, []byte("AB")); err != nil || n != 2 {
|
||||
t.Fatalf("Write: %d, %v", n, err)
|
||||
}
|
||||
got, err := l.Read(h, 0, 10)
|
||||
if err != nil {
|
||||
t.Fatalf("Read: %v", err)
|
||||
}
|
||||
if string(got) != "01AB456789" {
|
||||
t.Fatalf("Read: %q", got)
|
||||
}
|
||||
raw, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile: %v", err)
|
||||
}
|
||||
if string(raw) != "01AB456789" {
|
||||
t.Fatalf("file on disk: %q", raw)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheConcurrent drives reads and writes of one file from many
|
||||
// goroutines; the race detector is the judge.
|
||||
func TestCacheConcurrent(t *testing.T) {
|
||||
l, h, _ := cacheTestRoot(t, "concurrent")
|
||||
var wg sync.WaitGroup
|
||||
for i := range 8 {
|
||||
wg.Go(func() {
|
||||
for range 50 {
|
||||
if _, err := l.Read(h, 0, 4); err != nil {
|
||||
t.Errorf("Read: %v", err)
|
||||
return
|
||||
}
|
||||
if i%2 == 0 {
|
||||
if _, err := l.Write(h, 0, []byte("writ")); err != nil {
|
||||
t.Errorf("Write: %v", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,29 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build darwin
|
||||
|
||||
package nfsfs
|
||||
|
||||
import "syscall"
|
||||
|
||||
// mknod creates a device node. It needs the superuser on darwin, and a
|
||||
// failure of permission is reported as such rather than as an io error.
|
||||
// The kind is picked before the single call: a second mknod over an
|
||||
// existing node fails EEXIST and leaves the wrong kind behind.
|
||||
func mknod(path string, spec CreateSpec, perm uint32) error {
|
||||
dev := makedev(spec.Major, spec.Minor)
|
||||
kind := uint32(syscall.S_IFCHR)
|
||||
if spec.Kind == KindBlk {
|
||||
kind = syscall.S_IFBLK
|
||||
}
|
||||
return syscall.Mknod(path, perm|kind, int(dev))
|
||||
}
|
||||
|
||||
// makedev assembles a device number the way the kernel expects it: the
|
||||
// encoding of makedev in bsd/sys/types.h of xnu, where the major number
|
||||
// sits at bits twenty-four through thirty-one and the minor number keeps
|
||||
// bits zero through twenty-three.
|
||||
func makedev(major, minor uint32) uint64 {
|
||||
return uint64(major&0xff)<<24 | uint64(minor&0xffffff)
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build freebsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
import "syscall"
|
||||
|
||||
// mknod creates a device node. It needs the superuser on FreeBSD, and a
|
||||
// failure of permission is reported as such rather than as an io error.
|
||||
// The kind is picked before the single call: a second mknod over an
|
||||
// existing node fails EEXIST and leaves the wrong kind behind.
|
||||
func mknod(path string, spec CreateSpec, perm uint32) error {
|
||||
dev := makedev(spec.Major, spec.Minor)
|
||||
kind := uint32(syscall.S_IFCHR)
|
||||
if spec.Kind == KindBlk {
|
||||
kind = syscall.S_IFBLK
|
||||
}
|
||||
return syscall.Mknod(path, perm|kind, dev)
|
||||
}
|
||||
|
||||
// makedev assembles a device number the way the kernel expects it: the
|
||||
// encoding of makedev in sys/sys/types.h, where the low byte of the
|
||||
// major number sits at bits eight to fifteen with the rest of it above
|
||||
// the thirty-second bit, and the minor number keeps its low byte at bit
|
||||
// zero with the byte at bits eight to fifteen lifted above the
|
||||
// thirty-second bit.
|
||||
func makedev(major, minor uint32) uint64 {
|
||||
return uint64(major&0xffffff00)<<32 | uint64(major&0xff)<<8 |
|
||||
uint64(minor&0xff00)<<24 | uint64(minor&0xffff00ff)
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import "syscall"
|
||||
|
||||
// mknod creates a device node. It needs CAP_MKNOD on Linux, and a failure
|
||||
// of permission is reported as such rather than as an io error. The kind
|
||||
// is picked before the single call: a second mknod over an existing node
|
||||
// fails EEXIST and leaves the wrong kind behind.
|
||||
func mknod(path string, spec CreateSpec, perm uint32) error {
|
||||
dev := int(makedev(spec.Major, spec.Minor))
|
||||
kind := uint32(syscall.S_IFCHR)
|
||||
if spec.Kind == KindBlk {
|
||||
kind = syscall.S_IFBLK
|
||||
}
|
||||
return syscall.Mknod(path, perm|kind, dev)
|
||||
}
|
||||
|
||||
// makedev assembles a device number the way the kernel expects it.
|
||||
func makedev(major, minor uint32) uint64 {
|
||||
return uint64(minor&0xff) | uint64(major&0xfff)<<8 |
|
||||
uint64(minor&0xfff00)<<12 | uint64(major&0xfffff000)<<32
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build netbsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
import "syscall"
|
||||
|
||||
// mknod creates a device node. It needs the superuser on NetBSD, and a
|
||||
// failure of permission is reported as such rather than as an io error.
|
||||
// The kind is picked before the single call: a second mknod over an
|
||||
// existing node fails EEXIST and leaves the wrong kind behind.
|
||||
func mknod(path string, spec CreateSpec, perm uint32) error {
|
||||
dev := makedev(spec.Major, spec.Minor)
|
||||
kind := uint32(syscall.S_IFCHR)
|
||||
if spec.Kind == KindBlk {
|
||||
kind = syscall.S_IFBLK
|
||||
}
|
||||
return syscall.Mknod(path, perm|kind, int(dev))
|
||||
}
|
||||
|
||||
// makedev assembles a device number the way the kernel expects it: the
|
||||
// encoding of makedev in sys/sys/types.h, where the major number sits at
|
||||
// bits eight to nineteen, the low byte of the minor number keeps bit zero
|
||||
// through seven and the rest of the minor number is lifted to bits twenty
|
||||
// through thirty-one.
|
||||
func makedev(major, minor uint32) uint64 {
|
||||
return uint64(major&0xfff)<<8 | uint64(minor&0xfff00)<<12 | uint64(minor&0xff)
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build openbsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
import "syscall"
|
||||
|
||||
// mknod creates a device node. It needs the superuser on OpenBSD, and a
|
||||
// failure of permission is reported as such rather than as an io error.
|
||||
// The kind is picked before the single call: a second mknod over an
|
||||
// existing node fails EEXIST and leaves the wrong kind behind.
|
||||
func mknod(path string, spec CreateSpec, perm uint32) error {
|
||||
dev := makedev(spec.Major, spec.Minor)
|
||||
kind := uint32(syscall.S_IFCHR)
|
||||
if spec.Kind == KindBlk {
|
||||
kind = syscall.S_IFBLK
|
||||
}
|
||||
return syscall.Mknod(path, perm|kind, int(dev))
|
||||
}
|
||||
|
||||
// makedev assembles a device number the way the kernel expects it: the
|
||||
// encoding of makedev in sys/sys/types.h, where the low byte of the
|
||||
// major number sits at bits eight to fifteen, the low byte of the minor
|
||||
// number keeps bit zero through seven, and the rest of the minor number
|
||||
// is lifted to bits sixteen upward.
|
||||
func makedev(major, minor uint32) uint64 {
|
||||
return uint64(major&0xff)<<8 | uint64(minor&0xff) | uint64(minor&0xffff00)<<8
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
)
|
||||
|
||||
// A ReadIntoer is the optional read half that fills a caller provided
|
||||
// buffer instead of allocating its own: the reply path of a server reads
|
||||
// straight into the buffer it is about to send. A backend that carries
|
||||
// only Read keeps its allocating behaviour.
|
||||
type ReadIntoer interface {
|
||||
// ReadInto reads up to len(buf) bytes at the offset into buf and
|
||||
// answers how many landed and whether the end of file was reached.
|
||||
ReadInto(h Handle, off int64, buf []byte) (int, bool, error)
|
||||
}
|
||||
|
||||
// ReadInto fills buf from the regular file the handle names. The
|
||||
// descriptor comes from the cache or a fresh verified open, and the
|
||||
// identity is revalidated before anything is read, exactly as Read.
|
||||
func (l *Local) ReadInto(h Handle, off int64, buf []byte) (int, bool, error) {
|
||||
f, release, err := l.dataFD(h, false)
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
defer release()
|
||||
n, rerr := f.ReadAt(buf, off)
|
||||
switch {
|
||||
case errors.Is(rerr, io.EOF):
|
||||
return n, true, nil
|
||||
case rerr != nil:
|
||||
return n, false, fmt.Errorf("%w: %v", ErrIO, rerr)
|
||||
}
|
||||
// A full buffer proves the end of file only against the size; the
|
||||
// descriptor's own stat answers it without a second path walk.
|
||||
if fst, serr := f.Stat(); serr == nil {
|
||||
return n, int64(off)+int64(n) >= fst.Size(), nil
|
||||
}
|
||||
return n, false, nil
|
||||
}
|
||||
|
||||
// ReadInto forwards to the wrapped export: a read only export reads as
|
||||
// its backend does.
|
||||
func (ro roFS) ReadInto(h Handle, off int64, buf []byte) (int, bool, error) {
|
||||
ri, ok := ro.FS.(ReadIntoer)
|
||||
if !ok {
|
||||
data, err := ro.Read(h, off, len(buf))
|
||||
return len(data), false, err
|
||||
}
|
||||
return ri.ReadInto(h, off, buf)
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
)
|
||||
|
||||
// ErrReadOnly marks an operation a read only export refuses, NFS4ERR_ROFS
|
||||
// on the wire.
|
||||
var ErrReadOnly = errors.New("nfsfs: the export is read only")
|
||||
|
||||
// A roFS wraps an export and hides its mutating halves. The Writer half
|
||||
// is gone, so the dispatcher answers NFS4ERR_ROFS for every operation
|
||||
// that would change anything through it. The optional halves keep their
|
||||
// reads and refuse their writes with ErrReadOnly, so an export served
|
||||
// read only still reports its extended attributes, its holes and its
|
||||
// named attributes, and refuses to touch them.
|
||||
type roFS struct{ FS }
|
||||
|
||||
// ReadOnly serves fs read only: the reads of every half pass through, the
|
||||
// writes of the mutating halves answer ErrReadOnly, and the mutating
|
||||
// Writer half disappears from the type.
|
||||
func ReadOnly(fs FS) FS {
|
||||
return roFS{fs}
|
||||
}
|
||||
|
||||
// GetXattr reads one named attribute of the object.
|
||||
func (ro roFS) GetXattr(h Handle, name string, max int) ([]byte, error) {
|
||||
x, ok := ro.FS.(XattrFS)
|
||||
if !ok {
|
||||
return nil, ErrXattrNotSupp
|
||||
}
|
||||
return x.GetXattr(h, name, max)
|
||||
}
|
||||
|
||||
// ListXattr names the attributes of the object.
|
||||
func (ro roFS) ListXattr(h Handle, max int) ([]string, error) {
|
||||
x, ok := ro.FS.(XattrFS)
|
||||
if !ok {
|
||||
return nil, ErrXattrNotSupp
|
||||
}
|
||||
return x.ListXattr(h, max)
|
||||
}
|
||||
|
||||
// SetXattr refuses a write on a read only export.
|
||||
func (ro roFS) SetXattr(h Handle, name string, value []byte, mode uint32) error {
|
||||
return ErrReadOnly
|
||||
}
|
||||
|
||||
// RemoveXattr refuses a write on a read only export.
|
||||
func (ro roFS) RemoveXattr(h Handle, name string) error {
|
||||
return ErrReadOnly
|
||||
}
|
||||
|
||||
// SeekHole reports the first hole at or after the offset.
|
||||
func (ro roFS) SeekHole(h Handle, offset int64) (int64, bool, error) {
|
||||
s, ok := ro.FS.(interface {
|
||||
SeekHole(Handle, int64) (int64, bool, error)
|
||||
SeekData(Handle, int64) (int64, bool, error)
|
||||
})
|
||||
if !ok {
|
||||
return 0, false, ErrNoSparse
|
||||
}
|
||||
return s.SeekHole(h, offset)
|
||||
}
|
||||
|
||||
// SeekData reports the first data byte at or after the offset.
|
||||
func (ro roFS) SeekData(h Handle, offset int64) (int64, bool, error) {
|
||||
s, ok := ro.FS.(interface {
|
||||
SeekHole(Handle, int64) (int64, bool, error)
|
||||
SeekData(Handle, int64) (int64, bool, error)
|
||||
})
|
||||
if !ok {
|
||||
return 0, false, ErrNoSparse
|
||||
}
|
||||
return s.SeekData(h, offset)
|
||||
}
|
||||
|
||||
// Allocate refuses a space reservation on a read only export.
|
||||
func (ro roFS) Allocate(h Handle, offset, length int64) error {
|
||||
return ErrReadOnly
|
||||
}
|
||||
|
||||
// Deallocate refuses a hole punch on a read only export.
|
||||
func (ro roFS) Deallocate(h Handle, offset, length int64) error {
|
||||
return ErrReadOnly
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// roTestRoot builds a backend over a fresh directory holding one file.
|
||||
func roTestRoot(t *testing.T) (*Local, Handle, string) {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatalf("NewLocal: %v", err)
|
||||
}
|
||||
p := filepath.Join(root, "file")
|
||||
if err := os.WriteFile(p, []byte("read only"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile: %v", err)
|
||||
}
|
||||
rh, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatalf("Root: %v", err)
|
||||
}
|
||||
h, _, err := l.Lookup(rh, "file")
|
||||
if err != nil {
|
||||
t.Fatalf("Lookup: %v", err)
|
||||
}
|
||||
return l, h, p
|
||||
}
|
||||
|
||||
// TestReadOnlyHidesWriter covers the contract the dispatcher relies on: a
|
||||
// read only export carries no Writer half, so every mutating operation
|
||||
// answers NFS4ERR_ROFS without the backend ever being asked.
|
||||
func TestReadOnlyHidesWriter(t *testing.T) {
|
||||
l, _, _ := roTestRoot(t)
|
||||
ro := ReadOnly(l)
|
||||
if _, ok := ro.(Writer); ok {
|
||||
t.Fatal("the read only export still carries a Writer")
|
||||
}
|
||||
var fs FS = l
|
||||
if _, ok := fs.(Writer); !ok {
|
||||
t.Fatal("the plain backend lost its Writer")
|
||||
}
|
||||
}
|
||||
|
||||
// TestReadOnlyRefusesWrites covers the optional halves: their reads pass
|
||||
// through, their writes answer ErrReadOnly.
|
||||
func TestReadOnlyRefusesWrites(t *testing.T) {
|
||||
l, h, _ := roTestRoot(t)
|
||||
ro := ReadOnly(l)
|
||||
if err := ro.(XattrFS).SetXattr(h, "user.note", []byte("x"), XattrModeCreate); !errors.Is(err, ErrReadOnly) {
|
||||
t.Fatalf("SetXattr: %v, want ErrReadOnly", err)
|
||||
}
|
||||
if err := ro.(XattrFS).RemoveXattr(h, "user.note"); !errors.Is(err, ErrReadOnly) {
|
||||
t.Fatalf("RemoveXattr: %v, want ErrReadOnly", err)
|
||||
}
|
||||
if err := ro.(interface {
|
||||
Allocate(Handle, int64, int64) error
|
||||
Deallocate(Handle, int64, int64) error
|
||||
}).Allocate(h, 0, 4096); !errors.Is(err, ErrReadOnly) {
|
||||
t.Fatalf("Allocate: %v, want ErrReadOnly", err)
|
||||
}
|
||||
if err := ro.(interface {
|
||||
Allocate(Handle, int64, int64) error
|
||||
Deallocate(Handle, int64, int64) error
|
||||
}).Deallocate(h, 0, 4096); !errors.Is(err, ErrReadOnly) {
|
||||
t.Fatalf("Deallocate: %v, want ErrReadOnly", err)
|
||||
}
|
||||
// The reads pass through: whatever the platform answers for a missing
|
||||
// attribute and a seeking question, neither is the read only refusal.
|
||||
if _, err := ro.(XattrFS).GetXattr(h, "user.note", 1024); errors.Is(err, ErrReadOnly) {
|
||||
t.Fatal("GetXattr refused as a write")
|
||||
}
|
||||
if _, _, err := ro.(interface {
|
||||
SeekHole(Handle, int64) (int64, bool, error)
|
||||
SeekData(Handle, int64) (int64, bool, error)
|
||||
}).SeekHole(h, 0); errors.Is(err, ErrReadOnly) {
|
||||
t.Fatal("SeekHole refused as a write")
|
||||
}
|
||||
}
|
||||
|
||||
// TestReadOnlyReadsWork covers that the read side of the wrapped export
|
||||
// is the export itself.
|
||||
func TestReadOnlyReadsWork(t *testing.T) {
|
||||
l, h, _ := roTestRoot(t)
|
||||
ro := ReadOnly(l)
|
||||
got, err := ro.Read(h, 0, 9)
|
||||
if err != nil {
|
||||
t.Fatalf("Read: %v", err)
|
||||
}
|
||||
if string(got) != "read only" {
|
||||
t.Fatalf("Read: %q", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build freebsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// The whence values of the hole seeking of FreeBSD, sys/sys/unistd.h.
|
||||
const (
|
||||
seekData = 3
|
||||
seekHole = 4
|
||||
)
|
||||
|
||||
// SeekHole finds the next hole at or after the offset. The eof flag
|
||||
// reports that none of the requested content follows: every file
|
||||
// carries a virtual hole at its end, so a dense tail answers the file
|
||||
// size with eof set, RFC 7862 section 15.11. ErrBeyondEOF answers a
|
||||
// request that starts past the end.
|
||||
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
|
||||
return l.seek(h, offset, seekHole)
|
||||
}
|
||||
|
||||
// SeekData finds the next data byte at or after the offset. The eof
|
||||
// flag reports that no data follows the offset, and the answer names
|
||||
// the file size; ErrBeyondEOF answers a request that starts past the
|
||||
// end.
|
||||
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
|
||||
return l.seek(h, offset, seekData)
|
||||
}
|
||||
|
||||
func (l *Local) seek(h Handle, offset int64, whence int) (int64, bool, error) {
|
||||
// The seek opens the same revalidated, unfollowed descriptor every
|
||||
// other data path uses, so a name swapped for a link never serves
|
||||
// bytes from outside the export.
|
||||
f, _, err := l.openVerified(h, syscall.O_RDONLY)
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
defer f.Close()
|
||||
fd := int(f.Fd())
|
||||
size, err := fstatSize(fd)
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
if offset > size {
|
||||
return 0, false, ErrBeyondEOF
|
||||
}
|
||||
at, err := syscall.Seek(fd, offset, whence)
|
||||
if err == syscall.ENXIO {
|
||||
// The range from the offset to the end carries none of the
|
||||
// requested content: the answer is the end of the file with the
|
||||
// eof flag set.
|
||||
return size, true, nil
|
||||
}
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
if at >= size {
|
||||
// The virtual hole at the end of every file: the eof flag tells
|
||||
// the client the search is over.
|
||||
return size, true, nil
|
||||
}
|
||||
return at, false, nil
|
||||
}
|
||||
|
||||
// Allocate reserves a range with posix_fallocate, the reservation call
|
||||
// of FreeBSD: it grows the file to offset plus length where the range
|
||||
// runs past the end, the behaviour RFC 7862 section 15.1 requires of
|
||||
// ALLOCATE. The handle is opened without following a final symlink
|
||||
// and revalidated against the descriptor before anything is reserved.
|
||||
func (l *Local) Allocate(h Handle, offset, length int64) error {
|
||||
kind, _, _, _, err := l.resolve(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if kind != typeFile {
|
||||
return ErrIsDir
|
||||
}
|
||||
f, _, err := l.openVerified(h, syscall.O_WRONLY)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
// posix_fallocate reports its failure as its return value, an
|
||||
// errno, and never through the system call errno itself.
|
||||
ret, _, _ := syscall.Syscall6(syscall.SYS_POSIX_FALLOCATE, f.Fd(),
|
||||
uintptr(offset), uintptr(length), 0, 0, 0)
|
||||
if ret != 0 {
|
||||
return syscall.Errno(ret)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Deallocate answers ErrNoSparse: the system call surface of FreeBSD
|
||||
// carries no call that punches a hole into a range and keeps the file
|
||||
// size, unlike the fallocate modes of Linux.
|
||||
func (l *Local) Deallocate(h Handle, offset, length int64) error {
|
||||
return ErrNoSparse
|
||||
}
|
||||
|
||||
// fstatSize reads the size of the open file.
|
||||
func fstatSize(fd int) (int64, error) {
|
||||
var st syscall.Stat_t
|
||||
if err := syscall.Fstat(fd, &st); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return st.Size, nil
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// The whence values of the hole seeking of Linux.
|
||||
const (
|
||||
seekData = 3
|
||||
seekHole = 4
|
||||
)
|
||||
|
||||
// SeekHole finds the next hole at or after the offset. The eof flag
|
||||
// reports that none of the requested content follows: every file
|
||||
// carries a virtual hole at its end, so a dense tail answers the file
|
||||
// size with eof set, RFC 7862 section 15.11. ErrBeyondEOF answers a
|
||||
// request that starts past the end.
|
||||
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
|
||||
return l.seek(h, offset, seekHole)
|
||||
}
|
||||
|
||||
// SeekData finds the next data byte at or after the offset. The eof
|
||||
// flag reports that no data follows the offset, and the answer names
|
||||
// the file size; ErrBeyondEOF answers a request that starts past the
|
||||
// end.
|
||||
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
|
||||
return l.seek(h, offset, seekData)
|
||||
}
|
||||
|
||||
func (l *Local) seek(h Handle, offset int64, whence int) (int64, bool, error) {
|
||||
// The seek opens the same revalidated, unfollowed descriptor every
|
||||
// other data path uses, so a name swapped for a link never serves
|
||||
// bytes from outside the export.
|
||||
f, _, err := l.openVerified(h, syscall.O_RDONLY)
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
defer f.Close()
|
||||
fd := int(f.Fd())
|
||||
size, err := fstatSize(fd)
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
if offset > size {
|
||||
return 0, false, ErrBeyondEOF
|
||||
}
|
||||
at, err := syscall.Seek(fd, offset, whence)
|
||||
if err == syscall.ENXIO {
|
||||
// The range from the offset to the end carries none of the
|
||||
// requested content: the answer is the end of the file with the
|
||||
// eof flag set.
|
||||
return size, true, nil
|
||||
}
|
||||
if err != nil {
|
||||
return 0, false, err
|
||||
}
|
||||
if at >= size {
|
||||
// The virtual hole at the end of every file: the eof flag tells
|
||||
// the client the search is over.
|
||||
return size, true, nil
|
||||
}
|
||||
return at, false, nil
|
||||
}
|
||||
|
||||
// Allocate reserves a range with fallocate in the default mode, which
|
||||
// grows the file to offset plus length, the behaviour RFC 7862 section
|
||||
// 15.1 requires of ALLOCATE; Deallocate punches a hole into the range and
|
||||
// keeps the size. The handle is opened without following a final symlink
|
||||
// and revalidated against the descriptor before anything is reserved.
|
||||
func (l *Local) Allocate(h Handle, offset, length int64) error {
|
||||
return l.fallocate(h, offset, length, 0)
|
||||
}
|
||||
|
||||
// Deallocate punches a hole into the range and keeps the file size.
|
||||
func (l *Local) Deallocate(h Handle, offset, length int64) error {
|
||||
return l.fallocate(h, offset, length, 0x03) // KEEP_SIZE|PUNCH_HOLE
|
||||
}
|
||||
|
||||
func (l *Local) fallocate(h Handle, offset, length int64, mode uint32) error {
|
||||
kind, _, _, _, err := l.resolve(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if kind != typeFile {
|
||||
return ErrIsDir
|
||||
}
|
||||
f, _, err := l.openVerified(h, syscall.O_WRONLY)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
_, _, errno := syscall.Syscall6(syscall.SYS_FALLOCATE, f.Fd(), uintptr(mode),
|
||||
uintptr(offset), uintptr(length), 0, 0)
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// fstatSize reads the size of the open file.
|
||||
func fstatSize(fd int) (int64, error) {
|
||||
var st syscall.Stat_t
|
||||
if err := syscall.Fstat(fd, &st); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return st.Size, nil
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestLocalSparse(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
if err := os.WriteFile(filepath.Join(root, "s.bin"), make([]byte, 8192), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rootHandle, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
fh, _, err := l.Lookup(rootHandle, "s.bin")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The file is all data: the only hole is the virtual one every file
|
||||
// carries at its end, so the seek answers the size with the eof flag,
|
||||
// RFC 7862 section 15.11.
|
||||
off, eof, err := l.SeekHole(fh, 0)
|
||||
if err != nil || off != 8192 || !eof {
|
||||
t.Fatalf("hole in a full file: %d %v %v", off, eof, err)
|
||||
}
|
||||
if off, eof, err := l.SeekData(fh, 0); err != nil || off != 0 || eof {
|
||||
t.Fatalf("seek data: %d %v %v", off, eof, err)
|
||||
}
|
||||
// A seek past the end is NXIO, not an answer.
|
||||
if _, _, err := l.SeekData(fh, 8193); err != ErrBeyondEOF {
|
||||
t.Fatalf("seek past the end: %v", err)
|
||||
}
|
||||
// Punching a hole moves the first hole to the punched offset.
|
||||
if err := l.Deallocate(fh, 4096, 4096); err != nil {
|
||||
t.Fatalf("deallocate: %v", err)
|
||||
}
|
||||
off, eof, err = l.SeekHole(fh, 0)
|
||||
if err != nil || off != 4096 || eof {
|
||||
t.Fatalf("seek hole: %d %v %v", off, eof, err)
|
||||
}
|
||||
// No data follows inside the hole: the answer names the size with
|
||||
// the eof flag.
|
||||
if off, eof, err := l.SeekData(fh, 4096); err != nil || off != 8192 || !eof {
|
||||
t.Fatalf("data in the hole: %d %v %v", off, eof, err)
|
||||
}
|
||||
// Allocate reserves the space without moving the hole back.
|
||||
if err := l.Allocate(fh, 4096, 4096); err != nil {
|
||||
t.Fatalf("allocate: %v", err)
|
||||
}
|
||||
if off, eof, err := l.SeekHole(fh, 0); err != nil || off != 4096 || eof {
|
||||
t.Fatalf("seek hole after allocate: %d %v %v", off, eof, err)
|
||||
}
|
||||
// Allocate past the end grows the file to the end of the reserved
|
||||
// range, which RFC 7862 section 15.1 requires.
|
||||
if err := l.Allocate(fh, 8192, 4096); err != nil {
|
||||
t.Fatalf("allocate past eof: %v", err)
|
||||
}
|
||||
info, err := l.Getattr(fh)
|
||||
if err != nil || info.Size != 12288 {
|
||||
t.Fatalf("size after the allocate past eof: %d, %v", info.Size, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build !linux && !freebsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
// SeekHole needs a system hole seeking call.
|
||||
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
|
||||
return 0, false, ErrNoSparse
|
||||
}
|
||||
|
||||
// SeekData needs a system hole seeking call.
|
||||
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
|
||||
return 0, false, ErrNoSparse
|
||||
}
|
||||
|
||||
// Allocate needs a system space reservation call.
|
||||
func (l *Local) Allocate(h Handle, offset, length int64) error { return ErrNoSparse }
|
||||
|
||||
// Deallocate needs a system hole punching call.
|
||||
func (l *Local) Deallocate(h Handle, offset, length int64) error { return ErrNoSparse }
|
||||
@@ -0,0 +1,23 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build freebsd || netbsd || openbsd || darwin
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// symlinkRefused reports whether an open refused a final component that is
|
||||
// a symlink, the answer O_NOFOLLOW exists for. The systems disagree on the
|
||||
// answer: FreeBSD and OpenBSD answer EMLINK, NetBSD answers EFTYPE, darwin
|
||||
// names EMLINK in its open(2), and the ELOOP spelling stays in the check
|
||||
// for the lineage it came from. FreeBSD, OpenBSD and NetBSD are verified
|
||||
// live; the darwin answer rests on its manual.
|
||||
func symlinkRefused(err error) bool {
|
||||
return errors.Is(err, syscall.ELOOP) ||
|
||||
errors.Is(err, syscall.EMLINK) ||
|
||||
errors.Is(err, syscall.EFTYPE)
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// symlinkRefused reports whether an open refused a final component that is
|
||||
// a symlink, the answer O_NOFOLLOW exists for. Linux answers ELOOP.
|
||||
func symlinkRefused(err error) bool {
|
||||
return errors.Is(err, syscall.ELOOP)
|
||||
}
|
||||
@@ -0,0 +1,253 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build freebsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"strings"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// xattrNamespace is the only namespace this backend serves: RFC 8276
|
||||
// section 3.3 names the attributes with their namespace, and the local
|
||||
// mapping of the FreeBSD server is the extattr user namespace.
|
||||
const xattrNamespace = "user."
|
||||
|
||||
// extattrNamespaceUser is EXTATTR_NAMESPACE_USER of sys/sys/extattr.h,
|
||||
// the numeric namespace the extattr family of system calls takes. The
|
||||
// syscall package of FreeBSD exports the traps of the family but not
|
||||
// this constant.
|
||||
const extattrNamespaceUser = 0x1
|
||||
|
||||
// xattrPath resolves a handle for the extattr family: the registered path
|
||||
// must still name the handle's device, inode and kind. A symlink is
|
||||
// refused the way the kernel refuses the user namespace on one, so the
|
||||
// path calls never follow a link out of the export.
|
||||
func (l *Local) xattrPath(h Handle) (string, error) {
|
||||
_, path, fi, err := l.revalidate(h)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if fi.Mode()&os.ModeSymlink != 0 {
|
||||
return "", syscall.EPERM
|
||||
}
|
||||
return path, nil
|
||||
}
|
||||
|
||||
// GetXattr reads one named attribute of the object.
|
||||
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
attr, ok := strings.CutPrefix(name, xattrNamespace)
|
||||
if !ok {
|
||||
return nil, ErrNoXattr
|
||||
}
|
||||
buf := make([]byte, maxOr(max, 256))
|
||||
for {
|
||||
n, err := extattrGetFile(path, attr, buf)
|
||||
if err == syscall.ERANGE {
|
||||
if len(buf) > 1<<20 {
|
||||
return nil, syscall.ERANGE
|
||||
}
|
||||
buf = make([]byte, len(buf)*2)
|
||||
continue
|
||||
}
|
||||
if err == syscall.ENOATTR {
|
||||
return nil, ErrNoXattr
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf[:n], nil
|
||||
}
|
||||
}
|
||||
|
||||
// SetXattr writes one named attribute under the RFC 8276 mode. The
|
||||
// extattr interface of FreeBSD carries no create and replace flags, so
|
||||
// the two strict modes ask for the attribute first and write it after:
|
||||
// a create of an existing name answers EEXIST and a replace of a
|
||||
// missing one ENOATTR, the same answers the flags of Linux produce.
|
||||
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
attr, ok := strings.CutPrefix(name, xattrNamespace)
|
||||
if !ok {
|
||||
return syscall.EOPNOTSUPP
|
||||
}
|
||||
switch mode {
|
||||
case XattrModeCreate, XattrModeReplace:
|
||||
_, err := extattrGetFile(path, attr, nil)
|
||||
switch {
|
||||
case err == nil && mode == XattrModeCreate:
|
||||
return syscall.EEXIST
|
||||
case err == syscall.ENOATTR && mode == XattrModeReplace:
|
||||
return syscall.ENOATTR
|
||||
case err != nil && err != syscall.ENOATTR:
|
||||
return err
|
||||
}
|
||||
}
|
||||
return extattrSetFile(path, attr, value)
|
||||
}
|
||||
|
||||
// ListXattr names the user namespace attributes of the object. The list
|
||||
// of extattr_list_file is a sequence of one length byte and name pairs
|
||||
// with no separators, extattr(2), and its names carry no namespace, so
|
||||
// each name is dressed with the namespace prefix the protocol speaks.
|
||||
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
buf := make([]byte, maxOr(max, 1024))
|
||||
for {
|
||||
n, err := extattrListFile(path, buf)
|
||||
if err == syscall.ERANGE {
|
||||
if len(buf) > 1<<20 {
|
||||
return nil, syscall.ERANGE
|
||||
}
|
||||
buf = make([]byte, len(buf)*2)
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
buf = buf[:n]
|
||||
break
|
||||
}
|
||||
var names []string
|
||||
for len(buf) > 0 {
|
||||
size := int(buf[0])
|
||||
if size+1 > len(buf) {
|
||||
break
|
||||
}
|
||||
names = append(names, xattrNamespace+string(buf[1:1+size]))
|
||||
buf = buf[1+size:]
|
||||
}
|
||||
return names, nil
|
||||
}
|
||||
|
||||
// RemoveXattr deletes one named attribute.
|
||||
func (l *Local) RemoveXattr(h Handle, name string) error {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
attr, ok := strings.CutPrefix(name, xattrNamespace)
|
||||
if !ok {
|
||||
return ErrNoXattr
|
||||
}
|
||||
if err := extattrDeleteFile(path, attr); err == syscall.ENOATTR {
|
||||
return ErrNoXattr
|
||||
} else if err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// maxOr replaces a zero budget with the given default.
|
||||
func maxOr(max, def int) int {
|
||||
if max == 0 || max > 1<<20 {
|
||||
return def
|
||||
}
|
||||
return max
|
||||
}
|
||||
|
||||
// extattrGetFile reads the attribute attr of path into buf, or names its
|
||||
// size when buf is nil, extattr(2).
|
||||
func extattrGetFile(path, attr string, buf []byte) (int, error) {
|
||||
name, err := syscall.ByteSliceFromString(attr)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
// SAFETY: the kernel reads the buffer for the length of the call
|
||||
// only, and data stays alive through the unsafe pointer until the
|
||||
// system call returns.
|
||||
var data unsafe.Pointer
|
||||
if len(buf) > 0 {
|
||||
data = unsafe.Pointer(&buf[0])
|
||||
}
|
||||
n, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_GET_FILE,
|
||||
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
|
||||
extattrNamespaceUser,
|
||||
uintptr(unsafe.Pointer(&name[0])),
|
||||
uintptr(data),
|
||||
uintptr(len(buf)),
|
||||
0)
|
||||
if errno != 0 {
|
||||
return 0, errno
|
||||
}
|
||||
return int(n), nil
|
||||
}
|
||||
|
||||
// extattrSetFile writes value into the attribute attr of path,
|
||||
// extattr(2).
|
||||
func extattrSetFile(path, attr string, value []byte) error {
|
||||
name, err := syscall.ByteSliceFromString(attr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// SAFETY: the kernel reads the buffer for the length of the call
|
||||
// only, and value stays alive through the unsafe pointer until the
|
||||
// system call returns.
|
||||
var data unsafe.Pointer
|
||||
if len(value) > 0 {
|
||||
data = unsafe.Pointer(&value[0])
|
||||
}
|
||||
_, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_SET_FILE,
|
||||
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
|
||||
extattrNamespaceUser,
|
||||
uintptr(unsafe.Pointer(&name[0])),
|
||||
uintptr(data),
|
||||
uintptr(len(value)),
|
||||
0)
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// extattrListFile names the user namespace attributes of path into buf,
|
||||
// or names the size of the list when buf is nil, extattr(2).
|
||||
func extattrListFile(path string, buf []byte) (int, error) {
|
||||
// SAFETY: the kernel writes the buffer for the length of the call
|
||||
// only, and buf stays alive through the unsafe pointer until the
|
||||
// system call returns.
|
||||
var data unsafe.Pointer
|
||||
if len(buf) > 0 {
|
||||
data = unsafe.Pointer(&buf[0])
|
||||
}
|
||||
n, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_LIST_FILE,
|
||||
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
|
||||
extattrNamespaceUser,
|
||||
uintptr(data),
|
||||
uintptr(len(buf)),
|
||||
0, 0)
|
||||
if errno != 0 {
|
||||
return 0, errno
|
||||
}
|
||||
return int(n), nil
|
||||
}
|
||||
|
||||
// extattrDeleteFile takes the attribute attr off path, extattr(2).
|
||||
func extattrDeleteFile(path, attr string) error {
|
||||
name, err := syscall.ByteSliceFromString(attr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
_, _, errno := syscall.Syscall(syscall.SYS_EXTATTR_DELETE_FILE,
|
||||
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
|
||||
extattrNamespaceUser,
|
||||
uintptr(unsafe.Pointer(&name[0])))
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"strings"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// xattrNamespace is the only namespace this backend serves: RFC 8276
|
||||
// section 3.3 names the attributes with their namespace, and the local
|
||||
// mapping of the Linux server is the user namespace.
|
||||
const xattrNamespace = "user."
|
||||
|
||||
// Flags of the xattr system calls.
|
||||
const (
|
||||
xattrCreateFlag = 0x1
|
||||
xattrReplaceFlag = 0x2
|
||||
)
|
||||
|
||||
// xattrPath resolves a handle for the xattr family: the registered path
|
||||
// must still name the handle's device, inode and kind. A symlink is
|
||||
// refused the way the kernel refuses the user namespace on one, so the
|
||||
// path calls never follow a link out of the export.
|
||||
func (l *Local) xattrPath(h Handle) (string, error) {
|
||||
_, path, fi, err := l.revalidate(h)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if fi.Mode()&os.ModeSymlink != 0 {
|
||||
return "", syscall.EPERM
|
||||
}
|
||||
return path, nil
|
||||
}
|
||||
|
||||
// GetXattr reads one named attribute of the object.
|
||||
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !strings.HasPrefix(name, xattrNamespace) {
|
||||
return nil, ErrNoXattr
|
||||
}
|
||||
buf := make([]byte, maxOr(max, 256))
|
||||
for {
|
||||
n, err := syscall.Getxattr(path, name, buf)
|
||||
if err == syscall.ERANGE {
|
||||
if len(buf) > 1<<20 {
|
||||
return nil, syscall.ERANGE
|
||||
}
|
||||
buf = make([]byte, len(buf)*2)
|
||||
continue
|
||||
}
|
||||
if err == syscall.ENODATA {
|
||||
return nil, ErrNoXattr
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf[:n], nil
|
||||
}
|
||||
}
|
||||
|
||||
// SetXattr writes one named attribute under the RFC 8276 mode.
|
||||
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !strings.HasPrefix(name, xattrNamespace) {
|
||||
return syscall.EOPNOTSUPP
|
||||
}
|
||||
flags := 0
|
||||
switch mode {
|
||||
case XattrModeCreate:
|
||||
flags = xattrCreateFlag
|
||||
case XattrModeReplace:
|
||||
flags = xattrReplaceFlag
|
||||
}
|
||||
return syscall.Setxattr(path, name, value, flags)
|
||||
}
|
||||
|
||||
// ListXattr names the user namespace attributes of the object.
|
||||
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
buf := make([]byte, maxOr(max, 1024))
|
||||
for {
|
||||
n, err := syscall.Listxattr(path, buf)
|
||||
if err == syscall.ERANGE {
|
||||
if len(buf) > 1<<20 {
|
||||
return nil, syscall.ERANGE
|
||||
}
|
||||
buf = make([]byte, len(buf)*2)
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var names []string
|
||||
for entry := range strings.SplitSeq(string(buf[:n]), "\x00") {
|
||||
if strings.HasPrefix(entry, xattrNamespace) {
|
||||
names = append(names, entry)
|
||||
}
|
||||
}
|
||||
return names, nil
|
||||
}
|
||||
}
|
||||
|
||||
// RemoveXattr deletes one named attribute.
|
||||
func (l *Local) RemoveXattr(h Handle, name string) error {
|
||||
path, err := l.xattrPath(h)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !strings.HasPrefix(name, xattrNamespace) {
|
||||
return ErrNoXattr
|
||||
}
|
||||
if err := syscall.Removexattr(path, name); err == syscall.ENODATA {
|
||||
return ErrNoXattr
|
||||
} else if err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// maxOr replaces a zero budget with the given default.
|
||||
func maxOr(max, def int) int {
|
||||
if max == 0 || max > 1<<20 {
|
||||
return def
|
||||
}
|
||||
return max
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build linux
|
||||
|
||||
package nfsfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestLocalXattr(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
if err := os.WriteFile(filepath.Join(root, "a.txt"), []byte("x"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
l, err := NewLocal(root)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rootHandle, err := l.Root()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
fh, _, err := l.Lookup(rootHandle, "a.txt")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := l.GetXattr(fh, "user.tag", 0); err != ErrNoXattr {
|
||||
t.Fatalf("missing attribute: %v", err)
|
||||
}
|
||||
if err := l.SetXattr(fh, "user.tag", []byte("value"), XattrModeCreate); err != nil {
|
||||
t.Fatalf("create: %v", err)
|
||||
}
|
||||
if err := l.SetXattr(fh, "user.tag", []byte("again"), XattrModeCreate); err == nil {
|
||||
t.Fatal("create over a live attribute succeeded")
|
||||
}
|
||||
got, err := l.GetXattr(fh, "user.tag", 0)
|
||||
if err != nil || string(got) != "value" {
|
||||
t.Fatalf("value after the refused create %q: %v", got, err)
|
||||
}
|
||||
if err := l.SetXattr(fh, "user.tag", []byte("again"), XattrModeReplace); err != nil {
|
||||
t.Fatalf("replace: %v", err)
|
||||
}
|
||||
got, err = l.GetXattr(fh, "user.tag", 0)
|
||||
if err != nil || string(got) != "again" {
|
||||
t.Fatalf("value after the replace %q: %v", got, err)
|
||||
}
|
||||
names, err := l.ListXattr(fh, 0)
|
||||
if err != nil || len(names) != 1 || names[0] != "user.tag" {
|
||||
t.Fatalf("names %v: %v", names, err)
|
||||
}
|
||||
if err := l.SetXattr(fh, "system.nfs", []byte("x"), XattrModeCreate); err == nil {
|
||||
t.Fatal("a foreign namespace was accepted")
|
||||
}
|
||||
if err := l.RemoveXattr(fh, "user.tag"); err != nil {
|
||||
t.Fatalf("remove: %v", err)
|
||||
}
|
||||
if err := l.RemoveXattr(fh, "user.tag"); err != ErrNoXattr {
|
||||
t.Fatalf("remove again: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
//go:build !linux && !freebsd
|
||||
|
||||
package nfsfs
|
||||
|
||||
// GetXattr answers that the backend carries no extended attributes; the
|
||||
// platforms without a system xattr interface serve none.
|
||||
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
|
||||
return nil, ErrXattrNotSupp
|
||||
}
|
||||
|
||||
// SetXattr answers that the backend carries no extended attributes.
|
||||
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
|
||||
return ErrXattrNotSupp
|
||||
}
|
||||
|
||||
// ListXattr answers that the backend carries no extended attributes.
|
||||
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
|
||||
return nil, ErrXattrNotSupp
|
||||
}
|
||||
|
||||
// RemoveXattr answers that the backend carries no extended attributes.
|
||||
func (l *Local) RemoveXattr(h Handle, name string) error {
|
||||
return ErrXattrNotSupp
|
||||
}
|
||||
Reference in New Issue
Block a user