feat: full NFSv4.2 server and client in pure Go
Test / test (push) Successful in 2m4s
Release / gates (push) Successful in 2m5s
Release / build (amd64, freebsd) (push) Successful in 1m27s
Release / build (amd64, linux) (push) Successful in 1m22s
Release / build (amd64, netbsd) (push) Successful in 1m19s
Release / build (amd64, openbsd) (push) Successful in 1m20s
Release / build (arm64, darwin) (push) Successful in 1m21s
Release / build (arm64, freebsd) (push) Successful in 1m26s
Release / build (arm64, linux) (push) Successful in 1m25s
Release / build (arm64, netbsd) (push) Successful in 1m31s
Release / build (arm64, openbsd) (push) Successful in 1m27s
Release / build (loong64, linux) (push) Successful in 1m37s
Release / build (riscv64, linux) (push) Successful in 1m21s
Release / release (push) Successful in 40s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 18:51:17 +02:00
commit a9b8039ef7
153 changed files with 34403 additions and 0 deletions
+112
View File
@@ -0,0 +1,112 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"runtime"
"syscall"
"unsafe"
)
// The reflink ioctl of linux/fs.h, FICLONERANGE, and the syscall number
// of copy_file_range(2), which the standard library does not export on
// linux. FICLONERANGE is _IOW(0x94, 13, struct file_clone_range) with a
// 32 byte struct: (1<<30)|(32<<16)|(0x94<<8)|13. The copy_file_range
// numbers follow the kernel's syscall tables: 319 on amd64 and 286 on
// the asm generic table of arm64, riscv64 and loong64. An architecture
// outside the table answers unshareable, and the caller keeps its
// userspace path.
const ioctlFICLONERANGE = 0x4020940D
// fileCloneRange mirrors struct file_clone_range of linux/fs.h, the
// argument of FICLONERANGE.
type fileCloneRange struct {
srcFD int64
srcOffset uint64
srcLength uint64
destOffset uint64
}
// CloneRange makes the destination carry the source's bytes through the
// reflink of the filesystem, XFS, btrfs and ZFS among them. The
// descriptor cache hands out both ends, so a range cloned through cached
// descriptors stays as verified as any other read or write.
func (l *Local) CloneRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error {
sf, releaseSrc, err := l.dataFD(src, false)
if err != nil {
return err
}
defer releaseSrc()
df, releaseDst, err := l.dataFD(dst, true)
if err != nil {
return err
}
defer releaseDst()
cr := fileCloneRange{
srcFD: int64(sf.Fd()),
srcOffset: uint64(srcOff),
srcLength: uint64(length),
destOffset: uint64(dstOff),
}
// SAFETY: ioctl takes the address of exactly the 32 byte struct the
// FICLONERANGE command name carries; the kernel reads it and writes
// nothing through it.
if _, _, errno := syscall.Syscall(syscall.SYS_IOCTL, df.Fd(), ioctlFICLONERANGE,
uintptr(unsafe.Pointer(&cr))); errno != 0 {
return errno
}
return nil
}
// CopyRange copies the bytes through copy_file_range(2), in chunks until
// the length is served. The syscall answers how much moved; a short move
// on the first call means the kernel refused for these files and the
// error travels to the caller's fallback.
func (l *Local) CopyRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error {
sf, releaseSrc, err := l.dataFD(src, false)
if err != nil {
return err
}
defer releaseSrc()
df, releaseDst, err := l.dataFD(dst, true)
if err != nil {
return err
}
defer releaseDst()
number := copyFileRangeSyscall()
if number == 0 {
return syscall.ENOSYS
}
var inOff, outOff int64 = srcOff, dstOff
for length > 0 {
chunk := min(length, 8<<20)
// SAFETY: the syscall copies from and to the file offsets behind
// the two pointers and updates them; both live across the call.
n, _, errno := syscall.Syscall6(number, sf.Fd(), uintptr(unsafe.Pointer(&inOff)),
df.Fd(), uintptr(unsafe.Pointer(&outOff)), uintptr(chunk), 0)
if errno != 0 {
return errno
}
if n == 0 {
return syscall.EINVAL
}
length -= int64(n)
}
return nil
}
// copyFileRangeSyscall answers the syscall number of copy_file_range on
// the architectures this project builds for linux, zero elsewhere.
func copyFileRangeSyscall() uintptr {
switch runtime.GOARCH {
case "amd64":
return 319
case "arm64", "riscv64", "loong64":
return 286
default:
return 0
}
}
+95
View File
@@ -0,0 +1,95 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"bytes"
"errors"
"os"
"path/filepath"
"syscall"
"testing"
)
// cloneTestFS builds a backend over a fresh directory holding a source
// file of the given content.
func cloneTestFS(t *testing.T, content []byte) (*Local, Handle, Handle, string) {
t.Helper()
root := t.TempDir()
l, err := NewLocal(root)
if err != nil {
t.Fatalf("NewLocal: %v", err)
}
if err := os.WriteFile(filepath.Join(root, "src"), content, 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
if err := os.WriteFile(filepath.Join(root, "dst"), make([]byte, len(content)), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
rh, err := l.Root()
if err != nil {
t.Fatalf("Root: %v", err)
}
src, _, err := l.Lookup(rh, "src")
if err != nil {
t.Fatalf("Lookup src: %v", err)
}
dst, _, err := l.Lookup(rh, "dst")
if err != nil {
t.Fatalf("Lookup dst: %v", err)
}
return l, src, dst, filepath.Join(root, "dst")
}
// TestCopyRangeKernel drives the kernel copy and verifies the bytes
// landed. Filesystems that refuse the syscall for their files skip the
// test: the handler falls back to its userspace path either way.
func TestCopyRangeKernel(t *testing.T) {
content := make([]byte, 1<<20)
for i := range content {
content[i] = byte(i * 3)
}
l, src, dst, dstPath := cloneTestFS(t, content)
if err := l.CopyRange(src, 0, dst, 0, int64(len(content))); err != nil {
if errors.Is(err, syscall.ENOSYS) || errors.Is(err, syscall.EOPNOTSUPP) ||
errors.Is(err, syscall.EXDEV) || errors.Is(err, syscall.EINVAL) {
t.Skipf("the filesystem refuses copy_file_range: %v", err)
}
t.Fatalf("CopyRange: %v", err)
}
got, err := os.ReadFile(dstPath)
if err != nil {
t.Fatalf("ReadFile: %v", err)
}
if !bytes.Equal(got, content) {
t.Fatalf("the copied bytes differ: %d vs %d", len(got), len(content))
}
}
// TestCloneRangeKernel drives the reflink where the filesystem provides
// one, and verifies the clone reads back as the source.
func TestCloneRangeKernel(t *testing.T) {
content := make([]byte, 1<<20)
for i := range content {
content[i] = byte(i * 7)
}
l, src, dst, dstPath := cloneTestFS(t, content)
err := l.CloneRange(src, 0, dst, 0, int64(len(content)))
if err != nil {
if errors.Is(err, syscall.EOPNOTSUPP) || errors.Is(err, syscall.EXDEV) ||
errors.Is(err, syscall.EINVAL) || errors.Is(err, syscall.EBADF) {
t.Skipf("the filesystem refuses FICLONERANGE: %v", err)
}
t.Fatalf("CloneRange: %v", err)
}
got, err := os.ReadFile(dstPath)
if err != nil {
t.Fatalf("ReadFile: %v", err)
}
if !bytes.Equal(got, content) {
t.Fatalf("the cloned bytes differ: %d vs %d", len(got), len(content))
}
}
+79
View File
@@ -0,0 +1,79 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
"os"
"path/filepath"
"syscall"
"testing"
)
// The create and write error wrappers map the errno families the
// protocol knows: a missing parent is NoEnt, a permission problem is
// Permission, a collision is Exist, and the rest stay raw.
func TestWrapErrFamilies(t *testing.T) {
// The wrapper level: each mapped errno.
cases := []struct {
err error
want error
}{
{nil, nil},
{&os.PathError{Op: "open", Err: syscall.EACCES}, ErrPermission},
{&os.PathError{Op: "open", Err: syscall.EPERM}, ErrPermission},
{&os.PathError{Op: "open", Err: syscall.EEXIST}, ErrExist},
{&os.PathError{Op: "open", Err: syscall.EISDIR}, ErrIsDir},
{&os.PathError{Op: "open", Err: syscall.ENOTDIR}, ErrNotDir},
{&os.PathError{Op: "write", Err: syscall.ENOSPC}, ErrNoSpace},
{&os.PathError{Op: "write", Err: syscall.EIO}, ErrIO},
}
for _, c := range cases {
if got := wrapCreateErr(c.err); !errors.Is(got, c.want) {
t.Fatalf("create wrap of %v: %v, want %v", c.err, got, c.want)
}
}
// The write wrapper answers a vanished path with stale, not noent.
if got := wrapWriteErr(&os.PathError{Op: "open", Err: syscall.ENOENT}); !errors.Is(got, ErrStale) {
t.Fatalf("write wrap of a vanished path: %v, want stale", got)
}
// An errno outside the families lands in the io error family with
// its cause kept in the text.
if got := wrapCreateErr(&os.PathError{Op: "open", Err: syscall.EDQUOT}); !errors.Is(got, ErrIO) {
t.Fatalf("unmapped errno: %v, want the io error family", got)
}
}
// A hard link to a missing target and one under a file both answer
// their own sentinels.
func TestLinkErrors(t *testing.T) {
root := t.TempDir()
l, err := NewLocal(root)
if err != nil {
t.Fatal(err)
}
rh, err := l.Root()
if err != nil {
t.Fatal(err)
}
if _, _, err := l.Link(nfsfsMissingHandle(), rh, "x"); err == nil {
t.Fatal("a link to a missing target succeeded")
}
target := filepath.Join(root, "t.txt")
if err := os.WriteFile(target, []byte("t"), 0o644); err != nil {
t.Fatal(err)
}
th, _, err := l.Lookup(rh, "t.txt")
if err != nil {
t.Fatal(err)
}
if _, _, err := l.Link(th, nfsfsMissingHandle(), "x"); err == nil {
t.Fatal("a link under a missing directory succeeded")
}
}
// nfsfsMissingHandle builds a handle that resolves to nothing.
func nfsfsMissingHandle() Handle {
return Handle("nfs\x02\x00\x00\x00stale-handle-bytes")
}
+244
View File
@@ -0,0 +1,244 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
"fmt"
"io/fs"
"os"
"syscall"
)
// fdCacheLimit bounds the descriptor cache. The cache holds at most this
// many idle descriptors across both access classes; descriptors checked
// out by running operations sit above the bound for their lifetime. The
// bound keeps a busy server under the process file descriptor ceiling:
// without it, one descriptor per file ever touched would grow without end.
const fdCacheLimit = 512
// An fdKey identifies one cached descriptor: the path it was opened
// through and the access class. The path is part of the key, not just the
// device and inode, so a file removed and recreated under a recycled inode
// number never inherits the old descriptor: the new object resolves to a
// fresh open, and the retired one fails its next identity check.
type fdKey struct {
path string
wr bool
}
// An fdEntry is one cached descriptor. refs counts the operations holding
// it right now; the entry leaves the cache, and its descriptor closes,
// when it is retired or evicted and the last reference lets go.
type fdEntry struct {
f *os.File
refs int
dead bool
use uint64
}
// dataFD hands out an open descriptor for the regular file the handle
// names: from the cache when one is held, from a fresh verified open
// otherwise. The identity is reverified on every use, a cache hit
// included, twice over: the registered path must still Lstat to the
// device, inode and kind the handle encodes, and the descriptor itself
// must still carry that identity with a link count above zero. A file
// removed while a descriptor of it sits in the cache therefore answers
// stale exactly as it does without the cache, and a name swapped for a
// symlink is never served through the cached descriptor.
//
// The returned release function must be called: it returns the descriptor
// to the cache, or closes it when the descriptor was retired, evicted or
// excluded from the cache while the lease was out.
func (l *Local) dataFD(h Handle, wr bool) (*os.File, func(), error) {
kind, dev, ino, path, err := l.resolve(h)
if err != nil {
return nil, nil, err
}
if kind != typeFile {
return nil, nil, ErrIsDir
}
key := fdKey{path: path, wr: wr}
if cached := l.fdCheckout(key); cached != nil {
// The descriptor is reverified against the path and against its
// own stat: a link count of zero means the cached descriptor is
// holding an unlinked inode, whatever the path names now.
fi, err := os.Lstat(path)
fst, ferr := cached.Stat()
switch {
case err == nil && ferr == nil && sameFile(fi, kind, dev, ino) &&
sameFile(fst, kind, dev, ino) && nlinkOf(fst) > 0:
return cached, func() { l.fdRelease(key, cached) }, nil
case err == nil:
// The path no longer names the inode, or the descriptor
// serves an unlinked one: retire and answer stale.
l.fdRetire(key)
return nil, nil, ErrStale
default:
l.fdRetire(key)
return nil, nil, revalidateStatErr(err)
}
}
flag := os.O_RDONLY
if wr {
flag = os.O_WRONLY
}
f, _, err := l.openVerified(h, flag)
if err != nil {
if em := l.fdRelieve(err); em {
// The open starved on descriptors; the cache gave its idle
// ones up. One retry is entitled to succeed now.
f, _, err = l.openVerified(h, flag)
}
if err != nil {
return nil, nil, err
}
}
l.fdInsert(key, f)
return f, func() { l.fdRelease(key, f) }, nil
}
// nlinkOf reports the link count a stat carried, zero when the platform
// data is missing.
func nlinkOf(fi os.FileInfo) uint64 {
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
return uint64(st.Nlink)
}
return 0
}
// revalidateStatErr maps the errors of the revalidating Lstat, the same
// mapping revalidate applies.
func revalidateStatErr(err error) error {
switch {
case errors.Is(err, fs.ErrNotExist), errors.Is(err, syscall.ENOTDIR):
return ErrStale
case errors.Is(err, fs.ErrPermission):
return ErrPermission
default:
return fmt.Errorf("%w: %v", ErrIO, err)
}
}
// fdCheckout hands the cached descriptor of key out to one operation and
// marks the entry busy, or reports nil when nothing usable is cached.
func (l *Local) fdCheckout(key fdKey) *os.File {
l.fdMu.Lock()
defer l.fdMu.Unlock()
e := l.fds[key]
if e == nil || e.dead {
return nil
}
e.refs++
l.fdUse++
e.use = l.fdUse
return e.f
}
// fdInsert admits a freshly opened, already verified descriptor into the
// cache with one reference held. When a retired entry under the same key
// is still draining its outstanding leases, the new descriptor bypasses
// the cache and closes on release instead.
func (l *Local) fdInsert(key fdKey, f *os.File) {
l.fdMu.Lock()
if old := l.fds[key]; old != nil {
// Only a drained entry leaves the map, so anything here is a
// retired one waiting for its leases; this descriptor stays out.
l.fdMu.Unlock()
return
}
l.fds[key] = &fdEntry{f: f, refs: 1, use: l.fdUse + 1}
l.fdUse++
closed := l.fdEvictLocked()
l.fdMu.Unlock()
for _, idle := range closed {
idle.Close()
}
}
// fdRelease ends one lease. A live entry takes the descriptor back; a
// retired, evicted or bypassed one closes it, at the last release.
func (l *Local) fdRelease(key fdKey, f *os.File) {
l.fdMu.Lock()
e := l.fds[key]
if e == nil || e.f != f {
l.fdMu.Unlock()
f.Close()
return
}
e.refs--
if e.refs > 0 {
l.fdMu.Unlock()
return
}
if e.dead {
delete(l.fds, key)
l.fdMu.Unlock()
f.Close()
return
}
l.fdMu.Unlock()
}
// fdRetire marks the cached descriptor of key dead: it is never handed
// out again, and it closes when its outstanding leases release.
func (l *Local) fdRetire(key fdKey) {
l.fdMu.Lock()
if e := l.fds[key]; e != nil {
e.dead = true
}
l.fdMu.Unlock()
}
// fdEvictLocked picks idle descriptors until the cache fits the bound and
// returns them for the caller to close outside the lock. Entries with
// outstanding leases are untouchable; a cache full of busy entries
// temporarily exceeds the bound by exactly the number of running
// operations.
func (l *Local) fdEvictLocked() []*os.File {
var closed []*os.File
for len(l.fds) > fdCacheLimit {
var victim *fdEntry
var victimKey fdKey
for key, e := range l.fds {
if e.dead || e.refs > 0 {
continue
}
if victim == nil || e.use < victim.use {
victim, victimKey = e, key
}
}
if victim == nil {
return closed
}
delete(l.fds, victimKey)
closed = append(closed, victim.f)
}
return closed
}
// fdRelieve answers whether err is an open refused for descriptor
// exhaustion and, when it is, retires every idle descriptor so one retry
// can run. The cache must never be the reason a server runs out of file
// descriptors.
func (l *Local) fdRelieve(err error) bool {
if !errors.Is(err, syscall.EMFILE) && !errors.Is(err, syscall.ENFILE) {
return false
}
l.fdMu.Lock()
var closed []*os.File
for key, e := range l.fds {
if e.refs > 0 {
e.dead = true
continue
}
delete(l.fds, key)
closed = append(closed, e.f)
}
l.fdMu.Unlock()
for _, f := range closed {
f.Close()
}
return true
}
+263
View File
@@ -0,0 +1,263 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
// Package nfsfs defines the virtual filesystem the NFS server serves, and
// provides a backend over a local directory.
//
// The interface carries exactly what the protocol layer needs and nothing
// more: handles that the backend itself interprets, the attributes each
// GETATTR turns into an fattr4, the data operations LOOKUP, READDIR and
// READ, and the Writer half that WRITE, CREATE and the other state
// changing operations reach.
package nfsfs
import (
"errors"
"io/fs"
"time"
)
// A Handle is an opaque file handle. The backend defines its layout; the
// protocol layer treats it as bytes.
type Handle []byte
// Sentinels the dispatcher maps onto NFS4ERR statuses. Use errors.Is.
var (
ErrStale = errors.New("nfsfs: unknown file handle")
ErrNoEnt = errors.New("nfsfs: no such file or directory")
ErrNotDir = errors.New("nfsfs: not a directory")
ErrIsDir = errors.New("nfsfs: is a directory")
ErrNameTooLong = errors.New("nfsfs: name too long")
ErrBadName = errors.New("nfsfs: invalid name")
ErrPermission = errors.New("nfsfs: permission denied")
ErrIO = errors.New("nfsfs: io error")
ErrExist = errors.New("nfsfs: file exists")
ErrNoSpace = errors.New("nfsfs: no space left")
ErrNotEmpty = errors.New("nfsfs: directory not empty")
ErrInval = errors.New("nfsfs: invalid argument")
ErrNotLnk = errors.New("nfsfs: not a symlink")
)
// Access mask bits, RFC 8881 section 15.2.2. The same values the protocol
// layer speaks.
const (
AccessRead = 1 << 0
AccessLookup = 1 << 1
AccessModify = 1 << 2
AccessExtend = 1 << 3
AccessDelete = 1 << 4
AccessExec = 1 << 5
)
// MaxName is the name length limit the backends enforce.
const MaxName = 255
// An Info carries the file attributes the backends report.
type Info struct {
Size int64
Mode fs.FileMode
ModTime time.Time
Dev uint64
Ino uint64
Nlink uint64
UID uint32
GID uint32
}
// IsDir reports whether the file is a directory.
func (i Info) IsDir() bool { return i.Mode.IsDir() }
// An Entry is one READDIR row: the cookie the client resumes from, the
// name, its handle and its attributes.
type Entry struct {
Cookie uint64
Name string
Handle Handle
Info Info
}
// A DirPage is one READDIR result page: the entries the cookie asked for,
// the verifier of the directory order, and whether the listing is complete.
type DirPage struct {
Entries []Entry
Verifier [8]byte
EOF bool
}
// An FS is the virtual filesystem the server serves. Implementations must
// be safe for concurrent use.
type FS interface {
// Root returns the handle of the export root.
Root() (Handle, error)
// Lookup resolves name under the parent handle.
Lookup(parent Handle, name string) (Handle, Info, error)
// Getattr reports the attributes of a handle.
Getattr(h Handle) (Info, error)
// ReadLink reports the target of a symlink. A handle that names
// anything else is an error.
ReadLink(h Handle) (string, error)
// Parent resolves the directory that holds h and the component name
// of h under it. The root of the export has no parent name.
Parent(h Handle) (Handle, string, error)
// ReadDir lists the directory from the given cookie, returning at most
// count entries, or all of them when count is zero or less.
ReadDir(h Handle, cookie uint64, count int) (DirPage, error)
// Read reads up to count bytes at the offset. A short result means end
// of file was reached.
Read(h Handle, off int64, count int) ([]byte, error)
// Access evaluates the requested mask bits for the credential and
// returns the bits granted.
Access(h Handle, mask uint32, uid, gid uint32, groups []uint32) (uint32, error)
}
// ValidName reports whether name can appear in a LOOKUP. Names carrying a
// separator or a control byte never belong to the client, because no
// backend resolves them.
func ValidName(name string) error {
switch {
case name == "":
return ErrBadName
case name == "." || name == "..":
return ErrBadName
case len(name) > MaxName:
return ErrNameTooLong
}
for i := range len(name) {
if name[i] == '/' || name[i] == 0 {
return ErrBadName
}
}
return nil
}
// The object kinds a CREATE may carry, the same values the protocol's
// createtype4 uses. A regular file is not among them: in NFSv4 regular
// files are created by OPEN.
const (
KindDir = 2
KindLnk = 5
KindSock = 6
KindFifo = 7
KindBlk = 3
KindChr = 4
)
// An Owner names the unix owner and group an object carries after its
// creation. The server runs under its own identity, so a backend that
// serves clients of several owners applies these values when a client
// makes a new object.
type Owner struct {
UID uint32
GID uint32
}
// A CreateSpec describes one object a CREATE makes.
type CreateSpec struct {
Kind uint32
Perm fs.FileMode // permission bits, applied exactly, umask aside
LinkData string // the target of a symlink
Major uint32 // device numbers of a character or block device
Minor uint32
Owner Owner // the owner a fresh object carries
}
// A Writer is the mutating half of an FS. A backend that serves reads only
// does not implement it, and the dispatcher answers NFS4ERR_ROFS.
type Writer interface {
// Create makes the object the spec describes under the parent handle.
// An existing target is an error for every kind.
Create(parent Handle, name string, spec CreateSpec) (Handle, Info, error)
// Write writes all of data at the offset of a regular file and
// returns how many bytes landed.
Write(h Handle, off int64, data []byte) (int, error)
// Remove takes the named entry out of the directory. Removing a
// directory that is not empty is an error.
Remove(dir Handle, name string) error
// Rename moves oldName from the oldDir directory to newName in the
// newDir directory, replacing an existing plain target the way POSIX
// rename does. The handles of the moved object and of its descendants
// keep working after the move.
Rename(oldDir Handle, oldName string, newDir Handle, newName string) error
// Setattr applies the named changes to a file. Changes that the
// backend cannot apply make the whole call fail.
Setattr(h Handle, s SetAttrs) error
// Link makes newName in dir a hard link to the target file.
Link(target Handle, dir Handle, name string) (Handle, Info, error)
// Sync flushes the file's dirty data to stable storage.
Sync(h Handle) error
// Open opens the regular file name under dir for writing. When create
// is set a missing file is made with perm and carried by owner; when
// guarded is set an existing name answers ErrExist instead of
// opening, which the GUARDED and EXCLUSIVE4_1 create modes require;
// when truncate is set an existing file is cut to zero first. The
// boolean reports whether the file was created by this call.
Open(dir Handle, name string, create, guarded, truncate bool, perm fs.FileMode, owner Owner) (h Handle, info Info, created bool, err error)
}
// A TimeSet is one time attribute of a SETATTR: either the server's
// current time or the time the client names.
type TimeSet struct {
Now bool
Time time.Time
}
// SetAttrs carries the changes a SETATTR names. A nil field is a change
// the client did not ask for.
type SetAttrs struct {
Mode *uint32
Size *int64
UID *uint32
GID *uint32
Atime *TimeSet
Mtime *TimeSet
}
// XattrFS is the optional extended attribute half of a backend. A backend
// that does not implement it answers NOT_SUPP to the xattr family.
type XattrFS interface {
// GetXattr reads one named attribute of the object.
GetXattr(h Handle, name string, max int) ([]byte, error)
// SetXattr writes one named attribute; the mode follows the
// SETXATTR4mode4 enum of RFC 8276.
SetXattr(h Handle, name string, value []byte, mode uint32) error
// ListXattr names the attributes of the object.
ListXattr(h Handle, max int) ([]string, error)
// RemoveXattr deletes one named attribute.
RemoveXattr(h Handle, name string) error
}
// ErrNoXattr marks a missing attribute, NFS4ERR_NOXATTR on the wire.
var ErrNoXattr = errors.New("nfsfs: no such attribute")
// ErrXattrNotSupp marks a backend that carries no extended attributes at
// all, NFS4ERR_NOT_SUPP on the wire.
var ErrXattrNotSupp = errors.New("nfsfs: extended attributes are not supported")
// ErrBeyondEOF marks a SEEK that starts past the end of the file,
// NFS4ERR_NXIO on the wire per RFC 7862 section 15.11.
var ErrBeyondEOF = errors.New("nfsfs: seek past the end")
// ErrNoSparse marks a backend whose platform carries no space
// reservation or hole seeking calls, NFS4ERR_NOT_SUPP on the wire.
var ErrNoSparse = errors.New("nfsfs: sparse file operations are not supported")
// The SETXATTR create modes of RFC 8276, mirrored from the wire
// protocol. They are platform independent: a backend that carries no
// extended attributes answers NOT_SUPP regardless of the mode.
const (
XattrModeCreate = 1
XattrModeReplace = 2
)
// A RangeCloner is the optional half that clones or copies a byte range
// between two regular files inside the kernel: the bytes never travel
// through userspace. A backend that does not carry it, or a filesystem
// that refuses the call for one pair of files, leaves the caller its
// userspace path.
type RangeCloner interface {
// CloneRange makes dstOff carry the same bytes as srcOff through a
// reflink where the filesystem provides one.
CloneRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error
// CopyRange copies length bytes through the kernel's copy syscall.
CopyRange(src Handle, srcOff int64, dst Handle, dstOff int64, length int64) error
}
File diff suppressed because it is too large Load Diff
+168
View File
@@ -0,0 +1,168 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"fmt"
"os"
"path/filepath"
"testing"
)
const benchChunk = 64 << 10
// benchRoot builds a backend over a fresh directory with one file of the
// given size, filled with a repeating pattern, and returns the backend
// and the file handle. Setup runs once, outside the measured region.
func benchRoot(b *testing.B, size int) (*Local, Handle) {
b.Helper()
root := b.TempDir()
l, err := NewLocal(root)
if err != nil {
b.Fatalf("NewLocal: %v", err)
}
p := filepath.Join(root, "file")
buf := make([]byte, 1<<20)
for i := range buf {
buf[i] = byte(i)
}
f, err := os.Create(p)
if err != nil {
b.Fatalf("Create: %v", err)
}
for written := 0; written < size; written += len(buf) {
if _, err := f.Write(buf); err != nil {
b.Fatalf("Write: %v", err)
}
}
if err := f.Close(); err != nil {
b.Fatalf("Close: %v", err)
}
rh, err := l.Root()
if err != nil {
b.Fatalf("Root: %v", err)
}
h, _, err := l.Lookup(rh, "file")
if err != nil {
b.Fatalf("Lookup: %v", err)
}
return l, h
}
// BenchmarkRead64K reads 64 KiB at a time from a 64 MiB file, cycling
// through the offsets so every read touches pages the previous read left.
func BenchmarkRead64K(b *testing.B) {
l, h := benchRoot(b, 64<<20)
off := int64(0)
b.SetBytes(benchChunk)
b.ResetTimer()
for b.Loop() {
if _, err := l.Read(h, off, benchChunk); err != nil {
b.Fatalf("Read: %v", err)
}
off += benchChunk
if off > 64<<20-benchChunk {
off = 0
}
}
}
// BenchmarkWrite64K writes 64 KiB at a time over a preallocated 64 MiB
// file, cycling through the offsets, so no read has to grow the file.
func BenchmarkWrite64K(b *testing.B) {
l, h := benchRoot(b, 64<<20)
buf := make([]byte, benchChunk)
off := int64(0)
b.SetBytes(benchChunk)
b.ResetTimer()
for b.Loop() {
if _, err := l.Write(h, off, buf); err != nil {
b.Fatalf("Write: %v", err)
}
off += benchChunk
if off > 64<<20-benchChunk {
off = 0
}
}
}
// BenchmarkGetattr reports the attributes of one file.
func BenchmarkGetattr(b *testing.B) {
l, h := benchRoot(b, 1<<20)
b.ResetTimer()
for b.Loop() {
if _, err := l.Getattr(h); err != nil {
b.Fatalf("Getattr: %v", err)
}
}
}
// BenchmarkLookup resolves one name under the export root of a directory
// holding a hundred files.
func BenchmarkLookup(b *testing.B) {
root := b.TempDir()
l, err := NewLocal(root)
if err != nil {
b.Fatalf("NewLocal: %v", err)
}
rh, err := l.Root()
if err != nil {
b.Fatalf("Root: %v", err)
}
for i := range 100 {
name := fmt.Sprintf("f%d", i)
if err := os.WriteFile(filepath.Join(root, name), []byte("x"), 0o644); err != nil {
b.Fatalf("WriteFile: %v", err)
}
}
b.ResetTimer()
i := 0
for b.Loop() {
if _, _, err := l.Lookup(rh, fmt.Sprintf("f%d", i%100)); err != nil {
b.Fatalf("Lookup: %v", err)
}
i++
}
}
// BenchmarkReadDirPage64 pages a 10 000 entry directory 64 entries at a
// time after the first page: with the listing cache the cost of a page
// is the page, and the benchmark holds that to the measurement.
func BenchmarkReadDirPage64(b *testing.B) {
root := b.TempDir()
l, err := NewLocal(root)
if err != nil {
b.Fatalf("NewLocal: %v", err)
}
dir := filepath.Join(root, "big")
if err := os.Mkdir(dir, 0o755); err != nil {
b.Fatalf("Mkdir: %v", err)
}
for i := range 10000 {
if err := os.WriteFile(filepath.Join(dir, fmt.Sprintf("f%04d", i)), []byte("x"), 0o644); err != nil {
b.Fatalf("WriteFile: %v", err)
}
}
rh, err := l.Root()
if err != nil {
b.Fatalf("Root: %v", err)
}
h, _, err := l.Lookup(rh, "big")
if err != nil {
b.Fatalf("Lookup: %v", err)
}
if _, err := l.ReadDir(h, 0, 64); err != nil {
b.Fatalf("first page: %v", err)
}
b.ResetTimer()
for b.Loop() {
page, err := l.ReadDir(h, 0, 64)
if err != nil {
b.Fatalf("ReadDir: %v", err)
}
if len(page.Entries) != 64 || page.EOF {
b.Fatalf("page: %d entries, eof %v", len(page.Entries), page.EOF)
}
}
}
+73
View File
@@ -0,0 +1,73 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"fmt"
"os"
"path/filepath"
"testing"
)
// openFDs counts the descriptors the process holds through /proc. The
// count is a lower bound of live descriptors and good enough to prove a
// cache does not leak: a thousand operations over one file must not add a
// thousand descriptors.
func openFDs(t *testing.T) int {
t.Helper()
entries, err := os.ReadDir("/proc/self/fd")
if err != nil {
t.Skipf("/proc/self/fd is unavailable: %v", err)
}
return len(entries)
}
// TestCacheNoDescriptorLeak drives more operations over one file than the
// cache can hold and requires the process descriptor count to stay flat.
func TestCacheNoDescriptorLeak(t *testing.T) {
l, h, _ := cacheTestRoot(t, "leak")
before := openFDs(t)
for range 1000 {
if _, err := l.Read(h, 0, 4); err != nil {
t.Fatalf("Read: %v", err)
}
}
after := openFDs(t)
if after-before > 8 {
t.Fatalf("descriptors grew from %d to %d over 1000 reads", before, after)
}
}
// TestCacheBound holds when many distinct files flow through: after
// touching twice the bound, at most the bound of cache entries may remain.
func TestCacheBound(t *testing.T) {
root := t.TempDir()
l, err := NewLocal(root)
if err != nil {
t.Fatalf("NewLocal: %v", err)
}
rh, err := l.Root()
if err != nil {
t.Fatalf("Root: %v", err)
}
for i := range 2 * fdCacheLimit {
name := fmt.Sprintf("f%d", i)
if err := os.WriteFile(filepath.Join(root, name), []byte("x"), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
h, _, err := l.Lookup(rh, name)
if err != nil {
t.Fatalf("Lookup: %v", err)
}
if _, err := l.Read(h, 0, 1); err != nil {
t.Fatalf("Read %s: %v", name, err)
}
}
l.fdMu.Lock()
n := len(l.fds)
l.fdMu.Unlock()
if n > fdCacheLimit {
t.Fatalf("cache holds %d entries, bound is %d", n, fdCacheLimit)
}
}
+174
View File
@@ -0,0 +1,174 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
"os"
"path/filepath"
"sync"
"testing"
)
// cacheTestRoot builds a backend over a fresh directory holding one file
// and returns the backend, the file handle and the path.
func cacheTestRoot(t *testing.T, content string) (*Local, Handle, string) {
t.Helper()
root := t.TempDir()
l, err := NewLocal(root)
if err != nil {
t.Fatalf("NewLocal: %v", err)
}
p := filepath.Join(root, "file")
if err := os.WriteFile(p, []byte(content), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
rh, err := l.Root()
if err != nil {
t.Fatalf("Root: %v", err)
}
h, _, err := l.Lookup(rh, "file")
if err != nil {
t.Fatalf("Lookup: %v", err)
}
return l, h, p
}
// TestCacheStaleAfterRemove covers the identity contract on a cache
// hit: a file removed while a descriptor of it is cached answers stale,
// exactly as it does without the cache.
func TestCacheStaleAfterRemove(t *testing.T) {
l, h, p := cacheTestRoot(t, "hello")
if _, err := l.Read(h, 0, 5); err != nil {
t.Fatalf("warm read: %v", err)
}
if err := os.Remove(p); err != nil {
t.Fatalf("Remove: %v", err)
}
if _, err := l.Read(h, 0, 5); !errors.Is(err, ErrStale) {
t.Fatalf("read after remove: %v, want ErrStale", err)
}
}
// TestCacheStaleAfterReplace covers a name swapped for another inode: the
// cached descriptor of the old inode must never serve through it.
func TestCacheStaleAfterReplace(t *testing.T) {
l, h, p := cacheTestRoot(t, "old")
if _, err := l.Read(h, 0, 3); err != nil {
t.Fatalf("warm read: %v", err)
}
if err := os.Remove(p); err != nil {
t.Fatalf("Remove: %v", err)
}
if err := os.WriteFile(p, []byte("new"), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
if _, err := l.Read(h, 0, 3); !errors.Is(err, ErrStale) {
t.Fatalf("read after replace: %v, want ErrStale", err)
}
}
// TestCacheInodeReuse covers the recycled inode number at the same path:
// a cached descriptor of the unlinked old inode must never serve the
// identity the recycled file now carries. The map entry is forged onto
// the new file, which is exactly the state a reuse produces.
func TestCacheInodeReuse(t *testing.T) {
l, h, p := cacheTestRoot(t, "stale data")
if _, err := l.Read(h, 0, 4); err != nil {
t.Fatalf("warm read: %v", err)
}
// Remove the file behind the backend's back and recreate a fresh one
// at the same path, then point the old handle's identity at it the
// way a recycled inode number would.
fi, err := os.Lstat(p)
if err != nil {
t.Fatalf("Lstat: %v", err)
}
oldID := fileID{dev: stat(fi).Dev, ino: stat(fi).Ino}
if err := os.Remove(p); err != nil {
t.Fatalf("Remove: %v", err)
}
if err := os.WriteFile(p, []byte("fresh"), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
fi2, err := os.Lstat(p)
if err != nil {
t.Fatalf("Lstat: %v", err)
}
newID := fileID{dev: stat(fi2).Dev, ino: stat(fi2).Ino}
l.mu.Lock()
l.paths[newID] = p
if newID == oldID {
// The filesystem handed back the very same inode; the scenario
// holds without forging anything.
l.paths[oldID] = p
} else {
delete(l.paths, oldID)
}
l.mu.Unlock()
// Whatever the inode numbers did, a fresh read must serve the fresh
// content, never the unlinked inode's bytes.
got, err := l.Read(h, 0, 5)
if err != nil {
// A stale answer is also safe: the identity broke and the cache
// refused. Serving the old bytes is the only failure.
if !errors.Is(err, ErrStale) {
t.Fatalf("read after reuse: %v", err)
}
return
}
if string(got) != "fresh" {
t.Fatalf("read after reuse: %q, want the fresh content", got)
}
}
// TestCacheWriteThrough covers that writes land and that a follow up read
// of the same cached file sees them.
func TestCacheWriteThrough(t *testing.T) {
l, h, p := cacheTestRoot(t, "0123456789")
if _, err := l.Read(h, 0, 10); err != nil {
t.Fatalf("warm read: %v", err)
}
if n, err := l.Write(h, 2, []byte("AB")); err != nil || n != 2 {
t.Fatalf("Write: %d, %v", n, err)
}
got, err := l.Read(h, 0, 10)
if err != nil {
t.Fatalf("Read: %v", err)
}
if string(got) != "01AB456789" {
t.Fatalf("Read: %q", got)
}
raw, err := os.ReadFile(p)
if err != nil {
t.Fatalf("ReadFile: %v", err)
}
if string(raw) != "01AB456789" {
t.Fatalf("file on disk: %q", raw)
}
}
// TestCacheConcurrent drives reads and writes of one file from many
// goroutines; the race detector is the judge.
func TestCacheConcurrent(t *testing.T) {
l, h, _ := cacheTestRoot(t, "concurrent")
var wg sync.WaitGroup
for i := range 8 {
wg.Go(func() {
for range 50 {
if _, err := l.Read(h, 0, 4); err != nil {
t.Errorf("Read: %v", err)
return
}
if i%2 == 0 {
if _, err := l.Write(h, 0, []byte("writ")); err != nil {
t.Errorf("Write: %v", err)
return
}
}
}
})
}
wg.Wait()
}
File diff suppressed because it is too large Load Diff
+29
View File
@@ -0,0 +1,29 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build darwin
package nfsfs
import "syscall"
// mknod creates a device node. It needs the superuser on darwin, and a
// failure of permission is reported as such rather than as an io error.
// The kind is picked before the single call: a second mknod over an
// existing node fails EEXIST and leaves the wrong kind behind.
func mknod(path string, spec CreateSpec, perm uint32) error {
dev := makedev(spec.Major, spec.Minor)
kind := uint32(syscall.S_IFCHR)
if spec.Kind == KindBlk {
kind = syscall.S_IFBLK
}
return syscall.Mknod(path, perm|kind, int(dev))
}
// makedev assembles a device number the way the kernel expects it: the
// encoding of makedev in bsd/sys/types.h of xnu, where the major number
// sits at bits twenty-four through thirty-one and the minor number keeps
// bits zero through twenty-three.
func makedev(major, minor uint32) uint64 {
return uint64(major&0xff)<<24 | uint64(minor&0xffffff)
}
+32
View File
@@ -0,0 +1,32 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build freebsd
package nfsfs
import "syscall"
// mknod creates a device node. It needs the superuser on FreeBSD, and a
// failure of permission is reported as such rather than as an io error.
// The kind is picked before the single call: a second mknod over an
// existing node fails EEXIST and leaves the wrong kind behind.
func mknod(path string, spec CreateSpec, perm uint32) error {
dev := makedev(spec.Major, spec.Minor)
kind := uint32(syscall.S_IFCHR)
if spec.Kind == KindBlk {
kind = syscall.S_IFBLK
}
return syscall.Mknod(path, perm|kind, dev)
}
// makedev assembles a device number the way the kernel expects it: the
// encoding of makedev in sys/sys/types.h, where the low byte of the
// major number sits at bits eight to fifteen with the rest of it above
// the thirty-second bit, and the minor number keeps its low byte at bit
// zero with the byte at bits eight to fifteen lifted above the
// thirty-second bit.
func makedev(major, minor uint32) uint64 {
return uint64(major&0xffffff00)<<32 | uint64(major&0xff)<<8 |
uint64(minor&0xff00)<<24 | uint64(minor&0xffff00ff)
}
+25
View File
@@ -0,0 +1,25 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import "syscall"
// mknod creates a device node. It needs CAP_MKNOD on Linux, and a failure
// of permission is reported as such rather than as an io error. The kind
// is picked before the single call: a second mknod over an existing node
// fails EEXIST and leaves the wrong kind behind.
func mknod(path string, spec CreateSpec, perm uint32) error {
dev := int(makedev(spec.Major, spec.Minor))
kind := uint32(syscall.S_IFCHR)
if spec.Kind == KindBlk {
kind = syscall.S_IFBLK
}
return syscall.Mknod(path, perm|kind, dev)
}
// makedev assembles a device number the way the kernel expects it.
func makedev(major, minor uint32) uint64 {
return uint64(minor&0xff) | uint64(major&0xfff)<<8 |
uint64(minor&0xfff00)<<12 | uint64(major&0xfffff000)<<32
}
+30
View File
@@ -0,0 +1,30 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build netbsd
package nfsfs
import "syscall"
// mknod creates a device node. It needs the superuser on NetBSD, and a
// failure of permission is reported as such rather than as an io error.
// The kind is picked before the single call: a second mknod over an
// existing node fails EEXIST and leaves the wrong kind behind.
func mknod(path string, spec CreateSpec, perm uint32) error {
dev := makedev(spec.Major, spec.Minor)
kind := uint32(syscall.S_IFCHR)
if spec.Kind == KindBlk {
kind = syscall.S_IFBLK
}
return syscall.Mknod(path, perm|kind, int(dev))
}
// makedev assembles a device number the way the kernel expects it: the
// encoding of makedev in sys/sys/types.h, where the major number sits at
// bits eight to nineteen, the low byte of the minor number keeps bit zero
// through seven and the rest of the minor number is lifted to bits twenty
// through thirty-one.
func makedev(major, minor uint32) uint64 {
return uint64(major&0xfff)<<8 | uint64(minor&0xfff00)<<12 | uint64(minor&0xff)
}
+30
View File
@@ -0,0 +1,30 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build openbsd
package nfsfs
import "syscall"
// mknod creates a device node. It needs the superuser on OpenBSD, and a
// failure of permission is reported as such rather than as an io error.
// The kind is picked before the single call: a second mknod over an
// existing node fails EEXIST and leaves the wrong kind behind.
func mknod(path string, spec CreateSpec, perm uint32) error {
dev := makedev(spec.Major, spec.Minor)
kind := uint32(syscall.S_IFCHR)
if spec.Kind == KindBlk {
kind = syscall.S_IFBLK
}
return syscall.Mknod(path, perm|kind, int(dev))
}
// makedev assembles a device number the way the kernel expects it: the
// encoding of makedev in sys/sys/types.h, where the low byte of the
// major number sits at bits eight to fifteen, the low byte of the minor
// number keeps bit zero through seven, and the rest of the minor number
// is lifted to bits sixteen upward.
func makedev(major, minor uint32) uint64 {
return uint64(major&0xff)<<8 | uint64(minor&0xff) | uint64(minor&0xffff00)<<8
}
+55
View File
@@ -0,0 +1,55 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
"fmt"
"io"
)
// A ReadIntoer is the optional read half that fills a caller provided
// buffer instead of allocating its own: the reply path of a server reads
// straight into the buffer it is about to send. A backend that carries
// only Read keeps its allocating behaviour.
type ReadIntoer interface {
// ReadInto reads up to len(buf) bytes at the offset into buf and
// answers how many landed and whether the end of file was reached.
ReadInto(h Handle, off int64, buf []byte) (int, bool, error)
}
// ReadInto fills buf from the regular file the handle names. The
// descriptor comes from the cache or a fresh verified open, and the
// identity is revalidated before anything is read, exactly as Read.
func (l *Local) ReadInto(h Handle, off int64, buf []byte) (int, bool, error) {
f, release, err := l.dataFD(h, false)
if err != nil {
return 0, false, err
}
defer release()
n, rerr := f.ReadAt(buf, off)
switch {
case errors.Is(rerr, io.EOF):
return n, true, nil
case rerr != nil:
return n, false, fmt.Errorf("%w: %v", ErrIO, rerr)
}
// A full buffer proves the end of file only against the size; the
// descriptor's own stat answers it without a second path walk.
if fst, serr := f.Stat(); serr == nil {
return n, int64(off)+int64(n) >= fst.Size(), nil
}
return n, false, nil
}
// ReadInto forwards to the wrapped export: a read only export reads as
// its backend does.
func (ro roFS) ReadInto(h Handle, off int64, buf []byte) (int, bool, error) {
ri, ok := ro.FS.(ReadIntoer)
if !ok {
data, err := ro.Read(h, off, len(buf))
return len(data), false, err
}
return ri.ReadInto(h, off, buf)
}
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
)
// ErrReadOnly marks an operation a read only export refuses, NFS4ERR_ROFS
// on the wire.
var ErrReadOnly = errors.New("nfsfs: the export is read only")
// A roFS wraps an export and hides its mutating halves. The Writer half
// is gone, so the dispatcher answers NFS4ERR_ROFS for every operation
// that would change anything through it. The optional halves keep their
// reads and refuse their writes with ErrReadOnly, so an export served
// read only still reports its extended attributes, its holes and its
// named attributes, and refuses to touch them.
type roFS struct{ FS }
// ReadOnly serves fs read only: the reads of every half pass through, the
// writes of the mutating halves answer ErrReadOnly, and the mutating
// Writer half disappears from the type.
func ReadOnly(fs FS) FS {
return roFS{fs}
}
// GetXattr reads one named attribute of the object.
func (ro roFS) GetXattr(h Handle, name string, max int) ([]byte, error) {
x, ok := ro.FS.(XattrFS)
if !ok {
return nil, ErrXattrNotSupp
}
return x.GetXattr(h, name, max)
}
// ListXattr names the attributes of the object.
func (ro roFS) ListXattr(h Handle, max int) ([]string, error) {
x, ok := ro.FS.(XattrFS)
if !ok {
return nil, ErrXattrNotSupp
}
return x.ListXattr(h, max)
}
// SetXattr refuses a write on a read only export.
func (ro roFS) SetXattr(h Handle, name string, value []byte, mode uint32) error {
return ErrReadOnly
}
// RemoveXattr refuses a write on a read only export.
func (ro roFS) RemoveXattr(h Handle, name string) error {
return ErrReadOnly
}
// SeekHole reports the first hole at or after the offset.
func (ro roFS) SeekHole(h Handle, offset int64) (int64, bool, error) {
s, ok := ro.FS.(interface {
SeekHole(Handle, int64) (int64, bool, error)
SeekData(Handle, int64) (int64, bool, error)
})
if !ok {
return 0, false, ErrNoSparse
}
return s.SeekHole(h, offset)
}
// SeekData reports the first data byte at or after the offset.
func (ro roFS) SeekData(h Handle, offset int64) (int64, bool, error) {
s, ok := ro.FS.(interface {
SeekHole(Handle, int64) (int64, bool, error)
SeekData(Handle, int64) (int64, bool, error)
})
if !ok {
return 0, false, ErrNoSparse
}
return s.SeekData(h, offset)
}
// Allocate refuses a space reservation on a read only export.
func (ro roFS) Allocate(h Handle, offset, length int64) error {
return ErrReadOnly
}
// Deallocate refuses a hole punch on a read only export.
func (ro roFS) Deallocate(h Handle, offset, length int64) error {
return ErrReadOnly
}
+99
View File
@@ -0,0 +1,99 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfsfs
import (
"errors"
"os"
"path/filepath"
"testing"
)
// roTestRoot builds a backend over a fresh directory holding one file.
func roTestRoot(t *testing.T) (*Local, Handle, string) {
t.Helper()
root := t.TempDir()
l, err := NewLocal(root)
if err != nil {
t.Fatalf("NewLocal: %v", err)
}
p := filepath.Join(root, "file")
if err := os.WriteFile(p, []byte("read only"), 0o644); err != nil {
t.Fatalf("WriteFile: %v", err)
}
rh, err := l.Root()
if err != nil {
t.Fatalf("Root: %v", err)
}
h, _, err := l.Lookup(rh, "file")
if err != nil {
t.Fatalf("Lookup: %v", err)
}
return l, h, p
}
// TestReadOnlyHidesWriter covers the contract the dispatcher relies on: a
// read only export carries no Writer half, so every mutating operation
// answers NFS4ERR_ROFS without the backend ever being asked.
func TestReadOnlyHidesWriter(t *testing.T) {
l, _, _ := roTestRoot(t)
ro := ReadOnly(l)
if _, ok := ro.(Writer); ok {
t.Fatal("the read only export still carries a Writer")
}
var fs FS = l
if _, ok := fs.(Writer); !ok {
t.Fatal("the plain backend lost its Writer")
}
}
// TestReadOnlyRefusesWrites covers the optional halves: their reads pass
// through, their writes answer ErrReadOnly.
func TestReadOnlyRefusesWrites(t *testing.T) {
l, h, _ := roTestRoot(t)
ro := ReadOnly(l)
if err := ro.(XattrFS).SetXattr(h, "user.note", []byte("x"), XattrModeCreate); !errors.Is(err, ErrReadOnly) {
t.Fatalf("SetXattr: %v, want ErrReadOnly", err)
}
if err := ro.(XattrFS).RemoveXattr(h, "user.note"); !errors.Is(err, ErrReadOnly) {
t.Fatalf("RemoveXattr: %v, want ErrReadOnly", err)
}
if err := ro.(interface {
Allocate(Handle, int64, int64) error
Deallocate(Handle, int64, int64) error
}).Allocate(h, 0, 4096); !errors.Is(err, ErrReadOnly) {
t.Fatalf("Allocate: %v, want ErrReadOnly", err)
}
if err := ro.(interface {
Allocate(Handle, int64, int64) error
Deallocate(Handle, int64, int64) error
}).Deallocate(h, 0, 4096); !errors.Is(err, ErrReadOnly) {
t.Fatalf("Deallocate: %v, want ErrReadOnly", err)
}
// The reads pass through: whatever the platform answers for a missing
// attribute and a seeking question, neither is the read only refusal.
if _, err := ro.(XattrFS).GetXattr(h, "user.note", 1024); errors.Is(err, ErrReadOnly) {
t.Fatal("GetXattr refused as a write")
}
if _, _, err := ro.(interface {
SeekHole(Handle, int64) (int64, bool, error)
SeekData(Handle, int64) (int64, bool, error)
}).SeekHole(h, 0); errors.Is(err, ErrReadOnly) {
t.Fatal("SeekHole refused as a write")
}
}
// TestReadOnlyReadsWork covers that the read side of the wrapped export
// is the export itself.
func TestReadOnlyReadsWork(t *testing.T) {
l, h, _ := roTestRoot(t)
ro := ReadOnly(l)
got, err := ro.Read(h, 0, 9)
if err != nil {
t.Fatalf("Read: %v", err)
}
if string(got) != "read only" {
t.Fatalf("Read: %q", got)
}
}
+112
View File
@@ -0,0 +1,112 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build freebsd
package nfsfs
import (
"syscall"
)
// The whence values of the hole seeking of FreeBSD, sys/sys/unistd.h.
const (
seekData = 3
seekHole = 4
)
// SeekHole finds the next hole at or after the offset. The eof flag
// reports that none of the requested content follows: every file
// carries a virtual hole at its end, so a dense tail answers the file
// size with eof set, RFC 7862 section 15.11. ErrBeyondEOF answers a
// request that starts past the end.
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
return l.seek(h, offset, seekHole)
}
// SeekData finds the next data byte at or after the offset. The eof
// flag reports that no data follows the offset, and the answer names
// the file size; ErrBeyondEOF answers a request that starts past the
// end.
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
return l.seek(h, offset, seekData)
}
func (l *Local) seek(h Handle, offset int64, whence int) (int64, bool, error) {
// The seek opens the same revalidated, unfollowed descriptor every
// other data path uses, so a name swapped for a link never serves
// bytes from outside the export.
f, _, err := l.openVerified(h, syscall.O_RDONLY)
if err != nil {
return 0, false, err
}
defer f.Close()
fd := int(f.Fd())
size, err := fstatSize(fd)
if err != nil {
return 0, false, err
}
if offset > size {
return 0, false, ErrBeyondEOF
}
at, err := syscall.Seek(fd, offset, whence)
if err == syscall.ENXIO {
// The range from the offset to the end carries none of the
// requested content: the answer is the end of the file with the
// eof flag set.
return size, true, nil
}
if err != nil {
return 0, false, err
}
if at >= size {
// The virtual hole at the end of every file: the eof flag tells
// the client the search is over.
return size, true, nil
}
return at, false, nil
}
// Allocate reserves a range with posix_fallocate, the reservation call
// of FreeBSD: it grows the file to offset plus length where the range
// runs past the end, the behaviour RFC 7862 section 15.1 requires of
// ALLOCATE. The handle is opened without following a final symlink
// and revalidated against the descriptor before anything is reserved.
func (l *Local) Allocate(h Handle, offset, length int64) error {
kind, _, _, _, err := l.resolve(h)
if err != nil {
return err
}
if kind != typeFile {
return ErrIsDir
}
f, _, err := l.openVerified(h, syscall.O_WRONLY)
if err != nil {
return err
}
defer f.Close()
// posix_fallocate reports its failure as its return value, an
// errno, and never through the system call errno itself.
ret, _, _ := syscall.Syscall6(syscall.SYS_POSIX_FALLOCATE, f.Fd(),
uintptr(offset), uintptr(length), 0, 0, 0)
if ret != 0 {
return syscall.Errno(ret)
}
return nil
}
// Deallocate answers ErrNoSparse: the system call surface of FreeBSD
// carries no call that punches a hole into a range and keeps the file
// size, unlike the fallocate modes of Linux.
func (l *Local) Deallocate(h Handle, offset, length int64) error {
return ErrNoSparse
}
// fstatSize reads the size of the open file.
func fstatSize(fd int) (int64, error) {
var st syscall.Stat_t
if err := syscall.Fstat(fd, &st); err != nil {
return 0, err
}
return st.Size, nil
}
+112
View File
@@ -0,0 +1,112 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"syscall"
)
// The whence values of the hole seeking of Linux.
const (
seekData = 3
seekHole = 4
)
// SeekHole finds the next hole at or after the offset. The eof flag
// reports that none of the requested content follows: every file
// carries a virtual hole at its end, so a dense tail answers the file
// size with eof set, RFC 7862 section 15.11. ErrBeyondEOF answers a
// request that starts past the end.
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
return l.seek(h, offset, seekHole)
}
// SeekData finds the next data byte at or after the offset. The eof
// flag reports that no data follows the offset, and the answer names
// the file size; ErrBeyondEOF answers a request that starts past the
// end.
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
return l.seek(h, offset, seekData)
}
func (l *Local) seek(h Handle, offset int64, whence int) (int64, bool, error) {
// The seek opens the same revalidated, unfollowed descriptor every
// other data path uses, so a name swapped for a link never serves
// bytes from outside the export.
f, _, err := l.openVerified(h, syscall.O_RDONLY)
if err != nil {
return 0, false, err
}
defer f.Close()
fd := int(f.Fd())
size, err := fstatSize(fd)
if err != nil {
return 0, false, err
}
if offset > size {
return 0, false, ErrBeyondEOF
}
at, err := syscall.Seek(fd, offset, whence)
if err == syscall.ENXIO {
// The range from the offset to the end carries none of the
// requested content: the answer is the end of the file with the
// eof flag set.
return size, true, nil
}
if err != nil {
return 0, false, err
}
if at >= size {
// The virtual hole at the end of every file: the eof flag tells
// the client the search is over.
return size, true, nil
}
return at, false, nil
}
// Allocate reserves a range with fallocate in the default mode, which
// grows the file to offset plus length, the behaviour RFC 7862 section
// 15.1 requires of ALLOCATE; Deallocate punches a hole into the range and
// keeps the size. The handle is opened without following a final symlink
// and revalidated against the descriptor before anything is reserved.
func (l *Local) Allocate(h Handle, offset, length int64) error {
return l.fallocate(h, offset, length, 0)
}
// Deallocate punches a hole into the range and keeps the file size.
func (l *Local) Deallocate(h Handle, offset, length int64) error {
return l.fallocate(h, offset, length, 0x03) // KEEP_SIZE|PUNCH_HOLE
}
func (l *Local) fallocate(h Handle, offset, length int64, mode uint32) error {
kind, _, _, _, err := l.resolve(h)
if err != nil {
return err
}
if kind != typeFile {
return ErrIsDir
}
f, _, err := l.openVerified(h, syscall.O_WRONLY)
if err != nil {
return err
}
defer f.Close()
_, _, errno := syscall.Syscall6(syscall.SYS_FALLOCATE, f.Fd(), uintptr(mode),
uintptr(offset), uintptr(length), 0, 0)
if errno != 0 {
return errno
}
return nil
}
// fstatSize reads the size of the open file.
func fstatSize(fd int) (int64, error) {
var st syscall.Stat_t
if err := syscall.Fstat(fd, &st); err != nil {
return 0, err
}
return st.Size, nil
}
+74
View File
@@ -0,0 +1,74 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"os"
"path/filepath"
"testing"
)
func TestLocalSparse(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "s.bin"), make([]byte, 8192), 0o644); err != nil {
t.Fatal(err)
}
l, err := NewLocal(root)
if err != nil {
t.Fatal(err)
}
rootHandle, err := l.Root()
if err != nil {
t.Fatal(err)
}
fh, _, err := l.Lookup(rootHandle, "s.bin")
if err != nil {
t.Fatal(err)
}
// The file is all data: the only hole is the virtual one every file
// carries at its end, so the seek answers the size with the eof flag,
// RFC 7862 section 15.11.
off, eof, err := l.SeekHole(fh, 0)
if err != nil || off != 8192 || !eof {
t.Fatalf("hole in a full file: %d %v %v", off, eof, err)
}
if off, eof, err := l.SeekData(fh, 0); err != nil || off != 0 || eof {
t.Fatalf("seek data: %d %v %v", off, eof, err)
}
// A seek past the end is NXIO, not an answer.
if _, _, err := l.SeekData(fh, 8193); err != ErrBeyondEOF {
t.Fatalf("seek past the end: %v", err)
}
// Punching a hole moves the first hole to the punched offset.
if err := l.Deallocate(fh, 4096, 4096); err != nil {
t.Fatalf("deallocate: %v", err)
}
off, eof, err = l.SeekHole(fh, 0)
if err != nil || off != 4096 || eof {
t.Fatalf("seek hole: %d %v %v", off, eof, err)
}
// No data follows inside the hole: the answer names the size with
// the eof flag.
if off, eof, err := l.SeekData(fh, 4096); err != nil || off != 8192 || !eof {
t.Fatalf("data in the hole: %d %v %v", off, eof, err)
}
// Allocate reserves the space without moving the hole back.
if err := l.Allocate(fh, 4096, 4096); err != nil {
t.Fatalf("allocate: %v", err)
}
if off, eof, err := l.SeekHole(fh, 0); err != nil || off != 4096 || eof {
t.Fatalf("seek hole after allocate: %d %v %v", off, eof, err)
}
// Allocate past the end grows the file to the end of the reserved
// range, which RFC 7862 section 15.1 requires.
if err := l.Allocate(fh, 8192, 4096); err != nil {
t.Fatalf("allocate past eof: %v", err)
}
info, err := l.Getattr(fh)
if err != nil || info.Size != 12288 {
t.Fatalf("size after the allocate past eof: %d, %v", info.Size, err)
}
}
+22
View File
@@ -0,0 +1,22 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build !linux && !freebsd
package nfsfs
// SeekHole needs a system hole seeking call.
func (l *Local) SeekHole(h Handle, offset int64) (int64, bool, error) {
return 0, false, ErrNoSparse
}
// SeekData needs a system hole seeking call.
func (l *Local) SeekData(h Handle, offset int64) (int64, bool, error) {
return 0, false, ErrNoSparse
}
// Allocate needs a system space reservation call.
func (l *Local) Allocate(h Handle, offset, length int64) error { return ErrNoSparse }
// Deallocate needs a system hole punching call.
func (l *Local) Deallocate(h Handle, offset, length int64) error { return ErrNoSparse }
+23
View File
@@ -0,0 +1,23 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build freebsd || netbsd || openbsd || darwin
package nfsfs
import (
"errors"
"syscall"
)
// symlinkRefused reports whether an open refused a final component that is
// a symlink, the answer O_NOFOLLOW exists for. The systems disagree on the
// answer: FreeBSD and OpenBSD answer EMLINK, NetBSD answers EFTYPE, darwin
// names EMLINK in its open(2), and the ELOOP spelling stays in the check
// for the lineage it came from. FreeBSD, OpenBSD and NetBSD are verified
// live; the darwin answer rests on its manual.
func symlinkRefused(err error) bool {
return errors.Is(err, syscall.ELOOP) ||
errors.Is(err, syscall.EMLINK) ||
errors.Is(err, syscall.EFTYPE)
}
+17
View File
@@ -0,0 +1,17 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"errors"
"syscall"
)
// symlinkRefused reports whether an open refused a final component that is
// a symlink, the answer O_NOFOLLOW exists for. Linux answers ELOOP.
func symlinkRefused(err error) bool {
return errors.Is(err, syscall.ELOOP)
}
+253
View File
@@ -0,0 +1,253 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build freebsd
package nfsfs
import (
"os"
"strings"
"syscall"
"unsafe"
)
// xattrNamespace is the only namespace this backend serves: RFC 8276
// section 3.3 names the attributes with their namespace, and the local
// mapping of the FreeBSD server is the extattr user namespace.
const xattrNamespace = "user."
// extattrNamespaceUser is EXTATTR_NAMESPACE_USER of sys/sys/extattr.h,
// the numeric namespace the extattr family of system calls takes. The
// syscall package of FreeBSD exports the traps of the family but not
// this constant.
const extattrNamespaceUser = 0x1
// xattrPath resolves a handle for the extattr family: the registered path
// must still name the handle's device, inode and kind. A symlink is
// refused the way the kernel refuses the user namespace on one, so the
// path calls never follow a link out of the export.
func (l *Local) xattrPath(h Handle) (string, error) {
_, path, fi, err := l.revalidate(h)
if err != nil {
return "", err
}
if fi.Mode()&os.ModeSymlink != 0 {
return "", syscall.EPERM
}
return path, nil
}
// GetXattr reads one named attribute of the object.
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
path, err := l.xattrPath(h)
if err != nil {
return nil, err
}
attr, ok := strings.CutPrefix(name, xattrNamespace)
if !ok {
return nil, ErrNoXattr
}
buf := make([]byte, maxOr(max, 256))
for {
n, err := extattrGetFile(path, attr, buf)
if err == syscall.ERANGE {
if len(buf) > 1<<20 {
return nil, syscall.ERANGE
}
buf = make([]byte, len(buf)*2)
continue
}
if err == syscall.ENOATTR {
return nil, ErrNoXattr
}
if err != nil {
return nil, err
}
return buf[:n], nil
}
}
// SetXattr writes one named attribute under the RFC 8276 mode. The
// extattr interface of FreeBSD carries no create and replace flags, so
// the two strict modes ask for the attribute first and write it after:
// a create of an existing name answers EEXIST and a replace of a
// missing one ENOATTR, the same answers the flags of Linux produce.
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
path, err := l.xattrPath(h)
if err != nil {
return err
}
attr, ok := strings.CutPrefix(name, xattrNamespace)
if !ok {
return syscall.EOPNOTSUPP
}
switch mode {
case XattrModeCreate, XattrModeReplace:
_, err := extattrGetFile(path, attr, nil)
switch {
case err == nil && mode == XattrModeCreate:
return syscall.EEXIST
case err == syscall.ENOATTR && mode == XattrModeReplace:
return syscall.ENOATTR
case err != nil && err != syscall.ENOATTR:
return err
}
}
return extattrSetFile(path, attr, value)
}
// ListXattr names the user namespace attributes of the object. The list
// of extattr_list_file is a sequence of one length byte and name pairs
// with no separators, extattr(2), and its names carry no namespace, so
// each name is dressed with the namespace prefix the protocol speaks.
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
path, err := l.xattrPath(h)
if err != nil {
return nil, err
}
buf := make([]byte, maxOr(max, 1024))
for {
n, err := extattrListFile(path, buf)
if err == syscall.ERANGE {
if len(buf) > 1<<20 {
return nil, syscall.ERANGE
}
buf = make([]byte, len(buf)*2)
continue
}
if err != nil {
return nil, err
}
buf = buf[:n]
break
}
var names []string
for len(buf) > 0 {
size := int(buf[0])
if size+1 > len(buf) {
break
}
names = append(names, xattrNamespace+string(buf[1:1+size]))
buf = buf[1+size:]
}
return names, nil
}
// RemoveXattr deletes one named attribute.
func (l *Local) RemoveXattr(h Handle, name string) error {
path, err := l.xattrPath(h)
if err != nil {
return err
}
attr, ok := strings.CutPrefix(name, xattrNamespace)
if !ok {
return ErrNoXattr
}
if err := extattrDeleteFile(path, attr); err == syscall.ENOATTR {
return ErrNoXattr
} else if err != nil {
return err
}
return nil
}
// maxOr replaces a zero budget with the given default.
func maxOr(max, def int) int {
if max == 0 || max > 1<<20 {
return def
}
return max
}
// extattrGetFile reads the attribute attr of path into buf, or names its
// size when buf is nil, extattr(2).
func extattrGetFile(path, attr string, buf []byte) (int, error) {
name, err := syscall.ByteSliceFromString(attr)
if err != nil {
return 0, err
}
// SAFETY: the kernel reads the buffer for the length of the call
// only, and data stays alive through the unsafe pointer until the
// system call returns.
var data unsafe.Pointer
if len(buf) > 0 {
data = unsafe.Pointer(&buf[0])
}
n, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_GET_FILE,
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
extattrNamespaceUser,
uintptr(unsafe.Pointer(&name[0])),
uintptr(data),
uintptr(len(buf)),
0)
if errno != 0 {
return 0, errno
}
return int(n), nil
}
// extattrSetFile writes value into the attribute attr of path,
// extattr(2).
func extattrSetFile(path, attr string, value []byte) error {
name, err := syscall.ByteSliceFromString(attr)
if err != nil {
return err
}
// SAFETY: the kernel reads the buffer for the length of the call
// only, and value stays alive through the unsafe pointer until the
// system call returns.
var data unsafe.Pointer
if len(value) > 0 {
data = unsafe.Pointer(&value[0])
}
_, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_SET_FILE,
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
extattrNamespaceUser,
uintptr(unsafe.Pointer(&name[0])),
uintptr(data),
uintptr(len(value)),
0)
if errno != 0 {
return errno
}
return nil
}
// extattrListFile names the user namespace attributes of path into buf,
// or names the size of the list when buf is nil, extattr(2).
func extattrListFile(path string, buf []byte) (int, error) {
// SAFETY: the kernel writes the buffer for the length of the call
// only, and buf stays alive through the unsafe pointer until the
// system call returns.
var data unsafe.Pointer
if len(buf) > 0 {
data = unsafe.Pointer(&buf[0])
}
n, _, errno := syscall.Syscall6(syscall.SYS_EXTATTR_LIST_FILE,
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
extattrNamespaceUser,
uintptr(data),
uintptr(len(buf)),
0, 0)
if errno != 0 {
return 0, errno
}
return int(n), nil
}
// extattrDeleteFile takes the attribute attr off path, extattr(2).
func extattrDeleteFile(path, attr string) error {
name, err := syscall.ByteSliceFromString(attr)
if err != nil {
return err
}
_, _, errno := syscall.Syscall(syscall.SYS_EXTATTR_DELETE_FILE,
uintptr(unsafe.Pointer(syscall.StringBytePtr(path))),
extattrNamespaceUser,
uintptr(unsafe.Pointer(&name[0])))
if errno != 0 {
return errno
}
return nil
}
+140
View File
@@ -0,0 +1,140 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"os"
"strings"
"syscall"
)
// xattrNamespace is the only namespace this backend serves: RFC 8276
// section 3.3 names the attributes with their namespace, and the local
// mapping of the Linux server is the user namespace.
const xattrNamespace = "user."
// Flags of the xattr system calls.
const (
xattrCreateFlag = 0x1
xattrReplaceFlag = 0x2
)
// xattrPath resolves a handle for the xattr family: the registered path
// must still name the handle's device, inode and kind. A symlink is
// refused the way the kernel refuses the user namespace on one, so the
// path calls never follow a link out of the export.
func (l *Local) xattrPath(h Handle) (string, error) {
_, path, fi, err := l.revalidate(h)
if err != nil {
return "", err
}
if fi.Mode()&os.ModeSymlink != 0 {
return "", syscall.EPERM
}
return path, nil
}
// GetXattr reads one named attribute of the object.
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
path, err := l.xattrPath(h)
if err != nil {
return nil, err
}
if !strings.HasPrefix(name, xattrNamespace) {
return nil, ErrNoXattr
}
buf := make([]byte, maxOr(max, 256))
for {
n, err := syscall.Getxattr(path, name, buf)
if err == syscall.ERANGE {
if len(buf) > 1<<20 {
return nil, syscall.ERANGE
}
buf = make([]byte, len(buf)*2)
continue
}
if err == syscall.ENODATA {
return nil, ErrNoXattr
}
if err != nil {
return nil, err
}
return buf[:n], nil
}
}
// SetXattr writes one named attribute under the RFC 8276 mode.
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
path, err := l.xattrPath(h)
if err != nil {
return err
}
if !strings.HasPrefix(name, xattrNamespace) {
return syscall.EOPNOTSUPP
}
flags := 0
switch mode {
case XattrModeCreate:
flags = xattrCreateFlag
case XattrModeReplace:
flags = xattrReplaceFlag
}
return syscall.Setxattr(path, name, value, flags)
}
// ListXattr names the user namespace attributes of the object.
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
path, err := l.xattrPath(h)
if err != nil {
return nil, err
}
buf := make([]byte, maxOr(max, 1024))
for {
n, err := syscall.Listxattr(path, buf)
if err == syscall.ERANGE {
if len(buf) > 1<<20 {
return nil, syscall.ERANGE
}
buf = make([]byte, len(buf)*2)
continue
}
if err != nil {
return nil, err
}
var names []string
for entry := range strings.SplitSeq(string(buf[:n]), "\x00") {
if strings.HasPrefix(entry, xattrNamespace) {
names = append(names, entry)
}
}
return names, nil
}
}
// RemoveXattr deletes one named attribute.
func (l *Local) RemoveXattr(h Handle, name string) error {
path, err := l.xattrPath(h)
if err != nil {
return err
}
if !strings.HasPrefix(name, xattrNamespace) {
return ErrNoXattr
}
if err := syscall.Removexattr(path, name); err == syscall.ENODATA {
return ErrNoXattr
} else if err != nil {
return err
}
return nil
}
// maxOr replaces a zero budget with the given default.
func maxOr(max, def int) int {
if max == 0 || max > 1<<20 {
return def
}
return max
}
+64
View File
@@ -0,0 +1,64 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build linux
package nfsfs
import (
"os"
"path/filepath"
"testing"
)
func TestLocalXattr(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "a.txt"), []byte("x"), 0o644); err != nil {
t.Fatal(err)
}
l, err := NewLocal(root)
if err != nil {
t.Fatal(err)
}
rootHandle, err := l.Root()
if err != nil {
t.Fatal(err)
}
fh, _, err := l.Lookup(rootHandle, "a.txt")
if err != nil {
t.Fatal(err)
}
if _, err := l.GetXattr(fh, "user.tag", 0); err != ErrNoXattr {
t.Fatalf("missing attribute: %v", err)
}
if err := l.SetXattr(fh, "user.tag", []byte("value"), XattrModeCreate); err != nil {
t.Fatalf("create: %v", err)
}
if err := l.SetXattr(fh, "user.tag", []byte("again"), XattrModeCreate); err == nil {
t.Fatal("create over a live attribute succeeded")
}
got, err := l.GetXattr(fh, "user.tag", 0)
if err != nil || string(got) != "value" {
t.Fatalf("value after the refused create %q: %v", got, err)
}
if err := l.SetXattr(fh, "user.tag", []byte("again"), XattrModeReplace); err != nil {
t.Fatalf("replace: %v", err)
}
got, err = l.GetXattr(fh, "user.tag", 0)
if err != nil || string(got) != "again" {
t.Fatalf("value after the replace %q: %v", got, err)
}
names, err := l.ListXattr(fh, 0)
if err != nil || len(names) != 1 || names[0] != "user.tag" {
t.Fatalf("names %v: %v", names, err)
}
if err := l.SetXattr(fh, "system.nfs", []byte("x"), XattrModeCreate); err == nil {
t.Fatal("a foreign namespace was accepted")
}
if err := l.RemoveXattr(fh, "user.tag"); err != nil {
t.Fatalf("remove: %v", err)
}
if err := l.RemoveXattr(fh, "user.tag"); err != ErrNoXattr {
t.Fatalf("remove again: %v", err)
}
}
+27
View File
@@ -0,0 +1,27 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
//go:build !linux && !freebsd
package nfsfs
// GetXattr answers that the backend carries no extended attributes; the
// platforms without a system xattr interface serve none.
func (l *Local) GetXattr(h Handle, name string, max int) ([]byte, error) {
return nil, ErrXattrNotSupp
}
// SetXattr answers that the backend carries no extended attributes.
func (l *Local) SetXattr(h Handle, name string, value []byte, mode uint32) error {
return ErrXattrNotSupp
}
// ListXattr answers that the backend carries no extended attributes.
func (l *Local) ListXattr(h Handle, max int) ([]string, error) {
return nil, ErrXattrNotSupp
}
// RemoveXattr answers that the backend carries no extended attributes.
func (l *Local) RemoveXattr(h Handle, name string) error {
return ErrXattrNotSupp
}