// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT // The handlers of the NFSv4.2 operations, RFC 7862. package nfs4server import ( "errors" "sourcedock.dev/petrbalvin/nfs/internal/nfs4" "sourcedock.dev/petrbalvin/nfs/internal/nfsfs" "sourcedock.dev/petrbalvin/nfs/internal/xdr" ) // A seeker is the backend's hole seeking half: the answer offset, the // eof flag that names the virtual hole at the end of every file, and // ErrBeyondEOF for a request past the end, RFC 7862 section 15.11. type seeker interface { SeekHole(nfsfs.Handle, int64) (int64, bool, error) SeekData(nfsfs.Handle, int64) (int64, bool, error) } // seeker resolves the backend's hole seeking half. func (h *Handler) seeker() seeker { s, _ := h.FS.(seeker) return s } // allocator resolves the backend's space reservation half. func (h *Handler) allocator() interface { Allocate(nfsfs.Handle, int64, int64) error Deallocate(nfsfs.Handle, int64, int64) error } { a, _ := h.FS.(interface { Allocate(nfsfs.Handle, int64, int64) error Deallocate(nfsfs.Handle, int64, int64) error }) return a } // readStateid pulls one stateid off the wire. func readStateid(d *xdr.Decoder) (nfs4.Stateid, error) { var st nfs4.Stateid raw, err := d.Raw(16) if err != nil { return st, err } copy(st[:], raw) return st, nil } // checkOpStateid validates the stateid of a stateful data operation: // it routes by the family mark to the store that minted it, the // caller's own open of the file, the caller's own lock state on it, or // the caller's own delegation of it, RFC 8881 section 10.3. func (h *Handler) checkOpStateid(st nfs4.Stateid, fh nfsfs.Handle, clientid uint64) uint32 { switch string(st[4:8]) { case "LOCK": if _, status := h.locks().byStateid(st, fh, clientid); status != nfs4.ErrOK { return status } return nfs4.ErrOK case "DELE": return h.delegs().checkDataStateid(st, fh, clientid) default: _, status := h.openStates().checkStateid(st, fh, clientid) return status } } // seekOp serves SEEK: the next hole or data byte from the offset. func (h *Handler) seekOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } offset, err := d.Uint64() if err != nil { return nil, 0, err } what, err := d.Uint32() if err != nil { return nil, 0, err } if what != nfs4.ContentData && what != nfs4.ContentHole { return nil, nfs4.ErrInval, nil } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } s := h.seeker() if s == nil { return nil, nfs4.ErrNotSupp, nil } seek := s.SeekData if what == nfs4.ContentHole { seek = s.SeekHole } if offset > 1<<62 { return nil, nfs4.ErrNXIO, nil } found, eof, serr := seek(reg.cur, int64(offset)) if serr == nfsfs.ErrBeyondEOF { return nil, nfs4.ErrNXIO, nil } if serr != nil { return nil, mapErr(serr), nil } return nfs4.AppendSeekRes(nil, eof, uint64(found)), nfs4.ErrOK, nil } // rangeOp serves ALLOCATE and DEALLOCATE through the backend's space // reservation half. func (h *Handler) rangeOp(d *xdr.Decoder, reg *fhreg, deallocate bool) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } offset, err := d.Uint64() if err != nil { return nil, 0, err } length, err := d.Uint64() if err != nil { return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } a := h.allocator() if a == nil { return nil, nfs4.ErrNotSupp, nil } if length == 0 { return nil, nfs4.ErrInval, nil } if deallocate { err = a.Deallocate(reg.cur, int64(offset), int64(length)) } else { err = a.Allocate(reg.cur, int64(offset), int64(length)) } if err != nil { return nil, mapErr(err), nil } return nil, nfs4.ErrOK, nil } // ioAdviseOp serves IO_ADVISE: the hints ride through, no state kept. func (h *Handler) ioAdviseOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } if _, err = d.Uint64(); err != nil { // offset return nil, 0, err } if _, err = d.Uint64(); err != nil { // count return nil, 0, err } hints, err := nfs4.ReadBitmap(d) if err != nil { return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } return nfs4.AppendIoAdviseRes(nil, hints), nfs4.ErrOK, nil } // copyOp serves COPY between two files of this server: the saved handle // is the source, the current one the destination, the copy runs // synchronously in this compound. func (h *Handler) copyOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { src, err := readStateid(d) if err != nil { return nil, 0, err } dst, err := readStateid(d) if err != nil { return nil, 0, err } srcOff, err := d.Uint64() if err != nil { return nil, 0, err } dstOff, err := d.Uint64() if err != nil { return nil, 0, err } count, err := d.Uint64() if err != nil { return nil, 0, err } consecutive, err := d.Bool() if err != nil { return nil, 0, err } synchronous, err := d.Bool() if err != nil { return nil, 0, err } sources, err := readNetlocList(d) if err != nil { return nil, 0, err } _ = sources if !reg.haveCur || reg.saved == nil { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(src, reg.saved, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } if status := h.checkOpStateid(dst, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } if count > nfs4.DefaultLimits.MaxRead { return nil, nfs4.ErrTooSmall, nil } // The kernel path moves the bytes without userspace touching them // where the backend and the filesystem provide it; anywhere else the // userspace loop carries the copy, the answer being the same. var moved uint64 if cl, ok := h.FS.(nfsfs.RangeCloner); ok { if err := cl.CopyRange(reg.saved, int64(srcOff), reg.cur, int64(dstOff), int64(count)); err == nil { moved = count } } if moved == 0 { data, rerr := h.FS.Read(reg.saved, int64(srcOff), int(count)) if rerr != nil { return nil, mapErr(rerr), nil } w := h.writer() if w == nil { return nil, nfs4.ErrROFS, nil } n, werr := w.Write(reg.cur, int64(dstOff), data) if werr != nil { return nil, mapErr(werr), nil } moved = uint64(n) } return nfs4.AppendCopyRes(nil, nfs4.Stateid{}, false, moved, nfs4.NfsSyncFileSync, h.writeVerifier(), consecutive, synchronous), nfs4.ErrOK, nil } // readNetlocList decodes the netloc4 list the copy family carries. func readNetlocList(d *xdr.Decoder) ([]nfs4.CopySourceServer, error) { n, err := d.Uint32() if err != nil { return nil, err } var out []nfs4.CopySourceServer for range n { var s nfs4.CopySourceServer if s.Type, err = d.Uint32(); err != nil { return nil, err } switch s.Type { case 1, 2: if s.Name, err = d.String(); err != nil { return nil, err } case 3: if s.Addr.Netid, err = d.String(); err != nil { return nil, err } if s.Addr.Uaddr, err = d.String(); err != nil { return nil, err } default: return nil, errors.New("nfs4server: unknown netloc type") } out = append(out, s) } return out, nil } // copyNotifyOp serves COPY_NOTIFY: the source server grants the copy to // the named destination. func (h *Handler) copyNotifyOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } dstType, err := d.Uint32() if err != nil { return nil, 0, err } var dst nfs4.CopySourceServer dst.Type = dstType switch dstType { case 1, 2: if dst.Name, err = d.String(); err != nil { return nil, 0, err } case 3: if dst.Addr.Netid, err = d.String(); err != nil { return nil, 0, err } if dst.Addr.Uaddr, err = d.String(); err != nil { return nil, 0, err } default: return nil, nfs4.ErrInval, nil } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } setStateidSeq(&st, 1) copy(st[4:], "CPYN") return nfs4.AppendCopyNotifyRes(nil, int64(h.leasePeriod().Seconds()), st, []nfs4.CopySourceServer{{Type: 3, Addr: nfs4.NetAddr{Netid: "tcp", Uaddr: h.deviceAddr(nil)}}}), nfs4.ErrOK, nil } // offloadCancelOp serves OFFLOAD_CANCEL: nothing async is in flight in // this build, so the stateid is checked and the call answered. func (h *Handler) offloadCancelOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } return nil, nfs4.ErrOK, nil } // offloadStatusOp serves OFFLOAD_STATUS: no asynchronous copy was ever // requested, which the standard answers OFFLOAD_NO_REQS for. func (h *Handler) offloadStatusOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { if _, err := readStateid(d); err != nil { return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } return nil, nfs4.ErrOffloadNoReqs, nil } // cloneOp serves CLONE between two files of this server: the saved // handle is the source, the current one the destination. func (h *Handler) cloneOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { src, err := readStateid(d) if err != nil { return nil, 0, err } dst, err := readStateid(d) if err != nil { return nil, 0, err } srcOff, err := d.Uint64() if err != nil { return nil, 0, err } dstOff, err := d.Uint64() if err != nil { return nil, 0, err } count, err := d.Uint64() if err != nil { return nil, 0, err } if !reg.haveCur || reg.saved == nil { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(src, reg.saved, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } if status := h.checkOpStateid(dst, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } // The offsets are signed on the wire in effect: a value beyond the // signed range cannot name a byte of any file this server serves, // and a count beyond the read limit would size one allocation from // the request. Both are refused before anything is read. if srcOff > 1<<62 || dstOff > 1<<62 || count > nfs4.DefaultLimits.MaxRead { return nil, nfs4.ErrInval, nil } srcInfo, gerr := h.FS.Getattr(reg.saved) if gerr != nil { return nil, mapErr(gerr), nil } dstInfo, gerr := h.FS.Getattr(reg.cur) if gerr != nil { return nil, mapErr(gerr), nil } // The ends are checked without addition, so a wrapped sum can never // slip past the size guard. if uint64(srcInfo.Size) < srcOff || uint64(srcInfo.Size)-srcOff < count || uint64(dstInfo.Size) < dstOff || uint64(dstInfo.Size)-dstOff < count { return nil, nfs4.ErrInval, nil } // CLONE is the reflink of the NFSv4.2 world: where the filesystem // provides it the kernel shares the bytes, and where it does not the // userspace copy answers instead, the same result either way. if cl, ok := h.FS.(nfsfs.RangeCloner); ok { if err := cl.CloneRange(reg.saved, int64(srcOff), reg.cur, int64(dstOff), int64(count)); err == nil { return nil, nfs4.ErrOK, nil } } data, rerr := h.FS.Read(reg.saved, int64(srcOff), int(count)) if rerr != nil { return nil, mapErr(rerr), nil } w := h.writer() if w == nil { return nil, nfs4.ErrROFS, nil } if _, werr := w.Write(reg.cur, int64(dstOff), data); werr != nil { return nil, mapErr(werr), nil } return nil, nfs4.ErrOK, nil } // layoutErrorOp serves LAYOUTERROR: the report is recorded and answered. func (h *Handler) layoutErrorOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { if _, err := d.Uint64(); err != nil { // offset return nil, 0, err } if _, err := d.Uint64(); err != nil { // length return nil, 0, err } if _, err := readStateid(d); err != nil { return nil, 0, err } n, err := d.Uint32() if err != nil { return nil, 0, err } for range n { if _, err = d.Raw(16); err != nil { // device id return nil, 0, err } if _, err = d.Uint32(); err != nil { // status return nil, 0, err } if _, err = d.Uint32(); err != nil { // opnum return nil, 0, err } } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } return nil, nfs4.ErrOK, nil } // layoutStatsOp serves LAYOUTSTATS: the counters are accepted and the // call answered. func (h *Handler) layoutStatsOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { for range 2 { if _, err := d.Uint64(); err != nil { // offset, length return nil, 0, err } } if _, err := readStateid(d); err != nil { return nil, 0, err } for range 4 { if _, err := d.Uint64(); err != nil { // io_info4 pairs return nil, 0, err } } if _, err := d.Raw(16); err != nil { // device id return nil, 0, err } if _, err := d.Uint32(); err != nil { // layout update type return nil, 0, err } if _, err := d.VarOpaque(); err != nil { // layout update body return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } return nil, nfs4.ErrOK, nil } // readPlusOp serves READ_PLUS: this backend has no sparse knowledge on // the read path, so the answer is one data segment. func (h *Handler) readPlusOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } offset, err := d.Uint64() if err != nil { return nil, 0, err } count, err := d.Uint32() if err != nil { return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } if uint64(count) > nfs4.DefaultLimits.MaxRead { count = uint32(nfs4.DefaultLimits.MaxRead) } data, rerr := h.FS.Read(reg.cur, int64(offset), int(count)) if rerr != nil { return nil, mapErr(rerr), nil } eof := false if info, gerr := h.FS.Getattr(reg.cur); gerr == nil { eof = int64(offset)+int64(len(data)) >= info.Size } return nfs4.AppendReadPlusDataRes(nil, eof, offset, data), nfs4.ErrOK, nil } // writeSameOp serves WRITE_SAME: the application data block pattern is // repeated over the block count at the offset. func (h *Handler) writeSameOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { st, err := readStateid(d) if err != nil { return nil, 0, err } stable, err := d.Uint32() if err != nil { return nil, 0, err } offset, err := d.Uint64() if err != nil { return nil, 0, err } blockSize, err := d.Uint64() if err != nil { return nil, 0, err } blockCount, err := d.Uint64() if err != nil { return nil, 0, err } if _, err = d.Uint64(); err != nil { // adb_reloff_blocknum return nil, 0, err } if _, err = d.Uint32(); err != nil { // adb_block_num, count4 return nil, 0, err } if _, err = d.Uint64(); err != nil { // adb_reloff_pattern return nil, 0, err } pattern, err := d.VarOpaque() if err != nil { return nil, 0, err } _ = stable if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if status := h.checkOpStateid(st, reg.cur, reg.clientID); status != nfs4.ErrOK { return nil, status, nil } // The block size and the total the operation may write are bounded // before anything is allocated: a wire controlled size beyond the // write limit is refused, never used as an allocation length. if blockSize == 0 || blockCount == 0 || len(pattern) == 0 { return nil, nfs4.ErrInval, nil } if blockSize > nfs4.DefaultLimits.MaxWrite || blockCount > 1<<20 || blockSize*blockCount > nfs4.DefaultLimits.MaxWrite { return nil, nfs4.ErrInval, nil } w := h.writer() if w == nil { return nil, nfs4.ErrROFS, nil } block := make([]byte, 0, blockSize) for len(block) < int(blockSize) { block = append(block, pattern...) } block = block[:blockSize] total := int64(0) for range blockCount { if _, err := w.Write(reg.cur, int64(offset)+total, block); err != nil { return nil, mapErr(err), nil } total += int64(blockSize) } return nfs4.AppendWriteSameRes(nil, uint64(total), nfs4.NfsSyncFileSync, h.writeVerifier()), nfs4.ErrOK, nil }