Files
petrbalvin a9b8039ef7
Test / test (push) Successful in 2m4s
Release / gates (push) Successful in 2m5s
Release / build (amd64, freebsd) (push) Successful in 1m27s
Release / build (amd64, linux) (push) Successful in 1m22s
Release / build (amd64, netbsd) (push) Successful in 1m19s
Release / build (amd64, openbsd) (push) Successful in 1m20s
Release / build (arm64, darwin) (push) Successful in 1m21s
Release / build (arm64, freebsd) (push) Successful in 1m26s
Release / build (arm64, linux) (push) Successful in 1m25s
Release / build (arm64, netbsd) (push) Successful in 1m31s
Release / build (arm64, openbsd) (push) Successful in 1m27s
Release / build (loong64, linux) (push) Successful in 1m37s
Release / build (riscv64, linux) (push) Successful in 1m21s
Release / release (push) Successful in 40s
feat: full NFSv4.2 server and client in pure Go
Assisted-by: GLM 5.3 Flash
2026-09-21 18:51:17 +02:00

538 lines
15 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package nfs4server
import (
"net"
"sync"
"sourcedock.dev/petrbalvin/nfs/internal/nfs4"
"sourcedock.dev/petrbalvin/nfs/internal/nfsfs"
"sourcedock.dev/petrbalvin/nfs/internal/xdr"
)
// layoutDeviceID names the one storage device behind every layout: the
// metadata server of this build is also its data server, so one identity
// serves both roles.
var layoutDeviceID = newDeviceID()
// newDeviceID builds the fixed device identity of this server.
func newDeviceID() [16]byte {
var id [16]byte
copy(id[:], "pnfs42mdsds00001")
return id
}
// A layout is one granted pNFS layout: the session and client it belongs
// to, the file it covers, the byte range and the stateid the client
// addresses it by.
type layout struct {
stateid nfs4.Stateid
sessID nfs4.SessionID
clientid uint64
fh nfsfs.Handle
offset uint64
length uint64
}
// layoutServer tracks the layouts the metadata server has granted. The
// store lives in memory: a server restart drops every layout and the
// clients re-request them through the grace window of the restart.
type layoutServer struct {
mu sync.Mutex
next uint64
layouts map[string]*layout // by layout stateid other
}
func newLayoutServer() *layoutServer {
return &layoutServer{next: randCounter(), layouts: make(map[string]*layout)}
}
// grant issues a layout over the requested range of the file handle,
// replacing the layout the same client already holds on the same file,
// so a re-request or a retry never piles entries up. The stateid other
// carries the LAYOUT mark and a counter; the sequence field starts at
// one, as RFC 8881 section 12.5.2 has it for layout stateids.
func (l *layoutServer) grant(sessID nfs4.SessionID, clientid uint64, fh nfsfs.Handle, offset, length uint64) *layout {
l.mu.Lock()
defer l.mu.Unlock()
key := fileKey(fh)
for _, lay := range l.layouts {
if lay.clientid == clientid && fileKey(lay.fh) == key {
lay.offset, lay.length, lay.sessID = offset, length, sessID
return lay
}
}
l.next++
var st nfs4.Stateid
setStateidSeq(&st, 1)
copy(st[4:], "LAYOUT")
for i := range 6 {
st[15-i] = byte(l.next >> (8 * i))
}
lay := &layout{stateid: st, sessID: sessID, clientid: clientid, fh: fh, offset: offset, length: length}
l.layouts[string(st[4:])] = lay
return lay
}
// lookup finds a live layout by the stateid the client presents.
func (l *layoutServer) lookup(st nfs4.Stateid, fh nfsfs.Handle) (*layout, uint32) {
l.mu.Lock()
defer l.mu.Unlock()
lay, ok := l.layouts[string(st[4:])]
if !ok || fileKey(lay.fh) != fileKey(fh) {
return nil, nfs4.ErrBadStateid
}
return lay, nfs4.ErrOK
}
// drop removes the layout identified by the stateid and answers the status
// of the removal.
func (l *layoutServer) drop(st nfs4.Stateid) uint32 {
l.mu.Lock()
defer l.mu.Unlock()
if _, ok := l.layouts[string(st[4:])]; !ok {
return nfs4.ErrBadStateid
}
delete(l.layouts, string(st[4:]))
return nfs4.ErrOK
}
// dropSession drops every layout of one session, which DESTROY_SESSION
// requires: layouts are session bound state.
func (l *layoutServer) dropSession(sessID nfs4.SessionID) {
l.mu.Lock()
defer l.mu.Unlock()
for other, lay := range l.layouts {
if lay.sessID == sessID {
delete(l.layouts, other)
}
}
}
// dropClient drops every layout of one client, which DESTROY_CLIENTID
// requires.
func (l *layoutServer) dropClient(clientid uint64) {
l.mu.Lock()
defer l.mu.Unlock()
for other, lay := range l.layouts {
if lay.clientid == clientid {
delete(l.layouts, other)
}
}
}
// count reports how many layouts the server has granted.
func (l *layoutServer) count() int {
l.mu.Lock()
defer l.mu.Unlock()
return len(l.layouts)
}
// layoutGetOp serves LAYOUTGET: it validates the open stateid the client
// presents, grants a flexfiles layout over the requested range and names
// this server as the data server the client reads and writes through.
func (h *Handler) layoutGetOp(d *xdr.Decoder, reg *fhreg, sessID nfs4.SessionID, clientid uint64) ([]byte, uint32, error) {
if _, err := d.Bool(); err != nil { // signal_avail
return nil, 0, err
}
layoutType, err := d.Uint32()
if err != nil {
return nil, 0, err
}
iomode, err := d.Uint32()
if err != nil {
return nil, 0, err
}
offset, err := d.Uint64()
if err != nil {
return nil, 0, err
}
length, err := d.Uint64()
if err != nil {
return nil, 0, err
}
if _, err = d.Uint64(); err != nil { // minlength
return nil, 0, err
}
var openSt nfs4.Stateid
raw, rerr := d.Raw(16)
if rerr != nil {
return nil, 0, rerr
}
copy(openSt[:], raw)
maxcount, err := d.Uint32()
if err != nil {
return nil, 0, err
}
if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 &&
layoutType != nfs4.LayoutTypeFiles && layoutType != nfs4.LayoutTypeBlock &&
layoutType != nfs4.LayoutTypeObjects && layoutType != nfs4.LayoutTypeScsi {
return nil, nfs4.ErrUnknownLayoutType, nil
}
if iomode == 0 || iomode > nfs4.IoModeRW {
return nil, nfs4.ErrBadIOMode, nil
}
if !reg.haveCur {
return nil, nfs4.ErrNoFileHandle, nil
}
// A layout hangs from a real open of the caller: the anonymous
// stateid forms never qualify.
if _, status := h.openStates().lookupOpen(openSt, reg.cur, clientid); status != nfs4.ErrOK {
return nil, status, nil
}
// The body is built through one closure so the maxcount check can
// run against a scratch stateid before anything is granted: a
// request whose answer cannot fit leaves no layout behind, RFC 8881
// section 18.43.
build := func(st nfs4.Stateid) []byte {
switch layoutType {
case nfs4.LayoutTypeFiles:
return nfs4.AppendFileLayoutBody(nil, layoutDeviceID,
0 /* util: no striping */, 0, 0, [][]byte{reg.cur})
case nfs4.LayoutTypeBlock:
return nfs4.AppendBlockDeviceAddr(nil, nfs4.BlockVolume{
DeviceID: layoutDeviceID,
BaseOffset: offset,
BlockCount: length,
})
case nfs4.LayoutTypeObjects:
return nfs4.AppendObjectLayoutBody(nil, layoutDeviceID, nfs4.ObjectLayout{
NumComponents: 1, StripeUnit: 4096, GroupWidth: 1, GroupDepth: 1,
})
case nfs4.LayoutTypeScsi:
return nfs4.AppendScsiLayoutBody(nil, layoutDeviceID, offset, length, offset)
case nfs4.LayoutTypeFlexFilesV2:
// The version two body of draft-haynes-nfsv4-flex-filesv2-00:
// one stateid per version and the AUTH_NONE credential, which
// the draft prescribes for tight coupling over synthetic
// identities.
return nfs4.AppendFlexFileLayoutBodyV2(nil, 0, 0, []nfs4.FlexMirrorV2{{
DataServers: []nfs4.FlexDataServerV2{{
DeviceID: layoutDeviceID,
Stateids: []nfs4.Stateid{st},
FHs: [][]byte{reg.cur},
AuthFlavor: 0, // AUTH_NONE
}},
}})
default:
return nfs4.AppendFlexFileLayoutBody(nil, 0, 0, []nfs4.FlexMirror{{
DataServers: []nfs4.FlexDataServer{{
DeviceID: layoutDeviceID,
Stateid: st,
FHs: [][]byte{reg.cur},
}},
}})
}
}
appendRes := func(st nfs4.Stateid) []byte {
return nfs4.AppendLayoutGetRes(nil, st, false, []nfs4.Layout4{{
Offset: offset,
Length: length,
IoMode: iomode,
Type: layoutType,
Body: build(st),
}})
}
if maxcount != 0 && uint32(len(appendRes(nfs4.Stateid{}))) > maxcount {
return nil, nfs4.ErrTooSmall, nil
}
lay := h.layouts().grant(sessID, clientid, reg.cur, offset, length)
return appendRes(lay.stateid), nfs4.ErrOK, nil
}
// layoutCommitOp serves LAYOUTCOMMIT: the layout must still be live and
// the client's last write offset may grow the file. The answer names the
// size the file is left at.
func (h *Handler) layoutCommitOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) {
if !reg.haveCur {
return nil, nfs4.ErrNoFileHandle, nil
}
offset, err := d.Uint64()
if err != nil {
return nil, 0, err
}
length, err := d.Uint64()
if err != nil {
return nil, 0, err
}
if _, err = d.Bool(); err != nil { // reclaim
return nil, 0, err
}
var st nfs4.Stateid
raw, rerr := d.Raw(16)
if rerr != nil {
return nil, 0, rerr
}
copy(st[:], raw)
lastWriteSet, err := d.Bool()
if err != nil {
return nil, 0, err
}
var lastWrite uint64
if lastWriteSet {
if lastWrite, err = d.Uint64(); err != nil {
return nil, 0, err
}
}
timeSet, err := d.Bool()
if err != nil {
return nil, 0, err
}
if timeSet {
if _, err = d.Int64(); err != nil {
return nil, 0, err
}
if _, err = d.Uint32(); err != nil {
return nil, 0, err
}
}
if _, err = d.Uint32(); err != nil { // layout update type
return nil, 0, err
}
if _, err = d.VarOpaque(); err != nil { // layout update body
return nil, 0, err
}
_ = offset
_ = length
if _, status := h.layouts().lookup(st, reg.cur); status != nfs4.ErrOK {
return nil, status, nil
}
info, ferr := h.FS.Getattr(reg.cur)
if ferr != nil {
return nil, mapErr(ferr), nil
}
newSize := uint64(info.Size)
if lastWriteSet && lastWrite+1 > newSize {
w := h.writer()
if w == nil {
return nil, nfs4.ErrROFS, nil
}
grown := int64(lastWrite + 1)
if err := w.Setattr(reg.cur, nfsfs.SetAttrs{Size: &grown}); err != nil {
return nil, mapErr(err), nil
}
newSize = lastWrite + 1
}
return nfs4.AppendLayoutCommitRes(nil, newSize), nfs4.ErrOK, nil
}
// layoutReturnOp serves LAYOUTRETURN: the client gives the layout back.
// A file return drops the named layout, a whole client or file system
// return drops every layout of the session.
func (h *Handler) layoutReturnOp(d *xdr.Decoder, reg *fhreg, sessID nfs4.SessionID, clientid uint64) ([]byte, uint32, error) {
reclaim, err := d.Bool()
if err != nil {
return nil, 0, err
}
layoutType, err := d.Uint32()
if err != nil {
return nil, 0, err
}
if _, err = d.Uint32(); err != nil { // iomode
return nil, 0, err
}
kind, err := d.Uint32()
if err != nil {
return nil, 0, err
}
_ = reclaim
if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 {
return nil, nfs4.ErrUnknownLayoutType, nil
}
switch kind {
case nfs4.ReturnFile:
if _, err = d.Uint64(); err != nil { // offset
return nil, 0, err
}
if _, err = d.Uint64(); err != nil { // length
return nil, 0, err
}
var st nfs4.Stateid
raw, rerr := d.Raw(16)
if rerr != nil {
return nil, 0, rerr
}
copy(st[:], raw)
if _, err = d.Uint32(); err != nil { // ffsid_info_type
return nil, 0, err
}
if !reg.haveCur {
return nil, nfs4.ErrNoFileHandle, nil
}
if _, status := h.layouts().lookup(st, reg.cur); status != nfs4.ErrOK {
return nil, status, nil
}
returned := st
returned[0]++
if status := h.layouts().drop(st); status != nfs4.ErrOK {
return nil, status, nil
}
return nfs4.AppendLayoutReturnRes(nil, returned), nfs4.ErrOK, nil
case nfs4.ReturnFsid, nfs4.ReturnAll:
h.layouts().dropClient(clientid)
h.layouts().dropSession(sessID)
return nfs4.AppendLayoutReturnRes(nil, nfs4.Stateid{}), nfs4.ErrOK, nil
default:
return nil, nfs4.ErrBadLayout, nil
}
}
// getDeviceInfoOp serves GETDEVICEINFO: the data server addresses the
// client needs to reach the storage behind the layout. The one device of
// this build is the metadata server itself.
func (h *Handler) getDeviceInfoOp(d *xdr.Decoder, ctx *connCB) ([]byte, uint32, error) {
var deviceID [16]byte
raw, err := d.Raw(16)
if err != nil {
return nil, 0, err
}
copy(deviceID[:], raw)
layoutType, err := d.Uint32()
if err != nil {
return nil, 0, err
}
maxcount, err := d.Uint32()
if err != nil {
return nil, 0, err
}
if _, err = nfs4.ReadBitmap(d); err != nil { // notification types
return nil, 0, err
}
if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 {
return nil, nfs4.ErrUnknownLayoutType, nil
}
if deviceID != layoutDeviceID {
return nil, nfs4.ErrNoEnt, nil
}
// The address body follows the layout type: flexfiles carries the
// server list with versions, the files layout the stripe indices over
// the multipath list, the others the emulated volume or component.
var addr []byte
switch layoutType {
case nfs4.LayoutTypeFiles:
addr = nfs4.AppendFileDeviceAddr(nil, []uint32{0},
[]nfs4.NetAddr{{Netid: "tcp", Uaddr: h.deviceAddr(ctx)}})
default:
addr = nfs4.AppendFlexDeviceAddr(nil, nfs4.FlexDeviceAddr{
NetAddrs: []nfs4.NetAddr{{Netid: "tcp", Uaddr: h.deviceAddr(ctx)}},
Versions: []nfs4.FlexVersion{{
Version: 4,
MinorVersion: nfs4.MinorVersion,
RSize: uint32(nfs4.DefaultLimits.MaxRead),
WSize: uint32(nfs4.DefaultLimits.MaxWrite),
}},
})
}
res := nfs4.AppendGetDeviceInfoRes(nil, addr)
if maxcount != 0 && uint32(len(res)) > maxcount {
return nil, nfs4.ErrTooSmall, nil
}
return res, nfs4.ErrOK, nil
}
// deviceAddr resolves the universal address the data server answers on:
// the configured value wins, then the local address of the connection the
// request rode in on, then the loopback default.
func (h *Handler) deviceAddr(ctx *connCB) string {
if h.DeviceAddr != "" {
return h.DeviceAddr
}
if ctx != nil && ctx.conn != nil {
if u := uaddrOf(ctx.conn.LocalAddr().String()); u != "" {
return u
}
}
return "127.0.0.1.8.1"
}
// uaddrOf turns a host:port address into the universal address form of
// RFC 8435: decimal octets and port for IPv4, hex nibbles for IPv6.
func uaddrOf(hostPort string) string {
host, portText, err := net.SplitHostPort(hostPort)
if err != nil {
return ""
}
ip := net.ParseIP(host)
if ip == nil {
return ""
}
port := 0
if portText == "" {
return ""
}
for _, r := range portText {
if r < '0' || r > '9' {
return ""
}
port = port*10 + int(r-'0')
if port > 0xffff {
return ""
}
}
var v4 [4]byte
if n := copy(v4[:], ip.To4()); n == 4 {
return itoa(int(v4[0])) + "." + itoa(int(v4[1])) + "." + itoa(int(v4[2])) + "." +
itoa(int(v4[3])) + "." + itoa(port>>8) + "." + itoa(port&0xff)
}
v6 := ip.To16()
if v6 == nil {
return ""
}
const hexDigits = "0123456789abcdef"
out := make([]byte, 0, 16*3+8)
for i, b := range v6 {
if i > 0 {
out = append(out, '.')
}
out = append(out, hexDigits[b>>4], hexDigits[b&0xf])
}
out = append(out, '.')
out = append(out, itoa(port>>8)...)
out = append(out, '.')
out = append(out, itoa(port&0xff)...)
return string(out)
}
// itoa renders a small non negative number in decimal.
func itoa(n int) string {
if n == 0 {
return "0"
}
var digits [8]byte
i := len(digits)
for n > 0 {
i--
digits[i] = byte('0' + n%10)
n /= 10
}
return string(digits[i:])
}
// getDeviceListOp serves GETDEVICELIST: the one device of this build is
// the answer for every layout type it carries, RFC 5661 section 18.41.
func (h *Handler) getDeviceListOp(d *xdr.Decoder, ctx *connCB) ([]byte, uint32, error) {
layoutType, err := d.Uint32()
if err != nil {
return nil, 0, err
}
if _, err = d.Uint32(); err != nil { // maxdevices
return nil, 0, err
}
if _, err = d.Uint64(); err != nil { // cookie
return nil, 0, err
}
if _, err = d.Raw(8); err != nil { // cookie verifier
return nil, 0, err
}
switch layoutType {
case nfs4.LayoutTypeFlexfiles, nfs4.LayoutTypeFlexFilesV2, nfs4.LayoutTypeFiles,
nfs4.LayoutTypeBlock, nfs4.LayoutTypeObjects, nfs4.LayoutTypeScsi:
return nfs4.AppendGetDeviceListRes(nil, 0, h.writeVerifier(),
[][16]byte{layoutDeviceID}, true), nfs4.ErrOK, nil
default:
return nil, nfs4.ErrUnknownLayoutType, nil
}
}