// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package nfs4server import ( "net" "sync" "sourcedock.dev/petrbalvin/nfs/internal/nfs4" "sourcedock.dev/petrbalvin/nfs/internal/nfsfs" "sourcedock.dev/petrbalvin/nfs/internal/xdr" ) // layoutDeviceID names the one storage device behind every layout: the // metadata server of this build is also its data server, so one identity // serves both roles. var layoutDeviceID = newDeviceID() // newDeviceID builds the fixed device identity of this server. func newDeviceID() [16]byte { var id [16]byte copy(id[:], "pnfs42mdsds00001") return id } // A layout is one granted pNFS layout: the session and client it belongs // to, the file it covers, the byte range and the stateid the client // addresses it by. type layout struct { stateid nfs4.Stateid sessID nfs4.SessionID clientid uint64 fh nfsfs.Handle offset uint64 length uint64 } // layoutServer tracks the layouts the metadata server has granted. The // store lives in memory: a server restart drops every layout and the // clients re-request them through the grace window of the restart. type layoutServer struct { mu sync.Mutex next uint64 layouts map[string]*layout // by layout stateid other } func newLayoutServer() *layoutServer { return &layoutServer{next: randCounter(), layouts: make(map[string]*layout)} } // grant issues a layout over the requested range of the file handle, // replacing the layout the same client already holds on the same file, // so a re-request or a retry never piles entries up. The stateid other // carries the LAYOUT mark and a counter; the sequence field starts at // one, as RFC 8881 section 12.5.2 has it for layout stateids. func (l *layoutServer) grant(sessID nfs4.SessionID, clientid uint64, fh nfsfs.Handle, offset, length uint64) *layout { l.mu.Lock() defer l.mu.Unlock() key := fileKey(fh) for _, lay := range l.layouts { if lay.clientid == clientid && fileKey(lay.fh) == key { lay.offset, lay.length, lay.sessID = offset, length, sessID return lay } } l.next++ var st nfs4.Stateid setStateidSeq(&st, 1) copy(st[4:], "LAYOUT") for i := range 6 { st[15-i] = byte(l.next >> (8 * i)) } lay := &layout{stateid: st, sessID: sessID, clientid: clientid, fh: fh, offset: offset, length: length} l.layouts[string(st[4:])] = lay return lay } // lookup finds a live layout by the stateid the client presents. func (l *layoutServer) lookup(st nfs4.Stateid, fh nfsfs.Handle) (*layout, uint32) { l.mu.Lock() defer l.mu.Unlock() lay, ok := l.layouts[string(st[4:])] if !ok || fileKey(lay.fh) != fileKey(fh) { return nil, nfs4.ErrBadStateid } return lay, nfs4.ErrOK } // drop removes the layout identified by the stateid and answers the status // of the removal. func (l *layoutServer) drop(st nfs4.Stateid) uint32 { l.mu.Lock() defer l.mu.Unlock() if _, ok := l.layouts[string(st[4:])]; !ok { return nfs4.ErrBadStateid } delete(l.layouts, string(st[4:])) return nfs4.ErrOK } // dropSession drops every layout of one session, which DESTROY_SESSION // requires: layouts are session bound state. func (l *layoutServer) dropSession(sessID nfs4.SessionID) { l.mu.Lock() defer l.mu.Unlock() for other, lay := range l.layouts { if lay.sessID == sessID { delete(l.layouts, other) } } } // dropClient drops every layout of one client, which DESTROY_CLIENTID // requires. func (l *layoutServer) dropClient(clientid uint64) { l.mu.Lock() defer l.mu.Unlock() for other, lay := range l.layouts { if lay.clientid == clientid { delete(l.layouts, other) } } } // count reports how many layouts the server has granted. func (l *layoutServer) count() int { l.mu.Lock() defer l.mu.Unlock() return len(l.layouts) } // layoutGetOp serves LAYOUTGET: it validates the open stateid the client // presents, grants a flexfiles layout over the requested range and names // this server as the data server the client reads and writes through. func (h *Handler) layoutGetOp(d *xdr.Decoder, reg *fhreg, sessID nfs4.SessionID, clientid uint64) ([]byte, uint32, error) { if _, err := d.Bool(); err != nil { // signal_avail return nil, 0, err } layoutType, err := d.Uint32() if err != nil { return nil, 0, err } iomode, err := d.Uint32() if err != nil { return nil, 0, err } offset, err := d.Uint64() if err != nil { return nil, 0, err } length, err := d.Uint64() if err != nil { return nil, 0, err } if _, err = d.Uint64(); err != nil { // minlength return nil, 0, err } var openSt nfs4.Stateid raw, rerr := d.Raw(16) if rerr != nil { return nil, 0, rerr } copy(openSt[:], raw) maxcount, err := d.Uint32() if err != nil { return nil, 0, err } if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 && layoutType != nfs4.LayoutTypeFiles && layoutType != nfs4.LayoutTypeBlock && layoutType != nfs4.LayoutTypeObjects && layoutType != nfs4.LayoutTypeScsi { return nil, nfs4.ErrUnknownLayoutType, nil } if iomode == 0 || iomode > nfs4.IoModeRW { return nil, nfs4.ErrBadIOMode, nil } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } // A layout hangs from a real open of the caller: the anonymous // stateid forms never qualify. if _, status := h.openStates().lookupOpen(openSt, reg.cur, clientid); status != nfs4.ErrOK { return nil, status, nil } // The body is built through one closure so the maxcount check can // run against a scratch stateid before anything is granted: a // request whose answer cannot fit leaves no layout behind, RFC 8881 // section 18.43. build := func(st nfs4.Stateid) []byte { switch layoutType { case nfs4.LayoutTypeFiles: return nfs4.AppendFileLayoutBody(nil, layoutDeviceID, 0 /* util: no striping */, 0, 0, [][]byte{reg.cur}) case nfs4.LayoutTypeBlock: return nfs4.AppendBlockDeviceAddr(nil, nfs4.BlockVolume{ DeviceID: layoutDeviceID, BaseOffset: offset, BlockCount: length, }) case nfs4.LayoutTypeObjects: return nfs4.AppendObjectLayoutBody(nil, layoutDeviceID, nfs4.ObjectLayout{ NumComponents: 1, StripeUnit: 4096, GroupWidth: 1, GroupDepth: 1, }) case nfs4.LayoutTypeScsi: return nfs4.AppendScsiLayoutBody(nil, layoutDeviceID, offset, length, offset) case nfs4.LayoutTypeFlexFilesV2: // The version two body of draft-haynes-nfsv4-flex-filesv2-00: // one stateid per version and the AUTH_NONE credential, which // the draft prescribes for tight coupling over synthetic // identities. return nfs4.AppendFlexFileLayoutBodyV2(nil, 0, 0, []nfs4.FlexMirrorV2{{ DataServers: []nfs4.FlexDataServerV2{{ DeviceID: layoutDeviceID, Stateids: []nfs4.Stateid{st}, FHs: [][]byte{reg.cur}, AuthFlavor: 0, // AUTH_NONE }}, }}) default: return nfs4.AppendFlexFileLayoutBody(nil, 0, 0, []nfs4.FlexMirror{{ DataServers: []nfs4.FlexDataServer{{ DeviceID: layoutDeviceID, Stateid: st, FHs: [][]byte{reg.cur}, }}, }}) } } appendRes := func(st nfs4.Stateid) []byte { return nfs4.AppendLayoutGetRes(nil, st, false, []nfs4.Layout4{{ Offset: offset, Length: length, IoMode: iomode, Type: layoutType, Body: build(st), }}) } if maxcount != 0 && uint32(len(appendRes(nfs4.Stateid{}))) > maxcount { return nil, nfs4.ErrTooSmall, nil } lay := h.layouts().grant(sessID, clientid, reg.cur, offset, length) return appendRes(lay.stateid), nfs4.ErrOK, nil } // layoutCommitOp serves LAYOUTCOMMIT: the layout must still be live and // the client's last write offset may grow the file. The answer names the // size the file is left at. func (h *Handler) layoutCommitOp(d *xdr.Decoder, reg *fhreg) ([]byte, uint32, error) { if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } offset, err := d.Uint64() if err != nil { return nil, 0, err } length, err := d.Uint64() if err != nil { return nil, 0, err } if _, err = d.Bool(); err != nil { // reclaim return nil, 0, err } var st nfs4.Stateid raw, rerr := d.Raw(16) if rerr != nil { return nil, 0, rerr } copy(st[:], raw) lastWriteSet, err := d.Bool() if err != nil { return nil, 0, err } var lastWrite uint64 if lastWriteSet { if lastWrite, err = d.Uint64(); err != nil { return nil, 0, err } } timeSet, err := d.Bool() if err != nil { return nil, 0, err } if timeSet { if _, err = d.Int64(); err != nil { return nil, 0, err } if _, err = d.Uint32(); err != nil { return nil, 0, err } } if _, err = d.Uint32(); err != nil { // layout update type return nil, 0, err } if _, err = d.VarOpaque(); err != nil { // layout update body return nil, 0, err } _ = offset _ = length if _, status := h.layouts().lookup(st, reg.cur); status != nfs4.ErrOK { return nil, status, nil } info, ferr := h.FS.Getattr(reg.cur) if ferr != nil { return nil, mapErr(ferr), nil } newSize := uint64(info.Size) if lastWriteSet && lastWrite+1 > newSize { w := h.writer() if w == nil { return nil, nfs4.ErrROFS, nil } grown := int64(lastWrite + 1) if err := w.Setattr(reg.cur, nfsfs.SetAttrs{Size: &grown}); err != nil { return nil, mapErr(err), nil } newSize = lastWrite + 1 } return nfs4.AppendLayoutCommitRes(nil, newSize), nfs4.ErrOK, nil } // layoutReturnOp serves LAYOUTRETURN: the client gives the layout back. // A file return drops the named layout, a whole client or file system // return drops every layout of the session. func (h *Handler) layoutReturnOp(d *xdr.Decoder, reg *fhreg, sessID nfs4.SessionID, clientid uint64) ([]byte, uint32, error) { reclaim, err := d.Bool() if err != nil { return nil, 0, err } layoutType, err := d.Uint32() if err != nil { return nil, 0, err } if _, err = d.Uint32(); err != nil { // iomode return nil, 0, err } kind, err := d.Uint32() if err != nil { return nil, 0, err } _ = reclaim if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 { return nil, nfs4.ErrUnknownLayoutType, nil } switch kind { case nfs4.ReturnFile: if _, err = d.Uint64(); err != nil { // offset return nil, 0, err } if _, err = d.Uint64(); err != nil { // length return nil, 0, err } var st nfs4.Stateid raw, rerr := d.Raw(16) if rerr != nil { return nil, 0, rerr } copy(st[:], raw) if _, err = d.Uint32(); err != nil { // ffsid_info_type return nil, 0, err } if !reg.haveCur { return nil, nfs4.ErrNoFileHandle, nil } if _, status := h.layouts().lookup(st, reg.cur); status != nfs4.ErrOK { return nil, status, nil } returned := st returned[0]++ if status := h.layouts().drop(st); status != nfs4.ErrOK { return nil, status, nil } return nfs4.AppendLayoutReturnRes(nil, returned), nfs4.ErrOK, nil case nfs4.ReturnFsid, nfs4.ReturnAll: h.layouts().dropClient(clientid) h.layouts().dropSession(sessID) return nfs4.AppendLayoutReturnRes(nil, nfs4.Stateid{}), nfs4.ErrOK, nil default: return nil, nfs4.ErrBadLayout, nil } } // getDeviceInfoOp serves GETDEVICEINFO: the data server addresses the // client needs to reach the storage behind the layout. The one device of // this build is the metadata server itself. func (h *Handler) getDeviceInfoOp(d *xdr.Decoder, ctx *connCB) ([]byte, uint32, error) { var deviceID [16]byte raw, err := d.Raw(16) if err != nil { return nil, 0, err } copy(deviceID[:], raw) layoutType, err := d.Uint32() if err != nil { return nil, 0, err } maxcount, err := d.Uint32() if err != nil { return nil, 0, err } if _, err = nfs4.ReadBitmap(d); err != nil { // notification types return nil, 0, err } if layoutType != nfs4.LayoutTypeFlexfiles && layoutType != nfs4.LayoutTypeFlexFilesV2 { return nil, nfs4.ErrUnknownLayoutType, nil } if deviceID != layoutDeviceID { return nil, nfs4.ErrNoEnt, nil } // The address body follows the layout type: flexfiles carries the // server list with versions, the files layout the stripe indices over // the multipath list, the others the emulated volume or component. var addr []byte switch layoutType { case nfs4.LayoutTypeFiles: addr = nfs4.AppendFileDeviceAddr(nil, []uint32{0}, []nfs4.NetAddr{{Netid: "tcp", Uaddr: h.deviceAddr(ctx)}}) default: addr = nfs4.AppendFlexDeviceAddr(nil, nfs4.FlexDeviceAddr{ NetAddrs: []nfs4.NetAddr{{Netid: "tcp", Uaddr: h.deviceAddr(ctx)}}, Versions: []nfs4.FlexVersion{{ Version: 4, MinorVersion: nfs4.MinorVersion, RSize: uint32(nfs4.DefaultLimits.MaxRead), WSize: uint32(nfs4.DefaultLimits.MaxWrite), }}, }) } res := nfs4.AppendGetDeviceInfoRes(nil, addr) if maxcount != 0 && uint32(len(res)) > maxcount { return nil, nfs4.ErrTooSmall, nil } return res, nfs4.ErrOK, nil } // deviceAddr resolves the universal address the data server answers on: // the configured value wins, then the local address of the connection the // request rode in on, then the loopback default. func (h *Handler) deviceAddr(ctx *connCB) string { if h.DeviceAddr != "" { return h.DeviceAddr } if ctx != nil && ctx.conn != nil { if u := uaddrOf(ctx.conn.LocalAddr().String()); u != "" { return u } } return "127.0.0.1.8.1" } // uaddrOf turns a host:port address into the universal address form of // RFC 8435: decimal octets and port for IPv4, hex nibbles for IPv6. func uaddrOf(hostPort string) string { host, portText, err := net.SplitHostPort(hostPort) if err != nil { return "" } ip := net.ParseIP(host) if ip == nil { return "" } port := 0 if portText == "" { return "" } for _, r := range portText { if r < '0' || r > '9' { return "" } port = port*10 + int(r-'0') if port > 0xffff { return "" } } var v4 [4]byte if n := copy(v4[:], ip.To4()); n == 4 { return itoa(int(v4[0])) + "." + itoa(int(v4[1])) + "." + itoa(int(v4[2])) + "." + itoa(int(v4[3])) + "." + itoa(port>>8) + "." + itoa(port&0xff) } v6 := ip.To16() if v6 == nil { return "" } const hexDigits = "0123456789abcdef" out := make([]byte, 0, 16*3+8) for i, b := range v6 { if i > 0 { out = append(out, '.') } out = append(out, hexDigits[b>>4], hexDigits[b&0xf]) } out = append(out, '.') out = append(out, itoa(port>>8)...) out = append(out, '.') out = append(out, itoa(port&0xff)...) return string(out) } // itoa renders a small non negative number in decimal. func itoa(n int) string { if n == 0 { return "0" } var digits [8]byte i := len(digits) for n > 0 { i-- digits[i] = byte('0' + n%10) n /= 10 } return string(digits[i:]) } // getDeviceListOp serves GETDEVICELIST: the one device of this build is // the answer for every layout type it carries, RFC 5661 section 18.41. func (h *Handler) getDeviceListOp(d *xdr.Decoder, ctx *connCB) ([]byte, uint32, error) { layoutType, err := d.Uint32() if err != nil { return nil, 0, err } if _, err = d.Uint32(); err != nil { // maxdevices return nil, 0, err } if _, err = d.Uint64(); err != nil { // cookie return nil, 0, err } if _, err = d.Raw(8); err != nil { // cookie verifier return nil, 0, err } switch layoutType { case nfs4.LayoutTypeFlexfiles, nfs4.LayoutTypeFlexFilesV2, nfs4.LayoutTypeFiles, nfs4.LayoutTypeBlock, nfs4.LayoutTypeObjects, nfs4.LayoutTypeScsi: return nfs4.AppendGetDeviceListRes(nil, 0, h.writeVerifier(), [][16]byte{layoutDeviceID}, true), nfs4.ErrOK, nil default: return nil, nfs4.ErrUnknownLayoutType, nil } }