Files
seaweedfs/weed/mount/weedfs_dir_mkrm.go
T
Chris LuandGitHub 86a189ff80 mount: keep metadata operations working on a removed open directory (#11073)
* mount: remember the entry of a directory removed while still referenced

A directory removed while a descriptor is open on it keeps its inode until
the kernel's final forget, but unlike a file it has no handle to live on
through: OpenDir hands out only a listing cursor. Keep the last-known entry
in memory, keyed by inode, from rmdir until that forget.

Claude-Session: https://claude.ai/code/session_01GYqLENjZzbV5hgt4L8cSAK

* mount: serve metadata ops on a removed open directory from its remembered entry

fchmod, futimens, and the f*xattr calls on a descriptor whose directory was
removed failed with ENOENT: maybeReadEntry resolved the inode to a path, and
rmdir had already dropped it. Fall back to the remembered entry the same way
an unlinked file falls back to its open handle. Mutations publish a changed
copy back rather than editing in place, so a concurrent reader never sees a
half-applied change, and the empty path keeps nlink 0 in every reply.

Claude-Session: https://claude.ai/code/session_01GYqLENjZzbV5hgt4L8cSAK

* mount: stash the entry the delete itself returned, not an earlier snapshot

A chmod landing between Rmdir's entry load and the delete RPC would be
resurrected pre-change: the remembered entry was the earlier local snapshot.
The filer serializes the delete against updates under the path lock and hands
the entry back in the delete event, so prefer that, keeping the local load
for the sticky-bit check and as fallback when no event comes back.

Claude-Session: https://claude.ai/code/session_01GYqLENjZzbV5hgt4L8cSAK

* mount: drop a remembered entry whose insert lost to the final forget

The forget's cleanup runs between RemovePath and the insert when the kernel
evicts the inode concurrently, finds nothing, and the entry would sit in the
map for the life of the mount. Re-check the inode after inserting and take
the entry back out; every interleaving now ends with the map empty.

Claude-Session: https://claude.ai/code/session_01GYqLENjZzbV5hgt4L8cSAK

* mount: insert the remembered entry under the inode table lock

The post-insert HasInode re-check could be fooled by inode number reuse: a
lookup landing between the forget and the check makes the number look alive
and the stale entry stays, keyed to someone else's inode. Do not check after
the fact — RemovePath now runs the retention callback inside its critical
section, where the forget that releases under the same lock cannot have run
and cannot be missed. Publishes need no such fence: their open descriptor
keeps the kernel from issuing the final forget in the first place.

Claude-Session: https://claude.ai/code/session_01GYqLENjZzbV5hgt4L8cSAK
2026-09-01 13:21:02 -07:00

201 lines
6.4 KiB
Go

package mount
import (
"context"
"os"
"syscall"
"time"
"github.com/seaweedfs/go-fuse/v2/fuse"
"google.golang.org/protobuf/proto"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
)
/** Create a directory
*
* Note that the mode argument may not have the type specification
* bits set, i.e. S_ISDIR(mode) can be false. To obtain the
* correct directory type bits use mode|S_IFDIR
* */
func (wfs *WFS) Mkdir(cancel <-chan struct{}, in *fuse.MkdirIn, name string, out *fuse.EntryOut) (code fuse.Status) {
if wfs.IsOverQuotaWithUncommitted() {
return fuse.Status(syscall.ENOSPC)
}
var s fuse.Status
if name, s = checkName(name); s != fuse.OK {
return s
}
now := time.Now().Unix()
dirFullPath, code := wfs.inodeToPath.GetPath(in.NodeId)
if code != fuse.OK {
return
}
entryFullPath := dirFullPath.Child(name)
// Pre-allocate the mount's local inode and stamp it into the create
// request so both the mount and the filer agree on object identity from
// the start. Without this, the filer assigns its own inode in CreateEntry
// and the cached entry then reports a different value than the one we
// return to the kernel here.
inode := wfs.inodeToPath.AllocateInode(entryFullPath, now)
newEntry := &filer_pb.Entry{
Name: name,
IsDirectory: true,
Attributes: &filer_pb.FuseAttributes{
Mtime: now,
Crtime: now,
Ctime: now,
FileMode: uint32(os.ModeDir) | in.Mode,
Uid: in.Uid,
Gid: in.Gid,
Inode: inode,
},
}
wfs.mapPbIdFromLocalToFiler(newEntry)
// Defer restoring to local uid/gid AFTER the entry is sent to the filer
// but BEFORE outputPbEntry writes attributes to the kernel. We restore
// explicitly below instead of using defer so the kernel gets local values.
request := &filer_pb.CreateEntryRequest{
// Defensive: dirFullPath is clean by construction for mount-originated
// mutations, but could carry invalid-UTF-8 bytes if metaCache was
// populated from a non-gRPC source (direct store write, legacy import).
// Sanitizing here keeps the marshal strictly per-request on the off
// chance invalid bytes do reach us.
Directory: dirFullPath.Sanitized(),
Entry: newEntry,
Signatures: []int32{wfs.signature},
SkipCheckParentDirectory: true,
}
glog.V(1).Infof("mkdir: %v", request)
resp, err := wfs.streamCreateEntry(context.Background(), request)
if err != nil {
glog.V(0).Infof("mkdir %s: %v", entryFullPath, err)
} else {
event := resp.GetMetadataEvent()
if event == nil {
event = metadataCreateEvent(string(dirFullPath), newEntry)
}
if applyErr := wfs.applyLocalMetadataEvent(context.Background(), event); applyErr != nil {
glog.Warningf("mkdir %s: best-effort metadata apply failed: %v", entryFullPath, applyErr)
wfs.inodeToPath.InvalidateChildrenCache(dirFullPath)
}
wfs.inodeToPath.TouchDirectory(dirFullPath)
wfs.touchDirMtimeCtimeBest(dirFullPath)
wfs.inodeToPath.AdjustSubdirCount(dirFullPath, 1)
}
glog.V(3).Infof("mkdir %s: %v", entryFullPath, err)
if err != nil {
wfs.mapPbIdFromFilerToLocal(newEntry)
return fuse.EIO
}
// Map uid/gid back to local-space before writing attributes to the
// kernel. The kernel (especially macFUSE) caches these and uses them
// for subsequent permission checks on children.
wfs.mapPbIdFromFilerToLocal(newEntry)
inode = wfs.inodeToPath.Lookup(entryFullPath, newEntry.Attributes.Crtime, true, false, inode, true)
// The newly created directory is guaranteed to be empty, so mark it as
// cached immediately to avoid a needless filer round-trip on the first
// Lookup or ReadDir inside this directory.
wfs.inodeToPath.MarkChildrenCached(entryFullPath)
wfs.outputPbEntry(out, inode, newEntry)
return fuse.OK
}
/** Remove a directory */
func (wfs *WFS) Rmdir(cancel <-chan struct{}, header *fuse.InHeader, name string) (code fuse.Status) {
if name == "." {
return fuse.Status(syscall.EINVAL)
}
if name == ".." {
return fuse.Status(syscall.ENOTEMPTY)
}
// Sanitize before it reaches DeleteEntryRequest.Name; see sanitizeFuseName.
name = sanitizeFuseName(name)
dirFullPath, code := wfs.inodeToPath.GetPath(header.NodeId)
if code != fuse.OK {
return
}
entryFullPath := dirFullPath.Child(name)
targetEntry, _, targetCode := wfs.maybeLoadEntry(entryFullPath)
if targetCode != fuse.OK {
targetEntry = nil
}
// POSIX: enforce sticky bit on the parent directory.
if dirEntry, _, dirCode := wfs.maybeLoadEntry(dirFullPath); dirCode == fuse.OK && dirEntry != nil && dirEntry.Attributes != nil {
targetUid := uint32(0)
if targetEntry != nil && targetEntry.Attributes != nil {
targetUid = targetEntry.Attributes.Uid
}
if code := checkStickyBit(dirEntry.Attributes.FileMode, dirEntry.Attributes.Uid, targetUid, header.Uid); code != fuse.OK {
return code
}
}
glog.V(3).Infof("remove directory: %v", entryFullPath)
deleteReq := &filer_pb.DeleteEntryRequest{
Directory: string(dirFullPath),
Name: name,
IsDeleteData: true,
IgnoreRecursiveError: true, // ignore recursion error since the OS should manage it
Signatures: []int32{wfs.signature},
}
resp, err := wfs.streamDeleteEntry(context.Background(), deleteReq)
if err != nil {
glog.V(1).Infof("remove %s: %v", entryFullPath, err)
if filer.IsNonEmptyFolderError(err) {
return fuse.Status(syscall.ENOTEMPTY)
}
return fuse.ENOENT
}
event := metadataDeleteEvent(string(dirFullPath), name, true)
if resp != nil && resp.MetadataEvent != nil {
event = resp.MetadataEvent
}
if applyErr := wfs.applyLocalMetadataEvent(context.Background(), event); applyErr != nil {
glog.Warningf("rmdir %s: best-effort metadata apply failed: %v", entryFullPath, applyErr)
wfs.inodeToPath.InvalidateChildrenCache(dirFullPath)
}
// The filer serialized the delete against concurrent updates and returned
// the entry as it stood; the snapshot loaded above may predate one.
if oldEntry := resp.GetMetadataEvent().GetEventNotification().GetOldEntry(); oldEntry.GetAttributes() != nil {
targetEntry = proto.Clone(oldEntry).(*filer_pb.Entry)
wfs.mapPbIdFromFilerToLocal(targetEntry)
}
wfs.inodeToPath.RemovePath(entryFullPath, func(inode uint64) {
if targetEntry != nil {
wfs.rememberRemovedDir(inode, targetEntry)
}
})
wfs.inodeToPath.TouchDirectory(dirFullPath)
wfs.touchDirMtimeCtimeBest(dirFullPath)
wfs.inodeToPath.AdjustSubdirCount(dirFullPath, -1)
return fuse.OK
}