mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-30 03:35:37 +00:00
* filer: resolve the collection a bucket delete drops A bucket delete dropped the collection named after the bucket, which assumes bucket name is collection name. With a collection rule the write path honors, deleting the bucket either orphaned its collection or, when a bucket was named after a shared collection, removed volumes other buckets still write to. Resolve the collection through the same rule chain the write path uses and drop it only when no other bucket resolves there too. A listing failure keeps the collection, the safe side of an unknown. * filer: prove collection exclusivity across all paths before dropping it The sibling-bucket scan missed every non-bucket writer: a broad rule like '/' or '/buckets/', a rule under a surviving bucket, or a rule on an unrelated path can route into the same collection. Check every storage rule's prefix instead, and mirror the grouped gateway's explicit <group>_<bucket> collection, which otherwise resolves a rule-named collection the bucket never wrote to. * s3: let the filer own the collection decision on bucket delete Both entry points deleted a name-derived collection around the filer's own resolved delete, bypassing its exclusivity check and wiping sibling data. The filer now resolves the collection a bucket actually used, including the grouped form. * filer: keep a collection the default write route also uses Rule-less writes outside buckets land in the filer's default collection, so a bucket resolving there shares it with them.
319 lines
11 KiB
Go
319 lines
11 KiB
Go
package filer
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
const (
|
|
MsgFailDelNonEmptyFolder = "fail to delete non-empty folder"
|
|
)
|
|
|
|
// ErrNonEmptyFolder is a non-recursive delete refused because the folder still
|
|
// has children. The marker leads the message the filer builds for it and is
|
|
// never wrapped on the way out, so the path that follows it, which the client
|
|
// chose, cannot forge one.
|
|
var ErrNonEmptyFolder = errors.New(MsgFailDelNonEmptyFolder)
|
|
|
|
// DeleteEntryError turns the text of DeleteEntryResponse.Error back into an
|
|
// error carrying the condition the filer reported. Call it on the response
|
|
// field, before formatting a path around it.
|
|
func DeleteEntryError(msg string) error {
|
|
if strings.HasPrefix(msg, MsgFailDelNonEmptyFolder) {
|
|
return &deleteEntryError{msg: msg, cause: ErrNonEmptyFolder}
|
|
}
|
|
return errors.New(msg)
|
|
}
|
|
|
|
// IsNonEmptyFolderError is for callers holding a delete failure that has not
|
|
// been wrapped yet: the sentinel when it survived, the leading marker when the
|
|
// error only crossed the wire as text.
|
|
func IsNonEmptyFolderError(err error) bool {
|
|
if err == nil {
|
|
return false
|
|
}
|
|
return errors.Is(err, ErrNonEmptyFolder) || strings.HasPrefix(err.Error(), MsgFailDelNonEmptyFolder)
|
|
}
|
|
|
|
type deleteEntryError struct {
|
|
msg string
|
|
cause error
|
|
}
|
|
|
|
func (e *deleteEntryError) Error() string { return e.msg }
|
|
func (e *deleteEntryError) Unwrap() error { return e.cause }
|
|
|
|
type OnChunksFunc func([]*filer_pb.FileChunk) error
|
|
type OnHardLinkIdsFunc func([]HardLinkId) error
|
|
|
|
func (f *Filer) DeleteEntryMetaAndData(ctx context.Context, p util.FullPath, isRecursive, ignoreRecursiveError, shouldDeleteChunks, isFromOtherCluster bool, signatures []int32, ifNotModifiedAfter int64) (err error) {
|
|
if p == "/" {
|
|
return nil
|
|
}
|
|
|
|
entry, findErr := f.FindEntry(ctx, p)
|
|
if findErr != nil {
|
|
return findErr
|
|
}
|
|
if ifNotModifiedAfter > 0 && entry.Attr.Mtime.Unix() > ifNotModifiedAfter {
|
|
return nil
|
|
}
|
|
isDeleteCollection := f.IsBucket(entry)
|
|
collectionName := ""
|
|
if isDeleteCollection {
|
|
collectionName = f.bucketCollection(ctx, entry.Name())
|
|
}
|
|
if entry.IsDirectory() {
|
|
// delete the folder children, not including the folder itself
|
|
err = f.doBatchDeleteFolderMetaAndData(ctx, entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks && !isDeleteCollection, isDeleteCollection, isFromOtherCluster, signatures, func(hardLinkIds []HardLinkId) error {
|
|
// A case not handled:
|
|
// what if the chunk is in a different collection?
|
|
if shouldDeleteChunks {
|
|
f.maybeDeleteHardLinks(ctx, hardLinkIds)
|
|
}
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
glog.V(2).InfofCtx(ctx, "delete directory %s: %v", p, err)
|
|
if errors.Is(err, ErrNonEmptyFolder) {
|
|
return err
|
|
}
|
|
return fmt.Errorf("delete directory %s: %v", p, err)
|
|
}
|
|
}
|
|
|
|
// delete the file or folder
|
|
err = f.doDeleteEntryMetaAndData(ctx, entry, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
if err != nil {
|
|
return fmt.Errorf("delete file %s: %v", p, err)
|
|
}
|
|
|
|
if shouldDeleteChunks && !isDeleteCollection {
|
|
if len(entry.HardLinkId) != 0 && entry.HardLinkCounter > 1 {
|
|
// if the file is a hard link and there are other hard links, do not delete the chunks
|
|
} else {
|
|
f.DeleteChunks(ctx, p, entry.GetChunks())
|
|
}
|
|
}
|
|
|
|
if isDeleteCollection {
|
|
if collectionName != "" {
|
|
// the entry is already gone: a caller that hung up must not leave the
|
|
// collection behind, so this cleanup outlives the request -- bounded all
|
|
// the same, or a master that is down parks this handler indefinitely and
|
|
// every client retry behind it parks another
|
|
collectionCtx, cancelCollection := context.WithTimeout(context.WithoutCancel(ctx), collectionDeleteTimeout)
|
|
f.DoDeleteCollection(collectionCtx, collectionName)
|
|
cancelCollection()
|
|
}
|
|
// drop bucket-labeled series held by this process; the S3 gateway
|
|
// only cleans its own registry
|
|
stats.DeleteBucketMetrics(entry.Name())
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) doBatchDeleteFolderMetaAndData(ctx context.Context, entry *Entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks, isDeletingBucket, isFromOtherCluster bool, signatures []int32, onHardLinkIdsFn OnHardLinkIdsFunc) (err error) {
|
|
|
|
//collect all the chunks of this layer and delete them together at the end
|
|
var chunksToDelete []*filer_pb.FileChunk
|
|
lastFileName := ""
|
|
includeLastFile := false
|
|
listedChildren := !isDeletingBucket || !f.Store.CanDropWholeBucket()
|
|
if listedChildren {
|
|
for {
|
|
entries, _, err := f.ListDirectoryEntries(ctx, entry.FullPath, lastFileName, includeLastFile, PaginationSize, "", "", "")
|
|
if err != nil {
|
|
glog.ErrorfCtx(ctx, "list folder %s: %v", entry.FullPath, err)
|
|
return fmt.Errorf("list folder %s: %v", entry.FullPath, err)
|
|
}
|
|
if lastFileName == "" && !isRecursive && len(entries) > 0 {
|
|
// only for first iteration in the loop
|
|
glog.V(2).InfofCtx(ctx, "deleting a folder %s has children: %+v ...", entry.FullPath, entries[0].Name())
|
|
return fmt.Errorf("%w: %s", ErrNonEmptyFolder, entry.FullPath)
|
|
}
|
|
|
|
for _, sub := range entries {
|
|
lastFileName = sub.Name()
|
|
if sub.IsDirectory() {
|
|
subIsDeletingBucket := f.IsBucket(sub)
|
|
err = f.doBatchDeleteFolderMetaAndData(ctx, sub, isRecursive, ignoreRecursiveError, shouldDeleteChunks, subIsDeletingBucket, isFromOtherCluster, nil, onHardLinkIdsFn)
|
|
} else {
|
|
if !isFromOtherCluster {
|
|
if _, remoteErr := f.maybeDeleteFromRemote(ctx, sub); remoteErr != nil {
|
|
glog.Warningf("remote delete child %s: %v", sub.FullPath, remoteErr)
|
|
if !ignoreRecursiveError {
|
|
err = remoteErr
|
|
}
|
|
}
|
|
}
|
|
if err != nil && !ignoreRecursiveError {
|
|
break
|
|
}
|
|
f.NotifyUpdateEvent(ctx, sub, nil, shouldDeleteChunks, isFromOtherCluster, nil)
|
|
if len(sub.HardLinkId) != 0 {
|
|
// hard link chunk data are deleted separately
|
|
err = onHardLinkIdsFn([]HardLinkId{sub.HardLinkId})
|
|
} else {
|
|
if shouldDeleteChunks {
|
|
chunksToDelete = append(chunksToDelete, sub.GetChunks()...)
|
|
}
|
|
}
|
|
}
|
|
if err != nil && !ignoreRecursiveError {
|
|
return err
|
|
}
|
|
}
|
|
|
|
if len(entries) < PaginationSize {
|
|
break
|
|
}
|
|
}
|
|
}
|
|
|
|
glog.V(3).InfofCtx(ctx, "deleting directory %v delete chunks: %v", entry.FullPath, shouldDeleteChunks)
|
|
|
|
// a non-recursive delete already proved the folder empty above, so sweeping the
|
|
// children now can only remove entries that raced in after that listing
|
|
if isRecursive || !listedChildren {
|
|
if storeDeletionErr := f.Store.DeleteFolderChildren(ctx, entry.FullPath); storeDeletionErr != nil {
|
|
return fmt.Errorf("filer store delete: %w", storeDeletionErr)
|
|
}
|
|
}
|
|
|
|
f.NotifyUpdateEvent(ctx, entry, nil, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
f.DeleteChunks(ctx, entry.FullPath, chunksToDelete)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) doDeleteEntryMetaAndData(ctx context.Context, entry *Entry, shouldDeleteChunks bool, isFromOtherCluster bool, signatures []int32) (err error) {
|
|
|
|
glog.V(3).InfofCtx(ctx, "deleting entry %v, delete chunks: %v", entry.FullPath, shouldDeleteChunks)
|
|
|
|
if !isFromOtherCluster {
|
|
if _, remoteDeletionErr := f.maybeDeleteFromRemote(ctx, entry); remoteDeletionErr != nil {
|
|
return remoteDeletionErr
|
|
}
|
|
}
|
|
|
|
if storeDeletionErr := f.Store.DeleteOneEntry(ctx, entry); storeDeletionErr != nil {
|
|
return fmt.Errorf("filer store delete: %w", storeDeletionErr)
|
|
}
|
|
|
|
if !entry.IsDirectory() {
|
|
f.NotifyUpdateEvent(ctx, entry, nil, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// collectionDeleteTimeout bounds the collection delete a bucket entry's own
|
|
// delete leaves behind, which carries no deadline of its own. It bounds the wait
|
|
// and not the work: the master keeps deleting on its own fan-out once asked, so
|
|
// giving up costs the confirmation. Short enough that the S3 client waiting on
|
|
// the bucket delete, which pays this and then the gateway's own follow-up
|
|
// DeleteCollection, still has retry budget left.
|
|
const collectionDeleteTimeout = 15 * time.Second
|
|
|
|
// bucketCollection resolves the collection a bucket's objects land in
|
|
// through the same chain the write path uses -- a grouped gateway's explicit
|
|
// collection, then the storage rules, then the bucket name -- and reports it
|
|
// only when nothing outside the bucket can still route into it. A shared
|
|
// collection must survive the bucket delete: dropping it removes volumes
|
|
// other paths still write to. A listing failure keeps the collection, the
|
|
// safe side of an unknown.
|
|
func (f *Filer) bucketCollection(ctx context.Context, bucket string) (collection string) {
|
|
bucketDir := f.DirBucketsPath + "/" + bucket + "/"
|
|
resolve := func(dir, name string) string {
|
|
if f.MasterClient != nil {
|
|
if group := f.MasterClient.FilerGroup; group != "" {
|
|
return group + "_" + name
|
|
}
|
|
}
|
|
return util.Nvl(f.FilerConf.MatchStorageRule(dir).Collection, name)
|
|
}
|
|
collection = resolve(bucketDir, bucket)
|
|
|
|
// Rule-less writes outside buckets fall back to the filer's default
|
|
// collection, so a bucket resolving there shares it with them.
|
|
if collection == f.metaLogCollection {
|
|
return ""
|
|
}
|
|
|
|
// A rule whose prefix escapes the bucket can route other paths into the
|
|
// same collection, including prefixes nested under surviving buckets.
|
|
for _, rule := range f.FilerConf.ToProto().Locations {
|
|
prefix := strings.TrimSuffix(rule.LocationPrefix, "/") + "/"
|
|
if strings.HasPrefix(prefix, bucketDir) {
|
|
continue
|
|
}
|
|
if f.FilerConf.MatchStorageRule(prefix).Collection == collection {
|
|
return ""
|
|
}
|
|
}
|
|
|
|
siblings, err := f.listBuckets(ctx)
|
|
if err != nil {
|
|
glog.ErrorfCtx(ctx, "list buckets for collection check: %v", err)
|
|
return ""
|
|
}
|
|
for _, sibling := range siblings {
|
|
if sibling != bucket && resolve(f.DirBucketsPath+"/"+sibling+"/", sibling) == collection {
|
|
return ""
|
|
}
|
|
}
|
|
return collection
|
|
}
|
|
|
|
func (f *Filer) listBuckets(ctx context.Context) (buckets []string, err error) {
|
|
lastFileName := ""
|
|
for {
|
|
entries, _, listErr := f.ListDirectoryEntries(ctx, util.FullPath(f.DirBucketsPath), lastFileName, false, PaginationSize, "", "", "")
|
|
if listErr != nil {
|
|
return nil, listErr
|
|
}
|
|
for _, entry := range entries {
|
|
lastFileName = entry.Name()
|
|
if f.IsBucket(entry) {
|
|
buckets = append(buckets, entry.Name())
|
|
}
|
|
}
|
|
if len(entries) < PaginationSize {
|
|
return buckets, nil
|
|
}
|
|
}
|
|
}
|
|
|
|
func (f *Filer) DoDeleteCollection(ctx context.Context, collectionName string) (err error) {
|
|
|
|
return f.MasterClient.WithClient(ctx, false, func(client master_pb.SeaweedClient) error {
|
|
_, err := client.CollectionDelete(ctx, &master_pb.CollectionDeleteRequest{
|
|
Name: collectionName,
|
|
})
|
|
if err != nil {
|
|
glog.Infof("delete collection %s: %v", collectionName, err)
|
|
}
|
|
return err
|
|
})
|
|
|
|
}
|
|
|
|
func (f *Filer) maybeDeleteHardLinks(ctx context.Context, hardLinkIds []HardLinkId) {
|
|
for _, hardLinkId := range hardLinkIds {
|
|
if err := f.Store.DeleteHardLink(ctx, hardLinkId); err != nil {
|
|
glog.ErrorfCtx(ctx, "delete hard link id %d : %v", hardLinkId, err)
|
|
}
|
|
}
|
|
}
|