Files
seaweedfs/weed/filer/filer_delete_collection_test.go
T
Chris LuandGitHub 11791fad6a filer: resolve the collection a bucket delete drops (#11439)
* filer: resolve the collection a bucket delete drops

A bucket delete dropped the collection named after the bucket, which
assumes bucket name is collection name. With a collection rule the
write path honors, deleting the bucket either orphaned its collection
or, when a bucket was named after a shared collection, removed volumes
other buckets still write to.

Resolve the collection through the same rule chain the write path uses
and drop it only when no other bucket resolves there too. A listing
failure keeps the collection, the safe side of an unknown.

* filer: prove collection exclusivity across all paths before dropping it

The sibling-bucket scan missed every non-bucket writer: a broad rule like
'/' or '/buckets/', a rule under a surviving bucket, or a rule on an
unrelated path can route into the same collection. Check every storage
rule's prefix instead, and mirror the grouped gateway's explicit
<group>_<bucket> collection, which otherwise resolves a rule-named
collection the bucket never wrote to.

* s3: let the filer own the collection decision on bucket delete

Both entry points deleted a name-derived collection around the filer's
own resolved delete, bypassing its exclusivity check and wiping sibling
data. The filer now resolves the collection a bucket actually used,
including the grouped form.

* filer: keep a collection the default write route also uses

Rule-less writes outside buckets land in the filer's default collection,
so a bucket resolving there shares it with them.
2026-09-25 07:30:42 +08:00

388 lines
13 KiB
Go

package filer
import (
"context"
"fmt"
"net"
"os"
"testing"
"time"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
"github.com/seaweedfs/seaweedfs/weed/cluster"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
"github.com/seaweedfs/seaweedfs/weed/wdclient"
)
// collectionDeleteMaster is just enough of a master for a MasterClient to
// consider itself connected, and records every CollectionDelete it is asked for.
type collectionDeleteMaster struct {
master_pb.UnimplementedSeaweedServer
calls chan collectionDeleteCall
}
type collectionDeleteCall struct {
name string
// budget is the time the RPC arrived with, or 0 when it carried no deadline.
budget time.Duration
}
// KeepConnected is what MasterClient waits on before it reports a master: one
// response is enough, then the stream idles until the server is torn down.
func (m *collectionDeleteMaster) KeepConnected(stream master_pb.Seaweed_KeepConnectedServer) error {
if _, err := stream.Recv(); err != nil {
return err
}
if err := stream.Send(&master_pb.KeepConnectedResponse{}); err != nil {
return err
}
<-stream.Context().Done()
return nil
}
func (m *collectionDeleteMaster) CollectionDelete(ctx context.Context, req *master_pb.CollectionDeleteRequest) (*master_pb.CollectionDeleteResponse, error) {
call := collectionDeleteCall{name: req.Name}
if deadline, ok := ctx.Deadline(); ok {
call.budget = time.Until(deadline)
}
m.calls <- call
return &master_pb.CollectionDeleteResponse{}, nil
}
// hookedStore is the stub store with one extra seam: a callback that runs once
// an entry has actually been removed, so a test can interleave an event -- a
// client hanging up, say -- between the store write and whatever follows it.
type hookedStore struct {
*stubFilerStore
onDeleteEntry func(util.FullPath)
}
func (s *hookedStore) DeleteEntry(ctx context.Context, p util.FullPath) error {
if err := s.stubFilerStore.DeleteEntry(ctx, p); err != nil {
return err
}
if s.onDeleteEntry != nil {
s.onDeleteEntry(p)
}
return nil
}
// newFilerWithFakeMaster builds a filer backed by the stub store, with its
// MasterClient connected to a fake master the test can observe.
func newFilerWithFakeMaster(t *testing.T) (*Filer, *hookedStore, *collectionDeleteMaster) {
t.Helper()
master := &collectionDeleteMaster{calls: make(chan collectionDeleteCall, 4)}
lis, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
grpcServer := grpc.NewServer()
master_pb.RegisterSeaweedServer(grpcServer, master)
go func() { _ = grpcServer.Serve(lis) }()
t.Cleanup(grpcServer.Stop)
_, port, err := net.SplitHostPort(lis.Addr().String())
if err != nil {
t.Fatalf("split listener address: %v", err)
}
masterAddress := pb.ServerAddress(fmt.Sprintf("127.0.0.1:0.%s", port))
mc := wdclient.NewMasterClient(
grpc.WithTransportCredentials(insecure.NewCredentials()),
"", cluster.FilerType, pb.ServerAddress("localhost:0"), "", "",
*pb.NewServiceDiscoveryFromMap(map[string]pb.ServerAddress{"m": masterAddress}),
)
connecting, stopConnecting := context.WithCancel(context.Background())
t.Cleanup(stopConnecting)
go mc.KeepConnectedToMaster(connecting)
waiting, cancelWaiting := context.WithTimeout(context.Background(), 20*time.Second)
defer cancelWaiting()
mc.WaitUntilConnected(waiting)
if waiting.Err() != nil {
t.Fatal("the master client never connected to the fake master")
}
store := &hookedStore{stubFilerStore: newStubFilerStore()}
f := &Filer{
DirBucketsPath: "/buckets",
RemoteStorage: NewFilerRemoteStorage(),
Store: NewFilerStoreWrapper(store),
FilerConf: NewFilerConf(),
MaxFilenameLength: 255,
MasterClient: mc,
FileIdDeletionQueue: util.NewUnboundedQueue(),
deletionQuit: make(chan struct{}),
LocalMetaLogBuffer: log_buffer.NewLogBuffer("test", time.Minute,
func(*log_buffer.LogBuffer, time.Time, time.Time, []byte, int64, int64) {}, nil, func() {}),
}
return f, store, master
}
// Deleting a bucket entry deletes its collection, and the request hanging up
// partway must not stop that: the bucket's metadata is already gone, so skipping
// it strands the bucket's volumes with nothing left to come back for them. The
// cleanup still has to carry a deadline of its own, or a master that is down
// parks this handler and every client retry behind it parks another.
//
// The cancellation lands between the store removing the entry and the collection
// delete, which is where a client disconnect actually bites: FilerStoreWrapper
// refuses a context that is already dead on entry, so an up-front cancellation
// fails the delete long before this point instead.
func TestDeleteEntryMetaAndDataDeletesCollectionWhenTheRequestIsCancelledMidDelete(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
const bucket = "bucket-a"
bucketPath := util.FullPath(f.DirBucketsPath + "/" + bucket)
if err := store.InsertEntry(context.Background(), &Entry{
FullPath: bucketPath,
Attr: Attr{Mode: os.ModeDir | 0755},
}); err != nil {
t.Fatalf("seed the bucket entry: %v", err)
}
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
// The client hangs up just as the bucket entry comes out of the store.
store.onDeleteEntry = func(util.FullPath) { cancel() }
if err := f.DeleteEntryMetaAndData(ctx, bucketPath, true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
if ctx.Err() == nil {
t.Fatal("test setup: the request was never cancelled, so nothing was exercised")
}
var call collectionDeleteCall
select {
case call = <-master.calls:
case <-time.After(20 * time.Second):
t.Fatal("CollectionDelete never reached the master")
}
if call.name != bucket {
t.Errorf("master was asked to delete collection %q, want %q", call.name, bucket)
}
if call.budget <= 0 {
t.Error("CollectionDelete arrived with no deadline; a master that stops answering would hold this handler open")
}
if call.budget > collectionDeleteTimeout {
t.Errorf("CollectionDelete budget = %v, want at most collectionDeleteTimeout %v", call.budget, collectionDeleteTimeout)
}
if store.getEntry(string(bucketPath)) != nil {
t.Error("the bucket entry survived the delete")
}
}
func seedBucket(t *testing.T, store *hookedStore, path util.FullPath) {
t.Helper()
if err := store.InsertEntry(context.Background(), &Entry{
FullPath: path,
Attr: Attr{Mode: os.ModeDir | 0755},
}); err != nil {
t.Fatalf("seed bucket %s: %v", path, err)
}
}
// Two buckets resolving to one collection: deleting either must leave the
// collection for the other. Previously the delete dropped the collection
// named after the bucket regardless of where its data actually lived.
func TestDeleteBucketKeepsSharedCollection(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets",
Collection: "shared",
})
seedBucket(t, store, util.FullPath("/buckets/a"))
seedBucket(t, store, util.FullPath("/buckets/b"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("shared collection was deleted: %q", call.name)
default:
}
if store.getEntry("/buckets/b") == nil {
t.Error("the surviving bucket's entry is gone")
}
}
// A bucket named after a collection other buckets resolve to is still just a
// bucket: deleting it must not take the shared collection down with it.
func TestDeleteBucketNamedAfterSharedCollection(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets",
Collection: "shared",
})
seedBucket(t, store, util.FullPath("/buckets/shared"))
seedBucket(t, store, util.FullPath("/buckets/other"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/shared", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("collection backing other buckets was deleted: %q", call.name)
default:
}
}
// A rule pointing a non-bucket path at the same collection keeps it: the
// collection serves files the bucket delete must not orphan.
func TestDeleteBucketKeepsCollectionUsedByNonBucketPath(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/a",
Collection: "cold",
})
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/archives",
Collection: "cold",
})
seedBucket(t, store, util.FullPath("/buckets/a"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("collection used by /archives was deleted: %q", call.name)
default:
}
}
// A broad prefix rule covering the whole tree keeps the collection even for a
// lone bucket: the same collection backs non-bucket paths too.
func TestDeleteBucketKeepsCollectionFromBroadRule(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/",
Collection: "everything",
})
seedBucket(t, store, util.FullPath("/buckets/a"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("collection from a / rule was deleted: %q", call.name)
default:
}
}
// A bucket resolving to the filer's default collection keeps it: rule-less
// writes outside buckets land there too, so it is never one bucket's alone.
func TestDeleteBucketKeepsDefaultCollection(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.metaLogCollection = "everything"
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/a",
Collection: "everything",
})
seedBucket(t, store, util.FullPath("/buckets/a"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("the filer's default collection was deleted: %q", call.name)
default:
}
}
// A rule nested under a surviving bucket keeps the collection: the other
// bucket resolves elsewhere at its root, but objects deeper inside it still
// land in the shared collection.
func TestDeleteBucketKeepsCollectionFromNestedSiblingRule(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/a",
Collection: "shared",
})
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/b/deep",
Collection: "shared",
})
seedBucket(t, store, util.FullPath("/buckets/a"))
seedBucket(t, store, util.FullPath("/buckets/b"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
t.Fatalf("collection used under /buckets/b/deep was deleted: %q", call.name)
default:
}
}
// A grouped gateway writes to <group>_<bucket> regardless of the storage
// rules, so that is the collection the delete must drop -- and a rule-named
// collection the bucket never used must survive.
func TestDeleteBucketUnderFilerGroup(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.MasterClient.FilerGroup = "tenant1"
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/photos",
Collection: "archive",
})
seedBucket(t, store, util.FullPath("/buckets/photos"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/photos", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
if call.name != "tenant1_photos" {
t.Fatalf("CollectionDelete = %q, want %q", call.name, "tenant1_photos")
}
case <-time.After(20 * time.Second):
t.Fatal("CollectionDelete never reached the master")
}
}
// A collection only the deleted bucket resolves to is dropped under its real
// name, so a dedicated custom collection does not leak its volumes.
func TestDeleteBucketDeletesResolvedCollection(t *testing.T) {
f, store, master := newFilerWithFakeMaster(t)
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
LocationPrefix: "/buckets/only",
Collection: "custom",
})
seedBucket(t, store, util.FullPath("/buckets/only"))
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/only", true, false, true, false, nil, 0); err != nil {
t.Fatalf("DeleteEntryMetaAndData: %v", err)
}
select {
case call := <-master.calls:
if call.name != "custom" {
t.Fatalf("CollectionDelete = %q, want %q", call.name, "custom")
}
case <-time.After(20 * time.Second):
t.Fatal("CollectionDelete never reached the master")
}
}