mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-30 12:47:00 +00:00
* filer.backup: key the checkpoint by source path and sink destination The checkpoint id hashed only sink name + directory, so two backups to different buckets or endpoints sharing a directory layout advanced one checkpoint: whichever job was running pushed the shared offset forward, and a stopped or failing job later resumed from the other's position, silently skipping changes. Backups of different source paths to the same destination shared a checkpoint the same way. Each sink now reports a destination identity (endpoint or account, bucket or container, directory) and the checkpoint is keyed by the source path plus that identity. Reads fall back to the historical name+directory key when the new key has no value, so existing backups resume where they left off; writes go only to the new key. * filer.sync: include the target path in the offset key The offset stored on the target filer was keyed by source path and source filer signature only, so two syncs from the same source cluster and path to different directories on the same target cluster advanced one shared checkpoint, and the slower one could resume past events it never applied. The target path now participates in the key; "/" keeps the historical form, and a sync with a non-root target path falls back to the historical key once when its own key has no value yet. * join checkpoint key fields with NUL so they cannot alias A path or configuration value spelling out the separator could concatenate two different field tuples to the same checkpoint key. NUL cannot appear in a CLI path argument or any sane configuration value, making the encoding injective.
346 lines
13 KiB
Go
346 lines
13 KiB
Go
package command
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/replication/sink"
|
|
"github.com/seaweedfs/seaweedfs/weed/replication/source"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
|
|
)
|
|
|
|
func TestMain(m *testing.M) {
|
|
util_http.InitGlobalHttpClient()
|
|
os.Exit(m.Run())
|
|
}
|
|
|
|
// readUrlError starts a test HTTP server returning the given status code
|
|
// and returns the error produced by ReadUrlAsStream.
|
|
//
|
|
// The error format is defined in ReadUrlAsStream:
|
|
// https://github.com/seaweedfs/seaweedfs/blob/3a765df2ff90839acb9acf910b73513417fa84d1/weed/util/http/http_global_client_util.go#L353
|
|
func readUrlError(t *testing.T, statusCode int) error {
|
|
t.Helper()
|
|
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
http.Error(w, http.StatusText(statusCode), statusCode)
|
|
}))
|
|
defer server.Close()
|
|
|
|
_, err := util_http.ReadUrlAsStream(context.Background(),
|
|
server.URL+"/437,03f591a3a2b95e?readDeleted=true", "",
|
|
nil, false, true, 0, 1024, func(data []byte) {})
|
|
if err == nil {
|
|
t.Fatal("expected error from ReadUrlAsStream, got nil")
|
|
}
|
|
return err
|
|
}
|
|
|
|
func TestIsIgnorable404_WrappedErrNotFound(t *testing.T) {
|
|
readErr := readUrlError(t, http.StatusNotFound)
|
|
// genProcessFunction wraps sink errors with %w:
|
|
// https://github.com/seaweedfs/seaweedfs/blob/3a765df2ff90839acb9acf910b73513417fa84d1/weed/command/filer_sync.go#L496
|
|
genErr := fmt.Errorf("create entry1 : %w", readErr)
|
|
|
|
if !isIgnorable404(genErr) {
|
|
t.Errorf("expected ignorable, got not: %v", genErr)
|
|
}
|
|
}
|
|
|
|
func TestIsIgnorable404_BrokenUnwrapChain(t *testing.T) {
|
|
readErr := readUrlError(t, http.StatusNotFound)
|
|
// AWS SDK v1 wraps transport errors via awserr.New which uses origErr.Error()
|
|
// instead of %w, so errors.Is cannot unwrap through it:
|
|
// https://github.com/aws/aws-sdk-go/blob/v1.55.8/aws/corehandlers/handlers.go#L173
|
|
// https://github.com/aws/aws-sdk-go/blob/v1.55.8/aws/awserr/types.go#L15
|
|
awsSdkErr := fmt.Errorf("RequestError: send request failed\n"+
|
|
"caused by: Put \"https://s3.amazonaws.com/bucket/key\": %s", readErr.Error())
|
|
genErr := fmt.Errorf("create entry1 : %w", awsSdkErr)
|
|
|
|
if !isIgnorable404(genErr) {
|
|
t.Errorf("expected ignorable, got not: %v", genErr)
|
|
}
|
|
}
|
|
|
|
func TestIsIgnorable404_NonIgnorableError(t *testing.T) {
|
|
readErr := readUrlError(t, http.StatusForbidden)
|
|
genErr := fmt.Errorf("create entry1 : %w", readErr)
|
|
|
|
if isIgnorable404(genErr) {
|
|
t.Errorf("expected not ignorable, got ignorable: %v", genErr)
|
|
}
|
|
}
|
|
|
|
// Regression for the partial-landing swallow: transient volume-lookup races
|
|
// ("LookupFileId ... failed", "volume id N not found") raised while replicating
|
|
// a checkpoint write burst must NOT be classified as a genuine source 404.
|
|
// They previously matched isIgnorable404 by substring, so the event was treated
|
|
// as a deletion — the subscription offset advanced and the file (typically a
|
|
// large manifest-backed .pt) was never replicated, with no error logged. They
|
|
// are now propagated so the offset stays put and the event is reprocessed; only
|
|
// a genuinely-gone source (verified live by the filer sink) is ever skipped.
|
|
func TestIsIgnorable404_TransientLookupNotSwallowed(t *testing.T) {
|
|
cases := []struct {
|
|
name string
|
|
err error
|
|
}{
|
|
{
|
|
"lookup file id race",
|
|
fmt.Errorf("create entry1 : %w",
|
|
fmt.Errorf("replicate manifest data chunks 3,01abc: LookupFileId 3,01abc failed, err: context deadline exceeded")),
|
|
},
|
|
{
|
|
"volume id not found race",
|
|
fmt.Errorf("create entry1 : %w",
|
|
fmt.Errorf("replicate entry chunks /buckets/x/model.pt: copy 7,02def: read part 7,02def: volume id 7 not found")),
|
|
},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
if isIgnorable404(tc.err) {
|
|
t.Errorf("transient lookup race must not be ignorable (would swallow a live file): %v", tc.err)
|
|
}
|
|
if !isSourceLookupError(tc.err) {
|
|
t.Errorf("lookup race must classify as a source lookup error (resolved via the live source): %v", tc.err)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestErrorClassifiersNilSafe(t *testing.T) {
|
|
if isIgnorable404(nil) {
|
|
t.Error("isIgnorable404(nil) must be false")
|
|
}
|
|
if isSourceLookupError(nil) {
|
|
t.Error("isSourceLookupError(nil) must be false")
|
|
}
|
|
}
|
|
|
|
func TestIsSourceLookupError_NonLookupErrors(t *testing.T) {
|
|
cases := []struct {
|
|
name string
|
|
err error
|
|
}{
|
|
{"genuine S3 404", fmt.Errorf("upload part: 404 Not Found: not found")},
|
|
{"network error", fmt.Errorf("dial tcp 10.0.0.1:8080: connection refused")},
|
|
{"plain not found without volume id", fmt.Errorf("entry not found")},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
if isSourceLookupError(tc.err) {
|
|
t.Errorf("must not classify as a source lookup error: %v", tc.err)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// Legacy events carry an empty NewParentPath; the probe must fall back to
|
|
// resp.Directory instead of building "/<name>", which would read as "gone"
|
|
// and skip a live file.
|
|
func TestEventSupersessionProbe_PathDerivation(t *testing.T) {
|
|
entry := &filer_pb.Entry{
|
|
Name: "f",
|
|
Attributes: &filer_pb.FuseAttributes{Mtime: 123},
|
|
}
|
|
cases := []struct {
|
|
name string
|
|
resp *filer_pb.SubscribeMetadataResponse
|
|
want string
|
|
}{
|
|
{
|
|
"legacy event without NewParentPath",
|
|
&filer_pb.SubscribeMetadataResponse{
|
|
Directory: "/buckets/x",
|
|
EventNotification: &filer_pb.EventNotification{NewEntry: entry},
|
|
},
|
|
"/buckets/x/f",
|
|
},
|
|
{
|
|
"rename event with NewParentPath",
|
|
&filer_pb.SubscribeMetadataResponse{
|
|
Directory: "/buckets/x",
|
|
EventNotification: &filer_pb.EventNotification{
|
|
NewParentPath: "/buckets/y",
|
|
NewEntry: entry,
|
|
},
|
|
},
|
|
"/buckets/y/f",
|
|
},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
path, mtimeNs, ok := eventSupersessionProbe(tc.resp)
|
|
if !ok {
|
|
t.Fatal("probe must succeed when NewEntry is present")
|
|
}
|
|
if string(path) != tc.want {
|
|
t.Errorf("path = %q, want %q", path, tc.want)
|
|
}
|
|
if mtimeNs != 123*int64(1e9) {
|
|
t.Errorf("mtimeNs = %d, want %d", mtimeNs, 123*int64(1e9))
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// When supersession cannot be proven, never skip — that would drop a live file.
|
|
func TestEventSourceSuperseded_Guards(t *testing.T) {
|
|
if eventSourceSuperseded(nil, nil) {
|
|
t.Error("nil response must not be skippable")
|
|
}
|
|
if eventSourceSuperseded(nil, &filer_pb.SubscribeMetadataResponse{
|
|
EventNotification: &filer_pb.EventNotification{},
|
|
}) {
|
|
t.Error("event without NewEntry must not be skippable")
|
|
}
|
|
if eventSourceSuperseded(nil, &filer_pb.SubscribeMetadataResponse{
|
|
EventNotification: &filer_pb.EventNotification{
|
|
NewParentPath: "/buckets/x",
|
|
NewEntry: &filer_pb.Entry{
|
|
Name: "model.pt",
|
|
Attributes: &filer_pb.FuseAttributes{Mtime: 1234567890},
|
|
},
|
|
},
|
|
}) {
|
|
t.Error("nil filerSource must not be skippable")
|
|
}
|
|
}
|
|
|
|
// stubSink is a minimal ReplicationSink used to exercise initialSnapshotTargetKey
|
|
// and backupCheckpointIds without standing up a real sink; the methods those
|
|
// read (GetName, IsIncremental, GetSinkToDirectory, GetDestinationIdentity)
|
|
// reflect the fields, the rest satisfy the interface.
|
|
type stubSink struct {
|
|
name string
|
|
dir string
|
|
destination string
|
|
isIncremental bool
|
|
}
|
|
|
|
func (s *stubSink) GetName() string { return s.name }
|
|
func (s *stubSink) Initialize(util.Configuration, string) error { return nil }
|
|
func (s *stubSink) DeleteEntry(string, bool, bool, []int32) error {
|
|
return nil
|
|
}
|
|
func (s *stubSink) CreateEntry(string, *filer_pb.Entry, []int32) error { return nil }
|
|
func (s *stubSink) UpdateEntry(string, *filer_pb.Entry, string, *filer_pb.Entry, bool, []int32) (bool, error) {
|
|
return false, nil
|
|
}
|
|
func (s *stubSink) GetSinkToDirectory() string { return s.dir }
|
|
func (s *stubSink) GetDestinationIdentity() string { return s.destination }
|
|
func (s *stubSink) SetSourceFiler(*source.FilerSource) {}
|
|
func (s *stubSink) IsIncremental() bool { return s.isIncremental }
|
|
|
|
var _ sink.ReplicationSink = (*stubSink)(nil)
|
|
|
|
func TestInitialSnapshotTargetKey(t *testing.T) {
|
|
// Mirror the non-incremental path of buildKey so a refactor of one without
|
|
// the other will fail this test.
|
|
mirror := &stubSink{name: "mirror", isIncremental: false}
|
|
got := initialSnapshotTargetKey(mirror, "/backup", "/data", util.FullPath("/data/sub/file.txt"), &filer_pb.Entry{})
|
|
if got != "/backup/sub/file.txt" {
|
|
t.Errorf("mirror sink: got %q, want %q", got, "/backup/sub/file.txt")
|
|
}
|
|
|
|
// Incremental sinks partition by entry mtime, so the seed must use the same
|
|
// YYYY-MM-DD prefix a replayed CreateEntry would produce. buildKey in
|
|
// filer_sync.go formats the date in local time, so compute the expected
|
|
// key the same way to keep the test timezone-independent.
|
|
inc := &stubSink{name: "inc", isIncremental: true}
|
|
mtime := int64(1704196800) // 2024-01-02T12:00:00 UTC — unambiguously Jan 2 in nearly all timezones
|
|
gotInc := initialSnapshotTargetKey(inc, "/backup", "/data", util.FullPath("/data/sub/file.txt"), &filer_pb.Entry{
|
|
Attributes: &filer_pb.FuseAttributes{Mtime: mtime},
|
|
})
|
|
wantInc := "/backup/" + time.Unix(mtime, 0).Format("2006-01-02") + "/sub/file.txt"
|
|
if gotInc != wantInc {
|
|
t.Errorf("incremental sink: got %q, want %q", gotInc, wantInc)
|
|
}
|
|
|
|
// Trailing-slash sourcePath still produces a clean relative key.
|
|
gotTrail := initialSnapshotTargetKey(mirror, "/backup", "/data/", util.FullPath("/data/file.txt"), &filer_pb.Entry{})
|
|
if gotTrail != "/backup/file.txt" {
|
|
t.Errorf("trailing-slash sourcePath: got %q, want %q", gotTrail, "/backup/file.txt")
|
|
}
|
|
|
|
// Edge cases CodeRabbit called out: sourceKey equal to sourcePath
|
|
// (non-trailing and trailing variants). Real TraverseBfs walks never emit
|
|
// the root itself, but the helper must not panic if something else does.
|
|
if got := initialSnapshotTargetKey(mirror, "/backup", "/data", util.FullPath("/data"), &filer_pb.Entry{}); got != "/backup" {
|
|
t.Errorf("sourceKey == sourcePath (no slash): got %q, want %q", got, "/backup")
|
|
}
|
|
if got := initialSnapshotTargetKey(mirror, "/backup", "/data/", util.FullPath("/data"), &filer_pb.Entry{}); got != "/backup" {
|
|
t.Errorf("sourceKey == sourcePath (trailing slash mismatch): got %q, want %q", got, "/backup")
|
|
}
|
|
}
|
|
|
|
// The scenario from the collision report: two backups to different S3
|
|
// endpoints/buckets that share the destination directory "/" must not share
|
|
// a checkpoint, or the stopped one resumes from the other's position and
|
|
// skips changes.
|
|
func TestBackupCheckpointIds_DistinctDestinations(t *testing.T) {
|
|
backupA := &stubSink{name: "s3", dir: "/", destination: "s3.us-west-004.backblazeb2.com|seaweed-backup-a|/"}
|
|
backupB := &stubSink{name: "s3", dir: "/", destination: "s3.us-east-005.backblazeb2.com|seaweed-backup-b|/"}
|
|
|
|
idA, legacyA := backupCheckpointIds("/", backupA)
|
|
idB, legacyB := backupCheckpointIds("/", backupB)
|
|
|
|
if idA == idB {
|
|
t.Errorf("backups to different destinations share checkpoint id %d", idA)
|
|
}
|
|
// Both historically hashed to the same key — that is the bug the
|
|
// destination-scoped key fixes, and the shared value both fall back to.
|
|
if legacyA != legacyB {
|
|
t.Errorf("legacy ids differ: %d vs %d", legacyA, legacyB)
|
|
}
|
|
}
|
|
|
|
// Two backups of different source paths to the same destination must not
|
|
// share a checkpoint either: each stream sees a different event subset, so a
|
|
// shared offset lets the faster one push the slower one past unseen events.
|
|
func TestBackupCheckpointIds_DistinctSourcePaths(t *testing.T) {
|
|
s := &stubSink{name: "s3", dir: "/", destination: "endpoint|bucket|/"}
|
|
idA, _ := backupCheckpointIds("/buckets/a", s)
|
|
idB, _ := backupCheckpointIds("/buckets/b", s)
|
|
if idA == idB {
|
|
t.Errorf("backups of different source paths share checkpoint id %d", idA)
|
|
}
|
|
}
|
|
|
|
// The fallback key must keep the exact historical formula
|
|
// hash(GetName() + GetSinkToDirectory()) truncated to int32, or existing
|
|
// backups lose their checkpoint on upgrade and replay from zero.
|
|
func TestBackupCheckpointIds_LegacyFormulaUnchanged(t *testing.T) {
|
|
s := &stubSink{name: "s3", dir: "/data", destination: "endpoint|bucket|/data"}
|
|
_, legacy := backupCheckpointIds("/", s)
|
|
if want := int32(util.HashStringToLong("s3" + "/data")); legacy != want {
|
|
t.Errorf("legacy id = %d, want historical formula value %d", legacy, want)
|
|
}
|
|
}
|
|
|
|
// The NUL joins keep the hash input injective: field values spelling out
|
|
// other fields' content must not concatenate to the same input.
|
|
func TestBackupCheckpointIds_NoAliasing(t *testing.T) {
|
|
idA, _ := backupCheckpointIds("/src", &stubSink{name: "s3", dir: "/", destination: "/d=>s3|/other"})
|
|
idB, _ := backupCheckpointIds("/src=>s3|/d", &stubSink{name: "s3", dir: "/", destination: "/other"})
|
|
if idA == idB {
|
|
t.Errorf("field content spelling a separator aliases checkpoint id %d", idA)
|
|
}
|
|
}
|
|
|
|
// Restarting the same configuration must derive the same key, or every
|
|
// restart would orphan its checkpoint.
|
|
func TestBackupCheckpointIds_Stable(t *testing.T) {
|
|
s := &stubSink{name: "s3", dir: "/", destination: "endpoint|bucket|/"}
|
|
id1, legacy1 := backupCheckpointIds("/buckets/a", s)
|
|
id2, legacy2 := backupCheckpointIds("/buckets/a", s)
|
|
if id1 != id2 || legacy1 != legacy2 {
|
|
t.Errorf("ids not stable: (%d,%d) vs (%d,%d)", id1, legacy1, id2, legacy2)
|
|
}
|
|
}
|