mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-21 22:56:55 +00:00
fix: CP13-7 rev2 — real handshake gap detection, reclassify rebuild test
Two fixes: 1. TestReconnect_GapBeyondRetainedWal_NeedsRebuild: rewritten to test the real reconnect handshake gap detection path (R < S in reconnectWithHandshake). Sequence: establish sync → disconnect → release retention hold via timeout → write + flush to advance WAL past replica position → reconnect → handshake detects R=0 < S=9 → NeedsRebuild. Log proves: "reconnect: gap too large R=0 H=8 S=9" 2. TestReplicaState_RebuildComplete_ReentersInSync: reclassified from primary proof to support evidence (does not start from live NeedsRebuild shipper state, but proves rebuild mechanics work end-to-end). Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
88c336b1c1
commit
ec63c18438
@@ -53,7 +53,7 @@ Code change: `sync_all_adversarial_test.go` + `sync_all_protocol_test.go` test r
|
||||
| Test | Was | Now | What it proves |
|
||||
|------|-----|-----|----------------|
|
||||
| `TestAdversarial_NeedsRebuildBlocksAllPaths` | FAIL | PASS | NeedsRebuild blocks Ship (drops) + Barrier (rejects) + is sticky across retries |
|
||||
| `TestReconnect_GapBeyondRetainedWal_NeedsRebuild` | PASS* | PASS | Unrecoverable gap → NeedsRebuild state assertion + SyncCache fails |
|
||||
| `TestReconnect_GapBeyondRetainedWal_NeedsRebuild` | PASS* | PASS | Real reconnect handshake gap detection (R < S path), not budget trigger |
|
||||
|
||||
## Proof Promotion
|
||||
|
||||
@@ -62,12 +62,17 @@ Code change: `sync_all_adversarial_test.go` + `sync_all_protocol_test.go` test r
|
||||
| Test | What it proves for CP13-7 |
|
||||
|------|--------------------------|
|
||||
| `TestAdversarial_NeedsRebuildBlocksAllPaths` | 5 assertions: NeedsRebuild state, Ship drops, Barrier rejects, state sticky after barrier, second SyncCache still fails |
|
||||
| `TestReconnect_GapBeyondRetainedWal_NeedsRebuild` | Unrecoverable gap → hard NeedsRebuild assertion + SyncCache failure |
|
||||
| `TestReconnect_GapBeyondRetainedWal_NeedsRebuild` | Real reconnect handshake detects R < S (gap beyond retained WAL) → SyncCache fails |
|
||||
| `TestHeartbeat_ReportsNeedsRebuild` | Heartbeat carries per-replica `needs_rebuild` state |
|
||||
| `TestReplicaState_RebuildComplete_ReentersInSync` | Full rebuild cycle: NeedsRebuild → rebuild → fresh shipper → InSync |
|
||||
| `TestRebuild_AbortOnEpochChange` | Epoch mismatch during rebuild → abort |
|
||||
| `TestRebuild_PostRebuild_FlushedLSN_IsCheckpoint` | Post-rebuild `flushedLSN = checkpointLSN` (not stale/zero) |
|
||||
|
||||
### Support evidence
|
||||
|
||||
| Test | What it supports |
|
||||
|------|-----------------|
|
||||
| `TestReplicaState_RebuildComplete_ReentersInSync` | Rebuild completion flow (reopen volume → RoleRebuilding → StartRebuild → fresh shipper → InSync). Support evidence: does not start from live NeedsRebuild shipper state, but proves the rebuild mechanics work end-to-end. |
|
||||
|
||||
## Updated Baseline Summary
|
||||
|
||||
| | PASS | FAIL | PASS* |
|
||||
|
||||
@@ -298,10 +298,19 @@ func TestReconnect_CatchupFromRetainedWal(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestReconnect_GapBeyondRetainedWal_NeedsRebuild verifies that when the
|
||||
// replica's gap exceeds the retained WAL range, the system transitions to
|
||||
// NeedsRebuild instead of silently losing data.
|
||||
// replica's gap exceeds the retained WAL range, the reconnect handshake
|
||||
// detects this and transitions to NeedsRebuild.
|
||||
//
|
||||
// CP13-7 proof: unrecoverable gap → NeedsRebuild transition + SyncCache fails.
|
||||
// CP13-7 proof: real reconnect handshake gap detection (R < S path in
|
||||
// reconnectWithHandshake), not just budget-triggered escalation.
|
||||
//
|
||||
// Sequence:
|
||||
// 1. Establish sync (replica at LSN 1)
|
||||
// 2. Disconnect replica
|
||||
// 3. Release retention hold via timeout budget on old shipper
|
||||
// 4. Write + flush to advance WAL tail past replica's flushedLSN
|
||||
// 5. Reconnect (new shipper seeded with hasFlushedProgress=true)
|
||||
// 6. SyncCache → reconnectWithHandshake → detects R < S → NeedsRebuild
|
||||
func TestReconnect_GapBeyondRetainedWal_NeedsRebuild(t *testing.T) {
|
||||
primary, replica := createSyncAllPair(t)
|
||||
defer primary.Close()
|
||||
@@ -315,7 +324,7 @@ func TestReconnect_GapBeyondRetainedWal_NeedsRebuild(t *testing.T) {
|
||||
|
||||
primary.SetReplicaAddr(recv.DataAddr(), recv.CtrlAddr())
|
||||
|
||||
// Write and sync while healthy.
|
||||
// Step 1: Write and sync while healthy.
|
||||
if err := primary.WriteLBA(0, makeBlock('A')); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -325,50 +334,79 @@ func TestReconnect_GapBeyondRetainedWal_NeedsRebuild(t *testing.T) {
|
||||
|
||||
sg := primary.shipperGroup
|
||||
s := sg.Shipper(0)
|
||||
if s.State() != ReplicaInSync {
|
||||
t.Fatalf("expected InSync, got %s", s.State())
|
||||
}
|
||||
replicaFlushed := s.ReplicaFlushedLSN()
|
||||
t.Logf("replica flushedLSN after sync: %d", replicaFlushed)
|
||||
|
||||
// Disconnect replica.
|
||||
// Step 2: Disconnect replica.
|
||||
recv.Stop()
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
|
||||
// Write entries (within WAL capacity).
|
||||
// Step 3: Release retention hold via timeout budget on the old shipper.
|
||||
// This transitions the old shipper to NeedsRebuild, so
|
||||
// MinRecoverableFlushedLSN no longer pins the WAL.
|
||||
sg.EvaluateRetentionBudgets(RetentionBudgetParams{
|
||||
Timeout: 1 * time.Nanosecond, // force timeout
|
||||
MaxBytes: 0,
|
||||
PrimaryHeadLSN: primary.nextLSN.Load() - 1,
|
||||
BlockSize: primary.super.BlockSize,
|
||||
})
|
||||
if s.State() != ReplicaNeedsRebuild {
|
||||
t.Fatalf("old shipper should be NeedsRebuild after timeout, got %s", s.State())
|
||||
}
|
||||
|
||||
// Step 4: Write + flush to advance WAL tail past replica's flushedLSN.
|
||||
// The retention hold is released, so writes won't block on WAL admission.
|
||||
for i := uint64(1); i < 8; i++ {
|
||||
if err := primary.WriteLBA(i, makeBlock(byte('0'+i))); err != nil {
|
||||
t.Fatalf("write %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
primary.flusher.FlushOnce()
|
||||
primary.flusher.FlushOnce()
|
||||
|
||||
// CP13-6 → CP13-7: trigger NeedsRebuild via max-bytes budget.
|
||||
// This simulates the case where the gap exceeds retention budget.
|
||||
sg.EvaluateRetentionBudgets(RetentionBudgetParams{
|
||||
Timeout: 5 * time.Minute,
|
||||
MaxBytes: 4 * 1024, // small budget, lag exceeds it
|
||||
PrimaryHeadLSN: primary.nextLSN.Load() - 1,
|
||||
BlockSize: primary.super.BlockSize,
|
||||
})
|
||||
|
||||
// Assert NeedsRebuild state.
|
||||
if s.State() != ReplicaNeedsRebuild {
|
||||
t.Fatalf("CP13-7: expected NeedsRebuild after unrecoverable gap, got %s", s.State())
|
||||
// Verify checkpoint advanced past replica's position (WAL reclaimed).
|
||||
checkpointAfterFlush := primary.flusher.CheckpointLSN()
|
||||
t.Logf("after flush: checkpoint=%d replicaFlushed=%d", checkpointAfterFlush, replicaFlushed)
|
||||
if checkpointAfterFlush <= replicaFlushed {
|
||||
t.Fatalf("checkpoint should advance past replicaFlushed after hold released: checkpoint=%d replicaFlushed=%d",
|
||||
checkpointAfterFlush, replicaFlushed)
|
||||
}
|
||||
|
||||
// SyncCache must fail — NeedsRebuild shipper cannot satisfy barrier.
|
||||
// Step 5: Reconnect with new receiver.
|
||||
recv2, err := NewReplicaReceiver(replica, "127.0.0.1:0", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
recv2.Serve()
|
||||
defer recv2.Stop()
|
||||
|
||||
// SetReplicaAddrs creates a new shipper seeded with hasFlushedProgress=true (CP13-5).
|
||||
primary.SetReplicaAddr(recv2.DataAddr(), recv2.CtrlAddr())
|
||||
|
||||
// Step 6: SyncCache triggers reconnect handshake on the new shipper.
|
||||
// The handshake sends ResumeShipReq{HeadLSN, RetainStart}.
|
||||
// Replica responds with its flushedLSN (~1).
|
||||
// Handshake gap analysis: R(1) < S(retainStart) → NeedsRebuild.
|
||||
syncDone := make(chan error, 1)
|
||||
go func() {
|
||||
syncDone <- primary.SyncCache()
|
||||
}()
|
||||
|
||||
select {
|
||||
case err := <-syncDone:
|
||||
if err == nil {
|
||||
t.Fatal("CP13-7: SyncCache should fail with NeedsRebuild")
|
||||
t.Fatal("SyncCache should fail — handshake should detect gap beyond retained WAL")
|
||||
}
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("SyncCache hung")
|
||||
}
|
||||
|
||||
t.Logf("CP13-7: unrecoverable gap → NeedsRebuild, SyncCache correctly fails")
|
||||
// Verify the NEW shipper detected the gap via handshake (not just budget).
|
||||
newS := primary.shipperGroup.Shipper(0)
|
||||
if newS.State() != ReplicaNeedsRebuild && newS.State() != ReplicaDegraded {
|
||||
t.Fatalf("new shipper should be NeedsRebuild or Degraded after handshake gap detection, got %s", newS.State())
|
||||
}
|
||||
t.Logf("CP13-7: reconnect handshake detected gap beyond retained WAL (state=%s)", newS.State())
|
||||
}
|
||||
|
||||
// ---------- WAL retention ----------
|
||||
|
||||
Reference in New Issue
Block a user