mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-16 20:26:45 +00:00
A decode deletes the shards only after the regenerated volume is mounted and verified, so a run interrupted in that last phase leaves the volume in place with its shards partway through deletion. The re-run then finds both, tries to collect the shards again to rebuild a volume that already exists, and fails on the first shard the interrupted run had removed: generate normal volume 3 ...: ec volume 3 missing shard 6 Nothing recovers from there: the shard set is deliberately being destroyed, so every retry fails the same way while the decoded volume sits there, already complete. Finish that cleanup instead. A volume beside the shards is not enough to act on -- an encode interrupted before it deleted the original leaves the same shape, as does a decode killed while generating, whose volume may be half written -- so require a data shard to be gone. Only the deletion phase removes one, and it is also exactly the state no decode can recover from, so finishing is the only move left rather than a choice between two. The deletion still runs behind verifyDecodedVolumeBeforeDelete, the check that guards it in a normal run.
92 lines
2.9 KiB
Go
92 lines
2.9 KiB
Go
package ec
|
|
|
|
import (
|
|
"testing"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
|
)
|
|
|
|
// shardsOn builds one holder's inventory.
|
|
func shardsOn(ids ...int) *erasure_coding.ShardsInfo {
|
|
si := erasure_coding.NewShardsInfo()
|
|
for _, id := range ids {
|
|
si.Set(erasure_coding.NewShardInfo(erasure_coding.ShardId(id), 1024))
|
|
}
|
|
return si
|
|
}
|
|
|
|
// missingDataShard decides whether a decode may delete the shards of a volume
|
|
// that already exists elsewhere. Only the deletion phase of a decode removes a
|
|
// data shard, so the answer separates "an earlier decode was interrupted while
|
|
// cleaning up" from the two states that look the same from the outside: an
|
|
// encode interrupted before it deleted the original, and a decode killed while
|
|
// generating. Both of those leave every shard in place, and the volume beside
|
|
// them may be the untouched original or a half-written one -- neither is safe
|
|
// to trade for the shards.
|
|
func TestMissingDataShard(t *testing.T) {
|
|
const dataShards = 10
|
|
|
|
tests := []struct {
|
|
name string
|
|
holders map[pb.ServerAddress]*erasure_coding.ShardsInfo
|
|
wantID erasure_coding.ShardId
|
|
wantMissed bool
|
|
}{
|
|
{
|
|
name: "complete set on one holder decodes, so nothing is finished",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{
|
|
"server1:8080": shardsOn(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13),
|
|
},
|
|
},
|
|
{
|
|
name: "complete set spread across holders is still complete",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{
|
|
"server1:8080": shardsOn(0, 1, 2, 3, 4),
|
|
"server2:8080": shardsOn(5, 6, 7, 8, 9),
|
|
"server3:8080": shardsOn(10, 11, 12, 13),
|
|
},
|
|
},
|
|
{
|
|
name: "parity gone is not the deletion phase: the volume still decodes",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{
|
|
"server1:8080": shardsOn(0, 1, 2, 3, 4, 5, 6, 7, 8, 9),
|
|
},
|
|
},
|
|
{
|
|
name: "a data shard gone is a decode that was interrupted mid-cleanup",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{
|
|
"server1:8080": shardsOn(0, 1, 2, 3, 4, 5, 7, 8, 9, 10, 11, 12, 13),
|
|
},
|
|
wantID: 6,
|
|
wantMissed: true,
|
|
},
|
|
{
|
|
name: "the first gap is reported, not the last",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{
|
|
"server1:8080": shardsOn(0, 3, 4, 5, 6, 7, 8, 9),
|
|
},
|
|
wantID: 1,
|
|
wantMissed: true,
|
|
},
|
|
{
|
|
name: "no holders at all leaves every data shard missing",
|
|
holders: map[pb.ServerAddress]*erasure_coding.ShardsInfo{},
|
|
wantID: 0,
|
|
wantMissed: true,
|
|
},
|
|
}
|
|
|
|
for _, tt := range tests {
|
|
t.Run(tt.name, func(t *testing.T) {
|
|
id, missed := missingDataShard(tt.holders, dataShards)
|
|
if missed != tt.wantMissed {
|
|
t.Fatalf("missingDataShard = %v, want %v", missed, tt.wantMissed)
|
|
}
|
|
if missed && id != tt.wantID {
|
|
t.Errorf("missing shard id = %d, want %d", id, tt.wantID)
|
|
}
|
|
})
|
|
}
|
|
}
|