Files
seaweedfs/weed/storage/erasure_coding/verification.go
T
Chris LuandGitHub 34f9b91d69 fix(storage): never let an empty .dat delete healthy distributed EC shards (#9930)
* fix(storage): never let an empty .dat delete healthy distributed EC shards

A leftover empty .dat stub (a phantom from the pre-fix loader; zero
needles) next to a distributed EC volume's local shards made startup
classify the volume as an interrupted local encode: validateEcVolume
requires >= dataShards local shards when a .dat is present, fails with
the 1-2 shards a distributed volume keeps per disk, and the cleanup
deletes those shards -- the only copies of that part of the volume.
Repeated across restart waves this destroys enough shards cluster-wide
to make the volume unrecoverable.

Go:
- loadExistingVolume: hoist the empty-stub sweep above the EC presence
  checks. Previously the .vif-next-to-.ecx guard returned before the
  sweep ever ran, so exactly the dangerous layout (stub + .ecx + local
  shards) kept its stub and then lost its shards in loadAllEcShards.
- validateEcVolume / checkDatFileExists: treat a .dat <= a superblock
  (zero needles) as absent. An empty .dat cannot be the encode source,
  so it must never gate shard deletion; this also covers stubs without
  a .vif, which the sweep cannot prove are EC leftovers.

Rust mirror (seaweed-volume): the same gate in validate_ec_volume and
check_dat_file_exists (the Rust sweep already ran before validation);
the volume-load skip keeps a plain existence check so fresh,
needle-less volumes still load.

Regression tests in Go and Rust reproduce the production layout (a
zero-byte .dat beside .ecx/.ecj and two shards of a 10+4 volume, with
and without a .vif) and fail without the fix with the shards deleted.

* fix(ec): gate source volume deletion on a recoverable shard set

After EC encode, the shell command and the (plugin) worker task refused
to delete the source volume unless every shard was present, and aborted
otherwise -- leaving the source .dat next to live shards, exactly the
mixed state the startup cleanup mishandles.

Replace the full-set requirement with a recoverability gate shared by
both callers (RequireRecoverableShardSet): deleting a non-empty source
.dat requires at least dataShards distinct shards cluster-wide. Below
that the source is kept and the encode fails as before. A degraded but
recoverable set (>= dataShards, < total) now proceeds with a warning
instead of aborting: the missing shards can be rebuilt from the
survivors, while keeping the source would preserve the dangerous mixed
state. Empty stub replicas are still swept unguarded (OnlyEmpty) -- an
empty .dat has nothing to lose.

dataShards/totalShards stay parameters so enterprise custom EC ratios
share the helper verbatim.

* test(ec): use recoverable shard verification gate
2026-06-11 20:26:20 -07:00

140 lines
4.1 KiB
Go

package erasure_coding
import (
"context"
"fmt"
"sort"
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
"google.golang.org/grpc"
)
type ServerShardInventory struct {
Bits ShardBits
QueryError error
}
// Query errors are recorded per-server and treated as zero shards rather
// than aborting the scan, so the caller still sees partial coverage from
// healthy peers when one server is down. The caller gates destructive
// actions on RequireRecoverableShardSet against the returned union.
func VerifyShardsAcrossServers(ctx context.Context, volumeID uint32,
servers []string, dialOption grpc.DialOption) (
union ShardBits, perServer map[string]ServerShardInventory) {
perServer = make(map[string]ServerShardInventory, len(servers))
for _, server := range servers {
if server == "" {
continue
}
if _, seen := perServer[server]; seen {
continue
}
var inv ServerShardInventory
callErr := operation.WithVolumeServerClient(false, pb.ServerAddress(server), dialOption,
func(client volume_server_pb.VolumeServerClient) error {
resp, e := client.VolumeEcShardsInfo(ctx, &volume_server_pb.VolumeEcShardsInfoRequest{
VolumeId: volumeID,
})
if e != nil {
return e
}
for _, s := range resp.EcShardInfos {
if s.VolumeId != volumeID || s.ShardId >= MaxShardCount {
continue
}
inv.Bits = inv.Bits.Set(ShardId(s.ShardId))
}
return nil
})
if callErr != nil {
inv.QueryError = callErr
}
perServer[server] = inv
union = ShardBits(uint32(union) | uint32(inv.Bits))
}
return union, perServer
}
// RequireRecoverableShardSet gates source-volume deletion after EC encode:
// a non-empty .dat may only be deleted when enough distinct shards exist to
// reconstruct the volume (>= dataShards). A full set returns (false, nil); a
// degraded-but-recoverable set returns (true, nil) so the caller can warn and
// proceed -- the missing shards can be rebuilt from the survivors, while
// keeping the source next to live shards is the more dangerous mixed state.
// Below dataShards it returns an error and the source must be kept.
// dataShards/totalShards are passed as parameters (not derived from the
// package constants) so enterprise builds with custom EC ratios share this
// helper verbatim.
func RequireRecoverableShardSet(volumeID uint32, shardsPresent ShardBits, dataShards, totalShards int) (degraded bool, err error) {
if totalShards <= 0 || totalShards > MaxShardCount {
return false, fmt.Errorf("invalid totalShards %d for volume %d (must be in [1, %d])",
totalShards, volumeID, MaxShardCount)
}
if dataShards <= 0 || dataShards > totalShards {
return false, fmt.Errorf("invalid dataShards %d for volume %d (must be in [1, %d])",
dataShards, volumeID, totalShards)
}
var missing []int
for id := 0; id < totalShards; id++ {
if !shardsPresent.Has(ShardId(id)) {
missing = append(missing, id)
}
}
if len(missing) == 0 {
return false, nil
}
if totalShards-len(missing) >= dataShards {
return true, nil
}
sort.Ints(missing)
return false, fmt.Errorf("EC shard set unrecoverable for volume %d: %d/%d shards present, need %d to reconstruct, missing shard ids %v",
volumeID, totalShards-len(missing), totalShards, dataShards, missing)
}
func SummarizeShardInventory(perServer map[string]ServerShardInventory) string {
servers := make([]string, 0, len(perServer))
for s := range perServer {
servers = append(servers, s)
}
sort.Strings(servers)
var b []byte
for i, s := range servers {
if i > 0 {
b = append(b, ' ')
}
inv := perServer[s]
b = append(b, s...)
b = append(b, '=')
b = append(b, '[')
ids := make([]int, 0)
for id := 0; id < MaxShardCount; id++ {
if inv.Bits.Has(ShardId(id)) {
ids = append(ids, id)
}
}
for j, id := range ids {
if j > 0 {
b = append(b, ' ')
}
b = append(b, []byte(fmt.Sprintf("%d", id))...)
}
if inv.QueryError != nil {
if len(ids) > 0 {
b = append(b, ' ')
}
b = append(b, []byte("ERR:"+inv.QueryError.Error())...)
}
b = append(b, ']')
}
return string(b)
}