Files
seaweedfs/weed/shell/command_ec_check_replication.go
Chris LuandGitHub 944d967502 refactor: extract EC orchestration into a shared weed/ec package (#10760)
* shell: move ErrorWaitGroup to weed/util

* shell: remove unused CandidateEcNode and EcRack types

* ec: extract EC orchestration logic from weed/shell into weed/ec

Move the EC node/topology model, balance engine, encode pipeline, decode
pipeline, and rebuild engine into a new weed/ec package so shell commands
and maintenance workers can share the logic. Shell commands keep flag
parsing and delegate through a small ec.Env (dial option, topology fetch,
volume locations, lock check). Tests move along with the code.

* shell: remove unused proportional-rebalance type stubs

* ec: move scrub, replication check, and shard unmount engines into weed/ec

* worker: share the EC generation-aware shard counter from weed/ec

* ec: gofmt

* shell: drop EC aliases with no remaining callers

* ec: guard a missing topology hook and nil disk entries in topology helpers

* ec: drop trailing newlines from decode error strings

* ec: re-check the shell lock before applying shard unmounts

* shell: trim -node entries in ec.scrub
2026-08-14 13:54:12 -07:00

88 lines
2.5 KiB
Go

package shell
import (
"flag"
"fmt"
"io"
"strconv"
"strings"
"github.com/seaweedfs/seaweedfs/weed/ec"
)
func init() {
Commands = append(Commands, &commandEcCheckReplication{})
}
type commandEcCheckReplication struct {
}
func (c *commandEcCheckReplication) Name() string {
return "ec.check.replication"
}
func (c *commandEcCheckReplication) Help() string {
return `check EC volumes for under- or over-replicated shards
ec.check.replication [-volumeIds=<id>,<id>...] [-details]
Reports EC volumes whose shards are:
- under-replicated: at least one shard is missing from every node
- over-replicated: at least one shard has more than one copy, whether on
different nodes or on more than one disk of a node
Over-replication is normal and transient while ec.balance or ec.encode is
running, since shards are copied before the redundant copies are deleted;
re-check once those operations finish before treating it as a problem.
Each volume is checked against its own data+parity ratio rather than a fixed
shard count, so volumes with non-default erasure coding ratios are reported
correctly.
Options:
-volumeIds: comma-separated EC volume IDs to check (default: all EC volumes)
-details: print the per-shard node placement for each flagged volume
`
}
func (c *commandEcCheckReplication) HasTag(CommandTag) bool {
return false
}
func (c *commandEcCheckReplication) Do(args []string, commandEnv *CommandEnv, writer io.Writer) (err error) {
checkReplicationCommand := flag.NewFlagSet(c.Name(), flag.ContinueOnError)
volumeIDsStr := checkReplicationCommand.String("volumeIds", "", "comma-separated EC volume IDs to process (optional)")
showDetails := checkReplicationCommand.Bool("details", false, "display result details, if available")
if err = checkReplicationCommand.Parse(args); err != nil {
return err
}
if err = commandEnv.confirmIsLocked(args); err != nil {
return
}
volumeIDMap := map[uint32]bool{}
if *volumeIDsStr != "" {
for _, vids := range strings.Split(*volumeIDsStr, ",") {
vids = strings.TrimSpace(vids)
if vids == "" {
continue
}
if vid, err := strconv.ParseUint(vids, 10, 32); err == nil {
volumeIDMap[uint32(vid)] = true
} else {
return fmt.Errorf("invalid volume ID %q", vids)
}
}
}
dataNodes, err := collectDataNodes(commandEnv, 0)
if err != nil {
return err
}
return ec.CheckEcVolumeReplication(writer, dataNodes, volumeIDMap, *showDetails)
}
// ecCheckReplicationRunner holds the state for a single ec.check.replication invocation.