From 613ad207b3da1fda565f0ec5ab8cd023f7b7e4b4 Mon Sep 17 00:00:00 2001 From: Chris Lu Date: Sat, 22 Aug 2026 14:27:17 -0700 Subject: [PATCH] master: never re-seed a raft cluster over committed state -raftBootstrap deleted logs.dat, stable.dat and snapshots on every start and then bootstrapped a fresh cluster. Since hashicorp raft only snapshots after 8192 log entries, the TopologyId lives in the log, not in a snapshot, so the pre-wipe snapshot recovery found nothing and each restart minted a new cluster identity. A master that came up while it could not reach its peers seeded a rival cluster; when the two logs met, SetTopologyId's split-brain guard fatally stopped every master holding the other id, and the master layer crash-looped with no quorum. Bootstrapping is genesis. Drop the wipe and the inline bootstrap. The first master in -peers already mints a cluster once it has confirmed no peer has a leader, so the flag has nothing left to do and is now ignored; keeping that one master the sole bootstrap authority is what stops a partition from minting two clusters, so the flag must not widen it either. A master with state rejoins its peers, and one whose data dir was reset is admitted by the sitting leader instead of forking again. --- weed/command/master.go | 13 +++++++---- weed/server/raft_hashicorp.go | 42 +++-------------------------------- weed/server/raft_server.go | 4 ++-- 3 files changed, 14 insertions(+), 45 deletions(-) diff --git a/weed/command/master.go b/weed/command/master.go index ebe5b8f0d..87a3cd68a 100644 --- a/weed/command/master.go +++ b/weed/command/master.go @@ -104,7 +104,7 @@ func init() { m.heartbeatInterval = cmdMaster.Flag.Duration("heartbeatInterval", 300*time.Millisecond, "heartbeat interval of master servers, and will be randomly multiplied by [1, 1.25)") m.electionTimeout = cmdMaster.Flag.Duration("electionTimeout", 10*time.Second, "election timeout of master servers") m.raftHashicorp = cmdMaster.Flag.Bool("raftHashicorp", false, "use hashicorp raft") - m.raftBootstrap = cmdMaster.Flag.Bool("raftBootstrap", false, "Whether to bootstrap the Raft cluster") + m.raftBootstrap = cmdMaster.Flag.Bool("raftBootstrap", false, "deprecated and ignored: the first master in -peers mints the Raft cluster on its own once it sees no leader anywhere") m.telemetryUrl = cmdMaster.Flag.String("telemetry.url", "https://telemetry.seaweedfs.com/api/collect", "telemetry server URL to send usage statistics") m.telemetryEnabled = cmdMaster.Flag.Bool("telemetry", true, "report anonymous cluster statistics to telemetry.url, use -telemetry=false to opt out") m.debug = cmdMaster.Flag.Bool("debug", false, "serves runtime profiling data via pprof on the port specified by -debug.port") @@ -211,6 +211,10 @@ func startMaster(masterOption MasterOptions, masterWhiteList []string) { isSingleMaster := isSingleMasterMode(*masterOption.peers) + if *masterOption.raftBootstrap { + glog.V(0).Infof("-raftBootstrap is ignored: masters mint a cluster on their own when no peer has a leader, and never over existing raft state") + } + raftServerOption := &weed_server.RaftServerOption{ GrpcDialOption: security.LoadClientTLS(util.GetViper(), "grpc.master"), Peers: masterPeers, @@ -221,7 +225,6 @@ func startMaster(masterOption MasterOptions, masterWhiteList []string) { SingleMaster: isSingleMaster, HeartbeatInterval: *masterOption.heartbeatInterval, ElectionTimeout: *masterOption.electionTimeout, - RaftBootstrap: *masterOption.raftBootstrap, } var raftServer *weed_server.RaftServer var err error @@ -278,8 +281,10 @@ func startMaster(masterOption MasterOptions, masterWhiteList []string) { // raft implementation lets a server outside the configuration campaign — so // it has to be pulled in by a leader. Keep asking the peers who the leader is // until we are in: the leader admits us once our master client registers, and - // only when nobody has one does the first peer mint a new cluster. Restarting - // a master alone, or scaling the peer list up, both land here. + // only when nobody has one does the first peer mint a new cluster. Keeping + // that one peer the sole authority is what stops a partition from minting + // two clusters. Restarting a master alone, or scaling the peer list up, + // both land here. if !isSingleMaster { go func() { // Stagger bootstrap by peer index so masters don't all check diff --git a/weed/server/raft_hashicorp.go b/weed/server/raft_hashicorp.go index 7734fb364..f64c0d26d 100644 --- a/weed/server/raft_hashicorp.go +++ b/weed/server/raft_hashicorp.go @@ -6,7 +6,6 @@ package weed_server import ( "encoding/json" "fmt" - "io" "math/rand/v2" "os" "path" @@ -36,28 +35,6 @@ func raftServerID(server pb.ServerAddress) string { return server.ToHttpAddress() } -// recoverTopologyIdFromHashicorpSnapshot reads the TopologyId from the latest -// hashicorp raft snapshot before state cleanup. -func recoverTopologyIdFromHashicorpSnapshot(dataDir string, topo *topology.Topology) { - fss, err := raft.NewFileSnapshotStore(dataDir, 1, io.Discard) - if err != nil { - return - } - snapshots, err := fss.List() - if err != nil || len(snapshots) == 0 { - return - } - _, rc, err := fss.Open(snapshots[0].ID) - if err != nil { - return - } - defer rc.Close() - - if b, err := io.ReadAll(rc); err == nil { - recoverTopologyIdFromState(b, topo) - } -} - func (s *RaftServer) AddPeersConfiguration() (cfg raft.Configuration) { for _, peer := range s.peers { cfg.Servers = append(cfg.Servers, raft.Server{ @@ -169,13 +146,6 @@ func NewHashicorpRaftServer(option *RaftServerOption) (*RaftServer, error) { return nil, fmt.Errorf("raft.ValidateConfig: %w", err) } - if option.RaftBootstrap { - recoverTopologyIdFromHashicorpSnapshot(s.dataDir, option.Topo) - - os.RemoveAll(path.Join(s.dataDir, ldbFile)) - os.RemoveAll(path.Join(s.dataDir, sdbFile)) - os.RemoveAll(path.Join(s.dataDir, "snapshots")) - } if err := os.MkdirAll(path.Join(s.dataDir, "snapshots"), os.ModePerm); err != nil { return nil, err } @@ -204,16 +174,10 @@ func NewHashicorpRaftServer(option *RaftServerOption) (*RaftServer, error) { return nil, fmt.Errorf("raft.NewRaft: %w", err) } - // An explicit -raftBootstrap mints the cluster right here. Otherwise the - // caller bootstraps, once it has confirmed no peer already has a leader: - // bootstrapping next to a live leader forms a second cluster instead of - // joining the first one. + // The caller bootstraps, once it has confirmed no peer already has a + // leader: bootstrapping next to a live leader forms a second cluster + // instead of joining the first one. updatePeers := len(s.RaftHashicorp.GetConfiguration().Configuration().Servers) > 0 - if option.RaftBootstrap { - if err := s.Bootstrap(); err != nil { - return nil, fmt.Errorf("raft.Raft.BootstrapCluster: %w", err) - } - } go s.monitorLeaderLoop(updatePeers) diff --git a/weed/server/raft_server.go b/weed/server/raft_server.go index ef6799512..55d92254c 100644 --- a/weed/server/raft_server.go +++ b/weed/server/raft_server.go @@ -35,7 +35,6 @@ type RaftServerOption struct { SingleMaster bool HeartbeatInterval time.Duration ElectionTimeout time.Duration - RaftBootstrap bool } type RaftServer struct { @@ -306,7 +305,8 @@ func (s *RaftServer) HasExistingState() bool { // Bootstrap mints a new raft cluster out of the configured peers. Only call it // when no leader exists anywhere: a master that starts with no raft state and // finds a leader must be admitted by that leader instead, or the two clusters -// never merge. +// never merge. Committed state is never bootstrapped over — hashicorp refuses, +// and goraft callers reach here only past HasExistingState. func (s *RaftServer) Bootstrap() error { glog.V(0).Infoln("Initializing new cluster")