Files
seaweedfs/sw-block/engine/replication/runtime/pending.go
T
pingqiuandClaude Opus 4.6 b304b8e212 refactor: make bounded recovery command addressing replica-scoped
Replace the remaining volume-scoped recovery command and pending slot
with replica-scoped addressing on the bounded core-present path. This
preserves the current single-replica catch-up and rebuilding behavior
while removing the structural blocker for later multi-replica startup
ownership.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-04 09:05:36 -07:00

166 lines
4.8 KiB
Go

// Package runtime provides reusable recovery-coordination helpers for the
// V2 engine. These helpers are independent of weed/ host specifics and can
// be used by any adapter shell.
package runtime
import (
"sync"
engine "github.com/seaweedfs/seaweedfs/sw-block/engine/replication"
)
// PendingExecution holds the state needed to execute a planned recovery
// action. The coordinator stores one pending execution per replica target and
// matches it against incoming commands.
//
// All fields are typed — no interface{} handles. The host adapter builds
// these from concrete BlockVol bindings; the coordinator and execution
// helpers consume them without type assertions.
type PendingExecution struct {
VolumeID string
ReplicaID string
// CatchUpTarget is the target LSN for catch-up execution.
// Used by TakeCatchUp for fail-closed target matching.
CatchUpTarget uint64
// RebuildTargetLSN is the target LSN for rebuild execution.
// Used by TakeRebuild for fail-closed target matching.
RebuildTargetLSN uint64
// Typed recovery handles — no interface{} drift.
Driver *engine.RecoveryDriver
Plan *engine.RecoveryPlan
CatchUpIO engine.CatchUpIO
RebuildIO engine.RebuildIO
}
// CancelFunc is called when a pending execution is cancelled due to
// mismatch or supersession. The reason string explains why.
type CancelFunc func(pending *PendingExecution, reason string)
// PendingCoordinator manages pending recovery executions with fail-closed
// command matching. It is safe for concurrent use.
//
// Flow:
// 1. Store() caches a planned execution after recovery planning completes
// 2. TakeCatchUp/TakeRebuild matches an incoming command against the cached plan
// 3. If the target doesn't match, the pending plan is cancelled (fail-closed)
// 4. Cancel() explicitly cancels a pending execution
type PendingCoordinator struct {
mu sync.Mutex
pending map[string]*PendingExecution
cancelFn CancelFunc
}
// NewPendingCoordinator creates a coordinator with the given cancel callback.
// The cancel function is called when a pending execution is cancelled due to
// target mismatch or explicit cancellation.
func NewPendingCoordinator(cancelFn CancelFunc) *PendingCoordinator {
return &PendingCoordinator{
pending: make(map[string]*PendingExecution),
cancelFn: cancelFn,
}
}
// Store caches a pending execution for one replica target, replacing any
// previous one.
func (pc *PendingCoordinator) Store(replicaID string, pe *PendingExecution) {
pc.mu.Lock()
defer pc.mu.Unlock()
pc.pending[replicaID] = pe
}
// TakeCatchUp takes the pending execution for the replica if the catch-up
// target matches. If there's a mismatch, the pending execution is cancelled
// (fail-closed) and nil is returned. If no pending execution exists, nil
// is returned.
func (pc *PendingCoordinator) TakeCatchUp(replicaID string, targetLSN uint64) *PendingExecution {
pc.mu.Lock()
pe, ok := pc.pending[replicaID]
if ok {
delete(pc.pending, replicaID)
}
pc.mu.Unlock()
if !ok || pe == nil {
return nil
}
if pe.CatchUpTarget != targetLSN {
if pc.cancelFn != nil {
pc.cancelFn(pe, "start_catchup_target_mismatch")
}
return nil
}
return pe
}
// TakeRebuild takes the pending execution for the replica if the rebuild
// target matches. Same fail-closed semantics as TakeCatchUp.
func (pc *PendingCoordinator) TakeRebuild(replicaID string, targetLSN uint64) *PendingExecution {
pc.mu.Lock()
pe, ok := pc.pending[replicaID]
if ok {
delete(pc.pending, replicaID)
}
pc.mu.Unlock()
if !ok || pe == nil {
return nil
}
if pe.RebuildTargetLSN != targetLSN {
if pc.cancelFn != nil {
pc.cancelFn(pe, "start_rebuild_target_mismatch")
}
return nil
}
return pe
}
// Has returns true if a pending execution exists for the replica target.
func (pc *PendingCoordinator) Has(replicaID string) bool {
pc.mu.Lock()
defer pc.mu.Unlock()
_, ok := pc.pending[replicaID]
return ok
}
// Cancel explicitly cancels and removes the pending execution for a replica
// target.
func (pc *PendingCoordinator) Cancel(replicaID, reason string) {
pc.mu.Lock()
pe, ok := pc.pending[replicaID]
if ok {
delete(pc.pending, replicaID)
}
pc.mu.Unlock()
if ok && pe != nil && pc.cancelFn != nil {
pc.cancelFn(pe, reason)
}
}
// Peek returns the pending execution without removing it. Returns nil if none.
func (pc *PendingCoordinator) Peek(replicaID string) *PendingExecution {
pc.mu.Lock()
defer pc.mu.Unlock()
return pc.pending[replicaID]
}
// CancelAll cancels and removes all pending executions.
func (pc *PendingCoordinator) CancelAll(reason string) {
pc.mu.Lock()
all := make(map[string]*PendingExecution, len(pc.pending))
for k, v := range pc.pending {
all[k] = v
}
pc.pending = make(map[string]*PendingExecution)
pc.mu.Unlock()
if pc.cancelFn != nil {
for _, pe := range all {
pc.cancelFn(pe, reason)
}
}
}