mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-01 12:16:07 +00:00
fix(scheduler): give worker tasks a real per-attempt execution deadline (#9041)
* fix(scheduler): give worker tasks a real per-attempt execution deadline The plugin scheduler derived the per-attempt execution deadline as DetectionTimeoutSeconds * 2, which capped every worker task at twice the cluster-scan budget regardless of actual work. For volume_balance batches this was 240s — far too short for 20 large volume copies, so every attempt died at "context deadline exceeded" and all in-flight sub-RPCs surfaced as "context canceled". Retries restarted from move 1 and hit the same wall. Add an explicit ExecutionTimeoutSeconds field to the plugin proto and make each handler declare its own baseline (1800s for vacuum, balance, EC; 3600s for iceberg). Size-aware handlers also emit an estimated_runtime_seconds parameter on each proposal so the scheduler extends the per-attempt deadline based on actual workload: - volume_balance batch: max(largest single move, total / concurrency) at 5 min/GB, so a skewed batch with one big volume isn't averaged away. - volume_balance single, vacuum (already), erasure_coding (10 min/GB), ec_balance (5 min/GB): per-volume budgets. admin_script and iceberg keep the configurable handler default since their workloads are opaque to the detector. * fix(scheduler): apply descriptor defaults to existing persisted configs The previous commit added execution_timeout_seconds to the proto and each handler's descriptor defaults, but two paths still left existing deployments broken: 1. deriveSchedulerAdminRuntime returned stored AdminRuntime configs as-is. Persisted configs from older versions have no execution_timeout_seconds, so the scheduler fell back to the 90s default — worse than the prior 240s behavior. Overlay descriptor defaults for any zero numeric fields when loading. 2. The admin form did not round-trip execution_timeout_seconds, so a normal save would clear it back to zero. Add the input field, the fillAdminSettings/collectAdminSettings hooks, and as defense in depth reapply descriptor defaults in UpdatePluginJobTypeConfigAPI before persisting so a stale form can never silently clobber a baseline. * fix(volume_balance): account for partial scheduling rounds in batch estimate With N moves and C slots, the busiest slot processes ceil(N/C) moves, not N/C. Dividing total seconds by C underestimates wall-clock time whenever N is not a multiple of C — e.g. 6 moves at concurrency 5 needs 2 rounds, not 1.2. Use avg * ceil(N/C) so partial rounds are counted as full ones. * fix(volume_balance): scale minBudget per wave instead of per move Orchestration overhead (setup/teardown for the parallel move runner) happens once per wave, not once per move. Use numRounds*60 as the floor instead of len(moves)*60 so the minimum doesn't inflate linearly with batch size when individual moves are tiny.
This commit is contained in:
@@ -123,6 +123,7 @@ func (h *AdminScriptHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 0,
|
||||
RetryBackoffSeconds: 30,
|
||||
JobTypeMaxRuntimeSeconds: 1800,
|
||||
ExecutionTimeoutSeconds: 1800,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{},
|
||||
}
|
||||
|
||||
@@ -167,6 +167,7 @@ func (h *ECBalanceHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 1,
|
||||
RetryBackoffSeconds: 30,
|
||||
JobTypeMaxRuntimeSeconds: 1800,
|
||||
ExecutionTimeoutSeconds: 1800,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{
|
||||
"imbalance_threshold": {Kind: &plugin_pb.ConfigValue_DoubleValue{DoubleValue: 0.2}},
|
||||
@@ -456,6 +457,12 @@ func buildECBalanceProposal(result *workertypes.TaskDetectionResult) (*plugin_pb
|
||||
summary = fmt.Sprintf("Move EC shard of volume %d: %s → %s", result.VolumeID, sourceNode, targetNode)
|
||||
}
|
||||
|
||||
// EC shard moves only relocate one shard (1/14 of the volume). Budget
|
||||
// 5 min/GB of full volume size, which conservatively covers the shard
|
||||
// transfer plus mount/registration overhead.
|
||||
volumeSizeGB := int64(result.TypedParams.VolumeSize/1024/1024/1024) + 1
|
||||
estimatedRuntimeSeconds := volumeSizeGB * 5 * 60
|
||||
|
||||
return &plugin_pb.JobProposal{
|
||||
ProposalId: proposalID,
|
||||
DedupeKey: dedupeKey,
|
||||
@@ -479,6 +486,9 @@ func buildECBalanceProposal(result *workertypes.TaskDetectionResult) (*plugin_pb
|
||||
"collection": {
|
||||
Kind: &plugin_pb.ConfigValue_StringValue{StringValue: result.Collection},
|
||||
},
|
||||
"estimated_runtime_seconds": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: estimatedRuntimeSeconds},
|
||||
},
|
||||
},
|
||||
Labels: map[string]string{
|
||||
"task_type": "ec_balance",
|
||||
|
||||
@@ -166,6 +166,7 @@ func (h *ErasureCodingHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 1,
|
||||
RetryBackoffSeconds: 30,
|
||||
JobTypeMaxRuntimeSeconds: 1800,
|
||||
ExecutionTimeoutSeconds: 1800,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{
|
||||
"quiet_for_seconds": {
|
||||
@@ -634,6 +635,12 @@ func buildErasureCodingProposal(
|
||||
summary = fmt.Sprintf("Erasure code volume %d from %s", result.VolumeID, sourceNode)
|
||||
}
|
||||
|
||||
// EC encoding reads the full volume, computes shards, and writes 14
|
||||
// shards out to target nodes. Budget 10 min/GB (roughly 2x a plain copy)
|
||||
// so the scheduler grants a deadline scaled to volume size.
|
||||
volumeSizeGB := int64(result.TypedParams.VolumeSize/1024/1024/1024) + 1
|
||||
estimatedRuntimeSeconds := volumeSizeGB * 10 * 60
|
||||
|
||||
return &plugin_pb.JobProposal{
|
||||
ProposalId: proposalID,
|
||||
DedupeKey: dedupeKey,
|
||||
@@ -657,6 +664,9 @@ func buildErasureCodingProposal(
|
||||
"target_count": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: int64(len(params.Targets))},
|
||||
},
|
||||
"estimated_runtime_seconds": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: estimatedRuntimeSeconds},
|
||||
},
|
||||
},
|
||||
Labels: map[string]string{
|
||||
"task_type": "erasure_coding",
|
||||
|
||||
@@ -332,6 +332,7 @@ func (h *Handler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 1,
|
||||
RetryBackoffSeconds: 60,
|
||||
JobTypeMaxRuntimeSeconds: 3600, // 1 hour max
|
||||
ExecutionTimeoutSeconds: 3600,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{
|
||||
"target_file_size_mb": {Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: defaultTargetFileSizeMB}},
|
||||
|
||||
@@ -200,6 +200,7 @@ func (h *VacuumHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 1,
|
||||
RetryBackoffSeconds: 10,
|
||||
JobTypeMaxRuntimeSeconds: 1800,
|
||||
ExecutionTimeoutSeconds: 1800,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{
|
||||
"garbage_threshold": {
|
||||
|
||||
@@ -246,6 +246,7 @@ func (h *VolumeBalanceHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
RetryLimit: 1,
|
||||
RetryBackoffSeconds: 15,
|
||||
JobTypeMaxRuntimeSeconds: 1800,
|
||||
ExecutionTimeoutSeconds: 1800,
|
||||
},
|
||||
WorkerDefaultValues: map[string]*plugin_pb.ConfigValue{
|
||||
"imbalance_threshold": {
|
||||
@@ -1160,6 +1161,12 @@ func buildVolumeBalanceProposal(
|
||||
summary = fmt.Sprintf("Move volume %d from %s to %s", result.VolumeID, sourceNode, targetNode)
|
||||
}
|
||||
|
||||
// Estimate runtime at 5 min/GB (matches vacuum) so the scheduler grants
|
||||
// a per-attempt deadline that scales with volume size instead of the
|
||||
// 2*DetectionTimeout default.
|
||||
volumeSizeGB := int64(result.TypedParams.VolumeSize/1024/1024/1024) + 1
|
||||
estimatedRuntimeSeconds := volumeSizeGB * 5 * 60
|
||||
|
||||
return &plugin_pb.JobProposal{
|
||||
ProposalId: proposalID,
|
||||
DedupeKey: dedupeKey,
|
||||
@@ -1183,6 +1190,9 @@ func buildVolumeBalanceProposal(
|
||||
"collection": {
|
||||
Kind: &plugin_pb.ConfigValue_StringValue{StringValue: result.Collection},
|
||||
},
|
||||
"estimated_runtime_seconds": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: estimatedRuntimeSeconds},
|
||||
},
|
||||
},
|
||||
Labels: map[string]string{
|
||||
"task_type": "balance",
|
||||
@@ -1311,6 +1321,36 @@ func buildBatchVolumeBalanceProposals(
|
||||
continue
|
||||
}
|
||||
|
||||
// Estimate runtime so the scheduler grants a per-attempt deadline
|
||||
// large enough for the whole batch. Mirrors vacuum's 5 min/GB budget.
|
||||
// Use max(largest single move, total / concurrency) so a skewed batch
|
||||
// with one big move isn't underestimated by the average.
|
||||
var totalSeconds, maxMoveSeconds int64
|
||||
for _, m := range moves {
|
||||
gb := int64(m.VolumeSize/1024/1024/1024) + 1
|
||||
s := gb * 5 * 60
|
||||
totalSeconds += s
|
||||
if s > maxMoveSeconds {
|
||||
maxMoveSeconds = s
|
||||
}
|
||||
}
|
||||
// Round up to whole scheduling rounds: with N moves and C slots,
|
||||
// the busiest slot processes ceil(N/C) moves, not N/C. Using
|
||||
// avg * ceil(N/C) avoids underestimating when N is not a multiple
|
||||
// of C (e.g. 6 moves at concurrency 5 → 2 rounds, not 1.2).
|
||||
avgSeconds := totalSeconds / int64(len(moves))
|
||||
numRounds := (int64(len(moves)) + int64(maxConcurrentMoves) - 1) / int64(maxConcurrentMoves)
|
||||
estimatedRuntimeSeconds := avgSeconds * numRounds
|
||||
if maxMoveSeconds > estimatedRuntimeSeconds {
|
||||
estimatedRuntimeSeconds = maxMoveSeconds
|
||||
}
|
||||
// Floor for orchestration overhead — scales per wave (one round of
|
||||
// concurrent moves) rather than per move, since setup/teardown
|
||||
// happens once per wave, not once per move.
|
||||
if minBudget := numRounds * 60; estimatedRuntimeSeconds < minBudget {
|
||||
estimatedRuntimeSeconds = minBudget
|
||||
}
|
||||
|
||||
proposalID := fmt.Sprintf("volume-balance-batch-%d-%d", batchStart, time.Now().UnixNano())
|
||||
summary := fmt.Sprintf("Batch balance %d volumes (%s)", len(moves), strings.Join(volumeIDs, ","))
|
||||
if len(summary) > maxProposalStringLength {
|
||||
@@ -1341,6 +1381,9 @@ func buildBatchVolumeBalanceProposals(
|
||||
"batch_size": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: int64(len(moves))},
|
||||
},
|
||||
"estimated_runtime_seconds": {
|
||||
Kind: &plugin_pb.ConfigValue_Int64Value{Int64Value: estimatedRuntimeSeconds},
|
||||
},
|
||||
},
|
||||
Labels: map[string]string{
|
||||
"task_type": "balance",
|
||||
|
||||
Reference in New Issue
Block a user