mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 14:45:51 +00:00
Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
56c51e7f50 | ||
|
|
d1665750e1 | ||
|
|
0566fbd552 | ||
|
|
d4e39b499b |
@@ -54,6 +54,70 @@ func (at *ActiveTopology) GetEffectiveAvailableCapacityDetailed(nodeID string, d
|
||||
return at.getEffectiveAvailableCapacityUnsafe(disk)
|
||||
}
|
||||
|
||||
// GetEffectiveAvailableEcShardSlots returns a disk's free EC shard slots,
|
||||
// accounting for in-flight task reservations at shard granularity. Unlike the
|
||||
// volume-slot views (GetDisksWithEffectiveCapacity / GetEffectiveAvailableCapacity),
|
||||
// this does not truncate sub-volume shard reservations: it subtracts the full
|
||||
// reservation impact (volume slots converted to shard slots, plus the raw shard
|
||||
// slots) so a reservation that is not a whole multiple of ShardsPerVolumeSlot is
|
||||
// not lost. It does NOT subtract the EC shards already persisted on the disk;
|
||||
// callers that track those (from EcShardInfos) subtract them separately.
|
||||
//
|
||||
// shardsPerVolume is the number of EC shards of the target collection that fit in
|
||||
// one volume slot (i.e. its data-shard count): a 4+2 volume's shards are ~1/4 of a
|
||||
// volume each, so one volume slot holds 4 of them, not the default
|
||||
// ShardsPerVolumeSlot. Pass <= 0 to use the default. Using the target ratio keeps
|
||||
// Place from over-filling a disk for low-data-shard layouts.
|
||||
func (at *ActiveTopology) GetEffectiveAvailableEcShardSlots(nodeID string, diskID uint32, shardsPerVolume int) int {
|
||||
if shardsPerVolume <= 0 {
|
||||
shardsPerVolume = ShardsPerVolumeSlot
|
||||
}
|
||||
|
||||
at.mutex.RLock()
|
||||
defer at.mutex.RUnlock()
|
||||
|
||||
diskKey := fmt.Sprintf("%s:%d", nodeID, diskID)
|
||||
disk, exists := at.disks[diskKey]
|
||||
if !exists || disk.DiskInfo == nil || disk.DiskInfo.DiskInfo == nil {
|
||||
return 0
|
||||
}
|
||||
|
||||
info := disk.DiskInfo.DiskInfo
|
||||
base := info.MaxVolumeCount - info.VolumeCount
|
||||
if base <= 0 && info.MaxVolumeCount == 0 && info.VolumeCount == 0 &&
|
||||
len(info.VolumeInfos) == 0 && len(info.EcShardInfos) == 0 {
|
||||
// Freshly started empty servers can report max=0 before publishing concrete
|
||||
// limits; keep one provisional slot so EC placement still sees the disk,
|
||||
// mirroring getEffectiveAvailableCapacityUnsafe.
|
||||
base = 1
|
||||
}
|
||||
if base < 0 {
|
||||
base = 0
|
||||
}
|
||||
// calculateTaskStorageImpact reports consumption as positive, so subtract it.
|
||||
// Volume-slot reservations scale by the target ratio; the sub-volume shard-slot
|
||||
// remainder is in default units and subtracted as-is (a small approximation).
|
||||
impact := at.getEffectiveCapacityUnsafe(disk)
|
||||
// impact.ShardSlots is recorded in default ShardsPerVolumeSlot units; convert it
|
||||
// to the target ratio's shard slots before subtracting (identity when
|
||||
// shardsPerVolume == ShardsPerVolumeSlot). Round a positive reservation up so a
|
||||
// sub-slot reservation (e.g. 1 default slot against a 4-shard target) is not
|
||||
// truncated to zero and wrongly counted as free.
|
||||
scaledShardImpact := int64(impact.ShardSlots) * int64(shardsPerVolume)
|
||||
if scaledShardImpact > 0 {
|
||||
scaledShardImpact = (scaledShardImpact + int64(ShardsPerVolumeSlot) - 1) / int64(ShardsPerVolumeSlot)
|
||||
} else {
|
||||
scaledShardImpact /= int64(ShardsPerVolumeSlot)
|
||||
}
|
||||
free := base*int64(shardsPerVolume) -
|
||||
int64(impact.VolumeSlots)*int64(shardsPerVolume) -
|
||||
scaledShardImpact
|
||||
if free < 0 {
|
||||
free = 0
|
||||
}
|
||||
return int(free)
|
||||
}
|
||||
|
||||
// GetEffectiveCapacityImpact returns the StorageSlotChange impact for a disk
|
||||
// This shows the net impact from all pending and assigned tasks
|
||||
func (at *ActiveTopology) GetEffectiveCapacityImpact(nodeID string, diskID uint32) StorageSlotChange {
|
||||
|
||||
@@ -374,6 +374,7 @@ message ErasureCodingTaskConfig {
|
||||
int32 min_volume_size_mb = 3; // Minimum volume size for EC
|
||||
string collection_filter = 4; // Only process volumes from specific collections
|
||||
repeated string preferred_tags = 5; // Disk tags to prioritize for EC shard placement
|
||||
string replica_placement = 6; // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
|
||||
@@ -2960,6 +2960,7 @@ type ErasureCodingTaskConfig struct {
|
||||
MinVolumeSizeMb int32 `protobuf:"varint,3,opt,name=min_volume_size_mb,json=minVolumeSizeMb,proto3" json:"min_volume_size_mb,omitempty"` // Minimum volume size for EC
|
||||
CollectionFilter string `protobuf:"bytes,4,opt,name=collection_filter,json=collectionFilter,proto3" json:"collection_filter,omitempty"` // Only process volumes from specific collections
|
||||
PreferredTags []string `protobuf:"bytes,5,rep,name=preferred_tags,json=preferredTags,proto3" json:"preferred_tags,omitempty"` // Disk tags to prioritize for EC shard placement
|
||||
ReplicaPlacement string `protobuf:"bytes,6,opt,name=replica_placement,json=replicaPlacement,proto3" json:"replica_placement,omitempty"` // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
@@ -3029,6 +3030,13 @@ func (x *ErasureCodingTaskConfig) GetPreferredTags() []string {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (x *ErasureCodingTaskConfig) GetReplicaPlacement() string {
|
||||
if x != nil {
|
||||
return x.ReplicaPlacement
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
type BalanceTaskConfig struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
@@ -4218,13 +4226,14 @@ const file_worker_proto_rawDesc = "" +
|
||||
"\x10VacuumTaskConfig\x12+\n" +
|
||||
"\x11garbage_threshold\x18\x01 \x01(\x01R\x10garbageThreshold\x12/\n" +
|
||||
"\x14min_volume_age_hours\x18\x02 \x01(\x05R\x11minVolumeAgeHours\x120\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\xed\x01\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\x9a\x02\n" +
|
||||
"\x17ErasureCodingTaskConfig\x12%\n" +
|
||||
"\x0efullness_ratio\x18\x01 \x01(\x01R\rfullnessRatio\x12*\n" +
|
||||
"\x11quiet_for_seconds\x18\x02 \x01(\x05R\x0fquietForSeconds\x12+\n" +
|
||||
"\x12min_volume_size_mb\x18\x03 \x01(\x05R\x0fminVolumeSizeMb\x12+\n" +
|
||||
"\x11collection_filter\x18\x04 \x01(\tR\x10collectionFilter\x12%\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\"n\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\x12+\n" +
|
||||
"\x11replica_placement\x18\x06 \x01(\tR\x10replicaPlacement\"n\n" +
|
||||
"\x11BalanceTaskConfig\x12/\n" +
|
||||
"\x13imbalance_threshold\x18\x01 \x01(\x01R\x12imbalanceThreshold\x12(\n" +
|
||||
"\x10min_server_count\x18\x02 \x01(\x05R\x0eminServerCount\"I\n" +
|
||||
|
||||
@@ -9,8 +9,6 @@ import (
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/operation"
|
||||
@@ -186,6 +184,13 @@ func (fs *FilerServer) CreateEntry(ctx context.Context, req *filer_pb.CreateEntr
|
||||
newEntry.TtlSec = 0
|
||||
}
|
||||
|
||||
// Serialize concurrent mutations to the same path on this filer so the
|
||||
// read (existence/condition) and the write are atomic. Callers route a
|
||||
// key's writes to this owner filer, making this local lock sufficient.
|
||||
fullpath := util.NewFullPath(req.Directory, req.Entry.Name)
|
||||
pathLock := fs.entryLockTable.AcquireLock("CreateEntry", fullpath, util.ExclusiveLock)
|
||||
defer fs.entryLockTable.ReleaseLock(fullpath, pathLock)
|
||||
|
||||
ctx, eventSink := filer.WithMetadataEventSink(ctx)
|
||||
createErr := fs.filer.CreateEntry(ctx, newEntry, req.OExcl, req.IsFromOtherCluster, req.Signatures, req.SkipCheckParentDirectory, so.MaxFileNameLength)
|
||||
|
||||
@@ -328,9 +333,11 @@ func (fs *FilerServer) AppendToEntry(ctx context.Context, req *filer_pb.AppendTo
|
||||
glog.V(4).InfofCtx(ctx, "AppendToEntry %v", req)
|
||||
fullpath := util.NewFullPath(req.Directory, req.EntryName)
|
||||
|
||||
lockClient := cluster.NewLockClient(fs.grpcDialOption, fs.option.Host)
|
||||
lock := lockClient.NewShortLivedLock(string(fullpath), string(fs.option.Host))
|
||||
defer lock.StopShortLivedLock()
|
||||
// Serialize the read-modify-write against concurrent mutations to the same
|
||||
// path on this filer. The append must route to this entry's owner filer for
|
||||
// this local lock to be authoritative.
|
||||
pathLock := fs.entryLockTable.AcquireLock("AppendToEntry", fullpath, util.ExclusiveLock)
|
||||
defer fs.entryLockTable.ReleaseLock(fullpath, pathLock)
|
||||
|
||||
var offset int64 = 0
|
||||
entry, err := fs.filer.FindEntry(ctx, fullpath)
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// Concurrent OExcl creates for the same path must yield exactly one winner. The
|
||||
// filer's CreateEntry is a FindEntry-then-Insert; without the per-path lock both
|
||||
// racers observe "not found" and both insert. The exclusive entry lock makes the
|
||||
// check-then-act atomic so the losers see ErrEntryAlreadyExists.
|
||||
func TestCreateEntryOExclSerialized(t *testing.T) {
|
||||
store := newRenameTestStore()
|
||||
store.findDelay = 5 * time.Millisecond
|
||||
f := newRenameTestFiler(store)
|
||||
f.DirBucketsPath = "/buckets"
|
||||
|
||||
fs := &FilerServer{
|
||||
filer: f,
|
||||
option: &FilerOption{},
|
||||
entryLockTable: util.NewLockTable[util.FullPath](),
|
||||
}
|
||||
|
||||
const racers = 8
|
||||
var success int32
|
||||
var wg sync.WaitGroup
|
||||
start := make(chan struct{})
|
||||
for i := 0; i < racers; i++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
<-start
|
||||
resp, err := fs.CreateEntry(context.Background(), &filer_pb.CreateEntryRequest{
|
||||
Directory: "/test",
|
||||
OExcl: true,
|
||||
SkipCheckParentDirectory: true,
|
||||
Entry: &filer_pb.Entry{
|
||||
Name: "obj",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000, FileMode: 0644, Inode: 1},
|
||||
},
|
||||
})
|
||||
if err == nil && resp.Error == "" {
|
||||
atomic.AddInt32(&success, 1)
|
||||
}
|
||||
}()
|
||||
}
|
||||
close(start)
|
||||
wg.Wait()
|
||||
|
||||
if success != 1 {
|
||||
t.Fatalf("expected exactly 1 OExcl winner, got %d", success)
|
||||
}
|
||||
}
|
||||
@@ -29,6 +29,7 @@ type renameTestStore struct {
|
||||
findCalls map[string]int
|
||||
commitErr error
|
||||
deleteErr error
|
||||
findDelay time.Duration // optional: widen check-then-act windows in tests
|
||||
}
|
||||
|
||||
func newRenameTestStore() *renameTestStore {
|
||||
@@ -69,6 +70,9 @@ func (s *renameTestStore) UpdateEntry(_ context.Context, entry *filer.Entry) err
|
||||
}
|
||||
|
||||
func (s *renameTestStore) FindEntry(_ context.Context, p util.FullPath) (*filer.Entry, error) {
|
||||
if s.findDelay > 0 {
|
||||
time.Sleep(s.findDelay)
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.findCalls[string(p)]++
|
||||
|
||||
@@ -124,6 +124,13 @@ type FilerServer struct {
|
||||
// mountPeerRegistry backs the MountRegister / MountList RPCs for peer
|
||||
// chunk sharing (tier 1). Always populated.
|
||||
mountPeerRegistry *filer.MountPeerRegistry
|
||||
|
||||
// entryLockTable serializes mutations to the same entry path on this filer.
|
||||
// It is the local serialization point for read-modify-write operations
|
||||
// (conditional create, append) once writers for a key are routed to this
|
||||
// node, replacing the distributed lock for that purpose. Idle keys are
|
||||
// evicted automatically, so the table stays bounded.
|
||||
entryLockTable *util.LockTable[util.FullPath]
|
||||
}
|
||||
|
||||
func NewFilerServer(defaultMux, readonlyMux *http.ServeMux, option *FilerOption) (fs *FilerServer, err error) {
|
||||
@@ -162,6 +169,7 @@ func NewFilerServer(defaultMux, readonlyMux *http.ServeMux, option *FilerOption)
|
||||
inFlightDataLimitCond: sync.NewCond(new(sync.Mutex)),
|
||||
recentCopyRequests: make(map[string]recentCopyRequest),
|
||||
CredentialManager: option.CredentialManager,
|
||||
entryLockTable: util.NewLockTable[util.FullPath](),
|
||||
}
|
||||
fs.mountPeerRegistry = filer.NewMountPeerRegistry()
|
||||
go fs.runMountPeerRegistrySweeper()
|
||||
|
||||
@@ -781,6 +781,10 @@ func (ecb *ecBalancer) balance(collections []string) error {
|
||||
ImbalanceThreshold: 0, // the shell balances to an even distribution
|
||||
ReplicaPlacement: ecb.replicaPlacement,
|
||||
Ratio: shellECRatio,
|
||||
// Balance the global phase by fractional fullness so heterogeneous-capacity
|
||||
// nodes fill proportionally (matching the worker). This is identical to raw
|
||||
// shard count when capacities are uniform.
|
||||
GlobalUtilizationBased: true,
|
||||
})
|
||||
return ecb.executeMoves(moves)
|
||||
}
|
||||
|
||||
@@ -340,7 +340,11 @@ func TestCommandEcBalanceIssue8793Topology(t *testing.T) {
|
||||
// Simulate 22 EC volumes across 14 nodes (each volume has 14 shards, 1 per node).
|
||||
// Give nodes 0-3 an extra volume each (vol 23-26, all 14 shards) to create imbalance.
|
||||
// Before balancing: nodes 0-3 have 22+14=36 shards each, nodes 4-13 have 22 shards each.
|
||||
// Total = 4*36 + 10*22 = 144+220 = 364. Average = ceil(364/14) = 26.
|
||||
// Total = 4*36 + 10*22 = 144+220 = 364. Capacities are heterogeneous (max 80 vs
|
||||
// 33), so the utilization-based global phase balances by fullness, not count:
|
||||
// 364 shards over 9*80+5*33=885 slots is ~41% full, so large nodes settle near
|
||||
// 33 shards and small nodes near 14 (all ~41% full) rather than ~26 each, which
|
||||
// would drive the small nodes to ~79%.
|
||||
|
||||
type nodeSpec struct {
|
||||
id string
|
||||
@@ -396,8 +400,17 @@ func TestCommandEcBalanceIssue8793Topology(t *testing.T) {
|
||||
|
||||
ecb.balance([]string{"cldata"})
|
||||
|
||||
// Verify: no node should exceed the average
|
||||
// Verify even FULLNESS (shards/capacity), not even count: with heterogeneous
|
||||
// capacities the utilization-based global phase fills nodes proportionally, so
|
||||
// large nodes hold more shards than small ones while every node ends near the
|
||||
// overall fullness. (An even-count result would over-fill the small nodes.)
|
||||
capacityByID := make(map[string]int, len(nodes))
|
||||
for _, ns := range nodes {
|
||||
capacityByID[ns.id] = ns.maxSlot
|
||||
}
|
||||
|
||||
totalShards := 0
|
||||
totalCapacity := 0
|
||||
shardCounts := make(map[string]int)
|
||||
for _, node := range ecb.ecNodes {
|
||||
count := 0
|
||||
@@ -408,14 +421,22 @@ func TestCommandEcBalanceIssue8793Topology(t *testing.T) {
|
||||
}
|
||||
shardCounts[node.info.Id] = count
|
||||
totalShards += count
|
||||
totalCapacity += capacityByID[node.info.Id]
|
||||
}
|
||||
avg := ceilDivide(totalShards, len(ecNodes))
|
||||
overallFullness := float64(totalShards) / float64(totalCapacity)
|
||||
|
||||
// Tolerance well below the gap a count-even result would show (small nodes
|
||||
// would sit ~38 points above overall), but above integer-rounding skew.
|
||||
const tolerance = 0.05
|
||||
for _, node := range ecb.ecNodes {
|
||||
count := shardCounts[node.info.Id]
|
||||
t.Logf("AFTER node %s: %d shards (avg %d)", node.info.Id, count, avg)
|
||||
if count > avg {
|
||||
t.Errorf("node %s has %d shards, expected at most %d (avg)", node.info.Id, count, avg)
|
||||
capacity := capacityByID[node.info.Id]
|
||||
fullness := float64(count) / float64(capacity)
|
||||
t.Logf("AFTER node %s: %d/%d shards (%.0f%% full, overall %.0f%%)",
|
||||
node.info.Id, count, capacity, fullness*100, overallFullness*100)
|
||||
if diff := fullness - overallFullness; diff > tolerance || diff < -tolerance {
|
||||
t.Errorf("node %s fullness %.1f%% deviates from overall %.1f%% by more than %.0f points",
|
||||
node.info.Id, fullness*100, overallFullness*100, tolerance*100)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package erasure_coding
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"iter"
|
||||
"math/bits"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -40,6 +41,20 @@ func (sb ShardBits) Count() int {
|
||||
return bits.OnesCount32(uint32(sb))
|
||||
}
|
||||
|
||||
// All iterates the shard ids present in the bitmap, in ascending order. It walks
|
||||
// only the set bits (trailing-zero scan), so cost scales with the number of
|
||||
// shards present rather than the full id range. Prefer this over scanning
|
||||
// 0..MaxShardCount and calling Has on each id.
|
||||
func (sb ShardBits) All() iter.Seq[ShardId] {
|
||||
return func(yield func(ShardId) bool) {
|
||||
for b := uint32(sb); b != 0; b &= b - 1 {
|
||||
if !yield(ShardId(bits.TrailingZeros32(b))) {
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ShardsInfo encapsulates information for EC shards with memory-efficient storage
|
||||
type ShardsInfo struct {
|
||||
mu sync.RWMutex
|
||||
|
||||
@@ -43,6 +43,7 @@ type Node struct {
|
||||
type disk struct {
|
||||
diskID uint32
|
||||
diskType string
|
||||
tags []string // placement tags, for preferred-tag tiering
|
||||
freeSlots int
|
||||
shardCount int // total EC shards on this disk across all volumes
|
||||
}
|
||||
@@ -88,8 +89,8 @@ type Options struct {
|
||||
GlobalMaxMovesPerRack int
|
||||
// GlobalUtilizationBased selects the global phase's balance metric: when true,
|
||||
// nodes are balanced by fractional fullness (shards/capacity), which suits
|
||||
// heterogeneous-capacity racks; when false, by raw shard count. The worker
|
||||
// uses utilization; the shell uses raw count.
|
||||
// heterogeneous-capacity racks; when false, by raw shard count. Both the worker
|
||||
// and the shell enable it; the two metrics agree when capacities are uniform.
|
||||
GlobalUtilizationBased bool
|
||||
}
|
||||
|
||||
@@ -132,6 +133,14 @@ func (n *Node) AddDisk(diskID uint32, diskType string, freeSlots, shardCount int
|
||||
n.disks[diskID] = &disk{diskID: diskID, diskType: diskType, freeSlots: freeSlots, shardCount: shardCount}
|
||||
}
|
||||
|
||||
// AddDiskTags records placement tags (e.g. "ssd","fast") for a disk, used by
|
||||
// preferred-tag tiering in Place. Call after AddDisk; a no-op if the disk is unknown.
|
||||
func (n *Node) AddDiskTags(diskID uint32, tags []string) {
|
||||
if d, ok := n.disks[diskID]; ok {
|
||||
d.tags = append([]string(nil), tags...)
|
||||
}
|
||||
}
|
||||
|
||||
// AddShards records that the volume's shards in bits live on diskID. Call it
|
||||
// only for the volumes that should be balanced; the disk's overall occupancy is
|
||||
// reported separately via AddDisk.
|
||||
@@ -251,10 +260,8 @@ func detectDuplicateShards(vk volKey, nodes map[string]*Node) []*move {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for shardID := 0; shardID < erasure_coding.MaxShardCount; shardID++ {
|
||||
if info.shardBits.Has(erasure_coding.ShardId(shardID)) {
|
||||
shardLocations[shardID] = append(shardLocations[shardID], node)
|
||||
}
|
||||
for sid := range info.shardBits.All() {
|
||||
shardLocations[int(sid)] = append(shardLocations[int(sid)], node)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -354,8 +361,11 @@ func balanceShardTypeAcrossRacks(vk volKey, nodes map[string]*Node, racks map[st
|
||||
destRack, ok := pickTarget(rackKeys, shardsPerRack, maxPerRack, antiAffinity,
|
||||
func(r string) bool { return racks[r].freeSlots > 0 },
|
||||
func(r string) bool {
|
||||
if rp != nil && rp.DiffRackCount > 0 {
|
||||
return rackShardCount[r] < rp.DiffRackCount
|
||||
if rp == nil {
|
||||
return true
|
||||
}
|
||||
if rp.DiffRackCount > 0 && rackShardCount[r] >= rp.DiffRackCount {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
})
|
||||
@@ -403,7 +413,7 @@ func pickNodeInRack(r *rack, vk volKey, rp *super_block.ReplicaPlacement) *Node
|
||||
continue
|
||||
}
|
||||
count := volumeShardCount(node, vk)
|
||||
if rp != nil && rp.SameRackCount > 0 && count >= rp.SameRackCount+1 {
|
||||
if rp != nil && rp.SameRackCount > 0 && count >= rp.SameRackCount {
|
||||
continue
|
||||
}
|
||||
if best == nil || count < bestCount {
|
||||
@@ -479,7 +489,7 @@ func balanceShardTypeAcrossNodes(vk volKey, r *rack, diskType string, dataShards
|
||||
func(n string) bool { return n != pm.src.id && r.nodes[n].freeSlots > 0 },
|
||||
func(n string) bool {
|
||||
if rp != nil && rp.SameRackCount > 0 {
|
||||
return nodeShardCount[n] < rp.SameRackCount+1
|
||||
return nodeShardCount[n] < rp.SameRackCount
|
||||
}
|
||||
return true
|
||||
})
|
||||
@@ -610,13 +620,10 @@ func detectGlobalImbalance(nodes map[string]*Node, racks map[string]*rack, diskT
|
||||
if pass == 1 && !volumeOnMin {
|
||||
continue // pass 1: only volumes already on the destination
|
||||
}
|
||||
// Iterate the full shard-id space so custom ratios with more than
|
||||
// the standard total (ids 14..MaxShardCount-1) are candidates too.
|
||||
for shardID := 0; shardID < erasure_coding.MaxShardCount; shardID++ {
|
||||
sid := erasure_coding.ShardId(shardID)
|
||||
if !info.shardBits.Has(sid) {
|
||||
continue
|
||||
}
|
||||
// Walk the volume's actual shard bitmap so custom ratios with more
|
||||
// than the standard total (ids 14..MaxShardCount-1) are candidates too.
|
||||
for sid := range info.shardBits.All() {
|
||||
shardID := int(sid)
|
||||
if minInfo != nil && minInfo.shardBits.Has(sid) {
|
||||
continue
|
||||
}
|
||||
@@ -669,10 +676,8 @@ func shardsByGroup(vk volKey, nodes map[string]*Node, dataShards int, key func(*
|
||||
continue
|
||||
}
|
||||
k := key(node)
|
||||
for s := 0; s < erasure_coding.MaxShardCount; s++ {
|
||||
if !info.shardBits.Has(erasure_coding.ShardId(s)) {
|
||||
continue
|
||||
}
|
||||
for sid := range info.shardBits.All() {
|
||||
s := int(sid)
|
||||
if s < dataShards {
|
||||
dataPer[k] = append(dataPer[k], s)
|
||||
} else {
|
||||
@@ -747,11 +752,8 @@ func pickBestDiskOnNode(node *Node, vk volKey, diskType string, shardID, dataSha
|
||||
bits := info.diskShardBits[diskID]
|
||||
existingShards = bits.Count()
|
||||
if dataShardCount > 0 {
|
||||
for s := 0; s < erasure_coding.MaxShardCount; s++ {
|
||||
if !bits.Has(erasure_coding.ShardId(s)) {
|
||||
continue
|
||||
}
|
||||
if s < dataShardCount {
|
||||
for sid := range bits.All() {
|
||||
if int(sid) < dataShardCount {
|
||||
hasData = true
|
||||
} else {
|
||||
hasParity = true
|
||||
@@ -807,9 +809,11 @@ func reserveShard(node *Node, vk volKey, shardID int, diskID uint32) {
|
||||
info.diskShardBits[diskID] = info.diskShardBits[diskID].Set(sid)
|
||||
if d, ok := node.disks[diskID]; ok {
|
||||
d.shardCount++
|
||||
if d.freeSlots > 0 {
|
||||
d.freeSlots--
|
||||
}
|
||||
// Decrement unconditionally so reserve/release stay symmetric (releaseShard
|
||||
// credits a slot unconditionally). Callers only reserve onto disks
|
||||
// pickBestDisk* already vetted as having free slots, so this won't go
|
||||
// negative; if it ever did, freeSlots<=0 correctly reads as full.
|
||||
d.freeSlots--
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,558 @@
|
||||
package ecbalancer
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
storagetypes "github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
)
|
||||
|
||||
// Constraints configures a Place call. Ratio resolves a collection's
|
||||
// (dataShards, parityShards); nil uses the standard scheme. ReplicaPlacement,
|
||||
// when non-nil, caps shards per rack (DiffRackCount = max shards/rack) and per
|
||||
// node within a rack (SameRackCount = max shards/node); both digits are direct
|
||||
// hard caps. The data-center digit (DiffDataCenterCount) is not honored:
|
||||
// the 1-byte volume ReplicaPlacement can only encode 0-2 there, too small to be a
|
||||
// meaningful per-DC EC shard cap, so EC relies on the rack/node even spread instead.
|
||||
//
|
||||
// DiskTypePolicy controls how DiskType constrains placement (Any / Prefer /
|
||||
// Require). PreferredTags drives whole-plan tag tiering: Place tries disks
|
||||
// carrying the earliest tags first and widens to all disks only if a tier cannot
|
||||
// place every shard.
|
||||
type Constraints struct {
|
||||
DiskType string
|
||||
DiskTypePolicy DiskTypePolicy
|
||||
PreferredTags []string
|
||||
ReplicaPlacement *super_block.ReplicaPlacement
|
||||
Ratio func(collection string) (dataShards, parityShards int)
|
||||
}
|
||||
|
||||
// DiskTypePolicy controls how Constraints.DiskType constrains placement.
|
||||
type DiskTypePolicy int
|
||||
|
||||
const (
|
||||
DiskTypeAny DiskTypePolicy = iota // any disk type
|
||||
DiskTypePrefer // prefer DiskType, spill to other types if needed
|
||||
DiskTypeRequire // only DiskType (HardDriveType when "")
|
||||
)
|
||||
|
||||
// diskTypeEqual compares disk types after normalization, so "" and "hdd" (both
|
||||
// HardDriveType) are equal.
|
||||
func diskTypeEqual(a, b string) bool {
|
||||
return storagetypes.ToDiskType(a).String() == storagetypes.ToDiskType(b).String()
|
||||
}
|
||||
|
||||
// diskHasAnyTag reports whether the disk carries any of the given tags.
|
||||
func diskHasAnyTag(d *disk, tags []string) bool {
|
||||
for _, want := range tags {
|
||||
for _, have := range d.tags {
|
||||
if have == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Destination is a chosen target for one shard. DataCenter and Rack are kept as
|
||||
// separate values (matching topology.DiskInfo) rather than a "dc:rack" composite,
|
||||
// so callers read them directly instead of parsing.
|
||||
type Destination struct {
|
||||
Node string
|
||||
DiskID uint32
|
||||
DataCenter string
|
||||
Rack string // bare rack id within DataCenter
|
||||
}
|
||||
|
||||
// PlaceResult holds the chosen destinations, which constraints had to be relaxed
|
||||
// (durability-first only), and whether placement spilled outside the preferred
|
||||
// disk type or tag tiers (for parity with today's logging).
|
||||
type PlaceResult struct {
|
||||
Destinations map[int]Destination
|
||||
Relaxed []string
|
||||
SpilledToOtherDiskType bool
|
||||
SpilledOutsidePreferredTags bool
|
||||
}
|
||||
|
||||
// PlacementMode selects the strictness/relaxation policy.
|
||||
type PlacementMode int
|
||||
|
||||
const (
|
||||
// PlaceStrict: caps and ReplicaPlacement are hard. Place fails rather than
|
||||
// violate them, so the caller can defer (leave the volume as-is and retry).
|
||||
PlaceStrict PlacementMode = iota
|
||||
// PlaceDurabilityFirst (used by both encode and repair): relax per-type caps ->
|
||||
// data/parity anti-affinity -> ReplicaPlacement, in that order, until each shard
|
||||
// lands, reporting what was relaxed in PlaceResult.Relaxed. The per-disk
|
||||
// durability cap (<= parityShards per disk) is never relaxed. Fails only if no
|
||||
// disk has free capacity. Encode places best-effort this way and rebalancing
|
||||
// tightens the spread afterward.
|
||||
PlaceDurabilityFirst
|
||||
)
|
||||
|
||||
// relaxation controls which placement-quality constraints are enforced on an
|
||||
// attempt. preferring fresh nodes (repair's "avoid surviving-shard nodes") is not
|
||||
// listed: pickNodeInRack already selects the node with the fewest shards of the
|
||||
// volume, so survivors are deprioritized with built-in fallback.
|
||||
type relaxation struct {
|
||||
caps bool
|
||||
antiAffinity bool
|
||||
rp bool
|
||||
}
|
||||
|
||||
func (r relaxation) relaxedNames() []string {
|
||||
var n []string
|
||||
if !r.caps {
|
||||
n = append(n, "caps")
|
||||
}
|
||||
if !r.antiAffinity {
|
||||
n = append(n, "anti-affinity")
|
||||
}
|
||||
if !r.rp {
|
||||
n = append(n, "replica-placement")
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
var strictAttempts = []relaxation{{caps: true, antiAffinity: true, rp: true}}
|
||||
|
||||
var durabilityAttempts = []relaxation{
|
||||
{caps: true, antiAffinity: true, rp: true},
|
||||
{caps: false, antiAffinity: true, rp: true},
|
||||
{caps: false, antiAffinity: false, rp: true},
|
||||
{caps: false, antiAffinity: false, rp: false},
|
||||
}
|
||||
|
||||
type placedEntry struct {
|
||||
node *Node
|
||||
sid int
|
||||
rackKey string
|
||||
}
|
||||
|
||||
// Place assigns destinations for the `need` shard ids of volume (collection,vid),
|
||||
// reading the volume's already-placed shards from the snapshot (so encode passes
|
||||
// an empty-for-this-volume snapshot, repair passes one seeded with the surviving
|
||||
// shards).
|
||||
//
|
||||
// Tag tiering (whole-plan retry): it tries the preferred-tag tiers in order, each
|
||||
// a complete candidate set, and returns the first tier that places every shard;
|
||||
// only when it falls through to the no-tag tier does it set
|
||||
// SpilledOutsidePreferredTags. Within a tier, disk-type Prefer spills to other
|
||||
// types per shard (SpilledToOtherDiskType); Require filters strictly.
|
||||
func (t *Topology) Place(vid uint32, collection string, need []int, c Constraints, mode PlacementMode) (*PlaceResult, error) {
|
||||
if len(need) == 0 {
|
||||
return &PlaceResult{Destinations: map[int]Destination{}}, nil
|
||||
}
|
||||
|
||||
vk := volKey{collection: collection, vid: vid}
|
||||
dataShards, parityShards := erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount
|
||||
if c.Ratio != nil {
|
||||
if d, p := c.Ratio(collection); d > 0 && p > 0 {
|
||||
dataShards, parityShards = d, p
|
||||
}
|
||||
}
|
||||
|
||||
racks := buildRacks(t.nodes)
|
||||
if len(racks) == 0 {
|
||||
return nil, fmt.Errorf("no racks available for EC placement")
|
||||
}
|
||||
rackKeys := sortedKeys(racks)
|
||||
|
||||
// Disk-type eligibility (Require filters; Any/Prefer admit all) and the soft
|
||||
// type preference applied in scoring under Prefer.
|
||||
typeEligible := func(d *disk) bool {
|
||||
if c.DiskTypePolicy == DiskTypeRequire {
|
||||
return diskTypeEqual(d.diskType, c.DiskType)
|
||||
}
|
||||
return true
|
||||
}
|
||||
var prefer func(*disk) bool
|
||||
if c.DiskTypePolicy == DiskTypePrefer {
|
||||
prefer = func(d *disk) bool { return diskTypeEqual(d.diskType, c.DiskType) }
|
||||
}
|
||||
|
||||
// Whole-plan retry over preferred-tag tiers; the first tier that places every
|
||||
// shard wins. Reaching the no-tag tier means we spilled outside the tags.
|
||||
tiers := tagTiers(c.PreferredTags)
|
||||
var lastErr error
|
||||
for _, tierTags := range tiers {
|
||||
tt := tierTags
|
||||
eligible := func(d *disk) bool {
|
||||
return typeEligible(d) && (len(tt) == 0 || diskHasAnyTag(d, tt))
|
||||
}
|
||||
res, err := t.tryPlace(vk, need, dataShards, parityShards, racks, rackKeys, mode, c.ReplicaPlacement, eligible, prefer)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
continue
|
||||
}
|
||||
if len(c.PreferredTags) > 0 && len(tierTags) == 0 {
|
||||
res.SpilledOutsidePreferredTags = true
|
||||
}
|
||||
return res, nil
|
||||
}
|
||||
return nil, lastErr
|
||||
}
|
||||
|
||||
// tagTiers returns the eligibility tag-sets in increasing breadth, ending with an
|
||||
// empty set ("any disk"). Empty preferredTags yields a single any-disk tier.
|
||||
func tagTiers(preferredTags []string) [][]string {
|
||||
if len(preferredTags) == 0 {
|
||||
return [][]string{nil}
|
||||
}
|
||||
tiers := make([][]string, 0, len(preferredTags)+1)
|
||||
for k := range preferredTags {
|
||||
tiers = append(tiers, append([]string(nil), preferredTags[:k+1]...))
|
||||
}
|
||||
return append(tiers, nil)
|
||||
}
|
||||
|
||||
// tryPlace runs one whole-plan placement attempt restricted to disks satisfying
|
||||
// `eligible`, with `prefer` (may be nil) ranking soft-preferred disks first. It
|
||||
// journals reservations and rolls them all back if any shard cannot be placed, so
|
||||
// a failed tier leaves the snapshot unchanged for the next attempt.
|
||||
func (t *Topology) tryPlace(vk volKey, need []int, dataShards, parityShards int, racks map[string]*rack, rackKeys []string, mode PlacementMode, rp *super_block.ReplicaPlacement, eligible func(*disk) bool, prefer func(*disk) bool) (*PlaceResult, error) {
|
||||
result := &PlaceResult{Destinations: make(map[int]Destination, len(need))}
|
||||
|
||||
// Per-type shard ids per rack (even caps), total shard count per rack
|
||||
// (DiffRackCount), and the racks bearing each type (anti-affinity) — all seeded
|
||||
// from the volume's existing shards.
|
||||
shardsPerRack := map[bool]map[string][]int{true: {}, false: {}}
|
||||
rackShardCount := map[string]int{}
|
||||
bearing := map[bool]map[string]bool{true: {}, false: {}}
|
||||
for _, n := range t.nodes {
|
||||
info, ok := n.shards[vk]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for sid := range info.shardBits.All() {
|
||||
s := int(sid)
|
||||
isData := s < dataShards
|
||||
shardsPerRack[isData][n.rack] = append(shardsPerRack[isData][n.rack], s)
|
||||
rackShardCount[n.rack]++
|
||||
bearing[isData][n.rack] = true
|
||||
}
|
||||
}
|
||||
|
||||
// Even per-rack caps divide by racks that actually have an eligible free disk,
|
||||
// not all racks (the snapshot keeps every disk type/tag), so a valid tiered
|
||||
// cluster — e.g. SSDs in only 2 of 4 racks — is not capped impossibly low.
|
||||
numEligibleRacks := 0
|
||||
for _, rk := range rackKeys {
|
||||
if rackHasFreeDisk(racks[rk], eligible) {
|
||||
numEligibleRacks++
|
||||
}
|
||||
}
|
||||
if numEligibleRacks < 1 {
|
||||
numEligibleRacks = 1
|
||||
}
|
||||
|
||||
attempts := strictAttempts
|
||||
if mode == PlaceDurabilityFirst {
|
||||
attempts = durabilityAttempts
|
||||
}
|
||||
|
||||
var journal []placedEntry
|
||||
relaxedSeen := map[string]bool{}
|
||||
spilledType := false
|
||||
|
||||
placeShard := func(sid int, isData bool) bool {
|
||||
typeTotal := dataShards
|
||||
if !isData {
|
||||
typeTotal = parityShards
|
||||
}
|
||||
for _, rl := range attempts {
|
||||
node, diskID, spilled, ok := chooseShardDest(vk, sid, isData, dataShards, typeTotal, numEligibleRacks, parityShards, racks, rackKeys, rp, eligible, prefer, shardsPerRack[isData], rackShardCount, bearing, rl)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
reserveShard(node, vk, sid, diskID)
|
||||
node.freeSlots--
|
||||
racks[node.rack].freeSlots--
|
||||
shardsPerRack[isData][node.rack] = append(shardsPerRack[isData][node.rack], sid)
|
||||
rackShardCount[node.rack]++
|
||||
bearing[isData][node.rack] = true
|
||||
journal = append(journal, placedEntry{node: node, sid: sid, rackKey: node.rack})
|
||||
result.Destinations[sid] = Destination{
|
||||
Node: node.id,
|
||||
DiskID: diskID,
|
||||
DataCenter: node.dc,
|
||||
Rack: strings.TrimPrefix(node.rack, node.dc+":"),
|
||||
}
|
||||
if spilled {
|
||||
spilledType = true
|
||||
}
|
||||
for _, name := range rl.relaxedNames() {
|
||||
relaxedSeen[name] = true
|
||||
}
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Data shards first, then parity, so parity can avoid data-bearing racks.
|
||||
for _, isData := range []bool{true, false} {
|
||||
for _, sid := range shardsOfType(need, isData, dataShards) {
|
||||
if placeShard(sid, isData) {
|
||||
continue
|
||||
}
|
||||
for _, e := range journal {
|
||||
releaseShard(e.node, vk, e.sid)
|
||||
e.node.freeSlots++
|
||||
racks[e.rackKey].freeSlots++
|
||||
}
|
||||
return nil, fmt.Errorf("cannot place EC shard %d of volume %d (collection %q)", sid, vk.vid, vk.collection)
|
||||
}
|
||||
}
|
||||
|
||||
result.SpilledToOtherDiskType = spilledType
|
||||
for name := range relaxedSeen {
|
||||
result.Relaxed = append(result.Relaxed, name)
|
||||
}
|
||||
sort.Strings(result.Relaxed)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// chooseShardDest selects a (node, disk) for one shard at the given relaxation
|
||||
// level: pick a rack (even per-type cap + ReplicaPlacement caps + two-pass
|
||||
// anti-affinity to the opposite type), then the least-loaded eligible node, then
|
||||
// the best eligible disk. The third return reports whether the disk spilled off
|
||||
// the soft-preferred type. ok=false when no rack/node/disk fits.
|
||||
func chooseShardDest(vk volKey, sid int, isData bool, dataShards, typeTotal, numEligibleRacks, maxPerDisk int, racks map[string]*rack, rackKeys []string, rp *super_block.ReplicaPlacement, eligible func(*disk) bool, prefer func(*disk) bool, shardsPerRackType map[string][]int, rackShardCount map[string]int, bearing map[bool]map[string]bool, rl relaxation) (*Node, uint32, bool, bool) {
|
||||
maxPerRack := numEligibleRacks*typeTotal + 1 // effectively unlimited when caps are relaxed
|
||||
if rl.caps {
|
||||
if maxPerRack = ceilDivide(typeTotal, numEligibleRacks); maxPerRack < 1 {
|
||||
maxPerRack = 1
|
||||
}
|
||||
}
|
||||
|
||||
var anti map[string]bool
|
||||
if rl.antiAffinity {
|
||||
anti = bearing[!isData] // racks already holding the opposite shard type
|
||||
}
|
||||
|
||||
if !rl.rp {
|
||||
rp = nil
|
||||
}
|
||||
// A rack is eligible only if it is under the per-rack shard cap (DiffRackCount),
|
||||
// enforced only when set (and relaxed with rp).
|
||||
withinLimit := func(r string) bool {
|
||||
if rp == nil {
|
||||
return true
|
||||
}
|
||||
if rp.DiffRackCount > 0 && rackShardCount[r] >= rp.DiffRackCount {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
destRack, ok := pickTarget(rackKeys, shardsPerRackType, maxPerRack, anti,
|
||||
func(r string) bool { return racks[r].freeSlots > 0 && rackHasFreeDisk(racks[r], eligible) },
|
||||
withinLimit)
|
||||
if !ok {
|
||||
return nil, 0, false, false
|
||||
}
|
||||
node := pickNodeInRackEligible(racks[destRack], vk, rp, eligible)
|
||||
if node == nil {
|
||||
return nil, 0, false, false
|
||||
}
|
||||
diskID, ok, spilled := pickBestDiskEligible(node, vk, eligible, prefer, sid, dataShards, maxPerDisk)
|
||||
if !ok {
|
||||
return nil, 0, false, false
|
||||
}
|
||||
return node, diskID, spilled, true
|
||||
}
|
||||
|
||||
// nodeHasFreeDisk reports whether the node has a free disk satisfying eligible.
|
||||
func nodeHasFreeDisk(n *Node, eligible func(*disk) bool) bool {
|
||||
for _, d := range n.disks {
|
||||
if d.freeSlots > 0 && eligible(d) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// rackHasFreeDisk reports whether any node in the rack has a free eligible disk.
|
||||
func rackHasFreeDisk(r *rack, eligible func(*disk) bool) bool {
|
||||
for _, n := range r.nodes {
|
||||
if n.freeSlots > 0 && nodeHasFreeDisk(n, eligible) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// pickNodeInRackEligible is pickNodeInRack restricted to nodes that have a free
|
||||
// eligible disk. FromActiveTopology keeps all disk types/tags in the snapshot, so
|
||||
// without this a node with free volume slots but no eligible disk could be chosen.
|
||||
func pickNodeInRackEligible(r *rack, vk volKey, rp *super_block.ReplicaPlacement, eligible func(*disk) bool) *Node {
|
||||
var best *Node
|
||||
bestCount := -1
|
||||
for _, id := range sortedNodeKeys(r.nodes) {
|
||||
node := r.nodes[id]
|
||||
if node.freeSlots <= 0 {
|
||||
continue
|
||||
}
|
||||
if !nodeHasFreeDisk(node, eligible) {
|
||||
continue
|
||||
}
|
||||
count := volumeShardCount(node, vk)
|
||||
if rp != nil && rp.SameRackCount > 0 && count >= rp.SameRackCount {
|
||||
continue
|
||||
}
|
||||
if best == nil || count < bestCount {
|
||||
best, bestCount = node, count
|
||||
}
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
// pickBestDiskEligible chooses the best eligible disk on a node, ranking
|
||||
// soft-preferred disks (prefer != nil && prefer(d)) ahead of others so disk-type
|
||||
// Prefer uses the preferred type when available but spills otherwise. Returns the
|
||||
// disk id, whether one was found, and whether the chosen disk spilled off the
|
||||
// preferred type.
|
||||
func pickBestDiskEligible(node *Node, vk volKey, eligible func(*disk) bool, prefer func(*disk) bool, shardID, dataShardCount, maxPerDisk int) (uint32, bool, bool) {
|
||||
isDataShard := dataShardCount > 0 && shardID < dataShardCount
|
||||
info := node.shards[vk]
|
||||
var bestDiskID uint32
|
||||
bestScore := -1
|
||||
bestPreferred := false
|
||||
for _, diskID := range sortedDiskKeys(node.disks) {
|
||||
d := node.disks[diskID]
|
||||
if !eligible(d) || d.freeSlots <= 0 {
|
||||
continue
|
||||
}
|
||||
existingShards := 0
|
||||
hasData := false
|
||||
hasParity := false
|
||||
if info != nil {
|
||||
bits := info.diskShardBits[diskID]
|
||||
existingShards = bits.Count()
|
||||
if dataShardCount > 0 {
|
||||
for sid := range bits.All() {
|
||||
if int(sid) < dataShardCount {
|
||||
hasData = true
|
||||
} else {
|
||||
hasParity = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Durability: never put more than maxPerDisk (parityShards) shards of this
|
||||
// volume on one disk, or losing that disk would lose more than EC can
|
||||
// recover. Hard cap, enforced even under durability-first relaxation.
|
||||
if maxPerDisk > 0 && existingShards >= maxPerDisk {
|
||||
continue
|
||||
}
|
||||
score := d.shardCount*10 + existingShards*100
|
||||
if dataShardCount > 0 {
|
||||
if isDataShard && hasParity {
|
||||
score += 1000
|
||||
} else if !isDataShard && hasData {
|
||||
score += 1000
|
||||
}
|
||||
}
|
||||
preferred := prefer == nil || prefer(d)
|
||||
if !preferred {
|
||||
score += 100000 // strongly deprioritize spilling to a non-preferred type
|
||||
}
|
||||
if bestScore == -1 || score < bestScore {
|
||||
bestScore = score
|
||||
bestDiskID = diskID
|
||||
bestPreferred = preferred
|
||||
}
|
||||
}
|
||||
if bestScore == -1 {
|
||||
return 0, false, false
|
||||
}
|
||||
return bestDiskID, true, prefer != nil && !bestPreferred
|
||||
}
|
||||
|
||||
// clearShardAccounting removes one shard copy of a volume from the snapshot's
|
||||
// per-domain accounting (the volume's shard bits) WITHOUT crediting disk capacity.
|
||||
// It clears only the given physical disk's bit, then recomputes the node-level
|
||||
// union from the remaining disk bits, so a kept copy of the same shard on another
|
||||
// disk of the same node still counts toward caps / ReplicaPlacement / anti-affinity.
|
||||
//
|
||||
// Repair uses this to drop the duplicate/mismatched copies it plans to delete
|
||||
// before placing missing shards, so those copies do not inflate placement
|
||||
// accounting. Capacity is deliberately NOT credited: the deletes run only after
|
||||
// the rebuilt shards are distributed, so the slots are not free at plan time. This
|
||||
// is distinct from releaseShard, which credits freeSlots and clears the union.
|
||||
func clearShardAccounting(node *Node, vk volKey, shardID int, diskID uint32) {
|
||||
info, ok := node.shards[vk]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
sid := erasure_coding.ShardId(shardID)
|
||||
if bits, ok := info.diskShardBits[diskID]; ok {
|
||||
info.diskShardBits[diskID] = bits.Clear(sid)
|
||||
}
|
||||
var union erasure_coding.ShardBits
|
||||
for _, b := range info.diskShardBits {
|
||||
union |= b
|
||||
}
|
||||
info.shardBits = union
|
||||
}
|
||||
|
||||
// ClearShardAccounting drops one shard copy of a volume from placement accounting
|
||||
// without crediting capacity (see clearShardAccounting). Repair calls it for each
|
||||
// copy it plans to delete before placing missing shards, so those copies do not
|
||||
// inflate caps/RP/anti-affinity. No-op for an unknown node.
|
||||
func (t *Topology) ClearShardAccounting(nodeID, collection string, vid uint32, shardID int, diskID uint32) {
|
||||
n, ok := t.nodes[nodeID]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
clearShardAccounting(n, volKey{collection: collection, vid: vid}, shardID, diskID)
|
||||
}
|
||||
|
||||
// ReleaseVolumeShards removes every shard of a volume from the snapshot and
|
||||
// credits the freed disk capacity. A greenfield encode calls this so any stale
|
||||
// EC shards left by a prior failed attempt (which the encode task deletes before
|
||||
// distributing the new shards) neither occupy capacity nor skew anti-affinity /
|
||||
// per-disk caps during planning. Unlike repair's ClearShardAccounting, it credits
|
||||
// freeSlots because the deletes run before the new writes.
|
||||
func (t *Topology) ReleaseVolumeShards(collection string, vid uint32) {
|
||||
vk := volKey{collection: collection, vid: vid}
|
||||
for _, n := range t.nodes {
|
||||
info, ok := n.shards[vk]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// freed is the total disk-slots the volume occupies on this node (a shard may
|
||||
// sit on more than one disk). releaseShard credits each disk's freeSlots;
|
||||
// credit the node's freeSlots by the same total, since rack capacity is summed
|
||||
// from node freeSlots (buildRacks) and node freeSlots gates node eligibility.
|
||||
freed := 0
|
||||
for _, bits := range info.diskShardBits {
|
||||
freed += bits.Count()
|
||||
}
|
||||
sids := make([]int, 0, info.shardBits.Count())
|
||||
for sid := range info.shardBits.All() {
|
||||
sids = append(sids, int(sid))
|
||||
}
|
||||
for _, sid := range sids {
|
||||
releaseShard(n, vk, sid)
|
||||
}
|
||||
n.freeSlots += freed
|
||||
delete(n.shards, vk)
|
||||
}
|
||||
}
|
||||
|
||||
// shardsOfType returns the sorted subset of need that are data shards (id <
|
||||
// dataShards) when isData, else the parity subset.
|
||||
func shardsOfType(need []int, isData bool, dataShards int) []int {
|
||||
var out []int
|
||||
for _, s := range need {
|
||||
if (s < dataShards) == isData {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
sort.Ints(out)
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,422 @@
|
||||
package ecbalancer
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
)
|
||||
|
||||
// buildPlaceTopo makes a topology of racks x nodesPerRack, each node one disk with
|
||||
// perDiskFree free EC shard slots.
|
||||
func buildPlaceTopo(racks, nodesPerRack, perDiskFree int) *Topology {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < racks; r++ {
|
||||
rackKey := fmt.Sprintf("dc1:rack%d", r)
|
||||
for n := 0; n < nodesPerRack; n++ {
|
||||
id := fmt.Sprintf("10.0.%d.%d:8080", r, n)
|
||||
node := topo.AddNode(id, "dc1", rackKey, perDiskFree)
|
||||
node.AddDisk(0, "", perDiskFree, 0)
|
||||
}
|
||||
}
|
||||
return topo
|
||||
}
|
||||
|
||||
func allShards() []int {
|
||||
out := make([]int, erasure_coding.TotalShardsCount)
|
||||
for i := range out {
|
||||
out[i] = i
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestPlaceStrictSpreadAndCaps places a fresh 10+4 volume and checks every shard
|
||||
// lands on a distinct node and no rack exceeds the even per-type cap.
|
||||
func TestPlaceStrictSpreadAndCaps(t *testing.T) {
|
||||
const racks = 4
|
||||
topo := buildPlaceTopo(racks, 4, 50)
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d shards, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
|
||||
usedNodes := map[string]bool{}
|
||||
dataPerRack := map[string]int{}
|
||||
parityPerRack := map[string]int{}
|
||||
for sid, d := range res.Destinations {
|
||||
if usedNodes[d.Node] {
|
||||
t.Errorf("node %s reused for shard %d (expected distinct nodes with ample capacity)", d.Node, sid)
|
||||
}
|
||||
usedNodes[d.Node] = true
|
||||
if sid < erasure_coding.DataShardsCount {
|
||||
dataPerRack[d.Rack]++
|
||||
} else {
|
||||
parityPerRack[d.Rack]++
|
||||
}
|
||||
}
|
||||
dataCap := ceilDivide(erasure_coding.DataShardsCount, racks)
|
||||
parityCap := ceilDivide(erasure_coding.ParityShardsCount, racks)
|
||||
for rk, n := range dataPerRack {
|
||||
if n > dataCap {
|
||||
t.Errorf("rack %s holds %d data shards, cap %d", rk, n, dataCap)
|
||||
}
|
||||
}
|
||||
for rk, n := range parityPerRack {
|
||||
if n > parityCap {
|
||||
t.Errorf("rack %s holds %d parity shards, cap %d", rk, n, parityCap)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceStrictFailsAndRollsBack: a single tiny disk cannot hold 14 shards, so
|
||||
// strict Place fails and leaves the snapshot untouched.
|
||||
func TestPlaceStrictFailsAndRollsBack(t *testing.T) {
|
||||
topo := buildPlaceTopo(1, 1, 2) // one node, room for 2 shards
|
||||
node := topo.nodes["10.0.0.0:8080"]
|
||||
freeBefore := node.freeSlots
|
||||
diskFreeBefore := node.disks[0].freeSlots
|
||||
|
||||
_, err := topo.Place(1, "c1", allShards(), Constraints{}, PlaceStrict)
|
||||
if err == nil {
|
||||
t.Fatal("expected Place to fail on insufficient capacity")
|
||||
}
|
||||
if info, ok := node.shards[volKey{collection: "c1", vid: 1}]; ok && info.shardBits.Count() != 0 {
|
||||
t.Errorf("volume shard bits left on node after failed strict Place (rollback incomplete): %b", info.shardBits)
|
||||
}
|
||||
if node.freeSlots != freeBefore {
|
||||
t.Errorf("node freeSlots = %d after rollback, want %d", node.freeSlots, freeBefore)
|
||||
}
|
||||
if node.disks[0].freeSlots != diskFreeBefore {
|
||||
t.Errorf("disk freeSlots = %d after rollback, want %d", node.disks[0].freeSlots, diskFreeBefore)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceDurabilityFirstRelaxesRP: a ReplicaPlacement rack limit too tight for
|
||||
// the shard count makes strict fail, while durability-first relaxes RP to place
|
||||
// everything and reports the relaxation.
|
||||
func TestPlaceDurabilityFirstRelaxesRP(t *testing.T) {
|
||||
rp := &super_block.ReplicaPlacement{DiffRackCount: 3} // <=3 shards per rack
|
||||
topo := buildPlaceTopo(2, 8, 50) // 2 racks: 2*3=6 < 14 under RP
|
||||
|
||||
if _, err := topo.Place(1, "c1", allShards(), Constraints{ReplicaPlacement: rp}, PlaceStrict); err == nil {
|
||||
t.Fatal("strict Place should fail when RP rack limit cannot fit all shards")
|
||||
}
|
||||
|
||||
topo = buildPlaceTopo(2, 8, 50)
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{ReplicaPlacement: rp}, PlaceDurabilityFirst)
|
||||
if err != nil {
|
||||
t.Fatalf("durability-first Place: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d shards, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
relaxedRP := false
|
||||
for _, r := range res.Relaxed {
|
||||
if r == "replica-placement" {
|
||||
relaxedRP = true
|
||||
}
|
||||
}
|
||||
if !relaxedRP {
|
||||
t.Errorf("expected replica-placement relaxation, got %v", res.Relaxed)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceSameRackCountIsDirectPerNodeCap: the 3rd ReplicaPlacement digit
|
||||
// (SameRackCount) caps shards per node directly (max == digit), matching the
|
||||
// per-rack DiffRackCount cap rather than allowing digit+1 per node.
|
||||
func TestPlaceSameRackCountIsDirectPerNodeCap(t *testing.T) {
|
||||
rp := &super_block.ReplicaPlacement{SameRackCount: 2} // <=2 shards per node
|
||||
|
||||
// 5 single-node racks: 5 nodes * 2 = 10 < 14, so a strict 10+4 placement
|
||||
// cannot satisfy the per-node cap and must fail. Under the old digit+1 reading
|
||||
// the cap would be 3/node => 15 slots and this would have wrongly succeeded.
|
||||
topo := buildPlaceTopo(5, 1, 50)
|
||||
if _, err := topo.Place(1, "c1", allShards(), Constraints{ReplicaPlacement: rp}, PlaceStrict); err == nil {
|
||||
t.Fatal("strict Place should fail: 5 nodes cannot hold 14 shards at <=2 per node")
|
||||
}
|
||||
|
||||
// Durability-first relaxes the unsatisfiable per-node cap, still places every
|
||||
// shard, and reports the relaxation so it isn't silently weakened.
|
||||
topo = buildPlaceTopo(5, 1, 50)
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{ReplicaPlacement: rp}, PlaceDurabilityFirst)
|
||||
if err != nil {
|
||||
t.Fatalf("durability-first Place: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d shards, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
relaxedRP := false
|
||||
for _, r := range res.Relaxed {
|
||||
if r == "replica-placement" {
|
||||
relaxedRP = true
|
||||
}
|
||||
}
|
||||
if !relaxedRP {
|
||||
t.Errorf("expected replica-placement relaxation, got %v", res.Relaxed)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceDiskTypeHardFilter: with DiskType set, shards land only on disks of
|
||||
// that type, even though the snapshot also contains other-typed disks.
|
||||
func TestPlaceDiskTypeHardFilter(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 4; r++ {
|
||||
rackKey := fmt.Sprintf("dc1:rack%d", r)
|
||||
ssd := topo.AddNode(fmt.Sprintf("ssd-%d:8080", r), "dc1", rackKey, 50)
|
||||
ssd.AddDisk(0, "ssd", 50, 0)
|
||||
hdd := topo.AddNode(fmt.Sprintf("hdd-%d:8080", r), "dc1", rackKey, 50)
|
||||
hdd.AddDisk(0, "hdd", 50, 0)
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{DiskType: "ssd", DiskTypePolicy: DiskTypeRequire}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place ssd: %v", err)
|
||||
}
|
||||
for sid, d := range res.Destinations {
|
||||
node := topo.nodes[d.Node]
|
||||
disk := node.disks[d.DiskID]
|
||||
if disk == nil || disk.diskType != "ssd" {
|
||||
t.Errorf("shard %d placed on non-ssd disk: node=%s diskID=%d", sid, d.Node, d.DiskID)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceDiskTypeUnavailableFails: a request for a disk type with no matching
|
||||
// disks fails rather than silently placing on the wrong tier.
|
||||
func TestPlaceDiskTypeUnavailableFails(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 4; r++ {
|
||||
n := topo.AddNode(fmt.Sprintf("hdd-%d:8080", r), "dc1", fmt.Sprintf("dc1:rack%d", r), 50)
|
||||
n.AddDisk(0, "hdd", 50, 0)
|
||||
}
|
||||
if _, err := topo.Place(1, "c1", allShards(), Constraints{DiskType: "ssd", DiskTypePolicy: DiskTypeRequire}, PlaceStrict); err == nil {
|
||||
t.Fatal("expected Place to fail when no disks of the requested type exist")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceHDDRequestMatchesEmptyTypeDisks: a "hdd" request normalizes to
|
||||
// HardDriveType ("") and must land on the HDD disk (reported as ""), never the SSD
|
||||
// disk, even on nodes that have both.
|
||||
func TestPlaceHDDRequestMatchesEmptyTypeDisks(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 6; r++ {
|
||||
n := topo.AddNode(fmt.Sprintf("n%d:8080", r), "dc1", fmt.Sprintf("dc1:rack%d", r), 100)
|
||||
n.AddDisk(0, "", 50, 0) // HDD (HardDriveType, reported as "")
|
||||
n.AddDisk(1, "ssd", 50, 0) // SSD
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{DiskType: "hdd", DiskTypePolicy: DiskTypeRequire}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place hdd: %v", err)
|
||||
}
|
||||
for sid, d := range res.Destinations {
|
||||
if d.DiskID != 0 { // disk 0 is the HDD disk on every node
|
||||
t.Errorf("shard %d placed on disk %d (expected HDD disk 0) on node %s", sid, d.DiskID, d.Node)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceDurabilityCapRejectsSkewed: in a near-full cluster where only one disk
|
||||
// has spare room, Place must not pile more than parityShards shards onto it (losing
|
||||
// it would then lose more than EC can recover). It fails instead, so the caller
|
||||
// leaves the volume unencoded rather than minting an unrecoverable layout.
|
||||
func TestPlaceDurabilityCapRejectsSkewed(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
// One spacious node plus four nearly-full ones, all in a single rack.
|
||||
a := topo.AddNode("a:8080", "dc1", "dc1:rack0", 100)
|
||||
a.AddDisk(0, "", 100, 0)
|
||||
for i := 0; i < 4; i++ {
|
||||
n := topo.AddNode(fmt.Sprintf("b%d:8080", i), "dc1", "dc1:rack0", 1)
|
||||
n.AddDisk(0, "", 1, 0)
|
||||
}
|
||||
|
||||
// 14 shards, parity 4: node a is capped at 4, the others hold 1 each -> at most
|
||||
// 4+4=8 placeable without exceeding parityShards on a disk, so Place must fail.
|
||||
// (Without the per-disk cap, a would greedily absorb 10 shards and "succeed".)
|
||||
if _, err := topo.Place(1, "c1", allShards(), Constraints{}, PlaceDurabilityFirst); err == nil {
|
||||
t.Fatal("expected Place to fail rather than pile >parityShards shards on one disk")
|
||||
}
|
||||
}
|
||||
|
||||
// TestReleaseVolumeShards: removes all of a volume's shards from the snapshot and
|
||||
// credits the freed capacity at BOTH disk and node level (rack capacity sums node
|
||||
// freeSlots), mirroring how FromActiveTopology accounts stale shards.
|
||||
func TestReleaseVolumeShards(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
// Node total 20 = two disks of 10. Two stale shards occupy one slot each, so the
|
||||
// snapshot would show disk0/disk1 at 9 and node at 18 (as FromActiveTopology does).
|
||||
n := topo.AddNode("n0:8080", "dc1", "dc1:rack0", 20)
|
||||
n.AddDisk(0, "", 10, 0)
|
||||
n.AddDisk(1, "", 10, 0)
|
||||
vk := volKey{collection: "c1", vid: 1}
|
||||
n.AddShards(1, "c1", 0, erasure_coding.ShardBits(uint32(1)<<3))
|
||||
n.AddShards(1, "c1", 1, erasure_coding.ShardBits(uint32(1)<<7))
|
||||
n.disks[0].freeSlots = 9
|
||||
n.disks[1].freeSlots = 9
|
||||
n.freeSlots = 18
|
||||
|
||||
topo.ReleaseVolumeShards("c1", 1)
|
||||
|
||||
if _, ok := n.shards[vk]; ok {
|
||||
t.Error("volume shards should be gone after ReleaseVolumeShards")
|
||||
}
|
||||
if n.disks[0].freeSlots != 10 || n.disks[1].freeSlots != 10 {
|
||||
t.Errorf("disk freeSlots not restored: disk0=%d, disk1=%d (want 10, 10)", n.disks[0].freeSlots, n.disks[1].freeSlots)
|
||||
}
|
||||
if n.freeSlots != 20 {
|
||||
t.Errorf("node freeSlots = %d, want 20 (must be credited at node level too)", n.freeSlots)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClearShardAccounting: dropping one disk's copy of a shard preserves a kept
|
||||
// copy of the same shard on another disk of the same node, and credits no capacity.
|
||||
func TestClearShardAccounting(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
n := topo.AddNode("n0:8080", "dc1", "dc1:rack0", 50)
|
||||
n.AddDisk(0, "", 50, 0)
|
||||
n.AddDisk(1, "", 50, 0)
|
||||
vk := volKey{collection: "c1", vid: 1}
|
||||
// Shard 3 lives on disk 0 (keep) and disk 1 (duplicate to delete).
|
||||
n.AddShards(1, "c1", 0, erasure_coding.ShardBits(uint32(1)<<3))
|
||||
n.AddShards(1, "c1", 1, erasure_coding.ShardBits(uint32(1)<<3))
|
||||
if got := n.shards[vk].shardBits.Count(); got != 1 {
|
||||
t.Fatalf("union count = %d, want 1", got)
|
||||
}
|
||||
freeBefore := n.disks[1].freeSlots
|
||||
|
||||
clearShardAccounting(n, vk, 3, 1)
|
||||
|
||||
if !n.shards[vk].shardBits.Has(erasure_coding.ShardId(3)) {
|
||||
t.Error("kept copy of shard 3 (disk 0) lost from the node-level union")
|
||||
}
|
||||
if n.shards[vk].diskShardBits[1].Has(erasure_coding.ShardId(3)) {
|
||||
t.Error("disk-1 copy of shard 3 was not cleared")
|
||||
}
|
||||
if n.disks[1].freeSlots != freeBefore {
|
||||
t.Errorf("freeSlots changed %d -> %d; clearShardAccounting must not credit capacity", freeBefore, n.disks[1].freeSlots)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceDiskTypePreferSpills: DiskTypePrefer fills the preferred type first and
|
||||
// spills the remainder to other types, reporting SpilledToOtherDiskType. SSD is
|
||||
// scarce (one tiny SSD per node) so the volume must spill to HDD, but there are
|
||||
// enough disks to keep each within the parityShards durability cap.
|
||||
func TestPlaceDiskTypePreferSpills(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 8; r++ {
|
||||
n := topo.AddNode(fmt.Sprintf("n%d:8080", r), "dc1", fmt.Sprintf("dc1:rack%d", r), 100)
|
||||
n.AddDisk(0, "ssd", 1, 0) // tiny SSD: 1 shard
|
||||
n.AddDisk(1, "", 50, 0) // roomy HDD
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{DiskType: "ssd", DiskTypePolicy: DiskTypePrefer}, PlaceDurabilityFirst)
|
||||
if err != nil {
|
||||
t.Fatalf("Place: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
ssd, hdd := 0, 0
|
||||
for _, d := range res.Destinations {
|
||||
if topo.nodes[d.Node].disks[d.DiskID].diskType == "ssd" {
|
||||
ssd++
|
||||
} else {
|
||||
hdd++
|
||||
}
|
||||
}
|
||||
if ssd == 0 || hdd == 0 {
|
||||
t.Errorf("expected prefer-then-spill: some shards on SSD and some on HDD, got ssd=%d hdd=%d", ssd, hdd)
|
||||
}
|
||||
if !res.SpilledToOtherDiskType {
|
||||
t.Error("expected SpilledToOtherDiskType when SSD cannot hold every shard")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlacePreferredTagsUseTaggedDisks: when tagged disks can hold the whole plan,
|
||||
// every shard lands on a tagged disk and no spill is reported.
|
||||
func TestPlacePreferredTagsUseTaggedDisks(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 4; r++ {
|
||||
rackKey := fmt.Sprintf("dc1:rack%d", r)
|
||||
fast := topo.AddNode(fmt.Sprintf("fast-%d:8080", r), "dc1", rackKey, 50)
|
||||
fast.AddDisk(0, "", 50, 0)
|
||||
fast.AddDiskTags(0, []string{"fast"})
|
||||
topo.AddNode(fmt.Sprintf("slow-%d:8080", r), "dc1", rackKey, 50).AddDisk(0, "", 50, 0)
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{PreferredTags: []string{"fast"}}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place: %v", err)
|
||||
}
|
||||
for sid, d := range res.Destinations {
|
||||
if !diskHasAnyTag(topo.nodes[d.Node].disks[d.DiskID], []string{"fast"}) {
|
||||
t.Errorf("shard %d placed on an untagged disk (node %s)", sid, d.Node)
|
||||
}
|
||||
}
|
||||
if res.SpilledOutsidePreferredTags {
|
||||
t.Error("did not expect tag spill when tagged disks suffice")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlacePreferredTagsSpillWhenInsufficient: when the tagged tier cannot hold the
|
||||
// whole plan, Place falls back to all disks and reports SpilledOutsidePreferredTags.
|
||||
func TestPlacePreferredTagsSpillWhenInsufficient(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
for r := 0; r < 4; r++ {
|
||||
n := topo.AddNode(fmt.Sprintf("n-%d:8080", r), "dc1", fmt.Sprintf("dc1:rack%d", r), 50)
|
||||
if r == 0 {
|
||||
n.AddDisk(0, "", 5, 0) // the only tagged disk, too small for 14 shards
|
||||
n.AddDiskTags(0, []string{"fast"})
|
||||
} else {
|
||||
n.AddDisk(0, "", 50, 0)
|
||||
}
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{PreferredTags: []string{"fast"}}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
if !res.SpilledOutsidePreferredTags {
|
||||
t.Error("expected SpilledOutsidePreferredTags when the single fast disk cannot hold the plan")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlaceStrictCapsCountEligibleRacks: with DiskTypeRequire, the even per-rack
|
||||
// cap divides by racks that have a matching disk, not all racks, so SSDs in only
|
||||
// some racks still place successfully.
|
||||
func TestPlaceStrictCapsCountEligibleRacks(t *testing.T) {
|
||||
topo := NewTopology()
|
||||
// SSDs live in only 2 of 4 racks, with several SSD nodes per rack so 14 shards
|
||||
// fit at <= parityShards per disk. The even per-rack cap must divide by the 2
|
||||
// eligible racks (ceil(10/2)=5 data/rack), not all 4 (ceil(10/4)=3 -> infeasible).
|
||||
for r := 0; r < 4; r++ {
|
||||
rackKey := fmt.Sprintf("dc1:rack%d", r)
|
||||
topo.AddNode(fmt.Sprintf("hdd-%d:8080", r), "dc1", rackKey, 50).AddDisk(0, "", 50, 0)
|
||||
if r < 2 {
|
||||
for n := 0; n < 4; n++ {
|
||||
topo.AddNode(fmt.Sprintf("ssd-%d-%d:8080", r, n), "dc1", rackKey, 50).AddDisk(0, "ssd", 50, 0)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
res, err := topo.Place(1, "c1", allShards(), Constraints{DiskType: "ssd", DiskTypePolicy: DiskTypeRequire}, PlaceStrict)
|
||||
if err != nil {
|
||||
t.Fatalf("Place ssd in 2/4 racks: %v", err)
|
||||
}
|
||||
if len(res.Destinations) != erasure_coding.TotalShardsCount {
|
||||
t.Fatalf("placed %d, want %d", len(res.Destinations), erasure_coding.TotalShardsCount)
|
||||
}
|
||||
for sid, d := range res.Destinations {
|
||||
if disk := topo.nodes[d.Node].disks[d.DiskID]; disk == nil || disk.diskType != "ssd" {
|
||||
t.Errorf("shard %d not on an SSD disk: node=%s disk=%d", sid, d.Node, d.DiskID)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
package ecbalancer
|
||||
|
||||
import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
)
|
||||
|
||||
// shardDataShards returns the data-shard count of the volume an EC shard belongs
|
||||
// to, used to size the shard's disk footprint. OSS uses the standard ratio for
|
||||
// every volume; custom per-volume ratios are an enterprise feature, so the
|
||||
// enterprise build overrides this to read the per-shard ratio.
|
||||
func shardDataShards(eci *master_pb.VolumeEcShardInformationMessage) int {
|
||||
return erasure_coding.DataShardsCount
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
package ecbalancer
|
||||
|
||||
import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/topology"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
)
|
||||
|
||||
// FromActiveTopology builds a Topology snapshot from the cluster's ActiveTopology
|
||||
// using the reservation-aware effective-capacity view that EC encode and repair
|
||||
// rely on. It collects ALL EC-eligible disks with no hard disk-type filter;
|
||||
// disk-type preference is applied later by callers (Place). Per-disk free EC shard
|
||||
// slots come from GetEffectiveAvailableEcShardSlots (shard-granular, so in-flight
|
||||
// task reservations that are not whole-volume multiples are not lost) minus the EC
|
||||
// shards already persisted on the disk. Rack keys are composite "dc:rack".
|
||||
//
|
||||
// dataShards is the target collection's data-shard count, used to size free EC
|
||||
// shard slots correctly for custom ratios (a 4+2 volume's shards are larger, so
|
||||
// fewer fit per volume slot). Pass <= 0 for the default scheme. Because of this,
|
||||
// the snapshot is ratio-specific; build one per collection ratio being placed.
|
||||
//
|
||||
// This is the encode/repair-side constructor; balance keeps its own raw-topology
|
||||
// builder (buildBalancerTopology) until the snapshot sources are reconciled.
|
||||
func FromActiveTopology(at *topology.ActiveTopology, dataShards int) *Topology {
|
||||
topo := NewTopology()
|
||||
if at == nil {
|
||||
return topo
|
||||
}
|
||||
|
||||
disks := at.GetDisksWithEffectiveCapacity(topology.TaskTypeErasureCoding, "", 0)
|
||||
|
||||
// Accumulate node-level free slots and group the node's disks together.
|
||||
nodeFree := make(map[string]int)
|
||||
nodeDC := make(map[string]string)
|
||||
nodeRack := make(map[string]string)
|
||||
byNode := make(map[string][]*topology.DiskInfo)
|
||||
for _, d := range disks {
|
||||
if d == nil || d.DiskInfo == nil {
|
||||
continue
|
||||
}
|
||||
if free := perDiskFreeECSlots(at, d, dataShards); free > 0 {
|
||||
nodeFree[d.NodeID] += free
|
||||
}
|
||||
nodeDC[d.NodeID] = d.DataCenter
|
||||
nodeRack[d.NodeID] = d.DataCenter + ":" + d.Rack
|
||||
byNode[d.NodeID] = append(byNode[d.NodeID], d)
|
||||
}
|
||||
|
||||
for nodeID, ds := range byNode {
|
||||
node := topo.AddNode(nodeID, nodeDC[nodeID], nodeRack[nodeID], nodeFree[nodeID])
|
||||
for _, d := range ds {
|
||||
free := perDiskFreeECSlots(at, d, dataShards)
|
||||
if free < 0 {
|
||||
free = 0
|
||||
}
|
||||
node.AddDisk(d.DiskID, d.DiskType, free, ecShardCountOnDisk(d))
|
||||
node.AddDiskTags(d.DiskID, d.DiskInfo.Tags)
|
||||
for _, eci := range d.DiskInfo.EcShardInfos {
|
||||
if eci.DiskId != d.DiskID {
|
||||
continue
|
||||
}
|
||||
node.AddShards(eci.Id, eci.Collection, d.DiskID, erasure_coding.ShardBits(eci.EcIndexBits))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return topo
|
||||
}
|
||||
|
||||
// perDiskFreeECSlots returns the disk's free EC shard slots for a volume with
|
||||
// dataShards data shards: the topology's shard-granular effective availability
|
||||
// (sized by the target ratio) minus the EC shards already on the disk, also
|
||||
// expressed in the target ratio's shard slots so mixed-ratio disks are charged by
|
||||
// size rather than raw count.
|
||||
func perDiskFreeECSlots(at *topology.ActiveTopology, d *topology.DiskInfo, dataShards int) int {
|
||||
return at.GetEffectiveAvailableEcShardSlots(d.NodeID, d.DiskID, dataShards) - ecShardSlotsOnDisk(d, dataShards)
|
||||
}
|
||||
|
||||
// ecShardCountOnDisk counts the EC shards physically on this disk (matching
|
||||
// eci.DiskId), across all volumes. Used as a per-disk load metric for scoring.
|
||||
func ecShardCountOnDisk(d *topology.DiskInfo) int {
|
||||
count := 0
|
||||
for _, eci := range d.DiskInfo.EcShardInfos {
|
||||
if eci.DiskId == d.DiskID {
|
||||
count += erasure_coding.GetShardCount(eci)
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// ecShardSlotsOnDisk returns the EC shards already on the disk expressed in the
|
||||
// TARGET collection's shard slots. A shard of a collection with d data shards
|
||||
// occupies ~1/d of a volume, i.e. targetDataShards/d target slots, so a 2+1 shard
|
||||
// counted against a 10+4 snapshot consumes ~5 slots, not 1.
|
||||
func ecShardSlotsOnDisk(d *topology.DiskInfo, targetDataShards int) int {
|
||||
if targetDataShards <= 0 {
|
||||
targetDataShards = erasure_coding.DataShardsCount
|
||||
}
|
||||
total := 0
|
||||
for _, eci := range d.DiskInfo.EcShardInfos {
|
||||
if eci.DiskId != d.DiskID {
|
||||
continue
|
||||
}
|
||||
ds := shardDataShards(eci)
|
||||
if ds <= 0 {
|
||||
ds = erasure_coding.DataShardsCount
|
||||
}
|
||||
// Round up so an existing shard always consumes at least its fractional
|
||||
// footprint; flooring lets a low-data-shard volume (targetDataShards < ds)
|
||||
// count as zero target slots and overstate the disk's free capacity.
|
||||
total += (erasure_coding.GetShardCount(eci)*targetDataShards + ds - 1) / ds
|
||||
}
|
||||
return total
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
package ecbalancer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/topology"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
)
|
||||
|
||||
// TestFromActiveTopology verifies the encode/repair-side snapshot constructor maps
|
||||
// nodes, per-disk EC shard counts, per-volume shard bits, and free slots from an
|
||||
// ActiveTopology. Shard accounting is asserted exactly; free slots are asserted to
|
||||
// be positive (their exact value depends on effective-capacity internals).
|
||||
func TestFromActiveTopology(t *testing.T) {
|
||||
const vid uint32 = 7
|
||||
at := topology.NewActiveTopology(10)
|
||||
|
||||
// Node A holds one EC shard (id 3) of volume 7 on disk 0; node B is empty.
|
||||
nodeA := &master_pb.DataNodeInfo{
|
||||
Id: "10.0.0.1:8080",
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"hdd": {
|
||||
DiskId: 0,
|
||||
MaxVolumeCount: 100,
|
||||
VolumeCount: 1,
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{{
|
||||
Id: vid,
|
||||
Collection: "c1",
|
||||
EcIndexBits: uint32(1) << 3,
|
||||
DiskId: 0,
|
||||
}},
|
||||
},
|
||||
},
|
||||
}
|
||||
nodeB := &master_pb.DataNodeInfo{
|
||||
Id: "10.0.0.2:8080",
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"hdd": {DiskId: 0, MaxVolumeCount: 100, VolumeCount: 0},
|
||||
},
|
||||
}
|
||||
if err := at.UpdateTopology(&master_pb.TopologyInfo{
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{{
|
||||
Id: "dc1",
|
||||
RackInfos: []*master_pb.RackInfo{{
|
||||
Id: "rack1",
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{nodeA, nodeB},
|
||||
}},
|
||||
}},
|
||||
}); err != nil {
|
||||
t.Fatalf("UpdateTopology: %v", err)
|
||||
}
|
||||
|
||||
topo := FromActiveTopology(at, 0)
|
||||
|
||||
if got := len(topo.nodes); got != 2 {
|
||||
t.Fatalf("node count = %d, want 2", got)
|
||||
}
|
||||
|
||||
a := topo.nodes["10.0.0.1:8080"]
|
||||
if a == nil {
|
||||
t.Fatal("node A missing from snapshot")
|
||||
}
|
||||
if a.rack != "dc1:rack1" {
|
||||
t.Errorf("node A rack = %q, want dc1:rack1", a.rack)
|
||||
}
|
||||
diskA := a.disks[0]
|
||||
if diskA == nil {
|
||||
t.Fatal("node A disk 0 missing")
|
||||
}
|
||||
if diskA.shardCount != 1 {
|
||||
t.Errorf("node A disk 0 shardCount = %d, want 1", diskA.shardCount)
|
||||
}
|
||||
vs := a.shards[volKey{collection: "c1", vid: vid}]
|
||||
if vs == nil {
|
||||
t.Fatal("volume 7 shards not recorded on node A")
|
||||
}
|
||||
if !vs.shardBits.Has(erasure_coding.ShardId(3)) {
|
||||
t.Errorf("node A volume 7 shardBits %b missing shard 3", vs.shardBits)
|
||||
}
|
||||
if vs.shardBits.Count() != 1 {
|
||||
t.Errorf("node A volume 7 shard count = %d, want 1", vs.shardBits.Count())
|
||||
}
|
||||
|
||||
b := topo.nodes["10.0.0.2:8080"]
|
||||
if b == nil {
|
||||
t.Fatal("node B missing from snapshot")
|
||||
}
|
||||
if b.rack != "dc1:rack1" {
|
||||
t.Errorf("node B rack = %q, want dc1:rack1", b.rack)
|
||||
}
|
||||
if diskB := b.disks[0]; diskB == nil || diskB.shardCount != 0 {
|
||||
t.Errorf("node B disk 0 shardCount = %v, want 0", diskB)
|
||||
}
|
||||
if len(b.shards) != 0 {
|
||||
t.Errorf("node B should hold no volume shards, got %d", len(b.shards))
|
||||
}
|
||||
|
||||
// Free slots should be positive on both near-empty disks.
|
||||
if a.freeSlots <= 0 || b.freeSlots <= 0 {
|
||||
t.Errorf("free slots not positive: A=%d B=%d", a.freeSlots, b.freeSlots)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEcShardSlotsOnDiskRoundsUp covers the mixed-ratio (targetDataShards <
|
||||
// existingDataShards) conversion: an existing shard's fractional footprint must
|
||||
// round up so it is never floored to zero, which would overstate free capacity.
|
||||
// OSS always uses the standard ratio at runtime, but ecShardSlotsOnDisk takes the
|
||||
// target data-shard count as a parameter, so the fractional path is exercised
|
||||
// directly here; the enterprise build reaches it with real per-volume ratios.
|
||||
func TestEcShardSlotsOnDiskRoundsUp(t *testing.T) {
|
||||
// A single shard (id 3) of a standard 10-data-shard volume on disk 0.
|
||||
disk := &topology.DiskInfo{
|
||||
DiskID: 0,
|
||||
DiskInfo: &master_pb.DiskInfo{
|
||||
DiskId: 0,
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{{
|
||||
Id: 7,
|
||||
Collection: "c1",
|
||||
EcIndexBits: uint32(1) << 3,
|
||||
DiskId: 0,
|
||||
}},
|
||||
},
|
||||
}
|
||||
|
||||
// Against a 2-data-shard target the shard occupies 2/10 of a slot, which must
|
||||
// round up to 1 rather than floor to 0.
|
||||
if got := ecShardSlotsOnDisk(disk, 2); got != 1 {
|
||||
t.Errorf("ecShardSlotsOnDisk(target=2) = %d, want 1 (rounded up from 0.2)", got)
|
||||
}
|
||||
|
||||
// Identity case: target equals the existing data-shard count, so the shard
|
||||
// consumes exactly its whole-number footprint.
|
||||
if got := ecShardSlotsOnDisk(disk, erasure_coding.DataShardsCount); got != 1 {
|
||||
t.Errorf("ecShardSlotsOnDisk(target=%d) = %d, want 1", erasure_coding.DataShardsCount, got)
|
||||
}
|
||||
}
|
||||
@@ -1,461 +0,0 @@
|
||||
// Package placement provides consolidated EC shard placement logic used by
|
||||
// both shell commands and worker tasks.
|
||||
//
|
||||
// This package encapsulates the algorithms for:
|
||||
// - Selecting destination nodes/disks for EC shards
|
||||
// - Ensuring proper spread across racks, servers, and disks
|
||||
// - Balancing shards across the cluster
|
||||
package placement
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// DiskCandidate represents a disk that can receive EC shards
|
||||
type DiskCandidate struct {
|
||||
NodeID string
|
||||
DiskID uint32
|
||||
DataCenter string
|
||||
Rack string
|
||||
DiskType string // disk type (hdd/ssd/...) — empty means HardDrive
|
||||
|
||||
// Capacity information
|
||||
VolumeCount int64
|
||||
MaxVolumeCount int64
|
||||
ShardCount int // Current number of EC shards on this disk
|
||||
FreeSlots int // Available slots for new shards
|
||||
|
||||
// Load information
|
||||
LoadCount int // Number of active tasks on this disk
|
||||
}
|
||||
|
||||
// NodeCandidate represents a server node that can receive EC shards
|
||||
type NodeCandidate struct {
|
||||
NodeID string
|
||||
DataCenter string
|
||||
Rack string
|
||||
FreeSlots int
|
||||
ShardCount int // Total shards across all disks
|
||||
Disks []*DiskCandidate // All disks on this node
|
||||
}
|
||||
|
||||
// PlacementRequest configures EC shard placement behavior
|
||||
type PlacementRequest struct {
|
||||
// ShardsNeeded is the total number of shards to place
|
||||
ShardsNeeded int
|
||||
|
||||
// MaxShardsPerServer limits how many shards can be placed on a single server
|
||||
// 0 means no limit (but prefer spreading when possible)
|
||||
MaxShardsPerServer int
|
||||
|
||||
// MaxShardsPerRack limits how many shards can be placed in a single rack
|
||||
// 0 means no limit
|
||||
MaxShardsPerRack int
|
||||
|
||||
// MaxTaskLoad is the maximum task load count for a disk to be considered
|
||||
MaxTaskLoad int
|
||||
|
||||
// PreferDifferentServers when true, spreads shards across different servers
|
||||
// before using multiple disks on the same server
|
||||
PreferDifferentServers bool
|
||||
|
||||
// PreferDifferentRacks when true, spreads shards across different racks
|
||||
// before using multiple servers in the same rack
|
||||
PreferDifferentRacks bool
|
||||
|
||||
// PreferredDiskType, when non-empty, biases placement toward disks of
|
||||
// this type. Disks of the preferred type are exhausted (subject to the
|
||||
// other diversity preferences) before disks of any other type are
|
||||
// considered. Empty means no disk-type bias — all suitable disks form a
|
||||
// single pool, matching pre-#9423 behavior.
|
||||
PreferredDiskType string
|
||||
}
|
||||
|
||||
// PlacementResult contains the selected destinations for EC shards
|
||||
type PlacementResult struct {
|
||||
SelectedDisks []*DiskCandidate
|
||||
|
||||
// Statistics
|
||||
ServersUsed int
|
||||
RacksUsed int
|
||||
DCsUsed int
|
||||
|
||||
// Distribution maps
|
||||
ShardsPerServer map[string]int
|
||||
ShardsPerRack map[string]int
|
||||
ShardsPerDC map[string]int
|
||||
|
||||
// SpilledToOtherDiskType is set when PlacementRequest.PreferredDiskType
|
||||
// was non-empty but the preferred-type pool could not satisfy
|
||||
// ShardsNeeded, so placement had to spill onto disks of other types.
|
||||
// Callers can log a warning when this is true.
|
||||
SpilledToOtherDiskType bool
|
||||
}
|
||||
|
||||
// SelectDestinations selects the best disks for EC shard placement.
|
||||
// This is the main entry point for EC placement logic.
|
||||
//
|
||||
// Disk-type preference (#9423): when config.PreferredDiskType is non-empty,
|
||||
// suitable disks are partitioned into a matching-type tier and a
|
||||
// fallback tier. Each tier is run through the diversity passes below;
|
||||
// the fallback tier is only consulted if the matching tier runs out of
|
||||
// candidates before ShardsNeeded is satisfied. Empty PreferredDiskType
|
||||
// processes all suitable disks as one tier, preserving prior behavior.
|
||||
//
|
||||
// Within each tier, the algorithm works in multiple passes:
|
||||
// 1. First pass: Select one disk from each rack (maximize rack diversity)
|
||||
// 2. Second pass: Select one disk from each unused server in used racks (maximize server diversity)
|
||||
// 3. Third pass: Select additional disks from servers already used (maximize disk diversity)
|
||||
func SelectDestinations(disks []*DiskCandidate, config PlacementRequest) (*PlacementResult, error) {
|
||||
if len(disks) == 0 {
|
||||
return nil, fmt.Errorf("no disk candidates provided")
|
||||
}
|
||||
if config.ShardsNeeded <= 0 {
|
||||
return nil, fmt.Errorf("shardsNeeded must be positive, got %d", config.ShardsNeeded)
|
||||
}
|
||||
|
||||
// Filter suitable disks
|
||||
suitable := filterSuitableDisks(disks, config)
|
||||
if len(suitable) == 0 {
|
||||
return nil, fmt.Errorf("no suitable disks found after filtering")
|
||||
}
|
||||
|
||||
result := &PlacementResult{
|
||||
SelectedDisks: make([]*DiskCandidate, 0, config.ShardsNeeded),
|
||||
ShardsPerServer: make(map[string]int),
|
||||
ShardsPerRack: make(map[string]int),
|
||||
ShardsPerDC: make(map[string]int),
|
||||
}
|
||||
|
||||
usedDisks := make(map[string]bool) // "nodeID:diskID" -> bool
|
||||
usedServers := make(map[string]bool) // nodeID -> bool
|
||||
usedRacks := make(map[string]bool) // "dc:rack" -> bool
|
||||
|
||||
// Partition suitable into preferred-disk-type / fallback tiers.
|
||||
// Process the preferred tier first; only spill to fallback when the
|
||||
// preferred pool can't satisfy ShardsNeeded.
|
||||
preferredTier, fallbackTier := partitionByDiskType(suitable, config.PreferredDiskType)
|
||||
selectFromTier(preferredTier, result, usedDisks, usedServers, usedRacks, config)
|
||||
if config.PreferredDiskType != "" && len(result.SelectedDisks) < config.ShardsNeeded && len(fallbackTier) > 0 {
|
||||
before := len(result.SelectedDisks)
|
||||
selectFromTier(fallbackTier, result, usedDisks, usedServers, usedRacks, config)
|
||||
if len(result.SelectedDisks) > before {
|
||||
result.SpilledToOtherDiskType = true
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate final statistics
|
||||
result.ServersUsed = len(usedServers)
|
||||
result.RacksUsed = len(usedRacks)
|
||||
dcSet := make(map[string]bool)
|
||||
for _, disk := range result.SelectedDisks {
|
||||
dcSet[disk.DataCenter] = true
|
||||
}
|
||||
result.DCsUsed = len(dcSet)
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// partitionByDiskType splits disks into (matching, fallback) based on the
|
||||
// preferred disk type. If preferred is empty, everything goes into the
|
||||
// matching tier and fallback is empty — i.e. existing single-pool behavior.
|
||||
//
|
||||
// Empty DiskCandidate.DiskType is treated as HardDriveType ("hdd") to
|
||||
// mirror weed/storage/types.ToDiskType's normalization, so a
|
||||
// PreferredDiskType of "hdd" matches disks reporting "" — otherwise EC
|
||||
// shards from an HDD source would always spill onto disks that happen to
|
||||
// report their type as "" (HardDriveType).
|
||||
func partitionByDiskType(disks []*DiskCandidate, preferred string) (matching, fallback []*DiskCandidate) {
|
||||
if preferred == "" {
|
||||
return disks, nil
|
||||
}
|
||||
pref := normalizeDiskType(preferred)
|
||||
for _, d := range disks {
|
||||
if normalizeDiskType(d.DiskType) == pref {
|
||||
matching = append(matching, d)
|
||||
} else {
|
||||
fallback = append(fallback, d)
|
||||
}
|
||||
}
|
||||
return matching, fallback
|
||||
}
|
||||
|
||||
// normalizeDiskType lower-cases the input and folds "" to "hdd" so the
|
||||
// HardDriveType sentinel ("") and explicit "hdd"/"HDD" all compare equal.
|
||||
func normalizeDiskType(t string) string {
|
||||
t = strings.ToLower(t)
|
||||
if t == "" {
|
||||
return "hdd"
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
// selectFromTier runs the three diversity passes against `tier`, mutating
|
||||
// `result` and the used* maps in place. Passes stop as soon as ShardsNeeded
|
||||
// is reached. The function is a no-op when the tier is empty or the result
|
||||
// already has enough shards, so it is safe to call once per tier.
|
||||
func selectFromTier(tier []*DiskCandidate, result *PlacementResult,
|
||||
usedDisks, usedServers, usedRacks map[string]bool,
|
||||
config PlacementRequest) {
|
||||
|
||||
if len(tier) == 0 || len(result.SelectedDisks) >= config.ShardsNeeded {
|
||||
return
|
||||
}
|
||||
|
||||
rackToDisks := groupDisksByRack(tier)
|
||||
|
||||
// Pass 1: Select one disk from each rack (maximize rack diversity).
|
||||
// When this is the fallback tier (preferred tier already populated
|
||||
// usedRacks), skip those racks so the spillover still spreads onto
|
||||
// new racks instead of doubling up on ones already picked.
|
||||
if config.PreferDifferentRacks {
|
||||
// Sort racks by number of available servers (descending) to prioritize racks with more options
|
||||
sortedRacks := sortRacksByServerCount(rackToDisks)
|
||||
for _, rackKey := range sortedRacks {
|
||||
if len(result.SelectedDisks) >= config.ShardsNeeded {
|
||||
break
|
||||
}
|
||||
if usedRacks[rackKey] {
|
||||
continue
|
||||
}
|
||||
rackDisks := rackToDisks[rackKey]
|
||||
// Select best disk from this rack, preferring a new server
|
||||
disk := selectBestDiskFromRack(rackDisks, usedServers, usedDisks, config)
|
||||
if disk != nil {
|
||||
addDiskToResult(result, disk, usedDisks, usedServers, usedRacks)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: Select disks from unused servers in already-used racks
|
||||
if config.PreferDifferentServers && len(result.SelectedDisks) < config.ShardsNeeded {
|
||||
for _, rackKey := range getSortedRackKeys(rackToDisks) {
|
||||
if len(result.SelectedDisks) >= config.ShardsNeeded {
|
||||
break
|
||||
}
|
||||
rackDisks := rackToDisks[rackKey]
|
||||
for _, disk := range sortDisksByScore(rackDisks) {
|
||||
if len(result.SelectedDisks) >= config.ShardsNeeded {
|
||||
break
|
||||
}
|
||||
diskKey := getDiskKey(disk)
|
||||
if usedDisks[diskKey] {
|
||||
continue
|
||||
}
|
||||
// Skip if server already used (we want different servers in this pass)
|
||||
if usedServers[disk.NodeID] {
|
||||
continue
|
||||
}
|
||||
// Check server limit
|
||||
if config.MaxShardsPerServer > 0 && result.ShardsPerServer[disk.NodeID] >= config.MaxShardsPerServer {
|
||||
continue
|
||||
}
|
||||
// Check rack limit
|
||||
if config.MaxShardsPerRack > 0 && result.ShardsPerRack[getRackKey(disk)] >= config.MaxShardsPerRack {
|
||||
continue
|
||||
}
|
||||
addDiskToResult(result, disk, usedDisks, usedServers, usedRacks)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3: Fill remaining slots from already-used servers (different disks)
|
||||
// Use round-robin across servers to balance shards evenly
|
||||
if len(result.SelectedDisks) < config.ShardsNeeded {
|
||||
// Group remaining disks by server (within this tier)
|
||||
serverToRemainingDisks := make(map[string][]*DiskCandidate)
|
||||
for _, disk := range tier {
|
||||
if !usedDisks[getDiskKey(disk)] {
|
||||
serverToRemainingDisks[disk.NodeID] = append(serverToRemainingDisks[disk.NodeID], disk)
|
||||
}
|
||||
}
|
||||
|
||||
// Sort each server's disks by score
|
||||
for serverID := range serverToRemainingDisks {
|
||||
serverToRemainingDisks[serverID] = sortDisksByScore(serverToRemainingDisks[serverID])
|
||||
}
|
||||
|
||||
// Round-robin: repeatedly select from the server with the fewest shards
|
||||
for len(result.SelectedDisks) < config.ShardsNeeded {
|
||||
// Find server with fewest shards that still has available disks
|
||||
var bestServer string
|
||||
minShards := -1
|
||||
for serverID, disks := range serverToRemainingDisks {
|
||||
if len(disks) == 0 {
|
||||
continue
|
||||
}
|
||||
// Check server limit
|
||||
if config.MaxShardsPerServer > 0 && result.ShardsPerServer[serverID] >= config.MaxShardsPerServer {
|
||||
continue
|
||||
}
|
||||
shardCount := result.ShardsPerServer[serverID]
|
||||
if minShards == -1 || shardCount < minShards {
|
||||
minShards = shardCount
|
||||
bestServer = serverID
|
||||
} else if shardCount == minShards && serverID < bestServer {
|
||||
// Tie-break by server name for determinism
|
||||
bestServer = serverID
|
||||
}
|
||||
}
|
||||
|
||||
if bestServer == "" {
|
||||
// No more servers with available disks
|
||||
break
|
||||
}
|
||||
|
||||
// Pop the best disk from this server
|
||||
disks := serverToRemainingDisks[bestServer]
|
||||
disk := disks[0]
|
||||
serverToRemainingDisks[bestServer] = disks[1:]
|
||||
|
||||
// Check rack limit
|
||||
if config.MaxShardsPerRack > 0 && result.ShardsPerRack[getRackKey(disk)] >= config.MaxShardsPerRack {
|
||||
continue
|
||||
}
|
||||
|
||||
addDiskToResult(result, disk, usedDisks, usedServers, usedRacks)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// filterSuitableDisks filters disks that are suitable for EC placement
|
||||
func filterSuitableDisks(disks []*DiskCandidate, config PlacementRequest) []*DiskCandidate {
|
||||
var suitable []*DiskCandidate
|
||||
for _, disk := range disks {
|
||||
if disk.FreeSlots <= 0 {
|
||||
continue
|
||||
}
|
||||
if config.MaxTaskLoad > 0 && disk.LoadCount > config.MaxTaskLoad {
|
||||
continue
|
||||
}
|
||||
suitable = append(suitable, disk)
|
||||
}
|
||||
return suitable
|
||||
}
|
||||
|
||||
// groupDisksByRack groups disks by their rack (dc:rack key)
|
||||
func groupDisksByRack(disks []*DiskCandidate) map[string][]*DiskCandidate {
|
||||
result := make(map[string][]*DiskCandidate)
|
||||
for _, disk := range disks {
|
||||
key := getRackKey(disk)
|
||||
result[key] = append(result[key], disk)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// getRackKey returns the unique key for a rack (dc:rack)
|
||||
func getRackKey(disk *DiskCandidate) string {
|
||||
return fmt.Sprintf("%s:%s", disk.DataCenter, disk.Rack)
|
||||
}
|
||||
|
||||
// getDiskKey returns the unique key for a disk (nodeID:diskID)
|
||||
func getDiskKey(disk *DiskCandidate) string {
|
||||
return fmt.Sprintf("%s:%d", disk.NodeID, disk.DiskID)
|
||||
}
|
||||
|
||||
// sortRacksByServerCount returns rack keys sorted by number of servers (ascending)
|
||||
func sortRacksByServerCount(rackToDisks map[string][]*DiskCandidate) []string {
|
||||
// Count unique servers per rack
|
||||
rackServerCount := make(map[string]int)
|
||||
for rackKey, disks := range rackToDisks {
|
||||
servers := make(map[string]bool)
|
||||
for _, disk := range disks {
|
||||
servers[disk.NodeID] = true
|
||||
}
|
||||
rackServerCount[rackKey] = len(servers)
|
||||
}
|
||||
|
||||
keys := getSortedRackKeys(rackToDisks)
|
||||
sort.Slice(keys, func(i, j int) bool {
|
||||
// Sort by server count (descending) to pick from racks with more options first
|
||||
return rackServerCount[keys[i]] > rackServerCount[keys[j]]
|
||||
})
|
||||
return keys
|
||||
}
|
||||
|
||||
// getSortedRackKeys returns rack keys in a deterministic order
|
||||
func getSortedRackKeys(rackToDisks map[string][]*DiskCandidate) []string {
|
||||
keys := make([]string, 0, len(rackToDisks))
|
||||
for k := range rackToDisks {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
return keys
|
||||
}
|
||||
|
||||
// selectBestDiskFromRack selects the best disk from a rack for EC placement
|
||||
// It prefers servers that haven't been used yet
|
||||
func selectBestDiskFromRack(disks []*DiskCandidate, usedServers, usedDisks map[string]bool, config PlacementRequest) *DiskCandidate {
|
||||
var bestDisk *DiskCandidate
|
||||
bestScore := -1.0
|
||||
bestIsFromUnusedServer := false
|
||||
|
||||
for _, disk := range disks {
|
||||
if usedDisks[getDiskKey(disk)] {
|
||||
continue
|
||||
}
|
||||
isFromUnusedServer := !usedServers[disk.NodeID]
|
||||
score := calculateDiskScore(disk)
|
||||
|
||||
// Prefer unused servers
|
||||
if isFromUnusedServer && !bestIsFromUnusedServer {
|
||||
bestDisk = disk
|
||||
bestScore = score
|
||||
bestIsFromUnusedServer = true
|
||||
} else if isFromUnusedServer == bestIsFromUnusedServer && score > bestScore {
|
||||
bestDisk = disk
|
||||
bestScore = score
|
||||
}
|
||||
}
|
||||
|
||||
return bestDisk
|
||||
}
|
||||
|
||||
// sortDisksByScore returns disks sorted by score (best first)
|
||||
func sortDisksByScore(disks []*DiskCandidate) []*DiskCandidate {
|
||||
sorted := make([]*DiskCandidate, len(disks))
|
||||
copy(sorted, disks)
|
||||
sort.Slice(sorted, func(i, j int) bool {
|
||||
return calculateDiskScore(sorted[i]) > calculateDiskScore(sorted[j])
|
||||
})
|
||||
return sorted
|
||||
}
|
||||
|
||||
// calculateDiskScore calculates a score for a disk candidate
|
||||
// Higher score is better
|
||||
func calculateDiskScore(disk *DiskCandidate) float64 {
|
||||
score := 0.0
|
||||
|
||||
// Primary factor: available capacity (lower utilization is better)
|
||||
if disk.MaxVolumeCount > 0 {
|
||||
utilization := float64(disk.VolumeCount) / float64(disk.MaxVolumeCount)
|
||||
score += (1.0 - utilization) * 60.0 // Up to 60 points
|
||||
} else {
|
||||
score += 30.0 // Default if no max count
|
||||
}
|
||||
|
||||
// Secondary factor: fewer shards already on this disk is better
|
||||
score += float64(10-disk.ShardCount) * 2.0 // Up to 20 points
|
||||
|
||||
// Tertiary factor: lower load is better
|
||||
score += float64(10 - disk.LoadCount) // Up to 10 points
|
||||
|
||||
return score
|
||||
}
|
||||
|
||||
// addDiskToResult adds a disk to the result and updates tracking maps
|
||||
func addDiskToResult(result *PlacementResult, disk *DiskCandidate,
|
||||
usedDisks, usedServers, usedRacks map[string]bool) {
|
||||
diskKey := getDiskKey(disk)
|
||||
rackKey := getRackKey(disk)
|
||||
|
||||
result.SelectedDisks = append(result.SelectedDisks, disk)
|
||||
usedDisks[diskKey] = true
|
||||
usedServers[disk.NodeID] = true
|
||||
usedRacks[rackKey] = true
|
||||
result.ShardsPerServer[disk.NodeID]++
|
||||
result.ShardsPerRack[rackKey]++
|
||||
result.ShardsPerDC[disk.DataCenter]++
|
||||
}
|
||||
@@ -1,143 +0,0 @@
|
||||
package placement
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// makeDisk builds a DiskCandidate with sensible defaults; tests override
|
||||
// only the fields they care about.
|
||||
func makeDisk(node, rack, diskType string, diskID uint32) *DiskCandidate {
|
||||
return &DiskCandidate{
|
||||
NodeID: node,
|
||||
DiskID: diskID,
|
||||
DataCenter: "dc1",
|
||||
Rack: rack,
|
||||
DiskType: diskType,
|
||||
VolumeCount: 0,
|
||||
MaxVolumeCount: 100,
|
||||
FreeSlots: 100,
|
||||
}
|
||||
}
|
||||
|
||||
func disksByType(disks []*DiskCandidate) map[string]int {
|
||||
out := map[string]int{}
|
||||
for _, d := range disks {
|
||||
out[d.DiskType]++
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func newRequest(shards int, preferred string) PlacementRequest {
|
||||
return PlacementRequest{
|
||||
ShardsNeeded: shards,
|
||||
PreferDifferentServers: true,
|
||||
PreferDifferentRacks: true,
|
||||
PreferredDiskType: preferred,
|
||||
}
|
||||
}
|
||||
|
||||
// Plenty of SSD disks available: placement should fill entirely from SSD
|
||||
// when PreferredDiskType="ssd", leaving HDDs untouched and not flagging
|
||||
// spillover.
|
||||
func TestSelectDestinations_PrefersMatchingDiskType(t *testing.T) {
|
||||
var disks []*DiskCandidate
|
||||
for i := 0; i < 6; i++ {
|
||||
disks = append(disks, makeDisk("ssd-"+strconv.Itoa(i), "r"+strconv.Itoa(i%3), "ssd", uint32(i)))
|
||||
}
|
||||
for i := 0; i < 6; i++ {
|
||||
disks = append(disks, makeDisk("hdd-"+strconv.Itoa(i), "r"+strconv.Itoa(i%3), "", uint32(i)))
|
||||
}
|
||||
|
||||
result, err := SelectDestinations(disks, newRequest(4, "ssd"))
|
||||
if err != nil {
|
||||
t.Fatalf("SelectDestinations: %v", err)
|
||||
}
|
||||
if got := len(result.SelectedDisks); got != 4 {
|
||||
t.Fatalf("selected %d disks, want 4", got)
|
||||
}
|
||||
if counts := disksByType(result.SelectedDisks); counts["ssd"] != 4 {
|
||||
t.Fatalf("disk-type counts = %v, want ssd=4", counts)
|
||||
}
|
||||
if result.SpilledToOtherDiskType {
|
||||
t.Fatalf("SpilledToOtherDiskType should be false when preferred pool was sufficient")
|
||||
}
|
||||
}
|
||||
|
||||
// Only one SSD disk available but 4 shards needed: placement must consume
|
||||
// the SSD first, then spill to HDD for the remainder, and report spillover.
|
||||
func TestSelectDestinations_SpillsWhenPreferredScarce(t *testing.T) {
|
||||
disks := []*DiskCandidate{
|
||||
makeDisk("ssd-0", "r0", "ssd", 0),
|
||||
makeDisk("hdd-0", "r1", "", 0),
|
||||
makeDisk("hdd-1", "r2", "", 0),
|
||||
makeDisk("hdd-2", "r3", "", 0),
|
||||
}
|
||||
|
||||
result, err := SelectDestinations(disks, newRequest(4, "ssd"))
|
||||
if err != nil {
|
||||
t.Fatalf("SelectDestinations: %v", err)
|
||||
}
|
||||
if got := len(result.SelectedDisks); got != 4 {
|
||||
t.Fatalf("selected %d disks, want 4", got)
|
||||
}
|
||||
counts := disksByType(result.SelectedDisks)
|
||||
if counts["ssd"] != 1 || counts[""] != 3 {
|
||||
t.Fatalf("disk-type counts = %v, want ssd=1 hdd=3", counts)
|
||||
}
|
||||
if !result.SpilledToOtherDiskType {
|
||||
t.Fatalf("SpilledToOtherDiskType should be true after falling back to HDD")
|
||||
}
|
||||
}
|
||||
|
||||
// PreferredDiskType="hdd" must match disks whose DiskType is "" (the
|
||||
// HardDriveType sentinel) — otherwise EC encoding of an HDD source would
|
||||
// always spill onto HDDs that happen to report disk_type="" even though
|
||||
// the cluster has plenty of matching capacity.
|
||||
func TestSelectDestinations_PreferredHddMatchesEmptyDiskType(t *testing.T) {
|
||||
disks := []*DiskCandidate{
|
||||
makeDisk("hdd-0", "r0", "", 0), // HardDriveType sentinel
|
||||
makeDisk("hdd-1", "r1", "", 0), // HardDriveType sentinel
|
||||
makeDisk("ssd-0", "r2", "ssd", 0),
|
||||
}
|
||||
|
||||
result, err := SelectDestinations(disks, newRequest(2, "hdd"))
|
||||
if err != nil {
|
||||
t.Fatalf("SelectDestinations: %v", err)
|
||||
}
|
||||
if got := len(result.SelectedDisks); got != 2 {
|
||||
t.Fatalf("selected %d disks, want 2", got)
|
||||
}
|
||||
// Both selected disks must be HDD-reporting (i.e. DiskType == ""),
|
||||
// and no spillover should have been required.
|
||||
for _, d := range result.SelectedDisks {
|
||||
if d.DiskType != "" {
|
||||
t.Errorf("selected disk %s has DiskType=%q, want \"\" (HardDriveType)", d.NodeID, d.DiskType)
|
||||
}
|
||||
}
|
||||
if result.SpilledToOtherDiskType {
|
||||
t.Fatalf("SpilledToOtherDiskType should be false when HDD pool matches preferred=hdd")
|
||||
}
|
||||
}
|
||||
|
||||
// Empty PreferredDiskType: pre-#9423 behavior, single pool, no spillover
|
||||
// flag regardless of disk-type mix.
|
||||
func TestSelectDestinations_EmptyPreferredDiskTypeKeepsPriorBehavior(t *testing.T) {
|
||||
disks := []*DiskCandidate{
|
||||
makeDisk("ssd-0", "r0", "ssd", 0),
|
||||
makeDisk("hdd-0", "r1", "", 0),
|
||||
makeDisk("hdd-1", "r2", "", 0),
|
||||
makeDisk("ssd-1", "r3", "ssd", 0),
|
||||
}
|
||||
|
||||
result, err := SelectDestinations(disks, newRequest(3, ""))
|
||||
if err != nil {
|
||||
t.Fatalf("SelectDestinations: %v", err)
|
||||
}
|
||||
if got := len(result.SelectedDisks); got != 3 {
|
||||
t.Fatalf("selected %d disks, want 3", got)
|
||||
}
|
||||
if result.SpilledToOtherDiskType {
|
||||
t.Fatalf("SpilledToOtherDiskType should never be set when PreferredDiskType is empty")
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,8 @@ package super_block
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
)
|
||||
|
||||
type ReplicaPlacement struct {
|
||||
@@ -77,3 +79,27 @@ func (rp *ReplicaPlacement) String() string {
|
||||
func (rp *ReplicaPlacement) GetCopyCount() int {
|
||||
return rp.DiffDataCenterCount + rp.DiffRackCount + rp.SameRackCount + 1
|
||||
}
|
||||
|
||||
// ResolveReplicaPlacement picks the EC shard replica placement constraint: an
|
||||
// explicit spec wins; otherwise the cluster default (typically the master's
|
||||
// configured default replication). A missing, invalid, or zero-replication value
|
||||
// yields nil, meaning even spread / no constraint. Shared by EC encode, repair,
|
||||
// and balance so the three resolve replica placement identically.
|
||||
func ResolveReplicaPlacement(explicitSpec, clusterDefault string) *ReplicaPlacement {
|
||||
spec := explicitSpec
|
||||
if spec == "" {
|
||||
spec = clusterDefault
|
||||
}
|
||||
if spec == "" {
|
||||
return nil
|
||||
}
|
||||
rp, err := NewReplicaPlacementFromString(spec)
|
||||
if err != nil {
|
||||
glog.Warningf("ignoring invalid replica placement %q: %v", spec, err)
|
||||
return nil
|
||||
}
|
||||
if !rp.HasReplication() {
|
||||
return nil
|
||||
}
|
||||
return rp
|
||||
}
|
||||
|
||||
@@ -242,22 +242,11 @@ func resolveECRatio(_ *types.ClusterInfo, _ string) (int, int) {
|
||||
// replication (matching the shell ec.balance default). A missing, invalid, or
|
||||
// zero-replication value yields nil, meaning even spread / no constraint.
|
||||
func resolveReplicaPlacement(ecConfig *Config, clusterInfo *types.ClusterInfo) *super_block.ReplicaPlacement {
|
||||
spec := ecConfig.ReplicaPlacement
|
||||
if spec == "" && clusterInfo != nil {
|
||||
spec = clusterInfo.DefaultReplicaPlacement
|
||||
clusterDefault := ""
|
||||
if clusterInfo != nil {
|
||||
clusterDefault = clusterInfo.DefaultReplicaPlacement
|
||||
}
|
||||
if spec == "" {
|
||||
return nil
|
||||
}
|
||||
rp, err := super_block.NewReplicaPlacementFromString(spec)
|
||||
if err != nil {
|
||||
glog.Warningf("EC balance: ignoring invalid replica placement %q: %v", spec, err)
|
||||
return nil
|
||||
}
|
||||
if !rp.HasReplication() {
|
||||
return nil
|
||||
}
|
||||
return rp
|
||||
return super_block.ResolveReplicaPlacement(ecConfig.ReplicaPlacement, clusterDefault)
|
||||
}
|
||||
|
||||
func normalizeECShardCounts(dataShards, parityShards int) (int, int) {
|
||||
|
||||
@@ -17,6 +17,7 @@ type Config struct {
|
||||
CollectionFilter string `json:"collection_filter"`
|
||||
MinSizeMB int `json:"min_size_mb"`
|
||||
PreferredTags []string `json:"preferred_tags"`
|
||||
ReplicaPlacement string `json:"replica_placement"` // e.g. "020"; empty falls back to the master default replication
|
||||
}
|
||||
|
||||
// NewDefaultConfig creates a new default erasure coding configuration
|
||||
@@ -157,6 +158,19 @@ func GetConfigSpec() base.ConfigSpec {
|
||||
InputType: "text",
|
||||
CSSClasses: "form-control",
|
||||
},
|
||||
{
|
||||
Name: "replica_placement",
|
||||
JSONName: "replica_placement",
|
||||
Type: config.FieldTypeString,
|
||||
DefaultValue: "",
|
||||
Required: false,
|
||||
DisplayName: "Replica Placement",
|
||||
Description: "EC shard replica placement constraint (e.g. 020)",
|
||||
HelpText: "Leave empty to use the master default replication. When set, the 2nd/3rd digits cap EC shards per rack and per node (best-effort during encode: relaxed rather than failing if the cluster can't satisfy them, then enforced by rebalancing). The 1st (data-center) digit is ignored for EC placement",
|
||||
Placeholder: "020",
|
||||
InputType: "text",
|
||||
CSSClasses: "form-control",
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -177,6 +191,7 @@ func (c *Config) ToTaskPolicy() *worker_pb.TaskPolicy {
|
||||
MinVolumeSizeMb: int32(c.MinSizeMB),
|
||||
CollectionFilter: c.CollectionFilter,
|
||||
PreferredTags: preferredTagsCopy,
|
||||
ReplicaPlacement: c.ReplicaPlacement,
|
||||
},
|
||||
},
|
||||
}
|
||||
@@ -200,6 +215,7 @@ func (c *Config) FromTaskPolicy(policy *worker_pb.TaskPolicy) error {
|
||||
c.MinSizeMB = int(ecConfig.MinVolumeSizeMb)
|
||||
c.CollectionFilter = ecConfig.CollectionFilter
|
||||
c.PreferredTags = append([]string(nil), ecConfig.PreferredTags...)
|
||||
c.ReplicaPlacement = ecConfig.ReplicaPlacement
|
||||
}
|
||||
|
||||
return nil
|
||||
|
||||
@@ -14,8 +14,8 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding/placement"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding/ecbalancer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/wildcard"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/base"
|
||||
workerutil "github.com/seaweedfs/seaweedfs/weed/worker/tasks/util"
|
||||
@@ -56,7 +56,17 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
skippedTooFewNodes := 0
|
||||
consecutivePlanningFailures := 0
|
||||
|
||||
var planner *ecPlacementPlanner
|
||||
// EC shard replica placement: explicit config wins, else the master default.
|
||||
var replicaPlacement *super_block.ReplicaPlacement
|
||||
if clusterInfo != nil {
|
||||
replicaPlacement = super_block.ResolveReplicaPlacement(ecConfig.ReplicaPlacement, clusterInfo.DefaultReplicaPlacement)
|
||||
}
|
||||
// EC placement honors only the rack/node digits; the data-center digit can't
|
||||
// express a useful per-DC EC shard cap (it maxes at 2). Warn once per cycle so a
|
||||
// 1xx/2xx setting isn't silently ineffective.
|
||||
if replicaPlacement != nil && replicaPlacement.DiffDataCenterCount > 0 {
|
||||
glog.Warningf("EC Detection: replica placement data-center digit (%d) is ignored for EC; only rack/node digits are honored", replicaPlacement.DiffDataCenterCount)
|
||||
}
|
||||
|
||||
allowedCollections := wildcard.CompileWildcardMatchers(ecConfig.CollectionFilter)
|
||||
|
||||
@@ -219,12 +229,9 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
}
|
||||
|
||||
glog.Infof("EC Detection: ActiveTopology available, planning destinations for volume %d", metric.VolumeID)
|
||||
if planner == nil {
|
||||
planner = newECPlacementPlanner(clusterInfo.ActiveTopology, ecConfig.PreferredTags)
|
||||
}
|
||||
dataShards := erasure_coding.DataShardsCount
|
||||
parityShards := erasure_coding.ParityShardsCount
|
||||
multiPlan, err := planECDestinations(planner, metric, ecConfig, dataShards, parityShards)
|
||||
multiPlan, shardsPerPlan, err := planECDestinations(clusterInfo.ActiveTopology, metric, ecConfig, replicaPlacement, dataShards, parityShards)
|
||||
if err != nil {
|
||||
glog.V(2).Infof("Failed to plan EC destinations for volume %d: %v", metric.VolumeID, err)
|
||||
consecutivePlanningFailures++
|
||||
@@ -304,14 +311,13 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
glog.V(2).Infof("Found %d volume replicas and %d existing EC shards for volume %d (total %d cleanup sources)",
|
||||
len(replicaLocations), len(existingECShards), metric.VolumeID, len(sources))
|
||||
|
||||
// Convert shard destinations to TaskDestinationSpec. With fewer
|
||||
// disks than shards a destination holds several shards, so reserve
|
||||
// capacity for the actual per-disk shard count (round-robin matches
|
||||
// createECTargets) rather than assuming one shard each.
|
||||
// Convert shard destinations to TaskDestinationSpec. A destination may
|
||||
// hold several shards (small clusters), so reserve capacity for the
|
||||
// actual per-disk shard count that Place assigned (shardsPerPlan),
|
||||
// which is exactly what createECTargets writes.
|
||||
destinations := make([]topology.TaskDestinationSpec, len(shardDestinations))
|
||||
shardsPerDest := distributeECShards(dataShards+parityShards, len(shardDestinations))
|
||||
for i, dest := range shardDestinations {
|
||||
shardCount := len(shardsPerDest[i])
|
||||
shardCount := len(shardsPerPlan[i])
|
||||
shardImpact := topology.CalculateECShardStorageImpact(int32(shardCount), int64(expectedShardSize))
|
||||
destSize := int64(expectedShardSize) * int64(shardCount)
|
||||
destinations[i] = topology.TaskDestinationSpec{
|
||||
@@ -342,9 +348,9 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
continue // Skip this volume if topology task addition fails
|
||||
}
|
||||
|
||||
if planner != nil {
|
||||
planner.applyTaskReservations(int64(metric.Size), sources, destinations)
|
||||
}
|
||||
// Cross-volume in-cycle capacity is tracked by ActiveTopology via the
|
||||
// pending task above, which the next volume's FromActiveTopology snapshot
|
||||
// reflects; no separate planner reservation is needed.
|
||||
|
||||
glog.V(2).Infof("Added pending EC shard task %s to ActiveTopology for volume %d with %d cleanup sources and %d shard destinations",
|
||||
taskID, metric.VolumeID, len(sources), len(multiPlan.Plans))
|
||||
@@ -360,7 +366,7 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
Sources: sourcesProto,
|
||||
|
||||
// Unified targets - all EC shard destinations
|
||||
Targets: createECTargets(multiPlan, dataShards, parityShards),
|
||||
Targets: createECTargets(multiPlan, shardsPerPlan),
|
||||
|
||||
TaskParams: &worker_pb.TaskParams_ErasureCodingParams{
|
||||
ErasureCodingParams: createECTaskParams(dataShards, parityShards, metric.DiskType),
|
||||
@@ -413,273 +419,6 @@ func Detection(ctx context.Context, metrics []*types.VolumeHealthMetrics, cluste
|
||||
return results, hasMore, nil
|
||||
}
|
||||
|
||||
type ecDiskState struct {
|
||||
baseAvailable int64
|
||||
reservedVolumes int32
|
||||
reservedShardSlots int32
|
||||
}
|
||||
|
||||
type ecPlacementPlanner struct {
|
||||
activeTopology *topology.ActiveTopology
|
||||
candidates []*placement.DiskCandidate
|
||||
candidateByKey map[string]*placement.DiskCandidate
|
||||
diskStates map[string]*ecDiskState
|
||||
diskTags map[string][]string
|
||||
preferredTags []string
|
||||
}
|
||||
|
||||
func newECPlacementPlanner(activeTopology *topology.ActiveTopology, preferredTags []string) *ecPlacementPlanner {
|
||||
if activeTopology == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
disks := activeTopology.GetDisksWithEffectiveCapacity(topology.TaskTypeErasureCoding, "", 0)
|
||||
candidates := diskInfosToCandidates(disks)
|
||||
tagsByKey := collectDiskTags(disks)
|
||||
normalizedPreferredTags := util.NormalizeTagList(preferredTags)
|
||||
if len(candidates) == 0 {
|
||||
return &ecPlacementPlanner{
|
||||
activeTopology: activeTopology,
|
||||
candidates: candidates,
|
||||
candidateByKey: map[string]*placement.DiskCandidate{},
|
||||
diskStates: map[string]*ecDiskState{},
|
||||
diskTags: tagsByKey,
|
||||
preferredTags: normalizedPreferredTags,
|
||||
}
|
||||
}
|
||||
|
||||
candidateByKey := make(map[string]*placement.DiskCandidate, len(candidates))
|
||||
diskStates := make(map[string]*ecDiskState, len(candidates))
|
||||
for _, candidate := range candidates {
|
||||
key := ecDiskKey(candidate.NodeID, candidate.DiskID)
|
||||
candidateByKey[key] = candidate
|
||||
diskStates[key] = &ecDiskState{
|
||||
baseAvailable: int64(candidate.FreeSlots),
|
||||
}
|
||||
}
|
||||
|
||||
return &ecPlacementPlanner{
|
||||
activeTopology: activeTopology,
|
||||
candidates: candidates,
|
||||
candidateByKey: candidateByKey,
|
||||
diskStates: diskStates,
|
||||
diskTags: tagsByKey,
|
||||
preferredTags: normalizedPreferredTags,
|
||||
}
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) selectDestinations(sourceRack, sourceDC, sourceDiskType string, shardsNeeded int) ([]*placement.DiskCandidate, error) {
|
||||
if p == nil || p.activeTopology == nil {
|
||||
return nil, fmt.Errorf("ec placement planner is not initialized")
|
||||
}
|
||||
if shardsNeeded <= 0 {
|
||||
return nil, fmt.Errorf("invalid shardsNeeded %d", shardsNeeded)
|
||||
}
|
||||
|
||||
config := placement.PlacementRequest{
|
||||
ShardsNeeded: shardsNeeded,
|
||||
MaxShardsPerServer: 0,
|
||||
MaxShardsPerRack: 0,
|
||||
MaxTaskLoad: topology.MaxTaskLoadForECPlacement,
|
||||
PreferDifferentServers: true,
|
||||
PreferDifferentRacks: true,
|
||||
// Bias placement toward disks matching the source volume's disk
|
||||
// type; placement spills to other types only if the preferred
|
||||
// pool can't satisfy ShardsNeeded (#9423).
|
||||
PreferredDiskType: sourceDiskType,
|
||||
}
|
||||
|
||||
var lastErr error
|
||||
for _, candidates := range p.buildCandidateSets(shardsNeeded) {
|
||||
if len(candidates) == 0 {
|
||||
continue
|
||||
}
|
||||
result, err := placement.SelectDestinations(candidates, config)
|
||||
if err == nil {
|
||||
if result.SpilledToOtherDiskType {
|
||||
glog.Warningf("EC placement spilled to disks outside preferred disk type %q to reach %d shards (source rack=%s dc=%s)",
|
||||
sourceDiskType, shardsNeeded, sourceRack, sourceDC)
|
||||
}
|
||||
return result.SelectedDisks, nil
|
||||
}
|
||||
lastErr = err
|
||||
}
|
||||
if lastErr == nil {
|
||||
lastErr = fmt.Errorf("no EC placement candidates available")
|
||||
}
|
||||
return nil, lastErr
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) applyTaskReservations(volumeSize int64, sources []topology.TaskSourceSpec, destinations []topology.TaskDestinationSpec) {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
touched := make(map[string]bool)
|
||||
|
||||
for _, source := range sources {
|
||||
impact := p.sourceImpact(source, volumeSize)
|
||||
p.applyImpact(source.ServerID, source.DiskID, impact)
|
||||
p.bumpShardCount(source.ServerID, source.DiskID, impact.ShardSlots)
|
||||
key := ecDiskKey(source.ServerID, source.DiskID)
|
||||
if !touched[key] {
|
||||
p.bumpLoad(source.ServerID, source.DiskID)
|
||||
touched[key] = true
|
||||
}
|
||||
}
|
||||
|
||||
for _, dest := range destinations {
|
||||
impact := p.destinationImpact(dest, volumeSize)
|
||||
p.applyImpact(dest.ServerID, dest.DiskID, impact)
|
||||
p.bumpShardCount(dest.ServerID, dest.DiskID, impact.ShardSlots)
|
||||
key := ecDiskKey(dest.ServerID, dest.DiskID)
|
||||
if !touched[key] {
|
||||
p.bumpLoad(dest.ServerID, dest.DiskID)
|
||||
touched[key] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) sourceImpact(source topology.TaskSourceSpec, volumeSize int64) topology.StorageSlotChange {
|
||||
if source.StorageImpact != nil {
|
||||
return *source.StorageImpact
|
||||
}
|
||||
if source.CleanupType == topology.CleanupECShards {
|
||||
return topology.CalculateECShardCleanupImpact(volumeSize)
|
||||
}
|
||||
impact, _ := topology.CalculateTaskStorageImpact(topology.TaskTypeErasureCoding, volumeSize)
|
||||
return impact
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) destinationImpact(dest topology.TaskDestinationSpec, volumeSize int64) topology.StorageSlotChange {
|
||||
if dest.StorageImpact != nil {
|
||||
return *dest.StorageImpact
|
||||
}
|
||||
_, impact := topology.CalculateTaskStorageImpact(topology.TaskTypeErasureCoding, volumeSize)
|
||||
return impact
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) applyImpact(nodeID string, diskID uint32, impact topology.StorageSlotChange) {
|
||||
if impact.IsZero() {
|
||||
return
|
||||
}
|
||||
key := ecDiskKey(nodeID, diskID)
|
||||
state, ok := p.diskStates[key]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
|
||||
state.reservedVolumes += impact.VolumeSlots
|
||||
state.reservedShardSlots += impact.ShardSlots
|
||||
|
||||
available := state.baseAvailable - int64(state.reservedVolumes) - int64(state.reservedShardSlots)/int64(topology.ShardsPerVolumeSlot)
|
||||
if available < 0 {
|
||||
available = 0
|
||||
}
|
||||
|
||||
if candidate, ok := p.candidateByKey[key]; ok {
|
||||
candidate.FreeSlots = int(available)
|
||||
candidate.VolumeCount = candidate.MaxVolumeCount - available
|
||||
}
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) bumpLoad(nodeID string, diskID uint32) {
|
||||
key := ecDiskKey(nodeID, diskID)
|
||||
if candidate, ok := p.candidateByKey[key]; ok {
|
||||
candidate.LoadCount++
|
||||
}
|
||||
}
|
||||
|
||||
func (p *ecPlacementPlanner) bumpShardCount(nodeID string, diskID uint32, delta int32) {
|
||||
if delta == 0 {
|
||||
return
|
||||
}
|
||||
key := ecDiskKey(nodeID, diskID)
|
||||
if candidate, ok := p.candidateByKey[key]; ok {
|
||||
candidate.ShardCount += int(delta)
|
||||
if candidate.ShardCount < 0 {
|
||||
candidate.ShardCount = 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func ecDiskKey(nodeID string, diskID uint32) string {
|
||||
return fmt.Sprintf("%s:%d", nodeID, diskID)
|
||||
}
|
||||
|
||||
func collectDiskTags(disks []*topology.DiskInfo) map[string][]string {
|
||||
tagMap := make(map[string][]string, len(disks))
|
||||
for _, disk := range disks {
|
||||
if disk == nil || disk.DiskInfo == nil {
|
||||
continue
|
||||
}
|
||||
key := ecDiskKey(disk.NodeID, disk.DiskID)
|
||||
tags := util.NormalizeTagList(disk.DiskInfo.Tags)
|
||||
if len(tags) > 0 {
|
||||
tagMap[key] = tags
|
||||
}
|
||||
}
|
||||
return tagMap
|
||||
}
|
||||
|
||||
func diskHasTag(tags []string, tag string) bool {
|
||||
if tag == "" || len(tags) == 0 {
|
||||
return false
|
||||
}
|
||||
for _, candidate := range tags {
|
||||
if candidate == tag {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// buildCandidateSets builds tiered candidate sets for preferred-tag prioritized placement.
|
||||
// For a planner with preferredTags, it accumulates disks matching each tag in order into
|
||||
// progressively larger tiers. It emits a candidate set once a tier reaches shardsNeeded,
|
||||
// then continues accumulating for subsequent tags. Finally, it falls back to the full
|
||||
// p.candidates set if preferred-tag tiers are insufficient. This ensures tagged disks
|
||||
// are selected first before falling back to all available candidates.
|
||||
func (p *ecPlacementPlanner) buildCandidateSets(shardsNeeded int) [][]*placement.DiskCandidate {
|
||||
if p == nil {
|
||||
return nil
|
||||
}
|
||||
if len(p.preferredTags) == 0 {
|
||||
return [][]*placement.DiskCandidate{p.candidates}
|
||||
}
|
||||
selected := make(map[string]bool, len(p.candidates))
|
||||
var tier []*placement.DiskCandidate
|
||||
var candidateSets [][]*placement.DiskCandidate
|
||||
for _, tag := range p.preferredTags {
|
||||
for _, candidate := range p.candidates {
|
||||
key := ecDiskKey(candidate.NodeID, candidate.DiskID)
|
||||
if selected[key] {
|
||||
continue
|
||||
}
|
||||
if diskHasTag(p.diskTags[key], tag) {
|
||||
selected[key] = true
|
||||
tier = append(tier, candidate)
|
||||
}
|
||||
}
|
||||
if shardsNeeded > 0 && len(tier) >= shardsNeeded {
|
||||
candidateSets = append(candidateSets, append([]*placement.DiskCandidate(nil), tier...))
|
||||
}
|
||||
}
|
||||
// Defensive check: selectDestinations always ensures shardsNeeded > 0 before calling
|
||||
// buildCandidateSets, but this branch handles direct callers and edge cases.
|
||||
if shardsNeeded <= 0 && len(tier) > 0 {
|
||||
candidateSets = append(candidateSets, append([]*placement.DiskCandidate(nil), tier...))
|
||||
}
|
||||
if len(tier) < len(p.candidates) {
|
||||
candidateSets = append(candidateSets, p.candidates)
|
||||
} else if len(candidateSets) == 0 {
|
||||
candidateSets = append(candidateSets, p.candidates)
|
||||
}
|
||||
return candidateSets
|
||||
}
|
||||
|
||||
// planECDestinations plans the destinations for erasure coding operation.
|
||||
// dataShards/parityShards are parameters so callers can drive non-10+4 ratios.
|
||||
// countTopologyNodes counts volume-server nodes in the active topology, used by
|
||||
// the min-node safety gate.
|
||||
func countTopologyNodes(at *topology.ActiveTopology) int {
|
||||
@@ -699,12 +438,20 @@ func countTopologyNodes(at *topology.ActiveTopology) int {
|
||||
return n
|
||||
}
|
||||
|
||||
func planECDestinations(planner *ecPlacementPlanner, metric *types.VolumeHealthMetrics, ecConfig *Config, dataShards, parityShards int) (*topology.MultiDestinationPlan, error) {
|
||||
if planner == nil || planner.activeTopology == nil {
|
||||
return nil, fmt.Errorf("active topology not available for EC placement")
|
||||
// planECDestinations places all shards of the volume via the shared ecbalancer
|
||||
// policy and returns the per-disk destination plans plus, parallel to them, the
|
||||
// shard ids ecbalancer.Place assigned to each disk (so createECTargets and the
|
||||
// capacity reservations use the real assignment, not a round-robin guess).
|
||||
//
|
||||
// Encode is lenient (PlaceDurabilityFirst): it relaxes caps/anti-affinity/RP as
|
||||
// needed rather than fail, and prefers the source disk type but spills if that
|
||||
// type can't hold every shard. rp is the resolved replica placement (may be nil).
|
||||
func planECDestinations(at *topology.ActiveTopology, metric *types.VolumeHealthMetrics, ecConfig *Config, rp *super_block.ReplicaPlacement, dataShards, parityShards int) (*topology.MultiDestinationPlan, [][]uint32, error) {
|
||||
if at == nil {
|
||||
return nil, nil, fmt.Errorf("active topology not available for EC placement")
|
||||
}
|
||||
if dataShards <= 0 || parityShards <= 0 {
|
||||
return nil, fmt.Errorf("invalid EC ratio: dataShards=%d parityShards=%d", dataShards, parityShards)
|
||||
return nil, nil, fmt.Errorf("invalid EC ratio: dataShards=%d parityShards=%d", dataShards, parityShards)
|
||||
}
|
||||
totalShards := dataShards + parityShards
|
||||
// Survive losing one disk: each disk holds at most parityShards shards,
|
||||
@@ -712,159 +459,121 @@ func planECDestinations(planner *ecPlacementPlanner, metric *types.VolumeHealthM
|
||||
minTotalDisks := (totalShards + parityShards - 1) / parityShards
|
||||
expectedShardSize := uint64(metric.Size) / uint64(dataShards)
|
||||
|
||||
// Get source node information from topology
|
||||
var sourceRack, sourceDC string
|
||||
|
||||
// Extract rack and DC from topology info
|
||||
topologyInfo := planner.activeTopology.GetTopologyInfo()
|
||||
if topologyInfo != nil {
|
||||
for _, dc := range topologyInfo.DataCenterInfos {
|
||||
for _, rack := range dc.RackInfos {
|
||||
for _, dataNodeInfo := range rack.DataNodeInfos {
|
||||
if dataNodeInfo.Id == metric.Server {
|
||||
sourceDC = dc.Id
|
||||
sourceRack = rack.Id
|
||||
break
|
||||
}
|
||||
}
|
||||
if sourceRack != "" {
|
||||
break
|
||||
}
|
||||
}
|
||||
if sourceDC != "" {
|
||||
break
|
||||
}
|
||||
}
|
||||
snap := ecbalancer.FromActiveTopology(at, dataShards)
|
||||
// Encode is greenfield: any EC shards already present for this volume are stale
|
||||
// leftovers from a prior failed attempt, which the task deletes
|
||||
// (cleanupStaleEcShards) before distributing the new shards. Release them so they
|
||||
// don't occupy capacity or skew anti-affinity / per-disk caps during planning.
|
||||
snap.ReleaseVolumeShards(metric.Collection, metric.VolumeID)
|
||||
need := make([]int, totalShards)
|
||||
for i := range need {
|
||||
need[i] = i
|
||||
}
|
||||
|
||||
// Select best disks for EC placement with rack/DC diversity using the cached planner.
|
||||
// Pass source disk type so placement prefers matching-type disks (#9423).
|
||||
selectedDisks, err := planner.selectDestinations(sourceRack, sourceDC, metric.DiskType, totalShards)
|
||||
res, err := snap.Place(metric.VolumeID, metric.Collection, need, ecbalancer.Constraints{
|
||||
DiskType: metric.DiskType,
|
||||
DiskTypePolicy: ecbalancer.DiskTypePrefer,
|
||||
PreferredTags: ecConfig.PreferredTags,
|
||||
ReplicaPlacement: rp,
|
||||
Ratio: func(string) (int, int) { return dataShards, parityShards },
|
||||
}, ecbalancer.PlaceDurabilityFirst)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
if len(selectedDisks) < minTotalDisks {
|
||||
return nil, fmt.Errorf("found %d disks, but EC %d+%d needs at least %d disks so no disk holds more than %d shards",
|
||||
len(selectedDisks), dataShards, parityShards, minTotalDisks, parityShards)
|
||||
if res.SpilledToOtherDiskType {
|
||||
glog.Warningf("EC volume %d: placed shards outside preferred disk type %q", metric.VolumeID, metric.DiskType)
|
||||
}
|
||||
// Fewer than totalShards disks is fine: createECTargets round-robins the
|
||||
// shards across the available disks, packing several distinct shards onto a
|
||||
// disk when needed (matching ec.encode's "spread as 4,4,3,3" fallback for
|
||||
// small clusters). A disk holding several shards of one volume is safe —
|
||||
// each is a separate .ecNN file and ReceiveFile keys by that extension. The
|
||||
// minTotalDisks floor above keeps any single disk under parityShards shards,
|
||||
// so the volume still survives losing any one disk.
|
||||
if len(selectedDisks) < totalShards {
|
||||
glog.V(1).Infof("EC volume %d: only %d disks for %d shards, packing up to %d shards per disk",
|
||||
metric.VolumeID, len(selectedDisks), totalShards, (totalShards+len(selectedDisks)-1)/len(selectedDisks))
|
||||
if res.SpilledOutsidePreferredTags {
|
||||
glog.Warningf("EC volume %d: placed shards outside preferred tags %v", metric.VolumeID, ecConfig.PreferredTags)
|
||||
}
|
||||
if len(res.Relaxed) > 0 {
|
||||
// Encode is best-effort (PlaceDurabilityFirst): it relaxes these constraints
|
||||
// rather than defer when the cluster can't satisfy them. Surface it so a tight
|
||||
// replica placement isn't silently weakened; rebalancing tightens the spread.
|
||||
glog.Warningf("EC volume %d: placed with relaxed constraints %v; replica placement not fully satisfied (rebalancing will adjust)", metric.VolumeID, res.Relaxed)
|
||||
}
|
||||
|
||||
// Group the per-shard destinations into one plan per (node,disk), iterating
|
||||
// shard ids in order for determinism.
|
||||
type diskGroup struct {
|
||||
node, rack, dc string
|
||||
diskID uint32
|
||||
shards []uint32
|
||||
}
|
||||
type diskKey struct {
|
||||
node string
|
||||
diskID uint32
|
||||
}
|
||||
groups := make(map[diskKey]*diskGroup, totalShards)
|
||||
order := make([]diskKey, 0, totalShards)
|
||||
for sid := 0; sid < totalShards; sid++ {
|
||||
d, ok := res.Destinations[sid]
|
||||
if !ok {
|
||||
return nil, nil, fmt.Errorf("EC volume %d: shard %d was not placed", metric.VolumeID, sid)
|
||||
}
|
||||
key := diskKey{node: d.Node, diskID: d.DiskID}
|
||||
g := groups[key]
|
||||
if g == nil {
|
||||
g = &diskGroup{node: d.Node, rack: d.Rack, dc: d.DataCenter, diskID: d.DiskID}
|
||||
groups[key] = g
|
||||
order = append(order, key)
|
||||
}
|
||||
g.shards = append(g.shards, uint32(sid))
|
||||
}
|
||||
if len(order) < minTotalDisks {
|
||||
return nil, nil, fmt.Errorf("placed onto %d disks, but EC %d+%d needs at least %d so no disk holds more than %d shards",
|
||||
len(order), dataShards, parityShards, minTotalDisks, parityShards)
|
||||
}
|
||||
|
||||
var plans []*topology.DestinationPlan
|
||||
shardsPerPlan := make([][]uint32, 0, len(order))
|
||||
rackCount := make(map[string]int)
|
||||
dcCount := make(map[string]int)
|
||||
|
||||
for _, disk := range selectedDisks {
|
||||
// Get the target server address
|
||||
targetAddress, err := workerutil.ResolveServerAddress(disk.NodeID, planner.activeTopology)
|
||||
for _, key := range order {
|
||||
g := groups[key]
|
||||
targetAddress, err := workerutil.ResolveServerAddress(g.node, at)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to resolve address for target server %s: %v", disk.NodeID, err)
|
||||
return nil, nil, fmt.Errorf("failed to resolve address for target server %s: %v", g.node, err)
|
||||
}
|
||||
|
||||
plan := &topology.DestinationPlan{
|
||||
TargetNode: disk.NodeID,
|
||||
TargetAddress: targetAddress,
|
||||
TargetDisk: disk.DiskID,
|
||||
TargetRack: disk.Rack,
|
||||
TargetDC: disk.DataCenter,
|
||||
ExpectedSize: expectedShardSize, // Set calculated EC shard size
|
||||
PlacementScore: calculateECScoreCandidate(disk, sourceRack, sourceDC),
|
||||
}
|
||||
plans = append(plans, plan)
|
||||
|
||||
// Count rack and DC diversity
|
||||
rackKey := fmt.Sprintf("%s:%s", disk.DataCenter, disk.Rack)
|
||||
rackCount[rackKey]++
|
||||
dcCount[disk.DataCenter]++
|
||||
plans = append(plans, &topology.DestinationPlan{
|
||||
TargetNode: g.node,
|
||||
TargetAddress: targetAddress,
|
||||
TargetDisk: g.diskID,
|
||||
TargetRack: g.rack,
|
||||
TargetDC: g.dc,
|
||||
ExpectedSize: expectedShardSize,
|
||||
})
|
||||
shardsPerPlan = append(shardsPerPlan, g.shards)
|
||||
rackCount[fmt.Sprintf("%s:%s", g.dc, g.rack)]++
|
||||
dcCount[g.dc]++
|
||||
}
|
||||
|
||||
// Log capacity utilization information using ActiveTopology's encapsulated logic
|
||||
totalEffectiveCapacity := int64(0)
|
||||
for _, plan := range plans {
|
||||
key := ecDiskKey(plan.TargetNode, plan.TargetDisk)
|
||||
if candidate, ok := planner.candidateByKey[key]; ok {
|
||||
totalEffectiveCapacity += int64(candidate.FreeSlots)
|
||||
}
|
||||
}
|
||||
|
||||
glog.V(1).Infof("Planned EC destinations for volume %d (size=%d bytes): expected shard size=%d bytes, %d shards across %d racks, %d DCs, total effective capacity=%d slots",
|
||||
metric.VolumeID, metric.Size, expectedShardSize, len(plans), len(rackCount), len(dcCount), totalEffectiveCapacity)
|
||||
|
||||
// Log storage impact for EC task (source only - EC has multiple targets handled individually)
|
||||
sourceChange, _ := topology.CalculateTaskStorageImpact(topology.TaskTypeErasureCoding, int64(metric.Size))
|
||||
glog.V(2).Infof("EC task capacity management: source_reserves_with_zero_impact={VolumeSlots:%d, ShardSlots:%d}, %d_targets_will_receive_shards, estimated_size=%d",
|
||||
sourceChange.VolumeSlots, sourceChange.ShardSlots, len(plans), metric.Size)
|
||||
glog.V(2).Infof("EC source reserves capacity but with zero StorageSlotChange impact")
|
||||
glog.V(1).Infof("Planned EC destinations for volume %d (size=%d bytes): expected shard size=%d bytes, %d shards across %d disks, %d racks, %d DCs",
|
||||
metric.VolumeID, metric.Size, expectedShardSize, totalShards, len(plans), len(rackCount), len(dcCount))
|
||||
|
||||
return &topology.MultiDestinationPlan{
|
||||
Plans: plans,
|
||||
TotalShards: len(plans),
|
||||
TotalShards: totalShards,
|
||||
SuccessfulRack: len(rackCount),
|
||||
SuccessfulDCs: len(dcCount),
|
||||
}, nil
|
||||
}, shardsPerPlan, nil
|
||||
}
|
||||
|
||||
// distributeECShards assigns shard ids 0..totalShards-1 across numTargets
|
||||
// targets round-robin, so each target holds either floor or ceil of
|
||||
// totalShards/numTargets shards. When numTargets < totalShards this packs
|
||||
// several shards onto a target; planECDestinations guarantees numTargets is at
|
||||
// least ceil(totalShards/parityShards), so no target exceeds parityShards shards.
|
||||
func distributeECShards(totalShards, numTargets int) [][]uint32 {
|
||||
targetShards := make([][]uint32, numTargets)
|
||||
for i := range targetShards {
|
||||
targetShards[i] = make([]uint32, 0)
|
||||
}
|
||||
for shardId := 0; shardId < totalShards; shardId++ {
|
||||
targetIndex := shardId % numTargets
|
||||
targetShards[targetIndex] = append(targetShards[targetIndex], uint32(shardId))
|
||||
}
|
||||
return targetShards
|
||||
}
|
||||
|
||||
// createECTargets builds TaskTargets, round-robining shards across the plan
|
||||
// entries. With fewer disks than shards a target receives several shard ids.
|
||||
func createECTargets(multiPlan *topology.MultiDestinationPlan, dataShards, parityShards int) []*worker_pb.TaskTarget {
|
||||
var targets []*worker_pb.TaskTarget
|
||||
numTargets := len(multiPlan.Plans)
|
||||
totalShards := dataShards + parityShards
|
||||
|
||||
targetShards := distributeECShards(totalShards, numTargets)
|
||||
|
||||
// createECTargets builds TaskTargets from the per-disk plans and the shard ids
|
||||
// ecbalancer.Place assigned to each (shardsPerPlan is parallel to multiPlan.Plans).
|
||||
func createECTargets(multiPlan *topology.MultiDestinationPlan, shardsPerPlan [][]uint32) []*worker_pb.TaskTarget {
|
||||
targets := make([]*worker_pb.TaskTarget, 0, len(multiPlan.Plans))
|
||||
for i, plan := range multiPlan.Plans {
|
||||
target := &worker_pb.TaskTarget{
|
||||
shardIDs := shardsPerPlan[i]
|
||||
targets = append(targets, &worker_pb.TaskTarget{
|
||||
Node: plan.TargetAddress,
|
||||
DiskId: plan.TargetDisk,
|
||||
Rack: plan.TargetRack,
|
||||
DataCenter: plan.TargetDC,
|
||||
ShardIds: targetShards[i],
|
||||
ShardIds: shardIDs,
|
||||
EstimatedSize: plan.ExpectedSize,
|
||||
}
|
||||
targets = append(targets, target)
|
||||
|
||||
assignedData := make([]uint32, 0)
|
||||
assignedParity := make([]uint32, 0)
|
||||
for _, shardId := range targetShards[i] {
|
||||
if int(shardId) < dataShards {
|
||||
assignedData = append(assignedData, shardId)
|
||||
} else {
|
||||
assignedParity = append(assignedParity, shardId)
|
||||
}
|
||||
}
|
||||
glog.V(2).Infof("EC planning: target %s assigned shards %v (data: %v, parity: %v)",
|
||||
plan.TargetNode, targetShards[i], assignedData, assignedParity)
|
||||
})
|
||||
glog.V(2).Infof("EC planning: target %s disk %d assigned shards %v", plan.TargetNode, plan.TargetDisk, shardIDs)
|
||||
}
|
||||
|
||||
glog.V(1).Infof("EC planning: distributed %d shards across %d targets using round-robin (data shards 0-%d, parity shards %d-%d)",
|
||||
totalShards, numTargets, dataShards-1, dataShards, totalShards-1)
|
||||
return targets
|
||||
}
|
||||
|
||||
@@ -918,68 +627,6 @@ func createECTaskParams(dataShards, parityShards int, sourceDiskType string) *wo
|
||||
}
|
||||
}
|
||||
|
||||
// diskInfosToCandidates converts topology.DiskInfo slice to placement.DiskCandidate slice
|
||||
func diskInfosToCandidates(disks []*topology.DiskInfo) []*placement.DiskCandidate {
|
||||
var candidates []*placement.DiskCandidate
|
||||
for _, disk := range disks {
|
||||
if disk.DiskInfo == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
// Calculate free slots (using default max if not set)
|
||||
freeSlots := int(disk.DiskInfo.MaxVolumeCount - disk.DiskInfo.VolumeCount)
|
||||
if freeSlots < 0 {
|
||||
freeSlots = 0
|
||||
}
|
||||
|
||||
// Calculate EC shard count for this specific disk
|
||||
// EcShardInfos contains all shards, so we need to filter by DiskId and sum actual shard counts
|
||||
ecShardCount := 0
|
||||
if disk.DiskInfo.EcShardInfos != nil {
|
||||
for _, shardInfo := range disk.DiskInfo.EcShardInfos {
|
||||
if shardInfo.DiskId == disk.DiskID {
|
||||
ecShardCount += erasure_coding.GetShardCount(shardInfo)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
candidates = append(candidates, &placement.DiskCandidate{
|
||||
NodeID: disk.NodeID,
|
||||
DiskID: disk.DiskID,
|
||||
DataCenter: disk.DataCenter,
|
||||
Rack: disk.Rack,
|
||||
DiskType: disk.DiskType,
|
||||
VolumeCount: disk.DiskInfo.VolumeCount,
|
||||
MaxVolumeCount: disk.DiskInfo.MaxVolumeCount,
|
||||
ShardCount: ecShardCount,
|
||||
FreeSlots: freeSlots,
|
||||
LoadCount: disk.LoadCount,
|
||||
})
|
||||
}
|
||||
return candidates
|
||||
}
|
||||
|
||||
// calculateECScoreCandidate calculates placement score for EC operations.
|
||||
// Used for logging and plan metadata.
|
||||
func calculateECScoreCandidate(disk *placement.DiskCandidate, sourceRack, sourceDC string) float64 {
|
||||
if disk == nil {
|
||||
return 0.0
|
||||
}
|
||||
|
||||
score := 0.0
|
||||
|
||||
// Prefer disks with available capacity (primary factor)
|
||||
if disk.MaxVolumeCount > 0 {
|
||||
utilization := float64(disk.VolumeCount) / float64(disk.MaxVolumeCount)
|
||||
score += (1.0 - utilization) * 60.0 // Up to 60 points for available capacity
|
||||
}
|
||||
|
||||
// Consider current load (secondary factor)
|
||||
score += (10.0 - float64(disk.LoadCount)) // Up to 10 points for low load
|
||||
|
||||
return score
|
||||
}
|
||||
|
||||
// findVolumeReplicaLocations finds all replica locations (server + disk) for the specified volume
|
||||
// Uses O(1) indexed lookup for optimal performance on large clusters.
|
||||
func findVolumeReplicaLocations(activeTopology *topology.ActiveTopology, volumeID uint32, collection string) []topology.VolumeReplica {
|
||||
|
||||
@@ -19,9 +19,6 @@ func TestPlanECDestinationsPrefersSourceDiskType_FullCluster(t *testing.T) {
|
||||
// for a 10+4 layout with one-shard-per-(server,disk) diversity.
|
||||
activeTopology := buildActiveTopology(t, erasure_coding.TotalShardsCount, []string{"hdd", "ssd"}, 100, 0)
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 1,
|
||||
Server: "10.0.0.1:8080",
|
||||
@@ -30,7 +27,7 @@ func TestPlanECDestinationsPrefersSourceDiskType_FullCluster(t *testing.T) {
|
||||
DiskType: "ssd", // the property being plumbed end-to-end
|
||||
}
|
||||
|
||||
plan, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
plan, _, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, plan.Plans, erasure_coding.TotalShardsCount)
|
||||
|
||||
@@ -67,9 +64,6 @@ func TestPlanECDestinationsSpillsToOtherDiskType_WhenPreferredScarce(t *testing.
|
||||
}
|
||||
require.NoError(t, activeTopology.UpdateTopology(topo))
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 2,
|
||||
Server: "10.0.0.1:8080",
|
||||
@@ -78,7 +72,7 @@ func TestPlanECDestinationsSpillsToOtherDiskType_WhenPreferredScarce(t *testing.
|
||||
DiskType: "ssd",
|
||||
}
|
||||
|
||||
plan, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
plan, _, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, plan.Plans, erasure_coding.TotalShardsCount)
|
||||
|
||||
|
||||
@@ -14,41 +14,8 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestECPlacementPlannerApplyReservations(t *testing.T) {
|
||||
activeTopology := buildActiveTopology(t, 1, []string{"hdd"}, 10, 0)
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
key := ecDiskKey("10.0.0.1:8080", 0)
|
||||
candidate, ok := planner.candidateByKey[key]
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 10, candidate.FreeSlots)
|
||||
assert.Equal(t, 0, candidate.ShardCount)
|
||||
assert.Equal(t, 0, candidate.LoadCount)
|
||||
|
||||
shardImpact := topology.CalculateECShardStorageImpact(1, 1)
|
||||
destinations := make([]topology.TaskDestinationSpec, 10)
|
||||
for i := 0; i < 10; i++ {
|
||||
destinations[i] = topology.TaskDestinationSpec{
|
||||
ServerID: "10.0.0.1:8080",
|
||||
DiskID: 0,
|
||||
StorageImpact: &shardImpact,
|
||||
}
|
||||
}
|
||||
|
||||
planner.applyTaskReservations(1024, nil, destinations)
|
||||
|
||||
candidate = planner.candidateByKey[key]
|
||||
assert.Equal(t, 9, candidate.FreeSlots, "10 shard slots should reduce available volume slots by 1")
|
||||
assert.Equal(t, 10, candidate.ShardCount)
|
||||
assert.Equal(t, 1, candidate.LoadCount, "load should only be incremented once per disk")
|
||||
}
|
||||
|
||||
func TestPlanECDestinationsUsesPlanner(t *testing.T) {
|
||||
activeTopology := buildActiveTopology(t, 7, []string{"hdd", "ssd"}, 100, 0)
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 1,
|
||||
@@ -57,74 +24,10 @@ func TestPlanECDestinationsUsesPlanner(t *testing.T) {
|
||||
Collection: "",
|
||||
}
|
||||
|
||||
plan, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
plan, shardsPerPlan, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, plan)
|
||||
assert.Equal(t, erasure_coding.TotalShardsCount, len(plan.Plans))
|
||||
}
|
||||
|
||||
func TestECPlacementPlannerPrefersTaggedDisks(t *testing.T) {
|
||||
activeTopology := buildActiveTopology(t, 3, []string{"hdd"}, 10, 0)
|
||||
topo := activeTopology.GetTopologyInfo()
|
||||
for _, dc := range topo.DataCenterInfos {
|
||||
for _, rack := range dc.RackInfos {
|
||||
for k, node := range rack.DataNodeInfos {
|
||||
for diskType := range node.DiskInfos {
|
||||
if k < 2 {
|
||||
node.DiskInfos[diskType].Tags = []string{"fast"}
|
||||
} else {
|
||||
node.DiskInfos[diskType].Tags = []string{"slow"}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
require.NoError(t, activeTopology.UpdateTopology(topo))
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, []string{"fast"})
|
||||
require.NotNil(t, planner)
|
||||
|
||||
selected, err := planner.selectDestinations("", "", "", 2)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, selected, 2)
|
||||
|
||||
for _, candidate := range selected {
|
||||
key := ecDiskKey(candidate.NodeID, candidate.DiskID)
|
||||
assert.True(t, diskHasTag(planner.diskTags[key], "fast"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestECPlacementPlannerFallsBackWhenTagsInsufficient(t *testing.T) {
|
||||
activeTopology := buildActiveTopology(t, 3, []string{"hdd"}, 10, 0)
|
||||
topo := activeTopology.GetTopologyInfo()
|
||||
for _, dc := range topo.DataCenterInfos {
|
||||
for _, rack := range dc.RackInfos {
|
||||
for i, node := range rack.DataNodeInfos {
|
||||
for diskType := range node.DiskInfos {
|
||||
if i == 0 {
|
||||
node.DiskInfos[diskType].Tags = []string{"fast"}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
require.NoError(t, activeTopology.UpdateTopology(topo))
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, []string{"fast"})
|
||||
require.NotNil(t, planner)
|
||||
|
||||
selected, err := planner.selectDestinations("", "", "", 3)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, selected, 3)
|
||||
|
||||
taggedCount := 0
|
||||
for _, candidate := range selected {
|
||||
key := ecDiskKey(candidate.NodeID, candidate.DiskID)
|
||||
if diskHasTag(planner.diskTags[key], "fast") {
|
||||
taggedCount++
|
||||
}
|
||||
}
|
||||
assert.Less(t, taggedCount, len(selected))
|
||||
requireAllShardsPlaced(t, plan, shardsPerPlan)
|
||||
}
|
||||
|
||||
// TestDetectionSkipsWhenECShardsAlreadyExist guards against issue #9448: a
|
||||
@@ -363,9 +266,6 @@ func TestPlanECDestinationsSpreadsAcrossPhysicalDisks(t *testing.T) {
|
||||
}},
|
||||
}))
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 42,
|
||||
Server: "127.0.0.1:8081",
|
||||
@@ -373,23 +273,14 @@ func TestPlanECDestinationsSpreadsAcrossPhysicalDisks(t *testing.T) {
|
||||
Collection: "",
|
||||
}
|
||||
|
||||
plan, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
plan, shardsPerPlan, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, plan)
|
||||
require.Equal(t, erasure_coding.TotalShardsCount, len(plan.Plans))
|
||||
|
||||
seen := make(map[string]bool, len(plan.Plans))
|
||||
for _, p := range plan.Plans {
|
||||
key := fmt.Sprintf("%s:%d", p.TargetNode, p.TargetDisk)
|
||||
assert.False(t, seen[key], "duplicate (server,disk_id) target %s", key)
|
||||
seen[key] = true
|
||||
}
|
||||
requireAllShardsPlaced(t, plan, shardsPerPlan)
|
||||
}
|
||||
|
||||
func TestPlanECDestinationsFailsWithInsufficientCapacity(t *testing.T) {
|
||||
activeTopology := buildActiveTopology(t, 1, []string{"hdd"}, 1, 1)
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 2,
|
||||
@@ -398,7 +289,7 @@ func TestPlanECDestinationsFailsWithInsufficientCapacity(t *testing.T) {
|
||||
Collection: "",
|
||||
}
|
||||
|
||||
_, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
_, _, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.Error(t, err)
|
||||
}
|
||||
|
||||
@@ -440,9 +331,6 @@ func TestPlanECDestinationsPacksWhenFewerDisksThanShards(t *testing.T) {
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{{Id: "dc1", RackInfos: rackInfos}},
|
||||
}))
|
||||
|
||||
planner := newECPlacementPlanner(activeTopology, nil)
|
||||
require.NotNil(t, planner)
|
||||
|
||||
metric := &types.VolumeHealthMetrics{
|
||||
VolumeID: 4569,
|
||||
Server: "192.168.1.145:8081",
|
||||
@@ -450,16 +338,18 @@ func TestPlanECDestinationsPacksWhenFewerDisksThanShards(t *testing.T) {
|
||||
Collection: "",
|
||||
}
|
||||
|
||||
plan, err := planECDestinations(planner, metric, NewDefaultConfig(), erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
plan, shardsPerPlan, err := planECDestinations(activeTopology, metric, NewDefaultConfig(), nil, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, plan)
|
||||
// One plan entry per available disk; fewer than the 14 shards.
|
||||
require.Equal(t, numServers, len(plan.Plans))
|
||||
// Packed onto the available disks: more than one shard per disk but never more
|
||||
// than the 8 disks, and at least the durability floor of distinct disks.
|
||||
require.LessOrEqual(t, len(plan.Plans), numServers)
|
||||
require.GreaterOrEqual(t, len(plan.Plans), (erasure_coding.TotalShardsCount+erasure_coding.ParityShardsCount-1)/erasure_coding.ParityShardsCount)
|
||||
|
||||
// createECTargets must cover all 14 shards exactly once, packing onto the
|
||||
// available disks without any disk exceeding parityShards shards.
|
||||
targets := createECTargets(plan, erasure_coding.DataShardsCount, erasure_coding.ParityShardsCount)
|
||||
require.Equal(t, numServers, len(targets))
|
||||
targets := createECTargets(plan, shardsPerPlan)
|
||||
require.Equal(t, len(plan.Plans), len(targets))
|
||||
|
||||
seenShards := make(map[uint32]bool)
|
||||
for _, target := range targets {
|
||||
@@ -473,6 +363,28 @@ func TestPlanECDestinationsPacksWhenFewerDisksThanShards(t *testing.T) {
|
||||
require.Len(t, seenShards, erasure_coding.TotalShardsCount, "every shard must be placed exactly once")
|
||||
}
|
||||
|
||||
// requireAllShardsPlaced asserts every EC shard landed exactly once, on a distinct
|
||||
// (node,disk) target, with no disk holding more than parityShards shards (so losing
|
||||
// any one disk cannot lose the volume). shardsPerPlan is parallel to plan.Plans.
|
||||
func requireAllShardsPlaced(t *testing.T, plan *topology.MultiDestinationPlan, shardsPerPlan [][]uint32) {
|
||||
t.Helper()
|
||||
require.Equal(t, len(plan.Plans), len(shardsPerPlan), "one shard list per plan entry")
|
||||
keys := make(map[string]bool, len(plan.Plans))
|
||||
seen := make(map[uint32]bool)
|
||||
for i, p := range plan.Plans {
|
||||
key := fmt.Sprintf("%s:%d", p.TargetNode, p.TargetDisk)
|
||||
require.False(t, keys[key], "duplicate (node,disk) target %s", key)
|
||||
keys[key] = true
|
||||
require.LessOrEqual(t, len(shardsPerPlan[i]), erasure_coding.ParityShardsCount,
|
||||
"disk %s holds %d shards, over parityShards", key, len(shardsPerPlan[i]))
|
||||
for _, s := range shardsPerPlan[i] {
|
||||
require.False(t, seen[s], "shard %d placed more than once", s)
|
||||
seen[s] = true
|
||||
}
|
||||
}
|
||||
require.Len(t, seen, erasure_coding.TotalShardsCount, "every shard must be placed exactly once")
|
||||
}
|
||||
|
||||
func buildVolumeMetricsForIDs(count int) []*types.VolumeHealthMetrics {
|
||||
metrics := make([]*types.VolumeHealthMetrics, 0, count)
|
||||
now := time.Now()
|
||||
|
||||
@@ -271,8 +271,17 @@ func (t *ErasureCodingTask) Validate(params *worker_pb.TaskParams) error {
|
||||
return fmt.Errorf("invalid parity shards: %d (must be >= 1)", ecParams.ParityShards)
|
||||
}
|
||||
|
||||
if len(params.Targets) < int(ecParams.DataShards+ecParams.ParityShards) {
|
||||
return fmt.Errorf("insufficient targets: got %d, need %d", len(params.Targets), ecParams.DataShards+ecParams.ParityShards)
|
||||
// Count distinct shard ids across targets, not target rows: Place packs several
|
||||
// shards onto one (node,disk) target when there are fewer disks than shards, so
|
||||
// a valid plan can have fewer target rows than total shards.
|
||||
distinctShards := make(map[uint32]struct{})
|
||||
for _, target := range params.Targets {
|
||||
for _, sid := range target.ShardIds {
|
||||
distinctShards[sid] = struct{}{}
|
||||
}
|
||||
}
|
||||
if total := int(ecParams.DataShards + ecParams.ParityShards); len(distinctShards) < total {
|
||||
return fmt.Errorf("insufficient shard targets: got %d distinct shards across %d targets, need %d", len(distinctShards), len(params.Targets), total)
|
||||
}
|
||||
|
||||
return nil
|
||||
|
||||
@@ -138,6 +138,14 @@ func (h *ErasureCodingHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
FieldType: plugin_pb.ConfigFieldType_CONFIG_FIELD_TYPE_STRING,
|
||||
Widget: plugin_pb.ConfigWidget_CONFIG_WIDGET_TEXT,
|
||||
},
|
||||
{
|
||||
Name: "replica_placement",
|
||||
Label: "Replica Placement",
|
||||
Description: "EC shard placement (e.g. 020): 2nd/3rd digits cap shards per rack/node (best-effort during encode, enforced by rebalancing); the data-center digit is ignored. Empty uses the master default.",
|
||||
Placeholder: "020",
|
||||
FieldType: plugin_pb.ConfigFieldType_CONFIG_FIELD_TYPE_STRING,
|
||||
Widget: plugin_pb.ConfigWidget_CONFIG_WIDGET_TEXT,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
@@ -154,6 +162,9 @@ func (h *ErasureCodingHandler) Descriptor() *plugin_pb.JobTypeDescriptor {
|
||||
"preferred_tags": {
|
||||
Kind: &plugin_pb.ConfigValue_StringValue{StringValue: ""},
|
||||
},
|
||||
"replica_placement": {
|
||||
Kind: &plugin_pb.ConfigValue_StringValue{StringValue: ""},
|
||||
},
|
||||
},
|
||||
},
|
||||
AdminRuntimeDefaults: &plugin_pb.AdminRuntimeDefaults{
|
||||
@@ -217,7 +228,11 @@ func (h *ErasureCodingHandler) Detect(
|
||||
return err
|
||||
}
|
||||
|
||||
clusterInfo := &workertypes.ClusterInfo{ActiveTopology: activeTopology, GrpcDialOption: h.grpcDialOption}
|
||||
clusterInfo := &workertypes.ClusterInfo{
|
||||
ActiveTopology: activeTopology,
|
||||
GrpcDialOption: h.grpcDialOption,
|
||||
DefaultReplicaPlacement: pluginworker.FetchDefaultReplicaPlacement(ctx, masters, h.grpcDialOption),
|
||||
}
|
||||
maxResults := int(request.MaxResults)
|
||||
if maxResults < 0 {
|
||||
maxResults = 0
|
||||
@@ -592,6 +607,8 @@ func deriveErasureCodingWorkerConfig(values map[string]*plugin_pb.ConfigValue) *
|
||||
|
||||
taskConfig.PreferredTags = util.NormalizeTagList(pluginworker.ReadStringListConfig(values, "preferred_tags"))
|
||||
|
||||
taskConfig.ReplicaPlacement = strings.TrimSpace(pluginworker.ReadStringConfig(values, "replica_placement", taskConfig.ReplicaPlacement))
|
||||
|
||||
return &erasureCodingWorkerConfig{
|
||||
TaskConfig: taskConfig,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user