mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-21 22:56:55 +00:00
balance: share replica-placement logic between shell and worker (#10169)
The replica-placement rule (data-center/rack/same-node limits plus host anti-affinity) existed three times: the shell's satisfyReplicaPlacement/isGoodMove used by volume.balance, fix.replication, and tier.move, and a line-for-line port in the maintenance balance worker. Move the canonical logic into weed/topology/balancer on a shared Location type; the shell and worker keep thin adapters that convert their own location representation and call it. Behavior is unchanged (the shared IsGoodMove keeps the shell's reject-move-to-self guard, and all four replica test suites pass).
This commit is contained in:
@@ -659,43 +659,29 @@ func moveVolume(commandEnv *CommandEnv, v *master_pb.VolumeInformationMessage, f
|
||||
return nil
|
||||
}
|
||||
|
||||
// toBalancerLocation converts a shell replica location to the shared placement
|
||||
// abstraction, resolving the physical host for machine anti-affinity.
|
||||
func toBalancerLocation(loc *location) balancer.Location {
|
||||
return balancer.Location{
|
||||
DataCenter: loc.dc,
|
||||
Rack: loc.rack,
|
||||
NodeID: loc.dataNode.Id,
|
||||
Host: pb.NewServerAddressFromDataNode(loc.dataNode).ToHost(),
|
||||
}
|
||||
}
|
||||
|
||||
func isGoodMove(placement *super_block.ReplicaPlacement, existingReplicas []*VolumeReplica, sourceNode, targetNode *Node) bool {
|
||||
for _, replica := range existingReplicas {
|
||||
if replica.location.dataNode.Id == targetNode.info.Id &&
|
||||
replica.location.rack == targetNode.rack &&
|
||||
replica.location.dc == targetNode.dc {
|
||||
// never move to existing nodes
|
||||
return false
|
||||
}
|
||||
locs := make([]balancer.Location, len(existingReplicas))
|
||||
for i, replica := range existingReplicas {
|
||||
locs[i] = toBalancerLocation(replica.location)
|
||||
}
|
||||
|
||||
// existing replicas except the one on sourceNode
|
||||
existingReplicasExceptSourceNode := make([]*VolumeReplica, 0)
|
||||
for _, replica := range existingReplicas {
|
||||
if replica.location.dataNode.Id != sourceNode.info.Id {
|
||||
existingReplicasExceptSourceNode = append(existingReplicasExceptSourceNode, replica)
|
||||
}
|
||||
target := balancer.Location{
|
||||
DataCenter: targetNode.dc,
|
||||
Rack: targetNode.rack,
|
||||
NodeID: targetNode.info.Id,
|
||||
Host: pb.NewServerAddressFromDataNode(targetNode.info).ToHost(),
|
||||
}
|
||||
|
||||
// Don't move a replica onto a machine (host) that already holds one of this
|
||||
// volume's replicas: servers sharing a host are one fault domain, so both would
|
||||
// die together. Best-effort -- skip and let balancing try the next target.
|
||||
targetHost := pb.NewServerAddressFromDataNode(targetNode.info).ToHost()
|
||||
for _, replica := range existingReplicasExceptSourceNode {
|
||||
if pb.NewServerAddressFromDataNode(replica.location.dataNode).ToHost() == targetHost {
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// target location
|
||||
targetLocation := location{
|
||||
dc: targetNode.dc,
|
||||
rack: targetNode.rack,
|
||||
dataNode: targetNode.info,
|
||||
}
|
||||
|
||||
// check if this satisfies replication requirements
|
||||
return satisfyReplicaPlacement(placement, existingReplicasExceptSourceNode, targetLocation)
|
||||
return balancer.IsGoodMove(placement, locs, sourceNode.info.Id, target)
|
||||
}
|
||||
|
||||
// addDiskFreeBytes adjusts a disk's reported free bytes by delta (negative when a
|
||||
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle_map"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
"github.com/seaweedfs/seaweedfs/weed/topology/balancer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/wildcard"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
@@ -474,95 +475,16 @@ func satisfyReplicaCurrentLocation(replicaPlacement *super_block.ReplicaPlacemen
|
||||
return false
|
||||
}
|
||||
*/
|
||||
// satisfyReplicaPlacement reports whether placing a replica at possibleLocation
|
||||
// is consistent with the replication policy given the existing replicas. Thin
|
||||
// adapter over weed/topology/balancer so the shell and the maintenance worker
|
||||
// share one placement implementation.
|
||||
func satisfyReplicaPlacement(replicaPlacement *super_block.ReplicaPlacement, replicas []*VolumeReplica, possibleLocation location) bool {
|
||||
|
||||
existingDataCenters, _, existingDataNodes := countReplicas(replicas)
|
||||
|
||||
if _, found := existingDataNodes[possibleLocation.String()]; found {
|
||||
// avoid duplicated volume on the same data node
|
||||
return false
|
||||
locs := make([]balancer.Location, len(replicas))
|
||||
for i, r := range replicas {
|
||||
locs[i] = toBalancerLocation(r.location)
|
||||
}
|
||||
|
||||
primaryDataCenters, _ := findTopKeys(existingDataCenters)
|
||||
|
||||
// ensure data center count is within limit
|
||||
if _, found := existingDataCenters[possibleLocation.DataCenter()]; !found {
|
||||
// different from existing dcs
|
||||
if len(existingDataCenters) < replicaPlacement.DiffDataCenterCount+1 {
|
||||
// lack on different dcs
|
||||
return true
|
||||
} else {
|
||||
// adding this would go over the different dcs limit
|
||||
return false
|
||||
}
|
||||
}
|
||||
// now this is same as one of the existing data center
|
||||
if !isAmong(possibleLocation.DataCenter(), primaryDataCenters) {
|
||||
// not on one of the primary dcs
|
||||
return false
|
||||
}
|
||||
|
||||
// now this is one of the primary dcs
|
||||
primaryDcRacks := make(map[string]int)
|
||||
for _, replica := range replicas {
|
||||
if replica.location.DataCenter() != possibleLocation.DataCenter() {
|
||||
continue
|
||||
}
|
||||
primaryDcRacks[replica.location.Rack()] += 1
|
||||
}
|
||||
primaryRacks, _ := findTopKeys(primaryDcRacks)
|
||||
sameRackCount := primaryDcRacks[possibleLocation.Rack()]
|
||||
|
||||
// ensure rack count is within limit
|
||||
if _, found := primaryDcRacks[possibleLocation.Rack()]; !found {
|
||||
// different from existing racks
|
||||
if len(primaryDcRacks) < replicaPlacement.DiffRackCount+1 {
|
||||
// lack on different racks
|
||||
return true
|
||||
} else {
|
||||
// adding this would go over the different racks limit
|
||||
return false
|
||||
}
|
||||
}
|
||||
// now this is same as one of the existing racks
|
||||
if !isAmong(possibleLocation.Rack(), primaryRacks) {
|
||||
// not on the primary rack
|
||||
return false
|
||||
}
|
||||
|
||||
// now this is on the primary rack
|
||||
|
||||
// different from existing data nodes
|
||||
if sameRackCount < replicaPlacement.SameRackCount+1 {
|
||||
// lack on same rack
|
||||
return true
|
||||
} else {
|
||||
// adding this would go over the same data node limit
|
||||
return false
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
func findTopKeys(m map[string]int) (topKeys []string, max int) {
|
||||
for k, c := range m {
|
||||
if max < c {
|
||||
topKeys = topKeys[:0]
|
||||
topKeys = append(topKeys, k)
|
||||
max = c
|
||||
} else if max == c {
|
||||
topKeys = append(topKeys, k)
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func isAmong(key string, keys []string) bool {
|
||||
for _, k := range keys {
|
||||
if k == key {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
return balancer.SatisfyReplicaPlacement(replicaPlacement, locs, toBalancerLocation(&possibleLocation))
|
||||
}
|
||||
|
||||
type VolumeReplica struct {
|
||||
|
||||
Reference in New Issue
Block a user