mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-29 11:15:34 +00:00
* vacuum: let the sweep release volumes that stay empty and quiet Vacuuming reclaims bytes but not slots: a fully emptied volume stays registered to its collection forever, and since growth is gated only on slot count a store at 99% free disk can still refuse writes to other collections (#11429). volume.deleteEmpty exists but is manual-only. With -vacuumDeleteEmptyAfterSeconds (or master.vacuumDeleteEmptyAfterSeconds under weed server/mini; default 0, off) the automatic sweep now deletes replica copies that have stayed empty and quiet for that long, the same rule volume.deleteEmpty applies on demand: remote-backed copies are skipped, and every delete carries the volume server's onlyEmpty / onlyGarbage guards so a copy written since the last report is refused rather than removed. Copies that still hold data or were written recently stay; only a volume whose every copy is deleted leaves the sweep's work map, sparing a compaction of bytes that are all deleted. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * vacuum: harden empty-volume sweep against partial and racing deletes Review follow-up on #11477: - delete a volume only when every replica copy is a verifiable empty-and-quiet candidate; deleting the empty copy of a volume whose sibling holds live files would silently cut its replica count (greptile P1). - drain the volume out of the writable list before deleting, the same drain the compact pass uses, so PickForWrite stops assigning it and pending writes settle (devin). - bound the VolumeDelete RPC so one stalled server cannot hold the vacuum lock indefinitely (greptile P1, reusing allocateVolumeTimeout). The vid2location panic scenario raised in review does not exist: VolumeLocationList methods are nil-receiver safe and a missing vid just fails enoughCopies, so a partially deleted volume skips compaction instead of crashing the sweep. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * vacuum: unregister deleted empty replicas and prune the sweep list A successful VolumeDelete only updates the volume server; the master still tracked the replica and kept it in the sweep's location list for the compaction pass (coderabbit on #11477). Unregister the replica right after its delete succeeds and drop it from the sweep copy, so a partially deleted volume only compacts copies that still exist. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * vacuum: pin deleting volumes out of the writable list across heartbeats Review follow-up on #11477 (greptile): DrainAndRemoveFromWritable only removed the volume once; a heartbeat landing between the drain and the replica deletes re-evaluated writability and re-added it, so a client write could reach a replica whose siblings were already gone and leave the volume under-replicated when the last copy refused its onlyEmpty delete. MarkDeleting records the vid in deletingVolumes — checked inside setVolumeWritable so heartbeat, capacity-recovery, and admin re-add paths all hold it out — and UnmarkDeleting releases it once the sweep finishes the copy pass. A partially deleted volume's surviving replicas then return to writable through the normal heartbeat path. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * vacuum: restore writability when a sweep delete survives Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
170 lines
6.0 KiB
Go
170 lines
6.0 KiB
Go
package command
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"time"
|
|
|
|
"github.com/aws/aws-sdk-go/aws"
|
|
"github.com/gorilla/mux"
|
|
"google.golang.org/grpc/reflection"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/security"
|
|
weed_server "github.com/seaweedfs/seaweedfs/weed/server"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
"github.com/seaweedfs/seaweedfs/weed/util/version"
|
|
)
|
|
|
|
var (
|
|
mf MasterOptions
|
|
)
|
|
|
|
func init() {
|
|
cmdMasterFollower.Run = runMasterFollower // break init cycle
|
|
mf.port = cmdMasterFollower.Flag.Int("port", 9334, "http listen port")
|
|
mf.portGrpc = cmdMasterFollower.Flag.Int("port.grpc", 0, "grpc listen port")
|
|
mf.ipBind = cmdMasterFollower.Flag.String("ip.bind", "", "ip address to bind to. Default to localhost.")
|
|
mf.peers = cmdMasterFollower.Flag.String("master", "localhost:9333", "all master nodes in comma separated ip:port list, example: 127.0.0.1:9093,127.0.0.1:9094,127.0.0.1:9095")
|
|
mf.mastersDeprecated = cmdMasterFollower.Flag.String("masters", "", "all master nodes in comma separated ip:port list (deprecated, use -master instead)")
|
|
// A follower serves /submit like the leader does, so it buffers uploads under
|
|
// the same limit and has to be told the same value. Left fixed at 256, a
|
|
// cluster raised above that would accept an upload through the leader and
|
|
// refuse the identical one through a follower.
|
|
mf.fileSizeLimitMB = cmdMasterFollower.Flag.Int("fileSizeLimitMB", 256, "limit the file size accepted by /submit, should match the leader's -fileSizeLimitMB")
|
|
|
|
mf.ip = aws.String(util.DetectedHostAddress())
|
|
mf.metaFolder = aws.String("")
|
|
mf.volumeSizeLimitMB = nil
|
|
mf.volumePreallocate = nil
|
|
mf.defaultReplication = nil
|
|
mf.garbageThreshold = aws.Float64(0.1)
|
|
mf.whiteList = nil
|
|
mf.disableHttp = aws.Bool(false)
|
|
mf.metricsAddress = aws.String("")
|
|
mf.metricsIntervalSec = aws.Int(0)
|
|
mf.raftResumeState = aws.Bool(false)
|
|
mf.maxParallelVacuumPerServer = aws.Int(1)
|
|
mf.vacuumIntervalSeconds = aws.Int(840)
|
|
mf.vacuumDeleteEmptyAfterSeconds = aws.Int(0)
|
|
mf.telemetryUrl = aws.String("https://telemetry.seaweedfs.com/api/collect")
|
|
mf.telemetryEnabled = aws.Bool(false)
|
|
}
|
|
|
|
var cmdMasterFollower = &Command{
|
|
UsageLine: "master.follower -port=9333 -master=<master1Host>:<master1Port>",
|
|
Short: "start a master follower",
|
|
Long: `start a master follower to provide volume=>location mapping service
|
|
|
|
The master follower does not participate in master election.
|
|
It just follow the existing masters, and listen for any volume location changes.
|
|
|
|
In most cases, the master follower is not needed. In big data centers with thousands of volume
|
|
servers. In theory, the master may have trouble to keep up with the write requests and read requests.
|
|
|
|
The master follower can relieve the master from read requests, which only needs to
|
|
lookup a fileId or volumeId.
|
|
|
|
The master follower currently can handle fileId lookup requests:
|
|
/dir/lookup?volumeId=4
|
|
/dir/lookup?fileId=4,49c50924569199
|
|
And gRPC API
|
|
rpc LookupVolume (LookupVolumeRequest) returns (LookupVolumeResponse) {}
|
|
|
|
This master follower is stateless and can run from any place.
|
|
|
|
`,
|
|
}
|
|
|
|
func runMasterFollower(cmd *Command, args []string) bool {
|
|
|
|
util.LoadSecurityConfiguration()
|
|
util.LoadConfiguration("master", false)
|
|
|
|
// Backward compatibility: if -masters is provided, use it
|
|
if *mf.mastersDeprecated != "" {
|
|
*mf.peers = *mf.mastersDeprecated
|
|
}
|
|
|
|
if *mf.portGrpc == 0 {
|
|
*mf.portGrpc = 10000 + *mf.port
|
|
}
|
|
|
|
startMasterFollower(mf)
|
|
|
|
return true
|
|
}
|
|
|
|
func startMasterFollower(masterOptions MasterOptions) {
|
|
|
|
// collect settings from main masters
|
|
masters := pb.ServerAddresses(*mf.peers).ToAddressMap()
|
|
|
|
var err error
|
|
grpcDialOption := security.LoadClientTLS(util.GetViper(), "grpc.master")
|
|
for i := 0; i < 10; i++ {
|
|
err = pb.WithOneOfGrpcMasterClients(false, masters, grpcDialOption, func(client master_pb.SeaweedClient) error {
|
|
resp, err := client.GetMasterConfiguration(context.Background(), &master_pb.GetMasterConfigurationRequest{})
|
|
if err != nil {
|
|
return fmt.Errorf("get master grpc address %v configuration: %w", masters, err)
|
|
}
|
|
masterOptions.defaultReplication = &resp.DefaultReplication
|
|
masterOptions.volumeSizeLimitMB = aws.Uint(uint(resp.VolumeSizeLimitMB))
|
|
masterOptions.volumePreallocate = &resp.VolumePreallocate
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
glog.V(0).Infof("failed to talk to filer %v: %v", masters, err)
|
|
glog.V(0).Infof("wait for %d seconds ...", i+1)
|
|
time.Sleep(time.Duration(i+1) * time.Second)
|
|
}
|
|
}
|
|
if err != nil {
|
|
glog.Errorf("failed to talk to filer %v: %v", masters, err)
|
|
return
|
|
}
|
|
|
|
option := masterOptions.toMasterOption(nil)
|
|
option.IsFollower = true
|
|
|
|
if *masterOptions.ipBind == "" {
|
|
*masterOptions.ipBind = *masterOptions.ip
|
|
}
|
|
|
|
r := mux.NewRouter()
|
|
ms := weed_server.NewMasterServer(r, option, masters)
|
|
listeningAddress := util.JoinHostPort(*masterOptions.ipBind, *masterOptions.port)
|
|
glog.V(0).Infof("Start Seaweed Master %s at %s", version.Version(), listeningAddress)
|
|
masterListener, masterLocalListener, e := util.NewIpAndLocalListeners(*masterOptions.ipBind, *masterOptions.port, 0)
|
|
if e != nil {
|
|
glog.Fatalf("Master startup error: %v", e)
|
|
}
|
|
|
|
// starting grpc server
|
|
grpcPort := *masterOptions.portGrpc
|
|
grpcL, grpcLocalL, err := util.NewIpAndLocalListeners(*masterOptions.ipBind, grpcPort, 0)
|
|
if err != nil {
|
|
glog.Fatalf("master failed to listen on grpc port %d: %v", grpcPort, err)
|
|
}
|
|
grpcS := pb.NewGrpcServer(security.LoadServerTLS(util.GetViper(), "grpc.master"))
|
|
master_pb.RegisterSeaweedServer(grpcS, ms)
|
|
reflection.Register(grpcS)
|
|
glog.V(0).Infof("Start Seaweed Master %s grpc server at %s:%d", version.Version(), *masterOptions.ip, grpcPort)
|
|
if grpcLocalL != nil {
|
|
go grpcS.Serve(grpcLocalL)
|
|
}
|
|
go grpcS.Serve(grpcL)
|
|
|
|
go ms.MasterClient.KeepConnectedToMaster(context.Background())
|
|
|
|
// start http server
|
|
if masterLocalListener != nil {
|
|
go newHttpServer(r, nil).Serve(masterLocalListener)
|
|
}
|
|
go newHttpServer(r, nil).Serve(masterListener)
|
|
|
|
select {}
|
|
}
|