mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-07-29 11:33:25 +00:00
* fix(master): let the growth initiator wait for the growth it triggered The growth-in-flight shed also fired on the request that initiated the growth: it sets the pending flag right before the shed check, so a cold-start assign enqueued growth and immediately failed itself with "volume growth in progress". With no concurrent assigns around to pick up the freshly grown volume, a single writer against an empty cluster never completes a write despite ample free space. Claim the pending flag with a compare-and-swap so exactly one request becomes the initiator, triggering growth at most once, and let it wait for that growth to land. Everyone else still sheds retryably instead of pinning a goroutine: followers behind an in-flight growth, an initiator whose growth concluded without yielding a writable volume, and an initiator whose growth outlives the 10s wait budget, which previously surfaced a non-retryable error (gRPC Unknown, HTTP 406) even though a retry would have succeeded moments later. * fix(master): stop assign waits when the request is cancelled The assign retry loops slept through client cancellation, keeping a goroutine spinning for the rest of the 10s budget after the caller had gone; StreamAssign also ran assigns on a background context detached from the stream. Wait on the request context and pass the stream context through. * topology: drop the unconditional grow-request setter Growth is only claimed through AddGrowRequestIfAbsent's compare-and-swap now; keeping the raw Store(true) around invites the check-then-set race back. * test: cover cold-start first write with a real cluster Boot a fresh master plus three empty volume servers and require the very first assign - HTTP and gRPC, each on a cold volume layout, no client retries - to complete a write. The assign that triggers volume growth must wait for it rather than answering "volume growth in progress"; unit tests stub the topology, so only a real cluster exercises the assign-grow-wait path end to end.
194 lines
6.3 KiB
Go
194 lines
6.3 KiB
Go
package weed_server
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
|
|
"github.com/seaweedfs/raft"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/security"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
|
"github.com/seaweedfs/seaweedfs/weed/topology"
|
|
"google.golang.org/grpc/codes"
|
|
"google.golang.org/grpc/status"
|
|
)
|
|
|
|
func (ms *MasterServer) StreamAssign(server master_pb.Seaweed_StreamAssignServer) error {
|
|
for {
|
|
req, err := server.Recv()
|
|
if err != nil {
|
|
glog.Errorf("StreamAssign failed to receive: %v", err)
|
|
return err
|
|
}
|
|
resp, err := ms.Assign(server.Context(), req)
|
|
if err != nil {
|
|
// Return transient errors (warmup, growth-in-progress shed) as in-band
|
|
// error responses instead of killing the stream, so pooled connections
|
|
// survive.
|
|
if st, ok := status.FromError(err); ok && (st.Code() == codes.Unavailable || st.Code() == codes.ResourceExhausted) {
|
|
glog.V(1).Infof("StreamAssign transient error: %v", err)
|
|
resp = &master_pb.AssignResponse{Error: st.Message()}
|
|
} else {
|
|
glog.Errorf("StreamAssign failed to assign: %v", err)
|
|
return err
|
|
}
|
|
}
|
|
if err = server.Send(resp); err != nil {
|
|
glog.Errorf("StreamAssign failed to send: %v", err)
|
|
return err
|
|
}
|
|
}
|
|
}
|
|
func (ms *MasterServer) Assign(ctx context.Context, req *master_pb.AssignRequest) (*master_pb.AssignResponse, error) {
|
|
|
|
if !ms.Topo.IsLeader() {
|
|
return nil, raft.NotLeaderError
|
|
}
|
|
|
|
if req.Count == 0 {
|
|
req.Count = 1
|
|
}
|
|
|
|
if req.Replication == "" {
|
|
req.Replication = ms.option.DefaultReplicaPlacement
|
|
}
|
|
replicaPlacement, err := super_block.NewReplicaPlacementFromString(req.Replication)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
ttl, err := needle.ReadTTL(req.Ttl)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
if ms.Topo.IsWarmingUp() {
|
|
return nil, status.Errorf(codes.Unavailable, "master is warming up, topology is still loading")
|
|
}
|
|
diskType := types.ToDiskType(req.DiskType)
|
|
|
|
ver := needle.GetCurrentVersion()
|
|
option := &topology.VolumeGrowOption{
|
|
Collection: req.Collection,
|
|
ReplicaPlacement: replicaPlacement,
|
|
Ttl: ttl,
|
|
DiskType: diskType,
|
|
Preallocate: ms.preallocateSize,
|
|
DataCenter: req.DataCenter,
|
|
Rack: req.Rack,
|
|
DataNode: req.DataNode,
|
|
MemoryMapMaxSizeMb: req.MemoryMapMaxSizeMb,
|
|
Version: uint32(ver),
|
|
}
|
|
|
|
if !ms.Topo.DataCenterExists(option.DataCenter) {
|
|
return nil, fmt.Errorf("data center %v not found in topology", option.DataCenter)
|
|
}
|
|
|
|
vl := ms.Topo.GetVolumeLayout(option.Collection, option.ReplicaPlacement, option.Ttl, option.DiskType)
|
|
if req.DiskType == "" {
|
|
if writable, _ := vl.GetWritableVolumeCount(); writable == 0 {
|
|
if hddVl := ms.Topo.GetVolumeLayout(option.Collection, option.ReplicaPlacement, option.Ttl, types.ToDiskType(types.HddType)); hddVl != nil {
|
|
if writable, _ := hddVl.GetWritableVolumeCount(); writable > 0 {
|
|
option.DiskType = types.ToDiskType(types.HddType)
|
|
vl = hddVl
|
|
}
|
|
}
|
|
}
|
|
}
|
|
vl.SetLastGrowCount(req.WritableVolumeCount)
|
|
|
|
var (
|
|
lastErr error
|
|
maxTimeout = time.Second * 10
|
|
startTime = time.Now()
|
|
initiatedGrow bool
|
|
)
|
|
|
|
for time.Now().Sub(startTime) < maxTimeout {
|
|
fid, count, dnList, shouldGrow, err := ms.Topo.PickForWrite(req.Count, option, vl, req.ExpectedDataSize)
|
|
if shouldGrow && !initiatedGrow && !ms.option.VolumeGrowthDisabled && vl.AddGrowRequestIfAbsent() {
|
|
initiatedGrow = true
|
|
if err != nil && ms.Topo.AvailableSpaceFor(option) <= 0 {
|
|
err = fmt.Errorf("%s and no free volumes left for %s", err.Error(), option.String())
|
|
}
|
|
ms.volumeGrowthRequestChan <- &topology.VolumeGrowRequest{
|
|
Option: option,
|
|
Count: req.WritableVolumeCount,
|
|
Reason: "grpc assign",
|
|
}
|
|
}
|
|
if err != nil {
|
|
glog.V(1).Infof("assign %v %v: %v", req, option.String(), err)
|
|
stats.MasterPickForWriteErrorCounter.Inc()
|
|
lastErr = err
|
|
if (req.DataCenter != "" || req.Rack != "") && strings.Contains(err.Error(), topology.NoWritableVolumes) {
|
|
glog.V(0).Infof("assign %v %v: %v", req, option.String(), err)
|
|
return nil, err
|
|
}
|
|
if shouldGrow {
|
|
if ms.Topo.AvailableSpaceFor(option) <= 0 {
|
|
break // out of space: surface the real error, not a retryable shed
|
|
}
|
|
// Only the initiator waits, and only while the growth it triggered
|
|
// is still pending: followers shed fast so a herd doesn't pin a
|
|
// goroutine each, and an initiator whose growth concluded without
|
|
// yielding a writable volume sheds so client retries re-trigger
|
|
// growth instead of looping it here. ResourceExhausted, not
|
|
// Unavailable: clients retry it (assign_file_id.go) without
|
|
// invalidating the shared master connection.
|
|
if initiatedGrow != vl.HasGrowRequest() {
|
|
return nil, status.Errorf(codes.ResourceExhausted, "no writable volumes for %s, volume growth in progress", option.String())
|
|
}
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
return nil, ctx.Err()
|
|
case <-time.After(200 * time.Millisecond):
|
|
}
|
|
continue
|
|
}
|
|
dn := dnList.Head()
|
|
if dn == nil {
|
|
continue
|
|
}
|
|
var replicas []*master_pb.Location
|
|
for _, r := range dnList.Rest() {
|
|
replicas = append(replicas, &master_pb.Location{
|
|
Url: r.Url(),
|
|
PublicUrl: r.PublicUrl,
|
|
GrpcPort: uint32(r.GrpcPort),
|
|
DataCenter: r.GetDataCenterId(),
|
|
})
|
|
}
|
|
return &master_pb.AssignResponse{
|
|
Fid: fid,
|
|
Location: &master_pb.Location{
|
|
Url: dn.Url(),
|
|
PublicUrl: dn.PublicUrl,
|
|
GrpcPort: uint32(dn.GrpcPort),
|
|
DataCenter: dn.GetDataCenterId(),
|
|
},
|
|
Count: count,
|
|
Auth: string(security.GenJwtForVolumeServer(ms.guard.SigningKey(), ms.guard.ExpiresAfterSec(), fid)),
|
|
Replicas: replicas,
|
|
}, nil
|
|
}
|
|
// The initiator timed out with its growth still pending: shed retryably
|
|
// rather than failing the write that growth is about to satisfy.
|
|
if initiatedGrow && vl.HasGrowRequest() && ms.Topo.AvailableSpaceFor(option) > 0 {
|
|
return nil, status.Errorf(codes.ResourceExhausted, "no writable volumes for %s, volume growth in progress", option.String())
|
|
}
|
|
if lastErr != nil {
|
|
glog.V(0).Infof("assign %v %v: %v", req, option.String(), lastErr)
|
|
}
|
|
return nil, lastErr
|
|
}
|