Files
seaweedfs/weed/server/master_grpc_server_assign.go
T
Chris Lu d3b8030a69 master: shed assigns retryably until volume servers register capacity (#11032)
An assign arriving before any volume server has heartbeated saw zero
available space and failed outright with a plain error no client retries,
so the first write to a fresh bucket answered 500 while the cluster was
still starting. Distinguish a topology with no registered capacity from a
genuinely full one: fail fast only when registered capacity is exhausted,
and shed ResourceExhausted otherwise so the client's retry budget rides
out the startup window.

Claude-Session: https://claude.ai/code/session_018G9kWFgy8BaBAEkYV3YL9n
2026-08-30 11:08:53 -07:00

210 lines
7.1 KiB
Go

package weed_server
import (
"context"
"fmt"
"strings"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/raft"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
"github.com/seaweedfs/seaweedfs/weed/security"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
"github.com/seaweedfs/seaweedfs/weed/storage/types"
"github.com/seaweedfs/seaweedfs/weed/topology"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
func (ms *MasterServer) StreamAssign(server master_pb.Seaweed_StreamAssignServer) error {
for {
req, err := server.Recv()
if err != nil {
glog.Errorf("StreamAssign failed to receive: %v", err)
return err
}
resp, err := ms.Assign(server.Context(), req)
if err != nil {
// Return transient errors (warmup, growth-in-progress shed) as in-band
// error responses instead of killing the stream, so pooled connections
// survive.
if st, ok := status.FromError(err); ok && (st.Code() == codes.Unavailable || st.Code() == codes.ResourceExhausted) {
glog.V(1).Infof("StreamAssign transient error: %v", err)
resp = &master_pb.AssignResponse{Error: st.Message()}
} else {
glog.Errorf("StreamAssign failed to assign: %v", err)
return err
}
}
if err = server.Send(resp); err != nil {
glog.Errorf("StreamAssign failed to send: %v", err)
return err
}
}
}
func (ms *MasterServer) Assign(ctx context.Context, req *master_pb.AssignRequest) (*master_pb.AssignResponse, error) {
if !ms.Topo.IsLeader() {
return nil, raft.NotLeaderError
}
if req.Count == 0 {
req.Count = 1
}
if req.Replication == "" {
req.Replication = ms.option.DefaultReplicaPlacement
}
replicaPlacement, err := super_block.NewReplicaPlacementFromString(req.Replication)
if err != nil {
return nil, err
}
ttl, err := needle.ReadTTL(req.Ttl)
if err != nil {
return nil, err
}
if ms.Topo.IsWarmingUp() {
return nil, status.Errorf(codes.Unavailable, "master is warming up, topology is still loading")
}
diskType := types.ToDiskType(req.DiskType)
ver := needle.GetCurrentVersion()
option := &topology.VolumeGrowOption{
Collection: req.Collection,
ReplicaPlacement: replicaPlacement,
Ttl: ttl,
DiskType: diskType,
Preallocate: ms.preallocateSize,
DataCenter: req.DataCenter,
Rack: req.Rack,
DataNode: req.DataNode,
MemoryMapMaxSizeMb: req.MemoryMapMaxSizeMb,
Version: uint32(ver),
}
if !ms.Topo.DataCenterExists(option.DataCenter) {
return nil, fmt.Errorf("data center %v not found in topology", option.DataCenter)
}
vl := ms.Topo.GetVolumeLayout(option.Collection, option.ReplicaPlacement, option.Ttl, option.DiskType)
if req.DiskType == "" {
if writable, _ := vl.GetWritableVolumeCount(); writable == 0 {
if hddVl := ms.Topo.GetVolumeLayout(option.Collection, option.ReplicaPlacement, option.Ttl, types.ToDiskType(types.HddType)); hddVl != nil {
if writable, _ := hddVl.GetWritableVolumeCount(); writable > 0 {
option.DiskType = types.ToDiskType(types.HddType)
vl = hddVl
}
}
}
}
vl.SetLastGrowCount(req.WritableVolumeCount)
var (
lastErr error
maxTimeout = time.Second * 10
startTime = time.Now()
initiatedGrow bool
repickedAfterGrow bool
)
for time.Now().Sub(startTime) < maxTimeout {
fid, count, dnList, shouldGrow, err := ms.Topo.PickForWrite(req.Count, option, vl, req.ExpectedDataSize)
if shouldGrow && !initiatedGrow && !ms.option.VolumeGrowthDisabled && vl.AddGrowRequestIfAbsent() {
initiatedGrow = true
if err != nil && ms.Topo.AvailableSpaceFor(option) <= 0 && ms.Topo.CapacityFor(option) > 0 {
err = fmt.Errorf("%s and no free volumes left for %s", err.Error(), option.String())
}
ms.volumeGrowthRequestChan <- &topology.VolumeGrowRequest{
Option: option,
Count: req.WritableVolumeCount,
Reason: "grpc assign",
}
}
if err != nil {
glog.V(1).Infof("assign %v %v: %v", req, option.String(), err)
stats.MasterPickForWriteErrorCounter.Inc()
lastErr = err
if (req.DataCenter != "" || req.Rack != "") && strings.Contains(err.Error(), topology.NoWritableVolumes) {
glog.V(0).Infof("assign %v %v: %v", req, option.String(), err)
return nil, err
}
if shouldGrow {
if ms.Topo.AvailableSpaceFor(option) <= 0 {
if ms.Topo.CapacityFor(option) > 0 {
break // out of space: surface the real error, not a retryable shed
}
// No capacity registered for this disk type yet, typically a
// just-started cluster whose volume servers have not
// heartbeated. Shed retryably so the first write rides out
// the startup window instead of failing outright.
return nil, status.Errorf(codes.ResourceExhausted, "no volume server capacity registered yet for %s", option.String())
}
// Only the initiator waits, and only while the growth it triggered
// is still pending: followers shed fast so a herd doesn't pin a
// goroutine each, and an initiator whose growth concluded without
// yielding a writable volume sheds so client retries re-trigger
// growth instead of looping it here. ResourceExhausted, not
// Unavailable: clients retry it (assign_file_id.go) without
// invalidating the shared master connection.
if initiatedGrow != vl.HasGrowRequest() {
// The failed pick above may predate the growth concluding, so
// re-pick once before shedding: growth registers its volumes
// before clearing the flag, and shedding here would bounce the
// client off a volume that just landed.
if initiatedGrow && !repickedAfterGrow {
repickedAfterGrow = true
continue
}
return nil, status.Errorf(codes.ResourceExhausted, "no writable volumes for %s, volume growth in progress", option.String())
}
}
select {
case <-ctx.Done():
return nil, ctx.Err()
case <-time.After(200 * time.Millisecond):
}
continue
}
dn := dnList.Head()
if dn == nil {
continue
}
var replicas []*master_pb.Location
for _, r := range dnList.Rest() {
replicas = append(replicas, &master_pb.Location{
Url: r.Url(),
PublicUrl: r.PublicUrl,
GrpcPort: uint32(r.GrpcPort),
DataCenter: r.GetDataCenterId(),
})
}
return &master_pb.AssignResponse{
Fid: fid,
Location: &master_pb.Location{
Url: dn.Url(),
PublicUrl: dn.PublicUrl,
GrpcPort: uint32(dn.GrpcPort),
DataCenter: dn.GetDataCenterId(),
},
Count: count,
Auth: string(security.GenJwtForVolumeServer(ms.guard.SigningKey(), ms.guard.ExpiresAfterSec(), fid)),
Replicas: replicas,
}, nil
}
// The initiator timed out with its growth still pending: shed retryably
// rather than failing the write that growth is about to satisfy.
if initiatedGrow && vl.HasGrowRequest() && ms.Topo.AvailableSpaceFor(option) > 0 {
return nil, status.Errorf(codes.ResourceExhausted, "no writable volumes for %s, volume growth in progress", option.String())
}
if lastErr != nil {
glog.V(0).Infof("assign %v %v: %v", req, option.String(), lastErr)
}
return nil, lastErr
}