mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-11 00:37:52 +02:00
* master: timeout AllocateVolume/DeleteVolume and defer growRequest cleanup The volume-grow goroutine clears the layout's growRequest flag only after ms.DoAutomaticVolumeGrow returns, and AllocateVolume / DeleteVolume were calling the volume-server RPC with context.Background(). A volume server that hung mid-call (heavy I/O, stuck lock, dead peer behind a stable VIP) would park the goroutine forever, leaving growRequest=true and silently blocking every subsequent automatic grow for that layout — Assign retries then drained their 30s budget with "context deadline exceeded" until the operator restarted the master. Bound both RPCs with a 5-minute deadline (creating/removing a volume is sub-second normally, generous for contended disks) and move the flag clear + filter delete into defers so a panic in DoAutomaticVolumeGrow doesn't strand the layout either. * allocate_volume: shorten timeout to 1m for faster recovery Volume create/delete is sub-second under normal conditions; 1 minute is generous even on a contended disk and clears the growRequest flag well before too many client Assigns drain their own retry budget. * trim comments
57 lines
1.7 KiB
Go
57 lines
1.7 KiB
Go
package topology
|
|
|
|
import (
|
|
"context"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"google.golang.org/grpc"
|
|
)
|
|
|
|
// Cap the RPC so a hung volume server can't strand the layout's
|
|
// growRequest flag and block all future automatic growth.
|
|
const allocateVolumeTimeout = 1 * time.Minute
|
|
|
|
type AllocateVolumeResult struct {
|
|
Error string
|
|
}
|
|
|
|
func AllocateVolume(dn *DataNode, grpcDialOption grpc.DialOption, vid needle.VolumeId, option *VolumeGrowOption) error {
|
|
|
|
return operation.WithVolumeServerClient(false, dn.ServerAddress(), grpcDialOption, func(client volume_server_pb.VolumeServerClient) error {
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), allocateVolumeTimeout)
|
|
defer cancel()
|
|
|
|
_, allocateErr := client.AllocateVolume(ctx, &volume_server_pb.AllocateVolumeRequest{
|
|
VolumeId: uint32(vid),
|
|
Collection: option.Collection,
|
|
Replication: option.ReplicaPlacement.String(),
|
|
Ttl: option.Ttl.String(),
|
|
Preallocate: option.Preallocate,
|
|
MemoryMapMaxSizeMb: option.MemoryMapMaxSizeMb,
|
|
DiskType: string(option.DiskType),
|
|
Version: option.Version,
|
|
})
|
|
return allocateErr
|
|
})
|
|
|
|
}
|
|
|
|
func DeleteVolume(dn *DataNode, grpcDialOption grpc.DialOption, vid needle.VolumeId) error {
|
|
|
|
return operation.WithVolumeServerClient(false, dn.ServerAddress(), grpcDialOption, func(client volume_server_pb.VolumeServerClient) error {
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), allocateVolumeTimeout)
|
|
defer cancel()
|
|
|
|
_, allocateErr := client.VolumeDelete(ctx, &volume_server_pb.VolumeDeleteRequest{
|
|
VolumeId: uint32(vid),
|
|
})
|
|
return allocateErr
|
|
})
|
|
|
|
}
|