mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-08 15:27:43 +02:00
* storage/topology: keep disk-full read-only volumes vacuumable The vacuum sweep skipped every read-only replica, so a volume that went read-only because its disk filled could never reclaim its garbage — the exact situation compaction exists for. The volume server now reports disk_space_low in VacuumVolumeCheckResponse, and the sweep skips a read-only replica only when the flag is clear. An explicit volumeId vacuum is unaffected: it already bypassed the read-only rule. The field takes number 4: 2 and 3 are downstream-allocated for tombstone retention, keeping the wire merge clean. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * storage: measure vacuum free space against live bytes The pre-compaction space check required the current .dat + .idx size free, which includes the garbage being reclaimed — on a nearly full disk that estimate can never fit, so the volume stayed garbage-bound forever. Measure against the estimated compacted output instead: superblock plus live index entries plus live content bytes, with the existing ten percent buffer unchanged. Mirrors the same check in the Rust volume server. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * vacuum: count per-needle framing in the compacted-size estimate The live-bytes estimate covered each live needle's content and index entry but not its .dat framing (header, checksum, timestamp, padding — ~32 bytes on version 3). For small-needle volumes that is more than the 10% headroom, so a disk with space between the estimate and the real output still ran out mid-compaction. Rust side mirrors the same formula. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * storage: report disk_space_low only when it is the sole read-only cause Review feedback (ihnokim, greptile, devin): a volume read-only for low disk space AND an operator mark or I/O quarantine was still eligible for the automatic sweep, rewriting a copy meant to stay protected. The flag now reports only the benign sole-cause case in both servers. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * topology: fail closed when the read-only lookup misses in the sweep A heartbeat can drop the volume from the DataNode cache between the location-list copy and VacuumVolumeCheck; a lookup error previously skipped the read-only check entirely. Review feedback (coderabbit). Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
137 lines
4.2 KiB
Go
137 lines
4.2 KiB
Go
package weed_server
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"strconv"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
|
|
"runtime"
|
|
|
|
"github.com/prometheus/procfs"
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
)
|
|
|
|
var numCPU = runtime.NumCPU()
|
|
|
|
func (vs *VolumeServer) VacuumVolumeCheck(ctx context.Context, req *volume_server_pb.VacuumVolumeCheckRequest) (*volume_server_pb.VacuumVolumeCheckResponse, error) {
|
|
|
|
resp := &volume_server_pb.VacuumVolumeCheckResponse{}
|
|
|
|
garbageRatio, diskSpaceLow, err := vs.store.CheckCompactVolume(needle.VolumeId(req.VolumeId))
|
|
|
|
resp.GarbageRatio = garbageRatio
|
|
resp.DiskSpaceLow = diskSpaceLow
|
|
|
|
if err != nil {
|
|
glog.V(3).Infof("check volume %d: %v", req.VolumeId, err)
|
|
return resp, volumeStatusError(err)
|
|
}
|
|
|
|
return resp, nil
|
|
|
|
}
|
|
|
|
func (vs *VolumeServer) VacuumVolumeCompact(req *volume_server_pb.VacuumVolumeCompactRequest, stream volume_server_pb.VolumeServer_VacuumVolumeCompactServer) error {
|
|
if err := vs.checkGrpcAdminAuth(stream.Context()); err != nil {
|
|
return err
|
|
}
|
|
if err := vs.CheckMaintenanceMode(); err != nil {
|
|
return err
|
|
}
|
|
|
|
start := time.Now()
|
|
defer func(start time.Time) {
|
|
stats.VolumeServerVacuumingHistogram.WithLabelValues("compact").Observe(time.Since(start).Seconds())
|
|
}(start)
|
|
|
|
resp := &volume_server_pb.VacuumVolumeCompactResponse{}
|
|
reportInterval := int64(1024 * 1024 * 128)
|
|
nextReportTarget := reportInterval
|
|
fs, fsErr := procfs.NewDefaultFS()
|
|
var sendErr error
|
|
err := vs.store.CompactVolume(needle.VolumeId(req.VolumeId), req.Preallocate, vs.compactionBytePerSecond, func(processed int64) bool {
|
|
if processed > nextReportTarget {
|
|
resp.ProcessedBytes = processed
|
|
if fsErr == nil && numCPU > 0 {
|
|
if fsLa, err := fs.LoadAvg(); err == nil {
|
|
resp.LoadAvg_1M = float32(fsLa.Load1 / float64(numCPU))
|
|
}
|
|
}
|
|
if sendErr = stream.Send(resp); sendErr != nil {
|
|
return false
|
|
}
|
|
nextReportTarget = processed + reportInterval
|
|
}
|
|
return true
|
|
})
|
|
|
|
stats.VolumeServerVacuumingCompactCounter.WithLabelValues(strconv.FormatBool(err == nil && sendErr == nil)).Inc()
|
|
if err != nil {
|
|
glog.Errorf("failed compact volume %d: %v", req.VolumeId, err)
|
|
return volumeStatusError(fmt.Errorf("compact volume %d: %w", req.VolumeId, err))
|
|
}
|
|
if sendErr != nil {
|
|
glog.Errorf("failed compact volume %d report progress: %v", req.VolumeId, sendErr)
|
|
return sendErr
|
|
}
|
|
|
|
glog.V(1).Infof("compact volume %d", req.VolumeId)
|
|
return nil
|
|
|
|
}
|
|
|
|
func (vs *VolumeServer) VacuumVolumeCommit(ctx context.Context, req *volume_server_pb.VacuumVolumeCommitRequest) (*volume_server_pb.VacuumVolumeCommitResponse, error) {
|
|
if err := vs.checkGrpcAdminAuth(ctx); err != nil {
|
|
return nil, err
|
|
}
|
|
if err := vs.CheckMaintenanceMode(); err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
start := time.Now()
|
|
defer func(start time.Time) {
|
|
stats.VolumeServerVacuumingHistogram.WithLabelValues("commit").Observe(time.Since(start).Seconds())
|
|
}(start)
|
|
|
|
resp := &volume_server_pb.VacuumVolumeCommitResponse{}
|
|
|
|
readOnly, volumeSize, err := vs.store.CommitCompactVolume(needle.VolumeId(req.VolumeId))
|
|
|
|
stats.VolumeServerVacuumingCommitCounter.WithLabelValues(strconv.FormatBool(err == nil)).Inc()
|
|
resp.IsReadOnly = readOnly
|
|
resp.VolumeSize = uint64(volumeSize)
|
|
if err != nil {
|
|
glog.Errorf("failed commit volume %d: %v", req.VolumeId, err)
|
|
return resp, volumeStatusError(fmt.Errorf("commit compact volume %d: %w", req.VolumeId, err))
|
|
}
|
|
glog.V(1).Infof("commit volume %d", req.VolumeId)
|
|
return resp, nil
|
|
|
|
}
|
|
|
|
func (vs *VolumeServer) VacuumVolumeCleanup(ctx context.Context, req *volume_server_pb.VacuumVolumeCleanupRequest) (*volume_server_pb.VacuumVolumeCleanupResponse, error) {
|
|
if err := vs.checkGrpcAdminAuth(ctx); err != nil {
|
|
return nil, err
|
|
}
|
|
if err := vs.CheckMaintenanceMode(); err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
resp := &volume_server_pb.VacuumVolumeCleanupResponse{}
|
|
err := vs.store.CommitCleanupVolume(needle.VolumeId(req.VolumeId))
|
|
|
|
if err != nil {
|
|
glog.Errorf("failed cleanup volume %d: %v", req.VolumeId, err)
|
|
return resp, volumeStatusError(fmt.Errorf("cleanup volume %d: %w", req.VolumeId, err))
|
|
}
|
|
glog.V(1).Infof("cleanup volume %d", req.VolumeId)
|
|
|
|
return resp, nil
|
|
|
|
}
|