mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* feat(master): drain pending size before marking volume readonly When vacuum, volume move, or EC encoding marks a volume readonly, in-flight assigned bytes may still be pending. This adds a drain step: immediately remove from writable list (stop new assigns), then wait for pending to decay below 4MB or 30s timeout. - Add volumeSizeTracking struct consolidating effectiveSize, reportedSize, and compactRevision into a single map - Add GetPendingSize, waitForPendingDrain, DrainAndRemoveFromWritable, DrainAndSetVolumeReadOnly to VolumeLayout - UpdateVolumeSize detects compaction via compactRevision change and resets effectiveSize instead of decaying - Wire drain into vacuum (topology_vacuum.go) and volume mark readonly (master_grpc_server_volume.go) * fix: use 2MB pending size drain threshold * fix: check crowded state on initial UpdateVolumeSize registration * fix: respect context cancellation in drain, relax test timing - DrainAndSetVolumeReadOnly now accepts context.Context and returns early on cancellation (for gRPC handler timeout/cancel) - waitForPendingDrain uses select on ctx.Done instead of time.Sleep - Increase concurrent heartbeat test timeout from 10s to 15s for CI * fix: use time-based dedup so decay runs even when reported size is unchanged The value-based dedup (same reportedSize + compactRevision = skip) prevented decay from running when pending bytes existed but no writes had landed on disk yet. The reported size stayed the same across heartbeats, so the excess never decayed. Fix: dedup replicas within the same heartbeat cycle using a 2-second time window instead of comparing values. This allows decay to run once per heartbeat cycle even when the reported size is unchanged. Also confirmed finding 1 (draining re-add race) is a false positive: - Vacuum: ensureCorrectWritables only runs for ReadOnly-changed volumes - Move/EC: readonlyVolumes flag prevents re-adding during drain * fix: make VolumeMarkReadonly non-blocking to fix EC integration test timeout The DrainAndSetVolumeReadOnly call in VolumeMarkReadonly gRPC blocked up to 30s waiting for pending bytes to decay. In integration tests (and real clusters during EC encoding), this caused timeouts because multiple volumes are marked readonly sequentially and heartbeats may not arrive fast enough to decay pending within the drain window. Fix: VolumeMarkReadonly now calls SetVolumeReadOnly immediately (stops new assigns) and only logs a warning if pending bytes remain. The drain wait is kept only for vacuum (DrainAndRemoveFromWritable) which runs inside the master's own goroutine pool. Remove DrainAndSetVolumeReadOnly as it's no longer used. * fix: relax test timing, rename test, add post-condition assert * test: add vacuum integration tests with CI workflow Full-cluster integration test for vacuum, modeled on the EC integration tests. Starts a real master + 2 volume servers, uploads data, deletes entries to create garbage, runs volume.vacuum via shell command, and verifies garbage cleanup and data integrity. Test flow: 1. Start cluster (master + 2 volume servers) 2. Upload 10 files to create volume with data 3. Delete 5 files to create ~50% garbage 4. Verify garbage ratio > 10% 5. Run volume.vacuum command 6. Verify garbage cleaned up 7. Verify remaining 5 files are still accessible CI workflow runs on push/PR to master with 15-minute timeout. Log collection on failure via artifact upload. * fix: use 500KB files and delete 75% to exceed vacuum garbage threshold * fix: add shell lock before vacuum command, fix compilation error * fix: strengthen vacuum integration test assertions - waitForServer: use net.DialTimeout instead of grpc.NewClient for real TCP readiness check - verify_garbage_before_vacuum: t.Fatal instead of warning when no garbage detected - verify_cleanup_after_vacuum: t.Fatal if no server reported the volume or cleanup wasn't verified - verify_remaining_data: read actual file contents via HTTP and compare byte-for-byte against original uploaded payloads * fix: use http.Client with timeout and close body before retry
262 lines
6.1 KiB
Go
262 lines
6.1 KiB
Go
package topology
|
|
|
|
import (
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
|
)
|
|
|
|
func TestGetPendingSize(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":1000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
_, vl := setupPickTest(t, layout, 10000)
|
|
|
|
// Initially no pending
|
|
if p := vl.GetPendingSize(1); p != 0 {
|
|
t.Fatalf("expected 0 pending, got %d", p)
|
|
}
|
|
|
|
// RecordAssign increases pending
|
|
vl.RecordAssign(1, 5000)
|
|
if p := vl.GetPendingSize(1); p != 5000 {
|
|
t.Fatalf("expected 5000 pending, got %d", p)
|
|
}
|
|
|
|
// UpdateVolumeSize (heartbeat) decays pending
|
|
vl.UpdateVolumeSize(1, 3000, 0)
|
|
// effective was 6000, reported 3000 → decay to 3000 + (6000-3000)/2 = 4500
|
|
// pending = 4500 - 3000 = 1500
|
|
if p := vl.GetPendingSize(1); p != 1500 {
|
|
t.Fatalf("expected 1500 pending after decay, got %d", p)
|
|
}
|
|
}
|
|
|
|
func TestGetPendingSize_CompactionResets(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":5000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
_, vl := setupPickTest(t, layout, 10000)
|
|
|
|
// Add large pending
|
|
vl.RecordAssign(1, 4000)
|
|
if p := vl.GetPendingSize(1); p != 4000 {
|
|
t.Fatalf("expected 4000 pending, got %d", p)
|
|
}
|
|
|
|
// Compaction happens — size drops from 5000 to 2000, revision changes.
|
|
// Without compaction awareness, decay would give: 2000 + (9000-2000)/2 = 5500.
|
|
// With compaction awareness, vid2size resets to 2000 (the real size).
|
|
vl.UpdateVolumeSize(1, 2000, 1) // revision 0 → 1
|
|
|
|
if p := vl.GetPendingSize(1); p != 0 {
|
|
t.Errorf("expected 0 pending after compaction reset, got %d", p)
|
|
}
|
|
|
|
// Verify vid2size is the reported size, not a decayed value
|
|
vl.accessLock.RLock()
|
|
if vl.sizeTracking[1].effectiveSize != 2000 {
|
|
t.Errorf("expected vid2size=2000 after compaction, got %d", vl.sizeTracking[1].effectiveSize)
|
|
}
|
|
vl.accessLock.RUnlock()
|
|
}
|
|
|
|
func TestDrainAndRemoveFromWritable_NoPending(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":1000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
_, vl := setupPickTest(t, layout, 10000)
|
|
|
|
// No pending — drain should return immediately
|
|
start := time.Now()
|
|
vl.DrainAndRemoveFromWritable(1)
|
|
if elapsed := time.Since(start); elapsed > 500*time.Millisecond {
|
|
t.Errorf("drain with no pending took %v, expected near-instant", elapsed)
|
|
}
|
|
|
|
// Verify volume is no longer writable
|
|
writable, _ := vl.GetWritableVolumeCount()
|
|
if writable != 0 {
|
|
t.Errorf("expected 0 writable after drain, got %d", writable)
|
|
}
|
|
}
|
|
|
|
func TestDrainAndRemoveFromWritable_WithPending(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":1000, "replication":"000"},
|
|
{"id":2, "size":1000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
_, vl := setupPickTest(t, layout, 10000)
|
|
|
|
// Add pending below threshold
|
|
vl.RecordAssign(1, int64(pendingSizeThreshold-1))
|
|
|
|
start := time.Now()
|
|
vl.DrainAndRemoveFromWritable(1)
|
|
elapsed := time.Since(start)
|
|
|
|
// Should return quickly since pending is below threshold
|
|
if elapsed > 500*time.Millisecond {
|
|
t.Errorf("drain with pending below threshold took %v", elapsed)
|
|
}
|
|
|
|
// Volume removed from writable
|
|
writables := vl.CloneWritableVolumes()
|
|
for _, vid := range writables {
|
|
if vid == 1 {
|
|
t.Error("volume 1 should not be writable after drain")
|
|
}
|
|
}
|
|
// Volume 2 still writable
|
|
found := false
|
|
for _, vid := range writables {
|
|
if vid == 2 {
|
|
found = true
|
|
}
|
|
}
|
|
if !found {
|
|
t.Error("volume 2 should still be writable")
|
|
}
|
|
}
|
|
|
|
func TestDrainAndRemoveFromWritable_DecaysViaConcurrentHeartbeat(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":1000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
_, vl := setupPickTest(t, layout, 10000)
|
|
|
|
// Add large pending (well above threshold)
|
|
vl.RecordAssign(1, 100*1024*1024) // 100 MB
|
|
|
|
// Simulate heartbeats in background that will decay the pending.
|
|
// Advance lastUpdateTime before each call to bypass the 2s replica dedup.
|
|
done := make(chan struct{})
|
|
go func() {
|
|
defer close(done)
|
|
for i := 0; i < 10; i++ {
|
|
time.Sleep(500 * time.Millisecond)
|
|
vl.accessLock.Lock()
|
|
if st := vl.sizeTracking[1]; st != nil {
|
|
st.lastUpdateTime = time.Time{} // reset to allow update
|
|
}
|
|
vl.accessLock.Unlock()
|
|
vl.UpdateVolumeSize(1, 1000+uint64(i+1)*1000, 0)
|
|
}
|
|
}()
|
|
|
|
start := time.Now()
|
|
vl.DrainAndRemoveFromWritable(1)
|
|
elapsed := time.Since(start)
|
|
|
|
// Should have drained within a few seconds (heartbeats every 500ms).
|
|
// Use generous margin for slow CI.
|
|
if elapsed > 15*time.Second {
|
|
t.Errorf("drain took %v, expected faster with concurrent heartbeats", elapsed)
|
|
}
|
|
|
|
<-done
|
|
|
|
// Verify volume is no longer writable
|
|
writable, _ := vl.GetWritableVolumeCount()
|
|
if writable != 0 {
|
|
t.Errorf("expected 0 writable after drain, got %d", writable)
|
|
}
|
|
}
|
|
|
|
func TestSetVolumeReadOnly_PreservesPending(t *testing.T) {
|
|
layout := `
|
|
{
|
|
"dc1":{
|
|
"rack1":{
|
|
"server1":{
|
|
"volumes":[
|
|
{"id":1, "size":1000, "replication":"000"}
|
|
],
|
|
"limit":10
|
|
}
|
|
}
|
|
}
|
|
}
|
|
`
|
|
topo := setupWithLimit(t, layout, 10000)
|
|
rp, _ := super_block.NewReplicaPlacementFromString("000")
|
|
vl := topo.GetVolumeLayout("", rp, needle.EMPTY_TTL, types.HardDriveType)
|
|
dn := vl.Lookup(1)[0]
|
|
|
|
// Add some pending
|
|
vl.RecordAssign(1, 5000)
|
|
|
|
// SetVolumeReadOnly should succeed immediately (non-blocking)
|
|
result := vl.SetVolumeReadOnly(dn, 1)
|
|
if !result {
|
|
t.Error("expected SetVolumeReadOnly to return true")
|
|
}
|
|
|
|
// Pending is still there (not drained), but volume is readonly
|
|
writable, _ := vl.GetWritableVolumeCount()
|
|
if writable != 0 {
|
|
t.Errorf("expected 0 writable after readonly, got %d", writable)
|
|
}
|
|
if p := vl.GetPendingSize(1); p != 5000 {
|
|
t.Errorf("expected 5000 pending (not drained), got %d", p)
|
|
}
|
|
}
|