mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-09 15:57:47 +02:00
* fix(filer): gate the aggregated metadata disk pass on real change A subscriber whose start position is past the end of the local persisted log re-ran the whole persisted-log pass - store listings, file opens, readahead - on every loop iteration. Each iteration is paced only by the shortest wake (the 20ms hold floor on a busy watermark), so one parked subscriber kept a full CPU core busy for the life of the stream. The aggregated loop now mirrors the local loop's gate: the disk pass runs on the first pass and afterwards only when something it cannot miss changed - a local flush landed, the peers' flush low-watermark advanced (more content admitted, or new files in a shared store), the cursor moved, or a disk hold is pending (the ring read that follows an empty pass parks internally, so skipping there would strand a held entry). Regression test: a subscriber parked past the persisted-log tail holds the listing rate near zero and still delivers once peers report progress. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: re-arm the aggregated disk pass on unobserved change Review found three staleness classes the gate could not see: the flush low-watermark only catching rises (a joining peer lowers the minimum and invalidates an earlier pass's proof), a peer past the minimum landing a file without moving it, and a chunk subscriber's refs-stop bound advancing with wall time. Re-read when the low-watermark moves in either direction, when the chunk listing bound admits more files, and on a slow re-probe cadence for files no watermark can signal. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: unwind the parked ring read so the disk re-probe runs, and re-read on cursor rewinds A caught-up subscriber parks inside LoopProcessLogData's wait loop, so the re-probe interval in the outer disk gate could never elapse there; the callback now unwinds the read once the cadence is due so the gate re-evaluates. The cursor trigger also needs to notice rewinds, not just advances, since ResumeFromDiskError moves the cursor backward. --------- Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
80 lines
2.7 KiB
Go
80 lines
2.7 KiB
Go
package weed_server
|
|
|
|
// The persisted log may end before a subscriber's start position (a filer
|
|
// whose own journal is older than the position a backup client resumes from).
|
|
// With nothing on disk and every ring entry held by a peer watermark, the
|
|
// aggregated loop used to re-list and re-read the persisted log on every wake
|
|
// - about a full CPU core per such subscriber. The disk pass may only re-run
|
|
// when something it cannot miss has changed.
|
|
|
|
import (
|
|
"context"
|
|
"strings"
|
|
"sync/atomic"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/filer"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
type countingStore struct {
|
|
filer.FilerStore
|
|
logLists *atomic.Int64
|
|
}
|
|
|
|
func (s *countingStore) ListDirectoryPrefixedEntries(ctx context.Context, dirPath util.FullPath, startFileName string, includeStartFile bool, limit int64, prefix string, eachEntryFunc filer.ListEachEntryFunc) (lastFileName string, err error) {
|
|
if strings.HasPrefix(string(dirPath), filer.SystemLogDir) {
|
|
s.logLists.Add(1)
|
|
}
|
|
return s.FilerStore.ListDirectoryPrefixedEntries(ctx, dirPath, startFileName, includeStartFile, limit, prefix, eachEntryFunc)
|
|
}
|
|
|
|
func TestSubscribeLoop_AggregatedNoPersistedEntryAfterStart(t *testing.T) {
|
|
h := newSubscribeHarness(t)
|
|
|
|
lists := &atomic.Int64{}
|
|
h.f.SetStore(&countingStore{FilerStore: h.f.GetStore(), logLists: lists})
|
|
|
|
// Persisted log ends at T1: one flushed window, nothing after.
|
|
h.append(h.tsAt(0, 0))
|
|
h.append(h.tsAt(0, 1))
|
|
h.f.LocalMetaLogBuffer.ForceFlush()
|
|
waitForFlushedFiles(t, h, h.tsAt(0, 1))
|
|
|
|
// Client cursor sits just past the last persisted entry.
|
|
cursor := h.tsAt(0, 1) + int64(time.Millisecond)
|
|
|
|
ma := h.startAggregator()
|
|
|
|
// Aggregated ring holds only much newer entries (peer events).
|
|
recent := time.Now().UnixNano()
|
|
h.appendAggregated(recent)
|
|
h.appendAggregated(recent + int64(time.Millisecond))
|
|
|
|
// Peers' watermarks are stuck at the old log tail.
|
|
reportPeersAt(ma, h.tsAt(0, 1), h.tsAt(0, 1))
|
|
|
|
r := h.subscribeAggregated(cursor)
|
|
|
|
// Warm up: the first pass plus the cursor-move re-read after the gap
|
|
// machinery re-arms the cursor are legitimate.
|
|
time.Sleep(150 * time.Millisecond)
|
|
|
|
before := lists.Load()
|
|
time.Sleep(500 * time.Millisecond)
|
|
rate := float64(lists.Load()-before) / 0.5
|
|
if rate > 4 {
|
|
t.Fatalf("%.1f persisted-log listings per second while parked; the disk pass re-ran on every wake", rate)
|
|
}
|
|
|
|
// Peer progress through the held entries releases the read: the held
|
|
// events are delivered without another disk pass.
|
|
lists.Store(0)
|
|
reportPeersAt(ma, recent+int64(time.Millisecond), h.tsAt(0, 1))
|
|
waitForEvents(t, r, []int64{recent, recent + int64(time.Millisecond)}, 3*time.Second)
|
|
if got := lists.Load(); got > 4 {
|
|
t.Fatalf("%d listings while draining held entries; delivery should come from the ring", got)
|
|
}
|
|
}
|