mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* telemetry: tidy the server module after the protobuf bump Claude-Session: https://claude.ai/code/session_01VGiphDxpsKMwu9XpUFhybQ * telemetry: keep only clusters that store at least 10 GiB Fresh weed server runs, CI jobs and throwaway containers each mint their own cluster id. They came in at tens of thousands a day, were most of the counted clusters and held almost none of the bytes, and the state file and the metrics page grew with every one of them. Reports under the floor are counted and dropped, and a state file written before the floor sheds them on the first restart. Claude-Session: https://claude.ai/code/session_01VGiphDxpsKMwu9XpUFhybQ * master: report telemetry only once the cluster stores 10 GiB A throwaway cluster no longer registers itself with its first report a minute after start; a real one begins reporting at the first daily tick after it crosses the floor. Claude-Session: https://claude.ai/code/session_01VGiphDxpsKMwu9XpUFhybQ
86 lines
2.5 KiB
Go
86 lines
2.5 KiB
Go
package storage
|
|
|
|
import (
|
|
"path/filepath"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/prometheus/client_golang/prometheus"
|
|
"github.com/seaweedfs/seaweedfs/telemetry/proto"
|
|
)
|
|
|
|
func TestClusterHistory(t *testing.T) {
|
|
s := newPrometheusStorage(prometheus.NewRegistry())
|
|
|
|
report := &proto.TelemetryData{
|
|
TopologyId: "hist-cluster",
|
|
Version: "4.40",
|
|
Os: "linux/amd64",
|
|
VolumeServerCount: 3,
|
|
TotalDiskBytes: proto.MinDiskBytes + 1000,
|
|
TotalVolumeCount: 10,
|
|
}
|
|
if err := s.StoreTelemetry(report); err != nil {
|
|
t.Fatalf("store: %v", err)
|
|
}
|
|
|
|
// A second report on the same UTC day replaces the day's sample.
|
|
report.TotalDiskBytes = proto.MinDiskBytes + 2000
|
|
if err := s.StoreTelemetry(report); err != nil {
|
|
t.Fatalf("store: %v", err)
|
|
}
|
|
samples, ok := s.GetHistory("hist-cluster", 90)
|
|
if !ok {
|
|
t.Fatal("cluster missing from history")
|
|
}
|
|
if len(samples) != 1 {
|
|
t.Fatalf("got %d samples, want 1 (same-day replace)", len(samples))
|
|
}
|
|
if samples[0].TotalDiskBytes != proto.MinDiskBytes+2000 {
|
|
t.Errorf("same-day sample not replaced: got %d", samples[0].TotalDiskBytes)
|
|
}
|
|
|
|
// An older sample from a previous day is appended and survives the
|
|
// state-file round trip.
|
|
s.mu.Lock()
|
|
s.histories["hist-cluster"] = append([]HistorySample{{
|
|
Ts: time.Now().AddDate(0, 0, -5).Unix(),
|
|
TotalDiskBytes: 500,
|
|
}}, s.histories["hist-cluster"]...)
|
|
s.mu.Unlock()
|
|
|
|
path := filepath.Join(t.TempDir(), "state.json")
|
|
if err := s.SaveStateIfDirty(path); err != nil {
|
|
t.Fatalf("save: %v", err)
|
|
}
|
|
s.instances = make(map[string]*telemetryData)
|
|
s.histories = make(map[string][]HistorySample)
|
|
if _, err := s.LoadState(path); err != nil {
|
|
t.Fatalf("load: %v", err)
|
|
}
|
|
samples, ok = s.GetHistory("hist-cluster", 90)
|
|
if !ok || len(samples) != 2 {
|
|
t.Fatalf("after round trip: ok=%v samples=%d, want 2", ok, len(samples))
|
|
}
|
|
if samples[0].TotalDiskBytes != 500 || samples[1].TotalDiskBytes != proto.MinDiskBytes+2000 {
|
|
t.Errorf("samples corrupted after round trip: %+v", samples)
|
|
}
|
|
|
|
// GetHistory filters by the requested window.
|
|
samples, _ = s.GetHistory("hist-cluster", 3)
|
|
if len(samples) != 1 {
|
|
t.Errorf("window filter: got %d samples, want 1", len(samples))
|
|
}
|
|
|
|
// Unknown cluster reports !ok.
|
|
if _, ok := s.GetHistory("nope", 90); ok {
|
|
t.Error("unknown cluster reported ok")
|
|
}
|
|
|
|
// Cleanup drops the history together with the instance.
|
|
s.CleanupOldInstances(0)
|
|
if _, ok := s.GetHistory("hist-cluster", 90); ok {
|
|
t.Error("history survived instance cleanup")
|
|
}
|
|
}
|