mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 06:22:05 +02:00
* fix(s3/shell): include EC volumes in bucket size metrics and collection.list S3 bucket size metrics exported to Prometheus (and fed through stats.UpdateBucketSizeMetrics) are computed by collectCollectionInfoFromTopology, which only walked diskInfo.VolumeInfos. As soon as a volume was encoded to EC it dropped out of every aggregate, so Grafana showed bucket sizes shrinking while physical disk usage kept climbing. The shell helper collectCollectionInfo — used by collection.list and s3.bucket.quota.enforce — had the same gap, with the EC branch left as a commented-out TODO. Fold EC shards into both paths using the same approach the admin dashboard already uses (PR #9093): - PhysicalSize / Size sum across shard holders: EC shards are node-local (not replicas), so per-node TotalSize() and MinusParityShards().TotalSize() sum to the whole-volume physical and logical sizes respectively. - FileCount is deduped via max across reporters (every shard holder reports the same .ecx count; a slow node with a not-yet-loaded .ecx reports 0 and must not pin the aggregate). - DeleteCount is summed (each delete tombstones exactly one node's .ecj). - VolumeCount increments once per unique EC volume id. Adds regression tests covering pure-EC, mixed regular+EC, and the slow-reporter FileCount dedupe case. Refs #9086 * Address PR review feedback: EC size helpers, composite key, VolumeCount dedupe - Add EcShardsTotalSize / EcShardsDataSize helpers in the erasure_coding package that walk the shard bitmap directly instead of materializing a ShardsInfo and copying it via MinusParityShards(). Keeps the DataShardsCount dependency encapsulated in one place and avoids the per-shard allocation/copy overhead in the metrics hot path. - Switch shell collectCollectionInfo ecVolumes map to a composite {collection, volumeId} key, matching the bucket_size_metrics collector and defending against any cross-collection volume id aliasing. - Dedupe VolumeCount in shell addToCollection by volume id so regular volumes aren't counted once per replica presence. Aligns the shell's collection.list output with the S3 metrics collector and the EC branch, all of which now report logical volume counts. - Add unit tests for the new helpers and for the regular-volume VolumeCount dedupe. * Parameterize EcShardsDataSize with dataShards for custom EC ratios Add a dataShards parameter to EcShardsDataSize so forks with per-volume ratio metadata (e.g. the enterprise data_shards field carried on an extended VolumeEcShardInformationMessage) can pass the configured value and get accurate logical sizes under custom EC policies like 6+3 or 16+6. Passing 0 or a negative value falls back to the upstream DataShardsCount default, which is correct for the fixed 10+4 layout — so OSS callers in s3api and shell pass 0 and keep their current behavior. Added table cases covering the custom 6+3 and 16+6 paths so the parameterization is pinned by tests.
302 lines
9.8 KiB
Go
302 lines
9.8 KiB
Go
package s3api
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
|
)
|
|
|
|
const (
|
|
bucketSizeMetricsInterval = 1 * time.Minute
|
|
listBucketPageSize = 1000 // Page size for paginated bucket listing
|
|
s3MetricsLockName = "s3.leader"
|
|
)
|
|
|
|
// CollectionInfo holds collection statistics
|
|
// Used for both metrics collection and quota enforcement
|
|
type CollectionInfo struct {
|
|
FileCount float64
|
|
DeleteCount float64
|
|
DeletedByteCount float64
|
|
Size float64 // Logical size (deduplicated by volume ID)
|
|
PhysicalSize float64 // Physical size (including all replicas)
|
|
VolumeCount int // Logical volume count (deduplicated by volume ID)
|
|
}
|
|
|
|
// volumeKey uniquely identifies a volume for deduplication
|
|
type volumeKey struct {
|
|
collection string
|
|
volumeId uint32
|
|
}
|
|
|
|
// startBucketSizeMetricsLoop periodically collects bucket size metrics and updates Prometheus gauges.
|
|
// Uses a distributed lock to ensure only one S3 instance collects metrics at a time.
|
|
// Should be called as a goroutine; stops when the provided context is cancelled.
|
|
func (s3a *S3ApiServer) startBucketSizeMetricsLoop(ctx context.Context) {
|
|
// Initial delay to let the system stabilize
|
|
select {
|
|
case <-time.After(10 * time.Second):
|
|
case <-ctx.Done():
|
|
return
|
|
}
|
|
|
|
// Create lock client for distributed lock
|
|
if len(s3a.option.Filers) == 0 {
|
|
glog.V(1).Infof("No filers configured, skipping bucket size metrics collection")
|
|
return
|
|
}
|
|
filer := s3a.option.Filers[0]
|
|
lockClient := cluster.NewLockClient(s3a.option.GrpcDialOption, filer)
|
|
owner := string(filer) + "-s3-metrics"
|
|
|
|
// Start long-lived lock - this S3 instance will only collect metrics when it holds the lock
|
|
lock := lockClient.StartLongLivedLock(s3MetricsLockName, owner, func(newLockOwner string) {
|
|
glog.V(1).Infof("S3 bucket size metrics lock owner changed to: %s", newLockOwner)
|
|
}, bucketSizeMetricsInterval)
|
|
defer lock.Stop()
|
|
|
|
ticker := time.NewTicker(bucketSizeMetricsInterval)
|
|
defer ticker.Stop()
|
|
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
glog.V(1).Infof("Stopping bucket size metrics collection")
|
|
return
|
|
case <-ticker.C:
|
|
// Only collect metrics if we hold the lock
|
|
if lock.IsLocked() {
|
|
s3a.collectAndUpdateBucketSizeMetrics(ctx)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// collectAndUpdateBucketSizeMetrics collects bucket sizes from master topology
|
|
// and updates Prometheus metrics. Uses the same approach as quota enforcement.
|
|
func (s3a *S3ApiServer) collectAndUpdateBucketSizeMetrics(ctx context.Context) {
|
|
// Collect collection info from master topology (same as quota enforcement)
|
|
collectionInfos, err := s3a.collectCollectionInfoFromMaster(ctx)
|
|
if err != nil {
|
|
glog.V(2).Infof("Failed to collect collection info from master: %v", err)
|
|
return
|
|
}
|
|
|
|
// Get list of buckets
|
|
buckets, err := s3a.listBucketNames(ctx)
|
|
if err != nil {
|
|
glog.V(2).Infof("Failed to list buckets for size metrics: %v", err)
|
|
return
|
|
}
|
|
|
|
// Map collections to buckets and update metrics
|
|
for _, bucket := range buckets {
|
|
collection := s3a.getCollectionName(bucket)
|
|
if info, found := collectionInfos[collection]; found {
|
|
stats.UpdateBucketSizeMetrics(bucket, info.Size, info.PhysicalSize, info.FileCount)
|
|
glog.V(3).Infof("Updated bucket size metrics: bucket=%s, logicalSize=%.0f, physicalSize=%.0f, objects=%.0f",
|
|
bucket, info.Size, info.PhysicalSize, info.FileCount)
|
|
} else {
|
|
// Bucket exists but no collection data (empty bucket)
|
|
stats.UpdateBucketSizeMetrics(bucket, 0, 0, 0)
|
|
}
|
|
}
|
|
}
|
|
|
|
// collectCollectionInfoFromMaster queries the master for topology info and extracts collection sizes.
|
|
// This is the same approach used by shell command s3.bucket.quota.enforce.
|
|
func (s3a *S3ApiServer) collectCollectionInfoFromMaster(ctx context.Context) (map[string]*CollectionInfo, error) {
|
|
if len(s3a.option.Masters) == 0 {
|
|
return nil, fmt.Errorf("no masters configured")
|
|
}
|
|
|
|
// Convert masters slice to map for WithOneOfGrpcMasterClients
|
|
masterMap := make(map[string]pb.ServerAddress)
|
|
for _, master := range s3a.option.Masters {
|
|
masterMap[string(master)] = master
|
|
}
|
|
|
|
// Connect to any available master and get volume list with topology
|
|
collectionInfos := make(map[string]*CollectionInfo)
|
|
|
|
err := pb.WithOneOfGrpcMasterClients(false, masterMap, s3a.option.GrpcDialOption, func(client master_pb.SeaweedClient) error {
|
|
resp, err := client.VolumeList(ctx, &master_pb.VolumeListRequest{})
|
|
if err != nil {
|
|
return fmt.Errorf("failed to get volume list: %w", err)
|
|
}
|
|
if resp == nil || resp.TopologyInfo == nil {
|
|
return fmt.Errorf("empty topology info from master")
|
|
}
|
|
collectCollectionInfoFromTopology(resp.TopologyInfo, collectionInfos)
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
return collectionInfos, nil
|
|
}
|
|
|
|
// listBucketNames returns a list of all bucket names using pagination
|
|
func (s3a *S3ApiServer) listBucketNames(ctx context.Context) ([]string, error) {
|
|
var buckets []string
|
|
|
|
err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
|
lastFileName := ""
|
|
for {
|
|
request := &filer_pb.ListEntriesRequest{
|
|
Directory: s3a.option.BucketsPath,
|
|
StartFromFileName: lastFileName,
|
|
Limit: listBucketPageSize,
|
|
InclusiveStartFrom: lastFileName == "",
|
|
}
|
|
|
|
stream, err := client.ListEntries(ctx, request)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
entriesReceived := 0
|
|
for {
|
|
resp, err := stream.Recv()
|
|
if err != nil {
|
|
if err == io.EOF {
|
|
break
|
|
}
|
|
return fmt.Errorf("error receiving bucket list entries: %w", err)
|
|
}
|
|
entriesReceived++
|
|
if resp.Entry != nil {
|
|
lastFileName = resp.Entry.Name
|
|
if resp.Entry.IsDirectory {
|
|
// Skip .uploads and other hidden directories
|
|
if !strings.HasPrefix(resp.Entry.Name, ".") {
|
|
buckets = append(buckets, resp.Entry.Name)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// If we got fewer entries than the limit, we're done
|
|
if entriesReceived < listBucketPageSize {
|
|
break
|
|
}
|
|
}
|
|
return nil
|
|
})
|
|
|
|
return buckets, err
|
|
}
|
|
|
|
// ecVolumeAgg accumulates per-volume EC counts across the shard holders.
|
|
// fileCount is volume-wide (every holder sees the same .ecx) so we take the
|
|
// max across reporters to avoid a slow node with a not-yet-loaded .ecx
|
|
// pinning the aggregate at 0. deleteCount is node-local to each .ecj
|
|
// deletion journal, so it's summed across reporters.
|
|
type ecVolumeAgg struct {
|
|
collection string
|
|
fileCount uint64
|
|
deleteCount uint64
|
|
}
|
|
|
|
// collectCollectionInfoFromTopology extracts collection info from topology.
|
|
// Deduplicates by volume ID to correctly handle missing replicas.
|
|
// Unlike dividing by copyCount (which would give wrong results if replicas are missing),
|
|
// we track seen volume IDs and only count each volume once for logical size/count.
|
|
// EC-encoded volumes are folded in via per-shard aggregation: every shard is
|
|
// node-local (not a replica), so shard sizes are summed across nodes; the
|
|
// per-volume file/delete counts carried on each shard message are deduped
|
|
// via max/sum so the aggregate doesn't double-count or drop after a volume
|
|
// is converted from regular to erasure coding.
|
|
func collectCollectionInfoFromTopology(t *master_pb.TopologyInfo, collectionInfos map[string]*CollectionInfo) {
|
|
// Track which volumes we've already seen to deduplicate by volume ID
|
|
seenVolumes := make(map[volumeKey]bool)
|
|
ecVolumes := make(map[volumeKey]*ecVolumeAgg)
|
|
|
|
for _, dc := range t.DataCenterInfos {
|
|
for _, r := range dc.RackInfos {
|
|
for _, dn := range r.DataNodeInfos {
|
|
for _, diskInfo := range dn.DiskInfos {
|
|
for _, vi := range diskInfo.VolumeInfos {
|
|
c := vi.Collection
|
|
cif, found := collectionInfos[c]
|
|
if !found {
|
|
cif = &CollectionInfo{}
|
|
collectionInfos[c] = cif
|
|
}
|
|
|
|
// Always add to physical size (all replicas)
|
|
cif.PhysicalSize += float64(vi.Size)
|
|
|
|
// Check if we've already counted this volume for logical stats
|
|
key := volumeKey{collection: c, volumeId: vi.Id}
|
|
if seenVolumes[key] {
|
|
// Already counted this volume, skip logical stats
|
|
continue
|
|
}
|
|
seenVolumes[key] = true
|
|
|
|
// First time seeing this volume - add to logical stats
|
|
cif.Size += float64(vi.Size)
|
|
cif.FileCount += float64(vi.FileCount)
|
|
cif.DeleteCount += float64(vi.DeleteCount)
|
|
cif.DeletedByteCount += float64(vi.DeletedByteCount)
|
|
cif.VolumeCount++
|
|
}
|
|
|
|
for _, esi := range diskInfo.EcShardInfos {
|
|
c := esi.Collection
|
|
cif, found := collectionInfos[c]
|
|
if !found {
|
|
cif = &CollectionInfo{}
|
|
collectionInfos[c] = cif
|
|
}
|
|
|
|
// EC shards are node-local (no replication), so both
|
|
// physical and logical shard sizes sum across nodes
|
|
// without any dedupe. Logical size excludes parity
|
|
// shards; physical size includes them. Upstream OSS
|
|
// uses the fixed 10+4 ratio (dataShards=0 → default);
|
|
// forks with per-volume ratio metadata can pass the
|
|
// configured value here.
|
|
cif.PhysicalSize += float64(erasure_coding.EcShardsTotalSize(esi))
|
|
cif.Size += float64(erasure_coding.EcShardsDataSize(esi, 0))
|
|
|
|
key := volumeKey{collection: c, volumeId: esi.Id}
|
|
agg, ok := ecVolumes[key]
|
|
if !ok {
|
|
agg = &ecVolumeAgg{collection: c}
|
|
ecVolumes[key] = agg
|
|
cif.VolumeCount++
|
|
}
|
|
if esi.FileCount > agg.fileCount {
|
|
agg.fileCount = esi.FileCount
|
|
}
|
|
agg.deleteCount += esi.DeleteCount
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Fold deduped EC file/delete counts into each collection's totals.
|
|
for _, agg := range ecVolumes {
|
|
cif := collectionInfos[agg.collection]
|
|
if cif == nil {
|
|
continue
|
|
}
|
|
cif.FileCount += float64(agg.fileCount)
|
|
cif.DeleteCount += float64(agg.deleteCount)
|
|
}
|
|
}
|