Files
seaweedfs/weed/admin/dash/cluster_topology.go
T
Chris Lu 0cf62a921a admin: dashboard counts chunks, not files (#10598)
* admin: count each chunk once in the dashboard total

The dashboard summed file_count from every node's volume list, so a chunk
was counted once per replica and deleted chunks were never subtracted.
Reuse the collection aggregation, which dedupes replicas and EC shard
holders and nets out tombstones.

* admin: the dashboard card counts chunks, so name it that

Volumes store chunks, and a file is split into one or more of them, so
the 'Total Files' card always read far higher than the number of files in
the filer. Rename it to 'Total Chunks' and say so in the tooltip.

* admin: collections pages count chunks once and say so

The collections list and detail pages summed file_count straight off the
topology, so replicas multiplied the count, tombstones stayed in it, and
the detail page ignored EC volumes entirely. Take the numbers from the
shared collection aggregation and label them chunks.

* admin: dedupe replica chunk counts per volume instead of dividing

Dividing each replica's live count by the copy count truncated a chunk
per odd-sized volume, and reported half the count while a volume's
second replica had not checked in yet. Replicas mirror each other's
needles and deletes, so keep the fullest report per volume id.

* admin: fix the collections CSV export column mapping

The exporter read chunks from the EC-volume cell and shifted size and
disk types with it. Read every column the table actually has.
2026-08-06 11:22:06 -07:00

230 lines
6.7 KiB
Go

package dash
import (
"context"
"encoding/json"
"fmt"
"io"
"net/http"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
)
const dirStatusTimeout = 5 * time.Second
// GetClusterTopology returns the current cluster topology with caching
func (s *AdminServer) GetClusterTopology() (*ClusterTopology, error) {
now := time.Now()
if s.cachedTopology != nil && now.Sub(s.lastCacheUpdate) < s.cacheExpiration {
return s.cachedTopology, nil
}
topology := &ClusterTopology{
UpdatedAt: now,
}
// Use gRPC only
err := s.getTopologyViaGRPC(topology)
if err != nil {
currentMaster := s.masterClient.GetMaster(context.Background())
glog.Errorf("Failed to connect to master server %s: %v", currentMaster, err)
return nil, fmt.Errorf("gRPC topology request failed: %w", err)
}
// Cache the result
s.cachedTopology = topology
s.lastCacheUpdate = now
return topology, nil
}
// fetchPublicUrlMap queries the master's /dir/status HTTP endpoint and returns
// a map from data node ID (ip:port) to its PublicUrl.
func (s *AdminServer) fetchPublicUrlMap() map[string]string {
currentMaster := s.masterClient.GetMaster(context.Background())
if currentMaster == "" {
return nil
}
ctx, cancel := context.WithTimeout(context.Background(), dirStatusTimeout)
defer cancel()
url := fmt.Sprintf("http://%s/dir/status", currentMaster.ToHttpAddress())
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
glog.V(1).Infof("Failed to build /dir/status request for %s: %v", currentMaster, err)
return nil
}
resp, err := util_http.GetGlobalHttpClient().Do(req)
if err != nil {
glog.V(1).Infof("Failed to fetch /dir/status from %s: %v", currentMaster, err)
return nil
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
glog.V(1).Infof("Non-OK response from /dir/status: %d", resp.StatusCode)
return nil
}
body, err := io.ReadAll(resp.Body)
if err != nil {
glog.V(1).Infof("Failed to read /dir/status response body: %v", err)
return nil
}
// Parse the JSON response to extract PublicUrl for each data node
var status struct {
Topology struct {
DataCenters []struct {
Racks []struct {
DataNodes []struct {
Url string `json:"Url"`
PublicUrl string `json:"PublicUrl"`
} `json:"DataNodes"`
} `json:"Racks"`
} `json:"DataCenters"`
} `json:"Topology"`
}
if err := json.Unmarshal(body, &status); err != nil {
glog.V(1).Infof("Failed to parse /dir/status response: %v", err)
return nil
}
publicUrls := make(map[string]string)
for _, dc := range status.Topology.DataCenters {
for _, rack := range dc.Racks {
for _, dn := range rack.DataNodes {
if dn.PublicUrl != "" {
publicUrls[dn.Url] = dn.PublicUrl
}
}
}
}
return publicUrls
}
// getTopologyViaGRPC gets topology using gRPC (original method)
func (s *AdminServer) getTopologyViaGRPC(topology *ClusterTopology) error {
// Fetch public URL mapping from master HTTP API
// The gRPC DataNodeInfo does not include PublicUrl, so we supplement it.
publicUrls := s.fetchPublicUrlMap()
// Get cluster status from master
err := s.WithMasterClient(func(client master_pb.SeaweedClient) error {
resp, err := client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
if err != nil {
currentMaster := s.masterClient.GetMaster(context.Background())
glog.Errorf("Failed to get volume list from master %s: %v", currentMaster, err)
return err
}
if resp.TopologyInfo != nil {
// Process gRPC response
for _, dc := range resp.TopologyInfo.DataCenterInfos {
dataCenter := DataCenter{
ID: dc.Id,
Racks: []Rack{},
}
for _, rack := range dc.RackInfos {
rackObj := Rack{
ID: rack.Id,
Nodes: []VolumeServer{},
}
for _, node := range rack.DataNodeInfos {
// Calculate totals from disk infos
var totalVolumes int64
var totalMaxVolumes int64
var totalSize int64
// Prefer the real physical disk capacity the volume server
// reports per disk; the slot-based estimate overstates capacity
// when maxVolumeCount is configured higher than the disk holds.
var diskCapacity int64
for _, diskInfo := range node.DiskInfos {
totalVolumes += diskInfo.VolumeCount
totalMaxVolumes += diskInfo.MaxVolumeCount
if diskInfo.DiskTotalBytes > 0 {
diskCapacity += int64(diskInfo.DiskTotalBytes)
} else {
diskCapacity += diskInfo.MaxVolumeCount * int64(resp.VolumeSizeLimitMb) * 1024 * 1024
}
// Sum up individual volume information
for _, volInfo := range diskInfo.VolumeInfos {
totalSize += int64(volInfo.Size)
}
// ShardSizes is local to this node, so summing
// across nodes gives the physical footprint.
for _, ecShardInfo := range diskInfo.EcShardInfos {
for _, shardSize := range ecShardInfo.ShardSizes {
totalSize += shardSize
}
}
}
// Look up PublicUrl from master HTTP API
// Use node.Address (ip:port) as the key, matching the Url field in /dir/status
nodeAddr := node.Address
if nodeAddr == "" {
nodeAddr = node.Id
}
publicUrl := publicUrls[nodeAddr]
if publicUrl == "" {
publicUrl = nodeAddr
}
vs := VolumeServer{
ID: node.Id,
Address: node.Id,
DataCenter: dc.Id,
Rack: rack.Id,
PublicURL: publicUrl,
Volumes: int(totalVolumes),
MaxVolumes: int(totalMaxVolumes),
DiskUsage: totalSize,
DiskCapacity: diskCapacity,
LastHeartbeat: time.Now(),
}
rackObj.Nodes = append(rackObj.Nodes, vs)
topology.VolumeServers = append(topology.VolumeServers, vs)
topology.TotalVolumes += vs.Volumes
topology.TotalSize += totalSize
}
dataCenter.Racks = append(dataCenter.Racks, rackObj)
}
topology.DataCenters = append(topology.DataCenters, dataCenter)
}
// Chunk counts come from the shared collection aggregation, which
// nets out tombstones and counts a chunk once no matter how many
// volume replicas or EC shard holders report it.
topology.TotalChunks = totalCollectionFileCount(resp.TopologyInfo)
}
return nil
})
return err
}
// InvalidateCache forces a refresh of cached data
func (s *AdminServer) InvalidateCache() {
s.lastCacheUpdate = time.Now().Add(-s.cacheExpiration)
s.cachedTopology = nil
s.lastFilerUpdate = time.Now().Add(-s.filerCacheExpiration)
s.cachedFilers = nil
s.lastCollectionStatsUpdate = time.Now().Add(-s.collectionStatsCacheThreshold)
s.collectionStatsCache = nil
}