Files
seaweedfs/weed/admin/dash/cluster_topology.go
T
Chris Lu 46b801aedb fix(admin): list all masters and dedupe EC file counts in dashboard (#9093)
* fix(admin): list all masters and dedupe EC file counts in dashboard

Dashboard -> Master Nodes only ever showed the currently connected master
because getMasterNodesStatus hard-coded a single entry. Replace it with a
RaftListClusterServers call that returns every master in the raft group and
tags the real leader, falling back to the current master only if the raft
call fails.

Buckets -> Object Store Buckets could render 0 objects for a bucket backed
by an EC volume. Every shard holder reports the same whole-volume
file_count (read from the replicated .ecx), so the first-seen value wins;
if that first node had not yet finished loading .ecx it reported 0 and
pinned the aggregate at 0. Take the max across reporting nodes instead.

The dashboard header total_files also dropped after volumes were converted
to erasure coding because getTopologyViaGRPC never folded EC file_count
into topology.TotalFiles. Aggregate it with the same max/sum dedupe.

* fix(admin): address PR review comments

- bound RaftListClusterServers with a 3s timeout so the dashboard endpoint
  cannot hang on a stalled master
- pre-validate raft addresses with net.SplitHostPort before calling
  pb.GrpcAddressToServerAddress, which otherwise glog.Fatalf's on a
  malformed entry and would crash the admin process
- when raft is unreachable, mark the fallback master as not-leader rather
  than claiming leadership the code cannot verify
- warn when summed EC delete_count exceeds file_count while folding into
  topology.TotalFiles, matching collectCollectionStats

* fix(admin): distinguish empty raft response from RPC failure

When RaftListClusterServers returns successfully with no servers, raft is
not initialized (standalone/non-raft cluster), so the single fallback
master is the leader. Only treat the fallback as a non-leader when the
RPC actually failed.

* fix(admin): remove misleading Objects column from S3 buckets page

The bucket "Objects" column displayed needle counts from volume
collection stats, not actual S3 object counts. This is confusing
because a single S3 object can span multiple needles (multipart
uploads, versions) and the count is inaccurate for EC volumes.

Remove the ObjectCount field from S3Bucket, the Objects table column,
the sort-by-objects handler, the detail-view row, and both CSV export
references.

* fix(admin): correct cell indexes in fallback bucket CSV export

After the Objects column was removed, the fallback CSV exporter in
admin.js still used stale cell indexes: cells[1] mapped to Owner
(not Created), cells[2] to Created (not Size), cells[3] to Logical
Size (not Quota). Align all indexes with the current table column
order and include Owner, Logical Size, and Physical Size.
2026-04-15 22:28:54 -07:00

239 lines
6.9 KiB
Go

package dash
import (
"context"
"encoding/json"
"fmt"
"io"
"net/http"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
)
var dirStatusClient = &http.Client{
Timeout: 5 * time.Second,
}
// GetClusterTopology returns the current cluster topology with caching
func (s *AdminServer) GetClusterTopology() (*ClusterTopology, error) {
now := time.Now()
if s.cachedTopology != nil && now.Sub(s.lastCacheUpdate) < s.cacheExpiration {
return s.cachedTopology, nil
}
topology := &ClusterTopology{
UpdatedAt: now,
}
// Use gRPC only
err := s.getTopologyViaGRPC(topology)
if err != nil {
currentMaster := s.masterClient.GetMaster(context.Background())
glog.Errorf("Failed to connect to master server %s: %v", currentMaster, err)
return nil, fmt.Errorf("gRPC topology request failed: %w", err)
}
// Cache the result
s.cachedTopology = topology
s.lastCacheUpdate = now
return topology, nil
}
// fetchPublicUrlMap queries the master's /dir/status HTTP endpoint and returns
// a map from data node ID (ip:port) to its PublicUrl.
func (s *AdminServer) fetchPublicUrlMap() map[string]string {
currentMaster := s.masterClient.GetMaster(context.Background())
if currentMaster == "" {
return nil
}
url := fmt.Sprintf("http://%s/dir/status", currentMaster.ToHttpAddress())
resp, err := dirStatusClient.Get(url)
if err != nil {
glog.V(1).Infof("Failed to fetch /dir/status from %s: %v", currentMaster, err)
return nil
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
glog.V(1).Infof("Non-OK response from /dir/status: %d", resp.StatusCode)
return nil
}
body, err := io.ReadAll(resp.Body)
if err != nil {
glog.V(1).Infof("Failed to read /dir/status response body: %v", err)
return nil
}
// Parse the JSON response to extract PublicUrl for each data node
var status struct {
Topology struct {
DataCenters []struct {
Racks []struct {
DataNodes []struct {
Url string `json:"Url"`
PublicUrl string `json:"PublicUrl"`
} `json:"DataNodes"`
} `json:"Racks"`
} `json:"DataCenters"`
} `json:"Topology"`
}
if err := json.Unmarshal(body, &status); err != nil {
glog.V(1).Infof("Failed to parse /dir/status response: %v", err)
return nil
}
publicUrls := make(map[string]string)
for _, dc := range status.Topology.DataCenters {
for _, rack := range dc.Racks {
for _, dn := range rack.DataNodes {
if dn.PublicUrl != "" {
publicUrls[dn.Url] = dn.PublicUrl
}
}
}
}
return publicUrls
}
// getTopologyViaGRPC gets topology using gRPC (original method)
func (s *AdminServer) getTopologyViaGRPC(topology *ClusterTopology) error {
// Fetch public URL mapping from master HTTP API
// The gRPC DataNodeInfo does not include PublicUrl, so we supplement it.
publicUrls := s.fetchPublicUrlMap()
// Get cluster status from master
err := s.WithMasterClient(func(client master_pb.SeaweedClient) error {
resp, err := client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
if err != nil {
currentMaster := s.masterClient.GetMaster(context.Background())
glog.Errorf("Failed to get volume list from master %s: %v", currentMaster, err)
return err
}
if resp.TopologyInfo != nil {
// Dedupe EC volume file counts across the nodes that report
// shards for the same volume: every shard holder reports the
// same .ecx-derived file_count, so we keep the max and sum
// node-local tombstones.
ecFile := make(map[uint32]uint64)
ecDel := make(map[uint32]uint64)
// Process gRPC response
for _, dc := range resp.TopologyInfo.DataCenterInfos {
dataCenter := DataCenter{
ID: dc.Id,
Racks: []Rack{},
}
for _, rack := range dc.RackInfos {
rackObj := Rack{
ID: rack.Id,
Nodes: []VolumeServer{},
}
for _, node := range rack.DataNodeInfos {
// Calculate totals from disk infos
var totalVolumes int64
var totalMaxVolumes int64
var totalSize int64
var totalFiles int64
for _, diskInfo := range node.DiskInfos {
totalVolumes += diskInfo.VolumeCount
totalMaxVolumes += diskInfo.MaxVolumeCount
// Sum up individual volume information
for _, volInfo := range diskInfo.VolumeInfos {
totalSize += int64(volInfo.Size)
totalFiles += int64(volInfo.FileCount)
}
// Sum up EC shard sizes on this node and collect
// volume-wide file/delete counts for later folding
// into topology.TotalFiles. ShardSizes is local to
// this node, so summing across nodes is correct;
// FileCount/DeleteCount are per-volume and must be
// deduped per volume id.
for _, ecShardInfo := range diskInfo.EcShardInfos {
for _, shardSize := range ecShardInfo.ShardSizes {
totalSize += shardSize
}
if ecShardInfo.FileCount > ecFile[ecShardInfo.Id] {
ecFile[ecShardInfo.Id] = ecShardInfo.FileCount
}
ecDel[ecShardInfo.Id] += ecShardInfo.DeleteCount
}
}
// Look up PublicUrl from master HTTP API
// Use node.Address (ip:port) as the key, matching the Url field in /dir/status
nodeAddr := node.Address
if nodeAddr == "" {
nodeAddr = node.Id
}
publicUrl := publicUrls[nodeAddr]
if publicUrl == "" {
publicUrl = nodeAddr
}
vs := VolumeServer{
ID: node.Id,
Address: node.Id,
DataCenter: dc.Id,
Rack: rack.Id,
PublicURL: publicUrl,
Volumes: int(totalVolumes),
MaxVolumes: int(totalMaxVolumes),
DiskUsage: totalSize,
DiskCapacity: totalMaxVolumes * int64(resp.VolumeSizeLimitMb) * 1024 * 1024,
LastHeartbeat: time.Now(),
}
rackObj.Nodes = append(rackObj.Nodes, vs)
topology.VolumeServers = append(topology.VolumeServers, vs)
topology.TotalVolumes += vs.Volumes
topology.TotalFiles += totalFiles
topology.TotalSize += totalSize
}
dataCenter.Racks = append(dataCenter.Racks, rackObj)
}
topology.DataCenters = append(topology.DataCenters, dataCenter)
}
// Fold deduped EC file counts into the cluster total so the
// dashboard header does not drop after volumes are converted
// to erasure coding.
for vid, fc := range ecFile {
dc := ecDel[vid]
if fc >= dc {
topology.TotalFiles += int64(fc - dc)
} else {
glog.Warningf("ec volume %d: summed delete_count=%d exceeds file_count=%d; skipping from TotalFiles", vid, dc, fc)
}
}
}
return nil
})
return err
}
// InvalidateCache forces a refresh of cached data
func (s *AdminServer) InvalidateCache() {
s.lastCacheUpdate = time.Now().Add(-s.cacheExpiration)
s.cachedTopology = nil
s.lastFilerUpdate = time.Now().Add(-s.filerCacheExpiration)
s.cachedFilers = nil
s.lastCollectionStatsUpdate = time.Now().Add(-s.collectionStatsCacheThreshold)
s.collectionStatsCache = nil
}