mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-16 04:06:44 +00:00
telemetry: stack volume servers per cluster, drop the total disk usage chart (#10550)
* telemetry: stack volume servers per cluster over time The fleet-wide server count says how many volume servers reported, but not who they belong to. Carry per-cluster counts in /api/cluster-sizes and draw them the same way as cluster sizes, sharing one cluster ranking so a cluster keeps its colour across both stacks. * telemetry: drop the total disk usage chart from the dashboard The stacked cluster sizes chart right below it has the same fleet total as its stack height, plus the per-cluster breakdown. /api/metrics still serves the aggregate for anyone graphing it elsewhere.
This commit is contained in:
+2
-1
@@ -181,7 +181,8 @@ GET /api/metrics?days=30
|
||||
# Get one cluster's daily usage history (disk bytes, volumes, volume servers)
|
||||
GET /api/history?cluster_id=<uuid>&days=90
|
||||
|
||||
# Get per-cluster disk usage over time, largest first, the rest summed as "other"
|
||||
# Get per-cluster disk usage and volume servers over time, largest first,
|
||||
# the rest summed as "other"
|
||||
GET /api/cluster-sizes?days=30&limit=20
|
||||
```
|
||||
|
||||
|
||||
@@ -151,20 +151,6 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
<canvas id="osChart" width="400" height="200"></canvas>
|
||||
</div>
|
||||
|
||||
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Volume Servers Over Time</div>
|
||||
<div class="chart-subtitle">Confirmed clusters only</div>
|
||||
<canvas id="serverChart" width="400" height="200"></canvas>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Total Disk Usage Over Time</div>
|
||||
<div class="chart-subtitle">Confirmed clusters only</div>
|
||||
<canvas id="diskChart" width="400" height="200"></canvas>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Cluster Sizes Over Time</div>
|
||||
<div class="chart-subtitle" id="clusterSizesTotal"></div>
|
||||
@@ -173,6 +159,14 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Volume Servers Over Time</div>
|
||||
<div class="chart-subtitle" id="clusterServersTotal"></div>
|
||||
<div style="position: relative; height: 420px;">
|
||||
<canvas id="serverChart"></canvas>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Per-Cluster History</div>
|
||||
<div class="cluster-lookup">
|
||||
@@ -198,17 +192,13 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
// Load stats
|
||||
const statsResponse = await fetch('/api/stats');
|
||||
const stats = await statsResponse.json();
|
||||
|
||||
// Load metrics
|
||||
const metricsResponse = await fetch('/api/metrics?days=30');
|
||||
const metrics = await metricsResponse.json();
|
||||
|
||||
// Load per-cluster sizes over time
|
||||
const sizesResponse = await fetch('/api/cluster-sizes?days=30&limit=20');
|
||||
const sizes = await sizesResponse.json();
|
||||
|
||||
updateStats(stats);
|
||||
updateCharts(stats, metrics);
|
||||
updateCharts(stats);
|
||||
updateClusterSizes(sizes);
|
||||
|
||||
document.getElementById('loading').style.display = 'none';
|
||||
@@ -227,25 +217,9 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
document.getElementById('totalOS').textContent = Object.keys(stats.os_distribution || {}).length;
|
||||
}
|
||||
|
||||
function updateCharts(stats, metrics) {
|
||||
// Version chart
|
||||
function updateCharts(stats) {
|
||||
createPieChart('versionChart', 'Version Distribution', stats.versions || {});
|
||||
|
||||
// OS chart
|
||||
createPieChart('osChart', 'Operating System Distribution', stats.os_distribution || {});
|
||||
|
||||
|
||||
|
||||
// Server count over time
|
||||
if (metrics.dates && metrics.server_counts) {
|
||||
createLineChart('serverChart', 'Volume Servers', metrics.dates, metrics.server_counts, '#2196F3');
|
||||
}
|
||||
|
||||
// Disk usage over time
|
||||
if (metrics.dates && metrics.disk_usage) {
|
||||
const diskUsageGB = metrics.disk_usage.map(bytes => Math.round(bytes / (1024 * 1024 * 1024)));
|
||||
createLineChart('diskChart', 'Disk Usage (GB)', metrics.dates, diskUsageGB, '#4CAF50');
|
||||
}
|
||||
}
|
||||
|
||||
function createPieChart(canvasId, title, data) {
|
||||
@@ -324,37 +298,50 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
return (unit === 0 ? value : value.toFixed(value >= 100 ? 0 : 1)) + ' ' + units[unit];
|
||||
}
|
||||
|
||||
// One stacked band per cluster over time: the band is that cluster's
|
||||
// size, the top of the stack is the fleet total. Clusters beyond the
|
||||
// requested limit are summed into a trailing "other" band so the stack
|
||||
// still adds up.
|
||||
// Disk usage and volume servers both drawn as one stacked band per
|
||||
// cluster over time: the band is that cluster's share, the top of the
|
||||
// stack is the fleet total. Clusters beyond the requested limit are
|
||||
// summed into a trailing "other" band so the stack still adds up.
|
||||
function updateClusterSizes(series) {
|
||||
const clusters = series.clusters || [];
|
||||
const dates = series.dates || [];
|
||||
const count = series.cluster_count || 0;
|
||||
|
||||
document.getElementById('clusterSizesTotal').textContent =
|
||||
formatBytes(series.total_disk) + ' across ' + count + ' cluster' + (count === 1 ? '' : 's') +
|
||||
const servers = series.total_servers || 0;
|
||||
const across = ' across ' + count + ' cluster' + (count === 1 ? '' : 's') +
|
||||
' on ' + (dates[dates.length - 1] || 'no data');
|
||||
|
||||
document.getElementById('clusterSizesTotal').textContent = formatBytes(series.total_disk) + across;
|
||||
document.getElementById('clusterServersTotal').textContent =
|
||||
servers + ' volume server' + (servers === 1 ? '' : 's') + across;
|
||||
|
||||
clusterSizeIds = clusters.map(c => c.cluster_id);
|
||||
if (series.other) {
|
||||
clusterSizeIds.push(null);
|
||||
}
|
||||
|
||||
stackedChart('clusterSizeChart', dates, clusters, series.other, 'disk', formatBytes);
|
||||
stackedChart('serverChart', dates, clusters, series.other, 'servers', value => value);
|
||||
}
|
||||
|
||||
// The stacks share one cluster order, so a cluster keeps its colour and
|
||||
// its legend entry across both of them.
|
||||
function stackedChart(canvasId, dates, clusters, other, key, format) {
|
||||
const datasets = clusters.map((c, i) => {
|
||||
// Evenly spaced hues keep neighbouring bands distinguishable.
|
||||
const hue = Math.round(i * 360 / clusters.length);
|
||||
return band(c.cluster_id.slice(0, 8), c.disk,
|
||||
return band(c.cluster_id.slice(0, 8), c[key],
|
||||
'hsl(' + hue + ', 65%, 45%)', 'hsla(' + hue + ', 65%, 55%, 0.75)');
|
||||
});
|
||||
if (series.other) {
|
||||
clusterSizeIds.push(null);
|
||||
datasets.push(band('other (' + series.other.count + ')', series.other.disk,
|
||||
if (other) {
|
||||
datasets.push(band('other (' + other.count + ')', other[key],
|
||||
'#9E9E9E', 'rgba(158, 158, 158, 0.6)'));
|
||||
}
|
||||
|
||||
const ctx = document.getElementById('clusterSizeChart').getContext('2d');
|
||||
if (charts.clusterSizeChart) {
|
||||
charts.clusterSizeChart.destroy();
|
||||
const ctx = document.getElementById(canvasId).getContext('2d');
|
||||
if (charts[canvasId]) {
|
||||
charts[canvasId].destroy();
|
||||
}
|
||||
charts.clusterSizeChart = new Chart(ctx, {
|
||||
charts[canvasId] = new Chart(ctx, {
|
||||
type: 'line',
|
||||
data: { labels: dates, datasets: datasets },
|
||||
options: {
|
||||
@@ -374,7 +361,7 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
callbacks: {
|
||||
label: item => {
|
||||
const id = clusterSizeIds[item.datasetIndex];
|
||||
return (id || item.dataset.label) + ': ' + formatBytes(item.raw);
|
||||
return (id || item.dataset.label) + ': ' + format(item.raw);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -384,7 +371,7 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
y: {
|
||||
stacked: true,
|
||||
beginAtZero: true,
|
||||
ticks: { callback: value => formatBytes(value) }
|
||||
ticks: { precision: 0, callback: value => format(value) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -80,8 +80,8 @@ func TestGetMetricsWindowStartsAtOldestSample(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// "Total Disk Usage Over Time" and the stacked "Cluster Sizes Over Time" sit on
|
||||
// the same dashboard, so their last day has to add up to the same number.
|
||||
// /api/metrics is the fleet total that /api/cluster-sizes breaks down per
|
||||
// cluster, so their last day has to add up to the same number.
|
||||
func TestGetMetricsAgreesWithClusterSizes(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
|
||||
@@ -5,32 +5,38 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// ClusterSeries is one cluster's daily disk usage, aligned to the shared date
|
||||
// axis of the enclosing ClusterSizeSeries.
|
||||
// ClusterSeries is one cluster's daily disk usage and volume server count,
|
||||
// aligned to the shared date axis of the enclosing ClusterSizeSeries.
|
||||
type ClusterSeries struct {
|
||||
ClusterId string `json:"cluster_id"`
|
||||
Disk []uint64 `json:"disk"`
|
||||
Servers []uint64 `json:"servers"`
|
||||
}
|
||||
|
||||
// OtherSeries is the clusters beyond the caller's limit, summed per day so a
|
||||
// stacked chart still adds up to the fleet total.
|
||||
type OtherSeries struct {
|
||||
Count int `json:"count"`
|
||||
Disk []uint64 `json:"disk"`
|
||||
Count int `json:"count"`
|
||||
Disk []uint64 `json:"disk"`
|
||||
Servers []uint64 `json:"servers"`
|
||||
}
|
||||
|
||||
// ClusterSizeSeries is per-cluster disk usage over time: one value per cluster
|
||||
// per day, largest cluster first, ranked by their most recent day.
|
||||
// ClusterSizeSeries is per-cluster disk usage and volume server count over
|
||||
// time: one value per cluster per day, largest cluster first, ranked by disk on
|
||||
// their most recent day. Both metrics share one ranking so a cluster keeps its
|
||||
// place, and its colour, across the charts drawn from this.
|
||||
type ClusterSizeSeries struct {
|
||||
Dates []string `json:"dates"`
|
||||
Clusters []ClusterSeries `json:"clusters"`
|
||||
Other *OtherSeries `json:"other,omitempty"`
|
||||
ClusterCount int `json:"cluster_count"`
|
||||
TotalDisk uint64 `json:"total_disk"` // across all clusters on the last day
|
||||
TotalDisk uint64 `json:"total_disk"` // across all clusters on the last day
|
||||
TotalServers uint64 `json:"total_servers"` // across all clusters on the last day
|
||||
}
|
||||
|
||||
// GetClusterSizeSeries returns the last `days` days of per-cluster disk usage
|
||||
// across confirmed clusters. Clusters beyond `limit` are folded into Other.
|
||||
// and volume server counts across confirmed clusters. Clusters beyond `limit`
|
||||
// are folded into Other.
|
||||
func (s *PrometheusStorage) GetClusterSizeSeries(days, limit int) ClusterSizeSeries {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
@@ -46,8 +52,10 @@ func (s *PrometheusStorage) GetClusterSizeSeries(days, limit int) ClusterSizeSer
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
series.Clusters = append(series.Clusters, ClusterSeries{ClusterId: id, Disk: disk})
|
||||
servers, _ := axis.align(history, activeSince, serverCount)
|
||||
series.Clusters = append(series.Clusters, ClusterSeries{ClusterId: id, Disk: disk, Servers: servers})
|
||||
series.TotalDisk += disk[last]
|
||||
series.TotalServers += servers[last]
|
||||
}
|
||||
series.ClusterCount = len(series.Clusters)
|
||||
|
||||
@@ -63,13 +71,17 @@ func (s *PrometheusStorage) GetClusterSizeSeries(days, limit int) ClusterSizeSer
|
||||
|
||||
if limit > 0 && len(series.Clusters) > limit {
|
||||
other := OtherSeries{
|
||||
Count: len(series.Clusters) - limit,
|
||||
Disk: make([]uint64, len(axis.dates)),
|
||||
Count: len(series.Clusters) - limit,
|
||||
Disk: make([]uint64, len(axis.dates)),
|
||||
Servers: make([]uint64, len(axis.dates)),
|
||||
}
|
||||
for _, c := range series.Clusters[limit:] {
|
||||
for i, v := range c.Disk {
|
||||
other.Disk[i] += v
|
||||
}
|
||||
for i, v := range c.Servers {
|
||||
other.Servers[i] += v
|
||||
}
|
||||
}
|
||||
series.Clusters = series.Clusters[:limit]
|
||||
series.Other = &other
|
||||
|
||||
@@ -8,12 +8,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/telemetry/proto"
|
||||
)
|
||||
|
||||
// seedHistory gives a cluster one sample per listed day offset (0 is today).
|
||||
func seedHistory(s *PrometheusStorage, id string, disk uint64, dayOffsets ...int) {
|
||||
seedSamples(s, id, HistorySample{TotalDiskBytes: disk}, dayOffsets...)
|
||||
}
|
||||
|
||||
// seedSamples is seedHistory for tests that care about more than disk usage.
|
||||
// seedSamples gives a cluster one sample per listed day offset (0 is today).
|
||||
// The sample's Ts is filled in per day offset.
|
||||
func seedSamples(s *PrometheusStorage, id string, sample HistorySample, dayOffsets ...int) {
|
||||
s.mu.Lock()
|
||||
@@ -28,14 +23,15 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
// Reported every day of the window.
|
||||
seedHistory(s, "daily", 300, -9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
-9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
// Reported two days ago and not since: still active, so its size is held
|
||||
// to the right edge instead of dropping out of the stack.
|
||||
seedHistory(s, "lagging", 200, -3, -2)
|
||||
seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2}, -3, -2)
|
||||
// Stopped reporting past the active window: its own days only.
|
||||
seedHistory(s, "gone", 900, -9, -8)
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9}, -9, -8)
|
||||
// One day of history only: unconfirmed, so it stays out of the stack.
|
||||
seedHistory(s, "oneshot", 400, -1)
|
||||
seedSamples(s, "oneshot", HistorySample{TotalDiskBytes: 400, VolumeServerCount: 4}, -1)
|
||||
|
||||
series := s.GetClusterSizeSeries(10, 0)
|
||||
if len(series.Dates) != 10 {
|
||||
@@ -45,27 +41,41 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
t.Fatalf("cluster_count = %d, want 3", series.ClusterCount)
|
||||
}
|
||||
|
||||
byId := map[string][]uint64{}
|
||||
byId := map[string]ClusterSeries{}
|
||||
for _, c := range series.Clusters {
|
||||
byId[c.ClusterId] = c.Disk
|
||||
byId[c.ClusterId] = c
|
||||
}
|
||||
if got := byId["daily"]; !equal(got, []uint64{300, 300, 300, 300, 300, 300, 300, 300, 300, 300}) {
|
||||
if got := byId["daily"].Disk; !equal(got, []uint64{300, 300, 300, 300, 300, 300, 300, 300, 300, 300}) {
|
||||
t.Errorf("daily = %v, want 300 every day", got)
|
||||
}
|
||||
if got := byId["lagging"]; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 200, 200, 200, 200}) {
|
||||
if got := byId["lagging"].Disk; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 200, 200, 200, 200}) {
|
||||
t.Errorf("lagging = %v, want its size carried to the right edge", got)
|
||||
}
|
||||
if got := byId["gone"]; !equal(got, []uint64{900, 900, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
if got := byId["gone"].Disk; !equal(got, []uint64{900, 900, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
t.Errorf("gone = %v, want no capacity after its last report", got)
|
||||
}
|
||||
if _, ok := byId["oneshot"]; ok {
|
||||
t.Errorf("unconfirmed cluster in the stack: %v", byId["oneshot"])
|
||||
t.Errorf("unconfirmed cluster in the stack: %+v", byId["oneshot"])
|
||||
}
|
||||
|
||||
// The total is the last day's stack height: daily + lagging, not gone.
|
||||
// Volume servers ride the same axis and the same hold-forward rule.
|
||||
if got := byId["daily"].Servers; !equal(got, []uint64{3, 3, 3, 3, 3, 3, 3, 3, 3, 3}) {
|
||||
t.Errorf("daily servers = %v, want 3 every day", got)
|
||||
}
|
||||
if got := byId["lagging"].Servers; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 2, 2, 2, 2}) {
|
||||
t.Errorf("lagging servers = %v, want carried to the right edge", got)
|
||||
}
|
||||
if got := byId["gone"].Servers; !equal(got, []uint64{9, 9, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
t.Errorf("gone servers = %v, want none after its last report", got)
|
||||
}
|
||||
|
||||
// The totals are the last day's stack height: daily + lagging, not gone.
|
||||
if series.TotalDisk != 500 {
|
||||
t.Errorf("total_disk = %d, want 500", series.TotalDisk)
|
||||
}
|
||||
if series.TotalServers != 5 {
|
||||
t.Errorf("total_servers = %d, want 5", series.TotalServers)
|
||||
}
|
||||
// Largest on the last day comes first so the stack reads top-down.
|
||||
if series.Clusters[0].ClusterId != "daily" {
|
||||
t.Errorf("order = %s first, want daily", series.Clusters[0].ClusterId)
|
||||
@@ -82,8 +92,12 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
if !equal(series.Other.Disk, []uint64{900, 900, 0, 0, 0, 0, 200, 200, 200, 200}) {
|
||||
t.Errorf("other = %v, want lagging+gone summed per day", series.Other.Disk)
|
||||
}
|
||||
if series.ClusterCount != 3 || series.TotalDisk != 500 {
|
||||
t.Errorf("limit changed totals: count=%d disk=%d, want 3/500", series.ClusterCount, series.TotalDisk)
|
||||
if !equal(series.Other.Servers, []uint64{9, 9, 0, 0, 0, 0, 2, 2, 2, 2}) {
|
||||
t.Errorf("other servers = %v, want lagging+gone summed per day", series.Other.Servers)
|
||||
}
|
||||
if series.ClusterCount != 3 || series.TotalDisk != 500 || series.TotalServers != 5 {
|
||||
t.Errorf("limit changed totals: count=%d disk=%d servers=%d, want 3/500/5",
|
||||
series.ClusterCount, series.TotalDisk, series.TotalServers)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -93,15 +107,17 @@ func TestClusterSizeSeriesUsesLatestDailySample(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
data := &proto.TelemetryData{
|
||||
TopologyId: "aaaaaaaa-0000-0000-0000-000000000001",
|
||||
Version: "4.40",
|
||||
Os: "linux/amd64",
|
||||
TotalDiskBytes: 100,
|
||||
TopologyId: "aaaaaaaa-0000-0000-0000-000000000001",
|
||||
Version: "4.40",
|
||||
Os: "linux/amd64",
|
||||
TotalDiskBytes: 100,
|
||||
VolumeServerCount: 4,
|
||||
}
|
||||
if err := s.StoreTelemetry(data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
data.TotalDiskBytes = 700
|
||||
data.VolumeServerCount = 6
|
||||
if err := s.StoreTelemetry(data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -118,8 +134,11 @@ func TestClusterSizeSeriesUsesLatestDailySample(t *testing.T) {
|
||||
if got := series.Clusters[0].Disk; !equal(got, []uint64{700}) {
|
||||
t.Errorf("disk = %v, want today's latest sample only", got)
|
||||
}
|
||||
if series.TotalDisk != 700 {
|
||||
t.Errorf("total_disk = %d, want 700", series.TotalDisk)
|
||||
if got := series.Clusters[0].Servers; !equal(got, []uint64{6}) {
|
||||
t.Errorf("servers = %v, want today's latest sample only", got)
|
||||
}
|
||||
if series.TotalDisk != 700 || series.TotalServers != 6 {
|
||||
t.Errorf("totals = %d disk / %d servers, want 700/6", series.TotalDisk, series.TotalServers)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user