diff --git a/telemetry/server/dashboard/dashboard.go b/telemetry/server/dashboard/dashboard.go index c977c4101..c4ea5f08e 100644 --- a/telemetry/server/dashboard/dashboard.go +++ b/telemetry/server/dashboard/dashboard.go @@ -155,11 +155,13 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
Volume Servers Over Time
+
Confirmed clusters only
Total Disk Usage Over Time
+
Confirmed clusters only
diff --git a/telemetry/server/storage/history.go b/telemetry/server/storage/history.go index 4321f9c82..8a3776650 100644 --- a/telemetry/server/storage/history.go +++ b/telemetry/server/storage/history.go @@ -40,6 +40,25 @@ func (s *PrometheusStorage) appendHistory(data *proto.TelemetryData, receivedAt s.histories[data.TopologyId] = h } +// seriesHistories picks the clusters the fleet-wide series are built from: the +// confirmed ones. A cluster that only ever reported on one day is usually a CI +// or test cluster that lived for a minute, and those arrive faster than they +// age out, so counting them makes every fleet total climb forever. Falls back to +// all clusters while none is confirmed yet, so a fresh server still draws its +// charts. Callers must hold s.mu. +func (s *PrometheusStorage) seriesHistories() map[string][]HistorySample { + confirmed := make(map[string][]HistorySample, len(s.histories)) + for id, history := range s.histories { + if len(history) >= confirmDays { + confirmed[id] = history + } + } + if len(confirmed) == 0 { + return s.histories + } + return confirmed +} + func sameUTCDay(a, b int64) bool { ta, tb := time.Unix(a, 0).UTC(), time.Unix(b, 0).UTC() return ta.Year() == tb.Year() && ta.YearDay() == tb.YearDay() diff --git a/telemetry/server/storage/metrics_test.go b/telemetry/server/storage/metrics_test.go index 8a55a2004..7c7166276 100644 --- a/telemetry/server/storage/metrics_test.go +++ b/telemetry/server/storage/metrics_test.go @@ -70,7 +70,7 @@ func TestGetMetricsWindowStartsAtOldestSample(t *testing.T) { } // History reaching past the requested window still clips to the window. - seedSamples(s, "old", HistorySample{TotalDiskBytes: 50, VolumeServerCount: 1}, -40) + seedSamples(s, "old", HistorySample{TotalDiskBytes: 50, VolumeServerCount: 1}, -40, -39) metrics, err = s.GetMetrics(10) if err != nil { t.Fatal(err) @@ -86,7 +86,7 @@ func TestGetMetricsAgreesWithClusterSizes(t *testing.T) { s := newPrometheusStorage(prometheus.NewRegistry()) seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3}, -9, -8, -7, -6, -5, -4, -3, -2, -1, 0) - seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2}, -2) + seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2}, -3, -2) seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9}, -9, -8) metrics, err := s.GetMetrics(10) @@ -104,6 +104,45 @@ func TestGetMetricsAgreesWithClusterSizes(t *testing.T) { } } +// Short-lived clusters -- a CI run, a docker-compose demo -- report once under a +// fresh raft topology id and never again. Held forward for the whole active +// window they would stack up into a fleet that grows every day, so they stay out +// of the totals until they are confirmed. +func TestGetMetricsExcludesUnconfirmedClusters(t *testing.T) { + s := newPrometheusStorage(prometheus.NewRegistry()) + + seedSamples(s, "real", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3}, -3, -2, -1, 0) + for _, id := range []string{"ci-1", "ci-2", "ci-3"} { + seedSamples(s, id, HistorySample{TotalDiskBytes: 5, VolumeServerCount: 14}, -1) + } + + metrics, err := s.GetMetrics(4) + if err != nil { + t.Fatal(err) + } + if got := metrics["server_counts"].([]int64); !equalInt64(got, []int64{3, 3, 3, 3}) { + t.Errorf("server_counts = %v, want the confirmed cluster alone", got) + } + if got := metrics["disk_usage"].([]uint64); !equal(got, []uint64{300, 300, 300, 300}) { + t.Errorf("disk_usage = %v, want the confirmed cluster alone", got) + } +} + +// Until any cluster has two days of history the charts fall back to every +// cluster, so a fresh server doesn't serve empty series. +func TestGetMetricsFallsBackWhenNoneConfirmed(t *testing.T) { + s := newPrometheusStorage(prometheus.NewRegistry()) + seedSamples(s, "new", HistorySample{TotalDiskBytes: 100, VolumeServerCount: 2}, 0) + + metrics, err := s.GetMetrics(7) + if err != nil { + t.Fatal(err) + } + if got := metrics["server_counts"].([]int64); !equalInt64(got, []int64{2}) { + t.Errorf("server_counts = %v, want today's only cluster", got) + } +} + func equalInt64(a, b []int64) bool { if len(a) != len(b) { return false diff --git a/telemetry/server/storage/prometheus.go b/telemetry/server/storage/prometheus.go index ee88e98a7..3a3eea59e 100644 --- a/telemetry/server/storage/prometheus.go +++ b/telemetry/server/storage/prometheus.go @@ -164,21 +164,22 @@ func (s *PrometheusStorage) GetInstances(limit int) ([]*telemetryData, error) { return instances, nil } -// GetMetrics returns fleet-wide daily totals for the last `days` days, in the -// parallel-array shape the dashboard charts expect. Totals come from the daily -// histories rather than from s.instances, which holds only each cluster's most -// recent report and so would credit every cluster to the single day it last -// reported on. +// GetMetrics returns fleet-wide daily totals across confirmed clusters for the +// last `days` days, in the parallel-array shape the dashboard charts expect. +// Totals come from the daily histories rather than from s.instances, which holds +// only each cluster's most recent report and so would credit every cluster to +// the single day it last reported on. func (s *PrometheusStorage) GetMetrics(days int) (map[string]interface{}, error) { s.mu.RLock() defer s.mu.RUnlock() - axis := newDailySeries(days, s.histories) + histories := s.seriesHistories() + axis := newDailySeries(days, histories) activeSince := time.Now().UTC().AddDate(0, 0, -activeDays).Unix() diskUsage := make([]uint64, len(axis.dates)) serverCounts := make([]int64, len(axis.dates)) - for _, history := range s.histories { + for _, history := range histories { if disk, ok := axis.align(history, activeSince, diskBytes); ok { for i, v := range disk { diskUsage[i] += v diff --git a/telemetry/server/storage/sizes.go b/telemetry/server/storage/sizes.go index 34f3204a3..3145f85f7 100644 --- a/telemetry/server/storage/sizes.go +++ b/telemetry/server/storage/sizes.go @@ -29,18 +29,19 @@ type ClusterSizeSeries struct { TotalDisk uint64 `json:"total_disk"` // across all clusters on the last day } -// GetClusterSizeSeries returns the last `days` days of per-cluster disk usage. -// Clusters beyond `limit` are folded into Other. +// GetClusterSizeSeries returns the last `days` days of per-cluster disk usage +// across confirmed clusters. Clusters beyond `limit` are folded into Other. func (s *PrometheusStorage) GetClusterSizeSeries(days, limit int) ClusterSizeSeries { s.mu.RLock() defer s.mu.RUnlock() - axis := newDailySeries(days, s.histories) + histories := s.seriesHistories() + axis := newDailySeries(days, histories) activeSince := time.Now().UTC().AddDate(0, 0, -activeDays).Unix() last := len(axis.dates) - 1 series := ClusterSizeSeries{Dates: axis.dates} - for id, history := range s.histories { + for id, history := range histories { disk, ok := axis.align(history, activeSince, diskBytes) if !ok { continue diff --git a/telemetry/server/storage/sizes_test.go b/telemetry/server/storage/sizes_test.go index 4c9948e7f..15f6aaee7 100644 --- a/telemetry/server/storage/sizes_test.go +++ b/telemetry/server/storage/sizes_test.go @@ -31,9 +31,11 @@ func TestClusterSizeSeries(t *testing.T) { seedHistory(s, "daily", 300, -9, -8, -7, -6, -5, -4, -3, -2, -1, 0) // Reported two days ago and not since: still active, so its size is held // to the right edge instead of dropping out of the stack. - seedHistory(s, "lagging", 200, -2) + seedHistory(s, "lagging", 200, -3, -2) // Stopped reporting past the active window: its own days only. seedHistory(s, "gone", 900, -9, -8) + // One day of history only: unconfirmed, so it stays out of the stack. + seedHistory(s, "oneshot", 400, -1) series := s.GetClusterSizeSeries(10, 0) if len(series.Dates) != 10 { @@ -50,12 +52,15 @@ func TestClusterSizeSeries(t *testing.T) { if got := byId["daily"]; !equal(got, []uint64{300, 300, 300, 300, 300, 300, 300, 300, 300, 300}) { t.Errorf("daily = %v, want 300 every day", got) } - if got := byId["lagging"]; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 0, 200, 200, 200}) { + if got := byId["lagging"]; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 200, 200, 200, 200}) { t.Errorf("lagging = %v, want its size carried to the right edge", got) } if got := byId["gone"]; !equal(got, []uint64{900, 900, 0, 0, 0, 0, 0, 0, 0, 0}) { t.Errorf("gone = %v, want no capacity after its last report", got) } + if _, ok := byId["oneshot"]; ok { + t.Errorf("unconfirmed cluster in the stack: %v", byId["oneshot"]) + } // The total is the last day's stack height: daily + lagging, not gone. if series.TotalDisk != 500 { @@ -74,7 +79,7 @@ func TestClusterSizeSeries(t *testing.T) { if series.Other == nil || series.Other.Count != 2 { t.Fatalf("other = %+v, want 2 clusters", series.Other) } - if !equal(series.Other.Disk, []uint64{900, 900, 0, 0, 0, 0, 0, 200, 200, 200}) { + if !equal(series.Other.Disk, []uint64{900, 900, 0, 0, 0, 0, 200, 200, 200, 200}) { t.Errorf("other = %v, want lagging+gone summed per day", series.Other.Disk) } if series.ClusterCount != 3 || series.TotalDisk != 500 {