Compare commits

..
2 Commits
10 changed files with 850 additions and 516 deletions
+6
View File
@@ -35,6 +35,12 @@ type CpuMetrics struct {
// getCpuMetrics calculates detailed CPU usage metrics using cached previous measurements.
// It returns percentages for total, user, system, iowait, and steal time.
func getCpuMetrics(cacheTimeMs uint16) (CpuMetrics, error) {
// Inside a container, /proc/stat reports the host cores' counters (via
// lxcfs on LXC, or the host's procfs elsewhere), not the container's own
// usage. Prefer the cgroup's own CPU accounting when available. (#2332)
if metrics, ok := containerCpuMetrics(cacheTimeMs); ok {
return metrics, nil
}
times, err := cpu.Times(false)
if err != nil || len(times) == 0 {
return CpuMetrics{}, err
+417
View File
@@ -0,0 +1,417 @@
//go:build linux
package agent
import (
"bytes"
"os"
"path/filepath"
"runtime"
"strconv"
"strings"
"sync"
"time"
"github.com/henrygd/beszel/agent/utils"
)
// Container-aware CPU accounting (issue #2332).
//
// Inside a container /proc/stat does not describe the container's own usage:
// lxcfs serves LXC guests the raw counters of the host cores in their cpuset,
// and plain runtimes (Docker, k8s) expose the host's /proc outright. An idle
// container sharing a host core with a busy neighbor then reports near-100%
// CPU while doing nothing. The cgroup's own accounting (cpu.stat /
// cpuacct.usage) reflects only the container's processes, so when the agent
// runs inside a container we derive CPU% from that instead.
// File paths and hooks are variables so tests can point them at fixtures.
var (
cpuCgroupRoot = "/sys/fs/cgroup" // default cgroup v2 mount point
cpuCgroupMountinfo = "/proc/self/mountinfo"
cpuProcSelfCgroup = "/proc/self/cgroup"
cpuProcOneEnviron = "/proc/1/environ"
cpuDockerenvPath = "/.dockerenv"
cpuContainerenv = "/run/.containerenv"
cpuSystemdContPath = "/run/systemd/container"
cpuNumCPU = runtime.NumCPU
cpuNow = time.Now
)
// cpuUserHZ is the USER_HZ jiffies-per-second rate cpuacct.stat reports in.
const cpuUserHZ = 100
var (
containerOnce sync.Once
containerDetected bool
)
// inContainer reports whether the agent itself runs inside a container.
// The result is cached because it cannot change during the process lifetime.
func inContainer() bool {
containerOnce.Do(func() { containerDetected = detectContainer() })
return containerDetected
}
// detectContainer looks for the usual container markers.
func detectContainer() bool {
// set by systemd-nspawn, LXC, Podman and others
if os.Getenv("container") != "" {
return true
}
for _, p := range []string{cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath} {
if utils.FileExists(p) {
return true
}
}
// liblxc always puts container=lxc in the container init's environment,
// which survives on non-systemd guests such as Alpine LXC.
if data, err := os.ReadFile(cpuProcOneEnviron); err == nil {
if bytes.HasPrefix(data, []byte("container=")) ||
bytes.Contains(data, []byte("\x00container=")) {
return true
}
}
// an lxcfs mount means /proc/stat is virtualized with host core counters
if data, err := os.ReadFile(cpuCgroupMountinfo); err == nil &&
bytes.Contains(data, []byte(" - fuse.lxcfs ")) {
return true
}
return false
}
// cgroupCpuSample is one read of the container's cumulative CPU accounting.
type cgroupCpuSample struct {
usageUsec uint64
userUsec uint64
systemUsec uint64
cores float64 // usable CPU cores: affinity ∩ cpuset ∩ quota
at time.Time
}
var lastCgroupCpuSamples = make(map[uint16]cgroupCpuSample)
// init seeds the container CPU baseline so the first reported value is a real
// delta since startup rather than zero.
func init() {
if !inContainer() {
return
}
if s, ok := readContainerCpuSample(); ok {
s.at = cpuNow()
lastCgroupCpuSamples[60000] = s
}
}
// containerCpuMetrics derives CPU metrics from the agent's own cgroup
// accounting when running inside a container. It returns ok=false on plain
// hosts and whenever cgroup accounting is unreadable, so callers keep the
// /proc/stat fallback.
func containerCpuMetrics(cacheTimeMs uint16) (CpuMetrics, bool) {
if !inContainer() {
return CpuMetrics{}, false
}
cur, ok := readContainerCpuSample()
if !ok {
return CpuMetrics{}, false
}
cur.at = cpuNow()
prev, ok := lastCgroupCpuSamples[cacheTimeMs]
if !ok {
prev = lastCgroupCpuSamples[60000]
}
lastCgroupCpuSamples[cacheTimeMs] = cur
// No baseline yet, a backwards counter (cgroup recreated), or a
// non-positive clock delta: report zero this tick instead of guessing.
elapsedUsec := cur.at.Sub(prev.at).Microseconds()
if prev.at.IsZero() || elapsedUsec <= 0 || cur.usageUsec < prev.usageUsec {
return CpuMetrics{}, true
}
cores := cur.cores
if cores <= 0 {
cores = 1
}
window := float64(elapsedUsec) * cores
metrics := CpuMetrics{
Total: clampPercent(float64(cur.usageUsec-prev.usageUsec) / window * 100),
User: clampPercent(float64(cur.userUsec-prev.userUsec) / window * 100),
System: clampPercent(float64(cur.systemUsec-prev.systemUsec) / window * 100),
}
// cgroup accounting has no iowait/steal; everything not busy is idle.
metrics.Idle = clampPercent(100 - metrics.Total)
return metrics, true
}
// readContainerCpuSample reads the container's cumulative CPU usage, preferring
// the cgroup v2 unified hierarchy and falling back to the v1 cpuacct
// controller.
func readContainerCpuSample() (cgroupCpuSample, bool) {
if s, ok := readCgroupV2CpuSample(); ok {
return s, true
}
return readCgroupV1CpuSample()
}
// readCgroupV2CpuSample reads usage from the unified hierarchy's cpu.stat.
//
// The mount root is usually the right cgroup to read: inside a private cgroup
// namespace (LXC, default Docker) /sys/fs/cgroup already is the container's
// root cgroup, and its cpu.stat accounts for every process in the container,
// including siblings of the agent's own service cgroup. When the hierarchy is
// not namespaced (e.g. docker run --cgroupns=host) /proc/self/cgroup instead
// holds the container's real host-side path, which is joined onto the mount.
func readCgroupV2CpuSample() (cgroupCpuSample, bool) {
rel := selfCgroupPath("0::")
if rel == "" {
return cgroupCpuSample{}, false // no v2 membership; try v1
}
dir := cpuCgroupRoot
if mount := cgroupMountPoint("cgroup2", ""); mount != "" {
dir = mount
}
if rel != "/" && hasContainerRuntimeMarker(rel) {
if cand := filepath.Join(dir, rel); directoryExistsOK(cand) {
dir = cand
}
}
stat := filepath.Join(dir, "cpu.stat")
usage, ok := cgroupStatValue(stat, "usage_usec")
if !ok {
return cgroupCpuSample{}, false
}
s := cgroupCpuSample{usageUsec: usage, cores: cpuCgroupCores(dir)}
s.userUsec, _ = cgroupStatValue(stat, "user_usec")
s.systemUsec, _ = cgroupStatValue(stat, "system_usec")
return s, true
}
// readCgroupV1CpuSample reads usage from the legacy cpuacct controller.
// Runtimes bind-mount the container's own cpuacct directory at the hierarchy
// mount, so the mount root is normally already the container's cgroup; if the
// process's cgroup path still resolves below the mount (shared host view),
// that subdirectory is used instead.
func readCgroupV1CpuSample() (cgroupCpuSample, bool) {
mount := cgroupMountPoint("cgroup", "cpuacct")
if mount == "" {
return cgroupCpuSample{}, false
}
dir := mount
if rel := selfCgroupPath("cpuacct"); rel != "" && rel != "/" {
if cand := filepath.Join(mount, rel); utils.FileExists(filepath.Join(cand, "cpuacct.usage")) {
dir = cand
}
}
usageNs, ok := utils.ReadUintFile(filepath.Join(dir, "cpuacct.usage"))
if !ok {
return cgroupCpuSample{}, false
}
s := cgroupCpuSample{usageUsec: usageNs / 1000, cores: cpuCgroupCores(dir)}
// cpuacct.stat reports user/system in USER_HZ jiffies.
if v, ok := cgroupStatValue(filepath.Join(dir, "cpuacct.stat"), "user"); ok {
s.userUsec = v * 1e6 / cpuUserHZ
}
if v, ok := cgroupStatValue(filepath.Join(dir, "cpuacct.stat"), "system"); ok {
s.systemUsec = v * 1e6 / cpuUserHZ
}
return s, true
}
// selfCgroupPath returns the agent's cgroup path from /proc/self/cgroup: the
// path after "0::" for the v2 unified hierarchy, or the path of the entry
// whose controller list contains the given v1 controller (e.g. "cpuacct").
func selfCgroupPath(selector string) string {
data, err := os.ReadFile(cpuProcSelfCgroup)
if err != nil {
return ""
}
for _, line := range strings.Split(string(data), "\n") {
parts := strings.SplitN(line, ":", 3)
if len(parts) != 3 {
continue
}
if selector == "0::" {
if parts[0] == "0" && parts[1] == "" {
return parts[2]
}
continue
}
for _, ctrl := range strings.Split(parts[1], ",") {
if ctrl == selector {
return parts[2]
}
}
}
return ""
}
// cgroupMountPoint returns the mount point of a cgroup hierarchy from
// /proc/self/mountinfo: the cgroup2 mount for v2, or the cgroup mount whose
// super options list the wanted v1 controller.
func cgroupMountPoint(fstype, v1ctrl string) string {
data, err := os.ReadFile(cpuCgroupMountinfo)
if err != nil {
return ""
}
for _, line := range strings.Split(string(data), "\n") {
left, right, found := strings.Cut(line, " - ")
if !found {
continue
}
post := strings.Fields(right)
if len(post) == 0 || post[0] != fstype {
continue
}
if v1ctrl != "" && !mountOptHas(post, v1ctrl) {
continue
}
fields := strings.Fields(left)
if len(fields) >= 5 {
return unescapeMountPoint(fields[4])
}
}
return ""
}
// mountOptHas reports whether the comma-separated super options (field 3 after
// the " - " separator) contain opt.
func mountOptHas(post []string, opt string) bool {
if len(post) < 3 {
return false
}
for _, o := range strings.Split(post[2], ",") {
if o == opt {
return true
}
}
return false
}
// unescapeMountPoint decodes octal escapes (e.g. \040 for space) used in
// mountinfo paths.
func unescapeMountPoint(s string) string {
return strings.NewReplacer(`\040`, " ", `\011`, "\t", `\012`, "\n", `\134`, `\`).Replace(s)
}
// hasContainerRuntimeMarker reports whether a cgroup path looks like a real
// host-side container cgroup path (docker/k8s/lxc/podman), meaning the visible
// hierarchy is not namespaced and the path can be resolved under the mount.
func hasContainerRuntimeMarker(path string) bool {
for _, m := range []string{"docker", "kubepods", "lxc", "crio", "libpod", "containerd", "podman"} {
if strings.Contains(path, m) {
return true
}
}
return false
}
// cpuCgroupCores returns how many CPU cores the cgroup at dir may use: the
// smallest of the process affinity mask, the cgroup cpuset, and the CPU quota.
func cpuCgroupCores(dir string) float64 {
cores := float64(cpuNumCPU())
if n := cpusetCount(dir); n > 0 && n < cores {
cores = n
}
if q, ok := cpuQuotaCores(dir); ok && q < cores {
cores = q
}
if cores <= 0 {
cores = 1
}
return cores
}
// cpusetCount returns the number of CPUs in the cgroup's cpuset, e.g. "0-3" or
// "2,5-7". An empty or missing file means unconstrained.
func cpusetCount(dir string) float64 {
for _, name := range []string{"cpuset.cpus.effective", "cpuset.cpus"} {
raw, err := os.ReadFile(filepath.Join(dir, name))
if err != nil {
continue
}
if n := countCpuList(strings.TrimSpace(string(raw))); n > 0 {
return float64(n)
}
}
return 0
}
// countCpuList counts the CPUs in a Linux CPU list like "0-3,5,8-9".
func countCpuList(list string) int {
total := 0
for _, part := range strings.Split(list, ",") {
lo, hi, ranged := strings.Cut(part, "-")
a, err := strconv.Atoi(lo)
if err != nil {
continue
}
b := a
if ranged {
if v, err := strconv.Atoi(hi); err == nil {
b = v
}
}
if b >= a {
total += b - a + 1
}
}
return total
}
// cpuQuotaCores returns the cgroup's CPU quota in cores. v2 uses cpu.max
// ("<quota|max> <period>"), v1 uses cpu.cfs_quota_us / cpu.cfs_period_us.
func cpuQuotaCores(dir string) (float64, bool) {
if raw, err := os.ReadFile(filepath.Join(dir, "cpu.max")); err == nil {
fields := strings.Fields(string(raw))
if len(fields) == 2 && fields[0] != "max" {
if quota, err := strconv.ParseFloat(fields[0], 64); err == nil && quota > 0 {
if period, err := strconv.ParseFloat(fields[1], 64); err == nil && period > 0 {
return quota / period, true
}
}
}
}
if quota, ok := readCgroupInt(filepath.Join(dir, "cpu.cfs_quota_us")); ok && quota > 0 {
if period, ok := readCgroupInt(filepath.Join(dir, "cpu.cfs_period_us")); ok && period > 0 {
return float64(quota) / float64(period), true
}
}
return 0, false
}
// cgroupStatValue returns the value of key in a cgroup "key value" stat file.
func cgroupStatValue(path, key string) (uint64, bool) {
data, err := os.ReadFile(path)
if err != nil {
return 0, false
}
for _, line := range strings.Split(string(data), "\n") {
name, value, found := strings.Cut(line, " ")
if !found || name != key {
continue
}
v, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64)
return v, err == nil
}
return 0, false
}
// readCgroupInt reads a file containing a single signed integer
// (cpu.cfs_quota_us is -1 when no quota is set).
func readCgroupInt(path string) (int64, bool) {
data, err := os.ReadFile(path)
if err != nil {
return 0, false
}
v, err := strconv.ParseInt(strings.TrimSpace(string(data)), 10, 64)
return v, err == nil
}
// directoryExistsOK reports whether path is a directory.
func directoryExistsOK(path string) bool {
ok, _ := directoryExists(path)
return ok
}
+337
View File
@@ -0,0 +1,337 @@
//go:build testing && linux
package agent
import (
"os"
"path/filepath"
"sync"
"testing"
"time"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
// swapCpuContainerSeams points every container-detection and cgroup path at
// empty fixtures under a temp dir, then restores them on cleanup.
func swapCpuContainerSeams(t *testing.T) {
t.Helper()
backup := struct {
root, mountinfo, selfCgroup, oneEnviron, dockerenv, containerenv, systemdCont string
numCPU func() int
now func() time.Time
}{
cpuCgroupRoot, cpuCgroupMountinfo, cpuProcSelfCgroup, cpuProcOneEnviron,
cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath, cpuNumCPU, cpuNow,
}
detected := containerDetected
samples := lastCgroupCpuSamples
env, hadEnv := os.LookupEnv("container")
t.Cleanup(func() {
cpuCgroupRoot, cpuCgroupMountinfo, cpuProcSelfCgroup, cpuProcOneEnviron = backup.root, backup.mountinfo, backup.selfCgroup, backup.oneEnviron
cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath = backup.dockerenv, backup.containerenv, backup.systemdCont
cpuNumCPU, cpuNow = backup.numCPU, backup.now
containerOnce = sync.Once{}
containerDetected = detected
lastCgroupCpuSamples = samples
if hadEnv {
os.Setenv("container", env)
}
})
containerOnce = sync.Once{}
containerDetected = false
lastCgroupCpuSamples = make(map[uint16]cgroupCpuSample)
os.Unsetenv("container")
tmp := t.TempDir()
cpuCgroupRoot = filepath.Join(tmp, "cgroup")
cpuCgroupMountinfo = filepath.Join(tmp, "mountinfo")
cpuProcSelfCgroup = filepath.Join(tmp, "self-cgroup")
cpuProcOneEnviron = filepath.Join(tmp, "one-environ")
cpuDockerenvPath = filepath.Join(tmp, "dockerenv")
cpuContainerenv = filepath.Join(tmp, "containerenv")
cpuSystemdContPath = filepath.Join(tmp, "systemd-container")
}
func writeCpuFixture(t *testing.T, path, contents string) {
t.Helper()
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
require.NoError(t, os.WriteFile(path, []byte(contents), 0o644))
}
// fakeNow installs a controllable clock and returns a function to advance it.
func fakeNow(t *testing.T) func(time.Duration) {
t.Helper()
cur := time.Unix(1_700_000_000, 0)
cpuNow = func() time.Time { return cur }
return func(d time.Duration) { cur = cur.Add(d) }
}
func TestDetectContainer(t *testing.T) {
tests := []struct {
name string
setup func(t *testing.T)
want bool
}{
{"plain host", func(t *testing.T) {}, false},
{"container env", func(t *testing.T) { t.Setenv("container", "lxc") }, true},
{".dockerenv", func(t *testing.T) { writeCpuFixture(t, cpuDockerenvPath, "") }, true},
{".containerenv", func(t *testing.T) { writeCpuFixture(t, cpuContainerenv, "") }, true},
{"systemd container", func(t *testing.T) { writeCpuFixture(t, cpuSystemdContPath, "lxc\n") }, true},
{"init environ container=lxc", func(t *testing.T) {
writeCpuFixture(t, cpuProcOneEnviron, "PATH=/sbin\x00container=lxc\x00HOME=/root\x00")
}, true},
{"init environ without marker", func(t *testing.T) {
writeCpuFixture(t, cpuProcOneEnviron, "PATH=/sbin\x00HOME=/root\x00")
}, false},
{"lxcfs serving /proc", func(t *testing.T) {
writeCpuFixture(t, cpuCgroupMountinfo,
"31 25 0:28 / /proc/stat rw,nosuid,nodev,relatime - fuse.lxcfs lxcfs rw,user_id=0,group_id=0\n")
}, true},
{"cgroup-only mountinfo", func(t *testing.T) {
writeCpuFixture(t, cpuCgroupMountinfo,
"36 25 0:32 / /sys/fs/cgroup rw - cgroup2 cgroup2 rw,nsdelegate\n")
}, false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
swapCpuContainerSeams(t)
tt.setup(t)
assert.Equal(t, tt.want, detectContainer())
})
}
}
func TestReadCgroupV2CpuSample(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755))
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"),
"usage_usec 3000000\nuser_usec 2000000\nsystem_usec 1000000\nnr_throttled 7\n")
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpuset.cpus.effective"), "2,5-7\n")
cpuNumCPU = func() int { return 8 }
s, ok := readCgroupV2CpuSample()
require.True(t, ok)
assert.EqualValues(t, 3000000, s.usageUsec)
assert.EqualValues(t, 2000000, s.userUsec)
assert.EqualValues(t, 1000000, s.systemUsec)
assert.InDelta(t, 4, s.cores, 0.001) // cpuset 2,5-7 = 4 cores
}
// In a namespaced container the agent may sit in a sub-cgroup (e.g. a systemd
// service); the mount root still accounts for the whole container and must win.
func TestReadCgroupV2PrefersContainerRoot(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuProcSelfCgroup, "0::/system.slice/beszel-agent.service\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 9000\n")
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "system.slice/beszel-agent.service/cpu.stat"), "usage_usec 5\n")
s, ok := readCgroupV2CpuSample()
require.True(t, ok)
assert.EqualValues(t, 9000, s.usageUsec)
}
// Without a cgroup namespace the mount shows the real host hierarchy and the
// container's own path (docker/kubepods/lxc markers) resolves under it.
func TestReadCgroupV2ResolvesRuntimePath(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuProcSelfCgroup, "0::/system.slice/docker-deadbeef.scope\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 9000\n")
sub := filepath.Join(cpuCgroupRoot, "system.slice/docker-deadbeef.scope")
writeCpuFixture(t, filepath.Join(sub, "cpu.stat"), "usage_usec 5\n")
writeCpuFixture(t, filepath.Join(sub, "cpu.max"), "100000 100000\n")
s, ok := readCgroupV2CpuSample()
require.True(t, ok)
assert.EqualValues(t, 5, s.usageUsec)
assert.InDelta(t, 1, s.cores, 0.001) // cpu.max quota of 1 core
}
func TestReadCgroupV1CpuSample(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuProcSelfCgroup, "3:cpuacct:/\n2:memory:/\n")
v1 := filepath.Join(t.TempDir(), "cpuacct")
writeCpuFixture(t, cpuCgroupMountinfo,
"30 25 0:26 / "+v1+" rw,nosuid,nodev,noexec,relatime - cgroup cgroup rw,cpuacct\n")
writeCpuFixture(t, filepath.Join(v1, "cpuacct.usage"), "2000000000\n")
writeCpuFixture(t, filepath.Join(v1, "cpuacct.stat"), "user 100\nsystem 50\n")
cpuNumCPU = func() int { return 4 }
s, ok := readContainerCpuSample() // no 0:: line -> falls through to v1
require.True(t, ok)
assert.EqualValues(t, 2000000, s.usageUsec) // ns -> usec
assert.EqualValues(t, 1000000, s.userUsec) // 100 jiffies * 1e6/100
assert.EqualValues(t, 500000, s.systemUsec) // 50 jiffies
assert.InDelta(t, 4, s.cores, 0.001)
}
func TestContainerCpuMetricsMath(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuDockerenvPath, "")
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755))
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpuset.cpus.effective"), "0-3\n")
cpuNumCPU = func() int { return 8 }
advance := fakeNow(t)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"),
"usage_usec 1000000\nuser_usec 600000\nsystem_usec 400000\n")
m, ok := containerCpuMetrics(60000)
require.True(t, ok)
assert.Zero(t, m.Total) // first call only seeds the baseline
// 1s elapsed, container burned 2 core-seconds on 4 usable cores
advance(time.Second)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"),
"usage_usec 3000000\nuser_usec 1600000\nsystem_usec 900000\n")
m, ok = containerCpuMetrics(60000)
require.True(t, ok)
assert.InDelta(t, 50, m.Total, 0.01)
assert.InDelta(t, 25, m.User, 0.01)
assert.InDelta(t, 12.5, m.System, 0.01)
assert.Zero(t, m.Iowait)
assert.Zero(t, m.Steal)
assert.InDelta(t, 50, m.Idle, 0.01)
}
func TestContainerCpuMetricsHonorsQuota(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuDockerenvPath, "")
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755))
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.max"), "200000 100000\n") // 2 cores
cpuNumCPU = func() int { return 8 }
advance := fakeNow(t)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 1000000\n")
containerCpuMetrics(60000)
advance(time.Second)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 2000000\n")
m, ok := containerCpuMetrics(60000)
require.True(t, ok)
assert.InDelta(t, 50, m.Total, 0.01) // 1 core-second against a 2-core quota
}
func TestContainerCpuMetricsZeroAndBackwardDelta(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuDockerenvPath, "")
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755))
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 5000000\n")
cpuNumCPU = func() int { return 4 }
advance := fakeNow(t)
// seed the baseline, then do not advance the clock: elapsed <= 0
containerCpuMetrics(60000)
m, ok := containerCpuMetrics(60000)
require.True(t, ok)
assert.Zero(t, m.Total)
// counter goes backwards (cgroup recreated): report zero and re-baseline
advance(time.Second)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 100000\n")
m, ok = containerCpuMetrics(60000)
require.True(t, ok)
assert.Zero(t, m.Total)
// next tick measures from the new baseline, not the stale one
advance(time.Second)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 1100000\n")
m, ok = containerCpuMetrics(60000)
require.True(t, ok)
assert.InDelta(t, 25, m.Total, 0.01) // 1e6 usec / (1s * 4 cores)
}
func TestContainerCpuMetricsFallbacks(t *testing.T) {
t.Run("not in container", func(t *testing.T) {
swapCpuContainerSeams(t)
_, ok := containerCpuMetrics(60000)
assert.False(t, ok)
})
t.Run("in container without cgroup accounting", func(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuDockerenvPath, "")
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
// cpuCgroupRoot has no cpu.stat
_, ok := containerCpuMetrics(60000)
assert.False(t, ok)
})
}
// The host path must keep reporting through gopsutil untouched.
func TestGetCpuMetricsHostFallback(t *testing.T) {
swapCpuContainerSeams(t)
m, err := getCpuMetrics(60000)
require.NoError(t, err)
assert.GreaterOrEqual(t, m.Total, 0.0)
assert.LessOrEqual(t, m.Total, 100.0)
}
// Inside a container getCpuMetrics must report the cgroup-derived value, not
// the host core counters from /proc/stat.
func TestGetCpuMetricsPrefersCgroup(t *testing.T) {
swapCpuContainerSeams(t)
writeCpuFixture(t, cpuDockerenvPath, "")
writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n")
writeCpuFixture(t, cpuCgroupMountinfo, "")
require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755))
cpuNumCPU = func() int { return 4 }
advance := fakeNow(t)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 0\n")
m, err := getCpuMetrics(60000)
require.NoError(t, err)
assert.Zero(t, m.Total)
advance(time.Second)
writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 2000000\n")
m, err = getCpuMetrics(60000)
require.NoError(t, err)
assert.InDelta(t, 50, m.Total, 0.01)
}
func TestCountCpuList(t *testing.T) {
assert.Equal(t, 4, countCpuList("0-3"))
assert.Equal(t, 4, countCpuList("2,5-7"))
assert.Equal(t, 1, countCpuList("2"))
assert.Equal(t, 0, countCpuList(""))
assert.Equal(t, 0, countCpuList("max"))
assert.Equal(t, 6, countCpuList("0-3,8-9"))
}
func TestCpuQuotaCores(t *testing.T) {
dir := t.TempDir()
_, ok := cpuQuotaCores(dir)
assert.False(t, ok) // no quota files
writeCpuFixture(t, filepath.Join(dir, "cpu.max"), "max 100000\n")
_, ok = cpuQuotaCores(dir)
assert.False(t, ok) // unlimited
writeCpuFixture(t, filepath.Join(dir, "cpu.max"), "150000 100000\n")
q, ok := cpuQuotaCores(dir)
require.True(t, ok)
assert.InDelta(t, 1.5, q, 0.001)
// v1 files
v1 := t.TempDir()
writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_quota_us"), "-1\n")
writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_period_us"), "100000\n")
_, ok = cpuQuotaCores(v1)
assert.False(t, ok)
writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_quota_us"), "50000\n")
q, ok = cpuQuotaCores(v1)
require.True(t, ok)
assert.InDelta(t, 0.5, q, 0.001)
}
+9
View File
@@ -0,0 +1,9 @@
//go:build !linux
package agent
// containerCpuMetrics is Linux-only (cgroup accounting). Other platforms keep
// the gopsutil /proc path.
func containerCpuMetrics(uint16) (CpuMetrics, bool) {
return CpuMetrics{}, false
}
@@ -1,4 +1,4 @@
import { type ReactNode, useCallback, useEffect, useRef, useState } from "react"
import { useCallback, useEffect, useRef, useState } from "react"
import { Trans, useLingui } from "@lingui/react/macro"
import { useStore } from "@nanostores/react"
import { pb } from "@/lib/api"
@@ -23,11 +23,10 @@ import { Input } from "@/components/ui/input"
import { Label } from "@/components/ui/label"
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"
import { Textarea } from "@/components/ui/textarea"
import { ChevronDownIcon, GlobeIcon, ListIcon, type LucideIcon, PlusIcon, SearchIcon, ServerIcon } from "lucide-react"
import { ChevronDownIcon, ListIcon, PlusIcon, SearchIcon, ServerIcon } from "lucide-react"
import { useToast } from "@/components/ui/use-toast"
import { $systems } from "@/lib/stores"
import { cn, supportsNetworkMonitors } from "@/lib/utils"
import { getMonitorTarget } from "@/lib/network-monitor-utils"
import type { NetworkMonitorRecord } from "@/types"
import * as v from "valibot"
@@ -203,7 +202,6 @@ export function SystemMultiSelect({
className,
systemIds,
placeholder,
canSelectMore,
}: {
id: string
selectedSystemIds: Set<string>
@@ -213,107 +211,8 @@ export function SystemMultiSelect({
/** Limit the options to these systems. Defaults to all systems that support network monitors. */
systemIds?: string[]
placeholder?: string
canSelectMore?: boolean
}) {
const systems = useStore($systems)
const { t } = useLingui()
const options = systems
.filter((system) => (systemIds ? systemIds.includes(system.id) : supportsNetworkMonitors(system)))
.map((system) => ({ id: system.id, label: system.name }))
return (
<MultiSelect
id={id}
options={options}
selectedIds={selectedSystemIds}
onChange={onChange}
disabled={disabled}
className={className}
icon={ServerIcon}
canSelectMore={canSelectMore}
placeholder={placeholder ?? t`Select systems`}
searchPlaceholder={t`Search systems`}
emptyText={<Trans>No systems found.</Trans>}
/>
)
}
/** Pick monitors by target, e.g. other targets on the same system to compare against. */
export function MonitorMultiSelect({
id,
monitors,
selectedMonitorIds,
onChange,
disabled,
className,
placeholder,
canSelectMore,
}: {
id: string
monitors: NetworkMonitorRecord[]
selectedMonitorIds: Set<string>
onChange: (ids: Set<string>) => void
disabled?: boolean
className?: string
placeholder?: string
canSelectMore?: boolean
}) {
const { t } = useLingui()
const options = monitors
.map((monitor) => ({ id: monitor.id, label: getMonitorTarget(monitor), server: monitor.server }))
.sort((a, b) => a.label.localeCompare(b.label))
return (
<MultiSelect
id={id}
options={options}
selectedIds={selectedMonitorIds}
onChange={onChange}
disabled={disabled}
className={cn("ps-9.5", className)}
icon={GlobeIcon}
canSelectMore={canSelectMore}
placeholder={placeholder ?? t`Select targets`}
searchPlaceholder={t`Search targets`}
emptyText={<Trans>No targets found.</Trans>}
renderOption={(option) => (
<>
<span className="truncate">{option.label}</span>
{option.server && <span className="ms-auto shrink-0 text-xs text-muted-foreground">{option.server}</span>}
</>
)}
/>
)
}
type MultiSelectOption = { id: string; label: string }
function MultiSelect<T extends MultiSelectOption>({
id,
options,
selectedIds,
onChange,
disabled,
className,
icon: Icon,
placeholder,
searchPlaceholder,
emptyText,
renderOption = (option) => <span className="truncate">{option.label}</span>,
canSelectMore = true,
}: {
id: string
options: T[]
selectedIds: Set<string>
onChange: (ids: Set<string>) => void
disabled?: boolean
className?: string
icon: LucideIcon
placeholder: string
searchPlaceholder: string
emptyText: ReactNode
renderOption?: (option: T) => ReactNode
/** False once the selection is full; only already selected options can then be toggled. */
canSelectMore?: boolean
}) {
const { t } = useLingui()
const [search, setSearch] = useState("")
const searchRef = useRef<HTMLInputElement>(null)
@@ -326,15 +225,19 @@ function MultiSelect<T extends MultiSelectOption>({
}, [])
const contentRef = useRef<HTMLDivElement>(null)
const query = search.trim().toLocaleLowerCase()
const filteredOptions = options.filter((option) => option.label.toLocaleLowerCase().includes(query))
const allSelected = filteredOptions.every((option) => selectedIds.has(option.id))
const anySelected = filteredOptions.some((option) => selectedIds.has(option.id))
const filteredSystems = systems.filter(
(system) =>
(systemIds ? systemIds.includes(system.id) : supportsNetworkMonitors(system)) &&
system.name.toLocaleLowerCase().includes(query)
)
const allSelected = filteredSystems.every((system) => selectedSystemIds.has(system.id))
const anySelected = filteredSystems.some((system) => selectedSystemIds.has(system.id))
const selectFiltered = (selected: boolean) => {
const next = new Set(selectedIds)
for (const option of filteredOptions) {
if (selected) next.add(option.id)
else next.delete(option.id)
const next = new Set(selectedSystemIds)
for (const system of filteredSystems) {
if (selected) next.add(system.id)
else next.delete(system.id)
}
onChange(next)
}
@@ -348,13 +251,13 @@ function MultiSelect<T extends MultiSelectOption>({
variant="outline"
className={cn("relative w-full min-w-0 ps-10 pe-10 justify-start font-normal text-start", className)}
>
<Icon className="size-3.5 absolute start-4 top-1/2 -translate-y-1/2 opacity-85" />
<ServerIcon className="size-3.5 absolute start-4 top-1/2 -translate-y-1/2 opacity-85" />
<span className="truncate">
{selectedIds.size === 0
? placeholder
: selectedIds.size === 1
? options.find((option) => selectedIds.has(option.id))?.label
: t`${selectedIds.size} selected`}
{selectedSystemIds.size === 0
? (placeholder ?? t`Select systems`)
: selectedSystemIds.size === 1
? systems.find((s) => selectedSystemIds.has(s.id))?.name
: t`${selectedSystemIds.size} selected`}
</span>
<ChevronDownIcon className="size-4 absolute end-4 top-1/2 -translate-y-1/2 opacity-50" />
</Button>
@@ -377,8 +280,8 @@ function MultiSelect<T extends MultiSelectOption>({
ref={focusSearchOnMount}
value={search}
onChange={(event) => setSearch(event.target.value)}
placeholder={searchPlaceholder}
aria-label={searchPlaceholder}
placeholder={t`Search systems`}
aria-label={t`Search systems`}
className="h-10 min-w-0 rounded-none border-0 bg-transparent px-0 shadow-none focus-visible:ring-0 focus-visible:ring-offset-0"
onKeyDown={(event) => {
if (event.key === "Escape") return
@@ -400,7 +303,7 @@ function MultiSelect<T extends MultiSelectOption>({
<div className="flex items-center">
<DropdownMenuItem
className="px-1.5 py-1 text-xs text-muted-foreground"
disabled={!filteredOptions.length || allSelected || !canSelectMore}
disabled={!filteredSystems.length || allSelected}
onSelect={(event) => {
event.preventDefault()
selectFiltered(true)
@@ -422,29 +325,32 @@ function MultiSelect<T extends MultiSelectOption>({
{query ? <Trans>Clear matches</Trans> : <Trans>Clear all</Trans>}
</DropdownMenuItem>
</div>
<span className="px-1.5 text-xs tabular-nums text-muted-foreground">{t`${selectedIds.size} selected`}</span>
<span className="px-1.5 text-xs tabular-nums text-muted-foreground">
{t`${selectedSystemIds.size} selected`}
</span>
</div>
</div>
<div className="min-h-0 overflow-y-auto">
{filteredOptions.length === 0 && (
<output className="block px-2.5 py-3 text-sm text-muted-foreground">{emptyText}</output>
{filteredSystems.length === 0 && (
<output className="block px-2.5 py-3 text-sm text-muted-foreground">
<Trans>No systems found.</Trans>
</output>
)}
{filteredOptions.map((option) => (
{filteredSystems.map((sys) => (
<DropdownMenuCheckboxItem
key={option.id}
checked={selectedIds.has(option.id)}
disabled={!canSelectMore && !selectedIds.has(option.id)}
key={sys.id}
checked={selectedSystemIds.has(sys.id)}
onSelect={(event) => event.preventDefault()}
onCheckedChange={(checked) => {
const next = new Set(selectedIds)
if (checked) next.add(option.id)
else next.delete(option.id)
const next = new Set(selectedSystemIds)
if (checked) next.add(sys.id)
else next.delete(sys.id)
onChange(next)
}}
className="group min-w-0 gap-2.5 py-2 ps-2.5"
indicatorClassName="static size-4 shrink-0 rounded border border-input group-data-[state=checked]:border-primary group-data-[state=checked]:bg-primary group-data-[state=checked]:text-primary-foreground [&_svg]:size-3"
>
{renderOption(option)}
<span className="truncate">{sys.name}</span>
</DropdownMenuCheckboxItem>
))}
</div>
@@ -39,7 +39,7 @@ import { SystemStatus } from "@/lib/enums"
import { $allSystemsById, $direction, $textMeasureVersion, $userSettings, getUserChartTime } from "@/lib/stores"
import { cn, formatShortDate, isVisuallyLonger, matchesFilterGroups, parseFilterGroups, parseSemVer } from "@/lib/utils"
import type { ChartOptions, MonitorCertInfo, NetworkMonitorRecord } from "@/types"
import { AddMonitorDialog, EditMonitorDialog, MonitorMultiSelect, SystemMultiSelect } from "./monitor-dialog"
import { AddMonitorDialog, EditMonitorDialog, SystemMultiSelect } from "./monitor-dialog"
import {
ArrowDownIcon,
ArrowLeftRightIcon,
@@ -67,8 +67,7 @@ import {
import { Sheet, SheetContent, SheetDescription, SheetHeader, SheetTitle } from "@/components/ui/sheet"
import ChartTimeSelect from "@/components/charts/chart-time-select"
import { LossChart, AvgMinMaxResponseChart, ResponseChart } from "@/components/routes/system/charts/monitors-charts"
import { getMonitorCompareState } from "@/lib/monitor-compare"
import { useCompareMonitors, useNetworkMonitorStats } from "@/lib/use-network-monitors"
import { useMatchingMonitors, useNetworkMonitorStats } from "@/lib/use-network-monitors"
import { useStore } from "@nanostores/react"
import { atom } from "nanostores"
import { Separator } from "../ui/separator"
@@ -471,7 +470,6 @@ export default function NetworkMonitorsTableNew({
visibleColumnsKey={visibleColumnsKey}
rowSelection={rowSelection}
isLoading={isLoading}
includesAllSystems={!systemId}
/>
</div>
</Card>
@@ -485,7 +483,6 @@ const NetworkMonitorsTable = memo(function NetworkMonitorTable({
visibleColumnsKey,
rowSelection,
isLoading,
includesAllSystems,
}: {
table: TableType<NetworkMonitorRecord>
rows: Row<NetworkMonitorRecord>[]
@@ -493,8 +490,6 @@ const NetworkMonitorsTable = memo(function NetworkMonitorTable({
visibleColumnsKey: string
rowSelection: RowSelectionState
isLoading: boolean
/** The table lists every system's monitors, so the sheet can compare without fetching. */
includesAllSystems: boolean
}) {
const scrollRef = useRef<HTMLDivElement>(null)
const [sheetOpen, setSheetOpen] = useState(false)
@@ -565,8 +560,6 @@ const NetworkMonitorsTable = memo(function NetworkMonitorTable({
setSheetOpen(nextOpen)
}}
monitor={activeMonitor}
monitors={table.options.data}
includesAllSystems={includesAllSystems}
/>
</div>
)
@@ -637,29 +630,16 @@ function NetworkMonitorSheet({
open,
onOpenChange,
monitor,
monitors,
includesAllSystems,
}: {
open: boolean
onOpenChange: (open: boolean) => void
monitor?: NetworkMonitorRecord
monitors: NetworkMonitorRecord[]
includesAllSystems: boolean
}) {
if (!monitor) {
return null
}
return (
<NetworkMonitorSheetContent
key={monitor.system}
open={open}
onOpenChange={onOpenChange}
monitor={monitor}
monitors={monitors}
includesAllSystems={includesAllSystems}
/>
)
return <NetworkMonitorSheetContent key={monitor.system} open={open} onOpenChange={onOpenChange} monitor={monitor} />
}
const certExpiryTextColors = { ok: "", warning: "text-yellow-600 dark:text-yellow-500", critical: "text-red-500" }
@@ -696,16 +676,10 @@ function NetworkMonitorSheetContent({
open,
onOpenChange,
monitor,
monitors,
includesAllSystems,
}: {
open: boolean
onOpenChange: (open: boolean) => void
monitor: NetworkMonitorRecord
/** Table monitors; used to find other targets on the same system to compare against. */
monitors: NetworkMonitorRecord[]
/** Whether `monitors` covers every system, so other systems' monitors needn't be fetched. */
includesAllSystems: boolean
}) {
// Keep monitor exploration independent of the system charts' time range.
const [chartTimeStore] = useState(() => {
@@ -717,7 +691,8 @@ function NetworkMonitorSheetContent({
const systems = useStore($allSystemsById)
const system = systems[monitor.system]
const [compareTargetIds, setCompareTargetIds] = useState<Set<string>>(() => new Set())
// Same target probed from other systems, for side-by-side comparison (#2385).
const matchingMonitors = useMatchingMonitors(monitor, open)
const [compareSystemIds, setCompareSystemIds] = useState<Set<string>>(() => new Set())
// Scoped to this sheet so a filter doesn't carry over to other monitors' sheets.
const [compareFilterStore, setCompareFilterStore] = useState(() => atom(""))
@@ -726,25 +701,16 @@ function NetworkMonitorSheetContent({
if (compareMonitorId !== monitor.id) {
setCompareMonitorId(monitor.id)
setCompareSystemIds(new Set())
setCompareTargetIds(new Set())
setCompareFilterStore(atom(""))
}
// Other systems' monitors come from the table when it lists every system, otherwise from one fetch.
const fetchedMonitors = useCompareMonitors(monitor.system, monitor.protocol, open && !includesAllSystems)
const compare = useMemo(
() =>
getMonitorCompareState({
monitor,
localMonitors: monitors,
otherMonitors: includesAllSystems ? monitors : fetchedMonitors,
selectedSystemIds: compareSystemIds,
selectedTargetIds: compareTargetIds,
getSystemName: (id) => systems[id]?.name ?? id,
}),
[monitor, monitors, includesAllSystems, fetchedMonitors, compareSystemIds, compareTargetIds, systems]
const matchingSystemIds = useMemo(() => matchingMonitors.map((m) => m.system), [matchingMonitors])
// The opened system is always charted; the picker only adds other systems to compare against.
const compareMonitors = useMemo(
() => [monitor, ...matchingMonitors.filter((m) => compareSystemIds.has(m.system))],
[monitor, matchingMonitors, compareSystemIds]
)
const { compareMonitors } = compare
const comparing = compareMonitors.length > 1
const getSystemName = useCallback((m: NetworkMonitorRecord) => systems[m.system]?.name ?? m.system, [systems])
const monitorStats = useNetworkMonitorStats({
systemId: monitor.system,
@@ -800,31 +766,21 @@ function NetworkMonitorSheetContent({
<div className="grid gap-4">
<div className="flex flex-wrap items-center gap-2">
<ChartTimeSelect
className="bg-card flex-1 min-w-0 basis-full sm:basis-0"
className="bg-card flex-1 basis-48"
agentVersion={chartData.agentVersion}
chartTimeStore={chartTimeStore}
allowRealtime={false}
/>
<MonitorMultiSelect
id="monitor-compare-targets"
className="flex-1 min-w-0 basis-full sm:basis-0 bg-card"
monitors={compare.targetOptions}
selectedMonitorIds={compare.selectedTargetIds}
onChange={setCompareTargetIds}
disabled={compare.targetOptions.length === 0}
canSelectMore={compare.canAddTarget}
placeholder={t`Compare with other targets`}
/>
<SystemMultiSelect
id="monitor-compare-systems"
className="flex-1 min-w-0 basis-full sm:basis-0 bg-card"
systemIds={compare.systemOptions}
selectedSystemIds={compare.selectedSystemIds}
onChange={setCompareSystemIds}
disabled={compare.systemOptions.length === 0}
canSelectMore={compare.canAddSystem}
placeholder={t`Compare with other systems`}
/>
{matchingMonitors.length > 0 && (
<SystemMultiSelect
id="monitor-compare-systems"
className="w-full sm:w-1/3 shrink-0 bg-card"
systemIds={matchingSystemIds}
selectedSystemIds={compareSystemIds}
onChange={setCompareSystemIds}
placeholder={t`Compare with other systems`}
/>
)}
</div>
{comparing ? (
<>
@@ -834,7 +790,7 @@ function NetworkMonitorSheetContent({
monitors={compareMonitors}
chartData={chartData}
empty={!hasMonitorStats}
getLabel={compare.getLabel}
getLabel={getSystemName}
filterStore={compareFilterStore}
/>
<LossChart
@@ -843,7 +799,7 @@ function NetworkMonitorSheetContent({
monitors={compareMonitors}
chartData={chartData}
empty={!hasMonitorStats}
getLabel={compare.getLabel}
getLabel={getSystemName}
filterStore={compareFilterStore}
/>
</>
@@ -1,173 +0,0 @@
import { expect, test } from "bun:test"
import type { NetworkMonitorRecord } from "@/types"
import {
getMonitorCompareState,
getMonitorIdentityKey,
getMonitorTarget,
MAX_COMPARE_MONITORS,
} from "./monitor-compare"
function mon(id: string, system: string, target: string, extra: Partial<NetworkMonitorRecord> = {}) {
return { id, system, target, protocol: "icmp", port: 0, server: "", interval: 30, ...extra } as NetworkMonitorRecord
}
const systemNames: Record<string, string> = { a: "Alpha", b: "Bravo", c: "Charlie" }
function state(
monitor: NetworkMonitorRecord,
monitors: NetworkMonitorRecord[],
selectedSystemIds: string[] = [],
selectedTargetIds: string[] = []
) {
return getMonitorCompareState({
monitor,
localMonitors: monitors,
otherMonitors: monitors,
selectedSystemIds: new Set(selectedSystemIds),
selectedTargetIds: new Set(selectedTargetIds),
getSystemName: (id) => systemNames[id] ?? id,
})
}
const ids = (monitors: NetworkMonitorRecord[]) => monitors.map((m) => m.id)
// a probes one and two; b probes one and two; c probes only one
const a1 = mon("a1", "a", "one.example")
const a2 = mon("a2", "a", "two.example")
const b1 = mon("b1", "b", "one.example")
const b2 = mon("b2", "b", "two.example")
const c1 = mon("c1", "c", "one.example")
const all = [a1, a2, b1, b2, c1]
test("formats tcp targets with their port", () => {
expect(getMonitorTarget({ target: "example.com", protocol: "icmp", port: 0 })).toBe("example.com")
expect(getMonitorTarget({ target: "example.com", protocol: "tcp", port: 443 })).toBe("example.com:443")
expect(getMonitorTarget({ target: "::1", protocol: "tcp", port: 22 })).toBe("[::1]:22")
})
test("identity ignores the system but not protocol, port or server", () => {
expect(getMonitorIdentityKey(a1)).toBe(getMonitorIdentityKey(b1))
expect(getMonitorIdentityKey(a1)).not.toBe(getMonitorIdentityKey({ ...a1, protocol: "http" }))
expect(getMonitorIdentityKey(a1)).not.toBe(getMonitorIdentityKey({ ...a1, port: 80 }))
expect(getMonitorIdentityKey(a1)).not.toBe(getMonitorIdentityKey({ ...a1, server: "1.1.1.1" }))
})
test("with nothing selected only the opened monitor is charted", () => {
const s = state(a1, all)
expect(s.systemOptions).toEqual(["b", "c"])
expect(ids(s.targetOptions)).toEqual(["a2"])
expect(ids(s.compareMonitors)).toEqual(["a1"])
})
test("only offers targets and systems with the same protocol", () => {
const aHttp = mon("aHttp", "a", "https://two.example", { protocol: "http" })
const bTcp = mon("bTcp", "b", "one.example", { protocol: "tcp", port: 443 })
const s = state(a1, [a1, a2, aHttp, bTcp, c1])
expect(ids(s.targetOptions)).toEqual(["a2"])
expect(s.systemOptions).toEqual(["c"])
})
test("comparing targets charts them on the opened system, labelled by target", () => {
const s = state(a1, all, [], ["a2"])
expect(ids(s.compareMonitors)).toEqual(["a1", "a2"])
expect(s.getLabel(a1)).toBe("one.example")
expect(s.getLabel(a2)).toBe("two.example")
})
test("comparing systems charts the opened target on them, labelled by system", () => {
const s = state(a1, all, ["b", "c"])
expect(ids(s.compareMonitors)).toEqual(["a1", "b1", "c1"])
expect(s.getLabel(a1)).toBe("Alpha")
expect(s.getLabel(c1)).toBe("Charlie")
})
test("each compared target is charted on every selected system", () => {
const s = state(a1, all, ["b"], ["a2"])
expect(ids(s.compareMonitors)).toEqual(["a1", "a2", "b1", "b2"])
expect(s.getLabel(b2)).toBe("Bravo · two.example")
})
test("selecting a system limits targets to ones it also probes", () => {
const s = state(a1, all, ["c"])
expect(ids(s.targetOptions)).toEqual([])
})
test("selecting a target limits systems to ones that also probe it", () => {
const s = state(a1, all, [], ["a2"])
expect(s.systemOptions).toEqual(["b"])
})
test("the pickers can't produce conflicting selections, but targets win if they conflict", () => {
const s = state(a1, all, ["c"], ["a2"])
expect(s.selectedSystemIds.size).toBe(0)
expect([...s.selectedTargetIds]).toEqual(["a2"])
expect(ids(s.compareMonitors)).toEqual(["a1", "a2"])
})
test("selected systems that are no longer options are ignored", () => {
const s = state(a1, all, ["gone"])
expect(s.selectedSystemIds.size).toBe(0)
expect(ids(s.compareMonitors)).toEqual(["a1"])
})
test("DNS lookups of the same name against different servers get the server in their label", () => {
const d1 = mon("d1", "a", "example.com", { protocol: "dns", server: "1.1.1.1" })
const d2 = mon("d2", "a", "example.com", { protocol: "dns", server: "8.8.8.8" })
const s = state(d1, [d1, d2], [], ["d2"])
expect(s.getLabel(d1)).toBe("example.com (1.1.1.1)")
expect(s.getLabel(d2)).toBe("example.com (8.8.8.8)")
})
test("uses separate local and fetched monitor lists on single-system pages", () => {
const s = getMonitorCompareState({
monitor: a1,
localMonitors: [a1, a2],
otherMonitors: [b1, b2],
selectedSystemIds: new Set(["b"]),
selectedTargetIds: new Set(["a2"]),
getSystemName: (id) => systemNames[id],
})
expect(ids(s.compareMonitors)).toEqual(["a1", "a2", "b1", "b2"])
})
test("duplicate labels without a server are left alone", () => {
const b1Dupe = mon("b1Dupe", "b", "one.example")
const s = state(a1, [...all, b1Dupe], ["b"])
expect(ids(s.compareMonitors)).toEqual(["a1", "b1", "b1Dupe"])
expect(s.getLabel(b1)).toBe("Bravo")
expect(s.getLabel(b1Dupe)).toBe("Bravo")
})
// Every system probes every target, so selections multiply into targets × systems lines.
function grid(systemCount: number, targetCount: number) {
const monitors: NetworkMonitorRecord[] = []
for (let s = 0; s < systemCount; s++) {
for (let t = 0; t < targetCount; t++) monitors.push(mon(`s${s}t${t}`, `s${s}`, `t${t}.example`))
}
return monitors
}
test("selecting everything is trimmed to the line limit, keeping targets and earlier picks", () => {
const monitors = grid(30, 5)
const allSystems = Array.from({ length: 29 }, (_, i) => `s${i + 1}`)
const allTargets = ["s0t1", "s0t2", "s0t3", "s0t4"]
const s = state(monitors[0], monitors, allSystems, allTargets)
expect([...s.selectedTargetIds]).toEqual(allTargets)
// 5 targets fit on 4 systems (the opened one plus 3) within the limit
expect([...s.selectedSystemIds]).toEqual(["s1", "s2", "s3"])
expect(s.compareMonitors.length).toBe(20)
expect(s.canAddSystem).toBe(false)
})
test("more can be added only while the next pick fits the line limit", () => {
const monitors = grid(12, 3)
const s = state(monitors[0], monitors, ["s1", "s2", "s3", "s4", "s5", "s6"])
// 1 target × 7 systems; another target makes 14, another system makes 8
expect(s.canAddTarget).toBe(true)
expect(s.canAddSystem).toBe(true)
const full = state(monitors[0], monitors, ["s1", "s2", "s3", "s4", "s5", "s6", "s7"], ["s0t1", "s0t2"])
// 3 targets × 8 systems = 24
expect(full.compareMonitors.length).toBe(MAX_COMPARE_MONITORS)
expect(full.canAddTarget).toBe(false)
expect(full.canAddSystem).toBe(false)
})
-128
View File
@@ -1,128 +0,0 @@
import type { NetworkMonitorRecord } from "@/types"
// Kept free of UI and store imports so it can be unit tested with bun.
/** Most lines a comparison charts, to keep it readable and the stats request's filter short. */
export const MAX_COMPARE_MONITORS = 24
type MonitorTarget = Pick<NetworkMonitorRecord, "target" | "protocol" | "port">
export function getMonitorTarget(monitor: MonitorTarget) {
if (monitor.protocol !== "tcp") return monitor.target
const host = monitor.target.includes(":") && !monitor.target.startsWith("[") ? `[${monitor.target}]` : monitor.target
return `${host}:${monitor.port}`
}
/** Identifies what a monitor probes, regardless of which system probes it. */
export function getMonitorIdentityKey({
protocol,
target,
port,
server,
}: Pick<NetworkMonitorRecord, "protocol" | "target" | "port" | "server">) {
return JSON.stringify([protocol, target, port, server])
}
interface MonitorCompareInput {
/** The monitor whose sheet is open; always charted. */
monitor: NetworkMonitorRecord
/** Monitors that may include other targets on the opened monitor's system. */
localMonitors: NetworkMonitorRecord[]
/** Monitors that may include other systems' monitors. */
otherMonitors: NetworkMonitorRecord[]
selectedSystemIds: Set<string>
selectedTargetIds: Set<string>
getSystemName: (systemId: string) => string
}
/**
* Works out what the monitor sheet can compare and what it charts. Comparisons are limited to the opened
* monitor's protocol, since response time and loss mean different things per protocol.
*/
export function getMonitorCompareState({
monitor,
localMonitors,
otherMonitors,
selectedSystemIds,
selectedTargetIds,
getSystemName,
}: MonitorCompareInput) {
const sameProtocol = (m: NetworkMonitorRecord) => m.protocol === monitor.protocol
const systemTargets = localMonitors.filter(
(m) => m.system === monitor.system && m.id !== monitor.id && sameProtocol(m)
)
const systemMonitors = otherMonitors.filter((m) => m.system !== monitor.system && sameProtocol(m))
const keysBySystem = new Map<string, Set<string>>()
for (const m of systemMonitors) {
const keys = keysBySystem.get(m.system) ?? new Set<string>()
keys.add(getMonitorIdentityKey(m))
keysBySystem.set(m.system, keys)
}
// Each picker only offers what fits the other's selection, so every pick charts a line per system:
// systems must probe the opened target and every selected target, and targets must be probed by
// every selected system. Selections outside the options (e.g. a monitor deleted while the sheet is
// open) are ignored, and ones past MAX_COMPARE_MONITORS lines are dropped, keeping targets over
// systems and earlier picks over later ones.
const systemTargetsById = new Map(systemTargets.map((m) => [m.id, m]))
const pickedTargets = [...selectedTargetIds]
.flatMap((id) => systemTargetsById.get(id) ?? [])
.slice(0, MAX_COMPARE_MONITORS - 1)
const targetMonitors = [monitor, ...pickedTargets]
const requiredKeys = targetMonitors.map(getMonitorIdentityKey)
const systemOptions = [...keysBySystem]
.filter(([, keys]) => requiredKeys.every((key) => keys.has(key)))
.map(([id]) => id)
const systemOptionSet = new Set(systemOptions)
const systemIds = [...selectedSystemIds]
.filter((id) => systemOptionSet.has(id))
.slice(0, Math.floor(MAX_COMPARE_MONITORS / targetMonitors.length) - 1)
const targetOptions = systemTargets.filter((m) => {
const key = getMonitorIdentityKey(m)
return systemIds.every((id) => keysBySystem.get(id)?.has(key))
})
const targetIds = new Set(pickedTargets.map((m) => m.id))
// Every charted target is also charted for each selected system.
const targetKeys = new Set(targetMonitors.map(getMonitorIdentityKey))
const selectedSystems = new Set(systemIds)
const compareMonitors = [
...targetMonitors,
...systemMonitors.filter((m) => selectedSystems.has(m.system) && targetKeys.has(getMonitorIdentityKey(m))),
]
// Label series by whatever differs between them: system, target, or both.
const multiSystem = systemIds.length > 0
const multiTarget = targetIds.size > 0
const labels = new Map<string, string>()
const counts = new Map<string, number>()
for (const m of compareMonitors) {
const systemName = getSystemName(m.system)
const target = getMonitorTarget(m)
const label = multiSystem && multiTarget ? `${systemName} · ${target}` : multiTarget ? target : systemName
labels.set(m.id, label)
counts.set(label, (counts.get(label) ?? 0) + 1)
}
// DNS lookups of the same name against different servers would otherwise share a label.
for (const m of compareMonitors) {
const label = labels.get(m.id) as string
if ((counts.get(label) ?? 0) > 1 && m.server) labels.set(m.id, `${label} (${m.server})`)
}
return {
/** Systems that can be selected to compare against. */
systemOptions,
selectedSystemIds: new Set(systemIds),
/** Other targets on the opened monitor's system that can be selected. */
targetOptions,
selectedTargetIds: targetIds,
/** Whether one more target or system still fits within MAX_COMPARE_MONITORS lines. */
canAddTarget: (targetMonitors.length + 1) * (systemIds.length + 1) <= MAX_COMPARE_MONITORS,
canAddSystem: targetMonitors.length * (systemIds.length + 2) <= MAX_COMPARE_MONITORS,
/** Monitors to chart, starting with the opened one. */
compareMonitors,
getLabel: (m: NetworkMonitorRecord) => labels.get(m.id) ?? getMonitorTarget(m),
}
}
@@ -98,7 +98,11 @@ export function mergeMonitorStats(rawRecords: RawMonitorStatsRecord[], bucketMs
.map(([created, stats]) => ({ created, stats }))
}
export { getMonitorTarget } from "./monitor-compare"
export function getMonitorTarget(monitor: Pick<NetworkMonitorRecord, "target" | "protocol" | "port">) {
if (monitor.protocol !== "tcp") return monitor.target
const host = monitor.target.includes(":") && !monitor.target.startsWith("[") ? `[${monitor.target}]` : monitor.target
return `${host}:${monitor.port}`
}
/** Whole days until the certificate expires; negative once expired. */
export function getCertDaysLeft(cert: Pick<MonitorCertInfo, "expires">, now = Date.now()) {
+16 -16
View File
@@ -323,38 +323,38 @@ export function useNetworkMonitorStats(props: UseNetworkMonitorStatsProps) {
}, [monitorStats, cacheKey, interval, chartTime])
}
/** Only what comparison charts and labels need. */
const COMPARE_MONITOR_FIELDS = "id,system,target,protocol,port,server,interval,resAvg1h"
/**
* Monitors of one protocol on all systems except the given one, to compare against (#2385).
* Fetched per open so it also works in single-system tables, which only hold one system's monitors.
* Monitors on other systems that probe the same target (same protocol, target, port, and DNS server).
* Fetched once per open so it also works in single-system tables, which only hold one system's monitors.
*/
export function useCompareMonitors(system: string, protocol: string, enabled = true) {
const key = `${system}:${protocol}`
const [result, setResult] = useState<{ key: string; monitors: NetworkMonitorRecord[] }>({ key, monitors: [] })
export function useMatchingMonitors(monitor: NetworkMonitorRecord, enabled = true) {
const [matches, setMatches] = useState<NetworkMonitorRecord[]>([])
const { id, system, protocol, target, port, server } = monitor
useEffect(() => {
setMatches([])
if (!enabled) return
let cancelled = false
pb.collection<NetworkMonitorRecord>("network_monitors")
.getFullList({
fields: COMPARE_MONITOR_FIELDS,
filter: pb.filter("system!={:system} && protocol={:protocol}", { system, protocol }),
fields: NETWORK_MONITOR_FIELDS,
filter: pb.filter(
"id!={:id} && system!={:system} && protocol={:protocol} && target={:target} && port={:port} && server={:server}",
{ id, system, protocol, target, port, server }
),
})
.then((monitors) => {
if (!cancelled) setResult({ key: `${system}:${protocol}`, monitors })
.then((records) => {
if (!cancelled) setMatches(records)
})
.catch((error) => {
if (!cancelled) console.error("Failed to fetch compare monitors:", error)
if (!cancelled) console.error("Failed to fetch matching monitors:", error)
})
return () => {
cancelled = true
}
}, [system, protocol, enabled])
}, [id, system, protocol, target, port, server, enabled])
// Keep showing the last result while reopening refreshes it, but never another monitor's.
return result.key === key ? result.monitors : []
return matches
}
async function fetchMonitors(system?: string) {