From ada8c69817609398592606c8d8fb523bd1ac19ba Mon Sep 17 00:00:00 2001 From: Santhi Prakash Date: Mon, 14 Sep 2026 09:26:37 +0000 Subject: [PATCH] fix(agent): report container-local CPU usage via cgroup cpu accounting --- agent/cpu.go | 6 + agent/cpu_linux.go | 417 +++++++++++++++++++++++++++++++++++++++ agent/cpu_linux_test.go | 337 +++++++++++++++++++++++++++++++ agent/cpu_unsupported.go | 9 + 4 files changed, 769 insertions(+) create mode 100644 agent/cpu_linux.go create mode 100644 agent/cpu_linux_test.go create mode 100644 agent/cpu_unsupported.go diff --git a/agent/cpu.go b/agent/cpu.go index 8ac0c2e9c..327a85343 100644 --- a/agent/cpu.go +++ b/agent/cpu.go @@ -35,6 +35,12 @@ type CpuMetrics struct { // getCpuMetrics calculates detailed CPU usage metrics using cached previous measurements. // It returns percentages for total, user, system, iowait, and steal time. func getCpuMetrics(cacheTimeMs uint16) (CpuMetrics, error) { + // Inside a container, /proc/stat reports the host cores' counters (via + // lxcfs on LXC, or the host's procfs elsewhere), not the container's own + // usage. Prefer the cgroup's own CPU accounting when available. (#2332) + if metrics, ok := containerCpuMetrics(cacheTimeMs); ok { + return metrics, nil + } times, err := cpu.Times(false) if err != nil || len(times) == 0 { return CpuMetrics{}, err diff --git a/agent/cpu_linux.go b/agent/cpu_linux.go new file mode 100644 index 000000000..282c0c882 --- /dev/null +++ b/agent/cpu_linux.go @@ -0,0 +1,417 @@ +//go:build linux + +package agent + +import ( + "bytes" + "os" + "path/filepath" + "runtime" + "strconv" + "strings" + "sync" + "time" + + "github.com/henrygd/beszel/agent/utils" +) + +// Container-aware CPU accounting (issue #2332). +// +// Inside a container /proc/stat does not describe the container's own usage: +// lxcfs serves LXC guests the raw counters of the host cores in their cpuset, +// and plain runtimes (Docker, k8s) expose the host's /proc outright. An idle +// container sharing a host core with a busy neighbor then reports near-100% +// CPU while doing nothing. The cgroup's own accounting (cpu.stat / +// cpuacct.usage) reflects only the container's processes, so when the agent +// runs inside a container we derive CPU% from that instead. + +// File paths and hooks are variables so tests can point them at fixtures. +var ( + cpuCgroupRoot = "/sys/fs/cgroup" // default cgroup v2 mount point + cpuCgroupMountinfo = "/proc/self/mountinfo" + cpuProcSelfCgroup = "/proc/self/cgroup" + cpuProcOneEnviron = "/proc/1/environ" + cpuDockerenvPath = "/.dockerenv" + cpuContainerenv = "/run/.containerenv" + cpuSystemdContPath = "/run/systemd/container" + cpuNumCPU = runtime.NumCPU + cpuNow = time.Now +) + +// cpuUserHZ is the USER_HZ jiffies-per-second rate cpuacct.stat reports in. +const cpuUserHZ = 100 + +var ( + containerOnce sync.Once + containerDetected bool +) + +// inContainer reports whether the agent itself runs inside a container. +// The result is cached because it cannot change during the process lifetime. +func inContainer() bool { + containerOnce.Do(func() { containerDetected = detectContainer() }) + return containerDetected +} + +// detectContainer looks for the usual container markers. +func detectContainer() bool { + // set by systemd-nspawn, LXC, Podman and others + if os.Getenv("container") != "" { + return true + } + for _, p := range []string{cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath} { + if utils.FileExists(p) { + return true + } + } + // liblxc always puts container=lxc in the container init's environment, + // which survives on non-systemd guests such as Alpine LXC. + if data, err := os.ReadFile(cpuProcOneEnviron); err == nil { + if bytes.HasPrefix(data, []byte("container=")) || + bytes.Contains(data, []byte("\x00container=")) { + return true + } + } + // an lxcfs mount means /proc/stat is virtualized with host core counters + if data, err := os.ReadFile(cpuCgroupMountinfo); err == nil && + bytes.Contains(data, []byte(" - fuse.lxcfs ")) { + return true + } + return false +} + +// cgroupCpuSample is one read of the container's cumulative CPU accounting. +type cgroupCpuSample struct { + usageUsec uint64 + userUsec uint64 + systemUsec uint64 + cores float64 // usable CPU cores: affinity ∩ cpuset ∩ quota + at time.Time +} + +var lastCgroupCpuSamples = make(map[uint16]cgroupCpuSample) + +// init seeds the container CPU baseline so the first reported value is a real +// delta since startup rather than zero. +func init() { + if !inContainer() { + return + } + if s, ok := readContainerCpuSample(); ok { + s.at = cpuNow() + lastCgroupCpuSamples[60000] = s + } +} + +// containerCpuMetrics derives CPU metrics from the agent's own cgroup +// accounting when running inside a container. It returns ok=false on plain +// hosts and whenever cgroup accounting is unreadable, so callers keep the +// /proc/stat fallback. +func containerCpuMetrics(cacheTimeMs uint16) (CpuMetrics, bool) { + if !inContainer() { + return CpuMetrics{}, false + } + cur, ok := readContainerCpuSample() + if !ok { + return CpuMetrics{}, false + } + cur.at = cpuNow() + + prev, ok := lastCgroupCpuSamples[cacheTimeMs] + if !ok { + prev = lastCgroupCpuSamples[60000] + } + lastCgroupCpuSamples[cacheTimeMs] = cur + + // No baseline yet, a backwards counter (cgroup recreated), or a + // non-positive clock delta: report zero this tick instead of guessing. + elapsedUsec := cur.at.Sub(prev.at).Microseconds() + if prev.at.IsZero() || elapsedUsec <= 0 || cur.usageUsec < prev.usageUsec { + return CpuMetrics{}, true + } + + cores := cur.cores + if cores <= 0 { + cores = 1 + } + window := float64(elapsedUsec) * cores + + metrics := CpuMetrics{ + Total: clampPercent(float64(cur.usageUsec-prev.usageUsec) / window * 100), + User: clampPercent(float64(cur.userUsec-prev.userUsec) / window * 100), + System: clampPercent(float64(cur.systemUsec-prev.systemUsec) / window * 100), + } + // cgroup accounting has no iowait/steal; everything not busy is idle. + metrics.Idle = clampPercent(100 - metrics.Total) + return metrics, true +} + +// readContainerCpuSample reads the container's cumulative CPU usage, preferring +// the cgroup v2 unified hierarchy and falling back to the v1 cpuacct +// controller. +func readContainerCpuSample() (cgroupCpuSample, bool) { + if s, ok := readCgroupV2CpuSample(); ok { + return s, true + } + return readCgroupV1CpuSample() +} + +// readCgroupV2CpuSample reads usage from the unified hierarchy's cpu.stat. +// +// The mount root is usually the right cgroup to read: inside a private cgroup +// namespace (LXC, default Docker) /sys/fs/cgroup already is the container's +// root cgroup, and its cpu.stat accounts for every process in the container, +// including siblings of the agent's own service cgroup. When the hierarchy is +// not namespaced (e.g. docker run --cgroupns=host) /proc/self/cgroup instead +// holds the container's real host-side path, which is joined onto the mount. +func readCgroupV2CpuSample() (cgroupCpuSample, bool) { + rel := selfCgroupPath("0::") + if rel == "" { + return cgroupCpuSample{}, false // no v2 membership; try v1 + } + dir := cpuCgroupRoot + if mount := cgroupMountPoint("cgroup2", ""); mount != "" { + dir = mount + } + if rel != "/" && hasContainerRuntimeMarker(rel) { + if cand := filepath.Join(dir, rel); directoryExistsOK(cand) { + dir = cand + } + } + stat := filepath.Join(dir, "cpu.stat") + usage, ok := cgroupStatValue(stat, "usage_usec") + if !ok { + return cgroupCpuSample{}, false + } + s := cgroupCpuSample{usageUsec: usage, cores: cpuCgroupCores(dir)} + s.userUsec, _ = cgroupStatValue(stat, "user_usec") + s.systemUsec, _ = cgroupStatValue(stat, "system_usec") + return s, true +} + +// readCgroupV1CpuSample reads usage from the legacy cpuacct controller. +// Runtimes bind-mount the container's own cpuacct directory at the hierarchy +// mount, so the mount root is normally already the container's cgroup; if the +// process's cgroup path still resolves below the mount (shared host view), +// that subdirectory is used instead. +func readCgroupV1CpuSample() (cgroupCpuSample, bool) { + mount := cgroupMountPoint("cgroup", "cpuacct") + if mount == "" { + return cgroupCpuSample{}, false + } + dir := mount + if rel := selfCgroupPath("cpuacct"); rel != "" && rel != "/" { + if cand := filepath.Join(mount, rel); utils.FileExists(filepath.Join(cand, "cpuacct.usage")) { + dir = cand + } + } + usageNs, ok := utils.ReadUintFile(filepath.Join(dir, "cpuacct.usage")) + if !ok { + return cgroupCpuSample{}, false + } + s := cgroupCpuSample{usageUsec: usageNs / 1000, cores: cpuCgroupCores(dir)} + // cpuacct.stat reports user/system in USER_HZ jiffies. + if v, ok := cgroupStatValue(filepath.Join(dir, "cpuacct.stat"), "user"); ok { + s.userUsec = v * 1e6 / cpuUserHZ + } + if v, ok := cgroupStatValue(filepath.Join(dir, "cpuacct.stat"), "system"); ok { + s.systemUsec = v * 1e6 / cpuUserHZ + } + return s, true +} + +// selfCgroupPath returns the agent's cgroup path from /proc/self/cgroup: the +// path after "0::" for the v2 unified hierarchy, or the path of the entry +// whose controller list contains the given v1 controller (e.g. "cpuacct"). +func selfCgroupPath(selector string) string { + data, err := os.ReadFile(cpuProcSelfCgroup) + if err != nil { + return "" + } + for _, line := range strings.Split(string(data), "\n") { + parts := strings.SplitN(line, ":", 3) + if len(parts) != 3 { + continue + } + if selector == "0::" { + if parts[0] == "0" && parts[1] == "" { + return parts[2] + } + continue + } + for _, ctrl := range strings.Split(parts[1], ",") { + if ctrl == selector { + return parts[2] + } + } + } + return "" +} + +// cgroupMountPoint returns the mount point of a cgroup hierarchy from +// /proc/self/mountinfo: the cgroup2 mount for v2, or the cgroup mount whose +// super options list the wanted v1 controller. +func cgroupMountPoint(fstype, v1ctrl string) string { + data, err := os.ReadFile(cpuCgroupMountinfo) + if err != nil { + return "" + } + for _, line := range strings.Split(string(data), "\n") { + left, right, found := strings.Cut(line, " - ") + if !found { + continue + } + post := strings.Fields(right) + if len(post) == 0 || post[0] != fstype { + continue + } + if v1ctrl != "" && !mountOptHas(post, v1ctrl) { + continue + } + fields := strings.Fields(left) + if len(fields) >= 5 { + return unescapeMountPoint(fields[4]) + } + } + return "" +} + +// mountOptHas reports whether the comma-separated super options (field 3 after +// the " - " separator) contain opt. +func mountOptHas(post []string, opt string) bool { + if len(post) < 3 { + return false + } + for _, o := range strings.Split(post[2], ",") { + if o == opt { + return true + } + } + return false +} + +// unescapeMountPoint decodes octal escapes (e.g. \040 for space) used in +// mountinfo paths. +func unescapeMountPoint(s string) string { + return strings.NewReplacer(`\040`, " ", `\011`, "\t", `\012`, "\n", `\134`, `\`).Replace(s) +} + +// hasContainerRuntimeMarker reports whether a cgroup path looks like a real +// host-side container cgroup path (docker/k8s/lxc/podman), meaning the visible +// hierarchy is not namespaced and the path can be resolved under the mount. +func hasContainerRuntimeMarker(path string) bool { + for _, m := range []string{"docker", "kubepods", "lxc", "crio", "libpod", "containerd", "podman"} { + if strings.Contains(path, m) { + return true + } + } + return false +} + +// cpuCgroupCores returns how many CPU cores the cgroup at dir may use: the +// smallest of the process affinity mask, the cgroup cpuset, and the CPU quota. +func cpuCgroupCores(dir string) float64 { + cores := float64(cpuNumCPU()) + if n := cpusetCount(dir); n > 0 && n < cores { + cores = n + } + if q, ok := cpuQuotaCores(dir); ok && q < cores { + cores = q + } + if cores <= 0 { + cores = 1 + } + return cores +} + +// cpusetCount returns the number of CPUs in the cgroup's cpuset, e.g. "0-3" or +// "2,5-7". An empty or missing file means unconstrained. +func cpusetCount(dir string) float64 { + for _, name := range []string{"cpuset.cpus.effective", "cpuset.cpus"} { + raw, err := os.ReadFile(filepath.Join(dir, name)) + if err != nil { + continue + } + if n := countCpuList(strings.TrimSpace(string(raw))); n > 0 { + return float64(n) + } + } + return 0 +} + +// countCpuList counts the CPUs in a Linux CPU list like "0-3,5,8-9". +func countCpuList(list string) int { + total := 0 + for _, part := range strings.Split(list, ",") { + lo, hi, ranged := strings.Cut(part, "-") + a, err := strconv.Atoi(lo) + if err != nil { + continue + } + b := a + if ranged { + if v, err := strconv.Atoi(hi); err == nil { + b = v + } + } + if b >= a { + total += b - a + 1 + } + } + return total +} + +// cpuQuotaCores returns the cgroup's CPU quota in cores. v2 uses cpu.max +// (" "), v1 uses cpu.cfs_quota_us / cpu.cfs_period_us. +func cpuQuotaCores(dir string) (float64, bool) { + if raw, err := os.ReadFile(filepath.Join(dir, "cpu.max")); err == nil { + fields := strings.Fields(string(raw)) + if len(fields) == 2 && fields[0] != "max" { + if quota, err := strconv.ParseFloat(fields[0], 64); err == nil && quota > 0 { + if period, err := strconv.ParseFloat(fields[1], 64); err == nil && period > 0 { + return quota / period, true + } + } + } + } + if quota, ok := readCgroupInt(filepath.Join(dir, "cpu.cfs_quota_us")); ok && quota > 0 { + if period, ok := readCgroupInt(filepath.Join(dir, "cpu.cfs_period_us")); ok && period > 0 { + return float64(quota) / float64(period), true + } + } + return 0, false +} + +// cgroupStatValue returns the value of key in a cgroup "key value" stat file. +func cgroupStatValue(path, key string) (uint64, bool) { + data, err := os.ReadFile(path) + if err != nil { + return 0, false + } + for _, line := range strings.Split(string(data), "\n") { + name, value, found := strings.Cut(line, " ") + if !found || name != key { + continue + } + v, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + return v, err == nil + } + return 0, false +} + +// readCgroupInt reads a file containing a single signed integer +// (cpu.cfs_quota_us is -1 when no quota is set). +func readCgroupInt(path string) (int64, bool) { + data, err := os.ReadFile(path) + if err != nil { + return 0, false + } + v, err := strconv.ParseInt(strings.TrimSpace(string(data)), 10, 64) + return v, err == nil +} + +// directoryExistsOK reports whether path is a directory. +func directoryExistsOK(path string) bool { + ok, _ := directoryExists(path) + return ok +} diff --git a/agent/cpu_linux_test.go b/agent/cpu_linux_test.go new file mode 100644 index 000000000..67119dd12 --- /dev/null +++ b/agent/cpu_linux_test.go @@ -0,0 +1,337 @@ +//go:build testing && linux + +package agent + +import ( + "os" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// swapCpuContainerSeams points every container-detection and cgroup path at +// empty fixtures under a temp dir, then restores them on cleanup. +func swapCpuContainerSeams(t *testing.T) { + t.Helper() + backup := struct { + root, mountinfo, selfCgroup, oneEnviron, dockerenv, containerenv, systemdCont string + numCPU func() int + now func() time.Time + }{ + cpuCgroupRoot, cpuCgroupMountinfo, cpuProcSelfCgroup, cpuProcOneEnviron, + cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath, cpuNumCPU, cpuNow, + } + detected := containerDetected + samples := lastCgroupCpuSamples + env, hadEnv := os.LookupEnv("container") + t.Cleanup(func() { + cpuCgroupRoot, cpuCgroupMountinfo, cpuProcSelfCgroup, cpuProcOneEnviron = backup.root, backup.mountinfo, backup.selfCgroup, backup.oneEnviron + cpuDockerenvPath, cpuContainerenv, cpuSystemdContPath = backup.dockerenv, backup.containerenv, backup.systemdCont + cpuNumCPU, cpuNow = backup.numCPU, backup.now + containerOnce = sync.Once{} + containerDetected = detected + lastCgroupCpuSamples = samples + if hadEnv { + os.Setenv("container", env) + } + }) + + containerOnce = sync.Once{} + containerDetected = false + lastCgroupCpuSamples = make(map[uint16]cgroupCpuSample) + os.Unsetenv("container") + + tmp := t.TempDir() + cpuCgroupRoot = filepath.Join(tmp, "cgroup") + cpuCgroupMountinfo = filepath.Join(tmp, "mountinfo") + cpuProcSelfCgroup = filepath.Join(tmp, "self-cgroup") + cpuProcOneEnviron = filepath.Join(tmp, "one-environ") + cpuDockerenvPath = filepath.Join(tmp, "dockerenv") + cpuContainerenv = filepath.Join(tmp, "containerenv") + cpuSystemdContPath = filepath.Join(tmp, "systemd-container") +} + +func writeCpuFixture(t *testing.T, path, contents string) { + t.Helper() + require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755)) + require.NoError(t, os.WriteFile(path, []byte(contents), 0o644)) +} + +// fakeNow installs a controllable clock and returns a function to advance it. +func fakeNow(t *testing.T) func(time.Duration) { + t.Helper() + cur := time.Unix(1_700_000_000, 0) + cpuNow = func() time.Time { return cur } + return func(d time.Duration) { cur = cur.Add(d) } +} + +func TestDetectContainer(t *testing.T) { + tests := []struct { + name string + setup func(t *testing.T) + want bool + }{ + {"plain host", func(t *testing.T) {}, false}, + {"container env", func(t *testing.T) { t.Setenv("container", "lxc") }, true}, + {".dockerenv", func(t *testing.T) { writeCpuFixture(t, cpuDockerenvPath, "") }, true}, + {".containerenv", func(t *testing.T) { writeCpuFixture(t, cpuContainerenv, "") }, true}, + {"systemd container", func(t *testing.T) { writeCpuFixture(t, cpuSystemdContPath, "lxc\n") }, true}, + {"init environ container=lxc", func(t *testing.T) { + writeCpuFixture(t, cpuProcOneEnviron, "PATH=/sbin\x00container=lxc\x00HOME=/root\x00") + }, true}, + {"init environ without marker", func(t *testing.T) { + writeCpuFixture(t, cpuProcOneEnviron, "PATH=/sbin\x00HOME=/root\x00") + }, false}, + {"lxcfs serving /proc", func(t *testing.T) { + writeCpuFixture(t, cpuCgroupMountinfo, + "31 25 0:28 / /proc/stat rw,nosuid,nodev,relatime - fuse.lxcfs lxcfs rw,user_id=0,group_id=0\n") + }, true}, + {"cgroup-only mountinfo", func(t *testing.T) { + writeCpuFixture(t, cpuCgroupMountinfo, + "36 25 0:32 / /sys/fs/cgroup rw - cgroup2 cgroup2 rw,nsdelegate\n") + }, false}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + swapCpuContainerSeams(t) + tt.setup(t) + assert.Equal(t, tt.want, detectContainer()) + }) + } +} + +func TestReadCgroupV2CpuSample(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755)) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), + "usage_usec 3000000\nuser_usec 2000000\nsystem_usec 1000000\nnr_throttled 7\n") + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpuset.cpus.effective"), "2,5-7\n") + cpuNumCPU = func() int { return 8 } + + s, ok := readCgroupV2CpuSample() + require.True(t, ok) + assert.EqualValues(t, 3000000, s.usageUsec) + assert.EqualValues(t, 2000000, s.userUsec) + assert.EqualValues(t, 1000000, s.systemUsec) + assert.InDelta(t, 4, s.cores, 0.001) // cpuset 2,5-7 = 4 cores +} + +// In a namespaced container the agent may sit in a sub-cgroup (e.g. a systemd +// service); the mount root still accounts for the whole container and must win. +func TestReadCgroupV2PrefersContainerRoot(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuProcSelfCgroup, "0::/system.slice/beszel-agent.service\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 9000\n") + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "system.slice/beszel-agent.service/cpu.stat"), "usage_usec 5\n") + + s, ok := readCgroupV2CpuSample() + require.True(t, ok) + assert.EqualValues(t, 9000, s.usageUsec) +} + +// Without a cgroup namespace the mount shows the real host hierarchy and the +// container's own path (docker/kubepods/lxc markers) resolves under it. +func TestReadCgroupV2ResolvesRuntimePath(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuProcSelfCgroup, "0::/system.slice/docker-deadbeef.scope\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 9000\n") + sub := filepath.Join(cpuCgroupRoot, "system.slice/docker-deadbeef.scope") + writeCpuFixture(t, filepath.Join(sub, "cpu.stat"), "usage_usec 5\n") + writeCpuFixture(t, filepath.Join(sub, "cpu.max"), "100000 100000\n") + + s, ok := readCgroupV2CpuSample() + require.True(t, ok) + assert.EqualValues(t, 5, s.usageUsec) + assert.InDelta(t, 1, s.cores, 0.001) // cpu.max quota of 1 core +} + +func TestReadCgroupV1CpuSample(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuProcSelfCgroup, "3:cpuacct:/\n2:memory:/\n") + v1 := filepath.Join(t.TempDir(), "cpuacct") + writeCpuFixture(t, cpuCgroupMountinfo, + "30 25 0:26 / "+v1+" rw,nosuid,nodev,noexec,relatime - cgroup cgroup rw,cpuacct\n") + writeCpuFixture(t, filepath.Join(v1, "cpuacct.usage"), "2000000000\n") + writeCpuFixture(t, filepath.Join(v1, "cpuacct.stat"), "user 100\nsystem 50\n") + cpuNumCPU = func() int { return 4 } + + s, ok := readContainerCpuSample() // no 0:: line -> falls through to v1 + require.True(t, ok) + assert.EqualValues(t, 2000000, s.usageUsec) // ns -> usec + assert.EqualValues(t, 1000000, s.userUsec) // 100 jiffies * 1e6/100 + assert.EqualValues(t, 500000, s.systemUsec) // 50 jiffies + assert.InDelta(t, 4, s.cores, 0.001) +} + +func TestContainerCpuMetricsMath(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuDockerenvPath, "") + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755)) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpuset.cpus.effective"), "0-3\n") + cpuNumCPU = func() int { return 8 } + advance := fakeNow(t) + + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), + "usage_usec 1000000\nuser_usec 600000\nsystem_usec 400000\n") + m, ok := containerCpuMetrics(60000) + require.True(t, ok) + assert.Zero(t, m.Total) // first call only seeds the baseline + + // 1s elapsed, container burned 2 core-seconds on 4 usable cores + advance(time.Second) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), + "usage_usec 3000000\nuser_usec 1600000\nsystem_usec 900000\n") + m, ok = containerCpuMetrics(60000) + require.True(t, ok) + assert.InDelta(t, 50, m.Total, 0.01) + assert.InDelta(t, 25, m.User, 0.01) + assert.InDelta(t, 12.5, m.System, 0.01) + assert.Zero(t, m.Iowait) + assert.Zero(t, m.Steal) + assert.InDelta(t, 50, m.Idle, 0.01) +} + +func TestContainerCpuMetricsHonorsQuota(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuDockerenvPath, "") + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755)) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.max"), "200000 100000\n") // 2 cores + cpuNumCPU = func() int { return 8 } + advance := fakeNow(t) + + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 1000000\n") + containerCpuMetrics(60000) + advance(time.Second) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 2000000\n") + m, ok := containerCpuMetrics(60000) + require.True(t, ok) + assert.InDelta(t, 50, m.Total, 0.01) // 1 core-second against a 2-core quota +} + +func TestContainerCpuMetricsZeroAndBackwardDelta(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuDockerenvPath, "") + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755)) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 5000000\n") + cpuNumCPU = func() int { return 4 } + advance := fakeNow(t) + + // seed the baseline, then do not advance the clock: elapsed <= 0 + containerCpuMetrics(60000) + m, ok := containerCpuMetrics(60000) + require.True(t, ok) + assert.Zero(t, m.Total) + + // counter goes backwards (cgroup recreated): report zero and re-baseline + advance(time.Second) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 100000\n") + m, ok = containerCpuMetrics(60000) + require.True(t, ok) + assert.Zero(t, m.Total) + + // next tick measures from the new baseline, not the stale one + advance(time.Second) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 1100000\n") + m, ok = containerCpuMetrics(60000) + require.True(t, ok) + assert.InDelta(t, 25, m.Total, 0.01) // 1e6 usec / (1s * 4 cores) +} + +func TestContainerCpuMetricsFallbacks(t *testing.T) { + t.Run("not in container", func(t *testing.T) { + swapCpuContainerSeams(t) + _, ok := containerCpuMetrics(60000) + assert.False(t, ok) + }) + t.Run("in container without cgroup accounting", func(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuDockerenvPath, "") + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + // cpuCgroupRoot has no cpu.stat + _, ok := containerCpuMetrics(60000) + assert.False(t, ok) + }) +} + +// The host path must keep reporting through gopsutil untouched. +func TestGetCpuMetricsHostFallback(t *testing.T) { + swapCpuContainerSeams(t) + m, err := getCpuMetrics(60000) + require.NoError(t, err) + assert.GreaterOrEqual(t, m.Total, 0.0) + assert.LessOrEqual(t, m.Total, 100.0) +} + +// Inside a container getCpuMetrics must report the cgroup-derived value, not +// the host core counters from /proc/stat. +func TestGetCpuMetricsPrefersCgroup(t *testing.T) { + swapCpuContainerSeams(t) + writeCpuFixture(t, cpuDockerenvPath, "") + writeCpuFixture(t, cpuProcSelfCgroup, "0::/\n") + writeCpuFixture(t, cpuCgroupMountinfo, "") + require.NoError(t, os.MkdirAll(cpuCgroupRoot, 0o755)) + cpuNumCPU = func() int { return 4 } + advance := fakeNow(t) + + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 0\n") + m, err := getCpuMetrics(60000) + require.NoError(t, err) + assert.Zero(t, m.Total) + + advance(time.Second) + writeCpuFixture(t, filepath.Join(cpuCgroupRoot, "cpu.stat"), "usage_usec 2000000\n") + m, err = getCpuMetrics(60000) + require.NoError(t, err) + assert.InDelta(t, 50, m.Total, 0.01) +} + +func TestCountCpuList(t *testing.T) { + assert.Equal(t, 4, countCpuList("0-3")) + assert.Equal(t, 4, countCpuList("2,5-7")) + assert.Equal(t, 1, countCpuList("2")) + assert.Equal(t, 0, countCpuList("")) + assert.Equal(t, 0, countCpuList("max")) + assert.Equal(t, 6, countCpuList("0-3,8-9")) +} + +func TestCpuQuotaCores(t *testing.T) { + dir := t.TempDir() + _, ok := cpuQuotaCores(dir) + assert.False(t, ok) // no quota files + + writeCpuFixture(t, filepath.Join(dir, "cpu.max"), "max 100000\n") + _, ok = cpuQuotaCores(dir) + assert.False(t, ok) // unlimited + + writeCpuFixture(t, filepath.Join(dir, "cpu.max"), "150000 100000\n") + q, ok := cpuQuotaCores(dir) + require.True(t, ok) + assert.InDelta(t, 1.5, q, 0.001) + + // v1 files + v1 := t.TempDir() + writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_quota_us"), "-1\n") + writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_period_us"), "100000\n") + _, ok = cpuQuotaCores(v1) + assert.False(t, ok) + + writeCpuFixture(t, filepath.Join(v1, "cpu.cfs_quota_us"), "50000\n") + q, ok = cpuQuotaCores(v1) + require.True(t, ok) + assert.InDelta(t, 0.5, q, 0.001) +} diff --git a/agent/cpu_unsupported.go b/agent/cpu_unsupported.go new file mode 100644 index 000000000..886941f82 --- /dev/null +++ b/agent/cpu_unsupported.go @@ -0,0 +1,9 @@ +//go:build !linux + +package agent + +// containerCpuMetrics is Linux-only (cgroup accounting). Other platforms keep +// the gopsutil /proc path. +func containerCpuMetrics(uint16) (CpuMetrics, bool) { + return CpuMetrics{}, false +}