Compare commits

...
46 Commits
Author SHA1 Message Date
henrygd ffcdb04167 i18n: update locale files 2026-09-03 11:22:37 -04:00
henrygd bc21da9cb3 fix(agent): prevent possible deadlock when stopping SSH server (#2280) 2026-09-02 20:31:38 -04:00
henrygd 71af06b31c chore: update changelog 2026-09-02 20:22:16 -04:00
hankandGitHub a8def47018 i18n: New Crowdin updates (#2284) 2026-09-02 18:06:01 -04:00
henrygd d2a253082f chore: update changelog for 0.19.0 2026-09-02 18:00:59 -04:00
henrygd 3d8fc39e94 update dev version and finalize migration file for 0.19.0 2026-09-02 18:00:40 -04:00
henrygd 7d347cfd6a deps: update go version and go deps 2026-09-02 17:58:51 -04:00
Santhi PrakashandGitHub 5790fbecce fix: GPU Power Draw chart renders full-width instead of half-width (#2269) 2026-09-02 17:09:32 -04:00
hankandGitHub f9309da9f0 i18n: New Crowdin updates (#2283) 2026-09-02 15:37:11 -04:00
henrygd 7d97b0d23a i18n: update locale files and source strings 2026-09-02 14:57:37 -04:00
hankandGitHub f104f31ee3 Merge commit from fork 2026-09-02 13:55:54 -04:00
henrygd a1ca51608a fix: add singleDesc back to container alert 2026-09-02 13:09:27 -04:00
henrygd a4de2e87c4 consolidate migrations and add triggeredDesc for container health alert 2026-09-02 13:02:42 -04:00
5969d36856 feat(alerts): add container health alerts with log excerpt on notifications (#2225)
Add a new "ContainerHealth" alert type that fires when a Docker container's
health check reports unhealthy, and resolves when it recovers. This mirrors
the existing Status (up/down) alert pattern: an alert can be armed per system
and honors the "min minutes" delay before firing.

When the alert fires, the notification (email and any configured webhook,
including Discord via shoutrrr) includes a log excerpt fetched live from the
agent for up to 2 of the unhealthy containers, prioritizing lines containing
"error" or "fatal" (falling back to the log tail if none match), capped to
keep the message well under Discord's size limit.

---------

Co-authored-by: hank <hank@henrygd.me>
2026-09-02 12:46:23 -04:00
henrygd b1895247ba alerts: defer system info unmarshalling for systemd alerts 2026-09-01 21:49:45 -04:00
097180e8d7 feat(alerts): add alert for failed systemd services (#2173)
Adds a user-configurable "Failed Services" alert that notifies when any
tracked systemd service enters the failed state, and again when all services
recover.

---------

Signed-off-by: Martin Stenröse <martin@stenrose.se>
Co-authored-by: henrygd <hank@henrygd.me>
2026-09-01 20:41:48 -04:00
spatiumstasandGitHub ed88e6efae feat(alerts): add CPU state notifications (#2249) 2026-09-01 19:10:31 -04:00
Steven HonsonandGitHub b670224ed8 ui: hide gpu indicator for host without gpu (#2279) 2026-09-01 18:36:09 -04:00
917d069ab3 feat: add ZFS monitoring (#2209)
- track pool capacity, health, I/O, scrub status, and vdev errors
- report dataset usage and correct ZFS filesystem metrics
- add pool charts, detail views, refresh controls, and health alerts
- persist pool details and include ZFS usage in disk alerts
- support configurable detail intervals and legacy agent compatibility

---------

Co-authored-by: hank <hank@henrygd.me>
2026-09-01 12:19:36 -04:00
Sven van GinkelandGitHub b38fb7dafa feat: Add cumulative disk read/write totals to Disk I/O sheet (#2179) 2026-08-30 15:44:18 -04:00
henrygd 8675199e20 fix: preserve battery array encoding with json v2 2026-08-30 14:40:13 -04:00
Aditya Raj SinghandGitHub 3af6512514 fix(hub): don't read the SSH client after it is closed (#2277)
createSessionWithTimeout checked sys.client for nil and then dereferenced
it again inside the goroutine that calls NewSession. update() runs the
SMART fetch in its own goroutine, so closeSSHConnection can clear the
field between those two reads and the goroutine dereferences a nil
client, panicking the whole hub process.

Make client an atomic.Pointer, load it once before starting the
goroutine, and clear it with Swap so a concurrent close cannot be
observed mid-session-creation. NewSession on an already-closed client
returns an error, which the existing retry path already handles.

Closes #2157
2026-08-30 13:34:32 -04:00
Sven van GinkelandGitHub 87620f3251 feat(hub/agent): alphabetical disk ordering and root disk renaming (#2006) 2026-08-30 13:09:04 -04:00
Aditya Raj SinghandGitHub fa9de55433 fix(agent): warn on critical ATA SMART attributes (#2275) 2026-08-30 11:23:52 -04:00
Aditya Raj SinghandGitHub e235c9935c fix(install): generate /etc/machine-id on systems without /proc (#2274)
* fix(install): generate /etc/machine-id on systems without /proc

The install script read the fingerprint UUID from
/proc/sys/kernel/random/uuid, which FreeBSD does not have. On pfSense
the redirect still created /etc/machine-id, cat failed, and the agent
was left with an empty machine-id file. The [ ! -f ] guard then skipped
regeneration on every later run.

Fall back to uuidgen when /proc is unavailable, mirroring how the
script already picks between sha256sum and FreeBSD's sha256, and leave
no file behind when neither source exists.

* fix(install): regenerate empty machine id
2026-08-30 11:12:21 -04:00
Aditya Raj SinghandGitHub 7c60f02802 fix(agent): don't read host CPU and memory totals from a Docker VM (#2272)
refreshSystemDetails() takes NCPU and MemTotal from the Docker daemon's
/info response. That only describes this machine when the daemon shares its
kernel. On macOS and Windows Docker runs inside a Linux VM, so the system
details header shows the VM's memory as the host total, and the VM's CPU
count clamps both cores and threads through the lxc branch below it.

Only consult Docker's host info on platforms where the daemon runs natively.
Everything else already falls back to gopsutil, which reads this host.
2026-08-30 11:09:39 -04:00
Ryan ChouandGitHub 6fe268e463 fix(agent): read TOKEN_FILE like KEY_FILE instead of sending the whole file (#2276) 2026-08-30 10:16:43 -04:00
henrygd 467f176713 i18n: add Greek and update locale files 2026-08-27 15:10:05 -04:00
Erkinjon YusupovandGitHub 4c8e3c69ba feat: add Uzbek (uz) translation (#2034) 2026-08-27 14:58:21 -04:00
hankandGitHub 8dfdacb8f5 i18n: New Crowdin updates (#2234) 2026-08-27 14:49:08 -04:00
henrygd d61b75ffdf deps: upgrade to go 1.27 + upgrade go packages 2026-08-26 14:48:52 -04:00
henrygd d7256c7af7 fix: widen coverage of internal ip space in isInternalIP 2026-08-26 13:42:29 -04:00
henrygd 0ad707288a fix windows sensor mocks and data directory tests 2026-08-26 11:50:19 -04:00
Aditya Raj SinghandGitHub 0bc5470f08 fix(agent): count swap cache as used space (#2267)
SwapCached pages have been read back into memory but still occupy allocated swap slots. Subtracting them from SwapTotal - SwapFree underreported swap usage compared with free, Glances, and gopsutil's canonical SwapMemory metric.
2026-08-26 11:42:40 -04:00
Luke WassandGitHub 4c48fe0c41 fix(agent): carry Intel GPU averages forward between samples (#2256)
Intel GPUs (intel_gpu_top) never report temperature or memory, so the
"suspended card" heuristic in calculateGPUAverage (temp == 0 &&
memoryUsed == 0) fired on every collection that landed between samples.

intel_gpu_top samples every 3.3s (intelGpuStatsInterval) while the hub's
realtime worker collects every 1s, so most realtime collections had no
new sample (delta count 0) and returned an empty GPUData with power
omitted (json "p"/"pp" are omitempty). The frontend derives the GPU
Power Draw series and legend from the latest sample, so the chart and
legend blanked on roughly two of every three or four one-second cycles.

NVIDIA/AMD were unaffected because they report temperature even when
idle, so the heuristic never fired and the last average was already
carried forward.

Gate the zero-return on non-engine (discrete) GPUs so Intel GPUs carry
the last average forward during between-sample gaps, matching the
existing NVIDIA/AMD behavior. Add a regression test.
2026-08-24 10:33:00 -04:00
dependabot[bot]andGitHub 6efe4be648 chore(deps): bump azure/setup-helm from 4 to 5 (#2255) 2026-08-23 12:31:58 -04:00
Aditya Raj SinghandGitHub f1e5797c76 fix(agent): round load average to two decimals (#2245)
Every other metric in getSystemStats is stored through utils.TwoDecimals.
The load averages were assigned straight from gopsutil, so whatever the
platform reported was recorded verbatim.

On Linux that goes unnoticed because /proc/loadavg is already two decimal
places. Everywhere else it is not. macOS and BSD divide a fixed point
value by fscale and produce numbers like 2.55322265625, and the Windows
implementation synthesises the average as a decaying EWMA over the
processor queue length counter, so an idle machine reports values like
1.3667392689044936e-73 instead of 0.

The hub already treats two decimals as the canonical precision for this
field, since records.go rounds the load average when it averages records.
That left the raw agent records as the only place carrying full precision.
2026-08-21 17:30:19 -04:00
henrygd 6f92b9396d fix(hub): user alerts idor
fixes very unlikely scenario where user guesses another user's 15
character random system id and adds alerts for it
2026-08-21 17:25:09 -04:00
MartinandGitHub 946f2e6be1 Report the configured listen address after install (#2243)
The final message always echoed $PORT, which falls back to the default when -p
is not passed. Existing service files are kept as they are, so a plain upgrade
on a host with a custom port reported that the agent runs on 45876 regardless
of the actual configuration.

Read the address from the active service file instead. LISTEN is checked before
PORT to match the agent's own precedence in GetAddress, and the value is read as
text so host:port and unix socket paths are reported as configured.
2026-08-19 11:48:01 -04:00
ba90daf4d6 fix(scripts): update agent env vars on reinstall instead of skipping (#2107)
Co-authored-by: henrygd <hank@henrygd.me>
2026-08-19 11:37:11 -04:00
Toomore ChiangandGitHub aa1d67a122 fix(agent): strip invalid UTF-8 from battery names (#2241)
Battery names come from firmware (sysfs model_name on Linux), which does not
guarantee valid UTF-8. The hub decodes agent payloads using the default
fxamacker/cbor decode mode, which rejects invalid UTF-8, so a single bad byte
in a battery name makes the hub drop the entire payload and mark the system
down until the agent is downgraded.
2026-08-19 11:02:33 -04:00
Alec RubinandGitHub 68a3f8962a fix(agent): read /proc/uptime on linux instead of sysinfo(2) (#2180)
gopsutil's host.Uptime() calls the sysinfo(2) syscall. Inside an LXC
container lxcfs virtualizes /proc/uptime but cannot intercept a
syscall, so every container reported the host's uptime.

Reads /proc/uptime on linux and falls back to host.Uptime() if the file
is missing or unparseable, so other platforms are unchanged.
2026-08-18 15:18:33 -04:00
Ilya MuratovandGitHub 0eb3426619 fix(agent): discover fans on legacy hwmon parent devices (#2238) 2026-08-18 11:32:52 -04:00
Jan DziąsłoandGitHub 96beadc8c9 fix(agent): add fallback for CPU model detection on MIPS architectures (#2138)
gopsutil's cpu.Info() does not parse the 'cpu model' field from
/proc/cpuinfo, which is the only source of CPU model names on MIPS.
Add a fallback that reads /proc/cpuinfo directly and combines
'cpu model' (e.g. 'MIPS 1004Kc V2.15') with 'system type'
(e.g. 'MediaTek MT7621 ver:1 eco:3') for a complete identifier.

The fallback only triggers when gopsutil returns an empty ModelName,
so x86/ARM/other architectures are unaffected.
2026-08-18 10:34:41 -04:00
Sven van GinkelandGitHub 65a6f60304 fix(agent): fix QNAP MD RAID arrays incorrectly reported as FAILED (#2065) 2026-08-18 10:10:12 -04:00
hankandGitHub 54dae08631 chore(helm): update app version to 0.18.8 (#2235) 2026-08-17 17:36:01 -04:00
145 changed files with 15046 additions and 1376 deletions
+1 -1
View File
@@ -63,7 +63,7 @@ jobs:
uses: actions/checkout@v7
- name: Set up Helm
uses: azure/setup-helm@v4
uses: azure/setup-helm@v5
- name: Lint chart
run: helm lint "${{ matrix.chart.path }}" --set env.KEY=ci-placeholder
+27 -1
View File
@@ -48,6 +48,7 @@ type Agent struct {
keys []gossh.PublicKey // SSH public keys
smartManager *SmartManager // Manages SMART data
systemdManager *systemdManager // Manages systemd services
zfsManager *ZfsManager // Manages ZFS pool and dataset data
}
// NewAgent creates a new agent with the given data directory for persisting data.
@@ -121,6 +122,19 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
// initialize handler registry
agent.handlerRegistry = NewHandlerRegistry()
agent.zfsManager = newZfsManager()
// ZFS_INTERVAL env var to update ZFS detail data at this interval
if zfsIntervalEnv, exists := utils.GetEnv("ZFS_INTERVAL"); exists {
if duration, err := time.ParseDuration(zfsIntervalEnv); err == nil && duration > 0 {
agent.zfsManager.detailInterval = duration
agent.systemDetails.ZfsInterval = duration
slog.Info("ZFS_INTERVAL", "duration", duration)
} else {
slog.Warn("Invalid ZFS_INTERVAL", "err", err)
}
}
// initialize disk info
agent.initializeDiskInfo()
@@ -187,13 +201,25 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
}
if a.systemdManager.hasFreshStats {
data.SystemdServices = a.systemdManager.getServiceStats(nil, false)
data.SystemdServicesUpdated = true
// Preserve an explicit zero count so the hub can distinguish a fresh
// empty snapshot from a response that omitted systemd data.
if totalCount == 0 {
data.Info.Services = []uint16{0, 0}
}
}
}
data.Stats.ExtraFs = make(map[string]*system.FsStats)
data.Info.ExtraFsPct = make(map[string]float64)
for name, stats := range a.fsStats {
if !stats.Root && stats.DiskTotal > 0 {
if stats.Root {
if stats.Name != "" {
data.Info.RootDiskName = stats.Name
}
continue
}
if stats.DiskTotal > 0 {
// Use custom name if available, otherwise use device name
key := name
if stats.Name != "" {
+4 -1
View File
@@ -33,7 +33,10 @@ var errNoBatteries = errors.New("no readable batteries")
func normalizeBatteries(batteries []Battery) []Battery {
nameCounts := make(map[string]int, len(batteries))
for i := range batteries {
name := strings.TrimSpace(batteries[i].Name)
// Names come from firmware (e.g. sysfs model_name) and are not guaranteed to
// be valid UTF-8. Invalid bytes are rejected when the hub decodes the CBOR
// payload, which drops every metric for the system, so strip them here.
name := strings.TrimSpace(strings.ToValidUTF8(batteries[i].Name, ""))
if name == "" {
name = "Battery " + strconv.Itoa(i+1)
}
+13
View File
@@ -2,6 +2,7 @@ package battery
import (
"testing"
"unicode/utf8"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
@@ -33,3 +34,15 @@ func TestNormalizeBatteriesFallbackNames(t *testing.T) {
bats := normalizeBatteries([]Battery{{}, {}, {Name: "Mouse"}, {Name: "Mouse"}})
assert.Equal(t, []string{"Battery 1", "Battery 2", "Mouse", "Mouse (2)"}, []string{bats[0].Name, bats[1].Name, bats[2].Name, bats[3].Name})
}
func TestNormalizeBatteriesStripsInvalidUTF8(t *testing.T) {
// Firmware occasionally reports names that are not valid UTF-8 (a ThinkPad
// reporting "LNV-5B11K63024@\xd0" in model_name is a real example).
bats := normalizeBatteries([]Battery{{Name: "LNV-5B11K63024@\xd0"}, {Name: "\xff\xfe"}})
assert.Equal(t, "LNV-5B11K63024@", bats[0].Name)
// A name made up entirely of invalid bytes falls back to the generic name.
assert.Equal(t, "Battery 2", bats[1].Name)
for _, b := range bats {
assert.True(t, utf8.ValidString(b.Name))
}
}
+65 -2
View File
@@ -2,6 +2,7 @@ package agent
import (
"crypto/tls"
"crypto/x509"
"errors"
"fmt"
"log/slog"
@@ -27,6 +28,18 @@ const (
wsDeadline = 70 * time.Second
)
type caCertFileError struct {
err error
}
func (e *caCertFileError) Error() string {
return e.err.Error()
}
func (e *caCertFileError) Unwrap() error {
return e.err
}
// WebSocketClient manages the WebSocket connection between the agent and hub.
// It handles authentication, message routing, and connection lifecycle management.
type WebSocketClient struct {
@@ -40,6 +53,7 @@ type WebSocketClient struct {
hubRequest *common.HubRequest[cbor.RawMessage] // Reusable request structure for message parsing
lastConnectAttempt time.Time // Timestamp of last connection attempt
hubVerified bool // Whether the hub has been cryptographically verified
tlsConfig *tls.Config // Optional TLS configuration with custom CA certificates
}
// newWebSocketClient creates a new WebSocket client for the given agent.
@@ -61,6 +75,10 @@ func newWebSocketClient(agent *Agent) (client *WebSocketClient, err error) {
if err != nil {
return nil, err
}
client.tlsConfig, err = getTLSConfig()
if err != nil {
return nil, err
}
client.agent = agent
client.hubRequest = &common.HubRequest[cbor.RawMessage]{}
@@ -87,7 +105,52 @@ func getToken() (string, error) {
if err != nil {
return "", err
}
return strings.TrimSpace(string(tokenBytes)), nil
return parseTokenFile(string(tokenBytes), tokenFile)
}
// parseTokenFile reads a single token from TOKEN_FILE.
// Blank lines and comments are ignored. Multiple tokens are rejected because
// the agent supports only one outbound hub connection.
func parseTokenFile(contents, path string) (string, error) {
var token string
for line := range strings.Lines(contents) {
line = strings.TrimSpace(line)
if len(line) == 0 || strings.HasPrefix(line, "#") {
continue
}
if token != "" {
return "", fmt.Errorf("%s must contain a single token", path)
}
token = line
}
// An empty file keeps returning an empty token, as before: the caller decides
// what to do about it.
return token, nil
}
// getTLSConfig returns a TLS configuration containing the system certificate
// pool plus any certificates configured through CA_CERT_FILE. A nil config lets
// gws use Go's default TLS configuration and system roots.
func getTLSConfig() (*tls.Config, error) {
caCertFile, _ := utils.GetEnv("CA_CERT_FILE")
if caCertFile == "" {
return nil, nil
}
caCertPEM, err := os.ReadFile(caCertFile)
if err != nil {
return nil, &caCertFileError{fmt.Errorf("read CA_CERT_FILE %q: %w", caCertFile, err)}
}
rootCAs, err := x509.SystemCertPool()
if err != nil {
return nil, &caCertFileError{fmt.Errorf("load system CA certificate pool: %w", err)}
}
if !rootCAs.AppendCertsFromPEM(caCertPEM) {
return nil, &caCertFileError{fmt.Errorf("CA_CERT_FILE %q does not contain any valid PEM certificates", caCertFile)}
}
return &tls.Config{RootCAs: rootCAs}, nil
}
// getOptions returns the WebSocket client options, creating them if necessary.
@@ -112,7 +175,7 @@ func (client *WebSocketClient) getOptions() *gws.ClientOption {
client.options = &gws.ClientOption{
Addr: client.hubURL.String(),
TlsConfig: &tls.Config{InsecureSkipVerify: true},
TlsConfig: client.tlsConfig,
RequestHeader: http.Header{
"User-Agent": []string{getUserAgent()},
"X-Token": []string{client.token},
+196
View File
@@ -4,8 +4,19 @@ package agent
import (
"crypto/ed25519"
"crypto/rand"
"crypto/rsa"
"crypto/tls"
"crypto/x509"
"crypto/x509/pkix"
"encoding/pem"
"math/big"
"net"
"net/http"
"net/http/httptest"
"net/url"
"os"
"path/filepath"
"strings"
"testing"
"time"
@@ -15,6 +26,7 @@ import (
"github.com/henrygd/beszel/internal/common"
"github.com/fxamacker/cbor/v2"
"github.com/lxzan/gws"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"golang.org/x/crypto/ssh"
@@ -164,6 +176,155 @@ func TestWebSocketClient_GetOptions(t *testing.T) {
}
}
func TestWebSocketClient_TLSVerification(t *testing.T) {
agent := createTestAgent(t)
serverCert, serverCertPEM := newSelfSignedServerCertificate(t)
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
server := httptest.NewUnstartedServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
conn, err := upgrader.Upgrade(w, r)
if err == nil {
go conn.ReadLoop()
}
}))
server.TLS = &tls.Config{Certificates: []tls.Certificate{serverCert}}
server.StartTLS()
t.Cleanup(server.Close)
caCertFile := filepath.Join(t.TempDir(), "hub-ca.crt")
require.NoError(t, os.WriteFile(caCertFile, serverCertPEM, 0600))
newClient := func(t *testing.T, caCertFile string) *WebSocketClient {
t.Helper()
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", caCertFile)
client, err := newWebSocketClient(agent)
require.NoError(t, err)
return client
}
t.Run("system roots are used by default", func(t *testing.T) {
client := newClient(t, "")
assert.Nil(t, client.getOptions().TlsConfig)
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
require.Error(t, err)
})
t.Run("custom CA trusts self-signed certificate", func(t *testing.T) {
systemRoots, err := x509.SystemCertPool()
require.NoError(t, err)
client := newClient(t, caCertFile)
assert.Greater(t, len(client.getOptions().TlsConfig.RootCAs.Subjects()), len(systemRoots.Subjects()))
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
require.NoError(t, err)
require.NoError(t, conn.NetConn().Close())
})
t.Run("custom CA does not bypass hostname verification", func(t *testing.T) {
client := newClient(t, caCertFile)
client.getOptions().TlsConfig.ServerName = "wrong.example.com"
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
require.Error(t, err)
})
}
func TestWebSocketClient_NonTLSConnection(t *testing.T) {
agent := createTestAgent(t)
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
conn, err := upgrader.Upgrade(w, r)
if err == nil {
go conn.ReadLoop()
}
}))
t.Cleanup(server.Close)
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", "")
client, err := newWebSocketClient(agent)
require.NoError(t, err)
assert.Nil(t, client.getOptions().TlsConfig)
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
require.NoError(t, err)
require.NoError(t, conn.NetConn().Close())
}
func TestGetTLSConfigErrors(t *testing.T) {
tempDir := t.TempDir()
testCases := []struct {
name string
path string
contents []byte
errorMatch string
}{
{
name: "missing file",
path: filepath.Join(tempDir, "missing.pem"),
errorMatch: "read CA_CERT_FILE",
},
{
name: "unreadable path",
path: tempDir,
errorMatch: "read CA_CERT_FILE",
},
{
name: "empty file",
path: filepath.Join(tempDir, "empty.pem"),
contents: []byte{},
errorMatch: "does not contain any valid PEM certificates",
},
{
name: "malformed file",
path: filepath.Join(tempDir, "malformed.pem"),
contents: []byte("not a PEM certificate"),
errorMatch: "does not contain any valid PEM certificates",
},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
if tc.contents != nil {
require.NoError(t, os.WriteFile(tc.path, tc.contents, 0600))
}
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", tc.path)
tlsConfig, err := getTLSConfig()
require.Error(t, err)
assert.Nil(t, tlsConfig)
assert.Contains(t, err.Error(), tc.errorMatch)
assert.Contains(t, err.Error(), tc.path)
})
}
}
func newSelfSignedServerCertificate(t *testing.T) (tls.Certificate, []byte) {
t.Helper()
privateKey, err := rsa.GenerateKey(rand.Reader, 2048)
require.NoError(t, err)
template := &x509.Certificate{
SerialNumber: big.NewInt(1),
Subject: pkix.Name{CommonName: "127.0.0.1"},
NotBefore: time.Now().Add(-time.Hour),
NotAfter: time.Now().Add(time.Hour),
IPAddresses: []net.IP{net.ParseIP("127.0.0.1")},
KeyUsage: x509.KeyUsageDigitalSignature | x509.KeyUsageKeyEncipherment | x509.KeyUsageCertSign,
ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageServerAuth},
BasicConstraintsValid: true,
IsCA: true,
}
certDER, err := x509.CreateCertificate(rand.Reader, template, template, &privateKey.PublicKey, privateKey)
require.NoError(t, err)
certPEM := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: certDER})
keyPEM := pem.EncodeToMemory(&pem.Block{Type: "RSA PRIVATE KEY", Bytes: x509.MarshalPKCS1PrivateKey(privateKey)})
certificate, err := tls.X509KeyPair(certPEM, keyPEM)
require.NoError(t, err)
return certificate, certPEM
}
// TestWebSocketClient_VerifySignature tests signature verification
func TestWebSocketClient_VerifySignature(t *testing.T) {
agent := createTestAgent(t)
@@ -409,6 +570,41 @@ func TestGetToken(t *testing.T) {
assert.Equal(t, expectedToken, token)
})
t.Run("TOKEN_FILE with surrounding blank lines and comments", func(t *testing.T) {
expectedToken := "test-token-with-noise"
tokenFile := filepath.Join(t.TempDir(), "token")
require.NoError(t, os.WriteFile(tokenFile, []byte("# hub token\n\n"+expectedToken+"\n\n"), 0o600))
t.Setenv("TOKEN_FILE", tokenFile)
token, err := getToken()
assert.NoError(t, err)
assert.Equal(t, expectedToken, token)
})
t.Run("TOKEN_FILE with multiple tokens is rejected", func(t *testing.T) {
tokenFile := filepath.Join(t.TempDir(), "token")
require.NoError(t, os.WriteFile(tokenFile, []byte("11111111-1111-1111-1111-111111111111\n22222222-2222-2222-2222-222222222222\n"), 0o600))
t.Setenv("TOKEN_FILE", tokenFile)
token, err := getToken()
require.Error(t, err)
assert.Empty(t, token)
assert.Contains(t, err.Error(), "must contain a single token")
})
t.Run("TOKEN_FILE holding only comments behaves like an empty file", func(t *testing.T) {
tokenFile := filepath.Join(t.TempDir(), "token")
require.NoError(t, os.WriteFile(tokenFile, []byte("\n# only a comment\n"), 0o600))
t.Setenv("TOKEN_FILE", tokenFile)
token, err := getToken()
assert.NoError(t, err)
assert.Equal(t, "", token)
})
t.Run("token from BESZEL_AGENT_TOKEN_FILE", func(t *testing.T) {
// Create a temporary token file
expectedToken := "test-token-from-beszel-file"
+7 -1
View File
@@ -87,6 +87,10 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
wsClient, err := newWebSocketClient(c.agent)
if err != nil {
var caCertErr *caCertFileError
if errors.As(err, &caCertErr) {
return err
}
slog.Warn("Error creating WebSocket client", "err", err)
}
c.wsClient = wsClient
@@ -151,7 +155,9 @@ func (c *ConnectionManager) handleEvent(event ConnectionEvent) {
case WebSocketConnect:
c.handleStateChange(WebSocketConnected)
case SSHConnect:
c.handleStateChange(SSHConnected)
if c.State == Disconnected {
c.handleStateChange(SSHConnected)
}
case WebSocketDisconnect:
if c.State == WebSocketConnected {
c.handleStateChange(Disconnected)
+19
View File
@@ -114,6 +114,12 @@ func TestConnectionManager_EventHandling(t *testing.T) {
event: SSHConnect,
expectedState: SSHConnected,
},
{
name: "SSH connect from WebSocket connected (no change)",
initialState: WebSocketConnected,
event: SSHConnect,
expectedState: WebSocketConnected,
},
{
name: "WebSocket disconnect from connected",
initialState: WebSocketConnected,
@@ -265,6 +271,19 @@ func TestConnectionManager_StartWithInvalidConfig(t *testing.T) {
assert.Error(t, err, "Should error when starting already started connection manager")
}
func TestConnectionManager_StartRejectsInvalidCACertFile(t *testing.T) {
agent := createTestAgent(t)
cm := agent.connectionManager
t.Setenv("BESZEL_AGENT_HUB_URL", "https://hub.example.com")
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", t.TempDir())
err := cm.Start(ServerOptions{})
require.Error(t, err)
assert.Contains(t, err.Error(), "read CA_CERT_FILE")
assert.Nil(t, cm.eventChan)
}
// TestConnectionManager_CloseWebSocket tests WebSocket closing
func TestConnectionManager_CloseWebSocket(t *testing.T) {
agent := createTestAgent(t)
+12 -4
View File
@@ -12,6 +12,14 @@ import (
"github.com/stretchr/testify/require"
)
func invalidDataDir(t *testing.T) string {
t.Helper()
filePath := filepath.Join(t.TempDir(), "file")
require.NoError(t, os.WriteFile(filePath, nil, 0644))
return filepath.Join(filePath, "data")
}
func TestGetDataDir(t *testing.T) {
// Test with explicit dataDir parameter
t.Run("explicit data dir", func(t *testing.T) {
@@ -48,7 +56,7 @@ func TestGetDataDir(t *testing.T) {
// Test with invalid explicit dataDir
t.Run("invalid explicit data dir", func(t *testing.T) {
invalidPath := "/invalid/path/that/cannot/be/created"
invalidPath := invalidDataDir(t)
_, err := GetDataDir(invalidPath)
assert.Error(t, err)
})
@@ -78,7 +86,7 @@ func TestTestDataDirs(t *testing.T) {
// Test with multiple directories, first one valid
t.Run("multiple dirs - first valid", func(t *testing.T) {
tempDir := t.TempDir()
invalidDir := "/invalid/path"
invalidDir := invalidDataDir(t)
result, err := testDataDirs([]string{tempDir, invalidDir})
require.NoError(t, err)
assert.Equal(t, tempDir, result)
@@ -87,7 +95,7 @@ func TestTestDataDirs(t *testing.T) {
// Test with multiple directories, second one valid
t.Run("multiple dirs - second valid", func(t *testing.T) {
tempDir := t.TempDir()
invalidDir := "/invalid/path"
invalidDir := invalidDataDir(t)
result, err := testDataDirs([]string{invalidDir, tempDir})
require.NoError(t, err)
assert.Equal(t, tempDir, result)
@@ -109,7 +117,7 @@ func TestTestDataDirs(t *testing.T) {
// Test with no valid directories
t.Run("no valid directories", func(t *testing.T) {
invalidPaths := []string{"/invalid/path1", "/invalid/path2"}
invalidPaths := []string{invalidDataDir(t), invalidDataDir(t)}
_, err := testDataDirs(invalidPaths)
assert.Error(t, err)
assert.Contains(t, err.Error(), "data directory not found")
+51 -18
View File
@@ -18,10 +18,11 @@ import (
// fsRegistrationContext holds the shared lookup state needed to resolve a
// filesystem into the tracked fsStats key and metadata.
type fsRegistrationContext struct {
filesystem string // value of optional FILESYSTEM env var
isWindows bool
efPath string // path to extra filesystems (default "/extra-filesystems")
diskIoCounters map[string]disk.IOCountersStat
filesystem string // device part of optional FILESYSTEM env var
filesystemName string // optional custom name from FILESYSTEM=device__name
isWindows bool
efPath string // path to extra filesystems (default "/extra-filesystems")
diskIoCounters map[string]disk.IOCountersStat
}
// diskDiscovery groups the transient state for a single initializeDiskInfo run so
@@ -177,7 +178,7 @@ func (d *diskDiscovery) addConfiguredRootFs() bool {
for _, p := range d.partitions {
if filesystemMatchesPartitionSetting(d.ctx.filesystem, p) {
d.addFsStat(p.Device, p.Mountpoint, true, "")
d.addFsStat(p.Device, p.Mountpoint, true, d.ctx.filesystemName)
return true
}
}
@@ -185,7 +186,7 @@ func (d *diskDiscovery) addConfiguredRootFs() bool {
// FILESYSTEM may name a physical disk absent from partitions (e.g. ZFS lists
// dataset paths like zroot/ROOT/default, not block devices).
if ioKey, match := findIoDevice(d.ctx.filesystem, d.ctx.diskIoCounters); match {
d.agent.fsStats[ioKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint}
d.agent.fsStats[ioKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint, Name: d.ctx.filesystemName}
return true
}
@@ -300,7 +301,8 @@ func (d *diskDiscovery) addExtraFilesystemFolders(folderNames []string) {
// Sets up the filesystems to monitor for disk usage and I/O.
func (a *Agent) initializeDiskInfo() {
filesystem, _ := utils.GetEnv("FILESYSTEM")
filesystemRaw, _ := utils.GetEnv("FILESYSTEM")
filesystem, filesystemName := parseFilesystemEntry(filesystemRaw)
hasRoot := false
isWindows := runtime.GOOS == "windows"
@@ -323,10 +325,11 @@ func (a *Agent) initializeDiskInfo() {
}
slog.Debug("Disk I/O", "diskstats", diskIoCounters)
ctx := fsRegistrationContext{
filesystem: filesystem,
isWindows: isWindows,
diskIoCounters: diskIoCounters,
efPath: "/extra-filesystems",
filesystem: filesystem,
filesystemName: filesystemName,
isWindows: isWindows,
diskIoCounters: diskIoCounters,
efPath: "/extra-filesystems",
}
// Get the appropriate root mount point for this system
@@ -534,7 +537,16 @@ func normalizeDeviceName(value string) string {
func (a *Agent) initializeDiskIoStats(diskIoCounters map[string]disk.IOCountersStat) {
a.fsNames = a.fsNames[:0]
now := time.Now()
// ZFS datasets have no /proc/diskstats entry, so they are excluded from
// I/O tracking instead of warning about a missing device (#1541).
var zfsMountpoints map[string]bool
if a.zfsManager != nil {
zfsMountpoints = a.zfsManager.ZfsMountpoints()
}
for device, stats := range a.fsStats {
if zfsMountpoints[stats.Mountpoint] {
continue
}
// skip if not in diskIoCounters
d, exists := diskIoCounters[device]
if !exists {
@@ -559,20 +571,31 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
!a.lastDiskUsageUpdate.IsZero() &&
time.Since(a.lastDiskUsageUpdate) < a.diskUsageCacheDuration
// ZFS dataset mountpoints use `zfs list` values because statfs(2) reports
// dataset-level usage that excludes child datasets (#1541).
var zfsUsage map[string]zfsDatasetUsage
if a.zfsManager != nil {
zfsUsage = a.zfsManager.DatasetUsage()
}
// disk usage
for _, stats := range a.fsStats {
// Skip non-root filesystems if caching is active
if cacheExtraFs && !stats.Root {
continue
}
if d, err := disk.Usage(stats.Mountpoint); err == nil {
stats.DiskTotal = utils.BytesToGigabytes(d.Total)
stats.DiskUsed = utils.BytesToGigabytes(d.Used)
if stats.Root {
systemStats.DiskTotal = utils.BytesToGigabytes(d.Total)
systemStats.DiskUsed = utils.BytesToGigabytes(d.Used)
systemStats.DiskPct = utils.TwoDecimals(d.UsedPercent)
var total, used uint64
var usedPct float64
if u, ok := zfsUsage[stats.Mountpoint]; ok {
total = u.used + u.avail
used = u.used
if total > 0 {
usedPct = float64(used) / float64(total) * 100
}
} else if d, err := disk.Usage(stats.Mountpoint); err == nil {
total = d.Total
used = d.Used
usedPct = d.UsedPercent
} else {
// reset stats if error (likely unmounted)
slog.Error("Error getting disk stats", "name", stats.Mountpoint, "err", err)
@@ -580,6 +603,14 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
stats.DiskUsed = 0
stats.TotalRead = 0
stats.TotalWrite = 0
continue
}
stats.DiskTotal = utils.BytesToGigabytes(total)
stats.DiskUsed = utils.BytesToGigabytes(used)
if stats.Root {
systemStats.DiskTotal = stats.DiskTotal
systemStats.DiskUsed = stats.DiskUsed
systemStats.DiskPct = utils.TwoDecimals(usedPct)
}
}
@@ -696,6 +727,8 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
systemStats.DiskWritePs = stats.DiskWritePs
systemStats.DiskIO[0] = diskIORead
systemStats.DiskIO[1] = diskIOWrite
systemStats.DiskIOTotal[0] = d.ReadBytes
systemStats.DiskIOTotal[1] = d.WriteBytes
systemStats.DiskIoStats[0] = diskReadTime
systemStats.DiskIoStats[1] = diskWriteTime
systemStats.DiskIoStats[2] = diskIoUtilPct
+9 -12
View File
@@ -78,14 +78,7 @@ func TestParseFilesystemEntry(t *testing.T) {
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
fsEntry := strings.TrimSpace(tt.input)
var fs, customName string
if parts := strings.SplitN(fsEntry, "__", 2); len(parts) == 2 {
fs = strings.TrimSpace(parts[0])
customName = strings.TrimSpace(parts[1])
} else {
fs = fsEntry
}
fs, customName := parseFilesystemEntry(tt.input)
assert.Equal(t, tt.expectedFs, fs)
assert.Equal(t, tt.expectedName, customName)
@@ -287,8 +280,9 @@ func TestAddConfiguredRootFs(t *testing.T) {
rootMountPoint: "/",
partitions: []disk.PartitionStat{{Device: "/dev/ada0p2", Mountpoint: "/"}},
ctx: fsRegistrationContext{
filesystem: "/dev/ada0p2",
isWindows: false,
filesystem: "/dev/ada0p2",
filesystemName: "root disk",
isWindows: false,
diskIoCounters: map[string]disk.IOCountersStat{
"ada0": {Name: "ada0", ReadBytes: 1000, WriteBytes: 1000},
},
@@ -302,6 +296,7 @@ func TestAddConfiguredRootFs(t *testing.T) {
assert.True(t, exists)
assert.True(t, stats.Root)
assert.Equal(t, "/", stats.Mountpoint)
assert.Equal(t, "root disk", stats.Name)
})
t.Run("adds root from io device when partition is missing", func(t *testing.T) {
@@ -310,8 +305,9 @@ func TestAddConfiguredRootFs(t *testing.T) {
agent: agent,
rootMountPoint: "/sysroot",
ctx: fsRegistrationContext{
filesystem: "zroot",
isWindows: false,
filesystem: "zroot",
filesystemName: "root pool",
isWindows: false,
diskIoCounters: map[string]disk.IOCountersStat{
"nda0": {Name: "nda0", Label: "zroot", ReadBytes: 1000, WriteBytes: 1000},
},
@@ -325,6 +321,7 @@ func TestAddConfiguredRootFs(t *testing.T) {
assert.True(t, exists)
assert.True(t, stats.Root)
assert.Equal(t, "/sysroot", stats.Mountpoint)
assert.Equal(t, "root pool", stats.Name)
})
t.Run("returns false when filesystem cannot be resolved", func(t *testing.T) {
+109
View File
@@ -0,0 +1,109 @@
//go:build testing
package agent
import (
"testing"
"github.com/henrygd/beszel/agent/zfs"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/shirou/gopsutil/v4/disk"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
// TestUpdateDiskUsageZfsMountpoint verifies that a filesystem whose mountpoint
// is a ZFS dataset reports `zfs list` usage (which includes child datasets)
// instead of the dataset-scoped statfs values (#1541).
func TestUpdateDiskUsageZfsMountpoint(t *testing.T) {
zm := &ZfsManager{}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
}, nil
}
agent := &Agent{
fsStats: map[string]*system.FsStats{
"tank": {Root: false, Mountpoint: "/tank"},
},
zfsManager: zm,
}
var stats system.Stats
agent.updateDiskUsage(&stats)
fs := agent.fsStats["tank"]
require.NotNil(t, fs)
assert.Equal(t, 22350.81, fs.DiskTotal) // (used + avail) in GiB
assert.Equal(t, 11175.87, fs.DiskUsed)
// Non-root filesystems do not populate system-level stats.
assert.Equal(t, float64(0), stats.DiskTotal)
}
// TestUpdateDiskUsageZfsRootPopulatesSystemStats verifies the root disk values
// are derived from ZFS usage when the root mountpoint is a ZFS dataset.
func TestUpdateDiskUsageZfsRootPopulatesSystemStats(t *testing.T) {
zm := &ZfsManager{}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{
{Name: "rpool/ROOT/pve-1", Used: 900000000000, Avail: 300000000000, Mountpoint: "/"},
}, nil
}
agent := &Agent{
fsStats: map[string]*system.FsStats{
"rpool/ROOT/pve-1": {Root: true, Mountpoint: "/"},
},
zfsManager: zm,
}
var stats system.Stats
agent.updateDiskUsage(&stats)
assert.Equal(t, 1117.59, agent.fsStats["rpool/ROOT/pve-1"].DiskTotal)
assert.Equal(t, 838.19, agent.fsStats["rpool/ROOT/pve-1"].DiskUsed)
assert.Equal(t, 75.0, stats.DiskPct)
assert.Equal(t, 1117.59, stats.DiskTotal)
assert.Equal(t, 838.19, stats.DiskUsed)
}
// TestUpdateDiskUsageWithoutZfsManager falls back to statfs when no manager is
// present (e.g. tests constructing bare Agent values).
func TestUpdateDiskUsageWithoutZfsManager(t *testing.T) {
agent := &Agent{
fsStats: map[string]*system.FsStats{
"root": {Root: true, Mountpoint: "/"},
},
}
var stats system.Stats
agent.updateDiskUsage(&stats)
assert.True(t, agent.fsStats["root"].DiskTotal > 0, "root usage should come from statfs")
assert.True(t, stats.DiskTotal > 0)
}
// TestInitializeDiskIoStatsSkipsZfsMountpoints verifies ZFS filesystems are
// excluded from diskstats I/O tracking instead of warning about a missing device.
func TestInitializeDiskIoStatsSkipsZfsMountpoints(t *testing.T) {
zm := &ZfsManager{}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{{Name: "tank", Mountpoint: "/tank"}}, nil
}
agent := &Agent{
fsStats: map[string]*system.FsStats{
"tank": {Root: false, Mountpoint: "/tank"},
"sda1": {Root: false, Mountpoint: "/mnt/data"},
},
zfsManager: zm,
diskPrev: make(map[uint16]map[string]prevDisk),
}
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
"sda1": {Name: "sda1", ReadBytes: 100, WriteBytes: 100},
})
assert.Equal(t, []string{"sda1"}, agent.fsNames)
assert.Equal(t, uint64(100), agent.fsStats["sda1"].TotalRead)
// ZFS entry is present but untouched by diskstats initialization.
assert.Equal(t, uint64(0), agent.fsStats["tank"].TotalRead)
}
+2
View File
@@ -729,6 +729,7 @@ func TestGetDockerStatsChecksDockerVersionAfterContainerList(t *testing.T) {
stats, err := dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
require.NotNil(t, stats, "A successful empty snapshot must remain distinguishable from a collection failure")
assert.Empty(t, stats)
assert.True(t, dm.dockerVersionChecked)
assert.Equal(t, tt.expectedGood, dm.goodDockerVersion)
@@ -742,6 +743,7 @@ func TestGetDockerStatsChecksDockerVersionAfterContainerList(t *testing.T) {
stats, err = dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
require.NotNil(t, stats, "A successful empty snapshot must remain distinguishable from a collection failure")
assert.Empty(t, stats)
assert.Equal(t, tt.expectedGood, dm.goodDockerVersion)
assert.Equal(t, tt.expectedPodman, dm.usingPodman)
+19 -3
View File
@@ -72,14 +72,30 @@ func discoverHwmonFans(root string) ([]fanSensor, error) {
var sensors []fanSensor
for _, entry := range entries {
chipDir := filepath.Join(root, entry.Name())
chipName := utils.ReadStringFile(filepath.Join(chipDir, "name"))
sensorDir := chipDir
inputs, _ := filepath.Glob(filepath.Join(sensorDir, "fan*_input"))
// Some legacy hwmon drivers (notably applesmc) register a hwmon class
// device but create fan attributes on the parent platform device. In
// sysfs that parent is exposed through hwmonN/device.
if len(inputs) == 0 {
deviceDir := filepath.Join(chipDir, "device")
if deviceInputs, _ := filepath.Glob(filepath.Join(deviceDir, "fan*_input")); len(deviceInputs) > 0 {
sensorDir = deviceDir
inputs = deviceInputs
}
}
chipName := utils.ReadStringFile(filepath.Join(sensorDir, "name"))
if chipName == "" {
chipName = utils.ReadStringFile(filepath.Join(chipDir, "name"))
}
if chipName == "" {
chipName = entry.Name()
}
inputs, _ := filepath.Glob(filepath.Join(chipDir, "fan*_input"))
for _, inputPath := range inputs {
base := strings.TrimSuffix(filepath.Base(inputPath), "_input")
label := utils.ReadStringFile(filepath.Join(chipDir, base+"_label"))
label := utils.ReadStringFile(filepath.Join(sensorDir, base+"_label"))
key := chipName + "_" + base
if label != "" {
key = chipName + "_" + label
+18
View File
@@ -50,6 +50,24 @@ func TestReadHwmonFans(t *testing.T) {
}, fans)
}
// TestReadHwmonFansLegacyParent verifies legacy hwmon layouts such as applesmc,
// where the hwmon class node exists but fan attributes live on hwmonN/device.
func TestReadHwmonFansLegacyParent(t *testing.T) {
root := t.TempDir()
deviceDir := filepath.Join(root, "devices", "applesmc.768")
writeFile(t, filepath.Join(deviceDir, "name"), "applesmc\n")
writeFile(t, filepath.Join(deviceDir, "fan1_input"), "1202\n")
writeFile(t, filepath.Join(deviceDir, "fan1_label"), "Exhaust\n")
chipDir := filepath.Join(root, "hwmon1")
require.NoError(t, os.MkdirAll(chipDir, 0o755))
require.NoError(t, os.Symlink(deviceDir, filepath.Join(chipDir, "device")))
fans, err := readHwmonFans(root)
require.NoError(t, err)
assert.Equal(t, map[string]uint16{"applesmc_Exhaust": 1202}, fans)
}
// TestReadHwmonFansMissingRoot returns an error rather than panicking when the
// hwmon root doesn't exist (e.g. running on a kernel without hwmon support).
func TestReadHwmonFansMissingRoot(t *testing.T) {
+3
View File
@@ -50,6 +50,9 @@ func generateFingerprint(hostname, cpuModel string) string {
if info, err := cpu.Info(); err == nil && len(info) > 0 {
cpuModel = info[0].ModelName
}
if cpuModel == "" {
cpuModel = getCpuModelFromCpuinfo()
}
}
fingerprint = hostname + cpuModel
}
+8 -4
View File
@@ -361,12 +361,16 @@ func (gm *GPUManager) calculateGPUAverage(id string, gpu *system.GPUData, cacheK
// If no new data arrived
if deltaCount == 0 {
// If GPU appears suspended (instantaneous values are 0), return zero values
// Otherwise return last known average for temporary collection gaps
if gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
// Only discrete GPUs report temp/memory, so treat all-zero as suspended (return zeros).
// Engine-based (Intel) GPUs don't, so carry the last average forward across sample gaps.
if gpu.Engines == nil && gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
return system.GPUData{Name: gpu.Name}
}
return gm.lastAvgData[id] // zero value if not found
lastAvg := gm.lastAvgData[id] // zero value if not found
if lastAvg.Name == "" {
lastAvg.Name = gpu.Name
}
return lastAvg
}
// Calculate new average
+36
View File
@@ -566,6 +566,42 @@ func TestGetCurrentData(t *testing.T) {
assert.EqualValues(t, 2, gm.GpuDataMap["0"].Count, "Count should still be 2")
})
t.Run("carries Intel GPU average forward between samples", func(t *testing.T) {
// Intel GPUs report no temp/memory, so between-sample gaps (delta 0) must
// reuse the last average instead of returning zeros and blanking the chart.
gm := &GPUManager{
GpuDataMap: map[string]*system.GPUData{
"0": {
Name: "GPU",
Usage: 0, // derived from engines for Intel
Power: 200, // averages to 100 over 2 counts
PowerPkg: 60, // averages to 30 over 2 counts
Count: 2,
Engines: map[string]float64{
"Render/3D": 80, // averages to 40
"Video": 20, // averages to 10
},
},
},
}
cacheKey := uint16(1000) // realtime cache key
// First collection - computes and stores averages
result1 := gm.GetCurrentData(cacheKey)
assert.InDelta(t, 100.0, result1["0"].Power, 0.01)
assert.InDelta(t, 30.0, result1["0"].PowerPkg, 0.01)
assert.InDelta(t, 40.0, result1["0"].Engines["Render/3D"], 0.01)
// Second collection with no new sample (count unchanged, temp/mem still 0).
// Must carry the last average forward rather than blanking to zero.
result2 := gm.GetCurrentData(cacheKey)
assert.Equal(t, "GPU", result2["0"].Name, "Name should be preserved")
assert.InDelta(t, 100.0, result2["0"].Power, 0.01, "Should reuse last average power, not 0")
assert.InDelta(t, 30.0, result2["0"].PowerPkg, 0.01, "Should reuse last average package power, not 0")
assert.InDelta(t, 40.0, result2["0"].Engines["Render/3D"], 0.01, "Should reuse last average engine usage")
})
t.Run("tracks separate averages per cache key", func(t *testing.T) {
gm := &GPUManager{
GpuDataMap: map[string]*system.GPUData{
+18
View File
@@ -51,6 +51,7 @@ func NewHandlerRegistry() *HandlerRegistry {
registry.Register(common.GetContainerInfo, &GetContainerInfoHandler{})
registry.Register(common.GetSmartData, &GetSmartDataHandler{})
registry.Register(common.GetSystemdInfo, &GetSystemdInfoHandler{})
registry.Register(common.GetZfsData, &GetZfsDataHandler{})
return registry
}
@@ -178,6 +179,23 @@ func (h *GetSmartDataHandler) Handle(hctx *HandlerContext) error {
}, hctx.RequestID)
}
////////////////////////////////////////////////////////////////////////////
////////////////////////////////////////////////////////////////////////////
// GetZfsDataHandler handles ZFS detail data requests
type GetZfsDataHandler struct{}
func (h *GetZfsDataHandler) Handle(hctx *HandlerContext) error {
if hctx.Agent.zfsManager == nil {
return hctx.SendResponse(nil, hctx.RequestID)
}
var req common.ZfsDataRequest
if err := cbor.Unmarshal(hctx.Request.Data, &req); err != nil {
return err
}
return hctx.SendResponse(hctx.Agent.zfsManager.GetDetail(req.Force), hctx.RequestID)
}
////////////////////////////////////////////////////////////////////////////
////////////////////////////////////////////////////////////////////////////
////////////////////////////////////////////////////////////////////////////
+28
View File
@@ -4,8 +4,10 @@ package agent
import (
"testing"
"time"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/agent/zfs"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/smart"
"github.com/stretchr/testify/assert"
@@ -30,6 +32,32 @@ func TestNewAgentResponseSmartData(t *testing.T) {
assert.True(t, response.SmartComplete)
}
func TestGetZfsDataHandlerForceRefresh(t *testing.T) {
poolCalls := 0
zm := &ZfsManager{detailInterval: time.Hour}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
poolCalls++
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
}
zm.poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
zm.GetDetail(false)
requestData, err := cbor.Marshal(common.ZfsDataRequest{Force: true})
assert.NoError(t, err)
ctx := &HandlerContext{
Agent: &Agent{zfsManager: zm},
Request: &common.HubRequest[cbor.RawMessage]{
Action: common.GetZfsData,
Data: requestData,
},
SendResponse: func(any, *uint32) error { return nil },
}
assert.NoError(t, (&GetZfsDataHandler{}).Handle(ctx))
assert.Equal(t, 2, poolCalls)
}
func (m *MockHandler) Handle(ctx *HandlerContext) error {
if m.handleFunc != nil {
return m.handleFunc(ctx)
+64 -9
View File
@@ -17,15 +17,17 @@ import (
var mdraidSysfsRoot = "/sys"
type mdraidHealth struct {
level string
arrayState string
degraded uint64
raidDisks uint64
syncAction string
syncCompleted string
syncSpeed string
mismatchCnt uint64
capacity uint64
level string
arrayState string
degraded uint64
faultyDisks uint64
populatedDisks uint64
raidDisks uint64
syncAction string
syncCompleted string
syncSpeed string
mismatchCnt uint64
capacity uint64
}
// scanMdraidDevices discovers Linux md arrays exposed in sysfs.
@@ -92,6 +94,9 @@ func (sm *SmartManager) collectMdraidHealth(deviceInfo *DeviceInfo) (bool, error
if health.degraded > 0 {
attrs = append(attrs, &smart.SmartAttribute{Name: "Degraded", RawValue: health.degraded})
}
if health.faultyDisks > 0 {
attrs = append(attrs, &smart.SmartAttribute{Name: "FaultyDisks", RawValue: health.faultyDisks})
}
if health.syncAction != "" {
attrs = append(attrs, &smart.SmartAttribute{Name: "SyncAction", RawString: health.syncAction})
}
@@ -152,6 +157,7 @@ func readMdraidHealth(blockName string) (mdraidHealth, bool) {
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "degraded")); ok {
out.degraded = val
}
out.faultyDisks, out.populatedDisks = countMdraidMemberStates(blockName, mdraidSysfsRoot)
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "mismatch_cnt")); ok {
out.mismatchCnt = val
}
@@ -177,7 +183,19 @@ func mdraidSmartStatus(health mdraidHealth) string {
case "resync", "recover", "reshape":
return "WARNING"
}
// Use actual faulty member count rather than the degraded counter, which
// equals raid_disks minus active_disks. On QNAP systems raid_disks may be
// set to a large value (e.g. 32) while only a few slots are ever used,
// making degraded misleadingly large despite zero failed disks.
if health.faultyDisks > 0 {
return "FAILED"
}
if health.degraded > 0 {
if isSparseSlotDegraded(health) {
// A sysfs snapshot cannot distinguish reserved slots from a removed
// member on sparse arrays, so report the ambiguity as a warning.
return "WARNING"
}
return "FAILED"
}
if health.mismatchCnt > 0 {
@@ -196,6 +214,43 @@ func mdraidSmartStatus(health mdraidHealth) string {
return "UNKNOWN"
}
// countMdraidMemberStates reads member device directories under
// block/<name>/md and returns how many are explicitly marked "faulty", plus
// how many are populated at all (regardless of state). populatedDisks lets
// callers distinguish RAID slots that were never used (QNAP reserves far
// more raid_disks than it ever populates) from members that went missing.
func countMdraidMemberStates(blockName, root string) (faultyDisks, populatedDisks uint64) {
devDir := filepath.Join(root, "block", blockName, "md")
entries, err := os.ReadDir(devDir)
if err != nil {
return 0, 0
}
for _, ent := range entries {
if !strings.HasPrefix(ent.Name(), "dev-") {
continue
}
populatedDisks++
statePath := filepath.Join(devDir, ent.Name(), "state")
state := utils.ReadStringFile(statePath)
if strings.Contains(state, "faulty") {
faultyDisks++
}
}
return faultyDisks, populatedDisks
}
// isSparseSlotDegraded reports whether a non-zero "degraded" count may be
// explained by RAID slots that were never populated. QNAP configures system
// arrays with raid_disks set to a large fixed maximum (e.g. 32) far beyond the
// handful of slots it ever populates, so sparse slots outnumber populated ones.
func isSparseSlotDegraded(health mdraidHealth) bool {
if health.populatedDisks == 0 || health.raidDisks <= health.populatedDisks {
return false
}
sparseSlots := health.raidDisks - health.populatedDisks
return sparseSlots > health.populatedDisks
}
// isMdraidBlockName matches /dev/mdN-style block device names.
func isMdraidBlockName(name string) bool {
if !strings.HasPrefix(name, "md") {
+74 -3
View File
@@ -40,6 +40,15 @@ func TestMdraidMockSysfsScanAndCollect(t *testing.T) {
write(filepath.Join(mdDir, "sync_completed"), "10%\n")
write(filepath.Join(mdDir, "sync_speed"), "100M\n")
write(filepath.Join(mdDir, "mismatch_cnt"), "0\n")
// Simulate two healthy member devices (no faulty state).
for _, dev := range []string{"dev-sda", "dev-sdb"} {
devPath := filepath.Join(mdDir, dev)
if err := os.MkdirAll(devPath, 0o755); err != nil {
t.Fatal(err)
}
write(filepath.Join(devPath, "state"), "in_sync\n")
}
write(filepath.Join(queueDir, "logical_block_size"), "512\n")
write(filepath.Join(tmp, "block", "md0", "size"), "2048\n")
@@ -81,15 +90,77 @@ func TestMdraidMockSysfsScanAndCollect(t *testing.T) {
}
}
func TestCountMdraidMemberStates(t *testing.T) {
tmp := t.TempDir()
write := func(path, content string) {
t.Helper()
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
mdDir := filepath.Join(tmp, "block", "md0", "md")
// No dev-* entries: zero faulty, zero populated.
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 0 {
t.Fatalf("no members: got (faulty=%d populated=%d), want (0,0)", faulty, populated)
}
// Two healthy members.
write(filepath.Join(mdDir, "dev-sda", "state"), "in_sync\n")
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 2 {
t.Fatalf("all in_sync: got (faulty=%d populated=%d), want (0,2)", faulty, populated)
}
// One faulty member.
write(filepath.Join(mdDir, "dev-sdb", "state"), "faulty\n")
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 1 || populated != 2 {
t.Fatalf("one faulty: got (faulty=%d populated=%d), want (1,2)", faulty, populated)
}
// QNAP-style: 28 degraded slots but no dev-* entries for them, 4 in_sync.
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
write(filepath.Join(mdDir, "dev-sdc", "state"), "in_sync\n")
write(filepath.Join(mdDir, "dev-sdd", "state"), "in_sync\n")
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 4 {
t.Fatalf("qnap sparse: got (faulty=%d populated=%d), want (0,4)", faulty, populated)
}
}
func TestMdraidSmartStatus(t *testing.T) {
if got := mdraidSmartStatus(mdraidHealth{arrayState: "inactive"}); got != "FAILED" {
t.Fatalf("mdraidSmartStatus(inactive) = %q, want FAILED", got)
}
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, syncAction: "recover"}); got != "WARNING" {
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1, syncAction: "recover"}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(degraded+recover) = %q, want WARNING", got)
}
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1}); got != "FAILED" {
t.Fatalf("mdraidSmartStatus(degraded) = %q, want FAILED", got)
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1}); got != "FAILED" {
t.Fatalf("mdraidSmartStatus(degraded+faulty) = %q, want FAILED", got)
}
// QNAP-style: raid_disks=32 but only 4 populated; degraded=28 but no faulty devices.
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 28, faultyDisks: 0, raidDisks: 32, populatedDisks: 4}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(qnap sparse) = %q, want WARNING", got)
}
// A member disappearing from the same sparse array is indistinguishable
// from another reserved slot, so it must not be reported as healthy.
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 29, faultyDisks: 0, raidDisks: 32, populatedDisks: 3}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(qnap sparse missing member) = %q, want WARNING", got)
}
// A genuinely missing member (removed dev-* entry, not just an unpopulated
// QNAP reserve slot) must still fail: raid_disks=4, only 3 populated, all
// of them in_sync, so faultyDisks==0 but degraded==1.
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 3}); got != "FAILED" {
t.Fatalf("mdraidSmartStatus(missing member) = %q, want FAILED", got)
}
// Degraded with no member-state info at all (e.g. sysfs read failed) must
// still fail rather than being silently treated as a sparse QNAP array.
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 0}); got != "FAILED" {
t.Fatalf("mdraidSmartStatus(degraded, no member info) = %q, want FAILED", got)
}
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", syncAction: "recover"}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(recover) = %q, want WARNING", got)
+2 -1
View File
@@ -602,8 +602,9 @@ func TestUpdateTemperaturesSkipsOnTimeout(t *testing.T) {
},
}
originalGetSensorTemps := getSensorTemps
t.Cleanup(func() {
getSensorTemps = sensors.TemperaturesWithContext
getSensorTemps = originalGetSensorTemps
})
getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
time.Sleep(50 * time.Millisecond)
+5 -2
View File
@@ -214,9 +214,12 @@ func (lhm *lhmProcess) getTemps(ctx context.Context) (temps []sensors.Temperatur
return temps, nil
}
// getSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
// getSensorTemps is a variable so tests can replace the platform sensor collector.
var getSensorTemps = getWindowsSensorTemps
// getWindowsSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
// NB: LibreHardwareMonitorLib requires admin privileges to access all available sensors.
func getSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
func getWindowsSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
defer func() {
if err != nil {
slog.Debug("Error reading sensors", "err", err)
-1
View File
@@ -265,6 +265,5 @@ func (a *Agent) StopServer() error {
slog.Info("Stopping SSH server")
_ = a.server.Close()
a.server = nil
a.connectionManager.eventChan <- SSHDisconnect
return nil
}
+22
View File
@@ -198,6 +198,28 @@ func TestStartServerDisableSSH(t *testing.T) {
assert.Contains(t, err.Error(), "SSH disabled")
}
func TestStopServerDoesNotBlockWhenEventQueueFull(t *testing.T) {
agent := createTestAgent(t)
agent.server = &ssh.Server{}
agent.connectionManager.eventChan = make(chan ConnectionEvent, 1)
agent.connectionManager.eventChan <- WebSocketConnect
done := make(chan error, 1)
go func() {
done <- agent.StopServer()
}()
select {
case err := <-done:
require.NoError(t, err)
case <-time.After(time.Second):
t.Fatal("StopServer blocked on the connection event queue")
}
assert.Nil(t, agent.server)
assert.Equal(t, WebSocketConnect, <-agent.connectionManager.eventChan)
}
/////////////////////////////////////////////////////////////////
//////////////////// ParseKeys Tests ////////////////////////////
/////////////////////////////////////////////////////////////////
+3
View File
@@ -931,6 +931,9 @@ func (sm *SmartManager) parseSmartForSata(output []byte, deviceType string) (boo
if parsed, ok := smart.ParseSmartRawValueString(attr.Raw.String); ok {
rawValue = parsed
}
if smartData.SmartStatus == "PASSED" && rawValue > 0 && (attr.ID == 5 || attr.ID == 197 || attr.ID == 198) {
smartData.SmartStatus = "WARNING"
}
smartAttr := &smart.SmartAttribute{
ID: attr.ID,
Name: attr.Name,
+48
View File
@@ -4,8 +4,10 @@ package agent
import (
"errors"
"fmt"
"os"
"path/filepath"
"strconv"
"testing"
"github.com/henrygd/beszel/internal/entities/smart"
@@ -88,6 +90,52 @@ func TestParseSmartForSata(t *testing.T) {
}
}
func TestParseSmartForSataWarnsForCriticalAttributes(t *testing.T) {
for _, attrID := range []int{5, 197, 198} {
t.Run("attribute "+strconv.Itoa(attrID), func(t *testing.T) {
jsonPayload := []byte(fmt.Sprintf(`{
"smartctl": {"exit_status": 0},
"device": {"name": "/dev/sda", "type": "sat"},
"model_name": "Example",
"serial_number": "WARNING%d",
"smart_status": {"passed": true},
"temperature": {"current": 30},
"ata_smart_attributes": {"table": [{"id": %d, "raw": {"value": 1, "string": "1"}}]}
}`, attrID, attrID))
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
require.True(t, hasData)
assert.Equal(t, "WARNING", sm.SmartDataMap[fmt.Sprintf("WARNING%d", attrID)].SmartStatus)
})
}
}
func TestParseSmartForSataPreservesFailedAndUnknownStatus(t *testing.T) {
for _, test := range []struct {
name string
temperature int
want string
}{
{name: "failed", temperature: 30, want: "FAILED"},
{name: "unknown", want: "UNKNOWN"},
} {
t.Run(test.name, func(t *testing.T) {
jsonPayload := []byte(fmt.Sprintf(`{
"device": {"name": "/dev/sda", "type": "sat"},
"serial_number": "PRESERVE%s",
"temperature": {"current": %d},
"ata_smart_attributes": {"table": [{"id": 197, "raw": {"value": 1, "string": "1"}}]}
}`, test.name, test.temperature))
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
require.True(t, hasData)
assert.Equal(t, test.want, sm.SmartDataMap["PRESERVE"+test.name].SmartStatus)
})
}
}
func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
jsonPayload := []byte(`{
"smartctl": {"exit_status": 0},
+74 -6
View File
@@ -4,6 +4,7 @@ import (
"bufio"
"errors"
"fmt"
"io"
"log/slog"
"os"
"runtime"
@@ -31,7 +32,11 @@ func (a *Agent) refreshSystemDetails() {
if a.dockerManager != nil {
a.systemDetails.Podman = a.dockerManager.IsPodman()
hostInfo, _ = a.dockerManager.GetHostInfo()
// Docker's host info describes the machine its daemon runs on. On macOS and
// Windows that is a Linux VM, so its CPU and memory totals are not this host's.
if runtime.GOOS != "darwin" && runtime.GOOS != "windows" {
hostInfo, _ = a.dockerManager.GetHostInfo()
}
}
a.systemDetails.Hostname, _ = os.Hostname()
@@ -78,6 +83,12 @@ func (a *Agent) refreshSystemDetails() {
if info, err := cpu.Info(); err == nil && len(info) > 0 {
a.systemDetails.CpuModel = info[0].ModelName
}
// gopsutil doesn't parse the "cpu model" field from /proc/cpuinfo, which
// is the only source of the CPU model name on MIPS. Fall back to reading
// it directly when ModelName is empty.
if a.systemDetails.CpuModel == "" {
a.systemDetails.CpuModel = getCpuModelFromCpuinfo()
}
// cores / threads
cores, _ := cpu.Counts(false)
threads := hostInfo.NCPU
@@ -164,9 +175,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
// load average
if avgstat, err := load.Avg(); err == nil {
systemStats.LoadAvg[0] = avgstat.Load1
systemStats.LoadAvg[1] = avgstat.Load5
systemStats.LoadAvg[2] = avgstat.Load15
systemStats.LoadAvg[0] = utils.TwoDecimals(avgstat.Load1)
systemStats.LoadAvg[1] = utils.TwoDecimals(avgstat.Load5)
systemStats.LoadAvg[2] = utils.TwoDecimals(avgstat.Load15)
slog.Debug("Load average", "5m", avgstat.Load5, "15m", avgstat.Load15)
} else {
slog.Error("Error getting load average", "err", err)
@@ -208,6 +219,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
// disk i/o (cache-aware per interval)
a.updateDiskIo(cacheTimeMs, &systemStats)
// zfs pool stats
a.zfsManager.Update(&systemStats)
// network stats (per cache interval)
a.updateNetworkStats(cacheTimeMs, &systemStats)
@@ -258,13 +272,66 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
a.systemInfo.MemPct = systemStats.MemPct
a.systemInfo.DiskPct = systemStats.DiskPct
a.systemInfo.Battery = systemStats.Battery
a.systemInfo.Uptime, _ = host.Uptime()
a.systemInfo.Uptime, _ = getUptime()
a.systemInfo.BandwidthBytes = systemStats.Bandwidth[0] + systemStats.Bandwidth[1]
a.systemInfo.Threads = a.systemDetails.Threads
return systemStats
}
// cpuModelFallbackKeys are the field names to look for in /proc/cpuinfo when
// gopsutil fails to return a ModelName. The "cpu model" key is used on MIPS
// (e.g. "MIPS 1004Kc V2.15"), while "system type" provides SoC information
// on various embedded architectures.
var cpuModelFallbackKeys = []string{"cpu model", "system type"}
// getCpuModelFromCpuinfo reads /proc/cpuinfo and returns a CPU model string.
// This is a fallback for architectures where gopsutil's cpu.Info() does not
// populate ModelName, most notably MIPS.
func getCpuModelFromCpuinfo() string {
file, err := os.Open("/proc/cpuinfo")
if err != nil {
return ""
}
defer file.Close()
return parseCpuModel(file)
}
// parseCpuModel scans r (expected to be /proc/cpuinfo content) and returns
// a combined CPU model string. It collects values from all matching keys
// and joins them with " / " when multiple are found.
func parseCpuModel(r io.Reader) string {
lines := readLines(r)
var parts []string
for _, key := range cpuModelFallbackKeys {
for _, line := range lines {
after, found := strings.CutPrefix(line, key)
if !found {
continue
}
after = strings.TrimSpace(after)
if len(after) < 2 || after[0] != ':' {
continue
}
if value := strings.TrimSpace(after[1:]); value != "" {
parts = append(parts, value)
break
}
}
}
return strings.Join(parts, " / ")
}
// readLines reads all lines from r into a slice.
func readLines(r io.Reader) []string {
scanner := bufio.NewScanner(r)
var lines []string
for scanner.Scan() {
lines = append(lines, scanner.Text())
}
return lines
}
// calculateHostMemoryUsage derives counters defensively because /proc/meminfo may
// change while gopsutil reads it. Invalid unsigned subtractions saturate at zero.
func calculateHostMemoryUsage(v *mem.VirtualMemoryStat, htop bool) (used, cacheBuff, swapUsed uint64) {
@@ -283,7 +350,8 @@ func calculateHostMemoryUsage(v *mem.VirtualMemoryStat, htop bool) (used, cacheB
if htop {
used = saturatingSub(v.Total, v.Free, cacheBuff)
}
return used, cacheBuff, saturatingSub(v.SwapTotal, v.SwapFree, v.SwapCached)
// Cached swap pages still occupy swap slots and are included in `free`'s used value.
return used, cacheBuff, saturatingSub(v.SwapTotal, v.SwapFree)
}
// saturatingSub subtracts each value, returning zero on underflow.
+82 -3
View File
@@ -1,6 +1,7 @@
package agent
import (
"strings"
"testing"
"github.com/henrygd/beszel/internal/common"
@@ -46,14 +47,14 @@ func TestCalculateHostMemoryUsage(t *testing.T) {
memory: mem.VirtualMemoryStat{Total: 100, Available: 40, Used: 60, Free: 20, Cached: 25, Buffers: 10, Shared: 5, SwapTotal: 20, SwapFree: 8, SwapCached: 2},
used: 60,
cacheBuff: 30,
swapUsed: 10,
swapUsed: 12,
},
{
name: "inconsistent counters saturate",
memory: mem.VirtualMemoryStat{Total: 100, Available: 110, Used: ^uint64(0) - 9, Free: 90, Cached: 5, Buffers: 10, Shared: 20, SwapTotal: 10, SwapFree: 9, SwapCached: 2},
used: 0,
cacheBuff: 0,
swapUsed: 0,
swapUsed: 1,
},
{
name: "htop subtraction saturates",
@@ -61,7 +62,7 @@ func TestCalculateHostMemoryUsage(t *testing.T) {
htop: true,
used: 0,
cacheBuff: 25,
swapUsed: 15,
swapUsed: 20,
},
{
name: "zero cache from shared cancellation does not fall back",
@@ -113,3 +114,81 @@ func TestUpdateSystemDetailsMarksDetailsDirty(t *testing.T) {
assert.False(t, agent.detailsDirty)
assert.Nil(t, original.Details)
}
func TestParseCpuModel(t *testing.T) {
tests := []struct {
name string
input string
expected string
}{
{
name: "MIPS with both cpu model and system type",
input: `system type : MediaTek MT7621 ver:1 eco:3
machine : ASUS RT-AX53U
processor : 0
cpu model : MIPS 1004Kc V2.15
BogoMIPS : 586.13
wait instruction : yes`,
expected: "MIPS 1004Kc V2.15 / MediaTek MT7621 ver:1 eco:3",
},
{
name: "MIPS with different SoC",
input: `system type : Atheros AR7161 rev 2
machine : NETGEAR WNDR3700
processor : 0
cpu model : MIPS 24Kc V7.4
BogoMIPS : 452.19`,
expected: "MIPS 24Kc V7.4 / Atheros AR7161 rev 2",
},
{
name: "only system type when cpu model missing",
input: `system type : Broadcom BCM47xx
processor : 0
BogoMIPS : 296.11`,
expected: "Broadcom BCM47xx",
},
{
name: "only cpu model when system type missing",
input: `processor : 0
cpu model : MIPS 34Kc V2.15
BogoMIPS : 300.00`,
expected: "MIPS 34Kc V2.15",
},
{
name: "x86 cpuinfo returns empty",
input: `processor : 0
vendor_id : GenuineIntel
cpu family : 6
model : 142
model name : Intel(R) Core(TM) i5-8250U CPU @ 1.60GHz
stepping : 10`,
expected: "",
},
{
name: "empty input",
input: "",
expected: "",
},
{
name: "cpu model with extra whitespace",
input: `processor : 0
cpu model : MIPS 34Kc V2.15
BogoMIPS : 300.00`,
expected: "MIPS 34Kc V2.15",
},
{
name: "cpu model without value",
input: `processor : 0
cpu model :
BogoMIPS : 300.00`,
expected: "",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
result := parseCpuModel(strings.NewReader(tt.input))
assert.Equal(t, tt.expected, result)
})
}
}
+9
View File
@@ -0,0 +1,9 @@
tank 12000000000000 11999000000000 /tank
tank/apps 1000000000000 11999000000000 /tank/apps
tank/backup 2000000000000 11999000000000 /tank/backup
tank/media 1000000000000 11999000000000 /tank/my media
rpool 900000000000 300000000000 -
rpool/ROOT 1000000000 300000000000 -
rpool/ROOT/pve-1 890000000000 300000000000 /
rpool/data 9000000000 300000000000 -
rpool/data/subvol-100-disk-0 400000000000 300000000000 /subvol-100-disk-0
+2
View File
@@ -0,0 +1,2 @@
tank 23999000000000 12000000000000 11999000000000 ONLINE
rpool 1200000000000 900000000000 300000000000 DEGRADED
+29
View File
@@ -0,0 +1,29 @@
pool: tank
state: ONLINE
scan: scrub repaired 0B in 00:05:12 with 0 errors on Sun Jun 1 02:00:12 2025
config:
NAME STATE READ WRITE CKSUM
tank ONLINE 0 0 0
mirror-0 ONLINE 0 0 0
sda ONLINE 0 0 0
sdb ONLINE 0 0 0
errors: No known data errors
pool: rpool
state: DEGRADED
status: One or more devices could not be used because the label is missing or
invalid. Sufficient replicas exist for the pool to continue functioning in a
degraded state.
scan: scrub in progress since Sun Jun 8 01:00:00 2025
10.00% done, 01:30:00 to go, 0.00/s
config:
NAME STATE READ WRITE CKSUM
rpool DEGRADED 0 0 0
mirror-0 DEGRADED 0 0 0
sda ONLINE 0 0 0
sdb FAULTED 1 2 3
errors: 1 data errors, use '-v' for a list
+44
View File
@@ -0,0 +1,44 @@
//go:build linux
package agent
import (
"math"
"os"
"strconv"
"strings"
"github.com/shirou/gopsutil/v4/host"
)
// uptimeFilePath is a variable so tests can point it at a fixture.
var uptimeFilePath = "/proc/uptime"
// getUptime returns the system uptime in seconds.
//
// This reads /proc/uptime instead of using host.Uptime(), which calls the
// sysinfo(2) syscall. Inside an LXC container lxcfs virtualizes /proc/uptime
// but cannot intercept a syscall, so sysinfo(2) reports the host's uptime
// rather than the container's.
//
// Falls back to host.Uptime() if /proc/uptime is missing or unparseable, so
// behavior is unchanged anywhere the file isn't available.
func getUptime() (uint64, error) {
data, err := os.ReadFile(uptimeFilePath)
if err != nil {
return host.Uptime()
}
fields := strings.Fields(string(data))
if len(fields) == 0 {
return host.Uptime()
}
seconds, err := strconv.ParseFloat(fields[0], 64)
if err != nil ||
math.IsNaN(seconds) ||
math.IsInf(seconds, 0) ||
seconds < 0 ||
seconds >= 1<<64 {
return host.Uptime()
}
return uint64(seconds), nil
}
+101
View File
@@ -0,0 +1,101 @@
//go:build linux
package agent
import (
"os"
"path/filepath"
"testing"
)
func TestGetUptimeFromProc(t *testing.T) {
tests := []struct {
name string
contents string
want uint64
}{
{"typical", "12345.67 98765.43\n", 12345},
{"zero", "0.00 0.00\n", 0},
{"no trailing newline", "42.99 7.00", 42},
{"single field", "600.5", 600},
{"large value", "266030.12 1000000.00\n", 266030},
}
prev := uptimeFilePath
t.Cleanup(func() { uptimeFilePath = prev })
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
path := filepath.Join(t.TempDir(), "uptime")
if err := os.WriteFile(path, []byte(tt.contents), 0o644); err != nil {
t.Fatal(err)
}
uptimeFilePath = path
got, err := getUptime()
if err != nil {
t.Fatalf("getUptime() returned error: %v", err)
}
if got != tt.want {
t.Errorf("getUptime() = %d, want %d", got, tt.want)
}
})
}
}
func writeUptime(contents string) func(t *testing.T) string {
return func(t *testing.T) string {
path := filepath.Join(t.TempDir(), "uptime")
if err := os.WriteFile(path, []byte(contents), 0o644); err != nil {
t.Fatal(err)
}
return path
}
}
// Malformed, missing, or out-of-range input must fall back to host.Uptime()
// rather than returning a bogus value, so the agent still reports something sane.
func TestGetUptimeFallsBack(t *testing.T) {
prev := uptimeFilePath
t.Cleanup(func() { uptimeFilePath = prev })
for _, tt := range []struct {
name string
prepare func(t *testing.T) string
}{
{"missing file", func(t *testing.T) string {
return filepath.Join(t.TempDir(), "does-not-exist")
}},
{"empty file", func(t *testing.T) string {
path := filepath.Join(t.TempDir(), "uptime")
if err := os.WriteFile(path, nil, 0o644); err != nil {
t.Fatal(err)
}
return path
}},
{"unparseable", func(t *testing.T) string {
path := filepath.Join(t.TempDir(), "uptime")
if err := os.WriteFile(path, []byte("not-a-number 1.0\n"), 0o644); err != nil {
t.Fatal(err)
}
return path
}},
{"NaN", writeUptime("NaN 1.0\n")},
{"positive infinity", writeUptime("+Inf 1.0\n")},
{"negative infinity", writeUptime("-Inf 1.0\n")},
{"negative", writeUptime("-42.5 1.0\n")},
{"exceeds uint64 range", writeUptime("1e20 1.0\n")},
} {
t.Run(tt.name, func(t *testing.T) {
uptimeFilePath = tt.prepare(t)
got, err := getUptime()
if err != nil {
t.Fatalf("getUptime() returned error: %v", err)
}
if got == 0 {
t.Error("getUptime() = 0, expected fallback to host.Uptime()")
}
})
}
}
+10
View File
@@ -0,0 +1,10 @@
//go:build !linux
package agent
import "github.com/shirou/gopsutil/v4/host"
// getUptime returns the system uptime in seconds.
func getUptime() (uint64, error) {
return host.Uptime()
}
+160
View File
@@ -0,0 +1,160 @@
// Package zfs provides functions to read ZFS statistics.
package zfs
import (
"bufio"
"bytes"
"context"
"errors"
"fmt"
"os"
"os/exec"
"strconv"
"strings"
"time"
)
var commandTimeout = 10 * time.Second
var commandOutput = func(name string, args ...string) ([]byte, error) {
ctx, cancel := context.WithTimeout(context.Background(), commandTimeout)
defer cancel()
cmd := exec.CommandContext(ctx, name, args...)
cmd.Env = append(os.Environ(), "LC_ALL=C", "LANG=C")
out, err := cmd.Output()
if ctx.Err() != nil {
return nil, fmt.Errorf("%s timed out after %s: %w", name, commandTimeout, ctx.Err())
}
return out, err
}
// ErrNoZfs is returned when the ZFS utilities or kernel interfaces are unavailable.
var ErrNoZfs = errors.New("zfs utilities unavailable")
// PoolStat is a snapshot of a ZFS pool's capacity and health.
type PoolStat struct {
Name string
Size uint64 // total capacity in bytes
Alloc uint64 // allocated bytes
Free uint64 // free bytes
Health string // ONLINE, DEGRADED, FAULTED, ...
}
// PoolKernelStat is the inexpensive pool telemetry exposed by the ZFS kernel.
// NRead and NWrite are cumulative byte counters since the pool was imported.
type PoolKernelStat struct {
Name string
Health string
NRead uint64
NWrite uint64
}
// PoolIoStats holds calculated per-second I/O rates for a pool.
type PoolIoStats struct {
NRead uint64
NWrite uint64
}
// Dataset is a single ZFS dataset with usage information.
type Dataset struct {
Name string
Used uint64
Avail uint64
Mountpoint string
}
// PoolStats returns capacity and health for all pools on the system using
// `zpool list`. Frequent health and I/O sampling uses PoolKernelStats instead.
func PoolStats() ([]PoolStat, error) {
out, err := commandOutput("zpool", "list", "-Hp", "-o", "name,size,alloc,free,health")
if err != nil {
var exitErr *exec.ExitError
if errors.As(err, &exitErr) && strings.Contains(string(exitErr.Stderr), "no pools available") {
return nil, nil
}
return nil, fmt.Errorf("zpool list: %w", err)
}
return parseZpoolListOutput(out)
}
// Datasets returns all datasets on the system with usage and mountpoint
// information using `zfs list` (recursive by default).
func Datasets() ([]Dataset, error) {
out, err := commandOutput("zfs", "list", "-Hp", "-o", "name,used,avail,mountpoint")
if err != nil {
return nil, fmt.Errorf("zfs list: %w", err)
}
return parseZfsListOutput(out)
}
// parseZpoolListOutput parses `zpool list -Hp -o name,size,alloc,free,health` output.
// Columns are tab-separated; numeric columns are raw bytes.
func parseZpoolListOutput(out []byte) ([]PoolStat, error) {
var pools []PoolStat
scanner := bufio.NewScanner(bytes.NewReader(out))
for scanner.Scan() {
line := strings.TrimSpace(scanner.Text())
if line == "" {
continue
}
if line == "no pools available" && len(pools) == 0 {
return nil, nil
}
fields := strings.Split(line, "\t")
if len(fields) < 5 {
return nil, fmt.Errorf("unexpected zpool list line: %q", line)
}
size, err := strconv.ParseUint(fields[1], 10, 64)
if err != nil {
return nil, fmt.Errorf("parsing size for pool %q: %w", fields[0], err)
}
alloc, err := strconv.ParseUint(fields[2], 10, 64)
if err != nil {
return nil, fmt.Errorf("parsing alloc for pool %q: %w", fields[0], err)
}
free, err := strconv.ParseUint(fields[3], 10, 64)
if err != nil {
return nil, fmt.Errorf("parsing free for pool %q: %w", fields[0], err)
}
pools = append(pools, PoolStat{
Name: fields[0],
Size: size,
Alloc: alloc,
Free: free,
Health: fields[4],
})
}
return pools, scanner.Err()
}
// parseZfsListOutput parses `zfs list -Hp -o name,used,avail,mountpoint` output.
// The mountpoint column may contain spaces, so it is split on tabs only.
func parseZfsListOutput(out []byte) ([]Dataset, error) {
var datasets []Dataset
scanner := bufio.NewScanner(bytes.NewReader(out))
for scanner.Scan() {
line := strings.TrimSpace(scanner.Text())
if line == "" {
continue
}
fields := strings.SplitN(line, "\t", 4)
if len(fields) < 4 {
return nil, fmt.Errorf("unexpected zfs list line: %q", line)
}
used, err := strconv.ParseUint(fields[1], 10, 64)
if err != nil {
return nil, fmt.Errorf("parsing used for dataset %q: %w", fields[0], err)
}
avail, err := strconv.ParseUint(fields[2], 10, 64)
if err != nil {
return nil, fmt.Errorf("parsing avail for dataset %q: %w", fields[0], err)
}
datasets = append(datasets, Dataset{
Name: fields[0],
Used: used,
Avail: avail,
Mountpoint: fields[3],
})
}
return datasets, scanner.Err()
}
+8
View File
@@ -3,9 +3,17 @@
package zfs
import (
"errors"
"golang.org/x/sys/unix"
)
func ARCSize() (uint64, error) {
return unix.SysctlUint64("kstat.zfs.misc.arcstats.size")
}
// FreeBSD does not expose Linux's per-pool procfs kstats. Capacity, health,
// and detail collection still work through the cached utilities.
func PoolKernelStats() ([]PoolKernelStat, error) {
return nil, errors.ErrUnsupported
}
+166 -1
View File
@@ -5,14 +5,18 @@ package zfs
import (
"bufio"
"errors"
"fmt"
"os"
"path/filepath"
"strconv"
"strings"
)
var procZfsPath = "/proc/spl/kstat/zfs"
func ARCSize() (uint64, error) {
file, err := os.Open("/proc/spl/kstat/zfs/arcstats")
file, err := os.Open(filepath.Join(procZfsPath, "arcstats"))
if err != nil {
return 0, err
}
@@ -29,6 +33,167 @@ func ARCSize() (uint64, error) {
return strconv.ParseUint(fields[2], 10, 64)
}
}
if err := scanner.Err(); err != nil {
return 0, err
}
return 0, fmt.Errorf("size field not found in arcstats")
}
// PoolKernelStats reads pool state and cumulative I/O counters directly from
// procfs. These kstats are the same interfaces used by node_exporter's Linux
// ZFS collector and avoid keeping a `zpool iostat` subprocess alive.
func PoolKernelStats() ([]PoolKernelStat, error) {
poolDirs := make(map[string]struct{})
for _, filename := range []string{"state", "io", "objset-*"} {
paths, err := filepath.Glob(filepath.Join(procZfsPath, "*", filename))
if err != nil {
return nil, err
}
for _, path := range paths {
poolDirs[filepath.Dir(path)] = struct{}{}
}
}
if len(poolDirs) == 0 {
return nil, ErrNoZfs
}
pools := make([]PoolKernelStat, 0, len(poolDirs))
for poolDir := range poolDirs {
nread, nwrite, err := readPoolCounters(poolDir)
if err != nil {
if errors.Is(err, os.ErrNotExist) {
continue // pool may have been exported after the glob
}
return nil, err
}
state, err := os.ReadFile(filepath.Join(poolDir, "state"))
if err != nil && !errors.Is(err, os.ErrNotExist) {
return nil, err
}
pools = append(pools, PoolKernelStat{
Name: filepath.Base(poolDir), Health: strings.ToUpper(strings.TrimSpace(string(state))),
NRead: nread, NWrite: nwrite,
})
}
if len(pools) == 0 {
return nil, ErrNoZfs
}
return pools, nil
}
// readPoolCounters supports both ZFS kernel interfaces. OpenZFS through 2.3
// exposes aggregate vdev counters in "io". When that file is unavailable, sum
// the logical I/O counters exposed for each dataset in the pool.
func readPoolCounters(poolDir string) (uint64, uint64, error) {
nread, nwrite, err := readPoolIO(filepath.Join(poolDir, "io"))
if err == nil || !errors.Is(err, os.ErrNotExist) {
return nread, nwrite, err
}
return readPoolObjsets(poolDir)
}
func readPoolIO(path string) (uint64, uint64, error) {
file, err := os.Open(path)
if err != nil {
return 0, 0, err
}
defer file.Close()
scanner := bufio.NewScanner(file)
for scanner.Scan() {
fields := strings.Fields(scanner.Text())
if len(fields) < 2 || fields[0] != "nread" {
continue
}
if !scanner.Scan() {
break
}
values := strings.Fields(scanner.Text())
if len(values) < 2 {
break
}
nread, err := strconv.ParseUint(values[0], 10, 64)
if err != nil {
return 0, 0, fmt.Errorf("parsing nread in %s: %w", path, err)
}
nwrite, err := strconv.ParseUint(values[1], 10, 64)
if err != nil {
return 0, 0, fmt.Errorf("parsing nwritten in %s: %w", path, err)
}
return nread, nwrite, nil
}
if err := scanner.Err(); err != nil {
return 0, 0, err
}
return 0, 0, fmt.Errorf("I/O counters not found in %s", path)
}
func readPoolObjsets(poolDir string) (uint64, uint64, error) {
paths, err := filepath.Glob(filepath.Join(poolDir, "objset-*"))
if err != nil {
return 0, 0, err
}
if len(paths) == 0 {
return 0, 0, fmt.Errorf("dataset I/O counters not found in %s", poolDir)
}
var totalRead, totalWrite uint64
objsetsRead := 0
for _, path := range paths {
nread, nwrite, err := readObjsetIO(path)
if errors.Is(err, os.ErrNotExist) {
continue // dataset may have been destroyed after the glob
}
if err != nil {
return 0, 0, err
}
totalRead += nread
totalWrite += nwrite
objsetsRead++
}
if objsetsRead == 0 {
return 0, 0, fmt.Errorf("dataset I/O counters not found in %s", poolDir)
}
return totalRead, totalWrite, nil
}
func readObjsetIO(path string) (uint64, uint64, error) {
file, err := os.Open(path)
if err != nil {
return 0, 0, err
}
defer file.Close()
var nread, nwrite uint64
var foundRead, foundWrite bool
scanner := bufio.NewScanner(file)
for scanner.Scan() {
fields := strings.Fields(scanner.Text())
if len(fields) < 3 {
continue
}
var target *uint64
switch fields[0] {
case "nread":
target = &nread
foundRead = true
case "nwritten":
target = &nwrite
foundWrite = true
default:
continue
}
value, err := strconv.ParseUint(fields[2], 10, 64)
if err != nil {
return 0, 0, fmt.Errorf("parsing %s in %s: %w", fields[0], path, err)
}
*target = value
}
if err := scanner.Err(); err != nil {
return 0, 0, err
}
if !foundRead || !foundWrite {
return 0, 0, fmt.Errorf("incomplete I/O counters in %s", path)
}
return nread, nwrite, nil
}
+90
View File
@@ -0,0 +1,90 @@
//go:build testing && linux
package zfs
import (
"os"
"path/filepath"
"testing"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestPoolKernelStats(t *testing.T) {
root := t.TempDir()
oldPath := procZfsPath
procZfsPath = root
t.Cleanup(func() { procZfsPath = oldPath })
poolDir := filepath.Join(root, "tank")
require.NoError(t, os.MkdirAll(poolDir, 0o755))
require.NoError(t, os.WriteFile(filepath.Join(poolDir, "io"), []byte(
"11 3 0x00 1 80 0 0\n"+
"nread nwritten reads writes wtime wlentime wupdate rtime rlentime rupdate wcnt rcnt\n"+
"1884160 6450688 22 978 0 0 0 0 0 0 0 0\n",
), 0o644))
require.NoError(t, os.WriteFile(filepath.Join(poolDir, "state"), []byte("DEGRADED\n"), 0o644))
stats, err := PoolKernelStats()
require.NoError(t, err)
require.Len(t, stats, 1)
assert.Equal(t, PoolKernelStat{
Name: "tank", Health: "DEGRADED", NRead: 1884160, NWrite: 6450688,
}, stats[0])
}
func TestPoolKernelStatsOpenZfs24(t *testing.T) {
root := t.TempDir()
oldPath := procZfsPath
procZfsPath = root
t.Cleanup(func() { procZfsPath = oldPath })
poolDir := filepath.Join(root, "tank")
require.NoError(t, os.MkdirAll(poolDir, 0o755))
require.NoError(t, os.WriteFile(filepath.Join(poolDir, "state"), []byte("ONLINE\n"), 0o644))
require.NoError(t, os.WriteFile(filepath.Join(poolDir, "objset-0x1"), []byte(
"34 1 0x01 28 7872 0 0\n"+
"name type data\n"+
"dataset_name 7 tank\n"+
"nwritten 4 2000\n"+
"nread 4 1000\n",
), 0o644))
require.NoError(t, os.WriteFile(filepath.Join(poolDir, "objset-0x2"), []byte(
"34 1 0x01 28 7872 0 0\n"+
"name type data\n"+
"dataset_name 7 tank/videos\n"+
"nwritten 4 400\n"+
"nread 4 300\n",
), 0o644))
stats, err := PoolKernelStats()
require.NoError(t, err)
require.Len(t, stats, 1)
assert.Equal(t, PoolKernelStat{
Name: "tank", Health: "ONLINE", NRead: 1300, NWrite: 2400,
}, stats[0])
}
func TestPoolKernelStatsNoZfs(t *testing.T) {
oldPath := procZfsPath
procZfsPath = t.TempDir()
t.Cleanup(func() { procZfsPath = oldPath })
_, err := PoolKernelStats()
assert.ErrorIs(t, err, ErrNoZfs)
}
func TestReadPoolIORejectsMalformedCounters(t *testing.T) {
path := filepath.Join(t.TempDir(), "io")
require.NoError(t, os.WriteFile(path, []byte("nread nwritten\nnope 10\n"), 0o644))
_, _, err := readPoolIO(path)
require.Error(t, err)
}
func TestReadObjsetIORequiresAllCounters(t *testing.T) {
path := filepath.Join(t.TempDir(), "objset-0x1")
require.NoError(t, os.WriteFile(path, []byte("nread 4 10\n"), 0o644))
_, _, err := readObjsetIO(path)
require.Error(t, err)
}
+150
View File
@@ -0,0 +1,150 @@
package zfs
import (
"bufio"
"bytes"
"fmt"
"regexp"
"strconv"
"strings"
)
// PoolStatus holds parsed `zpool status` information for one pool.
type PoolStatus struct {
Name string
State string // ONLINE, DEGRADED, FAULTED, ...
Scrub ScrubStatus
Vdevs []VdevStatus
}
// ScrubStatus holds the scrub (or resilver) status parsed from the scan line.
type ScrubStatus struct {
State string // NONE, SCANNING, FINISHED, CANCELED
Progress string // e.g. "10.00%" while scanning
Errors uint64
}
// VdevStatus is a single vdev row (mirror, raidz, or leaf disk).
type VdevStatus struct {
Name string
State string
ReadErrs uint64
WriteErrs uint64
ChecksumErrs uint64
}
var (
progressRe = regexp.MustCompile(`(\d+\.\d+)%\s+done`)
errorsRe = regexp.MustCompile(`with\s+(\d+)\s+errors`)
)
// PoolStatuses runs `zpool status` and parses per-pool state, scrub, and vdev
// information. The human-readable format has been stable across OpenZFS
// releases; rows are matched by their tabular shape rather than position.
func PoolStatuses() ([]PoolStatus, error) {
out, err := commandOutput("zpool", "status")
if err != nil {
return nil, fmt.Errorf("zpool status: %w", err)
}
return parseZpoolStatusOutput(out)
}
// parseZpoolStatusOutput parses the output of `zpool status`.
func parseZpoolStatusOutput(out []byte) ([]PoolStatus, error) {
var pools []PoolStatus
var current *PoolStatus
inConfig := false
scanContinuation := false // next non-blank line continues the scan line (progress)
scanner := bufio.NewScanner(bytes.NewReader(out))
for scanner.Scan() {
line := scanner.Text()
trimmed := strings.TrimSpace(line)
switch {
case strings.HasPrefix(trimmed, "pool:"):
pools = append(pools, PoolStatus{Name: strings.TrimSpace(strings.TrimPrefix(trimmed, "pool:"))})
current = &pools[len(pools)-1]
inConfig = false
scanContinuation = false
case current == nil:
continue
case strings.HasPrefix(trimmed, "state:"):
current.State = strings.TrimSpace(strings.TrimPrefix(trimmed, "state:"))
case strings.HasPrefix(trimmed, "scan:"):
current.Scrub = parseScanLine(trimmed)
// zpool status prints the progress percentage on the line after scan.
scanContinuation = true
case trimmed == "config:":
inConfig = true
case scanContinuation:
// The line after scan: may be an indented progress continuation.
if m := progressRe.FindStringSubmatch(trimmed); m != nil {
current.Scrub.Progress = m[1] + "%"
}
scanContinuation = false
case inConfig && (line == "" || strings.HasPrefix(line, " ") || strings.HasPrefix(line, "\t")):
// Table rows are indented; blank lines separate sections. The
// column header and the pool's own row are skipped.
if trimmed != "" && !strings.HasPrefix(trimmed, "NAME") {
if vdev, ok := parseVdevLine(trimmed, current.Name); ok {
current.Vdevs = append(current.Vdevs, vdev)
}
}
case inConfig:
// unindented line (errors:, status:, next pool:) ends the table
inConfig = false
}
}
return pools, scanner.Err()
}
// parseScanLine maps a `scan:` line to a ScrubStatus.
func parseScanLine(line string) ScrubStatus {
var scrub ScrubStatus
switch {
case strings.Contains(line, "in progress"):
scrub.State = "SCANNING"
case strings.Contains(line, "canceled"):
scrub.State = "CANCELED"
case strings.Contains(line, "repaired"), strings.Contains(line, "resilvered"):
scrub.State = "FINISHED"
default:
scrub.State = "NONE"
}
if m := progressRe.FindStringSubmatch(line); m != nil {
scrub.Progress = m[1] + "%"
}
if m := errorsRe.FindStringSubmatch(line); m != nil {
if n, err := strconv.ParseUint(m[1], 10, 64); err == nil {
scrub.Errors = n
}
}
return scrub
}
// parseVdevLine parses one row of the config table. Rows have the shape
// "NAME STATE READ WRITE CKSUM [extra...]". The first data row is the pool
// itself and is skipped since it duplicates pool-level info.
func parseVdevLine(line, poolName string) (VdevStatus, bool) {
fields := strings.Fields(line)
if len(fields) < 5 {
return VdevStatus{}, false
}
if fields[0] == poolName {
return VdevStatus{}, false
}
read, err1 := strconv.ParseUint(fields[2], 10, 64)
write, err2 := strconv.ParseUint(fields[3], 10, 64)
cksum, err3 := strconv.ParseUint(fields[4], 10, 64)
if err1 != nil || err2 != nil || err3 != nil {
return VdevStatus{}, false
}
return VdevStatus{
Name: fields[0],
State: fields[1],
ReadErrs: read,
WriteErrs: write,
ChecksumErrs: cksum,
}, true
}
+143
View File
@@ -0,0 +1,143 @@
//go:build testing
package zfs
import (
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"time"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func fixturePath(name string) string {
return filepath.Join("..", "test-data", "zfs", name)
}
func TestParseZpoolListOutput(t *testing.T) {
data, err := os.ReadFile(fixturePath("zpool_list.txt"))
require.NoError(t, err)
pools, err := parseZpoolListOutput(data)
require.NoError(t, err)
require.Len(t, pools, 2)
assert.Equal(t, PoolStat{Name: "tank", Size: 23999000000000, Alloc: 12000000000000, Free: 11999000000000, Health: "ONLINE"}, pools[0])
assert.Equal(t, PoolStat{Name: "rpool", Size: 1200000000000, Alloc: 900000000000, Free: 300000000000, Health: "DEGRADED"}, pools[1])
}
func TestParseZpoolListOutputIgnoresEmptyLines(t *testing.T) {
pools, err := parseZpoolListOutput([]byte("tank\t100\t50\t50\tONLINE\n\n"))
require.NoError(t, err)
require.Len(t, pools, 1)
assert.Equal(t, "tank", pools[0].Name)
}
func TestParseZpoolListOutputNoPools(t *testing.T) {
pools, err := parseZpoolListOutput([]byte("no pools available\n"))
require.NoError(t, err)
assert.Empty(t, pools)
}
func TestParseZpoolListOutputRejectsMalformedLine(t *testing.T) {
_, err := parseZpoolListOutput([]byte("tank\t100\t50\n"))
require.Error(t, err)
_, err = parseZpoolListOutput([]byte("tank\tnotanumber\t50\t50\tONLINE\n"))
require.Error(t, err)
}
func TestParseZfsListOutput(t *testing.T) {
data, err := os.ReadFile(fixturePath("zfs_list.txt"))
require.NoError(t, err)
datasets, err := parseZfsListOutput(data)
require.NoError(t, err)
require.Len(t, datasets, 9)
// Mountpoint with a space must be kept intact (tab-split only).
assert.Equal(t, "/tank/my media", datasets[3].Mountpoint)
// Unmounted datasets/zvols report "-".
assert.Equal(t, "-", datasets[4].Mountpoint)
assert.Equal(t, uint64(12000000000000), datasets[0].Used)
assert.Equal(t, uint64(11999000000000), datasets[0].Avail)
}
func TestParseZpoolStatusOutput(t *testing.T) {
data, err := os.ReadFile(fixturePath("zpool_status.txt"))
require.NoError(t, err)
pools, err := parseZpoolStatusOutput(data)
require.NoError(t, err)
require.Len(t, pools, 2)
tank := pools[0]
assert.Equal(t, "tank", tank.Name)
assert.Equal(t, "ONLINE", tank.State)
assert.Equal(t, "FINISHED", tank.Scrub.State)
assert.Equal(t, "", tank.Scrub.Progress)
assert.Equal(t, uint64(0), tank.Scrub.Errors)
// Pool row itself is skipped; mirror + 2 disks remain.
require.Len(t, tank.Vdevs, 3)
assert.Equal(t, "mirror-0", tank.Vdevs[0].Name)
assert.Equal(t, "sda", tank.Vdevs[1].Name)
assert.Equal(t, "sdb", tank.Vdevs[2].Name)
rpool := pools[1]
assert.Equal(t, "rpool", rpool.Name)
assert.Equal(t, "DEGRADED", rpool.State)
assert.Equal(t, "SCANNING", rpool.Scrub.State)
assert.Equal(t, "10.00%", rpool.Scrub.Progress)
require.Len(t, rpool.Vdevs, 3)
assert.Equal(t, "FAULTED", rpool.Vdevs[2].State)
assert.Equal(t, uint64(1), rpool.Vdevs[2].ReadErrs)
assert.Equal(t, uint64(2), rpool.Vdevs[2].WriteErrs)
assert.Equal(t, uint64(3), rpool.Vdevs[2].ChecksumErrs)
}
func TestParseScanLine(t *testing.T) {
assert.Equal(t, "FINISHED", parseScanLine("scan: scrub repaired 0B in 00:05:12 with 0 errors on Sun Jun 1 02:00:12 2025").State)
assert.Equal(t, uint64(3), parseScanLine("scan: scrub repaired 10G in 01:00:00 with 3 errors on Sun Jun 1 02:00:12 2025").Errors)
assert.Equal(t, "SCANNING", parseScanLine("scan: scrub in progress since Sun Jun 8 01:00:00 2025").State)
assert.Equal(t, "CANCELED", parseScanLine("scan: scrub canceled on Sun Jun 1 02:00:12 2025").State)
assert.Equal(t, "FINISHED", parseScanLine("scan: resilvered 1.23G in 00:01:00 with 0 errors on Sun Jun 1 02:00:12 2025").State)
assert.Equal(t, "NONE", parseScanLine("scan: none requested").State)
}
func TestCommandOutputForcesLocaleAndTimesOut(t *testing.T) {
t.Setenv("BESZEL_ZFS_COMMAND_HELPER", "1")
out, err := commandOutput(os.Args[0], "-test.run=TestZfsCommandHelperProcess", "--", "locale")
require.NoError(t, err)
assert.Equal(t, "C/C", string(out))
oldTimeout := commandTimeout
commandTimeout = 20 * time.Millisecond
t.Cleanup(func() { commandTimeout = oldTimeout })
_, err = commandOutput(os.Args[0], "-test.run=TestZfsCommandHelperProcess", "--", "sleep")
require.Error(t, err)
assert.Contains(t, err.Error(), "timed out")
}
func TestZfsCommandHelperProcess(t *testing.T) {
if os.Getenv("BESZEL_ZFS_COMMAND_HELPER") != "1" {
return
}
mode := ""
for i, arg := range os.Args {
if arg == "--" && i+1 < len(os.Args) {
mode = os.Args[i+1]
break
}
}
switch strings.TrimSpace(mode) {
case "locale":
_, _ = fmt.Printf("%s/%s", os.Getenv("LC_ALL"), os.Getenv("LANG"))
case "sleep":
time.Sleep(time.Second)
}
os.Exit(0)
}
+4
View File
@@ -7,3 +7,7 @@ import "errors"
func ARCSize() (uint64, error) {
return 0, errors.ErrUnsupported
}
func PoolKernelStats() ([]PoolKernelStat, error) {
return nil, errors.ErrUnsupported
}
+320
View File
@@ -0,0 +1,320 @@
package agent
import (
"log/slog"
"strings"
"sync"
"time"
"github.com/henrygd/beszel/agent/zfs"
"github.com/henrygd/beszel/internal/entities/system"
zfsentity "github.com/henrygd/beszel/internal/entities/zfs"
)
// zfsDatasetUsage holds usage values for a ZFS dataset mountpoint.
type zfsDatasetUsage struct {
used uint64
avail uint64
}
// datasetUsageRefreshInterval controls how often `zfs list` is re-run for the
// mountpoint usage map. Dataset inventory changes rarely.
const datasetUsageRefreshInterval = 5 * time.Minute
// poolStatsRefreshInterval controls how often `zpool list` is re-run for pool
// capacity. Health and I/O are read from procfs on Linux, so the utility only
// needs to refresh slow-moving space accounting.
const poolStatsRefreshInterval = time.Minute
type poolKernelSample struct {
nread uint64
nwrite uint64
at time.Time
}
// ZfsManager collects ZFS pool and dataset statistics. Collection functions
// are fields so unit tests can substitute them (same pattern as
// diskDiscovery.usageFn). It is safe for concurrent use by a single goroutine
// only; callers must hold the agent lock like updateDiskUsage does.
type ZfsManager struct {
poolStatsFn func() ([]zfs.PoolStat, error) // capacity/health source
datasetsFn func() ([]zfs.Dataset, error) // dataset inventory source
kernelStatsFn func() ([]zfs.PoolKernelStat, error) // procfs pool state/I/O source
poolStatusesFn func() ([]zfs.PoolStatus, error) // scrub/vdev detail source
poolData []zfs.PoolStat // cached pool inventory (TTL below)
lastPoolStats time.Time
kernelSamples map[string]poolKernelSample
datasetUsage map[string]zfsDatasetUsage // mountpoint -> usage
lastUsageRefresh time.Time
// Detail data (pools, vdevs, scrub, datasets) is cached and refreshed on
// an interval. Accessed from handler goroutines, so it is mutex-protected.
detailMu sync.Mutex
detail *zfsentity.ZfsData
lastDetailRefresh time.Time
detailInterval time.Duration
}
// newZfsManager creates a ZfsManager wired to the system's ZFS utilities.
func newZfsManager() *ZfsManager {
return &ZfsManager{
poolStatsFn: zfs.PoolStats,
datasetsFn: zfs.Datasets,
kernelStatsFn: zfs.PoolKernelStats,
poolStatusesFn: zfs.PoolStatuses,
detailInterval: time.Hour,
}
}
// Update refreshes systemStats.ZfsPools with the latest pool data. I/O
// throughput and health come from inexpensive kernel kstats on Linux. Pool
// capacity and dataset usage come from separately cached utility calls. It is
// a no-op when ZFS is absent.
func (zm *ZfsManager) Update(systemStats *system.Stats) {
pools := zm.poolStats()
if len(pools) == 0 {
return
}
kernelStats, ioRates := zm.kernelStats()
if systemStats.ZfsPools == nil {
systemStats.ZfsPools = make(map[string]*system.ZfsPool, len(pools))
}
for i := range pools {
pool := &pools[i]
// Full precision, matching the dataset values below; the frontend
// formats any magnitude.
stats := &system.ZfsPool{
Total: float64(pool.Size) / (1024 * 1024 * 1024),
Used: float64(pool.Alloc) / (1024 * 1024 * 1024),
Health: pool.Health,
}
if kernel, exists := kernelStats[pool.Name]; exists && kernel.Health != "" {
stats.Health = kernel.Health
}
if io, exists := ioRates[pool.Name]; exists {
stats.ReadBytes = io.NRead
stats.WriteBytes = io.NWrite
}
slog.Debug("ZFS pool sample", "pool", pool.Name, "health", stats.Health, "used_gb", stats.Used, "read_bps", stats.ReadBytes, "write_bps", stats.WriteBytes)
systemStats.ZfsPools[pool.Name] = stats
}
}
// poolStats returns the cached pool inventory, re-running `zpool list` at most
// every poolStatsRefreshInterval. On failure the previous inventory is
// retained and the refresh is retried on the next cadence.
func (zm *ZfsManager) poolStats() []zfs.PoolStat {
if zm.lastPoolStats.IsZero() || time.Since(zm.lastPoolStats) >= poolStatsRefreshInterval {
pools, err := zm.poolStatsFn()
if err != nil {
slog.Debug("ZFS pool stats unavailable", "err", err)
} else {
zm.poolData = pools
}
zm.lastPoolStats = time.Now()
}
return zm.poolData
}
// kernelStats reads cumulative pool counters and converts them to per-second
// rates. Counter decreases indicate a pool export/import and reset the
// baseline instead of producing an underflow spike.
func (zm *ZfsManager) kernelStats() (map[string]zfs.PoolKernelStat, map[string]zfs.PoolIoStats) {
if zm.kernelStatsFn == nil {
return nil, nil
}
stats, err := zm.kernelStatsFn()
if err != nil {
slog.Debug("ZFS kernel stats unavailable", "err", err)
return nil, nil
}
now := time.Now()
byName := make(map[string]zfs.PoolKernelStat, len(stats))
rates := make(map[string]zfs.PoolIoStats, len(stats))
nextSamples := make(map[string]poolKernelSample, len(stats))
for _, stat := range stats {
byName[stat.Name] = stat
if previous, ok := zm.kernelSamples[stat.Name]; ok && now.After(previous.at) &&
stat.NRead >= previous.nread && stat.NWrite >= previous.nwrite {
seconds := now.Sub(previous.at).Seconds()
rates[stat.Name] = zfs.PoolIoStats{
NRead: uint64(float64(stat.NRead-previous.nread) / seconds),
NWrite: uint64(float64(stat.NWrite-previous.nwrite) / seconds),
}
}
nextSamples[stat.Name] = poolKernelSample{nread: stat.NRead, nwrite: stat.NWrite, at: now}
}
zm.kernelSamples = nextSamples
return byName, rates
}
// refreshDatasetUsage re-runs `zfs list` when the refresh window has elapsed
// and rebuilds the mountpoint-keyed usage map.
func (zm *ZfsManager) refreshDatasetUsage() {
if !zm.lastUsageRefresh.IsZero() && time.Since(zm.lastUsageRefresh) < datasetUsageRefreshInterval {
return
}
datasets, err := zm.datasetsFn()
if err != nil {
slog.Debug("ZFS dataset usage unavailable", "err", err)
} else {
usage := make(map[string]zfsDatasetUsage, len(datasets))
for _, ds := range datasets {
if ds.Mountpoint != "" && ds.Mountpoint != "-" {
usage[ds.Mountpoint] = zfsDatasetUsage{used: ds.Used, avail: ds.Avail}
}
}
zm.datasetUsage = usage
}
zm.lastUsageRefresh = time.Now()
}
// DatasetUsage returns ZFS dataset usage keyed by mountpoint, refreshed at
// most every datasetUsageRefreshInterval. On failure the previous map is
// retained and a debug log is emitted.
func (zm *ZfsManager) DatasetUsage() map[string]zfsDatasetUsage {
zm.refreshDatasetUsage()
return zm.datasetUsage
}
// GetDetail returns ZFS detail data (pool health, scrub, vdevs, datasets).
// Scheduled requests use the cached snapshot until stale; manual requests can
// force collection. On failure the previous snapshot is retained.
func (zm *ZfsManager) GetDetail(force bool) *zfsentity.ZfsData {
zm.detailMu.Lock()
defer zm.detailMu.Unlock()
if force || zm.detail == nil || time.Since(zm.lastDetailRefresh) >= zm.detailInterval {
if data, err := zm.collectDetail(zm.detail); err != nil {
slog.Debug("ZFS detail collection failed", "err", err)
if zm.detail == nil {
return &zfsentity.ZfsData{}
}
return &zfsentity.ZfsData{Pools: zm.detail.Pools}
} else {
zm.detail = data
zm.lastDetailRefresh = time.Now()
}
}
if zm.detail == nil {
return &zfsentity.ZfsData{}
}
return zm.detail
}
// collectDetail builds a ZfsData payload from the current system state.
func (zm *ZfsManager) collectDetail(previous *zfsentity.ZfsData) (*zfsentity.ZfsData, error) {
pools, err := zm.poolStatsFn()
if err != nil {
return nil, err
}
if len(pools) == 0 {
return &zfsentity.ZfsData{Pools: []*zfsentity.PoolDetail{}, Complete: true}, nil
}
statuses, statusErr := zm.poolStatusesFn()
if statusErr != nil {
slog.Debug("ZFS pool status unavailable", "err", statusErr)
}
datasets, datasetsErr := zm.datasetsFn()
if datasetsErr != nil {
slog.Debug("ZFS datasets unavailable", "err", datasetsErr)
}
statusByPool := make(map[string]zfs.PoolStatus, len(statuses))
for _, st := range statuses {
statusByPool[st.Name] = st
}
previousByPool := make(map[string]*zfsentity.PoolDetail)
if previous != nil {
for _, pool := range previous.Pools {
if pool != nil {
previousByPool[pool.Name] = pool
}
}
}
data := &zfsentity.ZfsData{Pools: make([]*zfsentity.PoolDetail, 0, len(pools)), Complete: true}
for i := range pools {
p := &pools[i]
detail := &zfsentity.PoolDetail{
Name: p.Name,
Health: p.Health,
Size: p.Size,
Alloc: p.Alloc,
Free: p.Free,
}
if st, ok := statusByPool[p.Name]; statusErr == nil && ok {
if st.Scrub.State != "" && st.Scrub.State != "NONE" {
detail.Scrub = &zfsentity.Scrub{
State: st.Scrub.State,
Progress: st.Scrub.Progress,
Errors: st.Scrub.Errors,
}
}
for _, v := range st.Vdevs {
detail.Vdevs = append(detail.Vdevs, &zfsentity.Vdev{
Name: v.Name,
State: v.State,
ReadErrs: v.ReadErrs,
WriteErrs: v.WriteErrs,
ChecksumErrs: v.ChecksumErrs,
})
}
} else {
if cached := previousByPool[p.Name]; cached != nil {
detail.Scrub = cached.Scrub
detail.Vdevs = cached.Vdevs
}
}
if datasetsErr == nil {
foundDataset := false
for _, ds := range datasets {
if poolOfDataset(ds.Name) == p.Name {
foundDataset = true
detail.Datasets = append(detail.Datasets, &zfsentity.Dataset{
Name: ds.Name,
Used: ds.Used,
Avail: ds.Avail,
Mountpoint: ds.Mountpoint,
})
}
}
if !foundDataset {
if cached := previousByPool[p.Name]; cached != nil {
detail.Datasets = cached.Datasets
}
}
} else if cached := previousByPool[p.Name]; cached != nil {
detail.Datasets = cached.Datasets
}
data.Pools = append(data.Pools, detail)
}
return data, nil
}
// poolOfDataset returns the pool name for a dataset name (everything before
// the first '/'). Datasets without a separator belong to a pool of the same
// name.
func poolOfDataset(name string) string {
if idx := strings.IndexByte(name, '/'); idx >= 0 {
return name[:idx]
}
return name
}
// ZfsMountpoints returns the set of mountpoints backed by ZFS datasets.
func (zm *ZfsManager) ZfsMountpoints() map[string]bool {
usage := zm.DatasetUsage()
mountpoints := make(map[string]bool, len(usage))
for mountpoint := range usage {
mountpoints[mountpoint] = true
}
return mountpoints
}
+245
View File
@@ -0,0 +1,245 @@
//go:build testing
package agent
import (
"testing"
"time"
"github.com/henrygd/beszel/agent/zfs"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestUpdatePopulatesZfsPools(t *testing.T) {
zm := &ZfsManager{}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
return []zfs.PoolStat{{Name: "tank", Size: 23999000000000, Alloc: 12000000000000, Free: 11999000000000, Health: "DEGRADED"}}, nil
}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{
{Name: "tank/apps", Used: 5000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
{Name: "tank/backup", Used: 6000000000000, Avail: 11999000000000, Mountpoint: "/tank/backup"},
// Small zvol (Proxmox VM EFI disk): must not round to zero.
{Name: "rpool/vm-100-disk-2", Used: 4194304, Avail: 0, Mountpoint: "-"},
}, nil
}
var kernelCalls int
zm.kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
kernelCalls++
return []zfs.PoolKernelStat{{
Name: "tank", Health: "ONLINE",
NRead: uint64(kernelCalls-1) * 1250, NWrite: uint64(kernelCalls-1) * 5120,
}}, nil
}
var stats system.Stats
// The first kernel sample establishes the cumulative-counter baseline.
zm.Update(&stats)
zm.kernelSamples["tank"] = poolKernelSample{at: time.Now().Add(-time.Second)}
zm.Update(&stats)
require.NotNil(t, stats.ZfsPools)
require.Contains(t, stats.ZfsPools, "tank")
assert.InDelta(t, 22350.8105, stats.ZfsPools["tank"].Total, 0.0001) // Size in GiB
assert.InDelta(t, 11175.8709, stats.ZfsPools["tank"].Used, 0.0001) // Alloc in GiB
assert.Equal(t, "ONLINE", stats.ZfsPools["tank"].Health)
assert.InDelta(t, 1250, stats.ZfsPools["tank"].ReadBytes, 5)
assert.InDelta(t, 5120, stats.ZfsPools["tank"].WriteBytes, 5)
}
// TestUpdateKernelStatsMissing verifies pools without a kernel sample report zero
// I/O instead of erroring.
func TestUpdateKernelStatsMissing(t *testing.T) {
zm := &ZfsManager{}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
return []zfs.PoolStat{{Name: "tank", Size: 1, Alloc: 1, Health: "ONLINE"}}, nil
}
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
zm.kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
return nil, zfs.ErrNoZfs
}
var stats system.Stats
zm.Update(&stats)
require.NotNil(t, stats.ZfsPools)
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
}
func TestUpdateKernelCounterReset(t *testing.T) {
zm := &ZfsManager{}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
return []zfs.PoolStat{{Name: "tank", Health: "ONLINE"}}, nil
}
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
zm.kernelSamples = map[string]poolKernelSample{
"tank": {nread: 100, nwrite: 200, at: time.Now().Add(-time.Second)},
}
zm.kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
return []zfs.PoolKernelStat{{Name: "tank", Health: "ONLINE", NRead: 10, NWrite: 20}}, nil
}
var stats system.Stats
zm.Update(&stats)
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
}
func TestUpdateNoZfs(t *testing.T) {
zm := &ZfsManager{}
calls := 0
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
calls++
return nil, zfs.ErrNoZfs
}
var stats system.Stats
zm.Update(&stats)
zm.Update(&stats)
assert.Nil(t, stats.ZfsPools)
assert.Equal(t, 1, calls, "failed pool discovery should be cached until the next refresh interval")
}
func TestUpdateEmptyPools(t *testing.T) {
zm := &ZfsManager{}
calls := 0
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
calls++
return nil, nil
}
var stats system.Stats
zm.Update(&stats)
zm.Update(&stats)
assert.Nil(t, stats.ZfsPools)
assert.Equal(t, 1, calls, "an empty pool inventory should be cached until the next refresh interval")
}
func TestDatasetUsage(t *testing.T) {
zm := &ZfsManager{}
calls := 0
zm.datasetsFn = func() ([]zfs.Dataset, error) {
calls++
return []zfs.Dataset{
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
{Name: "tank/apps", Used: 1000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
{Name: "rpool", Used: 900000000000, Avail: 300000000000, Mountpoint: "-"}, // zvol/unmounted: excluded
}, nil
}
usage := zm.DatasetUsage()
require.Len(t, usage, 2)
assert.Equal(t, zfsDatasetUsage{used: 12000000000000, avail: 11999000000000}, usage["/tank"])
assert.Equal(t, zfsDatasetUsage{used: 1000000000000, avail: 11999000000000}, usage["/tank/apps"])
assert.Equal(t, 1, calls)
// Second call within the refresh window must not re-run the collector.
zm.DatasetUsage()
assert.Equal(t, 1, calls)
}
func TestDatasetUsageRefreshOnErrorKeepsPrevious(t *testing.T) {
zm := &ZfsManager{}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{{Name: "tank", Used: 1, Avail: 1, Mountpoint: "/tank"}}, nil
}
assert.Len(t, zm.DatasetUsage(), 1)
// Force refresh window expiry, then a failing collector.
zm.lastUsageRefresh = time.Now().Add(-10 * time.Minute)
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return nil, zfs.ErrNoZfs
}
usage := zm.DatasetUsage()
assert.Len(t, usage, 1, "previous usage should be retained on error")
}
func TestGetDetailForceRefresh(t *testing.T) {
zm := &ZfsManager{detailInterval: time.Hour}
poolCalls := 0
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
poolCalls++
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
}
zm.poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
first := zm.GetDetail(false)
assert.True(t, first.Complete)
require.Len(t, first.Pools, 1)
assert.Equal(t, uint64(1), first.Pools[0].Alloc)
cached := zm.GetDetail(false)
require.Len(t, cached.Pools, 1)
assert.Equal(t, uint64(1), cached.Pools[0].Alloc)
assert.Equal(t, 1, poolCalls)
refreshed := zm.GetDetail(true)
assert.True(t, refreshed.Complete)
require.Len(t, refreshed.Pools, 1)
assert.Equal(t, uint64(2), refreshed.Pools[0].Alloc)
assert.Equal(t, 2, poolCalls)
}
func TestGetDetailSuccessfulEmptyInventoryClearsCache(t *testing.T) {
zm := &ZfsManager{detailInterval: time.Hour}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
return []zfs.PoolStat{{Name: "tank"}}, nil
}
zm.poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
require.Len(t, zm.GetDetail(false).Pools, 1)
zm.poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, nil }
empty := zm.GetDetail(true)
assert.True(t, empty.Complete)
assert.Empty(t, empty.Pools)
}
func TestGetDetailFailureReturnsIncompleteCachedInventory(t *testing.T) {
zm := &ZfsManager{detailInterval: time.Hour}
zm.poolStatsFn = func() ([]zfs.PoolStat, error) {
return []zfs.PoolStat{{Name: "tank"}}, nil
}
zm.poolStatusesFn = func() ([]zfs.PoolStatus, error) {
return []zfs.PoolStatus{{Name: "tank", Vdevs: []zfs.VdevStatus{{Name: "mirror-0"}}}}, nil
}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{{Name: "tank/data"}}, nil
}
first := zm.GetDetail(false)
require.True(t, first.Complete)
require.Len(t, first.Pools[0].Vdevs, 1)
require.Len(t, first.Pools[0].Datasets, 1)
zm.poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, zfs.ErrNoZfs }
zm.datasetsFn = func() ([]zfs.Dataset, error) { return nil, zfs.ErrNoZfs }
partial := zm.GetDetail(true)
require.True(t, partial.Complete)
require.Len(t, partial.Pools[0].Vdevs, 1)
require.Len(t, partial.Pools[0].Datasets, 1)
zm.poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, zfs.ErrNoZfs }
lastSuccessfulRefresh := zm.lastDetailRefresh
failed := zm.GetDetail(true)
assert.False(t, failed.Complete)
require.Len(t, failed.Pools, 1)
assert.Equal(t, "tank", failed.Pools[0].Name)
assert.Equal(t, lastSuccessfulRefresh, zm.lastDetailRefresh)
}
func TestZfsMountpoints(t *testing.T) {
zm := &ZfsManager{}
zm.datasetsFn = func() ([]zfs.Dataset, error) {
return []zfs.Dataset{
{Name: "tank", Mountpoint: "/tank"},
{Name: "rpool/ROOT/pve-1", Mountpoint: "/"},
}, nil
}
mountpoints := zm.ZfsMountpoints()
assert.Len(t, mountpoints, 2)
assert.True(t, mountpoints["/tank"])
assert.True(t, mountpoints["/"])
}
+4 -1
View File
@@ -6,7 +6,7 @@ import "github.com/blang/semver"
const (
// Version is the current version of the application.
Version = "0.18.8"
Version = "0.19.0"
// AppName is the name of the application.
AppName = "beszel"
)
@@ -16,3 +16,6 @@ var MinVersionCbor = semver.MustParse("0.12.0")
// MinVersionAgentResponse is the minimum supported version for AgentResponse compatibility.
var MinVersionAgentResponse = semver.MustParse("0.13.0")
// MinVersionZfsData is the minimum agent version that supports ZFS detail requests.
var MinVersionZfsData = semver.MustParse("0.18.9")
+14 -13
View File
@@ -1,25 +1,25 @@
module github.com/henrygd/beszel
go 1.26.6
go 1.27.1
require (
github.com/blang/semver v3.5.1+incompatible
github.com/coreos/go-systemd/v22 v22.7.0
github.com/ebitengine/purego v0.10.2
github.com/fxamacker/cbor/v2 v2.9.2
github.com/ebitengine/purego v0.11.0
github.com/fxamacker/cbor/v2 v2.9.3
github.com/gliderlabs/ssh v0.3.8
github.com/google/uuid v1.6.0
github.com/lxzan/gws v1.10.1
github.com/nicholas-fedor/shoutrrr v0.17.0
github.com/nicholas-fedor/shoutrrr v0.19.0
github.com/pocketbase/dbx v1.12.0
github.com/pocketbase/pocketbase v0.39.11
github.com/shirou/gopsutil/v4 v4.26.7
github.com/pocketbase/pocketbase v0.40.2
github.com/shirou/gopsutil/v4 v4.26.8
github.com/spf13/cast v1.10.0
github.com/spf13/cobra v1.10.2
github.com/spf13/pflag v1.0.10
github.com/stretchr/testify v1.12.0
golang.org/x/crypto v0.55.0
golang.org/x/exp v0.0.0-20260813180055-c1d0aacb2297
github.com/stretchr/testify v1.12.1
golang.org/x/crypto v0.56.0
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa
golang.org/x/net v0.58.0
golang.org/x/sys v0.47.0
gopkg.in/yaml.v3 v3.0.1
@@ -43,7 +43,7 @@ require (
github.com/golang-jwt/jwt/v5 v5.3.1 // indirect
github.com/gorilla/websocket v1.5.3 // indirect
github.com/inconshreveable/mousetrap v1.1.0 // indirect
github.com/klauspost/compress v1.19.2 // indirect
github.com/klauspost/compress v1.20.0 // indirect
github.com/lufia/plan9stats v0.0.0-20260802145828-341c2f0c90b5 // indirect
github.com/mattn/go-colorable v0.1.15 // indirect
github.com/mattn/go-isatty v0.0.24 // indirect
@@ -55,13 +55,14 @@ require (
github.com/tklauser/numcpus v0.12.0 // indirect
github.com/x448/float16 v0.8.4 // indirect
github.com/yusufpapurcu/wmi v1.2.4 // indirect
go.yaml.in/yaml/v3 v3.0.5 // indirect
golang.org/x/image v0.45.0 // indirect
golang.org/x/oauth2 v0.36.0 // indirect
golang.org/x/sync v0.22.0 // indirect
golang.org/x/term v0.45.0 // indirect
golang.org/x/text v0.41.0 // indirect
modernc.org/libc v1.74.1 // indirect
modernc.org/libc v1.74.4 // indirect
modernc.org/mathutil v1.7.1 // indirect
modernc.org/memory v1.12.0 // indirect
modernc.org/sqlite v1.55.0 // indirect
modernc.org/memory v1.12.1 // indirect
modernc.org/sqlite v1.57.0 // indirect
)
+33 -34
View File
@@ -19,8 +19,8 @@ github.com/domodwyer/mailyak/v3 v3.6.2 h1:x3tGMsyFhTCaxp6ycgR0FE/bu5QiNp+hetUuCO
github.com/domodwyer/mailyak/v3 v3.6.2/go.mod h1:lOm/u9CyCVWHeaAmHIdF4RiKVxKUT/H5XX10lIKAL6c=
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
github.com/ebitengine/purego v0.10.2 h1:W809HbnvzAxgdm+aOvlSekrM16wGCdT/e76+9tS7gzE=
github.com/ebitengine/purego v0.10.2/go.mod h1:iIjxzd6CiRiOG0UyXP+V1+jWqUXVjPKLAI0mRfJZTmQ=
github.com/ebitengine/purego v0.11.0 h1:jhp/D+Nyv7UUW8HAcmcjt2N2rYrYi9m3SL21k0Ua/NI=
github.com/ebitengine/purego v0.11.0/go.mod h1:DCHPP08djqhNSoTfImcnHYQRZmd0qhakvrozqaEYhGQ=
github.com/eclipse/paho.golang v0.23.0 h1:KHgl2wz6EJo7cMBmkuhpt7C576vP+kpPv7jjvSyR6Mk=
github.com/eclipse/paho.golang v0.23.0/go.mod h1:nQRhTkoZv8EAiNs5UU0/WdQIx2NrnWUpL9nsGJTQN04=
github.com/fatih/color v1.19.0 h1:Zp3PiM21/9Ld6FzSKyL5c/BULoe/ONr9KlbYVOfG8+w=
@@ -29,8 +29,8 @@ github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHk
github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0=
github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx59Ho=
github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo=
github.com/fxamacker/cbor/v2 v2.9.2 h1:X4Ksno9+x3cz0TZv69ec1hxP/+tymuR8PXQJyDwfh78=
github.com/fxamacker/cbor/v2 v2.9.2/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ=
github.com/fxamacker/cbor/v2 v2.9.3 h1:oQBnFATpNdY8gJHTndDDv5Xl4QqNaz51G5LLEPhng3Q=
github.com/fxamacker/cbor/v2 v2.9.3/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ=
github.com/gabriel-vasile/mimetype v1.4.15 h1:05iP/CYtZ/w455R/KZM6rZ5ieAdh99UPtd+d3YzLmaI=
github.com/gabriel-vasile/mimetype v1.4.15/go.mod h1:azpTcoLcDZRNgFou5j+APrqQx9HqVPWa6ijYQIIVswQ=
github.com/ganigeorgiev/fexpr v0.6.0 h1:Fza3O/QMBKEudUvxV862qe6GjxM60GJjjKytdp+VQus=
@@ -54,8 +54,8 @@ github.com/golang-jwt/jwt/v5 v5.3.1/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArs
github.com/golang/protobuf v1.3.1/go.mod h1:6lQm79b+lXiMfvg/cZm0SGofjICqVBUtrP5yJMmIC1U=
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo=
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk=
github.com/google/pprof v0.0.0-20260902005441-ca85771921e4 h1:/6mPXfWmhv8eKck12I0YNIcIjwHtxP3YRIMKiEgTjWg=
github.com/google/pprof v0.0.0-20260902005441-ca85771921e4/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk=
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
github.com/gorilla/websocket v1.5.3 h1:saDtZ6Pbx/0u+bgYQ3q96pZgCzfhKXGPqt7kZ72aNNg=
@@ -67,8 +67,8 @@ github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLf
github.com/jarcoal/httpmock v1.4.2 h1:dKwiP/9zITCPfBLsDn3kchbSOu16JrnxtVEmL0fPRcI=
github.com/jarcoal/httpmock v1.4.2/go.mod h1:ftW1xULwo+j0R0JJkJIIi7UKigZUXCLLanykgjwBXL0=
github.com/jessevdk/go-flags v1.4.0/go.mod h1:4FA24M0QyGHXBuZZK/XkWh8h0e1EYbRYJSGM75WSRxI=
github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8=
github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
github.com/klauspost/compress v1.20.0 h1:a3C1ke2ohxFymNlb2HWAHjDeKCI90scRskErZkR0ezA=
github.com/klauspost/compress v1.20.0/go.mod h1:LUdAzn7YLVvxLpc7y3V1m40wESHTgc1422pwwBSKYuI=
github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE=
github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk=
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
@@ -83,19 +83,19 @@ github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsRe
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
github.com/nicholas-fedor/shoutrrr v0.17.0 h1:xfp3z5QbE8jXvUhUEwWDk47SJ/b912VoB8MJJDU+q4E=
github.com/nicholas-fedor/shoutrrr v0.17.0/go.mod h1:s4ldyLs6uwBy9lIjYrY+8lyTqJtPvZSrILw0CyMLock=
github.com/onsi/ginkgo/v2 v2.32.0 h1:Hw7s2pVrQo/8Yz5N77qdnpHaoc+c6cC9WIV1Jce+J6E=
github.com/onsi/ginkgo/v2 v2.32.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
github.com/onsi/gomega v1.42.1 h1:iN1rCUX+44NZ1Dc97MPoeFYbFR0vh8zxoxMFwKdyZ6I=
github.com/onsi/gomega v1.42.1/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg=
github.com/nicholas-fedor/shoutrrr v0.19.0 h1:Rl6bpK3DXuR2Trtx2JV8t+wjUwkHdRHrc8nBKoEpHr0=
github.com/nicholas-fedor/shoutrrr v0.19.0/go.mod h1:Glfdi8AGTbnEn2k2+hW62n8oL0i9vqRVFtXaUIthNks=
github.com/onsi/ginkgo/v2 v2.32.1 h1:6tlvcDm/3sE8lGJbZ4+d4mO3RLy24/tQWOFzVSQNIfw=
github.com/onsi/ginkgo/v2 v2.32.1/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
github.com/onsi/gomega v1.43.0 h1:VlG/1FxqNxhSO+lq/OHBNaaqwiBK/mO8JbVkX9Y+FeU=
github.com/onsi/gomega v1.43.0/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg=
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/pocketbase/dbx v1.12.0 h1:/oLErM+A0b4xI0PWTGPqSDVjzix48PqI/bng2l0PzoA=
github.com/pocketbase/dbx v1.12.0/go.mod h1:xXRCIAKTHMgUCyCKZm55pUOdvFziJjQfXaWKhu2vhMs=
github.com/pocketbase/ozzo-validation/v4 v4.3.0 h1:uKBDVma7bZqgR2a6AwE+k9hkuDFfiZMpBHQdZ1z3iQs=
github.com/pocketbase/ozzo-validation/v4 v4.3.0/go.mod h1:6XNjSTw/Jb2F8LOkKO3oyzIWExbrGiYoS4uVxVwz90g=
github.com/pocketbase/pocketbase v0.39.11 h1:cl/Kh13ukof/4BAEku3OozYrLl85M5/bmH62Ny4szFc=
github.com/pocketbase/pocketbase v0.39.11/go.mod h1:5CaCvp/52fZJ5/qyYsqpCGyVue/kjez889cQAATD2cY=
github.com/pocketbase/pocketbase v0.40.2 h1:7gTqvt3bmilkphyZZ1QNhX19g3BXHqT7ynDyU81RVT4=
github.com/pocketbase/pocketbase v0.40.2/go.mod h1:jc3YuyToy+ZXM4CeO7uSCN/htgR8yv+tjSE3eJZ8eh8=
github.com/power-devops/perfstat v0.0.0-20260805114148-88456608a4f6 h1:jL3a8soXdzuTCcRnKhOmtcsVOObdDTFf4O2B403HPRU=
github.com/power-devops/perfstat v0.0.0-20260805114148-88456608a4f6/go.mod h1:OmDBASR4679mdNQnz2pUhc2G8CO2JrUAVFDRBDP/hJE=
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
@@ -103,8 +103,8 @@ github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qq
github.com/rogpeppe/go-internal v1.9.0 h1:73kH8U+JUqXU8lRuOHeVHaa/SZPifC7BkcraZVejAe8=
github.com/rogpeppe/go-internal v1.9.0/go.mod h1:WtVeX8xhTBvf0smdhujwtBcq4Qrzq/fJaraNFVN+nFs=
github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
github.com/shirou/gopsutil/v4 v4.26.7 h1:IXzpHz/dkMRYAhKkOXr1HB6SuzWU3eoyyeWe7g3bNZc=
github.com/shirou/gopsutil/v4 v4.26.7/go.mod h1:5O9FjBiXoTDFatIWjZZosqj4pV0DRtLx598xGbBehzM=
github.com/shirou/gopsutil/v4 v4.26.8 h1:YQMTF/1J50B5+Y0vlo1eDRf5DoR7Gk69hY+8wjYkQeo=
github.com/shirou/gopsutil/v4 v4.26.8/go.mod h1:5O9FjBiXoTDFatIWjZZosqj4pV0DRtLx598xGbBehzM=
github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY=
github.com/spf13/cast v1.10.0/go.mod h1:jNfB8QC9IA6ZuY2ZjDp0KtFO2LZZlg4S/7bzP6qqeHo=
github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
@@ -116,8 +116,8 @@ github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+
github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4=
github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0=
github.com/stretchr/testify v1.4.0/go.mod h1:j7eGeouHqKxXV5pUuKE4zz7dFj8WfuZ+81PSLYec5m4=
github.com/stretchr/testify v1.12.0 h1:K6Mr6jO9JICuend/5xzTM03ydSV3vdNRYAdPSukj8uI=
github.com/stretchr/testify v1.12.0/go.mod h1:bOYBZb5qJ00vPzWfIqBUZPaxK8jWiXc6d3ErP4Ca9Gw=
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
github.com/tklauser/go-sysconf v0.4.0 h1:7H0uAN+7RkwWRaxhYXDLqa5V3LPrJeV8wmD9dRUgPQU=
github.com/tklauser/go-sysconf v0.4.0/go.mod h1:8mTNWyog7H+MpKijp4VmKJAd2bbYQ2zuUwkYRbUArPI=
github.com/tklauser/numcpus v0.12.0 h1:NR85qdvHA9pFse3x3weVZ0r0ST8R6l5RHbZrlRaqob4=
@@ -132,10 +132,10 @@ go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
golang.org/x/exp v0.0.0-20260813180055-c1d0aacb2297 h1:YXnL44eJ77R+ji4/ooy8UsXIhz+lbi2Qgdlc8iRN0gY=
golang.org/x/exp v0.0.0-20260813180055-c1d0aacb2297/go.mod h1:Mkmymgv+uMpSQ/XxJ/7GpdrdYoqm3u72jEbpCLiJmNk=
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa h1:QSyA8ishJCyT21kER9KwNt0b7BM3iRK4x9QXhjN5Fdk=
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa/go.mod h1:zeBbvyFKDaLwa7CH/zI8KXt7gTl14SF7sO08Pl5jBCM=
golang.org/x/image v0.0.0-20191009234506-e7c1f5e7dbb8/go.mod h1:FeLwcggjj3mMvU+oOTbSwawSJRM1uh48EjtB4UJZlP0=
golang.org/x/image v0.45.0 h1:FMb1nTbH5H9vF55SriQHgFw5GnNL9Jg6L25BwXKzhB0=
golang.org/x/image v0.45.0/go.mod h1:n62x/7RqlwXDvGsSU4u6IUTUf6KghUZ9Bt7cG/T9Fx4=
@@ -164,17 +164,16 @@ golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGm
golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI=
golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo=
google.golang.org/appengine v1.6.5/go.mod h1:8WjMMxjGQR8xUklV/ARdw2HLXBOI7O7uCIDZVag1xfc=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
gopkg.in/yaml.v1 v1.0.0-20140924161607-9f9df34309c0/go.mod h1:WDnlLJ4WF5VGsH/HVa3CI79GS0ol3YnhVnKP89i0kNg=
gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
howett.net/plist v1.0.1 h1:37GdZ8tP09Q35o9ych3ehygcsL+HqKSwzctveSlarvM=
howett.net/plist v1.0.1/go.mod h1:lqaXoTrLY4hg8tnEzNru53gicrbv7rrk+2xJA/7hw9g=
modernc.org/cc/v4 v4.29.0 h1:CXgwL8cvxmyzBQZzbSl/6xFtMCryb6u8IOqDci39cgc=
modernc.org/cc/v4 v4.29.0/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI=
modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=
modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk=
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
@@ -185,18 +184,18 @@ modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI=
modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
modernc.org/libc v1.74.1 h1:bdR4VTKFMC4966QSNZ05XLGI/VwzVa2kTUX51Dm0riQ=
modernc.org/libc v1.74.1/go.mod h1:uH4t5bOx3G3g9Xcmj10YKlTcVISlRDwv8VoQJG9n8Os=
modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k=
modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co=
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
modernc.org/memory v1.12.0 h1:twkmYNkGXCvtYWzoux02jtK6eovjZbdI0uHFUYp6kuU=
modernc.org/memory v1.12.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
modernc.org/memory v1.12.1 h1:nFMiWrpStgZczNl6XI9GnIk/rWhYIyHGUaR04pGbp9g=
modernc.org/memory v1.12.1/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
modernc.org/sqlite v1.55.0 h1:hIFh0MCH0rGinQ/4KYb5/UbCkRkb+UP+OkLCVWa5MTM=
modernc.org/sqlite v1.55.0/go.mod h1:4ntCLuNmnH8+GNqjka1wNg7KJd5/Hi5FYp8K+XQ7GZw=
modernc.org/sqlite v1.57.0 h1:qNQP6xnx5M0ISNtlnxoOX0+cD5bJ0/gr9aMmndFczzg=
modernc.org/sqlite v1.57.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
+17 -4
View File
@@ -20,10 +20,10 @@ type hubLike interface {
}
type AlertManager struct {
hub hubLike
stopOnce sync.Once
pendingAlerts sync.Map
alertsCache *AlertsCache
hub hubLike
stopOnce sync.Once
pendingAlerts sync.Map
alertsCache *AlertsCache
}
type AlertMessageData struct {
@@ -48,6 +48,7 @@ type SystemAlertFsStats struct {
// Values pulled from system_stats.stats that are relevant to alerts.
type SystemAlertStats struct {
Cpu float64 `json:"cpu"`
CpuBreakdown []float64 `json:"cpub"`
Mem float64 `json:"mp"`
Disk float64 `json:"dp"`
Bandwidth [2]uint64 `json:"b"`
@@ -57,12 +58,18 @@ type SystemAlertStats struct {
Battery [2]uint8 `json:"bat"`
Batteries map[string]uint8 `json:"bats"`
ExtraFs map[string]SystemAlertFsStats `json:"efs"`
ZfsPools map[string]SystemAlertZfsPool `json:"z"`
}
type SystemAlertGPUData struct {
Usage float64 `json:"u"`
}
type SystemAlertZfsPool struct {
Total float64 `json:"d"`
Used float64 `json:"du"`
}
type SystemAlertData struct {
systemRecord *core.Record
alertData CachedAlertData
@@ -111,6 +118,9 @@ func (am *AlertManager) bindEvents() {
am.hub.OnRecordAfterUpdateSuccess("alerts").BindFunc(updateHistoryOnAlertUpdate)
am.hub.OnRecordAfterDeleteSuccess("alerts").BindFunc(resolveHistoryOnAlertDelete)
am.hub.OnRecordAfterUpdateSuccess("smart_devices").BindFunc(am.handleSmartDeviceAlert)
am.hub.OnRecordAfterCreateSuccess("zfs_pools").BindFunc(am.handleZfsPoolCreateAlert)
am.hub.OnRecordAfterUpdateSuccess("zfs_pools").BindFunc(am.handleZfsPoolAlert)
am.hub.OnRecordAfterDeleteSuccess("zfs_pools").BindFunc(resolveZfsPoolHistoryOnDelete)
am.hub.OnServe().BindFunc(func(e *core.ServeEvent) error {
// Populate all alerts into cache on startup
@@ -119,6 +129,9 @@ func (am *AlertManager) bindEvents() {
if err := resolveStatusAlerts(e.App); err != nil {
e.App.Logger().Error("Failed to resolve stale status alerts", "err", err)
}
if err := resolveSystemdAlerts(e.App); err != nil {
e.App.Logger().Error("Failed to resolve stale systemd alerts", "err", err)
}
if err := am.restorePendingStatusAlerts(); err != nil {
e.App.Logger().Error("Failed to restore pending status alerts", "err", err)
}
+27 -1
View File
@@ -9,6 +9,7 @@ import (
"slices"
"strings"
"github.com/henrygd/beszel/internal/hub/utils"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
@@ -37,6 +38,9 @@ func UpsertUserAlerts(e *core.RequestEvent) error {
err = e.App.RunInTransaction(func(txApp core.App) error {
for _, systemId := range reqData.Systems {
if !userHasSystem(txApp, userID, systemId) {
continue
}
// find existing matching alert
alertRecord, err := txApp.FindFirstRecordByFilter(alertsCollection,
"system={:system} && name={:name} && user={:user}",
@@ -94,6 +98,9 @@ func DeleteUserAlerts(e *core.RequestEvent) error {
err = e.App.RunInTransaction(func(txApp core.App) error {
for _, systemId := range reqData.Systems {
if !userHasSystem(txApp, userID, systemId) {
continue
}
// Find existing alert to delete
alertRecord, err := txApp.FindFirstRecordByFilter("alerts",
"system={:system} && name={:name} && user={:user}",
@@ -122,6 +129,15 @@ func DeleteUserAlerts(e *core.RequestEvent) error {
return e.JSON(http.StatusOK, map[string]any{"success": true, "count": numDeleted})
}
func userHasSystem(app core.App, userID, systemID string) bool {
system, err := app.FindRecordById("systems", systemID)
if err != nil {
return false
}
shareAll, _ := utils.GetEnv("SHARE_ALL_SYSTEMS")
return shareAll == "true" || slices.Contains(system.GetStringSlice("users"), userID)
}
// SendTestNotification handles API request to send a test notification to a specified Shoutrrr URL
func (am *AlertManager) SendTestNotification(e *core.RequestEvent) error {
var data struct {
@@ -187,6 +203,16 @@ func isInternalURL(rawURL string) (bool, error) {
return false, nil
}
var cgnatNetwork = &net.IPNet{
IP: net.IPv4(100, 64, 0, 0),
Mask: net.CIDRMask(10, 32),
}
func isInternalIP(ip net.IP) bool {
return ip.IsPrivate() || ip.IsLoopback() || ip.IsUnspecified()
return ip.IsPrivate() ||
ip.IsLoopback() ||
ip.IsUnspecified() ||
ip.IsLinkLocalUnicast() ||
ip.IsMulticast() ||
cgnatNetwork.Contains(ip)
}
+64 -3
View File
@@ -36,11 +36,23 @@ func TestIsInternalURL(t *testing.T) {
internal bool
}{
{name: "loopback ipv4", url: "generic://127.0.0.1", internal: true},
{name: "private ipv4", url: "generic://10.0.0.1", internal: true},
{name: "localhost hostname", url: "generic://localhost", internal: true},
{name: "localhost hostname", url: "generic+http://localhost/api/v1/postStuff", internal: true},
{name: "localhost hostname", url: "generic+http://127.0.0.1:8080/api/v1/postStuff", internal: true},
{name: "localhost hostname", url: "generic+https://beszel.dev/api/v1/postStuff", internal: false},
{name: "localhost with path", url: "generic+http://localhost/api/v1/postStuff", internal: true},
{name: "loopback with port and path", url: "generic+http://127.0.0.1:8080/api/v1/postStuff", internal: true},
{name: "public hostname", url: "generic+https://beszel.dev/api/v1/postStuff", internal: false},
{name: "cloud metadata ipv4", url: "generic://169.254.169.254", internal: true},
{name: "link-local ipv4", url: "generic://169.254.1.1", internal: true},
{name: "link-local ipv6", url: "generic://[fe80::1]", internal: true},
{name: "mapped link-local ipv4", url: "generic://[::ffff:169.254.169.254]", internal: true},
{name: "cgnat lower boundary", url: "generic://100.64.0.0", internal: true},
{name: "cgnat upper boundary", url: "generic://100.127.255.255", internal: true},
{name: "below cgnat", url: "generic://100.63.255.255", internal: false},
{name: "above cgnat", url: "generic://100.128.0.0", internal: false},
{name: "multicast ipv4", url: "generic://224.0.0.1", internal: true},
{name: "multicast ipv6", url: "generic://[ff02::1]", internal: true},
{name: "public ipv4", url: "generic://8.8.8.8", internal: false},
{name: "public ipv6", url: "generic://[2001:4860:4860::8888]", internal: false},
{name: "token style service url", url: "discord://abc123@123456789", internal: false},
{name: "single label service url", url: "slack://token@team/channel", internal: false},
}
@@ -190,6 +202,30 @@ func TestUserAlertsApi(t *testing.T) {
assert.EqualValues(t, 3, user1Alerts, "should have 3 alerts")
},
},
{
Name: "POST ignores systems the user cannot access",
Method: http.MethodPost,
URL: "/api/beszel/user-alerts",
Headers: map[string]string{
"Authorization": user2Token,
},
ExpectedStatus: 200,
ExpectedContent: []string{"\"success\":true"},
TestAppFactory: testAppFactory,
Body: jsonReader(map[string]any{
"name": "CPU",
"systems": []string{system1.Id},
"value": 90,
"min": 10,
}),
BeforeTestFunc: func(t testing.TB, app *pbTests.TestApp, e *core.ServeEvent) {
beszelTests.ClearCollection(t, app, "alerts")
},
AfterTestFunc: func(t testing.TB, app *pbTests.TestApp, res *http.Response) {
alerts, _ := app.CountRecords("alerts")
assert.Zero(t, alerts)
},
},
{
Name: "Overwrite: false, should not overwrite existing alert",
Method: http.MethodPost,
@@ -347,6 +383,31 @@ func TestUserAlertsApi(t *testing.T) {
assert.Zero(t, alerts, "should have 0 alerts")
},
},
{
Name: "DELETE ignores systems the user cannot access",
Method: http.MethodDelete,
URL: "/api/beszel/user-alerts",
Headers: map[string]string{
"Authorization": user2Token,
},
ExpectedStatus: 200,
ExpectedContent: []string{"\"count\":0", "\"success\":true"},
TestAppFactory: testAppFactory,
Body: jsonReader(map[string]any{
"name": "CPU",
"systems": []string{system1.Id},
}),
BeforeTestFunc: func(t testing.TB, app *pbTests.TestApp, e *core.ServeEvent) {
beszelTests.ClearCollection(t, app, "alerts")
beszelTests.CreateRecord(app, "alerts", map[string]any{
"name": "CPU", "system": system1.Id, "user": user2.Id, "value": 80,
})
},
AfterTestFunc: func(t testing.TB, app *pbTests.TestApp, res *http.Response) {
alerts, _ := app.CountRecords("alerts")
assert.EqualValues(t, 1, alerts)
},
},
{
Name: "User 2 should not be able to delete alert of user 1",
Method: http.MethodDelete,
+11 -7
View File
@@ -1,6 +1,8 @@
package alerts
import (
"time"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/store"
@@ -8,13 +10,14 @@ import (
// CachedAlertData represents the relevant fields of an alert record for status checking and updates.
type CachedAlertData struct {
Id string
SystemID string
UserID string
Name string
Value float64
Triggered bool
Min uint8
Id string
SystemID string
UserID string
Name string
Value float64
Triggered bool
Min uint8
PendingSince time.Time
// Created types.DateTime
}
@@ -26,6 +29,7 @@ func (a *CachedAlertData) PopulateFromRecord(record *core.Record) {
a.Value = record.GetFloat("value")
a.Triggered = record.GetBool("triggered")
a.Min = uint8(record.GetInt("min"))
a.PendingSince = record.GetDateTime("pending_since").Time()
// a.Created = record.GetDateTime("created")
}
+318
View File
@@ -0,0 +1,318 @@
package alerts
import (
"errors"
"fmt"
"strings"
"time"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/pocketbase/pocketbase/core"
)
const (
// containerAlertName is the value stored in the alerts.name field for this alert type.
containerAlertName = "ContainerHealth"
// containerLogMaxLines caps how many matched (error/fatal) log lines are kept.
containerLogMaxLines = 12
// containerLogFallbackLines is how many trailing raw log lines are used when no
// line matches "error" or "fatal", so the notification still carries some context.
containerLogFallbackLines = 6
// containerLogExcerptMaxChars bounds a single container's log excerpt so a
// handful of containers can't blow past Discord's message size limit.
containerLogExcerptMaxChars = 500
// containerAlertMaxLogged is the max number of unhealthy containers we fetch
// and embed logs for in a single alert message.
containerAlertMaxLogged = 2
// containerAlertMessageMaxChars is a final safety cap on the whole message body.
containerAlertMessageMaxChars = 1800
)
// FetchContainerLogsFunc retrieves recent logs for a container ID from its
// connected agent. Implementations should apply their own timeout. This is a
// type alias (not a defined type) so it satisfies the hubLike interface in
// internal/hub/systems, which declares the same func signature without
// importing this package.
type FetchContainerLogsFunc = func(containerID string) (string, error)
// containerAlertTarget is an immutable snapshot of the fields needed after the
// alert fires. Keeping agent-owned container records out of notification work
// avoids retaining and concurrently reading data that is refreshed in place.
type containerAlertTarget struct {
id string
name string
}
// HandleContainerAlerts checks configured "ContainerHealth" alerts for a system
// against the Docker container health data included in the latest agent update.
// It persists when containers first become unhealthy, fires from a fresh poll
// once the configured delay has elapsed, and resolves once containers recover.
// fetchLogs is used when an alert actually fires so the notification can include
// a log excerpt (prioritizing lines containing "error"/"fatal") for context.
func (am *AlertManager) HandleContainerAlerts(systemRecord *core.Record, data *system.CombinedData, fetchLogs FetchContainerLogsFunc) error {
alerts := am.alertsCache.GetAlertsByName(systemRecord.Id, containerAlertName)
if len(alerts) == 0 {
return nil
}
if data.Containers == nil {
// An unknown Docker state must not resolve a triggered alert or count
// toward the minimum unhealthy duration.
var result error
for _, alertData := range alerts {
if err := am.clearPendingContainerAlert(alertData); err != nil {
result = errors.Join(result, err)
}
}
return result
}
var unhealthy []*container.Stats
for _, c := range data.Containers {
if c.Health == container.DockerHealthUnhealthy {
unhealthy = append(unhealthy, c)
}
}
systemName := systemRecord.GetString("name")
now := time.Now().UTC()
var result error
for _, alertData := range alerts {
if len(unhealthy) > 0 {
if alertData.Triggered {
continue
}
min := max(1, int(alertData.Min))
if alertData.PendingSince.IsZero() {
pendingSince, err := am.setPendingContainerAlert(alertData, now)
if err != nil {
result = errors.Join(result, err)
continue
}
if pendingSince.IsZero() {
continue
}
alertData.PendingSince = pendingSince
if min > 1 {
continue
}
}
if min > 1 && now.Before(alertData.PendingSince.Add(time.Duration(min)*time.Minute)) {
continue
}
if err := am.sendContainerHealthAlert(true, systemName, alertData, snapshotContainerAlertTargets(unhealthy), fetchLogs); err != nil {
result = errors.Join(result, err)
}
continue
}
// no unhealthy containers right now
if err := am.clearPendingContainerAlert(alertData); err != nil {
result = errors.Join(result, err)
}
if !alertData.Triggered {
continue
}
if err := am.sendContainerHealthAlert(false, systemName, alertData, nil, fetchLogs); err != nil {
result = errors.Join(result, err)
}
}
return result
}
func snapshotContainerAlertTargets(containers []*container.Stats) []containerAlertTarget {
targets := make([]containerAlertTarget, len(containers))
for i, c := range containers {
targets[i] = containerAlertTarget{id: c.Id, name: c.Name}
}
return targets
}
// setPendingContainerAlert durably records the first unhealthy observation and
// returns the persisted generation used to claim delivery.
func (am *AlertManager) setPendingContainerAlert(alertData CachedAlertData, since time.Time) (time.Time, error) {
record, err := am.hub.FindRecordById("alerts", alertData.Id)
if err != nil {
return time.Time{}, err
}
if record.GetBool("triggered") {
return time.Time{}, nil
}
if pendingSince := record.GetDateTime("pending_since").Time(); !pendingSince.IsZero() {
return pendingSince, nil
}
// PocketBase date fields are persisted with millisecond precision. Normalize
// before saving so the update-hook cache and a subsequent database read agree.
since = since.Truncate(time.Millisecond)
record.Set("pending_since", since)
return since, am.hub.Save(record)
}
func (am *AlertManager) clearPendingContainerAlert(alertData CachedAlertData) error {
if alertData.PendingSince.IsZero() {
return nil
}
record, err := am.hub.FindRecordById("alerts", alertData.Id)
if err != nil {
return err
}
if record.GetDateTime("pending_since").Time().IsZero() {
return nil
}
record.Set("pending_since", nil)
return am.hub.Save(record)
}
// claimPendingContainerAlert marks an alert triggered only if the pending
// generation is still current. A healthy/unknown update can clear the timestamp
// while logs are being fetched, causing this claim to become a no-op.
func (am *AlertManager) claimPendingContainerAlert(alertData CachedAlertData) (bool, error) {
record, err := am.hub.FindRecordById("alerts", alertData.Id)
if err != nil {
return false, err
}
pendingSince := record.GetDateTime("pending_since").Time()
if record.GetBool("triggered") || pendingSince.IsZero() || pendingSince.UnixMilli() != alertData.PendingSince.UnixMilli() {
return false, nil
}
record.Set("pending_since", nil)
record.Set("triggered", true)
return true, am.hub.Save(record)
}
// CancelPendingContainerAlerts clears pending container-health durations for a
// system. Called when monitoring pauses or the system goes down.
func (am *AlertManager) CancelPendingContainerAlerts(systemID string) {
for _, alertData := range am.alertsCache.GetAlertsByName(systemID, containerAlertName) {
if err := am.clearPendingContainerAlert(alertData); err != nil {
am.hub.Logger().Error("Failed to clear pending container alert", "err", err)
}
}
}
// sendContainerHealthAlert updates the alert's triggered state and sends the
// notification. When unhealthy is true, it embeds a log excerpt (prioritizing
// error/fatal lines) for up to containerAlertMaxLogged of the affected containers.
func (am *AlertManager) sendContainerHealthAlert(unhealthy bool, systemName string, alertData CachedAlertData, containers []containerAlertTarget, fetchLogs FetchContainerLogsFunc) error {
link := am.hub.MakeLink("system", alertData.SystemID)
linkText := "View " + systemName
if !unhealthy {
if err := am.setAlertTriggered(alertData, false); err != nil {
return err
}
title := fmt.Sprintf("%s containers are healthy ✅", systemName)
return am.SendAlert(AlertMessageData{
UserID: alertData.UserID,
SystemID: alertData.SystemID,
Title: title,
Message: strings.TrimSuffix(title, " ✅"),
Link: link,
LinkText: linkText,
})
}
names := make([]string, len(containers))
for i, c := range containers {
names[i] = c.name
}
var title string
if len(names) == 1 {
title = fmt.Sprintf("Unhealthy container %s on %s \U0001F534", names[0], systemName)
} else {
title = fmt.Sprintf("%d unhealthy containers on %s \U0001F534", len(names), systemName)
}
var body strings.Builder
fmt.Fprintf(&body, "Unhealthy: %s", strings.Join(names, ", "))
body.WriteString(am.buildContainerLogsSection(containers, fetchLogs))
message := body.String()
if len(message) > containerAlertMessageMaxChars {
message = message[:containerAlertMessageMaxChars] + "\n…(truncated)"
}
claimed, err := am.claimPendingContainerAlert(alertData)
if err != nil || !claimed {
return err
}
return am.SendAlert(AlertMessageData{
UserID: alertData.UserID,
SystemID: alertData.SystemID,
Title: title,
Message: message,
Link: link,
LinkText: linkText,
})
}
// buildContainerLogsSection attempts to fetch and format log excerpts for up to
// containerAlertMaxLogged unhealthy containers, to append to an alert message.
func (am *AlertManager) buildContainerLogsSection(containers []containerAlertTarget, fetchLogs FetchContainerLogsFunc) string {
if fetchLogs == nil {
return ""
}
var section strings.Builder
attempts := min(len(containers), containerAlertMaxLogged)
for _, c := range containers[:attempts] {
rawLogs, err := fetchLogs(c.id)
if err != nil {
am.hub.Logger().Warn("Failed to fetch container logs for alert", "container", c.name, "err", err)
continue
}
excerpt := buildContainerLogExcerpt(rawLogs)
if excerpt == "" {
continue
}
fmt.Fprintf(&section, "\n\n%s logs:\n```\n%s\n```", c.name, excerpt)
}
if len(containers) > containerAlertMaxLogged {
fmt.Fprintf(&section, "\n\n(+%d more unhealthy container(s), logs omitted)", len(containers)-containerAlertMaxLogged)
}
return section.String()
}
// buildContainerLogExcerpt filters raw container log output down to the lines
// most likely to explain why the container is unhealthy: lines containing
// "error" or "fatal" (case-insensitive) are preferred. If none match, the tail
// of the raw output is used instead so the notification still carries context.
func buildContainerLogExcerpt(raw string) string {
raw = strings.TrimSpace(raw)
if raw == "" {
return ""
}
lines := strings.Split(raw, "\n")
var matched []string
for _, line := range lines {
line = strings.TrimRight(line, "\r")
if line == "" {
continue
}
lower := strings.ToLower(line)
if strings.Contains(lower, "error") || strings.Contains(lower, "fatal") {
matched = append(matched, line)
}
}
selected := matched
if len(selected) == 0 {
start := max(0, len(lines)-containerLogFallbackLines)
selected = lines[start:]
} else if len(selected) > containerLogMaxLines {
selected = selected[len(selected)-containerLogMaxLines:]
}
excerpt := strings.TrimSpace(strings.Join(selected, "\n"))
if len(excerpt) > containerLogExcerptMaxChars {
excerpt = "…" + excerpt[len(excerpt)-containerLogExcerptMaxChars:]
}
return excerpt
}
+349
View File
@@ -0,0 +1,349 @@
//go:build testing
package alerts_test
import (
"fmt"
"strings"
"testing"
"testing/synctest"
"time"
"github.com/henrygd/beszel/internal/alerts"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/henrygd/beszel/internal/entities/system"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
type containerAlertTestFixture struct {
hub *beszelTests.TestHub
am *alerts.AlertManager
alertID string
systemRecord *core.Record
}
func newContainerAlertTestFixture(t *testing.T, min int) *containerAlertTestFixture {
t.Helper()
hub, user := beszelTests.GetHubWithUser(t)
systems, err := beszelTests.CreateSystems(hub, 1, user.Id, "up")
require.NoError(t, err)
systemRecord := systems[0]
userSettings, err := hub.FindFirstRecordByFilter("user_settings", "user={:user}", map[string]any{"user": user.Id})
require.NoError(t, err)
userSettings.Set("settings", `{"emails":["test@example.com"],"webhooks":[]}`)
require.NoError(t, hub.Save(userSettings))
alertRecord, err := beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "ContainerHealth",
"system": systemRecord.Id,
"user": user.Id,
"min": min,
})
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "Alert should not be triggered initially")
return &containerAlertTestFixture{
hub: hub,
am: alerts.NewTestAlertManagerWithoutWorker(hub),
alertID: alertRecord.Id,
systemRecord: systemRecord,
}
}
func (f *containerAlertTestFixture) cleanup() {
f.hub.Cleanup()
}
func (f *containerAlertTestFixture) submit(t *testing.T, containers []*container.Stats, fetchLogs alerts.FetchContainerLogsFunc) {
t.Helper()
data := &system.CombinedData{Containers: containers}
require.NoError(t, f.am.HandleContainerAlerts(f.systemRecord, data, fetchLogs))
}
func (f *containerAlertTestFixture) submitInvalid(t *testing.T) {
t.Helper()
require.NoError(t, f.am.HandleContainerAlerts(f.systemRecord, &system.CombinedData{}, nil))
}
func (f *containerAlertTestFixture) assertTriggered(t *testing.T, triggered bool, message string) {
t.Helper()
alertRecord, err := f.hub.FindRecordById("alerts", f.alertID)
require.NoError(t, err)
assert.Equal(t, triggered, alertRecord.GetBool("triggered"), message)
}
func (f *containerAlertTestFixture) assertPending(t *testing.T, pending bool) {
t.Helper()
alertRecord, err := f.hub.FindRecordById("alerts", f.alertID)
require.NoError(t, err)
assert.Equal(t, pending, !alertRecord.GetDateTime("pending_since").Time().IsZero())
}
func waitForContainerAlert(d time.Duration) {
time.Sleep(d)
synctest.Wait()
}
func healthyContainer(name string) *container.Stats {
return &container.Stats{Name: name, Id: "abc123def456", Health: container.DockerHealthHealthy}
}
func unhealthyContainer(name string) *container.Stats {
return &container.Stats{Name: name, Id: "abc123def456", Health: container.DockerHealthUnhealthy}
}
func TestContainerHealthAlertTriggersAndResolves(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("web")}, nil)
fixture.assertTriggered(t, true, "A one-minute alert should trigger on the first unhealthy update")
require.Equal(t, 1, fixture.hub.TestMailer.TotalSend(), "An email should have been sent")
msg := fixture.hub.TestMailer.LastMessage()
assert.Contains(t, msg.Subject, "web", "Subject should name the unhealthy container")
assert.Contains(t, strings.ToLower(msg.Subject), "unhealthy")
fixture.submit(t, []*container.Stats{unhealthyContainer("web")}, nil)
fixture.assertPending(t, false)
fixture.submitInvalid(t)
fixture.assertTriggered(t, true, "An invalid container snapshot should not resolve the alert")
assert.Equal(t, 1, fixture.hub.TestMailer.TotalSend(), "An invalid snapshot should not send a recovery")
fixture.submit(t, []*container.Stats{}, nil)
waitForContainerAlert(time.Second)
fixture.assertTriggered(t, false, "Alert should resolve once the container is healthy again")
assert.Equal(t, 2, fixture.hub.TestMailer.TotalSend(), "A second email should have been sent for the recovery")
assert.Contains(t, fixture.hub.TestMailer.LastMessage().Subject, " healthy")
})
}
func TestContainerHealthAlertInvalidSnapshotCancelsPending(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 5)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("db")}, nil)
fixture.assertPending(t, true)
waitForContainerAlert(time.Minute)
fixture.submitInvalid(t)
fixture.assertPending(t, false)
waitForContainerAlert(10 * time.Minute)
fixture.submit(t, []*container.Stats{unhealthyContainer("db")}, nil)
fixture.assertTriggered(t, false, "Stale unhealthy data should not trigger an alert")
fixture.assertPending(t, true)
assert.Equal(t, 0, fixture.hub.TestMailer.TotalSend())
})
}
func TestContainerHealthAlertSystemDownCancelsPending(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 5)
defer fixture.cleanup()
// Use the hub's alert manager because the system-manager status hook invokes
// cancellation on that instance.
am := fixture.hub.GetAlertManager()
require.NoError(t, am.HandleContainerAlerts(
fixture.systemRecord,
&system.CombinedData{Containers: []*container.Stats{unhealthyContainer("db")}},
nil,
))
fixture.assertPending(t, true)
fixture.systemRecord.Set("status", "down")
require.NoError(t, fixture.hub.Save(fixture.systemRecord))
fixture.assertPending(t, false)
}
func TestContainerHealthAlertResolvesBeforeMinDelayCancelsPending(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 5)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("db")}, nil)
waitForContainerAlert(time.Minute)
fixture.assertTriggered(t, false, "Alert should not fire until the min delay elapses")
fixture.assertPending(t, true)
assert.Equal(t, 0, fixture.hub.TestMailer.TotalSend())
// container recovers before the 5 minute delay elapses
fixture.submit(t, []*container.Stats{healthyContainer("db")}, nil)
waitForContainerAlert(10 * time.Minute)
fixture.submit(t, []*container.Stats{healthyContainer("db")}, nil)
fixture.assertTriggered(t, false, "Alert should remain untriggered")
fixture.assertPending(t, false)
assert.Equal(t, 0, fixture.hub.TestMailer.TotalSend(), "No email should be sent for a container that recovered before the delay")
})
}
func TestContainerHealthAlertPreservesPendingDurationAcrossManagerRestart(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 2)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("db")}, nil)
waitForContainerAlert(30 * time.Second)
restarted := alerts.NewTestAlertManagerWithoutWorker(fixture.hub)
waitForContainerAlert(91 * time.Second)
require.NoError(t, restarted.HandleContainerAlerts(
fixture.systemRecord,
&system.CombinedData{Containers: []*container.Stats{unhealthyContainer("db")}},
nil,
))
fixture.assertTriggered(t, true, "Restart should preserve the original unhealthy start time")
fixture.assertPending(t, false)
assert.Equal(t, 1, fixture.hub.TestMailer.TotalSend())
})
}
func TestContainerHealthAlertClaimsPendingTimestampAtDatabasePrecision(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
alertRecord, err := fixture.hub.FindRecordById("alerts", fixture.alertID)
require.NoError(t, err)
// PocketBase persists dates to milliseconds, while record update hooks can
// retain the original sub-millisecond value in the in-memory alert cache.
alertRecord.Set("pending_since", time.Now().UTC().Add(-2*time.Minute).Truncate(time.Millisecond).Add(123*time.Nanosecond))
require.NoError(t, fixture.hub.Save(alertRecord))
fixture.submit(t, []*container.Stats{unhealthyContainer("db")}, nil)
fixture.assertTriggered(t, true, "Equivalent persisted and cached timestamps should claim the alert")
fixture.assertPending(t, false)
assert.Equal(t, 1, fixture.hub.TestMailer.TotalSend())
}
func TestContainerHealthAlertRecoveryWhileFetchingLogsCancelsDelivery(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
fetchLogs := func(containerID string) (string, error) {
fixture.submit(t, []*container.Stats{healthyContainer("api")}, nil)
return "FATAL stale failure", nil
}
fixture.submit(t, []*container.Stats{unhealthyContainer("api")}, fetchLogs)
fixture.assertTriggered(t, false, "Recovery should cancel delivery while logs are fetched")
fixture.assertPending(t, false)
assert.Equal(t, 0, fixture.hub.TestMailer.TotalSend())
})
}
func TestContainerHealthAlertIncludesLogExcerpt(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
rawLogs := strings.Join([]string{
"2026-08-16T10:00:00Z booting",
"2026-08-16T10:00:01Z ERROR could not reach upstream",
"2026-08-16T10:00:02Z FATAL giving up after 3 retries",
}, "\n")
fetchLogs := func(containerID string) (string, error) {
assert.Equal(t, "abc123def456", containerID)
return rawLogs, nil
}
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("api")}, fetchLogs)
fixture.assertTriggered(t, true, "Alert should be triggered")
require.Equal(t, 1, fixture.hub.TestMailer.TotalSend())
body := fixture.hub.TestMailer.LastMessage().Text
assert.Contains(t, body, "could not reach upstream")
assert.Contains(t, body, "giving up after 3 retries")
assert.NotContains(t, body, "booting", "non error/fatal lines should be dropped when matches exist")
})
}
func TestContainerHealthAlertSkipsLogsOnFetchError(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
fetchLogs := func(containerID string) (string, error) {
return "", fmt.Errorf("agent unreachable")
}
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, []*container.Stats{unhealthyContainer("api")}, fetchLogs)
fixture.assertTriggered(t, true, "Alert should still be triggered even if logs can't be fetched")
require.Equal(t, 1, fixture.hub.TestMailer.TotalSend())
})
}
func TestContainerHealthAlertCapsLogFetchAttempts(t *testing.T) {
fixture := newContainerAlertTestFixture(t, 1)
defer fixture.cleanup()
containers := make([]*container.Stats, 100)
for i := range containers {
containers[i] = &container.Stats{
Name: fmt.Sprintf("container-%d", i),
Id: fmt.Sprintf("id-%d", i),
Health: container.DockerHealthUnhealthy,
}
}
attempts := 0
fetchLogs := func(containerID string) (string, error) {
attempts++
return "", fmt.Errorf("agent unreachable")
}
synctest.Test(t, func(t *testing.T) {
fixture.submit(t, containers, fetchLogs)
fixture.assertTriggered(t, true, "Alert should still fire when log retrieval fails")
assert.Equal(t, 2, attempts, "Log retrieval should attempt at most two containers")
require.Equal(t, 1, fixture.hub.TestMailer.TotalSend())
})
}
func TestBuildContainerLogExcerptPrefersErrorAndFatalLines(t *testing.T) {
raw := strings.Join([]string{
"2026-08-16T10:00:00Z starting up",
"2026-08-16T10:00:01Z listening on :8080",
"2026-08-16T10:00:02Z ERROR failed to connect to db",
"2026-08-16T10:00:03Z retrying connection",
"2026-08-16T10:00:04Z FATAL could not recover, exiting",
}, "\n")
excerpt := alerts.BuildContainerLogExcerpt(raw)
assert.Contains(t, excerpt, "failed to connect to db")
assert.Contains(t, excerpt, "could not recover, exiting")
assert.NotContains(t, excerpt, "starting up", "non-matching lines should be dropped when error/fatal lines exist")
}
func TestBuildContainerLogExcerptFallsBackToTailWhenNoMatches(t *testing.T) {
var lines []string
for i := range 20 {
lines = append(lines, fmt.Sprintf("line %d: all good here", i))
}
raw := strings.Join(lines, "\n")
excerpt := alerts.BuildContainerLogExcerpt(raw)
assert.Contains(t, excerpt, "line 19", "should keep the tail of the output")
assert.NotContains(t, excerpt, "line 0:", "should not keep the very start when falling back to a short tail")
}
func TestBuildContainerLogExcerptEmpty(t *testing.T) {
assert.Equal(t, "", alerts.BuildContainerLogExcerpt(" \n \n"))
}
+81 -4
View File
@@ -13,8 +13,41 @@ import (
"github.com/pocketbase/pocketbase/tools/types"
)
var cpuStateAlerts = map[string]struct {
index int
label string
}{
"CPUIOWait": {2, "CPU I/O Wait"},
"CPUSteal": {3, "CPU Steal Time"},
}
func cpuStateAlertValue(name string, breakdown []float64) (float64, bool) {
state, ok := cpuStateAlerts[name]
if !ok || len(breakdown) < 5 {
return 0, false
}
var total float64
for _, value := range breakdown {
total += value
}
if total <= 0 {
return 0, false
}
return breakdown[state.index], true
}
func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *system.CombinedData) error {
alerts := am.alertsCache.GetAlertsExcludingNames(systemRecord.Id, "Status")
// Systemd alerts are binary state, not numeric thresholds, so they're handled
// separately. They read their own state from the database and don't use data.
if err := am.HandleSystemdAlerts(systemRecord); err != nil {
am.hub.Logger().Error("Error handling systemd alerts", "err", err)
}
if data == nil {
return nil
}
alerts := am.alertsCache.GetAlertsExcludingNames(systemRecord.Id, "Status", alertNameSystemdFailed, containerAlertName)
if len(alerts) == 0 {
return nil
}
@@ -44,6 +77,14 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
maxUsedPct = usedPct
}
}
for _, pool := range data.Stats.ZfsPools {
if pool != nil && pool.Total > 0 {
usedPct := pool.Used / pool.Total * 100
if usedPct > maxUsedPct {
maxUsedPct = usedPct
}
}
}
val = maxUsedPct
case "Temperature":
if data.Info.DashboardTemp < 1 {
@@ -67,6 +108,11 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
continue
}
val = float64(data.Stats.Battery[0])
default:
var ok bool
if val, ok = cpuStateAlertValue(name, data.Stats.CpuBreakdown); !ok {
continue
}
}
triggered := alertData.Triggered
@@ -208,6 +254,16 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
alert.mapSums[key] += float32(fs.DiskUsed / fs.DiskTotal * 100)
}
}
// add zfs pool usage from historical record
for key, pool := range stats.ZfsPools {
if pool.Total > 0 {
zfsKey := zfsDiskAlertKey(key)
if _, ok := alert.mapSums[zfsKey]; !ok {
alert.mapSums[zfsKey] = 0.0
}
alert.mapSums[zfsKey] += float32(pool.Used / pool.Total * 100)
}
}
case "Temperature":
if alert.mapSums == nil {
alert.mapSums = make(map[string]float32, len(stats.Temperatures))
@@ -241,13 +297,20 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
}
alert.val += float64(stats.Battery[0])
default:
continue
value, ok := cpuStateAlertValue(alert.name, stats.CpuBreakdown)
if !ok {
continue
}
alert.val += value
}
alert.count++
}
}
// sum up vals for each alert
for _, alert := range validAlerts {
if alert.count == 0 {
continue
}
switch alert.name {
case "Disk":
maxPct := float32(0)
@@ -255,7 +318,7 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
sumPct := float32(value)
if sumPct > maxPct {
maxPct = sumPct
alert.descriptor = fmt.Sprintf("Usage of %s", key)
alert.descriptor = diskAlertDescriptor(key)
}
}
alert.val = float64(maxPct / float32(alert.count))
@@ -301,6 +364,17 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
return nil
}
func zfsDiskAlertKey(poolName string) string {
return "zfs:" + poolName
}
func diskAlertDescriptor(key string) string {
if poolName, ok := strings.CutPrefix(key, "zfs:"); ok {
return fmt.Sprintf("Usage of ZFS pool %s", poolName)
}
return fmt.Sprintf("Usage of %s", key)
}
func hasRepresentativeBattery(legacy [2]uint8, batteries map[string]uint8) bool {
return legacy != [2]uint8{} || len(batteries) > 0
}
@@ -309,6 +383,9 @@ func (am *AlertManager) sendSystemAlert(alert SystemAlertData) {
// log.Printf("Sending alert %s: val %f | count %d | threshold %f\n", alert.name, alert.val, alert.count, alert.threshold)
systemName := alert.systemRecord.GetString("name")
if state, ok := cpuStateAlerts[alert.name]; ok {
alert.name = state.label
}
// change Disk to Disk usage
if alert.name == "Disk" {
alert.name += " usage"
@@ -320,7 +397,7 @@ func (am *AlertManager) sendSystemAlert(alert SystemAlertData) {
// make title alert name lowercase if not CPU or GPU
titleAlertName := alert.name
if titleAlertName != "CPU" && titleAlertName != "GPU" {
if titleAlertName != "CPU" && titleAlertName != "GPU" && !strings.HasPrefix(titleAlertName, "CPU") {
titleAlertName = strings.ToLower(titleAlertName)
}
+38
View File
@@ -146,6 +146,20 @@ func setCPUAlertValue(info *system.Info, stats *system.Stats, value float64) {
stats.Cpu = value
}
func setCPUStateAlertValue(_ *system.Info, stats *system.Stats, value []float64) {
stats.CpuBreakdown = value
}
var cpuStateAlertTests = []struct {
name string
trigger []float64
resolve []float64
baseline []float64
}{
{"CPUIOWait", []float64{0, 0, 51, 0, 49}, []float64{0, 0, 48, 0, 52}, []float64{0, 0, 10, 0, 90}},
{"CPUSteal", []float64{0, 0, 0, 51, 49}, []float64{0, 0, 0, 48, 52}, []float64{0, 0, 0, 10, 90}},
}
func setMemoryAlertValue(info *system.Info, stats *system.Stats, value float64) {
info.MemPct = value
stats.MemPct = value
@@ -191,6 +205,11 @@ func setBatteryAlertValue(info *system.Info, stats *system.Stats, value [2]uint8
func TestSystemAlertsOneMin(t *testing.T) {
testOneMinuteSystemAlert(t, "CPU", 50, setCPUAlertValue, 51, 49)
for _, test := range cpuStateAlertTests {
t.Run(test.name, func(t *testing.T) {
testOneMinuteSystemAlert(t, test.name, 50, setCPUStateAlertValue, test.trigger, test.resolve)
})
}
testOneMinuteSystemAlert(t, "Memory", 50, setMemoryAlertValue, 51, 49)
testOneMinuteSystemAlert(t, "Disk", 50, setDiskAlertValue, 51, 49)
testOneMinuteSystemAlert(t, "Bandwidth", 50, setBandwidthAlertValue, [2]uint64{megabytesToBytes(26), megabytesToBytes(25)}, [2]uint64{megabytesToBytes(25), megabytesToBytes(24)})
@@ -204,6 +223,11 @@ func TestSystemAlertsOneMin(t *testing.T) {
func TestSystemAlertsTwoMin(t *testing.T) {
testMultiMinuteSystemAlert(t, "CPU", 50, 2, setCPUAlertValue, 10, 51, 48)
for _, test := range cpuStateAlertTests {
t.Run(test.name, func(t *testing.T) {
testMultiMinuteSystemAlert(t, test.name, 50, 2, setCPUStateAlertValue, test.baseline, test.trigger, test.resolve)
})
}
testMultiMinuteSystemAlert(t, "Memory", 50, 2, setMemoryAlertValue, 10, 51, 48)
testMultiMinuteSystemAlert(t, "Disk", 50, 2, setDiskAlertValue, 10, 51, 48)
testMultiMinuteSystemAlert(t, "Bandwidth", 50, 2, setBandwidthAlertValue, [2]uint64{megabytesToBytes(10), megabytesToBytes(10)}, [2]uint64{megabytesToBytes(26), megabytesToBytes(25)}, [2]uint64{megabytesToBytes(10), megabytesToBytes(10)})
@@ -214,3 +238,17 @@ func TestSystemAlertsTwoMin(t *testing.T) {
testMultiMinuteSystemAlert(t, "LoadAvg15", 4, 2, setLoadAvgAlertValue, [3]float64{0, 0, 2}, [3]float64{0, 0, 4.1}, [3]float64{0, 0, 3.5})
testMultiMinuteSystemAlert(t, "Battery", 20, 2, setBatteryAlertValue, [2]uint8{21, 0}, [2]uint8{19, 0}, [2]uint8{25, 1})
}
func TestCPUStateAlertWithoutBreakdown(t *testing.T) {
fixture := newSystemAlertTestFixture(t, "CPUSteal", 1, 1)
defer fixture.cleanup()
synctest.Test(t, func(t *testing.T) {
submitValue(fixture, t, []float64(nil), setCPUStateAlertValue)
submitValue(fixture, t, []float64{0, 0, 0, 0, 0}, setCPUStateAlertValue)
waitForSystemAlert(time.Second)
fixture.assertTriggered(t, false, "Alert should ignore missing CPU breakdown data")
assert.Zero(t, fixture.hub.TestMailer.TotalSend(), "No email should be sent without CPU breakdown data")
})
}
+190
View File
@@ -0,0 +1,190 @@
package alerts
import (
"fmt"
"strings"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
// alertNameSystemdFailed is the alerts.name value for the failed systemd services alert.
const alertNameSystemdFailed = "SystemdFailed"
// maxListedServices caps how many service names are listed in a notification body.
const maxListedServices = 10
// HandleSystemdAlerts manages alerts for systemd services in the failed state.
//
// This is a binary state alert and fires on the first observation of a failed
// service rather than using a delay. The agent only refreshes systemd state every
// 10 minutes, so a shorter delay could never observe new data before expiring, and
// that poll interval already hides services that fail and restart quickly.
func (am *AlertManager) HandleSystemdAlerts(systemRecord *core.Record) error {
alerts := am.alertsCache.GetAlertsByName(systemRecord.Id, alertNameSystemdFailed)
if len(alerts) == 0 {
return nil
}
// State is read from the systemd_services snapshot rather than the update payload.
// The payload is not a reliable source here: realtime dashboard subscriptions fetch
// from the agent with a shorter cache time, and the agent omits systemd services from
// those responses, overwriting the cached payload roughly once a second while a system
// is being viewed. The snapshot table is only written by the full update cycle.
total, failed, err := am.queryServiceStates(systemRecord.Id)
if err != nil {
return err
}
if total == 0 {
// No rows normally means no systemd data for this system (agent without
// systemd, or not yet reported), which must not be treated as a recovery.
// Read info only in this ambiguous case. The record being saved is used
// instead of data because dashboard polling can replace the system's
// in-memory payload concurrently.
var currentInfo system.Info
if err := systemRecord.UnmarshalJSONField("info", &currentInfo); err != nil ||
len(currentInfo.Services) == 0 || currentInfo.Services[0] != 0 {
return nil
}
}
systemName := systemRecord.GetString("name")
for _, alertData := range alerts {
triggered := len(failed) > 0
// Only notify on a change of state, so a service that stays failed across
// cycles doesn't re-notify every update.
if triggered == alertData.Triggered {
continue
}
if err := am.sendSystemdAlert(triggered, systemName, alertData, failed); err != nil {
am.hub.Logger().Error("Failed to send alert", "err", err)
}
}
return nil
}
// queryServiceStates returns the number of services reported in the most recent update
// for a system, and the names of those in the failed state.
//
// Rows are restricted to the latest update because systemd_services is upserted, never
// pruned on change: a service that no longer exists on the host stops being reported and
// its row keeps its last known state until the retention sweep removes it. Every row
// written in one cycle shares a single updated timestamp, so the newest timestamp
// identifies exactly the services the agent last reported.
func (am *AlertManager) queryServiceStates(systemID string) (total int, failed []string, err error) {
var rows []struct {
Name string `db:"name"`
State systemd.ServiceState `db:"state"`
}
err = am.hub.DB().
Select("name", "state").
From("systemd_services").
Where(dbx.NewExp(
"system={:system} AND updated=(SELECT MAX(updated) FROM systemd_services WHERE system={:system})",
dbx.Params{"system": systemID},
)).
OrderBy("name").
All(&rows)
if err != nil {
return 0, nil, err
}
for _, row := range rows {
if row.State == systemd.StatusFailed {
failed = append(failed, row.Name)
}
}
return len(rows), failed, nil
}
// sendSystemdAlert sends a failed or recovered systemd services alert to the alert's user.
func (am *AlertManager) sendSystemdAlert(triggered bool, systemName string, alertData CachedAlertData, failed []string) error {
// Update trigger state for alert record before sending alert
if err := am.setAlertTriggered(alertData, triggered); err != nil {
return err
}
var title, message string
if triggered {
title = fmt.Sprintf("Failed services on %s %v", systemName, "\U0001F534") // Red alert emoji
message = fmt.Sprintf("%s on %s: %s", pluralizeServices(len(failed)), systemName, formatServiceList(failed))
} else {
title = fmt.Sprintf("Services recovered on %s %v", systemName, "✅") // Green checkmark emoji
message = fmt.Sprintf("No services are in the failed state on %s.", systemName)
}
systemID := alertData.SystemID
return am.SendAlert(AlertMessageData{
UserID: alertData.UserID,
SystemID: systemID,
Title: title,
Message: message,
Link: am.hub.MakeLink("system", systemID),
LinkText: "View " + systemName,
})
}
// pluralizeServices returns a count label like "1 failed service" or "3 failed services".
func pluralizeServices(count int) string {
if count == 1 {
return "1 failed service"
}
return fmt.Sprintf("%d failed services", count)
}
// formatServiceList joins service names, truncating long lists.
func formatServiceList(names []string) string {
if len(names) <= maxListedServices {
return strings.Join(names, ", ")
}
remaining := len(names) - maxListedServices
return fmt.Sprintf("%s and %d more", strings.Join(names[:maxListedServices], ", "), remaining)
}
// resolveSystemdAlerts resolves triggered systemd alerts for systems that no longer
// have any failed services. This clears stale state left by a hub restart.
func resolveSystemdAlerts(app core.App) error {
db := app.DB()
var alertIds []string
err := db.NewQuery(`
SELECT a.id
FROM alerts a
JOIN systems sys ON sys.id = a.system
WHERE a.name = {:name}
AND a.triggered = true
AND (
EXISTS (
SELECT 1 FROM systemd_services cur
WHERE cur.system = a.system
AND cur.updated = (SELECT MAX(updated) FROM systemd_services WHERE system = a.system)
)
OR json_extract(sys.info, '$.sv[0]') = 0
)
AND NOT EXISTS (
SELECT 1 FROM systemd_services s
WHERE s.system = a.system AND s.state = {:state}
AND s.updated = (SELECT MAX(updated) FROM systemd_services WHERE system = a.system)
)
`).Bind(dbx.Params{
"name": alertNameSystemdFailed,
"state": systemd.StatusFailed,
}).Column(&alertIds)
if err != nil {
return err
}
for _, alertId := range alertIds {
alert, err := app.FindRecordById("alerts", alertId)
if err != nil {
return err
}
alert.Set("triggered", false)
if err := app.Save(alert); err != nil {
return err
}
}
return nil
}
+383
View File
@@ -0,0 +1,383 @@
//go:build testing
package alerts_test
import (
"testing"
"time"
"github.com/henrygd/beszel/internal/alerts"
systemEntity "github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
// setSystemdServiceState upserts a systemd_services row mirroring the raw SQL write
// path used by the hub (createSystemdStatsRecords), which bypasses record hooks.
func setSystemdServiceState(t *testing.T, hub core.App, systemID, name string, state systemd.ServiceState, updated int64) {
t.Helper()
_, err := hub.DB().NewQuery(
"INSERT INTO systemd_services (id, system, name, state, sub, cpu, cpuPeak, memory, memPeak, updated) " +
"VALUES ({:id}, {:system}, {:name}, {:state}, 0, 0, 0, 0, 0, {:updated}) " +
"ON CONFLICT(id) DO UPDATE SET state = excluded.state, updated = excluded.updated",
).Bind(dbx.Params{
"id": systemID + "-" + name,
"system": systemID,
"name": name,
"state": state,
"updated": updated,
}).Execute()
require.NoError(t, err)
}
// seedServices writes a set of services into the systemd_services snapshot, which is the
// source HandleSystemdAlerts reads from. All rows share one updated timestamp, matching
// how the hub writes a batch in createSystemdStatsRecords.
func seedServices(t *testing.T, hub core.App, systemID string, states ...systemd.ServiceState) {
t.Helper()
seedServicesAt(t, hub, systemID, time.Now().UTC().UnixMilli(), states...)
}
// seedServicesAt writes services with an explicit batch timestamp.
func seedServicesAt(t *testing.T, hub core.App, systemID string, updated int64, states ...systemd.ServiceState) {
t.Helper()
for i, state := range states {
setSystemdServiceState(t, hub, systemID, serviceName(i), state, updated)
}
}
func serviceName(i int) string {
return string(rune('a'+i)) + ".service"
}
// systemdTestSetup creates a user with an email, a system, and a SystemdFailed alert.
func systemdTestSetup(t *testing.T, triggered bool) (*beszelTests.TestHub, *core.Record, *core.Record) {
t.Helper()
hub, user := beszelTests.GetHubWithUser(t)
userSettings, err := hub.FindFirstRecordByFilter("user_settings", "user={:user}", map[string]any{"user": user.Id})
require.NoError(t, err)
userSettings.Set("settings", `{"emails":["test@example.com"],"webhooks":[]}`)
require.NoError(t, hub.Save(userSettings))
// "paused" avoids spawning a background updater goroutine that would outlive
// the test hub; these tests drive HandleSystemdAlerts directly.
systems, err := beszelTests.CreateSystems(hub, 1, user.Id, "paused")
require.NoError(t, err)
system := systems[0]
alert, err := beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "SystemdFailed",
"system": system.Id,
"user": user.Id,
"triggered": triggered,
})
require.NoError(t, err)
return hub, system, alert
}
func TestSystemdAlertFiresImmediately(t *testing.T) {
hub, system, alert := systemdTestSetup(t, false)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
seedServices(t, hub, system.Id, systemd.StatusFailed, systemd.StatusActive)
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend(), "failed service should notify on first observation")
messages := hub.TestMailer.Messages()
require.NotEmpty(t, messages)
last := messages[len(messages)-1]
assert.Contains(t, last.Subject, "Failed services")
assert.Contains(t, last.Text, "a.service", "notification should name the failed service")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, alertRecord.GetBool("triggered"), "alert should be marked triggered")
// history record should be created via the alerts update hook
historyCount, err := hub.CountRecords("alerts_history", dbx.HashExp{"resolved": ""})
require.NoError(t, err)
assert.EqualValues(t, 1, historyCount, "should have one unresolved alert history record")
}
func TestSystemdAlertFullCycle(t *testing.T) {
hub, system, alert := systemdTestSetup(t, false)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
// Fail, then recover.
seedServices(t, hub, system.Id, systemd.StatusFailed)
require.NoError(t, am.HandleSystemdAlerts(system))
seedServices(t, hub, system.Id, systemd.StatusActive)
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount+2, hub.TestMailer.TotalSend(), "should send a failure and a recovery notification")
messages := hub.TestMailer.Messages()
require.Len(t, messages, 2)
assert.Contains(t, messages[0].Subject, "Failed services")
assert.Contains(t, messages[1].Subject, "Services recovered")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "alert should be cleared after recovery")
// history record should be resolved
historyCount, err := hub.CountRecords("alerts_history", dbx.HashExp{"resolved": ""})
require.NoError(t, err)
assert.Zero(t, historyCount, "alert history record should be resolved")
}
func TestSystemdAlertSendsRecoveryWhenTriggered(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
seedServices(t, hub, system.Id, systemd.StatusActive, systemd.StatusInactive)
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend(), "recovery notification should be sent")
messages := hub.TestMailer.Messages()
require.NotEmpty(t, messages)
assert.Contains(t, messages[len(messages)-1].Subject, "Services recovered")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "alert should be cleared after recovery")
}
func TestSystemdAlertDoesNotResendWhileTriggered(t *testing.T) {
hub, system, _ := systemdTestSetup(t, true)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
// Still failing across several cycles — should not re-notify.
for range 3 {
seedServices(t, hub, system.Id, systemd.StatusFailed)
require.NoError(t, am.HandleSystemdAlerts(system))
}
assert.Equal(t, initialEmailCount, hub.TestMailer.TotalSend(), "should not re-notify while still triggered")
}
func TestSystemdAlertRepeatedFailureNotifiesOnce(t *testing.T) {
hub, system, _ := systemdTestSetup(t, false)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
for range 3 {
seedServices(t, hub, system.Id, systemd.StatusFailed)
require.NoError(t, am.HandleSystemdAlerts(system))
}
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend(), "repeated failures should only notify once")
}
// A service that no longer exists on the host stops being reported, but its row stays
// in systemd_services with its last known state until the retention sweep. That stale
// row must not keep the alert triggered.
func TestSystemdAlertIgnoresServicesNoLongerReported(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
now := time.Now().UTC().UnixMilli()
// Older batch still holding a failed service that has since been removed.
setSystemdServiceState(t, hub, system.Id, "gone.service", systemd.StatusFailed, now-60_000)
// Current batch reports only healthy services.
seedServicesAt(t, hub, system.Id, now, systemd.StatusActive, systemd.StatusActive)
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend(), "stale failed row should not block recovery")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "alert should resolve once the service stops being reported")
}
func TestResolveSystemdAlertsIgnoresStaleFailedRows(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
now := time.Now().UTC().UnixMilli()
setSystemdServiceState(t, hub, system.Id, "gone.service", systemd.StatusFailed, now-60_000)
seedServicesAt(t, hub, system.Id, now, systemd.StatusActive)
require.NoError(t, alerts.ResolveSystemdAlerts(hub))
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "stale failed row should not keep the alert triggered")
}
func TestSystemdAlertNoSystemdDataIsIgnored(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
// A system with no systemd_services rows (agent without systemd, or nothing
// reported yet) must not be treated as a recovery.
require.NoError(t, am.HandleSystemdAlerts(system))
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount, hub.TestMailer.TotalSend(), "missing systemd data should not send a recovery")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, alertRecord.GetBool("triggered"), "triggered state should be preserved when data is absent")
}
func TestSystemdAlertFreshEmptySnapshotResolves(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
// An explicit zero service count on the saved system record distinguishes a
// confirmed empty snapshot from an agent response that omitted systemd data.
system.Set("info", systemEntity.Info{Services: []uint16{0, 0}})
require.NoError(t, am.HandleSystemAlerts(system, nil))
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend(), "fresh empty snapshot should send a recovery")
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "fresh empty snapshot should resolve the alert")
}
func TestSystemdAlertNoAlertRecord(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
systems, err := beszelTests.CreateSystems(hub, 1, user.Id, "paused")
require.NoError(t, err)
system := systems[0]
initialEmailCount := hub.TestMailer.TotalSend()
am := alerts.NewTestAlertManagerWithoutWorker(hub)
seedServices(t, hub, system.Id, systemd.StatusFailed)
require.NoError(t, am.HandleSystemdAlerts(system))
assert.Equal(t, initialEmailCount, hub.TestMailer.TotalSend(), "no email when no alert record exists")
}
func TestResolveSystemdAlertsClearsStaleTriggered(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
// No failed services in the snapshot, but the alert is still marked triggered
// (e.g. the hub restarted while the alert was active).
setSystemdServiceState(t, hub, system.Id, "a.service", systemd.StatusActive, time.Now().UTC().UnixMilli())
require.NoError(t, alerts.ResolveSystemdAlerts(hub))
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "stale triggered flag should be cleared")
}
func TestResolveSystemdAlertsKeepsTriggeredWithoutSystemdData(t *testing.T) {
hub, _, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
// Missing rows do not prove recovery. This can happen when a system is offline
// and its last service snapshot has been removed by retention.
require.NoError(t, alerts.ResolveSystemdAlerts(hub))
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, alertRecord.GetBool("triggered"), "missing systemd data should preserve triggered state")
}
func TestResolveSystemdAlertsClearsConfirmedEmptySnapshot(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
// Update the persisted snapshot directly so record hooks don't alter alert state
// before the startup resolver is exercised.
_, err := hub.DB().NewQuery(
"UPDATE systems SET info = {:info} WHERE id = {:id}",
).Bind(dbx.Params{"info": `{"sv":[0,0]}`, "id": system.Id}).Execute()
require.NoError(t, err)
require.NoError(t, alerts.ResolveSystemdAlerts(hub))
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alertRecord.GetBool("triggered"), "confirmed empty snapshot should clear triggered state")
}
func TestResolveSystemdAlertsKeepsStillFailing(t *testing.T) {
hub, system, alert := systemdTestSetup(t, true)
defer hub.Cleanup()
setSystemdServiceState(t, hub, system.Id, "a.service", systemd.StatusFailed, time.Now().UTC().UnixMilli())
require.NoError(t, alerts.ResolveSystemdAlerts(hub))
alertRecord, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, alertRecord.GetBool("triggered"), "alert should stay triggered while a service is still failed")
}
func TestSystemdAlertMultipleUsersRespectOwnAlerts(t *testing.T) {
hub, user1 := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
setStatusAlertEmail(t, hub, user1.Id, "user1@example.com")
user2, err := beszelTests.CreateUser(hub, "user2@example.com", "password")
require.NoError(t, err)
_, err = beszelTests.CreateRecord(hub, "user_settings", map[string]any{
"user": user2.Id,
"settings": map[string]any{
"emails": []string{"user2@example.com"},
"webhooks": []string{},
},
})
require.NoError(t, err)
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "shared-system",
"users": []string{user1.Id, user2.Id},
"host": "127.0.0.1",
})
require.NoError(t, err)
for _, user := range []*core.Record{user1, user2} {
_, err = beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "SystemdFailed",
"system": system.Id,
"user": user.Id,
})
require.NoError(t, err)
}
am := alerts.NewTestAlertManagerWithoutWorker(hub)
seedServices(t, hub, system.Id, systemd.StatusFailed)
require.NoError(t, am.HandleSystemdAlerts(system))
messages := hub.TestMailer.Messages()
require.Len(t, messages, 2, "each user should receive their own alert")
}
+9
View File
@@ -88,6 +88,10 @@ func ResolveStatusAlerts(app core.App) error {
return resolveStatusAlerts(app)
}
func ResolveSystemdAlerts(app core.App) error {
return resolveSystemdAlerts(app)
}
func (am *AlertManager) RestorePendingStatusAlerts() error {
return am.restorePendingStatusAlerts()
}
@@ -99,3 +103,8 @@ func (am *AlertManager) SetAlertTriggered(alert CachedAlertData, triggered bool)
func IsInternalURL(rawURL string) (bool, error) {
return isInternalURL(rawURL)
}
// BuildContainerLogExcerpt exposes buildContainerLogExcerpt for testing.
func BuildContainerLogExcerpt(raw string) string {
return buildContainerLogExcerpt(raw)
}
+142
View File
@@ -0,0 +1,142 @@
package alerts
import (
"fmt"
"time"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
// handleZfsPoolAlert sends alerts when a ZFS pool health state worsens and
// resolves the alert history entry when the pool recovers. Like the SMART
// hook, this is automatic and does not require user opt-in.
func (am *AlertManager) handleZfsPoolAlert(e *core.RecordEvent) error {
return am.handleZfsPoolHealthAlert(e, e.Record.Original().GetString("health"))
}
func (am *AlertManager) handleZfsPoolCreateAlert(e *core.RecordEvent) error {
return am.handleZfsPoolHealthAlert(e, "")
}
func (am *AlertManager) handleZfsPoolHealthAlert(e *core.RecordEvent, oldHealth string) error {
newHealth := e.Record.GetString("health")
oldSeverity := zfsPoolSeverity(oldHealth)
newSeverity := zfsPoolSeverity(newHealth)
systemID := e.Record.GetString("system")
if systemID == "" {
return e.Next()
}
systemRecord, err := e.App.FindRecordById("systems", systemID)
if err != nil {
e.App.Logger().Error("Failed to find system for ZFS alert", "err", err, "systemID", systemID)
return e.Next()
}
// Pool recovered to a healthy state: resolve any open history entries.
if newSeverity == 1 && oldSeverity > 1 {
resolveAllAlertHistoryRecords(e.App, e.Record.Id)
return e.Next()
}
if !shouldSendZfsPoolAlert(oldSeverity, newSeverity) {
return e.Next()
}
systemName := systemRecord.GetString("name")
poolName := e.Record.GetString("name")
title := fmt.Sprintf("ZFS pool %s on %s: %s", newHealth, systemName, poolName)
message := fmt.Sprintf("ZFS pool %s (%s) was first observed as %s", poolName, systemName, newHealth)
if oldSeverity > 0 {
message = fmt.Sprintf("ZFS pool %s (%s) health changed from %s to %s", poolName, systemName, oldHealth, newHealth)
}
userIDs := systemRecord.GetStringSlice("users")
if len(userIDs) == 0 {
return e.Next()
}
for _, userID := range userIDs {
if err := am.SendAlert(AlertMessageData{
UserID: userID,
SystemID: systemID,
Title: title,
Message: message,
Link: am.hub.MakeLink("system", systemID),
LinkText: "View " + systemName,
}); err != nil {
e.App.Logger().Error("Failed to send ZFS alert", "err", err, "userID", userID)
}
_ = createZfsPoolHistoryRecord(e.App, userID, systemID, e.Record.Id, poolName)
}
return e.Next()
}
// resolveZfsPoolHistoryOnDelete resolves open alert history entries when a
// pool record is deleted (manually or because the pool disappeared), so the
// UI does not keep showing an ongoing alert for a pool that no longer exists.
func resolveZfsPoolHistoryOnDelete(e *core.RecordEvent) error {
resolveAllAlertHistoryRecords(e.App, e.Record.Id)
return e.Next()
}
// shouldSendZfsPoolAlert reports whether a health transition warrants an alert.
// First observations of unhealthy pools and worsening transitions are reported.
func shouldSendZfsPoolAlert(oldSeverity, newSeverity int) bool {
return newSeverity > 1 && (oldSeverity == 0 || newSeverity > oldSeverity)
}
// zfsPoolSeverity ranks pool health states: healthy (1), degraded (2),
// failed/unavailable (3), unknown (0).
func zfsPoolSeverity(health string) int {
switch health {
case "ONLINE":
return 1
case "DEGRADED":
return 2
case "FAULTED", "OFFLINE", "UNAVAIL", "REMOVED", "SUSPENDED":
return 3
default:
return 0
}
}
// createZfsPoolHistoryRecord logs a pool health alert in the alerts history so
// it is visible in the UI without creating an editable alert configuration.
func createZfsPoolHistoryRecord(app core.App, userID, systemID, alertID, poolName string) error {
collection, err := app.FindCachedCollectionByNameOrId("alerts_history")
if err != nil {
return err
}
record := core.NewRecord(collection)
record.Set("user", userID)
record.Set("system", systemID)
record.Set("alert_id", alertID)
record.Set("name", "ZFS Pool: "+poolName)
return app.Save(record)
}
// resolveAllAlertHistoryRecords resolves every open history entry for an alert
// record id (one per system user).
func resolveAllAlertHistoryRecords(app core.App, alertID string) {
records, err := app.FindRecordsByFilter(
"alerts_history",
"alert_id={:alert_id} && resolved=null",
"", 0, 0,
dbx.Params{"alert_id": alertID},
)
if err != nil || len(records) == 0 {
return
}
now := time.Now().UTC()
for _, record := range records {
record.Set("resolved", now)
if err := app.Save(record); err != nil {
app.Logger().Error("Failed to resolve ZFS alert history", "err", err, "recordId", record.Id)
}
}
}
+145
View File
@@ -0,0 +1,145 @@
//go:build testing
package alerts_test
import (
"encoding/json"
"testing"
"time"
"github.com/henrygd/beszel/internal/entities/system"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/tools/types"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
// TestDiskAlertZfsPoolMultiMinute verifies that ZFS pool usage participates in
// the Disk threshold alert using historical per-minute values, mirroring the
// extra-filesystem behavior.
func TestDiskAlertZfsPoolMultiMinute(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
systems, err := beszelTests.CreateSystems(hub, 1, user.Id, "up")
require.NoError(t, err)
systemRecord := systems[0]
diskAlert, err := beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "Disk",
"system": systemRecord.Id,
"user": user.Id,
"value": 80, // threshold: 80%
"min": 2, // requires historical averaging
})
require.NoError(t, err)
am := hub.GetAlertManager()
now := time.Now().UTC()
poolHigh := map[string]*system.ZfsPool{
"tank": {Total: 1000, Used: 920}, // 92% - above threshold
}
recordTimes := []time.Duration{
-180 * time.Second,
-90 * time.Second,
-60 * time.Second,
-30 * time.Second,
}
for _, offset := range recordTimes {
stats := system.Stats{
DiskPct: 30, // root disk at 30% - below threshold
ZfsPools: poolHigh,
}
statsJSON, _ := json.Marshal(stats)
recordTime := now.Add(offset)
record, err := beszelTests.CreateRecord(hub, "system_stats", map[string]any{
"system": systemRecord.Id,
"type": "1m",
"stats": string(statsJSON),
})
require.NoError(t, err)
record.SetRaw("created", recordTime.Format(types.DefaultDateLayout))
err = hub.SaveNoValidate(record)
require.NoError(t, err)
}
combinedDataHigh := &system.CombinedData{
Stats: system.Stats{
DiskPct: 30,
ZfsPools: poolHigh,
},
Info: system.Info{
DiskPct: 30,
},
}
systemRecord.Set("updated", now)
err = hub.SaveNoValidate(systemRecord)
require.NoError(t, err)
err = am.HandleSystemAlerts(systemRecord, combinedDataHigh)
require.NoError(t, err)
time.Sleep(20 * time.Millisecond)
diskAlert, err = hub.FindFirstRecordByFilter("alerts", "id={:id}", dbx.Params{"id": diskAlert.Id})
require.NoError(t, err)
assert.True(t, diskAlert.GetBool("triggered"),
"Alert should be triggered when ZFS pool average (92%%) exceeds threshold (80%%)")
// --- Resolution: pool drops to 50%, alert should resolve ---
poolLow := map[string]*system.ZfsPool{
"tank": {Total: 1000, Used: 500}, // 50% - below threshold
}
newNow := now.Add(2 * time.Minute)
for _, offset := range recordTimes {
stats := system.Stats{
DiskPct: 30,
ZfsPools: poolLow,
}
statsJSON, _ := json.Marshal(stats)
recordTime := newNow.Add(offset)
record, err := beszelTests.CreateRecord(hub, "system_stats", map[string]any{
"system": systemRecord.Id,
"type": "1m",
"stats": string(statsJSON),
})
require.NoError(t, err)
record.SetRaw("created", recordTime.Format(types.DefaultDateLayout))
err = hub.SaveNoValidate(record)
require.NoError(t, err)
}
combinedDataLow := &system.CombinedData{
Stats: system.Stats{
DiskPct: 30,
ZfsPools: poolLow,
},
Info: system.Info{
DiskPct: 30,
},
}
systemRecord.Set("updated", newNow)
err = hub.SaveNoValidate(systemRecord)
require.NoError(t, err)
err = am.HandleSystemAlerts(systemRecord, combinedDataLow)
require.NoError(t, err)
time.Sleep(20 * time.Millisecond)
diskAlert, err = hub.FindFirstRecordByFilter("alerts", "id={:id}", dbx.Params{"id": diskAlert.Id})
require.NoError(t, err)
assert.False(t, diskAlert.GetBool("triggered"),
"Alert should be resolved when ZFS pool average (50%%) drops below threshold (80%%)")
}
+15
View File
@@ -0,0 +1,15 @@
//go:build testing
package alerts
import (
"testing"
"github.com/stretchr/testify/assert"
)
func TestZfsDiskAlertKeyIsNamespaced(t *testing.T) {
assert.Equal(t, "zfs:tank", zfsDiskAlertKey("tank"))
assert.Equal(t, "Usage of ZFS pool tank", diskAlertDescriptor(zfsDiskAlertKey("tank")))
assert.Equal(t, "Usage of tank", diskAlertDescriptor("tank"))
}
+292
View File
@@ -0,0 +1,292 @@
//go:build testing
package alerts_test
import (
"testing"
"time"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestZfsPoolAlertOnlineToDegraded(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "ONLINE",
})
assert.NoError(t, err)
// Re-fetch so PocketBase tracks original values
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "DEGRADED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 1, hub.TestMailer.TotalSend(), "should have 1 email sent after pool became DEGRADED")
lastMessage := hub.TestMailer.LastMessage()
assert.Contains(t, lastMessage.Subject, "ZFS pool DEGRADED on test-system")
assert.Contains(t, lastMessage.Subject, "tank")
assert.Contains(t, lastMessage.Text, "ONLINE to DEGRADED")
}
func TestZfsPoolAlertDegradedToFaulted(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "rpool",
"health": "DEGRADED",
})
assert.NoError(t, err)
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "FAULTED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 2, hub.TestMailer.TotalSend(), "should alert on initial DEGRADED state and later FAULTED transition")
lastMessage := hub.TestMailer.LastMessage()
assert.Contains(t, lastMessage.Subject, "ZFS pool FAULTED on test-system")
}
func TestZfsPoolAlertNoAlertOnRecovery(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "DEGRADED",
})
assert.NoError(t, err)
// Trigger a worsening alert first
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "FAULTED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 2, hub.TestMailer.TotalSend(), "expected alerts for initial DEGRADED state and DEGRADED -> FAULTED")
// Recovery back to ONLINE must not send a new alert
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "ONLINE")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 2, hub.TestMailer.TotalSend(), "recovery should not send a new alert")
// And the open history entry should have been resolved
history, err := hub.FindRecordsByFilter("alerts_history", "alert_id={:alert_id}", "", 0, 0, map[string]any{"alert_id": pool.Id})
assert.NoError(t, err)
requireHistoryResolved(t, history)
}
func TestZfsPoolAlertUnknownHealthDoesNotResolve(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
require.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "DEGRADED",
})
require.NoError(t, err)
time.Sleep(50 * time.Millisecond)
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
require.NoError(t, err)
pool.Set("health", "")
require.NoError(t, hub.Save(pool))
time.Sleep(50 * time.Millisecond)
history, err := hub.FindRecordsByFilter("alerts_history", "alert_id={:alert_id} && resolved=null", "", 0, 0, map[string]any{"alert_id": pool.Id})
require.NoError(t, err)
require.Len(t, history, 1, "unknown health must not resolve an active alert")
}
func TestZfsPoolAlertUnknownToFaulted(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "",
})
assert.NoError(t, err)
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "FAULTED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 1, hub.TestMailer.TotalSend(), "should alert when a previously unknown pool becomes FAULTED")
}
func TestZfsPoolAlertOnInitialUnhealthyState(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
require.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "DEGRADED",
})
require.NoError(t, err)
time.Sleep(50 * time.Millisecond)
require.EqualValues(t, 1, hub.TestMailer.TotalSend())
assert.Contains(t, hub.TestMailer.LastMessage().Text, "first observed as DEGRADED")
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
require.NoError(t, err)
require.NoError(t, hub.Save(pool))
time.Sleep(50 * time.Millisecond)
assert.EqualValues(t, 1, hub.TestMailer.TotalSend(), "unchanged unhealthy health must not duplicate alerts")
}
func TestZfsPoolAlertWritesHistory(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "ONLINE",
})
assert.NoError(t, err)
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "FAULTED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
history, err := hub.FindRecordsByFilter("alerts_history", "alert_id={:alert_id}", "", 0, 0, map[string]any{"alert_id": pool.Id})
assert.NoError(t, err)
require.Len(t, history, 1, "expected one history entry per user")
assert.Equal(t, "ZFS Pool: tank", history[0].GetString("name"))
assert.Equal(t, system.Id, history[0].GetString("system"))
}
func TestZfsPoolAlertResolvedOnRecordDelete(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"host": "127.0.0.1",
})
assert.NoError(t, err)
pool, err := beszelTests.CreateRecord(hub, "zfs_pools", map[string]any{
"system": system.Id,
"name": "tank",
"health": "ONLINE",
})
assert.NoError(t, err)
// Trigger an alert so an open history entry exists.
pool, err = hub.FindRecordById("zfs_pools", pool.Id)
assert.NoError(t, err)
pool.Set("health", "FAULTED")
err = hub.Save(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
history, err := hub.FindRecordsByFilter("alerts_history", "alert_id={:alert_id} && resolved=null", "", 0, 0, map[string]any{"alert_id": pool.Id})
assert.NoError(t, err)
require.Len(t, history, 1, "expected one open history entry")
// Deleting the pool record must resolve the open entry.
err = hub.Delete(pool)
assert.NoError(t, err)
time.Sleep(50 * time.Millisecond)
history, err = hub.FindRecordsByFilter("alerts_history", "alert_id={:alert_id}", "", 0, 0, map[string]any{"alert_id": pool.Id})
assert.NoError(t, err)
require.Len(t, history, 1)
requireHistoryResolved(t, history)
}
func requireHistoryResolved(t *testing.T, history []*core.Record) {
t.Helper()
for _, record := range history {
assert.False(t, record.GetDateTime("resolved").Time().IsZero(), "expected history entry to be resolved")
}
}
+6
View File
@@ -22,6 +22,8 @@ const (
GetSmartData
// Request detailed systemd service info from agent
GetSystemdInfo
// Request ZFS detail data from agent
GetZfsData
// Add new actions here...
)
@@ -64,6 +66,10 @@ type DataRequestOptions struct {
IncludeDetails bool `cbor:"1,keyasint"`
}
type ZfsDataRequest struct {
Force bool `cbor:"0,keyasint,omitempty"`
}
type ContainerLogsRequest struct {
ContainerID string `cbor:"0,keyasint"`
}
+1 -1
View File
@@ -23,7 +23,7 @@ COPY --from=builder /agent /agent
# AMD GPU name lookup (used by agent on Linux when /usr/share/libdrm/amdgpu.ids is read)
COPY --from=builder /app/agent/test-data/amdgpu.ids /usr/share/libdrm/amdgpu.ids
RUN apk add --no-cache smartmontools
RUN apk add --no-cache smartmontools zfs
# Ensure data persistence across container recreations
VOLUME ["/var/lib/beszel-agent"]
+33 -7
View File
@@ -33,8 +33,6 @@ type Stats struct {
MaxNetworkSent float64 `json:"nsm,omitempty" cbor:"-"`
MaxNetworkRecv float64 `json:"nrm,omitempty" cbor:"-"`
Temperatures map[string]float64 `json:"t,omitempty" cbor:"20,keyasint,omitempty"`
Fans map[string]uint16 `json:"f,omitempty" cbor:"36,keyasint,omitempty"`
Batteries map[string]uint8 `json:"bats,omitempty" cbor:"37,keyasint,omitempty"`
ExtraFs map[string]*FsStats `json:"efs,omitempty" cbor:"21,keyasint,omitempty"`
GPUData map[string]GPUData `json:"g,omitempty" cbor:"22,keyasint,omitempty"`
// LoadAvg1 float64 `json:"l1,omitempty" cbor:"23,keyasint,omitempty"`
@@ -44,7 +42,7 @@ type Stats struct {
MaxBandwidth [2]uint64 `json:"bm,omitzero" cbor:"-"` // [sent bytes, recv bytes]
// TODO: remove other load fields in future release in favor of load avg array
LoadAvg [3]float64 `json:"la,omitempty" cbor:"28,keyasint"`
Battery [2]uint8 `json:"bat,omitzero" cbor:"29,keyasint,omitzero"` // [percent, charge state]
Battery Battery `json:"bat,omitzero" cbor:"29,keyasint,omitzero"` // [percent, charge state]
NetworkInterfaces map[string][4]uint64 `json:"ni,omitempty" cbor:"31,keyasint,omitempty"` // [upload bytes, download bytes, total upload, total download]
DiskIO [2]uint64 `json:"dio,omitzero" cbor:"32,keyasint,omitzero"` // [read bytes, write bytes]
MaxDiskIO [2]uint64 `json:"diom,omitzero" cbor:"-"` // [max read bytes, max write bytes]
@@ -52,6 +50,20 @@ type Stats struct {
CpuCoresUsage Uint8Slice `json:"cpus,omitempty" cbor:"34,keyasint,omitempty"` // per-core busy usage [CPU0..]
DiskIoStats [6]float64 `json:"dios,omitzero" cbor:"35,keyasint,omitzero"` // [read time %, write time %, io utilization %, r_await ms, w_await ms, weighted io %]
MaxDiskIoStats [6]float64 `json:"diosm,omitzero" cbor:"-"` // max values for DiskIoStats
Fans map[string]uint16 `json:"f,omitempty" cbor:"36,keyasint,omitempty"`
Batteries map[string]uint8 `json:"bats,omitempty" cbor:"37,keyasint,omitempty"`
ZfsPools map[string]*ZfsPool `json:"z,omitempty" cbor:"39,keyasint,omitempty"` // ZFS pool metrics, keyed by pool name
DiskIOTotal [2]uint64 `json:"diot,omitzero" cbor:"38,keyasint,omitzero"` // [total read bytes, total write bytes] cumulative device counters
}
// ZfsPool holds per-pool ZFS metrics for a single collection interval.
type ZfsPool struct {
Total float64 `json:"d" cbor:"0,keyasint"` // total capacity in GiB
Used float64 `json:"du" cbor:"1,keyasint"` // allocated in GiB
ReadBytes uint64 `json:"rb,omitzero" cbor:"2,keyasint,omitzero"` // read throughput in bytes/s
WriteBytes uint64 `json:"wb,omitzero" cbor:"3,keyasint,omitzero"` // write throughput in bytes/s
Health string `json:"h,omitempty" cbor:"4,keyasint,omitempty"` // ONLINE, DEGRADED, FAULTED, ...
}
// Uint8Slice wraps []uint8 to customize JSON encoding while keeping CBOR efficient.
@@ -71,6 +83,15 @@ func (s Uint8Slice) MarshalJSON() ([]byte, error) {
return json.Marshal(arr)
}
// Battery stores the representative battery's percent and charge state.
// Its custom JSON encoding keeps the public and persisted representation as a
// numeric tuple under both encoding/json v1 and v2.
type Battery [2]uint8
func (b Battery) MarshalJSON() ([]byte, error) {
return json.Marshal([2]uint16{uint16(b[0]), uint16(b[1])})
}
type GPUData struct {
Name string `json:"n" cbor:"0,keyasint"`
Temperature float64 `json:"-"`
@@ -90,8 +111,8 @@ type FsStats struct {
Name string `json:"-"`
DiskTotal float64 `json:"d" cbor:"0,keyasint"`
DiskUsed float64 `json:"du" cbor:"1,keyasint"`
TotalRead uint64 `json:"-"`
TotalWrite uint64 `json:"-"`
TotalRead uint64 `json:"tr,omitzero" cbor:"9,keyasint,omitzero"` // cumulative device read bytes
TotalWrite uint64 `json:"tw,omitzero" cbor:"10,keyasint,omitzero"` // cumulative device write bytes
DiskReadPs float64 `json:"r" cbor:"2,keyasint"`
DiskWritePs float64 `json:"w" cbor:"3,keyasint"`
MaxDiskReadPS float64 `json:"rm,omitempty" cbor:"-"`
@@ -155,8 +176,9 @@ type Info struct {
LoadAvg [3]float64 `json:"la,omitempty" cbor:"19,keyasint"`
ConnectionType ConnectionType `json:"ct,omitempty" cbor:"20,keyasint,omitempty,omitzero"`
ExtraFsPct map[string]float64 `json:"efs,omitempty" cbor:"21,keyasint,omitempty"`
Services []uint16 `json:"sv,omitempty" cbor:"22,keyasint,omitempty"` // [totalServices, numFailedServices]
Battery [2]uint8 `json:"bat,omitzero" cbor:"23,keyasint,omitzero"` // [percent, charge state]
Services []uint16 `json:"sv,omitempty" cbor:"22,keyasint,omitempty"` // [totalServices, numFailedServices]
Battery Battery `json:"bat,omitzero" cbor:"23,keyasint,omitzero"` // [percent, charge state]
RootDiskName string `json:"rdn,omitempty" cbor:"24,keyasint,omitempty"` // custom name for root disk (set via FILESYSTEM=device__name)
}
// Data that does not change during process lifetime and is not needed in All Systems table
@@ -172,6 +194,7 @@ type Details struct {
Podman bool `cbor:"8,keyasint,omitempty"`
MemoryTotal uint64 `cbor:"9,keyasint"`
SmartInterval time.Duration `cbor:"10,keyasint,omitempty"`
ZfsInterval time.Duration `cbor:"11,keyasint,omitempty"` // interval for ZFS detail refresh
}
// Final data structure to return to the hub
@@ -181,4 +204,7 @@ type CombinedData struct {
Containers []*container.Stats `json:"container" cbor:"2,keyasint"`
SystemdServices []*systemd.Service `json:"systemd,omitempty" cbor:"3,keyasint,omitempty"`
Details *Details `cbor:"4,keyasint,omitempty"`
// SystemdServicesUpdated distinguishes a fresh empty snapshot from a response
// that omitted systemd data (for example, a short-cache dashboard request).
SystemdServicesUpdated bool `json:"systemdUpdated,omitempty" cbor:"5,keyasint,omitempty"`
}
+88 -6
View File
@@ -2,9 +2,11 @@ package system
import (
"encoding/json"
jsonv2 "encoding/json/v2"
"testing"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
@@ -12,12 +14,19 @@ import (
func TestStatsBatteryTransport(t *testing.T) {
stats := Stats{Battery: [2]uint8{0, 1}, Batteries: map[string]uint8{"Primary": 0, "Mouse": 75}}
jsonData, err := json.Marshal(stats)
require.NoError(t, err)
var jsonPayload map[string]any
require.NoError(t, json.Unmarshal(jsonData, &jsonPayload))
assert.Equal(t, []any{float64(0), float64(1)}, jsonPayload["bat"])
assert.Equal(t, map[string]any{"Primary": float64(0), "Mouse": float64(75)}, jsonPayload["bats"])
for name, marshal := range map[string]func(any) ([]byte, error){
"json_v1": json.Marshal,
"json_v2": func(value any) ([]byte, error) { return jsonv2.Marshal(value) },
} {
t.Run(name, func(t *testing.T) {
jsonData, err := marshal(stats)
require.NoError(t, err)
var jsonPayload map[string]any
require.NoError(t, json.Unmarshal(jsonData, &jsonPayload))
assert.Equal(t, []any{float64(0), float64(1)}, jsonPayload["bat"])
assert.Equal(t, map[string]any{"Primary": float64(0), "Mouse": float64(75)}, jsonPayload["bats"])
})
}
cborData, err := cbor.Marshal(stats)
require.NoError(t, err)
@@ -27,6 +36,26 @@ func TestStatsBatteryTransport(t *testing.T) {
assert.Equal(t, stats.Batteries, decoded.Batteries)
}
func TestStatsDiskIOTotalAndFansTransport(t *testing.T) {
stats := Stats{
DiskIOTotal: [2]uint64{437348527104, 331522465792},
Fans: map[string]uint16{"cpu": 1200},
}
cborData, err := cbor.Marshal(stats)
require.NoError(t, err)
var decoded Stats
require.NoError(t, cbor.Unmarshal(cborData, &decoded))
assert.Equal(t, stats.DiskIOTotal, decoded.DiskIOTotal)
assert.Equal(t, stats.Fans, decoded.Fans)
}
func TestStatsBatteryNumericArrayUnmarshal(t *testing.T) {
var stats Stats
require.NoError(t, json.Unmarshal([]byte(`{"bat":[50,4]}`), &stats))
assert.Equal(t, Battery{50, 4}, stats.Battery)
}
func TestStatsLegacyBatteryPayload(t *testing.T) {
data, err := json.Marshal(Stats{Battery: [2]uint8{50, 4}})
require.NoError(t, err)
@@ -35,3 +64,56 @@ func TestStatsLegacyBatteryPayload(t *testing.T) {
assert.Contains(t, payload, "bat")
assert.NotContains(t, payload, "bats")
}
func TestCombinedDataSystemdUpdateMarkerTransport(t *testing.T) {
data := CombinedData{SystemdServicesUpdated: true}
jsonData, err := json.Marshal(data)
require.NoError(t, err)
var decodedJSON CombinedData
require.NoError(t, json.Unmarshal(jsonData, &decodedJSON))
assert.True(t, decodedJSON.SystemdServicesUpdated)
assert.Empty(t, decodedJSON.SystemdServices)
cborData, err := cbor.Marshal(data)
require.NoError(t, err)
var decodedCBOR CombinedData
require.NoError(t, cbor.Unmarshal(cborData, &decodedCBOR))
assert.True(t, decodedCBOR.SystemdServicesUpdated)
assert.Empty(t, decodedCBOR.SystemdServices)
var legacy CombinedData
require.NoError(t, json.Unmarshal([]byte(`{"stats":{},"info":{},"container":[]}`), &legacy))
assert.False(t, legacy.SystemdServicesUpdated)
}
func TestCombinedDataContainerValidityTransport(t *testing.T) {
validEmpty := CombinedData{Containers: []*container.Stats{}}
jsonData, err := json.Marshal(validEmpty)
require.NoError(t, err)
var decodedJSON CombinedData
require.NoError(t, json.Unmarshal(jsonData, &decodedJSON))
assert.NotNil(t, decodedJSON.Containers)
assert.Empty(t, decodedJSON.Containers)
jsonV2Data, err := jsonv2.Marshal(validEmpty)
require.NoError(t, err)
var decodedJSONV2 CombinedData
require.NoError(t, jsonv2.Unmarshal(jsonV2Data, &decodedJSONV2))
assert.NotNil(t, decodedJSONV2.Containers)
assert.Empty(t, decodedJSONV2.Containers)
cborData, err := cbor.Marshal(validEmpty)
require.NoError(t, err)
var decodedCBOR CombinedData
require.NoError(t, cbor.Unmarshal(cborData, &decodedCBOR))
assert.NotNil(t, decodedCBOR.Containers)
assert.Empty(t, decodedCBOR.Containers)
invalidData, err := cbor.Marshal(CombinedData{})
require.NoError(t, err)
var decodedInvalid CombinedData
require.NoError(t, cbor.Unmarshal(invalidData, &decodedInvalid))
assert.Nil(t, decodedInvalid.Containers)
}
+45
View File
@@ -0,0 +1,45 @@
// Package zfs defines the ZFS detail data exchanged between agent and hub.
package zfs
// ZfsData is the detail payload returned by the agent for the GetZfsData action.
type ZfsData struct {
Pools []*PoolDetail `json:"pools,omitempty"`
Complete bool `json:"complete,omitempty"`
}
// PoolDetail holds the verbose state of a single pool: capacity, health,
// scrub, vdev, and dataset information.
type PoolDetail struct {
Name string `json:"name"`
Health string `json:"health,omitempty"`
Size uint64 `json:"size,omitempty"` // bytes
Alloc uint64 `json:"alloc,omitempty"` // bytes
Free uint64 `json:"free,omitempty"` // bytes
Scrub *Scrub `json:"scrub,omitempty"`
Vdevs []*Vdev `json:"vdevs,omitempty"`
Datasets []*Dataset `json:"datasets,omitempty"`
}
// Scrub holds the scrub (or resilver) status of a pool.
type Scrub struct {
State string `json:"state,omitempty"` // NONE, SCANNING, FINISHED, CANCELED
Progress string `json:"progress,omitempty"`
Errors uint64 `json:"errors,omitempty"`
}
// Vdev is a single vdev (mirror, raidz, or leaf disk) with error counters.
type Vdev struct {
Name string `json:"name"`
State string `json:"state,omitempty"`
ReadErrs uint64 `json:"readErrs,omitempty"`
WriteErrs uint64 `json:"writeErrs,omitempty"`
ChecksumErrs uint64 `json:"checksumErrs,omitempty"`
}
// Dataset is a single ZFS dataset with usage information.
type Dataset struct {
Name string `json:"name"`
Used uint64 `json:"used,omitempty"`
Avail uint64 `json:"avail,omitempty"`
Mountpoint string `json:"mount,omitempty"`
}
+22
View File
@@ -125,6 +125,8 @@ func (h *Hub) registerApiRoutes(se *core.ServeEvent) error {
apiAuth.DELETE("/user-alerts", alerts.DeleteUserAlerts)
// refresh SMART devices for a system
apiAuth.POST("/smart/refresh", h.refreshSmartData).BindFunc(excludeReadOnlyRole)
// refresh ZFS pool details for a system
apiAuth.POST("/zfs/refresh", h.refreshZfsData).BindFunc(excludeReadOnlyRole)
// get systemd service details
apiAuth.GET("/systemd/info", h.getSystemdInfo)
// /containers routes
@@ -389,3 +391,23 @@ func (h *Hub) refreshSmartData(e *core.RequestEvent) error {
return e.JSON(http.StatusOK, map[string]string{"status": "ok"})
}
// refreshZfsData handles POST /api/beszel/zfs/refresh requests
// Fetches fresh ZFS detail data from the agent and updates the collection
func (h *Hub) refreshZfsData(e *core.RequestEvent) error {
systemID := e.Request.URL.Query().Get("system")
if systemID == "" {
return e.BadRequestError("Invalid system parameter", nil)
}
system, err := h.sm.GetSystem(systemID)
if err != nil || !system.HasUser(e.App, e.Auth) {
return e.NotFoundError("", nil)
}
if err := system.FetchAndSaveZfsPools(true); err != nil {
return e.InternalServerError("", err)
}
return e.JSON(http.StatusOK, map[string]string{"status": "ok"})
}
+43 -1
View File
@@ -11,6 +11,7 @@ import (
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/henrygd/beszel/internal/migrations"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
pbTests "github.com/pocketbase/pocketbase/tests"
"github.com/stretchr/testify/require"
@@ -55,7 +56,7 @@ func TestApiRoutesAuthentication(t *testing.T) {
// Create test system
system, err := beszelTests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"users": []string{user.Id},
"users": []string{user.Id, readOnlyUser.Id},
"host": "127.0.0.1",
})
require.NoError(t, err, "Failed to create test system")
@@ -277,6 +278,24 @@ func TestApiRoutesAuthentication(t *testing.T) {
"systems": []string{system.Id},
}),
},
{
Name: "POST /user-alerts - readonly user can create own alert",
Method: http.MethodPost,
URL: "/api/beszel/user-alerts",
Headers: map[string]string{
"Authorization": readOnlyUserToken,
},
ExpectedStatus: 200,
ExpectedContent: []string{"\"success\":true"},
TestAppFactory: testAppFactory,
Body: jsonReader(map[string]any{
"name": "CPU", "value": 80, "min": 10, "systems": []string{system.Id},
}),
AfterTestFunc: func(t testing.TB, app *pbTests.TestApp, res *http.Response) {
alerts, _ := app.CountRecords("alerts", dbx.HashExp{"user": readOnlyUser.Id})
require.EqualValues(t, 1, alerts)
},
},
{
Name: "DELETE /user-alerts - no auth should fail",
Method: http.MethodDelete,
@@ -314,6 +333,29 @@ func TestApiRoutesAuthentication(t *testing.T) {
})
},
},
{
Name: "DELETE /user-alerts - readonly user can delete own alert",
Method: http.MethodDelete,
URL: "/api/beszel/user-alerts",
Headers: map[string]string{
"Authorization": readOnlyUserToken,
},
ExpectedStatus: 200,
ExpectedContent: []string{"\"count\":1", "\"success\":true"},
TestAppFactory: testAppFactory,
Body: jsonReader(map[string]any{
"name": "CPU", "systems": []string{system.Id},
}),
BeforeTestFunc: func(t testing.TB, app *pbTests.TestApp, e *core.ServeEvent) {
beszelTests.CreateRecord(app, "alerts", map[string]any{
"name": "CPU", "system": system.Id, "user": readOnlyUser.Id, "value": 80,
})
},
AfterTestFunc: func(t testing.TB, app *pbTests.TestApp, res *http.Response) {
alerts, _ := app.CountRecords("alerts", dbx.HashExp{"user": readOnlyUser.Id})
require.Zero(t, alerts)
},
},
{
Name: "GET /containers/logs - no auth should fail",
Method: http.MethodGet,
+6
View File
@@ -91,6 +91,12 @@ func setCollectionAuthSettings(app core.App) error {
}); err != nil {
return err
}
if err := applyCollectionRules(app, []string{"zfs_pools"}, collectionRules{
list: &systemScopedReadRule,
view: &systemScopedReadRule,
}); err != nil {
return err
}
if err := applyCollectionRules(app, []string{"fingerprints"}, collectionRules{
list: &systemScopedReadRule,
+38
View File
@@ -50,6 +50,13 @@ func TestCollectionRulesDefault(t *testing.T) {
assert.Equal(t, isUserMatchesUser, *alertsCollection.CreateRule)
assert.Equal(t, isUserMatchesUser, *alertsCollection.UpdateRule)
assert.Equal(t, isUserMatchesUser, *alertsCollection.DeleteRule)
alertNames := alertsCollection.Fields.GetByName("name").(*core.SelectField).Values
for _, name := range []string{"CPUIOWait", "CPUSteal"} {
assert.Contains(t, alertNames, name)
}
for _, name := range []string{"CPUSystem", "CPUUser", "CPUIdle", "CPUOther"} {
assert.NotContains(t, alertNames, name)
}
// alerts_history collection
alertsHistoryCollection, err := hub.FindCollectionByNameOrId("alerts_history")
@@ -357,6 +364,13 @@ func TestApiCollectionsAuthRules(t *testing.T) {
"host": "127.0.0.2",
})
userOneAlert, _ := beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "CPU", "system": userOneSystem.Id, "user": user1.Id, "value": 80,
})
userTwoAlert, _ := beszelTests.CreateRecord(hub, "alerts", map[string]any{
"name": "CPU", "system": userTwoSystem.Id, "user": user2.Id, "value": 80,
})
userRecords, _ := hub.CountRecords("users")
assert.EqualValues(t, 3, userRecords, "all users should be created")
@@ -368,6 +382,30 @@ func TestApiCollectionsAuthRules(t *testing.T) {
}
scenarios := []beszelTests.ApiScenario{
{
Name: "Users can only list their own alerts",
Method: http.MethodGet,
URL: "/api/collections/alerts/records",
Headers: map[string]string{
"Authorization": user1Token,
},
ExpectedStatus: 200,
ExpectedContent: []string{userOneAlert.Id},
NotExpectedContent: []string{userTwoAlert.Id},
TestAppFactory: testAppFactory,
},
{
Name: "Users cannot view another user's alert by id",
Method: http.MethodGet,
URL: fmt.Sprintf("/api/collections/alerts/records/%s", userTwoAlert.Id),
Headers: map[string]string{
"Authorization": user1Token,
},
ExpectedStatus: 403,
ExpectedContent: []string{"Only superusers"},
NotExpectedContent: []string{userTwoAlert.Id},
TestAppFactory: testAppFactory,
},
{
Name: "Unauthorized user cannot list systems",
Method: http.MethodGet,
+37
View File
@@ -4,11 +4,13 @@ package systems
import (
"errors"
"sync"
"testing"
"testing/synctest"
"time"
"github.com/stretchr/testify/assert"
"golang.org/x/crypto/ssh"
)
// TestRunWithTimeout covers the guard added for issue #2041: the per-system SSH
@@ -54,3 +56,38 @@ func TestRunWithTimeout(t *testing.T) {
})
})
}
// closedConn stands in for a connection whose peer has gone away: opening a
// channel fails rather than succeeding, which is what NewSession does on a
// client that closeSSHConnection has already closed.
type closedConn struct{ ssh.Conn }
func (closedConn) OpenChannel(string, []byte) (ssh.Channel, <-chan *ssh.Request, error) {
return nil, nil, errors.New("use of closed network connection")
}
func (closedConn) Close() error { return nil }
// TestCreateSessionDuringClose covers issue #2157: the background SMART fetch
// creates a session while the updater can be tearing the same connection down,
// so session creation must not read the client field after it is cleared.
func TestCreateSessionDuringClose(t *testing.T) {
for range 500 {
sys := &System{ctx: t.Context()}
sys.client.Store(&ssh.Client{Conn: closedConn{}})
var wg sync.WaitGroup
wg.Add(2)
go func() {
defer wg.Done()
session, err := sys.createSessionWithTimeout(time.Second)
assert.Nil(t, session)
assert.Error(t, err, "a closed connection must surface an error, not a session")
}()
go func() {
defer wg.Done()
sys.closeSSHConnection()
}()
wg.Wait()
}
}
+83 -32
View File
@@ -21,6 +21,7 @@ import (
"github.com/henrygd/beszel/internal/entities/smart"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
"github.com/henrygd/beszel/internal/entities/zfs"
"github.com/henrygd/beszel"
@@ -33,22 +34,24 @@ import (
)
type System struct {
Id string `db:"id"`
Host string `db:"host"`
Port string `db:"port"`
Status string `db:"status"`
manager *SystemManager // Manager that this system belongs to
client *ssh.Client // SSH client for fetching data
sshTransport *transport.SSHTransport // SSH transport for requests
data *system.CombinedData // system data from agent
ctx context.Context // Context for stopping the updater
cancel context.CancelFunc // Stops and removes system from updater
WsConn *ws.WsConn // Handler for agent WebSocket connection
agentVersion semver.Version // Agent version
updateTicker *time.Ticker // Ticker for updating the system
detailsFetched atomic.Bool // True if static system details have been fetched and saved
smartFetching atomic.Bool // True if SMART devices are currently being fetched
smartInterval time.Duration // Interval for periodic SMART data updates
Id string `db:"id"`
Host string `db:"host"`
Port string `db:"port"`
Status string `db:"status"`
manager *SystemManager // Manager that this system belongs to
client atomic.Pointer[ssh.Client] // SSH client for fetching data
sshTransport *transport.SSHTransport // SSH transport for requests
data *system.CombinedData // system data from agent
ctx context.Context // Context for stopping the updater
cancel context.CancelFunc // Stops and removes system from updater
WsConn *ws.WsConn // Handler for agent WebSocket connection
agentVersion semver.Version // Agent version
updateTicker *time.Ticker // Ticker for updating the system
detailsFetched atomic.Bool // True if static system details have been fetched and saved
smartFetching atomic.Bool // True if SMART devices are currently being fetched
smartInterval time.Duration // Interval for periodic SMART data updates
zfsFetching atomic.Bool // True if ZFS pools are currently being fetched
zfsInterval time.Duration // Interval for periodic ZFS detail data updates
}
func (sm *SystemManager) NewSystem(systemId string) *System {
@@ -154,6 +157,12 @@ func (sys *System) update() error {
// to prevent premature expiration leading to new fetch if interval is different.
sys.manager.smartFetchMap.UpdateExpiration(sys.Id, sys.smartInterval+time.Minute)
}
// update zfs interval if it's set on the agent side
if data.Details.ZfsInterval > 0 {
sys.zfsInterval = data.Details.ZfsInterval
sys.manager.hub.Logger().Info("ZFS interval updated from agent details", "system", sys.Id, "interval", sys.zfsInterval.String())
sys.manager.zfsFetchMap.UpdateExpiration(sys.Id, sys.zfsInterval+time.Minute)
}
}
// Fetch and save SMART devices when system first comes online or at intervals
@@ -170,6 +179,20 @@ func (sys *System) update() error {
}
}
// Fetch and save ZFS pool details when system first comes online or at intervals
if backgroundZfsFetchEnabled() && sys.detailsFetched.Load() && sys.supportsZfsData() {
if sys.zfsInterval <= 0 {
sys.zfsInterval = time.Hour
}
if sys.shouldFetchZfs() && sys.zfsFetching.CompareAndSwap(false, true) {
sys.manager.hub.Logger().Info("ZFS fetch", "system", sys.Id, "interval", sys.zfsInterval.String())
go func() {
defer sys.zfsFetching.Store(false)
_ = sys.FetchAndSaveZfsPools(false)
}()
}
}
return err
}
@@ -227,8 +250,10 @@ func (sys *System) createRecords(data *system.CombinedData) (*core.Record, error
}
}
// add new systemd_stats record
if len(data.SystemdServices) > 0 {
// Update systemd service records when the agent reports a fresh snapshot.
// The length check keeps snapshots from older agents working, while the
// explicit marker lets newer agents report that a fresh snapshot is empty.
if data.SystemdServicesUpdated || len(data.SystemdServices) > 0 {
if err := createSystemdStatsRecords(txApp, data.SystemdServices, sys.Id); err != nil {
return err
}
@@ -241,6 +266,10 @@ func (sys *System) createRecords(data *system.CombinedData) (*core.Record, error
}
}
if err := sys.syncZfsPoolHealth(txApp, data.Stats.ZfsPools); err != nil {
return err
}
// update system record (do this last because it triggers alerts and we need above records to be inserted first)
systemRecord.Set("status", up)
systemRecord.Set("info", data.Info)
@@ -280,7 +309,10 @@ func createSystemDetailsRecord(app core.App, data *system.Details, systemId stri
func createSystemdStatsRecords(app core.App, data []*systemd.Service, systemId string) error {
if len(data) == 0 {
return nil
_, err := app.DB().NewQuery(
"DELETE FROM systemd_services WHERE system = {:system}",
).Bind(dbx.Params{"system": systemId}).Execute()
return err
}
// shared params for all records
params := dbx.Params{
@@ -305,7 +337,16 @@ func createSystemdStatsRecords(app core.App, data []*systemd.Service, systemId s
"INSERT INTO systemd_services (id, system, name, state, sub, cpu, cpuPeak, memory, memPeak, updated) VALUES %s ON CONFLICT(id) DO UPDATE SET system = excluded.system, name = excluded.name, state = excluded.state, sub = excluded.sub, cpu = excluded.cpu, cpuPeak = excluded.cpuPeak, memory = excluded.memory, memPeak = excluded.memPeak, updated = excluded.updated",
strings.Join(valueStrings, ","),
)
_, err := app.DB().NewQuery(queryString).Bind(params).Execute()
if _, err := app.DB().NewQuery(queryString).Bind(params).Execute(); err != nil {
return err
}
// Remove services the agent no longer reports. Every row in this batch shares the
// same updated timestamp, so anything older no longer exists on the host. Left in
// place these rows survive until the retention sweep and surface inconsistently
// across the dashboard, the services table, and alerts.
_, err := app.DB().NewQuery(
"DELETE FROM systemd_services WHERE system = {:system} AND updated < {:updated}",
).Bind(dbx.Params{"system": systemId, "updated": params["updated"]}).Execute()
return err
}
@@ -434,7 +475,7 @@ func (sys *System) request(ctx context.Context, action common.WebSocketAction, r
err := sys.sshTransport.RequestWithRetry(ctx, action, req, dest, 1)
// Keep legacy SSH client/version fields in sync for other code paths.
if sys.sshTransport != nil {
sys.client = sys.sshTransport.GetClient()
sys.client.Store(sys.sshTransport.GetClient())
sys.agentVersion = sys.sshTransport.GetAgentVersion()
}
return err
@@ -476,8 +517,8 @@ func (sys *System) ensureSSHTransport() error {
})
}
// Sync client state with transport
if sys.client != nil {
sys.sshTransport.SetClient(sys.client)
if client := sys.client.Load(); client != nil {
sys.sshTransport.SetClient(client)
sys.sshTransport.SetAgentVersion(sys.agentVersion)
}
return nil
@@ -558,6 +599,15 @@ func (sys *System) FetchSmartDataFromAgent() (smart.SmartDataResponse, error) {
return result, err
}
// FetchZfsDataFromAgent fetches ZFS detail data from the agent.
func (sys *System) FetchZfsDataFromAgent(force bool) (*zfs.ZfsData, error) {
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
defer cancel()
var result zfs.ZfsData
err := sys.request(ctx, common.GetZfsData, common.ZfsDataRequest{Force: force}, &result)
return &result, err
}
func makeStableHashId(strings ...string) string {
hash := fnv.New32a()
for _, str := range strings {
@@ -625,7 +675,7 @@ func (sys *System) fetchDataViaSSH(options common.DataRequestOptions) (*system.C
// The operation can request a retry by returning true as the first return value.
func (sys *System) runSSHOperation(timeout time.Duration, retries int, operation func(*ssh.Session) (bool, error)) error {
for attempt := 0; attempt <= retries; attempt++ {
if sys.client == nil || sys.Status == down {
if sys.client.Load() == nil || sys.Status == down {
if err := sys.createSSHClient(); err != nil {
return err
}
@@ -721,13 +771,14 @@ func (s *System) createSSHClient() error {
} else {
host = net.JoinHostPort(host, s.Port)
}
var err error
s.client, err = dialSSHWithKeepAlive(network, host, s.manager.sshConfig)
client, err := dialSSHWithKeepAlive(network, host, s.manager.sshConfig)
s.client.Store(client)
if err != nil {
return err
}
s.agentVersion, _ = extractAgentVersion(string(s.client.Conn.ServerVersion()))
s.agentVersion, _ = extractAgentVersion(string(client.Conn.ServerVersion()))
s.manager.resetFailedSmartFetchState(s.Id)
s.manager.resetFailedZfsFetchState(s.Id)
return nil
}
@@ -762,7 +813,8 @@ func dialSSHWithKeepAlive(network, addr string, config *ssh.ClientConfig) (*ssh.
// createSessionWithTimeout creates a new SSH session with a timeout to avoid hanging
// in case of network issues
func (sys *System) createSessionWithTimeout(timeout time.Duration) (*ssh.Session, error) {
if sys.client == nil {
client := sys.client.Load()
if client == nil {
return nil, fmt.Errorf("client not initialized")
}
@@ -773,7 +825,7 @@ func (sys *System) createSessionWithTimeout(timeout time.Duration) (*ssh.Session
errChan := make(chan error, 1)
go func() {
if session, err := sys.client.NewSession(); err != nil {
if session, err := client.NewSession(); err != nil {
errChan <- err
} else {
sessionChan <- session
@@ -795,9 +847,8 @@ func (sys *System) closeSSHConnection() {
if sys.sshTransport != nil {
sys.sshTransport.Close()
}
if sys.client != nil {
sys.client.Close()
sys.client = nil
if client := sys.client.Swap(nil); client != nil {
client.Close()
}
}
+22 -1
View File
@@ -46,6 +46,7 @@ type SystemManager struct {
systems *store.Store[string, *System] // Thread-safe store of active systems
sshConfig *ssh.ClientConfig // SSH client configuration for system connections
smartFetchMap *expirymap.ExpiryMap[smartFetchState] // Stores last SMART fetch time/result; TTL is only for cleanup
zfsFetchMap *expirymap.ExpiryMap[zfsFetchState] // Stores last ZFS fetch time/result; TTL is only for cleanup
ctx context.Context // Cancelled when the app terminates
cancel context.CancelFunc // Cancels ctx and all child system contexts
}
@@ -57,7 +58,9 @@ type hubLike interface {
GetSSHKey(dataDir string) (ssh.Signer, error)
HandleSystemAlerts(systemRecord *core.Record, data *system.CombinedData) error
HandleStatusAlerts(status string, systemRecord *core.Record) error
HandleContainerAlerts(systemRecord *core.Record, data *system.CombinedData, fetchLogs func(containerID string) (string, error)) error
CancelPendingStatusAlerts(systemID string)
CancelPendingContainerAlerts(systemID string)
}
// NewSystemManager creates a new SystemManager instance with the provided hub.
@@ -67,6 +70,7 @@ func NewSystemManager(hub hubLike) *SystemManager {
systems: store.New(map[string]*System{}),
hub: hub,
smartFetchMap: expirymap.New[smartFetchState](time.Hour),
zfsFetchMap: expirymap.New[zfsFetchState](time.Hour),
}
sm.ctx, sm.cancel = context.WithCancel(context.Background())
return sm
@@ -187,7 +191,7 @@ func (sm *SystemManager) onRecordUpdate(e *core.RecordEvent) error {
// - paused: Closes SSH connection and deactivates alerts
// - pending: Starts monitoring (reuses WebSocket if available)
// - up: Triggers system alerts
// - down: Triggers status change alerts
// - down: Cancels pending container alerts and triggers status change alerts
func (sm *SystemManager) onRecordAfterUpdateSuccess(e *core.RecordEvent) error {
newStatus := e.Record.GetString("status")
prevStatus := pending
@@ -205,6 +209,7 @@ func (sm *SystemManager) onRecordAfterUpdateSuccess(e *core.RecordEvent) error {
}
_ = deactivateAlerts(e.App, e.Record.Id)
sm.hub.CancelPendingStatusAlerts(e.Record.Id)
sm.hub.CancelPendingContainerAlerts(e.Record.Id)
return e.Next()
case pending:
// Resume monitoring, preferring existing WebSocket connection
@@ -218,6 +223,10 @@ func (sm *SystemManager) onRecordAfterUpdateSuccess(e *core.RecordEvent) error {
}
_ = deactivateAlerts(e.App, e.Record.Id)
return e.Next()
case down:
// Docker state is unknown while the system is unreachable. Do not let a
// delayed container-health alert fire from the last received snapshot.
sm.hub.CancelPendingContainerAlerts(e.Record.Id)
}
// Handle systems not in manager
@@ -230,6 +239,9 @@ func (sm *SystemManager) onRecordAfterUpdateSuccess(e *core.RecordEvent) error {
if err := sm.hub.HandleSystemAlerts(e.Record, system.data); err != nil {
e.App.Logger().Error("Error handling system alerts", "err", err)
}
if err := sm.hub.HandleContainerAlerts(e.Record, system.data, system.FetchContainerLogsFromAgent); err != nil {
e.App.Logger().Error("Error handling container alerts", "err", err)
}
}
// Trigger status change alerts for up/down transitions
@@ -343,6 +355,15 @@ func (sm *SystemManager) resetFailedSmartFetchState(systemID string) {
}
}
// resetFailedZfsFetchState clears only failed ZFS cooldown entries so a fresh
// agent reconnect retries ZFS discovery immediately after configuration changes.
func (sm *SystemManager) resetFailedZfsFetchState(systemID string) {
state, ok := sm.zfsFetchMap.GetOk(systemID)
if ok && !state.Successful {
sm.zfsFetchMap.Remove(systemID)
}
}
// createSSHClientConfig initializes the SSH client configuration for connecting to an agent's server
func (sm *SystemManager) createSSHClientConfig() error {
privateKey, err := sm.hub.GetSSHKey("")
+5 -1
View File
@@ -29,7 +29,11 @@ func (stubHub) HandleSystemAlerts(systemRecord *core.Record, data *esystem.Combi
return nil
}
func (stubHub) HandleStatusAlerts(status string, systemRecord *core.Record) error { return nil }
func (stubHub) CancelPendingStatusAlerts(systemID string) {}
func (stubHub) HandleContainerAlerts(systemRecord *core.Record, data *esystem.CombinedData, fetchLogs func(containerID string) (string, error)) error {
return nil
}
func (stubHub) CancelPendingStatusAlerts(systemID string) {}
func (stubHub) CancelPendingContainerAlerts(systemID string) {}
// newTestSystemWithHub creates a System backed by a real (temp) database, along
// with a matching "systems" record, for tests that need to exercise DB reads/writes.
+193
View File
@@ -0,0 +1,193 @@
package systems
import (
"database/sql"
"errors"
"fmt"
"time"
"github.com/henrygd/beszel"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/zfs"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
var errIncompleteZfsData = errors.New("incomplete ZFS pool inventory")
type zfsFetchState struct {
LastAttempt int64
Successful bool
}
func (sys *System) supportsZfsData() bool {
return sys.agentVersion.GTE(beszel.MinVersionZfsData)
}
// FetchAndSaveZfsPools fetches ZFS detail data from the agent and saves it to
// the database. force bypasses the agent's detail cache for manual refreshes.
func (sys *System) FetchAndSaveZfsPools(force bool) error {
zfsData, err := sys.FetchZfsDataFromAgent(force)
if err != nil {
sys.recordZfsFetchResult(err, 0)
return err
}
if zfsData == nil || !zfsData.Complete {
err = errIncompleteZfsData
sys.recordZfsFetchResult(err, 0)
return err
}
err = sys.saveZfsPools(zfsData)
sys.recordZfsFetchResult(err, len(zfsData.Pools))
return err
}
// recordZfsFetchResult stores a cooldown entry for the ZFS interval and marks
// whether the last fetch produced any pools, so failed setup can retry on reconnect.
func (sys *System) recordZfsFetchResult(err error, poolCount int) {
if sys.manager == nil {
return
}
interval := sys.zfsFetchInterval()
success := err == nil && poolCount > 0
if sys.manager.hub != nil {
sys.manager.hub.Logger().Info("ZFS fetch result", "system", sys.Id, "success", success, "pools", poolCount, "interval", interval.String(), "err", err)
}
sys.manager.zfsFetchMap.Set(sys.Id, zfsFetchState{LastAttempt: time.Now().UnixMilli(), Successful: success}, interval+time.Minute)
}
// shouldFetchZfs returns true when there is no active ZFS cooldown entry for this system.
func (sys *System) shouldFetchZfs() bool {
if sys.manager == nil {
return true
}
state, ok := sys.manager.zfsFetchMap.GetOk(sys.Id)
if !ok {
return true
}
return !time.UnixMilli(state.LastAttempt).Add(sys.zfsFetchInterval()).After(time.Now())
}
// zfsFetchInterval returns the agent-provided ZFS interval or the default when unset.
func (sys *System) zfsFetchInterval() time.Duration {
if sys.zfsInterval > 0 {
return sys.zfsInterval
}
return time.Hour
}
// saveZfsPools saves ZFS pool detail data to the zfs_pools collection and
// removes records for pools no longer reported by a complete agent inventory.
func (sys *System) saveZfsPools(zfsData *zfs.ZfsData) error {
if zfsData == nil || !zfsData.Complete {
return errIncompleteZfsData
}
hub := sys.manager.hub
collection, err := hub.FindCachedCollectionByNameOrId("zfs_pools")
if err != nil {
return err
}
return hub.RunInTransaction(func(txApp core.App) error {
alive := make(map[string]bool, len(zfsData.Pools))
for _, pool := range zfsData.Pools {
if pool == nil {
continue
}
alive[pool.Name] = true
if err := sys.upsertZfsPoolRecord(txApp, collection, pool); err != nil {
return err
}
}
existing, err := txApp.FindRecordsByFilter(
collection,
"system={:system}",
"", 0, 0,
dbx.Params{"system": sys.Id},
)
if err != nil {
return err
}
for _, record := range existing {
if !alive[record.GetString("name")] {
if err := txApp.Delete(record); err != nil {
return err
}
}
}
return nil
})
}
func (sys *System) upsertZfsPoolRecord(app core.App, collection *core.Collection, pool *zfs.PoolDetail) error {
recordID := makeStableHashId(sys.Id, pool.Name)
record, err := app.FindRecordById(collection, recordID)
if err != nil {
if !errors.Is(err, sql.ErrNoRows) {
return err
}
record = core.NewRecord(collection)
record.Set("id", recordID)
}
record.Set("system", sys.Id)
record.Set("name", pool.Name)
record.Set("health", pool.Health)
record.Set("size", pool.Size)
record.Set("alloc", pool.Alloc)
record.Set("free", pool.Free)
record.Set("scrub", pool.Scrub)
record.Set("vdevs", pool.Vdevs)
record.Set("datasets", pool.Datasets)
record.Set("details_updated", time.Now().UTC())
return app.SaveNoValidate(record)
}
// syncZfsPoolHealth persists newly discovered pools and health transitions from
// regular system samples. Detailed fields remain owned by the hourly refresh.
func (sys *System) syncZfsPoolHealth(app core.App, pools map[string]*system.ZfsPool) error {
if len(pools) == 0 {
return nil
}
collection, err := app.FindCachedCollectionByNameOrId("zfs_pools")
if err != nil {
return err
}
const gib = 1024 * 1024 * 1024
for name, pool := range pools {
if pool == nil {
continue
}
recordID := makeStableHashId(sys.Id, name)
record, err := app.FindRecordById(collection, recordID)
if err != nil {
if !errors.Is(err, sql.ErrNoRows) {
return err
}
record = core.NewRecord(collection)
record.Set("id", recordID)
record.Set("system", sys.Id)
record.Set("name", name)
record.Set("health", pool.Health)
record.Set("size", uint64(pool.Total*gib))
record.Set("alloc", uint64(pool.Used*gib))
record.Set("free", uint64(max(pool.Total-pool.Used, 0)*gib))
if err := app.SaveNoValidate(record); err != nil {
return fmt.Errorf("creating ZFS pool summary %q: %w", name, err)
}
continue
}
if record.GetString("health") == pool.Health {
continue
}
record.Set("health", pool.Health)
if err := app.SaveNoValidate(record); err != nil {
return fmt.Errorf("updating ZFS pool health %q: %w", name, err)
}
}
return nil
}
+153
View File
@@ -0,0 +1,153 @@
//go:build testing
package systems
import (
"errors"
"testing"
"time"
"github.com/blang/semver"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/zfs"
"github.com/henrygd/beszel/internal/hub/expirymap"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestSupportsZfsData(t *testing.T) {
sys := &System{agentVersion: semver.MustParse("0.18.8")}
assert.False(t, sys.supportsZfsData())
sys.agentVersion = semver.MustParse("0.18.9")
assert.True(t, sys.supportsZfsData())
}
func TestRecordZfsFetchResult(t *testing.T) {
sm := &SystemManager{zfsFetchMap: expirymap.New[zfsFetchState](time.Hour)}
t.Cleanup(sm.zfsFetchMap.StopCleaner)
sys := &System{
Id: "system-1",
manager: sm,
zfsInterval: time.Hour,
}
// Successful fetch with pools
sys.recordZfsFetchResult(nil, 2)
state, ok := sm.zfsFetchMap.GetOk(sys.Id)
assert.True(t, ok, "expected zfs fetch result to be stored")
assert.True(t, state.Successful, "expected successful fetch state to be recorded")
// Failed fetch
sys.recordZfsFetchResult(errors.New("failed"), 0)
state, ok = sm.zfsFetchMap.GetOk(sys.Id)
assert.True(t, ok, "expected failed zfs fetch state to be stored")
assert.False(t, state.Successful, "expected failed zfs fetch state to be marked unsuccessful")
// Successful fetch but no pools
sys.recordZfsFetchResult(nil, 0)
state, ok = sm.zfsFetchMap.GetOk(sys.Id)
assert.True(t, ok, "expected fetch with zero pools to be stored")
assert.False(t, state.Successful, "expected fetch with zero pools to be marked unsuccessful")
}
func TestShouldFetchZfs(t *testing.T) {
sm := &SystemManager{zfsFetchMap: expirymap.New[zfsFetchState](time.Hour)}
t.Cleanup(sm.zfsFetchMap.StopCleaner)
sys := &System{
Id: "system-1",
manager: sm,
zfsInterval: time.Hour,
}
assert.True(t, sys.shouldFetchZfs(), "expected initial zfs fetch to be allowed")
sys.recordZfsFetchResult(errors.New("failed"), 0)
assert.False(t, sys.shouldFetchZfs(), "expected zfs fetch to be blocked while interval entry exists")
sm.zfsFetchMap.Remove(sys.Id)
assert.True(t, sys.shouldFetchZfs(), "expected zfs fetch to be allowed after interval entry is cleared")
}
func TestZfsFetchIntervalDefault(t *testing.T) {
sys := &System{}
assert.Equal(t, time.Hour, sys.zfsFetchInterval())
sys.zfsInterval = 5 * time.Minute
assert.Equal(t, 5*time.Minute, sys.zfsFetchInterval())
}
func TestResetFailedZfsFetchState(t *testing.T) {
sm := &SystemManager{zfsFetchMap: expirymap.New[zfsFetchState](time.Hour)}
t.Cleanup(sm.zfsFetchMap.StopCleaner)
sm.zfsFetchMap.Set("system-1", zfsFetchState{LastAttempt: time.Now().UnixMilli(), Successful: false}, time.Hour)
sm.resetFailedZfsFetchState("system-1")
_, ok := sm.zfsFetchMap.GetOk("system-1")
assert.False(t, ok, "expected failed zfs fetch state to be cleared on reconnect")
sm.zfsFetchMap.Set("system-1", zfsFetchState{LastAttempt: time.Now().UnixMilli(), Successful: true}, time.Hour)
sm.resetFailedZfsFetchState("system-1")
_, ok = sm.zfsFetchMap.GetOk("system-1")
assert.True(t, ok, "expected successful zfs fetch state to be preserved")
}
func TestSaveZfsPoolsCompleteEmptyPrunesFinalPool(t *testing.T) {
sys, app := newTestSystemWithHub(t)
require.NoError(t, sys.saveZfsPools(&zfs.ZfsData{
Complete: true,
Pools: []*zfs.PoolDetail{{Name: "tank", Health: "ONLINE"}},
}))
records, err := app.FindRecordsByFilter("zfs_pools", "system={:system}", "", 0, 0, map[string]any{"system": sys.Id})
require.NoError(t, err)
require.Len(t, records, 1)
assert.False(t, records[0].GetDateTime("details_updated").Time().IsZero())
require.NoError(t, sys.saveZfsPools(&zfs.ZfsData{Complete: true}))
records, err = app.FindRecordsByFilter("zfs_pools", "system={:system}", "", 0, 0, map[string]any{"system": sys.Id})
require.NoError(t, err)
assert.Empty(t, records)
}
func TestSaveZfsPoolsIncompletePreservesRecords(t *testing.T) {
sys, app := newTestSystemWithHub(t)
require.NoError(t, sys.saveZfsPools(&zfs.ZfsData{
Complete: true,
Pools: []*zfs.PoolDetail{{Name: "tank", Health: "ONLINE"}},
}))
assert.ErrorIs(t, sys.saveZfsPools(&zfs.ZfsData{}), errIncompleteZfsData)
records, err := app.FindRecordsByFilter("zfs_pools", "system={:system}", "", 0, 0, map[string]any{"system": sys.Id})
require.NoError(t, err)
assert.Len(t, records, 1)
}
func TestSyncZfsPoolHealthWritesOnlyTransitions(t *testing.T) {
sys, app := newTestSystemWithHub(t)
collection, err := app.FindCachedCollectionByNameOrId("zfs_pools")
require.NoError(t, err)
require.NoError(t, sys.syncZfsPoolHealth(app, map[string]*system.ZfsPool{
"tank": {Total: 100, Used: 25, Health: "ONLINE"},
}))
record, err := app.FindRecordById(collection, makeStableHashId(sys.Id, "tank"))
require.NoError(t, err)
firstUpdated := record.GetDateTime("updated")
assert.Equal(t, "ONLINE", record.GetString("health"))
assert.EqualValues(t, 100*1024*1024*1024, record.GetInt("size"))
require.NoError(t, sys.syncZfsPoolHealth(app, map[string]*system.ZfsPool{
"tank": {Total: 100, Used: 30, Health: "ONLINE"},
}))
record, err = app.FindRecordById(collection, record.Id)
require.NoError(t, err)
assert.Equal(t, firstUpdated, record.GetDateTime("updated"))
require.NoError(t, sys.syncZfsPoolHealth(app, map[string]*system.ZfsPool{
"tank": {Total: 100, Used: 30, Health: "DEGRADED"},
}))
record, err = app.FindRecordById(collection, record.Id)
require.NoError(t, err)
assert.Equal(t, "DEGRADED", record.GetString("health"))
}
@@ -0,0 +1,126 @@
//go:build testing
package systems_test
import (
"testing"
"time"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
"github.com/henrygd/beszel/internal/hub/systems"
"github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestCreateRecordsHandlesSystemdAlertLifecycle(t *testing.T) {
hub, user := tests.GetHubWithUser(t)
defer hub.Cleanup()
settings, err := hub.FindFirstRecordByFilter("user_settings", "user={:user}", dbx.Params{"user": user.Id})
require.NoError(t, err)
settings.Set("settings", `{"emails":["test@example.com"],"webhooks":[]}`)
require.NoError(t, hub.Save(settings))
systemRecords, err := tests.CreateSystems(hub, 1, user.Id, "paused")
require.NoError(t, err)
systemRecord := systemRecords[0]
alert, err := tests.CreateRecord(hub, "alerts", map[string]any{
"name": "SystemdFailed",
"system": systemRecord.Id,
"user": user.Id,
})
require.NoError(t, err)
monitoredSystem, err := hub.GetSystemManager().GetSystem(systemRecord.Id)
require.NoError(t, err)
initialEmailCount := hub.TestMailer.TotalSend()
// Exercise the production path: persist the snapshot transactionally, save the
// system record, and let its update hook evaluate and deliver the alert.
_, err = monitoredSystem.CreateRecords(&system.CombinedData{
Info: system.Info{Services: []uint16{1, 1}},
SystemdServicesUpdated: true,
SystemdServices: []*systemd.Service{
{Name: "failed.service", State: systemd.StatusFailed},
},
})
require.NoError(t, err)
alert, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, alert.GetBool("triggered"))
assert.Equal(t, initialEmailCount+1, hub.TestMailer.TotalSend())
serviceCount, err := hub.CountRecords("systemd_services", dbx.HashExp{"system": systemRecord.Id})
require.NoError(t, err)
assert.EqualValues(t, 1, serviceCount)
unresolvedCount, err := hub.CountRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id, "resolved": ""})
require.NoError(t, err)
assert.EqualValues(t, 1, unresolvedCount)
// A fresh empty snapshot must delete the old failed row and resolve the alert.
_, err = monitoredSystem.CreateRecords(&system.CombinedData{
Info: system.Info{Services: []uint16{0, 0}},
SystemdServicesUpdated: true,
})
require.NoError(t, err)
alert, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alert.GetBool("triggered"))
assert.Equal(t, initialEmailCount+2, hub.TestMailer.TotalSend())
serviceCount, err = hub.CountRecords("systemd_services", dbx.HashExp{"system": systemRecord.Id})
require.NoError(t, err)
assert.Zero(t, serviceCount)
unresolvedCount, err = hub.CountRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id, "resolved": ""})
require.NoError(t, err)
assert.Zero(t, unresolvedCount)
}
// createSystemdStatsRecords upserts the reported services and must drop rows for
// services the agent has stopped reporting, so a unit removed from the host doesn't
// linger with its last known state until the retention sweep.
func TestCreateSystemdStatsRecordsRemovesUnreportedServices(t *testing.T) {
hub, err := tests.NewTestHub(t.TempDir())
require.NoError(t, err)
defer hub.Cleanup()
user, err := tests.CreateUser(hub, "test@example.com", "password")
require.NoError(t, err)
system, err := tests.CreateRecord(hub, "systems", map[string]any{
"name": "test-system",
"host": "127.0.0.1",
"users": []string{user.Id},
})
require.NoError(t, err)
serviceNames := func() []string {
var out []string
require.NoError(t, hub.DB().Select("name").From("systemd_services").
Where(dbx.NewExp("system={:s}", dbx.Params{"s": system.Id})).
OrderBy("name").Column(&out))
return out
}
require.NoError(t, systems.CreateSystemdStatsRecords(hub, []*systemd.Service{
{Name: "a.service", State: systemd.StatusActive},
{Name: "gone.service", State: systemd.StatusFailed},
}, system.Id))
assert.Equal(t, []string{"a.service", "gone.service"}, serviceNames())
// Batches are stamped with millisecond precision and update cycles are a minute
// apart in practice; ensure the next batch gets a distinct timestamp.
time.Sleep(2 * time.Millisecond)
// gone.service is no longer reported, so its row must not survive.
require.NoError(t, systems.CreateSystemdStatsRecords(hub, []*systemd.Service{
{Name: "a.service", State: systemd.StatusActive},
}, system.Id))
assert.Equal(t, []string{"a.service"}, serviceNames())
// A fresh empty snapshot means the agent no longer reports any services.
require.NoError(t, systems.CreateSystemdStatsRecords(hub, nil, system.Id))
assert.Empty(t, serviceNames())
}
@@ -7,3 +7,6 @@ package systems
// The hub integration tests create/replace systems and clean up the test apps quickly.
// Background SMART fetching can outlive teardown and crash in PocketBase internals (nil DB).
func backgroundSmartFetchEnabled() bool { return true }
// Background ZFS fetching follows the same policy as SMART fetching.
func backgroundZfsFetchEnabled() bool { return true }
@@ -7,6 +7,7 @@ import (
"fmt"
entities "github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
"github.com/pocketbase/pocketbase/core"
)
@@ -17,6 +18,9 @@ import (
// the automatic background fetch during tests.
func backgroundSmartFetchEnabled() bool { return false }
// Background ZFS fetching follows the same policy as SMART fetching.
func backgroundZfsFetchEnabled() bool { return false }
// TESTING ONLY: GetSystemCount returns the number of systems in the store
func (sm *SystemManager) GetSystemCount() int {
return sm.systems.Length()
@@ -115,6 +119,7 @@ func (sm *SystemManager) RemoveAllSystems() {
sm.RemoveSystem(system.Id)
}
sm.smartFetchMap.StopCleaner()
sm.zfsFetchMap.StopCleaner()
}
// ResetContextForTesting replaces the manager context for a new synctest bubble.
@@ -130,3 +135,7 @@ func (s *System) CreateRecords(data *entities.CombinedData) (*core.Record, error
s.data = data
return s.createRecords(data)
}
func CreateSystemdStatsRecords(app core.App, data []*systemd.Service, systemId string) error {
return createSystemdStatsRecords(app, data, systemId)
}
@@ -79,7 +79,11 @@ func init() {
"LoadAvg1",
"LoadAvg5",
"LoadAvg15",
"Battery"
"Battery",
"ContainerHealth",
"SystemdFailed",
"CPUIOWait",
"CPUSteal"
]
},
{
@@ -115,6 +119,17 @@ func init() {
"system": false,
"type": "bool"
},
{
"hidden": true,
"id": "date1302749137",
"max": "",
"min": "",
"name": "pending_since",
"presentable": false,
"required": false,
"system": false,
"type": "date"
},
{
"hidden": false,
"id": "autodate2990389176",
@@ -1699,7 +1714,165 @@ func init() {
"type": "base",
"updateRule": null,
"viewRule": null
}
},
{
"createRule": null,
"deleteRule": null,
"fields": [
{
"autogeneratePattern": "[a-z0-9]{15}",
"hidden": false,
"id": "text3208210256",
"max": 15,
"min": 15,
"name": "id",
"pattern": "^[a-z0-9]+$",
"presentable": false,
"primaryKey": true,
"required": true,
"system": true,
"type": "text"
},
{
"cascadeDelete": true,
"collectionId": "2hz5ncl8tizk5nx",
"hidden": false,
"id": "relation1204987316",
"maxSelect": 1,
"minSelect": 0,
"name": "system",
"presentable": false,
"required": true,
"system": false,
"type": "relation"
},
{
"autogeneratePattern": "",
"hidden": false,
"id": "text7739291048",
"max": 0,
"min": 0,
"name": "name",
"pattern": "",
"presentable": false,
"primaryKey": false,
"required": false,
"system": false,
"type": "text"
},
{
"autogeneratePattern": "",
"hidden": false,
"id": "text5528164482",
"max": 0,
"min": 0,
"name": "health",
"pattern": "",
"presentable": false,
"primaryKey": false,
"required": false,
"system": false,
"type": "text"
},
{
"hidden": false,
"id": "number8862034195",
"max": null,
"min": null,
"name": "size",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"hidden": false,
"id": "number4418907321",
"max": null,
"min": null,
"name": "alloc",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"hidden": false,
"id": "number2904183765",
"max": null,
"min": null,
"name": "free",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"hidden": false,
"id": "json4466109723",
"maxSize": 0,
"name": "scrub",
"presentable": false,
"required": false,
"system": false,
"type": "json"
},
{
"hidden": false,
"id": "json9012873456",
"maxSize": 0,
"name": "vdevs",
"presentable": false,
"required": false,
"system": false,
"type": "json"
},
{
"hidden": false,
"id": "json7182045639",
"maxSize": 0,
"name": "datasets",
"presentable": false,
"required": false,
"system": false,
"type": "json"
},
{
"hidden": false,
"id": "date9274163058",
"max": "",
"min": "",
"name": "details_updated",
"presentable": false,
"required": false,
"system": false,
"type": "date"
},
{
"hidden": false,
"id": "autodate3332085495",
"name": "updated",
"onCreate": true,
"onUpdate": true,
"presentable": false,
"system": false,
"type": "autodate"
}
],
"id": "pbc_8441057391",
"indexes": [
"CREATE INDEX ` + "`" + `idx_zfsPoolsSystem` + "`" + ` ON ` + "`" + `zfs_pools` + "`" + ` (` + "`" + `system` + "`" + `)"
],
"listRule": null,
"name": "zfs_pools",
"system": false,
"type": "base",
"updateRule": null,
"viewRule": null
}
]`
err := app.ImportCollectionsByMarshaledJSON([]byte(jsonData), false)
+39
View File
@@ -127,6 +127,7 @@ func (rm *RecordManager) CreateLongerRecords() {
"created": shorterRecordPeriod,
},
)).
OrderBy("created").
All(&recordIds)
// continue if not enough shorter records
@@ -196,6 +197,7 @@ func AverageSystemStatsSlice(records []system.Stats) system.Stats {
tempCount := float64(0)
var fanSums map[string]uint64
fanCount := uint64(0)
zfsPoolCounts := make(map[string]uint64)
// Accumulate totals
for i := range records {
@@ -266,6 +268,8 @@ func AverageSystemStatsSlice(records []system.Stats) system.Stats {
sum.MaxBandwidth[1] = max(sum.MaxBandwidth[1], stats.MaxBandwidth[1], stats.Bandwidth[1])
sum.MaxDiskIO[0] = max(sum.MaxDiskIO[0], stats.MaxDiskIO[0], stats.DiskIO[0])
sum.MaxDiskIO[1] = max(sum.MaxDiskIO[1], stats.MaxDiskIO[1], stats.DiskIO[1])
sum.DiskIOTotal[0] = max(sum.DiskIOTotal[0], stats.DiskIOTotal[0])
sum.DiskIOTotal[1] = max(sum.DiskIOTotal[1], stats.DiskIOTotal[1])
for i := range stats.DiskIoStats {
sum.MaxDiskIoStats[i] = max(sum.MaxDiskIoStats[i], stats.MaxDiskIoStats[i], stats.DiskIoStats[i])
}
@@ -325,6 +329,8 @@ func AverageSystemStatsSlice(records []system.Stats) system.Stats {
fs.DiskWriteBytes += value.DiskWriteBytes
fs.MaxDiskReadBytes = max(fs.MaxDiskReadBytes, value.MaxDiskReadBytes, value.DiskReadBytes)
fs.MaxDiskWriteBytes = max(fs.MaxDiskWriteBytes, value.MaxDiskWriteBytes, value.DiskWriteBytes)
fs.TotalRead = max(fs.TotalRead, value.TotalRead)
fs.TotalWrite = max(fs.TotalWrite, value.TotalWrite)
for i := range value.DiskIoStats {
fs.DiskIoStats[i] += value.DiskIoStats[i]
fs.MaxDiskIoStats[i] = max(fs.MaxDiskIoStats[i], value.MaxDiskIoStats[i], value.DiskIoStats[i])
@@ -332,6 +338,31 @@ func AverageSystemStatsSlice(records []system.Stats) system.Stats {
}
}
// Accumulate ZFS pool stats. Counts are tracked per entry so a pool
// missing from some samples is not averaged as zero.
if stats.ZfsPools != nil {
if sum.ZfsPools == nil {
sum.ZfsPools = make(map[string]*system.ZfsPool, len(stats.ZfsPools))
}
for name, value := range stats.ZfsPools {
if value == nil {
continue
}
pool := sum.ZfsPools[name]
if pool == nil {
pool = &system.ZfsPool{}
sum.ZfsPools[name] = pool
}
pool.Total += value.Total
pool.Used += value.Used
pool.ReadBytes += value.ReadBytes
pool.WriteBytes += value.WriteBytes
if value.Health != "" {
pool.Health = value.Health
}
zfsPoolCounts[name]++
}
}
// Accumulate GPU data
if stats.GPUData != nil {
if sum.GPUData == nil {
@@ -442,6 +473,14 @@ func AverageSystemStatsSlice(records []system.Stats) system.Stats {
}
}
// Average ZFS pool stats.
for name, pool := range sum.ZfsPools {
entryCount := zfsPoolCounts[name]
pool.Total = twoDecimals(pool.Total / float64(entryCount))
pool.Used = twoDecimals(pool.Used / float64(entryCount))
pool.ReadBytes /= entryCount
pool.WriteBytes /= entryCount
}
// Average GPU data
if sum.GPUData != nil {
for id := range sum.GPUData {
+28 -1
View File
@@ -620,7 +620,7 @@ func TestAverageSystemStatsSlice_ZeroRepresentativeBattery(t *testing.T) {
{Battery: [2]uint8{0, 1}, Batteries: map[string]uint8{"Primary": 0}},
{},
})
assert.Equal(t, [2]uint8{0, 1}, result.Battery)
assert.Equal(t, system.Battery{0, 1}, result.Battery)
assert.Equal(t, map[string]uint8{"Primary": 0}, result.Batteries)
}
@@ -669,6 +669,33 @@ func TestAverageSystemStatsSlice_MixedOptionalFields(t *testing.T) {
assert.Equal(t, 20.0, result.GPUData["gpu0"].Usage)
}
func TestAverageSystemStatsSlice_Zfs(t *testing.T) {
input := []system.Stats{
{
ZfsPools: map[string]*system.ZfsPool{
"tank": {Total: 100, Used: 40, ReadBytes: 100, WriteBytes: 200, Health: "ONLINE"},
},
},
{},
{
ZfsPools: map[string]*system.ZfsPool{
"tank": {Total: 120, Used: 60, ReadBytes: 300, WriteBytes: 400, Health: "DEGRADED"},
"backup": {Total: 50, Used: 10, ReadBytes: 25, WriteBytes: 50, Health: "ONLINE"},
},
},
}
result := records.AverageSystemStatsSlice(input)
require.Len(t, result.ZfsPools, 2)
assert.Equal(t, &system.ZfsPool{
Total: 110, Used: 50, ReadBytes: 200, WriteBytes: 300, Health: "DEGRADED",
}, result.ZfsPools["tank"])
assert.Equal(t, &system.ZfsPool{
Total: 50, Used: 10, ReadBytes: 25, WriteBytes: 50, Health: "ONLINE",
}, result.ZfsPools["backup"])
}
// Tests with 10 records matching the common real-world case (10 x 1m -> 1 x 10m).
func TestAverageSystemStatsSlice_TenRecords(t *testing.T) {
input := make([]system.Stats, 10)
+2
View File
@@ -8,6 +8,7 @@ export default defineConfig({
"cs",
"da",
"de",
"el",
"es",
"fa",
"fr",
@@ -28,6 +29,7 @@ export default defineConfig({
"sr",
"sv",
"uk",
"uz",
"vi",
"zh",
"zh-CN",
+2 -2
View File
@@ -1,7 +1,7 @@
{
"name": "beszel",
"private": true,
"version": "0.18.8",
"version": "0.19.0",
"type": "module",
"scripts": {
"dev": "vite --host",
@@ -74,4 +74,4 @@
"optionalDependencies": {
"@esbuild/linux-arm64": "^0.21.5"
}
}
}
@@ -22,7 +22,7 @@ export const ActiveAlerts = () => {
for (const alert of alerts[systemId].values()) {
if (alert.triggered && alert.name in alertInfo) {
activeAlerts.push(alert)
alertsKey.push(`${alert.system}${alert.value}${alert.min}`)
alertsKey.push(`${alert.id}${alert.value}${alert.min}`)
}
}
}
@@ -59,7 +59,9 @@ export const ActiveAlerts = () => {
{systems[alert.system]?.name} {info.name()}
</AlertTitle>
<AlertDescription>
{alert.name === "Status" ? (
{info.triggeredDesc ? (
info.triggeredDesc()
) : alert.name === "Status" ? (
<Trans>Connection is down</Trans>
) : info.invert ? (
<Trans>
@@ -60,11 +60,15 @@ export const alertsHistoryColumns: ColumnDef<AlertsHistoryRecord>[] = [
),
cell({ row, getValue }) {
const name = row.original.name
const info = alertInfo[name]
if (info?.triggeredDesc) {
return <span className="ps-2">{info.triggeredDesc()}</span>
}
if (name === "Status") {
return <span className="ps-2">{t`Down`}</span>
}
const value = getValue() as number
const unit = alertInfo[name]?.unit
const unit = info?.unit
return (
<span className="tabular-nums ps-2.5">
{toFixedFloat(value, value < 10 ? 2 : 1)}
@@ -236,10 +236,16 @@ export function AlertContent({
const { name } = alertData
const singleDescription = alertData.singleDesc?.()
/** Alerts that fire on first observation have no duration to configure */
const noDuration = alertData.noDuration === true
/** Binary alerts have no threshold to configure */
const noThreshold = !!singleDescription || noDuration
/** Whether enabling the alert reveals anything to configure */
const hasControls = !(noThreshold && noDuration)
const [checked, setChecked] = useState(global ? false : !!alert)
const [min, setMin] = useState(alert?.min || 10)
const [value, setValue] = useState(alert?.value || (singleDescription ? 0 : (alertData.start ?? 80)))
const [min, setMin] = useState(alert?.min || (noDuration ? 0 : 10))
const [value, setValue] = useState(alert?.value || (noThreshold ? 0 : (alertData.start ?? 80)))
const Icon = alertData.icon
@@ -277,14 +283,16 @@ export function AlertContent({
<label
htmlFor={`s${name}`}
className={cn("flex flex-row items-center justify-between gap-4 cursor-pointer p-4", {
"pb-0": checked,
"pb-0": checked && hasControls,
})}
>
<div className="grid gap-1 select-none">
<p className="font-semibold flex gap-3 items-center">
<Icon className="h-4 w-4 opacity-85" /> {alertData.name()}
</p>
{!checked && <span className="block text-sm text-muted-foreground">{alertData.desc()}</span>}
{(!checked || !hasControls) && (
<span className="block text-sm text-muted-foreground">{alertData.desc()}</span>
)}
</div>
<Switch
id={`s${name}`}
@@ -307,10 +315,10 @@ export function AlertContent({
}}
/>
</label>
{checked && (
{checked && hasControls && (
<div className="grid sm:grid-cols-2 mt-1.5 gap-5 px-4 pb-5 tabular-nums text-muted-foreground">
<Suspense fallback={<div className="h-10" />}>
{!singleDescription && (
{!noThreshold && (
<div>
<p id={`v${name}`} className="text-sm block h-6">
{alertData.invert ? (
@@ -361,46 +369,49 @@ export function AlertContent({
</div>
</div>
)}
<div className={cn(singleDescription && "col-span-full lowercase")}>
<p id={`t${name}`} className="text-sm block h-6 first-letter:uppercase">
{singleDescription && (
<>
{singleDescription}
{` `}
</>
)}
<Trans>
For <strong className="text-foreground">{min}</strong>{" "}
<Plural value={min} one="minute" other="minutes" />
</Trans>
</p>
<div className="flex gap-3 items-center">
<Slider
aria-labelledby={`t${name}`}
value={[min]}
onValueCommit={(val) => sendUpsert(val[0], value)}
onValueChange={(val) => setMin(val[0])}
min={1}
max={60}
/>
<Input
type="number"
value={min}
onChange={(e) => {
let val = parseInt(e.target.value, 10)
if (!Number.isNaN(val)) {
val = Math.max(1, Math.min(val, 60))
setMin(val)
sendUpsert(val, value)
}
}}
min={1}
max={60}
className="w-16 h-8 text-center px-1"
/>
{!noDuration && (
<div className={cn(singleDescription && "col-span-full lowercase")}>
<p id={`t${name}`} className="text-sm block h-6 first-letter:uppercase">
{singleDescription && (
<>
{singleDescription}
{` `}
</>
)}
<Trans>
For <strong className="text-foreground">{min}</strong>{" "}
<Plural value={min} one="minute" other="minutes" />
</Trans>
</p>
<div className="flex gap-3 items-center">
<Slider
aria-labelledby={`t${name}`}
value={[min]}
onValueCommit={(val) => sendUpsert(val[0], value)}
onValueChange={(val) => setMin(val[0])}
min={1}
max={60}
/>
<Input
type="number"
value={min}
onChange={(e) => {
let val = parseInt(e.target.value, 10)
if (!Number.isNaN(val)) {
val = Math.max(1, Math.min(val, 60))
setMin(val)
sendUpsert(val, value)
}
}}
min={1}
max={60}
className="w-16 h-8 text-center px-1"
/>
</div>
</div>
</div>
)}
</Suspense>
{checked && alertData.note && <span className="block col-span-full text-sm text-muted-foreground -mt-3">{alertData.note()}</span>}
</div>
)}
</div>
+19 -14
View File
@@ -8,10 +8,11 @@ import { useSystemData } from "./system/use-system-data"
import { CpuChart, ContainerCpuChart } from "./system/charts/cpu-charts"
import { MemoryChart, ContainerMemoryChart, SwapChart } from "./system/charts/memory-charts"
import { RootDiskCharts, ExtraFsCharts } from "./system/charts/disk-charts"
import { ZfsCharts } from "./system/charts/zfs-charts"
import { BandwidthChart, ContainerNetworkChart } from "./system/charts/network-charts"
import { TemperatureChart, FanChart, BatteryChart } from "./system/charts/sensor-charts"
import { GpuPowerChart, GpuDetailCharts } from "./system/charts/gpu-charts"
import { LazyContainersTable, LazySmartTable, LazySystemdTable } from "./system/lazy-tables"
import { GpuPowerChart, GpuCharts } from "./system/charts/gpu-charts"
import { LazyContainersTable, LazySmartTable, LazySystemdTable, LazyZfsTable } from "./system/lazy-tables"
import { LoadAverageChart } from "./system/charts/load-average-chart"
import { ContainerIcon, CpuIcon, HardDriveIcon, TerminalSquareIcon } from "lucide-react"
import { GpuIcon } from "../ui/icons"
@@ -63,6 +64,7 @@ export default memo(function SystemDetail({ id }: { id: string }) {
const hasContainersTable = hasContainers && compareSemVer(chartData.agentVersion, SEMVER_0_14_0) >= 0
const hasSystemd = system.info.sv
const hasGpu = hasGpuData || hasGpuPowerData
const hasZfs = Object.keys(systemStats.at(-1)?.stats?.z ?? {}).length > 0
// keep tabsRef in sync for keyboard navigation
const tabs = ["core", "disk"]
@@ -131,7 +133,7 @@ export default memo(function SystemDetail({ id }: { id: string }) {
</div>
{hasGpuData && lastGpus && (
<GpuDetailCharts
<GpuCharts
chartData={chartData}
grid={grid}
dataEmpty={dataEmpty}
@@ -142,6 +144,10 @@ export default memo(function SystemDetail({ id }: { id: string }) {
<ExtraFsCharts systemData={systemData} />
{hasZfs && <ZfsCharts systemData={systemData} />}
{hasZfs && <LazyZfsTable systemId={system.id} />}
{maybeHasSmartData && <LazySmartTable systemId={system.id} />}
{hasContainersTable && <LazyContainersTable systemId={system.id} />}
@@ -204,6 +210,8 @@ export default memo(function SystemDetail({ id }: { id: string }) {
<RootDiskCharts systemData={systemData} />
</div>
<ExtraFsCharts systemData={systemData} />
{hasZfs && <ZfsCharts systemData={systemData} />}
{hasZfs && <LazyZfsTable systemId={system.id} />}
{maybeHasSmartData && <LazySmartTable systemId={system.id} />}
</>
)}
@@ -211,18 +219,15 @@ export default memo(function SystemDetail({ id }: { id: string }) {
{hasGpu && (
<TabsContent value="gpu" forceMount className={activeTab === "gpu" ? "contents" : "hidden"}>
<div className="grid xl:grid-cols-2 gap-4">
<GpuCharts
chartData={chartData}
grid={grid}
dataEmpty={dataEmpty}
lastGpus={(lastGpus ?? {}) as Record<string, GPUData>}
hasGpuEnginesData={hasGpuEnginesData}
>
{hasGpuPowerData && <GpuPowerChart chartData={chartData} grid={grid} dataEmpty={dataEmpty} />}
</div>
{hasGpuData && lastGpus && (
<GpuDetailCharts
chartData={chartData}
grid={grid}
dataEmpty={dataEmpty}
lastGpus={lastGpus as Record<string, GPUData>}
hasGpuEnginesData={hasGpuEnginesData}
/>
)}
</GpuCharts>
</TabsContent>
)}
@@ -57,6 +57,17 @@ export const diskDataFns = {
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.efs?.[name]?.wbm ?? (stats?.efs?.[name]?.wm ?? 0) * 1024 * 1024,
// cumulative totals
totalRead: ({ stats }: SystemStatsRecord) => stats?.diot?.[0] ?? 0,
totalWrite: ({ stats }: SystemStatsRecord) => stats?.diot?.[1] ?? 0,
extraTotalRead:
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.efs?.[name]?.tr ?? 0,
extraTotalWrite:
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.efs?.[name]?.tw ?? 0,
// read/write time
readTime: dios(0),
readTimeMax: diosMax(0),
@@ -114,8 +125,10 @@ export function DiskUsageChart({ systemData, extraFsName }: { systemData: System
diskSize = Math.round(diskSize)
}
const title = extraFsName ? `${extraFsName} ${t`Usage`}` : t`Disk Usage`
const description = extraFsName ? t`Disk usage of ${extraFsName}` : t`Usage of root partition`
const rootName = systemData.system?.info?.rdn
const rootLabel = rootName ?? t({ message: `Root`, context: "Root disk label" })
const title = extraFsName ? `${extraFsName} ${t`Usage`}` : `${rootLabel} ${t`Usage`}`
const description = t`Disk usage of ${{extraFsName: extraFsName ?? rootLabel.toLowerCase()}}`
return (
<ChartCard empty={dataEmpty} grid={grid} title={title} description={description}>
@@ -152,8 +165,10 @@ export function DiskIOChart({ systemData, extraFsName }: { systemData: SystemDat
return null
}
const title = extraFsName ? `${extraFsName} I/O` : t`Disk I/O`
const description = extraFsName ? t`Throughput of ${extraFsName}` : t`Throughput of root filesystem`
const rootName = systemData.system?.info?.rdn
const rootLabel = rootName ?? t({ message: `Root`, context: "Root disk label" })
const title = t`${{diskName: extraFsName ?? rootLabel}} I/O`
const description = t`Throughput of ${{extraFsName: extraFsName ?? rootLabel.toLowerCase()}}`
const hasMoreIOMetrics = chartData.systemStats?.some((record) => record.stats?.dios?.at(0))
@@ -264,7 +279,7 @@ export function ExtraFsCharts({ systemData }: { systemData: SystemData }) {
return (
<div className="grid xl:grid-cols-2 gap-4">
{Object.keys(extraFs).map((extraFsName) => {
{Object.keys(extraFs).sort((a, b) => a.localeCompare(b)).map((extraFsName) => {
let diskSize = systemStats.at(-1)?.stats.efs?.[extraFsName].d ?? NaN
// round to nearest GB
if (diskSize >= 100) {
@@ -1,5 +1,5 @@
import { t } from "@lingui/core/macro"
import { useRef, useMemo } from "react"
import { Fragment, type ReactNode, useRef, useMemo } from "react"
import AreaChartDefault, { type DataPoint } from "@/components/charts/area-chart"
import LineChartDefault from "@/components/charts/line-chart"
import { Unit } from "@/lib/enums"
@@ -79,6 +79,7 @@ export function GpuPowerChart({
return (
<ChartCard
className={cn(grid && "!col-span-1")}
empty={dataEmpty}
grid={grid}
title={t`GPU Power Draw`}
@@ -96,22 +97,26 @@ export function GpuPowerChart({
)
}
/** GPU detail grid (engines + per-GPU usage/VRAM) — rendered outside the main 2-col grid */
export function GpuDetailCharts({
/** All GPU charts (optional power-draw slot + engines + per-GPU usage/VRAM) in a single 2-col grid, so the
* cards' odd:last-of-type parity rule flows across the whole tab and no row is left half-empty */
export function GpuCharts({
chartData,
grid,
dataEmpty,
lastGpus,
hasGpuEnginesData,
children,
}: {
chartData: ChartData
grid: boolean
dataEmpty: boolean
lastGpus: Record<string, GPUData>
hasGpuEnginesData: boolean
children?: ReactNode
}) {
return (
<div className="grid xl:grid-cols-2 gap-4">
{children}
{hasGpuEnginesData && (
<ChartCard
legend={true}
@@ -126,9 +131,8 @@ export function GpuDetailCharts({
{Object.keys(lastGpus).map((id) => {
const gpu = lastGpus[id] as GPUData
return (
<div key={id} className="contents">
<Fragment key={id}>
<ChartCard
className={cn(grid && "!col-span-1")}
empty={dataEmpty}
grid={grid}
title={`${gpu.n} ${t`Usage`}`}
@@ -178,7 +182,7 @@ export function GpuDetailCharts({
/>
</ChartCard>
)}
</div>
</Fragment>
)
})}
</div>
@@ -99,7 +99,7 @@ export function ContainerMemoryChart({
empty={dataEmpty}
grid={grid}
title={dockerOrPodman(t`Docker Memory Usage`, isPodman)}
description={dockerOrPodman(t`Memory usage of docker containers`, isPodman)}
description={t`Memory usage of containers`}
cornerEl={<FilterBar />}
>
<AreaChartDefault
@@ -159,7 +159,7 @@ export function ContainerNetworkChart({
empty={dataEmpty}
grid={grid}
title={dockerOrPodman(t`Docker Network I/O`, isPodman)}
description={dockerOrPodman(t`Network traffic of docker containers`, isPodman)}
description={t`Network traffic of containers`}
cornerEl={<FilterBar />}
>
<AreaChartDefault
@@ -0,0 +1,130 @@
import { t } from "@lingui/core/macro"
import AreaChartDefault from "@/components/charts/area-chart"
import { decimalString, formatBytes, toFixedFloat } from "@/lib/utils"
import type { SystemStatsRecord } from "@/types"
import { ChartCard } from "../chart-card"
import { Unit } from "@/lib/enums"
import { useStore } from "@nanostores/react"
import { $userSettings } from "@/lib/stores"
import type { SystemData } from "../use-system-data"
// Accessors for ZFS metrics
const poolUsage =
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.z?.[name]?.du ?? 0
const poolRead =
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.z?.[name]?.rb ?? 0
const poolWrite =
(name: string) =>
({ stats }: SystemStatsRecord) =>
stats?.z?.[name]?.wb ?? 0
export function ZfsPoolUsageChart({ systemData, poolName }: { systemData: SystemData; poolName: string }) {
const { chartData, grid, dataEmpty } = systemData
const latest = chartData.systemStats.at(-1)?.stats
const pool = latest?.z?.[poolName]
if (!pool) {
return null
}
let poolTotal = pool.d
// round to nearest GB
if (poolTotal >= 100) {
poolTotal = Math.round(poolTotal)
}
return (
<ChartCard
empty={dataEmpty}
grid={grid}
title={`${poolName} ${t`Usage`}`}
description={t`Usage of ZFS pool ${poolName}`}
>
<AreaChartDefault
chartData={chartData}
domain={[0, poolTotal]}
showTotal={true}
tickFormatter={(val) => {
const { value, unit } = formatBytes(val * 1024, false, Unit.Bytes, true)
return `${toFixedFloat(value, value >= 10 ? 0 : 1)} ${unit}`
}}
contentFormatter={({ value }) => {
const { value: convertedValue, unit } = formatBytes(value * 1024, false, Unit.Bytes, true)
return `${decimalString(convertedValue, convertedValue >= 100 ? 1 : 2)} ${unit}`
}}
dataPoints={[
{
label: t`Pool Usage`,
dataKey: poolUsage(poolName),
color: 4,
opacity: 0.4,
},
]}
/>
</ChartCard>
)
}
export function ZfsPoolIOChart({ systemData, poolName }: { systemData: SystemData; poolName: string }) {
const { chartData, grid, dataEmpty } = systemData
const userSettings = useStore($userSettings)
if (!chartData.systemStats?.length) {
return null
}
return (
<ChartCard
empty={dataEmpty}
grid={grid}
title={`${poolName} I/O`}
description={t`Throughput of ZFS pool ${poolName}`}
>
<AreaChartDefault
chartData={chartData}
showTotal={true}
dataPoints={[
{
label: t({ message: "Write", comment: "Disk write" }),
dataKey: poolWrite(poolName),
color: 3,
opacity: 0.3,
},
{
label: t({ message: "Read", comment: "Disk read" }),
dataKey: poolRead(poolName),
color: 1,
opacity: 0.3,
},
]}
tickFormatter={(val) => {
const { value, unit } = formatBytes(val, true, userSettings.unitDisk, false)
return `${toFixedFloat(value, value >= 10 ? 0 : 1)} ${unit}`
}}
contentFormatter={({ value }) => {
const { value: convertedValue, unit } = formatBytes(value, true, userSettings.unitDisk, false)
return `${decimalString(convertedValue, convertedValue >= 100 ? 1 : 2)} ${unit}`
}}
/>
</ChartCard>
)
}
/** ZFS section: one stacked usage card per pool plus per-pool I/O cards. */
export function ZfsCharts({ systemData }: { systemData: SystemData }) {
const latest = systemData.chartData.systemStats?.at(-1)?.stats
const pools = latest?.z ?? {}
if (Object.keys(pools).length === 0) {
return null
}
return (
<div className="grid xl:grid-cols-2 gap-4">
{Object.keys(pools).map((poolName) => (
<div key={poolName} className="contents">
<ZfsPoolUsageChart systemData={systemData} poolName={poolName} />
<ZfsPoolIOChart systemData={systemData} poolName={poolName} />
</div>
))}
</div>
)
}
@@ -52,6 +52,14 @@ export default memo(function DiskIOSheet({
writeTimeFn = showMax ? diskDataFns.extraWriteTimeMax(extraFsName) : diskDataFns.extraWriteTime(extraFsName)
}
// cumulative total functions, with extra fs variants if needed
let totalReadFn = diskDataFns.totalRead
let totalWriteFn = diskDataFns.totalWrite
if (extraFsName) {
totalReadFn = diskDataFns.extraTotalRead(extraFsName)
totalWriteFn = diskDataFns.extraTotalWrite(extraFsName)
}
// I/O await functions, with extra fs variants if needed
let rAwaitFn = showMax ? diskDataFns.rAwaitMax : diskDataFns.rAwait
let wAwaitFn = showMax ? diskDataFns.wAwaitMax : diskDataFns.wAwait
@@ -70,12 +78,16 @@ export default memo(function DiskIOSheet({
let hasUtilization = false
let hasAwait = false
let hasWeightedIO = false
let hasCumulative = false
for (const record of chartData.systemStats ?? []) {
const dios = record.stats?.dios
if ((dios?.at(2) ?? 0) > 0) hasUtilization = true
if ((dios?.at(3) ?? 0) > 0) hasAwait = true
if ((dios?.at(5) ?? 0) > 0) hasWeightedIO = true
if (hasUtilization && hasAwait && hasWeightedIO) {
if (!hasCumulative && (totalReadFn(record) > 0 || totalWriteFn(record) > 0)) {
hasCumulative = true
}
if (hasUtilization && hasAwait && hasWeightedIO && hasCumulative) {
break
}
}
@@ -258,6 +270,69 @@ export default memo(function DiskIOSheet({
/>
</ChartCard>
)}
{hasCumulative && (
<ChartCard
empty={dataEmpty}
grid={grid}
title={t`Cumulative Read`}
description={t`Cumulative data read since boot`}
className="min-h-auto"
>
<AreaChartDefault
chartData={chartData}
chartProps={{syncId: "c"}}
dataPoints={[
{
label: t`Read`,
dataKey: totalReadFn,
color: 1,
opacity: 0.4,
},
]}
tickFormatter={(val) => {
const { value, unit } = formatBytes(val, false, userSettings.unitDisk, false)
return `${toFixedFloat(value, value >= 10 ? 0 : 1)} ${unit}`
}}
contentFormatter={({ value }) => {
const { value: convertedValue, unit } = formatBytes(value, false, userSettings.unitDisk, false)
return `${decimalString(convertedValue, convertedValue >= 100 ? 1 : 2)} ${unit}`
}}
/>
</ChartCard>
)}
{hasCumulative && (
<ChartCard
empty={dataEmpty}
grid={grid}
title={t`Cumulative Write`}
description={t`Cumulative data written since boot`}
className="min-h-auto"
>
<AreaChartDefault
chartData={chartData}
chartProps={{syncId: "c"}}
dataPoints={[
{
label: t`Write`,
dataKey: totalWriteFn,
color: 3,
opacity: 0.4,
},
]}
tickFormatter={(val) => {
const { value, unit } = formatBytes(val, false, userSettings.unitDisk, false)
return `${toFixedFloat(value, value >= 10 ? 0 : 1)} ${unit}`
}}
contentFormatter={({ value }) => {
const { value: convertedValue, unit } = formatBytes(value, false, userSettings.unitDisk, false)
return `${decimalString(convertedValue, convertedValue >= 100 ? 1 : 2)} ${unit}`
}}
/>
</ChartCard>
)}
</SheetContent>
)}
</Sheet>
@@ -24,6 +24,17 @@ export function LazySmartTable({ systemId }: { systemId: string }) {
)
}
const ZfsTable = lazy(() => import("./zfs-table"))
export function LazyZfsTable({ systemId }: { systemId: string }) {
const { isIntersecting, ref } = useIntersectionObserver({ rootMargin: "90px" })
return (
<div ref={ref} className={cn(isIntersecting && "contents")}>
{isIntersecting && <ZfsTable systemId={systemId} />}
</div>
)
}
const SystemdTable = lazy(() => import("../../systemd-table/systemd-table"))
export function LazySystemdTable({ systemId }: { systemId: string }) {
@@ -0,0 +1,650 @@
import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert"
import { Badge } from "@/components/ui/badge"
import { Button } from "@/components/ui/button"
import { Card, CardDescription, CardHeader, CardTitle } from "@/components/ui/card"
import { DropdownMenu, DropdownMenuContent, DropdownMenuItem, DropdownMenuTrigger } from "@/components/ui/dropdown-menu"
import { Input } from "@/components/ui/input"
import { Separator } from "@/components/ui/separator"
import { Sheet, SheetContent, SheetDescription, SheetHeader, SheetTitle } from "@/components/ui/sheet"
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table"
import { isReadOnlyUser, pb } from "@/lib/api"
import { cn, formatBytes, formatShortDate, hourWithSeconds, toFixedFloat } from "@/lib/utils"
import type { ZfsDataset, ZfsPoolRecord, ZfsVdev } from "@/types"
import { t } from "@lingui/core/macro"
import { Trans } from "@lingui/react/macro"
import type { Column, ColumnDef } from "@tanstack/react-table"
import {
flexRender,
getCoreRowModel,
getFilteredRowModel,
getSortedRowModel,
useReactTable,
} from "@tanstack/react-table"
import {
ActivityIcon,
BinaryIcon,
CheckCircleIcon,
CircleAlertIcon,
ClockIcon,
HardDriveDownloadIcon,
HardDriveIcon,
HardDriveUploadIcon,
LoaderCircleIcon,
MoreHorizontalIcon,
RefreshCwIcon,
RotateCwIcon,
XCircleIcon,
XIcon,
} from "lucide-react"
import { useCallback, useEffect, useMemo, useState } from "react"
const ZFS_POOL_FIELDS = "id,system,name,health,size,alloc,free,scrub,details_updated,updated"
/** Maps a zpool health string to a Badge variant. */
function healthVariant(health: string): "success" | "warning" | "danger" | "outline" {
switch (health) {
case "ONLINE":
return "success"
case "DEGRADED":
return "warning"
case "FAULTED":
case "OFFLINE":
case "UNAVAIL":
case "REMOVED":
case "SUSPENDED":
return "danger"
default:
return "outline"
}
}
function formatCapacity(bytes: number): string {
if (!bytes) return "-"
const { value, unit } = formatBytes(bytes)
return `${toFixedFloat(value, value >= 10 ? 1 : 2)} ${unit}`
}
function HeaderButton<T>({ column, name, Icon }: { column: Column<T>; name: string; Icon: React.ElementType }) {
const isSorted = column.getIsSorted()
return (
<Button
className={cn(
"h-9 px-3 flex items-center gap-2 duration-50",
isSorted && "bg-accent/70 light:bg-accent text-accent-foreground/90"
)}
variant="ghost"
onClick={() => column.toggleSorting(column.getIsSorted() === "asc")}
>
<Icon className="size-4" />
{name}
</Button>
)
}
const columns: ColumnDef<ZfsPoolRecord>[] = [
{
accessorKey: "name",
sortingFn: (a, b) => a.original.name.localeCompare(b.original.name),
header: ({ column }) => <HeaderButton column={column} name={`Pool`} Icon={HardDriveIcon} />,
cell: ({ getValue }) => <span className="font-medium ms-1.5">{getValue() as string}</span>,
},
{
accessorKey: "health",
sortingFn: (a, b) => a.original.health.localeCompare(b.original.health),
header: ({ column }) => <HeaderButton column={column} name={t`Health`} Icon={ActivityIcon} />,
cell: ({ getValue }) => {
const health = (getValue() as string) || ""
return <Badge variant={healthVariant(health)}>{health || t`Unknown`}</Badge>
},
},
{
id: "size",
accessorFn: (record) => record.size,
invertSorting: true,
header: ({ column }) => <HeaderButton column={column} name={t`Capacity`} Icon={BinaryIcon} />,
cell: ({ getValue }) => <span className="ms-1.5 tabular-nums">{formatCapacity(getValue() as number)}</span>,
},
{
id: "used",
accessorFn: (record) => record.alloc,
invertSorting: true,
header: ({ column }) => <HeaderButton column={column} name={t`Used`} Icon={HardDriveDownloadIcon} />,
cell: ({ getValue }) => <span className="ms-1.5 tabular-nums">{formatCapacity(getValue() as number)}</span>,
},
{
id: "free",
accessorFn: (record) => record.free,
invertSorting: true,
header: ({ column }) => <HeaderButton column={column} name={t({ message: `Free`, context: "Free space" })} Icon={HardDriveUploadIcon} />,
cell: ({ getValue }) => <span className="ms-1.5 tabular-nums">{formatCapacity(getValue() as number)}</span>,
},
{
id: "scrub",
accessorFn: (record) => record.scrub?.state ?? "",
header: ({ column }) => <HeaderButton column={column} name={`Scrub`} Icon={RotateCwIcon} />,
cell: ({ row }) => {
const scrub = row.original.scrub
if (!scrub?.state) return <span className="ms-1.5 text-muted-foreground">{t`None`}</span>
return (
<span className="ms-1.5 tabular-nums">
{scrub.state}
{scrub.progress ? ` (${scrub.progress})` : ""}
</span>
)
},
},
{
id: "updated",
invertSorting: true,
accessorFn: (record) => record.details_updated || record.updated,
header: ({ column }) => <HeaderButton column={column} name={t`Updated`} Icon={ClockIcon} />,
cell: ({ getValue }) => {
const timestamp = getValue() as string
if (!timestamp) return null
const formatter =
new Date(timestamp).toDateString() === new Date().toDateString() ? hourWithSeconds : formatShortDate
return <span className="ms-1 tabular-nums">{formatter(timestamp)}</span>
},
},
]
function VdevTable({ vdevs }: { vdevs: ZfsVdev[] }) {
if (!vdevs?.length) return null
return (
<div className="overflow-x-auto rounded-md border">
<Table>
<TableHeader>
<TableRow>
<TableHead>Vdev</TableHead>
<TableHead>{t`State`}</TableHead>
<TableHead className="text-right">{t`Read errors`}</TableHead>
<TableHead className="text-right">{t`Write errors`}</TableHead>
<TableHead className="text-right">{t`Checksum errors`}</TableHead>
</TableRow>
</TableHeader>
<TableBody>
{vdevs.map((vdev) => (
<TableRow key={vdev.name}>
<TableCell className="font-mono text-xs">{vdev.name}</TableCell>
<TableCell>
<Badge variant={healthVariant(vdev.state ?? "")} className="font-normal">
{vdev.state ?? "-"}
</Badge>
</TableCell>
<TableCell
className={cn("text-right tabular-nums", (vdev.readErrs ?? 0) > 0 && "text-red-600 dark:text-red-400")}
>
{vdev.readErrs ?? 0}
</TableCell>
<TableCell
className={cn("text-right tabular-nums", (vdev.writeErrs ?? 0) > 0 && "text-red-600 dark:text-red-400")}
>
{vdev.writeErrs ?? 0}
</TableCell>
<TableCell
className={cn(
"text-right tabular-nums",
(vdev.checksumErrs ?? 0) > 0 && "text-red-600 dark:text-red-400"
)}
>
{vdev.checksumErrs ?? 0}
</TableCell>
</TableRow>
))}
</TableBody>
</Table>
</div>
)
}
const datasetColumns: ColumnDef<ZfsDataset>[] = [
{
accessorKey: "name",
sortingFn: (a, b) => a.original.name.localeCompare(b.original.name),
header: ({ column }) => <HeaderButton column={column} name={`Dataset`} Icon={HardDriveIcon} />,
cell: ({ getValue }) => <span className="font-mono text-xs">{getValue() as string}</span>,
},
{
id: "used",
accessorFn: (ds) => ds.used ?? 0,
invertSorting: true,
header: ({ column }) => <HeaderButton column={column} name={t`Used`} Icon={HardDriveDownloadIcon} />,
cell: ({ getValue }) => <span className="text-right tabular-nums">{formatCapacity(getValue() as number)}</span>,
},
{
id: "avail",
accessorFn: (ds) => ds.avail ?? 0,
invertSorting: true,
header: ({ column }) => <HeaderButton column={column} name={t({message:`Available`, context: "Disk space available"})} Icon={HardDriveUploadIcon} />,
cell: ({ getValue }) => <span className="text-right tabular-nums">{formatCapacity(getValue() as number)}</span>,
},
{
accessorKey: "mount",
sortingFn: (a, b) => (a.original.mount ?? "").localeCompare(b.original.mount ?? ""),
header: ({ column }) => <HeaderButton column={column} name={t`Mountpoint`} Icon={HardDriveIcon} />,
cell: ({ getValue }) => (
<span className="font-mono text-xs text-muted-foreground">{(getValue() as string) || "-"}</span>
),
},
]
function DatasetTable({ datasets }: { datasets: ZfsDataset[] }) {
const [filter, setFilter] = useState("")
const filtered = useMemo(() => {
if (!datasets) return []
if (!filter) return datasets
const needle = filter.toLowerCase()
return datasets.filter((ds) => ds.name.toLowerCase().includes(needle))
}, [datasets, filter])
const table = useReactTable({
data: filtered,
columns: datasetColumns,
getCoreRowModel: getCoreRowModel(),
getSortedRowModel: getSortedRowModel(),
})
if (!datasets?.length) return null
return (
<div>
<div className="mb-2 flex items-center justify-between gap-4">
<h3 className="text-base font-semibold">Datasets</h3>
<div className="relative w-64 max-w-full">
<Input
placeholder={t`Filter...`}
value={filter}
onChange={(e) => setFilter(e.target.value)}
className="px-4 w-full"
/>
{filter && (
<Button
type="button"
variant="ghost"
size="icon"
aria-label={t`Clear`}
className="absolute right-1 top-1/2 -translate-y-1/2 h-7 w-7 text-muted-foreground"
onClick={() => setFilter("")}
>
<XIcon className="h-4 w-4" />
</Button>
)}
</div>
</div>
<div className="max-h-80 overflow-auto rounded-md border">
<Table>
<TableHeader className="sticky top-0 z-10">
{table.getHeaderGroups().map((headerGroup) => (
<TableRow key={headerGroup.id}>
{headerGroup.headers.map((header) => (
<TableHead key={header.id} className="px-2">
{header.isPlaceholder ? null : flexRender(header.column.columnDef.header, header.getContext())}
</TableHead>
))}
</TableRow>
))}
</TableHeader>
<TableBody>
{table.getRowModel().rows.map((row) => (
<TableRow key={row.id}>
{row.getVisibleCells().map((cell) => (
<TableCell key={cell.id} className="ps-5 whitespace-pre">
{flexRender(cell.column.columnDef.cell, cell.getContext())}
</TableCell>
))}
</TableRow>
))}
</TableBody>
</Table>
</div>
</div>
)
}
function PoolSheet({
poolId,
open,
onOpenChange,
}: {
poolId: string | null
open: boolean
onOpenChange: (open: boolean) => void
}) {
const [pool, setPool] = useState<ZfsPoolRecord | null>(null)
const [isLoading, setIsLoading] = useState(false)
useEffect(() => {
let active = true
if (!poolId) {
setPool(null)
return
}
// Only fetch when opening, not when closing (keeps data visible during close animation)
if (!open) return
setIsLoading(true)
pb.collection("zfs_pools")
.getOne(poolId)
.then((record) => active && setPool(record as ZfsPoolRecord))
.catch(() => active && setPool(null))
.finally(() => active && setIsLoading(false))
return () => {
active = false
}
}, [open, poolId])
const health = pool?.health || ""
const healthVariantValue = healthVariant(health)
const HealthIcon =
healthVariantValue === "success"
? CheckCircleIcon
: healthVariantValue === "warning"
? CircleAlertIcon
: XCircleIcon
return (
<Sheet open={open} onOpenChange={onOpenChange}>
<SheetContent className="w-full sm:max-w-220 gap-0 overflow-y-auto">
<SheetHeader className="mb-0 border-b">
<SheetTitle className="flex items-center gap-2">
{pool ? pool.name : `ZFS Pool`}
{pool && <Badge variant={healthVariantValue}>{health}</Badge>}
</SheetTitle>
<SheetDescription className="flex flex-wrap items-center gap-x-2 gap-y-1">
{pool?.size ? formatCapacity(pool.size) : null}
{pool?.alloc ? (
<>
<Separator orientation="vertical" className="h-2.5 bg-muted-foreground opacity-70" />
<span>
<Trans>Used</Trans>: {formatCapacity(pool.alloc)}
</span>
</>
) : null}
{pool?.free ? (
<>
<Separator orientation="vertical" className="h-2.5 bg-muted-foreground opacity-70" />
<span>
<Trans context="Free space">Free</Trans>: {formatCapacity(pool.free)}
</span>
</>
) : null}
</SheetDescription>
</SheetHeader>
<div className="flex-1 p-4 flex flex-col gap-4">
{isLoading ? (
<div className="flex justify-center py-8">
<LoaderCircleIcon className="animate-spin size-10 opacity-60" />
</div>
) : (
<>
{pool && health && (
<Alert className="pb-3 shrink-0">
<HealthIcon className="size-4" />
<AlertTitle>
<Trans>Pool Health</Trans>: {health}
</AlertTitle>
{pool.scrub?.state && (
<AlertDescription>
Scrub: {pool.scrub.state}
{pool.scrub.progress ? ` (${pool.scrub.progress})` : ""}
{pool.scrub.errors ? `, ${pool.scrub.errors} errors` : ""}
</AlertDescription>
)}
</Alert>
)}
{pool?.vdevs?.length || pool?.datasets?.length ? (
<>
{pool.vdevs?.length ? <VdevTable vdevs={pool.vdevs} /> : null}
<DatasetTable datasets={pool.datasets ?? []} />
</>
) : (
!isLoading && (
<div className="py-8 text-center text-sm text-muted-foreground">
<Trans>No detail data for this pool.</Trans>
</div>
)
)}
</>
)}
</div>
</SheetContent>
</Sheet>
)
}
export default function ZfsTable({ systemId }: { systemId?: string }) {
const [zfsPools, setZfsPools] = useState<ZfsPoolRecord[]>()
const [globalFilter, setGlobalFilter] = useState("")
const [activePoolId, setActivePoolId] = useState<string | null>(null)
const [sheetOpen, setSheetOpen] = useState(false)
const [refreshingId, setRefreshingId] = useState<string | null>(null)
useEffect(() => {
let disposed = false
let unsubscribe: () => void = () => {}
// fetch initial records
pb.collection<ZfsPoolRecord>("zfs_pools")
.getFullList({
filter: systemId ? pb.filter("system={:id}", { id: systemId }) : "",
sort: "name",
fields: ZFS_POOL_FIELDS,
})
.then((records) => !disposed && setZfsPools(records))
.catch((error) => console.error("Failed to fetch ZFS pools:", error))
// subscribe to realtime updates
const pbOptions = systemId ? { filter: `system="${systemId}"` } : undefined
;(async () => {
try {
const unsubscribeNow = await pb.collection<ZfsPoolRecord>("zfs_pools").subscribe(
"*",
(event) => {
const record = event.record as ZfsPoolRecord
setZfsPools((current) => {
const pools = current ?? []
const matchesSystemScope = !systemId || record.system === systemId
if (event.action === "delete") {
return pools.filter((pool) => pool.id !== record.id)
}
if (!matchesSystemScope) {
return pools.filter((pool) => pool.id !== record.id)
}
const existingIndex = pools.findIndex((pool) => pool.id === record.id)
if (existingIndex === -1) {
return [record, ...pools]
}
const next = [...pools]
next[existingIndex] = record
return next
})
},
pbOptions
)
if (disposed) {
unsubscribeNow()
} else {
unsubscribe = unsubscribeNow
}
} catch (error) {
console.error("Failed to subscribe to ZFS pool updates:", error)
}
})()
return () => {
disposed = true
unsubscribe?.()
}
}, [systemId])
const refreshSystem = useCallback(async (systemId: string) => {
try {
await pb.send("/api/beszel/zfs/refresh", {
method: "POST",
query: { system: systemId },
})
} catch (error) {
console.error("Failed to refresh ZFS pools:", error)
}
}, [])
const handleRowRefresh = useCallback(
async (pool: ZfsPoolRecord) => {
if (!pool.system) return
setRefreshingId(pool.id)
try {
await refreshSystem(pool.system)
} finally {
setRefreshingId((id) => (id === pool.id ? null : id))
}
},
[refreshSystem]
)
const actionColumn = useMemo<ColumnDef<ZfsPoolRecord>>(
() => ({
id: "actions",
enableSorting: false,
header: () => (
<span className="sr-only">
<Trans>Actions</Trans>
</span>
),
cell: ({ row }) => {
const pool = row.original
const isRowRefreshing = refreshingId === pool.id
return (
<div className="flex justify-end">
<DropdownMenu>
<DropdownMenuTrigger asChild>
<Button
variant="ghost"
size="icon"
className="size-10"
onClick={(event) => event.stopPropagation()}
onMouseDown={(event) => event.stopPropagation()}
>
<span className="sr-only">
<Trans>Open menu</Trans>
</span>
<MoreHorizontalIcon className="w-5" />
</Button>
</DropdownMenuTrigger>
<DropdownMenuContent align="end" onClick={(event) => event.stopPropagation()}>
<DropdownMenuItem
onClick={(event) => {
event.stopPropagation()
handleRowRefresh(pool)
}}
disabled={isRowRefreshing}
>
<RefreshCwIcon className={cn("me-2.5 size-4", isRowRefreshing && "animate-spin")} />
<Trans>Refresh</Trans>
</DropdownMenuItem>
</DropdownMenuContent>
</DropdownMenu>
</div>
)
},
}),
[refreshingId, handleRowRefresh]
)
const tableColumns = useMemo(() => {
return isReadOnlyUser() ? columns : [...columns, actionColumn]
}, [actionColumn])
const table = useReactTable({
data: zfsPools || ([] as ZfsPoolRecord[]),
columns: tableColumns,
getCoreRowModel: getCoreRowModel(),
getSortedRowModel: getSortedRowModel(),
getFilteredRowModel: getFilteredRowModel(),
state: { globalFilter },
onGlobalFilterChange: setGlobalFilter,
globalFilterFn: (row, _columnId, filterValue) => {
const pool = row.original
const searchString = `${pool.name} ${pool.health ?? ""}`.toLowerCase()
return (filterValue as string)
.toLowerCase()
.split(" ")
.every((term) => searchString.includes(term))
},
})
const rows = table.getRowModel().rows
// Hide the table on system pages if there's no data
if (systemId && !zfsPools?.length && !globalFilter) {
return null
}
const openSheet = (pool: ZfsPoolRecord) => {
setActivePoolId(pool.id)
setSheetOpen(true)
}
return (
<div>
<Card className="@container w-full px-3 py-5 sm:py-6 sm:px-6">
<CardHeader className="p-0 mb-3 sm:mb-4">
<div className="grid md:flex gap-x-5 gap-y-3 w-full items-end">
<div className="px-2 sm:px-1">
<CardTitle className="mb-2">ZFS</CardTitle>
<CardDescription className="flex">
<Trans>Click on a pool to view vdev and dataset details.</Trans>
</CardDescription>
</div>
<div className="relative ms-auto w-full max-w-full md:w-64">
<Input
placeholder={t`Filter...`}
value={globalFilter}
onChange={(event) => setGlobalFilter(event.target.value)}
className="px-4 w-full max-w-full md:w-64"
/>
{globalFilter && (
<Button
type="button"
variant="ghost"
size="icon"
aria-label={t`Clear`}
className="absolute right-1 top-1/2 -translate-y-1/2 h-7 w-7 text-muted-foreground"
onClick={() => setGlobalFilter("")}
>
<XIcon className="h-4 w-4" />
</Button>
)}
</div>
</div>
</CardHeader>
<div className="h-min max-h-[calc(100dvh-17rem)] max-w-full relative overflow-auto rounded-md border">
<Table>
<TableHeader className="sticky top-0 z-50 w-full border-b-2">
{table.getHeaderGroups().map((headerGroup) => (
<TableRow key={headerGroup.id}>
{headerGroup.headers.map((header) => (
<TableHead key={header.id} className="px-2">
{header.isPlaceholder ? null : flexRender(header.column.columnDef.header, header.getContext())}
</TableHead>
))}
</TableRow>
))}
</TableHeader>
<TableBody>
{rows.map((row) => (
<TableRow
key={row.id}
data-state={row.getIsSelected() && "selected"}
className="cursor-pointer"
onClick={() => openSheet(row.original)}
>
{row.getVisibleCells().map((cell) => (
<TableCell key={cell.id}>{flexRender(cell.column.columnDef.cell, cell.getContext())}</TableCell>
))}
</TableRow>
))}
</TableBody>
</Table>
</div>
</Card>
<PoolSheet poolId={activePoolId} open={sheetOpen} onOpenChange={setSheetOpen} />
</div>
)
}
@@ -196,7 +196,13 @@ export function SystemsTableColumns(viewMode: "table" | "grid"): ColumnDef<Syste
accessorFn: ({ info }) => info.g || undefined,
id: "gpu",
name: () => "GPU",
cell: TableCellWithMeter,
cell: (info) => {
const val = info.getValue() as number | undefined
if (val === undefined) {
return null
}
return TableCellWithMeter(info)
},
Icon: GpuIcon,
header: sortableHeader,
},
@@ -484,9 +490,9 @@ function DiskCellWithMultiple(info: CellContext<SystemRecord, unknown>) {
const { info: sysInfo, status, id } = info.row.original
const extraFs = Object.entries(sysInfo.efs ?? {})
const rootDiskPct = sysInfo.dp
const rootDiskName = sysInfo.rdn
// sort extra disks by percentage descending
extraFs.sort((a, b) => b[1] - a[1])
extraFs.sort((a, b) => a[0].localeCompare(b[0]))
function getIndicatorColor(pct: number) {
const threshold = getMeterStateByThresholds(pct, colorWarn, colorCrit)
@@ -541,8 +547,8 @@ function DiskCellWithMultiple(info: CellContext<SystemRecord, unknown>) {
<TooltipContent side="right" className="max-w-xs pb-2">
<div className="grid gap-1">
<div className="grid gap-0.5">
<div className="text-[0.65rem] text-muted-foreground uppercase tracking-wide tabular-nums">
<Trans context="Root disk label">Root</Trans>
<div className="text-[0.65rem] max-w-40 text-muted-foreground uppercase tracking-wide truncate tabular-nums">
{rootDiskName ?? <Trans context="Root disk label">Root</Trans>}
</div>
<div className="flex gap-2 items-center tabular-nums text-xs">
<span className="min-w-7">{decimalString(rootDiskPct, rootDiskPct >= 10 ? 1 : 2)}%</span>
+33 -1
View File
@@ -1,5 +1,5 @@
import { t } from "@lingui/core/macro"
import { CpuIcon, HardDriveIcon, MemoryStickIcon, ServerIcon } from "lucide-react"
import { ContainerIcon, CpuIcon, HardDriveIcon, MemoryStickIcon, ServerCrashIcon, ServerIcon } from "lucide-react"
import type { RecordSubscription } from "pocketbase"
import { EthernetIcon, GpuIcon } from "@/components/ui/icons"
import { $alerts } from "@/lib/stores"
@@ -23,6 +23,18 @@ export const alertInfo: Record<string, AlertInfo> = {
icon: CpuIcon,
desc: () => t`Triggers when CPU usage exceeds a threshold`,
},
CPUIOWait: {
name: () => t`CPU I/O Wait`,
unit: "%",
icon: CpuIcon,
desc: () => t`Triggers when CPU I/O wait exceeds a threshold`,
},
CPUSteal: {
name: () => t`CPU Steal Time`,
unit: "%",
icon: CpuIcon,
desc: () => t`Triggers when CPU steal time exceeds a threshold`,
},
Memory: {
name: () => t`Memory Usage`,
unit: "%",
@@ -92,6 +104,26 @@ export const alertInfo: Record<string, AlertInfo> = {
start: 20,
invert: true,
},
ContainerHealth: {
name: () => t`Container Health`,
unit: "",
icon: ContainerIcon,
desc: () => t`Triggers when a container's health check reports unhealthy`,
note: () =>
t`Notifications may include recent container log excerpts.`,
triggeredDesc: () => t`One or more containers are unhealthy`,
singleDesc: () => `${t`Container`} ${t`Unhealthy`}`,
},
SystemdFailed: {
name: () => t`Failed Services`,
unit: "",
icon: ServerCrashIcon,
desc: () => t`Triggers when any systemd service enters the failed state`,
triggeredDesc: () => t`One or more services are in a failed state`,
/** Fires on first observation - the agent only polls systemd every 10 minutes */
noDuration: true,
},
} as const
/** Helper to manage user alerts */

Some files were not shown because too many files have changed in this diff Show More