mirror of
https://github.com/henrygd/beszel.git
synced 2026-09-13 19:44:40 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6a7b2772d9 | ||
|
|
086091a0fe | ||
|
|
f204dc17e6 | ||
|
|
5fe1583655 | ||
|
|
6d82ee70b1 | ||
|
|
bb270e02a8 | ||
|
|
312c109138 | ||
|
|
c938368089 | ||
|
|
8d6a5d5f6e | ||
|
|
98687be2f2 | ||
|
|
e39e153ca0 | ||
|
|
5b87f7d7cb | ||
|
|
9a0aa5a89e | ||
|
|
997adc19bb | ||
|
|
08d813620c | ||
|
|
6cb302fcf6 | ||
|
|
59eed073c3 | ||
|
|
266a74bab8 | ||
|
|
ad24484caa | ||
|
|
027d0c204d | ||
|
|
46d94a9804 | ||
|
|
82fc772882 | ||
|
|
c157c2026d | ||
|
|
e1d9ebc61d | ||
|
|
5af6b6b184 | ||
|
|
ffcdb04167 | ||
|
|
bc21da9cb3 | ||
|
|
71af06b31c | ||
|
|
a8def47018 | ||
|
|
d2a253082f | ||
|
|
3d8fc39e94 | ||
|
|
7d347cfd6a | ||
|
|
5790fbecce | ||
|
|
f9309da9f0 | ||
|
|
7d97b0d23a | ||
|
|
f104f31ee3 | ||
|
|
a1ca51608a | ||
|
|
a4de2e87c4 | ||
|
|
5969d36856 | ||
|
|
b1895247ba | ||
|
|
097180e8d7 | ||
|
|
ed88e6efae | ||
|
|
b670224ed8 | ||
|
|
917d069ab3 | ||
|
|
b38fb7dafa | ||
|
|
8675199e20 | ||
|
|
3af6512514 | ||
|
|
87620f3251 | ||
|
|
fa9de55433 | ||
|
|
e235c9935c | ||
|
|
7c60f02802 | ||
|
|
6fe268e463 | ||
|
|
467f176713 | ||
|
|
4c8e3c69ba | ||
|
|
8dfdacb8f5 | ||
|
|
d61b75ffdf | ||
|
|
d7256c7af7 | ||
|
|
0ad707288a | ||
|
|
0bc5470f08 | ||
|
|
4c48fe0c41 | ||
|
|
6efe4be648 | ||
|
|
f1e5797c76 | ||
|
|
6f92b9396d | ||
|
|
946f2e6be1 | ||
|
|
ba90daf4d6 | ||
|
|
aa1d67a122 | ||
|
|
68a3f8962a | ||
|
|
0eb3426619 | ||
|
|
96beadc8c9 | ||
|
|
65a6f60304 | ||
|
|
54dae08631 | ||
|
|
2df1f722e4 | ||
|
|
2054b276a7 | ||
|
|
19f250c7de | ||
|
|
e07f91b920 | ||
|
|
218aa8478a | ||
|
|
90b789aa72 | ||
|
|
2d01d71f46 | ||
|
|
f894d188cf | ||
|
|
46cc602d37 | ||
|
|
89ad51d4ce | ||
|
|
d5f41af3a6 | ||
|
|
1074503af1 | ||
|
|
90f1bdef1e | ||
|
|
5f383c0eb1 | ||
|
|
ae037b278e | ||
|
|
9f1128933f | ||
|
|
dbe10e3a8f | ||
|
|
7e7bcb3b35 | ||
|
|
418e3f0892 | ||
|
|
7556a63378 | ||
|
|
260b082c5e | ||
|
|
55054a75ab | ||
|
|
ccd1735a8e | ||
|
|
c67b69d17e | ||
|
|
da3ab62d4e | ||
|
|
ec4ec01a39 | ||
|
|
ca5497324c | ||
|
|
adaf6f338d | ||
|
|
1ab1229a61 | ||
|
|
9a54d844ba | ||
|
|
87405c5f10 | ||
|
|
bfa6a1e361 | ||
|
|
3688b2d033 | ||
|
|
b4e1f3fafe | ||
|
|
eebcd56462 | ||
|
|
eb5dd230cf | ||
|
|
3337dff64b | ||
|
|
b68acea5a8 | ||
|
|
66ac62a125 | ||
|
|
e2a18ec636 | ||
|
|
ffd4fc2c45 | ||
|
|
052489cada | ||
|
|
d50c09176f | ||
|
|
fe84cfa16d | ||
|
|
6d0b83f6de | ||
|
|
cf90249519 | ||
|
|
6607d4c0d6 | ||
|
|
bc55e249c4 | ||
|
|
977826e8f3 | ||
|
|
8450b40e0c | ||
|
|
ac4436bea3 | ||
|
|
fa5cda83c2 | ||
|
|
35af36fbd2 | ||
|
|
e380ab6917 | ||
|
|
c3a432101b | ||
|
|
98e86b4c9c | ||
|
|
d40372842b | ||
|
|
c9b6279e61 | ||
|
|
9887b662ad | ||
|
|
71bc6f9b9f | ||
|
|
87468965dc | ||
|
|
db0da58ac8 | ||
|
|
b9b3a23063 | ||
|
|
ac34e58113 | ||
|
|
3c8703d1c0 | ||
|
|
01efba50a5 | ||
|
|
d0453f1ca6 | ||
|
|
7ffc6e81ce | ||
|
|
1aa9fcd31d | ||
|
|
bd52134558 | ||
|
|
3fac02f0c0 | ||
|
|
e4c0522cae | ||
|
|
f2ccaacecb | ||
|
|
a81ce14046 | ||
|
|
6e4a2c8a3d | ||
|
|
8cf39be6ef | ||
|
|
146ca4284c | ||
|
|
6eae195d9c | ||
|
|
30bf65991b | ||
|
|
f57f5883ce | ||
|
|
0472343730 | ||
|
|
d3a1d61955 | ||
|
|
9708e24fd2 | ||
|
|
17e910a246 | ||
|
|
0e65a2373f | ||
|
|
c1c1cd1bcb | ||
|
|
cd9ea51039 | ||
|
|
a71617e058 | ||
|
|
e5507fa106 | ||
|
|
a024c3cfd0 | ||
|
|
07466804e7 | ||
|
|
981c788d6f | ||
|
|
f5576759de | ||
|
|
be0b708064 | ||
|
|
ab3a3de46c | ||
|
|
1556e53926 | ||
|
|
e3ade3aeb8 | ||
|
|
b013f06956 | ||
|
|
3793b27958 | ||
|
|
5b02158228 | ||
|
|
0ae8c42ae0 | ||
|
|
ea80f3c5a2 | ||
|
|
c3dffff5e4 | ||
|
|
06fdd0e7a8 | ||
|
|
6e3fd90834 | ||
|
|
5ab82183fa | ||
|
|
a68e02ca84 | ||
|
|
0f2e16c63c | ||
|
|
c4009f2b43 | ||
|
|
ef0c1420d1 | ||
|
|
eb9a8e1ef9 | ||
|
|
6b5e6ffa9a | ||
|
|
d656036d3b | ||
|
|
80b73c7faf | ||
|
|
afe9eb7a70 | ||
|
|
7f565a3086 | ||
|
|
77862d4cb1 | ||
|
|
e158a9001b | ||
|
|
f670e868e4 | ||
|
|
0fff699bf6 | ||
|
|
ba10da1b9f | ||
|
|
7f4f14b505 | ||
|
|
2fda4ff264 | ||
|
|
20b0b40ec8 | ||
|
|
d548a012b4 | ||
|
|
ce5d1217dd | ||
|
|
cef09d7cb1 | ||
|
|
f6440acb43 | ||
|
|
5463a38f0f | ||
|
|
80135fdad3 | ||
|
|
5db4eb4346 | ||
|
|
f6c5e2928a | ||
|
|
6a207c33fa | ||
|
|
9f19afccde | ||
|
|
f25f2469e3 | ||
|
|
5bd43ed461 | ||
|
|
afdc3f7779 | ||
|
|
a227c77526 | ||
|
|
8202d746af | ||
|
|
9840b99327 | ||
|
|
f7b5a505e8 | ||
|
|
3cb32ac046 | ||
|
|
e610d9bfc8 | ||
|
|
b53fdbe0ef | ||
|
|
c7261b56f1 | ||
|
|
3f4c3d51b6 | ||
|
|
ad21cab457 | ||
|
|
f04684b30a | ||
|
|
4d4e4fba9b | ||
|
|
62587919f4 | ||
|
|
35528332fd | ||
|
|
e3e453140e | ||
|
|
7a64da9f65 | ||
|
|
8e71c8ad97 | ||
|
|
97f3b8c61f | ||
|
|
0b0b5d16d7 | ||
|
|
b2fd50211e | ||
|
|
c159eaacd1 | ||
|
|
441bdd2ec5 | ||
|
|
ff36138229 | ||
|
|
be70840609 | ||
|
|
565162ef5f | ||
|
|
adbfe7cfb7 | ||
|
|
1ff7762c80 | ||
|
|
0ab8a606e0 | ||
|
|
e4e0affbc1 | ||
|
|
c3a0e645ee | ||
|
|
c6c3950fb0 | ||
|
|
48ddc96a0d | ||
|
|
704cb86de8 | ||
|
|
2854ce882f | ||
|
|
ed50367f70 | ||
|
|
4ebe869591 | ||
|
|
c9bbbe91f2 | ||
|
|
5bfe4f6970 | ||
|
|
380d2b1091 | ||
|
|
a7f99e7a8c | ||
|
|
bd94a9d142 | ||
|
|
8e2316f845 | ||
|
|
0d3dfcb207 | ||
|
|
b386ce5190 | ||
|
|
e527534016 | ||
|
|
ec7ad632a9 | ||
|
|
963fce5a33 | ||
|
|
d38c0da06d | ||
|
|
cae6ac4626 | ||
|
|
6b1ff264f2 | ||
|
|
35d0e792ad | ||
|
|
654cd06b19 | ||
|
|
5e1b028130 | ||
|
|
638e7dc12a | ||
|
|
73c262455d | ||
|
|
0c4d2edd45 | ||
|
|
8f23fff1c9 | ||
|
|
02c1a0c13d | ||
|
|
69fdcb36ab | ||
|
|
b91eb6de40 | ||
|
|
ec69f6c6e0 | ||
|
|
a86cb91e07 | ||
|
|
004841717a | ||
|
|
096296ba7b | ||
|
|
b012df5669 | ||
|
|
12545b4b6d | ||
|
|
9e2296452b | ||
|
|
ac79860d4a | ||
|
|
e13a99fdac | ||
|
|
4cfb2a86ad | ||
|
|
191f25f6e0 | ||
|
|
aa8b3711d7 | ||
|
|
1fb0b25988 | ||
|
|
04600d83cc | ||
|
|
5d8906c9b2 | ||
|
|
daac287b9d | ||
|
|
d526ea61a9 | ||
|
|
79616e1662 | ||
|
|
01e8bdf040 | ||
|
|
1e3a44e05d | ||
|
|
311095cfdd | ||
|
|
4869c834bb | ||
|
|
e1c1e97f0a | ||
|
|
f6b2824ccc | ||
|
|
f17ffc21b8 |
@@ -0,0 +1,12 @@
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: gomod
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
|
||||
@@ -41,7 +41,7 @@ jobs:
|
||||
# henrygd/beszel-agent-nvidia
|
||||
- image: henrygd/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia
|
||||
platforms: linux/amd64
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: docker.io
|
||||
username_secret: DOCKERHUB_USERNAME
|
||||
password_secret: DOCKERHUB_TOKEN
|
||||
@@ -52,6 +52,19 @@ jobs:
|
||||
type=semver,pattern={{major}}
|
||||
type=raw,value={{sha}},enable=${{ github.ref_type != 'tag' }}
|
||||
|
||||
# henrygd/beszel-agent-nvidia:slim
|
||||
- image: henrygd/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia_slim
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: docker.io
|
||||
username_secret: DOCKERHUB_USERNAME
|
||||
password_secret: DOCKERHUB_TOKEN
|
||||
tags: |
|
||||
type=raw,value=slim
|
||||
type=semver,pattern={{version}}-slim
|
||||
type=semver,pattern={{major}}.{{minor}}-slim
|
||||
type=semver,pattern={{major}}-slim
|
||||
|
||||
# henrygd/beszel-agent-intel
|
||||
- image: henrygd/beszel-agent-intel
|
||||
dockerfile: ./internal/dockerfile_agent_intel
|
||||
@@ -96,7 +109,7 @@ jobs:
|
||||
# ghcr.io/henrygd/beszel-agent-nvidia
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia
|
||||
platforms: linux/amd64
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password_secret: GITHUB_TOKEN
|
||||
@@ -107,6 +120,19 @@ jobs:
|
||||
type=semver,pattern={{major}}
|
||||
type=raw,value={{sha}},enable=${{ github.ref_type != 'tag' }}
|
||||
|
||||
# ghcr.io/henrygd/beszel-agent-nvidia:slim
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia_slim
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password_secret: GITHUB_TOKEN
|
||||
tags: |
|
||||
type=raw,value=slim
|
||||
type=semver,pattern={{version}}-slim
|
||||
type=semver,pattern={{major}}.{{minor}}-slim
|
||||
type=semver,pattern={{major}}-slim
|
||||
|
||||
# ghcr.io/henrygd/beszel-agent-intel
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-intel
|
||||
dockerfile: ./internal/dockerfile_agent_intel
|
||||
@@ -152,7 +178,7 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
@@ -164,14 +190,14 @@ jobs:
|
||||
run: bun run --cwd ./internal/site build
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
uses: docker/setup-qemu-action@v4
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
- name: Docker metadata
|
||||
id: metadata
|
||||
uses: docker/metadata-action@v5
|
||||
uses: docker/metadata-action@v6
|
||||
with:
|
||||
images: ${{ matrix.image }}
|
||||
tags: ${{ matrix.tags }}
|
||||
@@ -181,7 +207,7 @@ jobs:
|
||||
env:
|
||||
password_secret_exists: ${{ secrets[matrix.password_secret] != '' && 'true' || 'false' }}
|
||||
if: github.event_name != 'pull_request' && env.password_secret_exists == 'true'
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: ${{ matrix.username || secrets[matrix.username_secret] }}
|
||||
password: ${{ secrets[matrix.password_secret] }}
|
||||
@@ -190,11 +216,13 @@ jobs:
|
||||
# Build and push Docker image with Buildx (don't push on PR)
|
||||
# https://github.com/docker/build-push-action
|
||||
- name: Build and push Docker image
|
||||
uses: docker/build-push-action@v5
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: ./
|
||||
file: ${{ matrix.dockerfile }}
|
||||
platforms: ${{ matrix.platforms || 'linux/amd64,linux/arm64,linux/arm/v7' }}
|
||||
platforms: ${{ matrix.platforms || 'linux/amd64,linux/arm64,linux/arm/v6,linux/arm/v7' }}
|
||||
push: ${{ github.ref_type == 'tag' && secrets[matrix.password_secret] != '' }}
|
||||
provenance: mode=max
|
||||
sbom: true
|
||||
tags: ${{ steps.metadata.outputs.tags }}
|
||||
labels: ${{ steps.metadata.outputs.labels }}
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
name: Helm charts
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- "supplemental/helm/**"
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "supplemental/helm/**"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
env:
|
||||
OCI_REGISTRY: ghcr.io/henrygd/beszel-charts
|
||||
|
||||
jobs:
|
||||
changes:
|
||||
name: Detect changed charts
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
charts: ${{ steps.changes.outputs.charts }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Detect changed charts
|
||||
id: changes
|
||||
env:
|
||||
BASE_SHA: ${{ github.event_name == 'pull_request' && github.event.pull_request.base.sha || github.event.before }}
|
||||
run: |
|
||||
charts=()
|
||||
|
||||
for name in beszel-agent beszel-hub; do
|
||||
path="supplemental/helm/$name"
|
||||
if ! git diff --quiet "$BASE_SHA" "$GITHUB_SHA" -- "$path"; then
|
||||
charts+=("$name|$path")
|
||||
fi
|
||||
done
|
||||
|
||||
printf '%s\n' "${charts[@]}" \
|
||||
| jq -Rsc 'split("\n") | map(select(length > 0) | split("|") | {name: .[0], path: .[1]})' \
|
||||
| xargs -0 printf 'charts=%s\n' >> "$GITHUB_OUTPUT"
|
||||
|
||||
validate-and-publish:
|
||||
name: ${{ github.event_name == 'push' && 'Publish' || 'Validate' }} ${{ matrix.chart.name }}
|
||||
needs: changes
|
||||
if: needs.changes.outputs.charts != '[]'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
chart: ${{ fromJSON(needs.changes.outputs.charts) }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Helm
|
||||
uses: azure/setup-helm@v5
|
||||
|
||||
- name: Lint chart
|
||||
run: helm lint "${{ matrix.chart.path }}" --set env.KEY=ci-placeholder
|
||||
|
||||
- name: Render chart
|
||||
run: helm template "${{ matrix.chart.name }}" "${{ matrix.chart.path }}" --set env.KEY=ci-placeholder > /dev/null
|
||||
|
||||
- name: Package chart
|
||||
id: package
|
||||
env:
|
||||
CHART_NAME: ${{ matrix.chart.name }}
|
||||
CHART_PATH: ${{ matrix.chart.path }}
|
||||
run: |
|
||||
version=$(awk '/^version:/ { print $2 }' "$CHART_PATH/Chart.yaml")
|
||||
test -n "$version"
|
||||
|
||||
mkdir -p .helm-packages
|
||||
helm package "$CHART_PATH" --destination .helm-packages
|
||||
|
||||
package=".helm-packages/${CHART_NAME}-${version}.tgz"
|
||||
test -f "$package"
|
||||
echo "version=$version" >> "$GITHUB_OUTPUT"
|
||||
echo "package=$package" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Log in to GHCR
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: echo "$GITHUB_TOKEN" | helm registry login ghcr.io --username "$GITHUB_ACTOR" --password-stdin
|
||||
|
||||
- name: Check chart version is unpublished
|
||||
env:
|
||||
CHART_NAME: ${{ matrix.chart.name }}
|
||||
CHART_VERSION: ${{ steps.package.outputs.version }}
|
||||
run: |
|
||||
chart="oci://${OCI_REGISTRY}/${CHART_NAME}"
|
||||
if helm show chart "$chart" --version "$CHART_VERSION" > /dev/null 2>&1; then
|
||||
echo "${CHART_NAME} ${CHART_VERSION} is already published. Bump version in Chart.yaml." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Publish chart
|
||||
if: github.event_name == 'push'
|
||||
run: helm push "${{ steps.package.outputs.package }}" "oci://${OCI_REGISTRY}"
|
||||
@@ -15,7 +15,7 @@ jobs:
|
||||
name: Lock Inactive Issues
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- uses: klaasnicolaas/action-inactivity-lock@v1.1.3
|
||||
- uses: klaasnicolaas/action-inactivity-lock@v2.0.1
|
||||
id: lock
|
||||
with:
|
||||
days-inactive-issues: 14
|
||||
@@ -29,7 +29,7 @@ jobs:
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- name: Close Stale Issues
|
||||
uses: actions/stale@v10
|
||||
uses: actions/stale@v11
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
@@ -27,12 +27,12 @@ jobs:
|
||||
run: bun run --cwd ./internal/site build
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version: "^1.22.1"
|
||||
go-version: stable
|
||||
|
||||
- name: Set up .NET
|
||||
uses: actions/setup-dotnet@v4
|
||||
uses: actions/setup-dotnet@v6
|
||||
with:
|
||||
dotnet-version: "9.0.x"
|
||||
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: GoReleaser beszel
|
||||
uses: goreleaser/goreleaser-action@v6
|
||||
uses: goreleaser/goreleaser-action@v7
|
||||
with:
|
||||
workdir: ./
|
||||
distribution: goreleaser
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
name: Update Helm charts
|
||||
|
||||
on:
|
||||
release:
|
||||
types:
|
||||
- published
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: update-helm-charts
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
update:
|
||||
name: Propose chart update
|
||||
if: ${{ github.repository_owner == 'henrygd' && startsWith(github.event.release.tag_name, 'v') && !github.event.release.prerelease }}
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
BRANCH: automation/update-helm-app-version
|
||||
RELEASE_TAG: ${{ github.event.release.tag_name }}
|
||||
AUTOMATION_TOKEN: ${{ secrets.CR_TOKEN || github.token }}
|
||||
|
||||
steps:
|
||||
- name: Checkout main
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
ref: main
|
||||
token: ${{ env.AUTOMATION_TOKEN }}
|
||||
|
||||
- name: Update chart versions
|
||||
id: update
|
||||
run: |
|
||||
version="${RELEASE_TAG#v}"
|
||||
if [[ ! "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
|
||||
echo "Unsupported software release version: $version" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
changed=false
|
||||
for chart in supplemental/helm/beszel-agent supplemental/helm/beszel-hub; do
|
||||
current_app_version=$(awk -F '"' '/^appVersion:/ { print $2 }' "$chart/Chart.yaml")
|
||||
if [[ "$current_app_version" == "$version" ]]; then
|
||||
echo "$chart already uses appVersion $version"
|
||||
continue
|
||||
fi
|
||||
|
||||
newest_version=$(printf '%s\n' "$current_app_version" "$version" | sort -V | tail -n 1)
|
||||
if [[ "$newest_version" != "$version" ]]; then
|
||||
echo "Skipping stale update of $chart from $current_app_version to $version"
|
||||
continue
|
||||
fi
|
||||
|
||||
chart_version=$(awk '/^version:/ { print $2 }' "$chart/Chart.yaml")
|
||||
if [[ ! "$chart_version" =~ ^([0-9]+)\.([0-9]+)\.([0-9]+)$ ]]; then
|
||||
echo "Unsupported chart version in $chart/Chart.yaml: $chart_version" >&2
|
||||
exit 1
|
||||
fi
|
||||
next_chart_version="${BASH_REMATCH[1]}.${BASH_REMATCH[2]}.$((BASH_REMATCH[3] + 1))"
|
||||
|
||||
NEW_APP_VERSION="$version" NEW_CHART_VERSION="$next_chart_version" \
|
||||
perl -pi -e 's/^appVersion:.*$/appVersion: "$ENV{NEW_APP_VERSION}"/; s/^version:.*$/version: $ENV{NEW_CHART_VERSION}/' \
|
||||
"$chart/Chart.yaml"
|
||||
OLD_APP_VERSION="$current_app_version" NEW_APP_VERSION="$version" \
|
||||
perl -pi -e 's/\Q$ENV{OLD_APP_VERSION}\E/$ENV{NEW_APP_VERSION}/g' "$chart/README.md"
|
||||
|
||||
echo "$chart: appVersion $current_app_version -> $version, chart $chart_version -> $next_chart_version"
|
||||
changed=true
|
||||
done
|
||||
|
||||
echo "changed=$changed" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Open or update pull request
|
||||
if: steps.update.outputs.changed == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ env.AUTOMATION_TOKEN }}
|
||||
run: |
|
||||
version="${RELEASE_TAG#v}"
|
||||
title="chore(helm): update app version to ${version}"
|
||||
body="Updates the Helm charts for [Beszel ${version}](${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/releases/tag/${RELEASE_TAG}) and bumps their chart patch versions. Merging this pull request publishes the updated charts to GHCR."
|
||||
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git checkout -B "$BRANCH"
|
||||
git add supplemental/helm/beszel-agent/Chart.yaml \
|
||||
supplemental/helm/beszel-agent/README.md \
|
||||
supplemental/helm/beszel-hub/Chart.yaml \
|
||||
supplemental/helm/beszel-hub/README.md
|
||||
git commit -m "$title"
|
||||
|
||||
git fetch origin "$BRANCH" || true
|
||||
git push --force-with-lease origin "HEAD:refs/heads/${BRANCH}"
|
||||
|
||||
pr_number=$(gh pr list --head "$BRANCH" --base main --state open --json number --jq '.[0].number')
|
||||
if [[ -n "$pr_number" ]]; then
|
||||
gh pr edit "$pr_number" --title "$title" --body "$body"
|
||||
else
|
||||
gh pr create --base main --head "$BRANCH" --title "$title" --body "$body"
|
||||
fi
|
||||
@@ -2,10 +2,6 @@
|
||||
|
||||
name: VulnCheck
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
@@ -19,11 +15,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check out code into the Go module directory
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v7
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version: 1.25.x
|
||||
go-version: stable
|
||||
# cached: false
|
||||
- name: Get official govulncheck
|
||||
run: go install golang.org/x/vuln/cmd/govulncheck@latest
|
||||
|
||||
+2
-1
@@ -3,7 +3,6 @@ pb_data
|
||||
data
|
||||
temp
|
||||
.vscode
|
||||
beszel-agent
|
||||
beszel_data
|
||||
beszel_data*
|
||||
dist
|
||||
@@ -21,3 +20,5 @@ __debug_*
|
||||
agent/lhm/obj
|
||||
agent/lhm/bin
|
||||
dockerfile_agent_dev
|
||||
.cr-release-packages
|
||||
.tmp
|
||||
|
||||
@@ -31,12 +31,16 @@ builds:
|
||||
goarch: arm64
|
||||
- goos: freebsd
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: arm
|
||||
|
||||
- id: beszel-agent
|
||||
binary: beszel-agent
|
||||
main: internal/cmd/agent/agent.go
|
||||
env:
|
||||
- CGO_ENABLED=0
|
||||
ldflags:
|
||||
- -s -w -X github.com/henrygd/beszel/internal/ghupdate.buildGOARM={{ .Arm }}
|
||||
goos:
|
||||
- linux
|
||||
- darwin
|
||||
@@ -52,6 +56,10 @@ builds:
|
||||
- mipsle
|
||||
- mips
|
||||
- ppc64le
|
||||
goarm:
|
||||
- "5"
|
||||
- "6"
|
||||
- "7"
|
||||
gomips:
|
||||
- hardfloat
|
||||
- softfloat
|
||||
@@ -71,6 +79,8 @@ builds:
|
||||
gomips: hardfloat
|
||||
- goos: windows
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: riscv64
|
||||
- goos: windows
|
||||
@@ -97,6 +107,7 @@ archives:
|
||||
{{ .Binary }}_
|
||||
{{- .Os }}_
|
||||
{{- .Arch }}
|
||||
{{- if ne .Arm "6" }}{{ with .Arm }}v{{ . }}{{ end }}{{ end }}
|
||||
format_overrides:
|
||||
- goos: windows
|
||||
formats: [zip]
|
||||
|
||||
@@ -51,9 +51,8 @@ clean:
|
||||
lint:
|
||||
golangci-lint run
|
||||
|
||||
test: export GOEXPERIMENT=synctest
|
||||
test:
|
||||
go test -tags=testing ./...
|
||||
go test -tags='testing no_ui' ./...
|
||||
|
||||
tidy:
|
||||
go mod tidy
|
||||
|
||||
+4
-2
@@ -2,6 +2,8 @@
|
||||
|
||||
## Reporting a Vulnerability
|
||||
|
||||
If you find a vulnerability in the latest version, please [submit a private advisory](https://github.com/henrygd/beszel/security/advisories/new).
|
||||
**PLEASE ONLY USE SECURITY ADVISORIES FOR REAL HIGH SEVERITY VULNERABILITIES.**
|
||||
|
||||
If it's low severity (use best judgement) you may open an issue instead of an advisory.
|
||||
If you find a vulnerability in the latest version, and it is not high severity, open an issue instead of an advisory.
|
||||
|
||||
I am overwhelmed with advisories, often erroneous, which are clearly found and written by AI. I don't have the capacity to review all of them.
|
||||
|
||||
+41
-25
@@ -6,7 +6,6 @@ package agent
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -14,11 +13,14 @@ import (
|
||||
"github.com/gliderlabs/ssh"
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/agent/deltatracker"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
gossh "golang.org/x/crypto/ssh"
|
||||
)
|
||||
|
||||
const defaultDataCacheTimeMs uint16 = 60_000
|
||||
|
||||
type Agent struct {
|
||||
sync.Mutex // Used to lock agent while collecting data
|
||||
debug bool // true if LOG_LEVEL is set to debug
|
||||
@@ -36,6 +38,7 @@ type Agent struct {
|
||||
sensorConfig *SensorConfig // Sensors config
|
||||
systemInfo system.Info // Host system info (dynamic)
|
||||
systemDetails system.Details // Host system details (static, once-per-connection)
|
||||
detailsDirty bool // Whether system details have changed and need to be resent
|
||||
gpuManager *GPUManager // Manages GPU data
|
||||
cache *systemDataCache // Cache for system stats based on cache time
|
||||
connectionManager *ConnectionManager // Channel to signal connection events
|
||||
@@ -45,6 +48,7 @@ type Agent struct {
|
||||
keys []gossh.PublicKey // SSH public keys
|
||||
smartManager *SmartManager // Manages SMART data
|
||||
systemdManager *systemdManager // Manages systemd services
|
||||
storagePoolManager *StoragePoolManager // Manages storage pool and dataset data
|
||||
}
|
||||
|
||||
// NewAgent creates a new agent with the given data directory for persisting data.
|
||||
@@ -68,11 +72,11 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
slog.Info("Data directory", "path", agent.dataDir)
|
||||
}
|
||||
|
||||
agent.memCalc, _ = GetEnv("MEM_CALC")
|
||||
agent.memCalc, _ = utils.GetEnv("MEM_CALC")
|
||||
agent.sensorConfig = agent.newSensorConfig()
|
||||
|
||||
// Parse disk usage cache duration (e.g., "15m", "1h") to avoid waking sleeping disks
|
||||
if diskUsageCache, exists := GetEnv("DISK_USAGE_CACHE"); exists {
|
||||
if diskUsageCache, exists := utils.GetEnv("DISK_USAGE_CACHE"); exists {
|
||||
if duration, err := time.ParseDuration(diskUsageCache); err == nil {
|
||||
agent.diskUsageCacheDuration = duration
|
||||
slog.Info("DISK_USAGE_CACHE", "duration", duration)
|
||||
@@ -82,7 +86,7 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
}
|
||||
|
||||
// Set up slog with a log level determined by the LOG_LEVEL env var
|
||||
if logLevelStr, exists := GetEnv("LOG_LEVEL"); exists {
|
||||
if logLevelStr, exists := utils.GetEnv("LOG_LEVEL"); exists {
|
||||
switch strings.ToLower(logLevelStr) {
|
||||
case "debug":
|
||||
agent.debug = true
|
||||
@@ -97,13 +101,13 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
slog.Debug(beszel.Version)
|
||||
|
||||
// initialize docker manager
|
||||
agent.dockerManager = newDockerManager()
|
||||
agent.dockerManager = newDockerManager(agent)
|
||||
|
||||
// initialize system info
|
||||
agent.refreshSystemDetails()
|
||||
|
||||
// SMART_INTERVAL env var to update smart data at this interval
|
||||
if smartIntervalEnv, exists := GetEnv("SMART_INTERVAL"); exists {
|
||||
if smartIntervalEnv, exists := utils.GetEnv("SMART_INTERVAL"); exists {
|
||||
if duration, err := time.ParseDuration(smartIntervalEnv); err == nil && duration > 0 {
|
||||
agent.systemDetails.SmartInterval = duration
|
||||
slog.Info("SMART_INTERVAL", "duration", duration)
|
||||
@@ -118,6 +122,19 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
// initialize handler registry
|
||||
agent.handlerRegistry = NewHandlerRegistry()
|
||||
|
||||
agent.storagePoolManager = newStoragePoolManager()
|
||||
|
||||
// Retain ZFS_INTERVAL for the shared storage pool detail refresh interval.
|
||||
if zfsIntervalEnv, exists := utils.GetEnv("ZFS_INTERVAL"); exists {
|
||||
if duration, err := time.ParseDuration(zfsIntervalEnv); err == nil && duration > 0 {
|
||||
agent.storagePoolManager.detailInterval = duration
|
||||
agent.systemDetails.ZfsInterval = duration
|
||||
slog.Info("ZFS_INTERVAL", "duration", duration)
|
||||
} else {
|
||||
slog.Warn("Invalid ZFS_INTERVAL", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// initialize disk info
|
||||
agent.initializeDiskInfo()
|
||||
|
||||
@@ -142,21 +159,12 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
|
||||
// if debugging, print stats
|
||||
if agent.debug {
|
||||
slog.Debug("Stats", "data", agent.gatherStats(common.DataRequestOptions{CacheTimeMs: 60_000, IncludeDetails: true}))
|
||||
slog.Debug("Stats", "data", agent.gatherStats(common.DataRequestOptions{CacheTimeMs: defaultDataCacheTimeMs, IncludeDetails: true}))
|
||||
}
|
||||
|
||||
return agent, nil
|
||||
}
|
||||
|
||||
// GetEnv retrieves an environment variable with a "BESZEL_AGENT_" prefix, or falls back to the unprefixed key.
|
||||
func GetEnv(key string) (value string, exists bool) {
|
||||
if value, exists = os.LookupEnv("BESZEL_AGENT_" + key); exists {
|
||||
return value, exists
|
||||
}
|
||||
// Fallback to the old unprefixed key
|
||||
return os.LookupEnv(key)
|
||||
}
|
||||
|
||||
func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedData {
|
||||
a.Lock()
|
||||
defer a.Unlock()
|
||||
@@ -173,11 +181,6 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
Info: a.systemInfo,
|
||||
}
|
||||
|
||||
// Include static system details only when requested
|
||||
if options.IncludeDetails {
|
||||
data.Details = &a.systemDetails
|
||||
}
|
||||
|
||||
// slog.Info("System data", "data", data, "cacheTimeMs", cacheTimeMs)
|
||||
|
||||
if a.dockerManager != nil {
|
||||
@@ -190,7 +193,7 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
}
|
||||
|
||||
// skip updating systemd services if cache time is not the default 60sec interval
|
||||
if a.systemdManager != nil && cacheTimeMs == 60_000 {
|
||||
if a.systemdManager != nil && cacheTimeMs == defaultDataCacheTimeMs {
|
||||
totalCount := uint16(a.systemdManager.getServiceStatsCount())
|
||||
if totalCount > 0 {
|
||||
numFailed := a.systemdManager.getFailedServiceCount()
|
||||
@@ -198,13 +201,25 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
}
|
||||
if a.systemdManager.hasFreshStats {
|
||||
data.SystemdServices = a.systemdManager.getServiceStats(nil, false)
|
||||
data.SystemdServicesUpdated = true
|
||||
// Preserve an explicit zero count so the hub can distinguish a fresh
|
||||
// empty snapshot from a response that omitted systemd data.
|
||||
if totalCount == 0 {
|
||||
data.Info.Services = []uint16{0, 0}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
data.Stats.ExtraFs = make(map[string]*system.FsStats)
|
||||
data.Info.ExtraFsPct = make(map[string]float64)
|
||||
for name, stats := range a.fsStats {
|
||||
if !stats.Root && stats.DiskTotal > 0 {
|
||||
if stats.Root {
|
||||
if stats.Name != "" {
|
||||
data.Info.RootDiskName = stats.Name
|
||||
}
|
||||
continue
|
||||
}
|
||||
if stats.DiskTotal > 0 {
|
||||
// Use custom name if available, otherwise use device name
|
||||
key := name
|
||||
if stats.Name != "" {
|
||||
@@ -213,7 +228,7 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
data.Stats.ExtraFs[key] = stats
|
||||
// Add percentages to Info struct for dashboard
|
||||
if stats.DiskTotal > 0 {
|
||||
pct := twoDecimals((stats.DiskUsed / stats.DiskTotal) * 100)
|
||||
pct := utils.TwoDecimals((stats.DiskUsed / stats.DiskTotal) * 100)
|
||||
data.Info.ExtraFsPct[key] = pct
|
||||
}
|
||||
}
|
||||
@@ -221,7 +236,8 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
slog.Debug("Extra FS", "data", data.Stats.ExtraFs)
|
||||
|
||||
a.cache.Set(data, cacheTimeMs)
|
||||
return data
|
||||
|
||||
return a.attachSystemDetails(data, cacheTimeMs, options.IncludeDetails)
|
||||
}
|
||||
|
||||
// Start initializes and starts the agent with optional WebSocket connection
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
|
||||
+56
-67
@@ -1,84 +1,73 @@
|
||||
//go:build !freebsd
|
||||
|
||||
// Package battery provides functions to check if the system has a battery and to get the battery stats.
|
||||
// Package battery provides battery information for the host and connected devices.
|
||||
package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"math"
|
||||
|
||||
"github.com/distatus/battery"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
var (
|
||||
systemHasBattery = false
|
||||
haveCheckedBattery = false
|
||||
const (
|
||||
stateUnknown uint8 = iota
|
||||
stateEmpty
|
||||
stateFull
|
||||
stateCharging
|
||||
stateDischarging
|
||||
stateIdle
|
||||
)
|
||||
|
||||
// HasReadableBattery checks if the system has a battery and returns true if it does.
|
||||
func HasReadableBattery() bool {
|
||||
if haveCheckedBattery {
|
||||
return systemHasBattery
|
||||
}
|
||||
haveCheckedBattery = true
|
||||
batteries, err := battery.GetAll()
|
||||
for _, bat := range batteries {
|
||||
if bat != nil && (bat.Full > 0 || bat.Design > 0) {
|
||||
systemHasBattery = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !systemHasBattery {
|
||||
slog.Debug("No battery found", "err", err)
|
||||
}
|
||||
return systemHasBattery
|
||||
// Battery is a readable battery reported by the operating system.
|
||||
type Battery struct {
|
||||
Name string
|
||||
Percent uint8
|
||||
State uint8
|
||||
FullChargeCapacity uint64
|
||||
HasFullChargeCapacity bool
|
||||
System bool
|
||||
}
|
||||
|
||||
// GetBatteryStats returns the current battery percent and charge state
|
||||
// percent = (current charge of all batteries) / (sum of designed/full capacity of all batteries)
|
||||
func GetBatteryStats() (batteryPercent uint8, batteryState uint8, err error) {
|
||||
if !HasReadableBattery() {
|
||||
return batteryPercent, batteryState, errors.ErrUnsupported
|
||||
var errNoBatteries = errors.New("no readable batteries")
|
||||
|
||||
// normalizeBatteries supplies stable fallback names and disambiguates duplicates.
|
||||
func normalizeBatteries(batteries []Battery) []Battery {
|
||||
nameCounts := make(map[string]int, len(batteries))
|
||||
for i := range batteries {
|
||||
// Names come from firmware (e.g. sysfs model_name) and are not guaranteed to
|
||||
// be valid UTF-8. Invalid bytes are rejected when the hub decodes the CBOR
|
||||
// payload, which drops every metric for the system, so strip them here.
|
||||
name := strings.TrimSpace(strings.ToValidUTF8(batteries[i].Name, ""))
|
||||
if name == "" {
|
||||
name = "Battery " + strconv.Itoa(i+1)
|
||||
}
|
||||
nameCounts[name]++
|
||||
if nameCounts[name] > 1 {
|
||||
name += " (" + strconv.Itoa(nameCounts[name]) + ")"
|
||||
}
|
||||
batteries[i].Name = name
|
||||
}
|
||||
batteries, err := battery.GetAll()
|
||||
// we'll handle errors later by skipping batteries with errors, rather
|
||||
// than skipping everything because of the presence of some errors.
|
||||
return batteries
|
||||
}
|
||||
|
||||
// Primary returns the representative battery. Reported full-charge capacity wins,
|
||||
// then system-scoped devices, then name for deterministic ties.
|
||||
func Primary(batteries []Battery) (Battery, bool) {
|
||||
if len(batteries) == 0 {
|
||||
return batteryPercent, batteryState, errors.New("no batteries")
|
||||
return Battery{}, false
|
||||
}
|
||||
|
||||
totalCapacity := float64(0)
|
||||
totalCharge := float64(0)
|
||||
errs, partialErrs := err.(battery.Errors)
|
||||
|
||||
batteryState = math.MaxUint8
|
||||
|
||||
for i, bat := range batteries {
|
||||
if partialErrs && errs[i] != nil {
|
||||
// if there were some errors, like missing data, skip it
|
||||
continue
|
||||
ordered := append([]Battery(nil), batteries...)
|
||||
sort.SliceStable(ordered, func(i, j int) bool {
|
||||
a, b := ordered[i], ordered[j]
|
||||
if a.HasFullChargeCapacity != b.HasFullChargeCapacity {
|
||||
return a.HasFullChargeCapacity
|
||||
}
|
||||
if bat == nil || bat.Full == 0 {
|
||||
// skip batteries with no capacity. Charge is unlikely to ever be zero, but
|
||||
// we can't guarantee that, so don't skip based on charge.
|
||||
continue
|
||||
if a.HasFullChargeCapacity && a.FullChargeCapacity != b.FullChargeCapacity {
|
||||
return a.FullChargeCapacity > b.FullChargeCapacity
|
||||
}
|
||||
totalCapacity += bat.Full
|
||||
totalCharge += min(bat.Current, bat.Full)
|
||||
if bat.State.Raw >= 0 {
|
||||
batteryState = uint8(bat.State.Raw)
|
||||
if a.System != b.System {
|
||||
return a.System
|
||||
}
|
||||
}
|
||||
|
||||
if totalCapacity == 0 || batteryState == math.MaxUint8 {
|
||||
// for macs there's sometimes a ghost battery with 0 capacity
|
||||
// https://github.com/distatus/battery/issues/34
|
||||
// Instead of skipping over those batteries, we'll check for total 0 capacity
|
||||
// and return an error. This also prevents a divide by zero.
|
||||
return batteryPercent, batteryState, errors.New("no battery capacity")
|
||||
}
|
||||
|
||||
batteryPercent = uint8(totalCharge / totalCapacity * 100)
|
||||
return batteryPercent, batteryState, nil
|
||||
return a.Name < b.Name
|
||||
})
|
||||
return ordered[0], true
|
||||
}
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
//go:build darwin
|
||||
|
||||
package battery
|
||||
|
||||
import (
|
||||
"os/exec"
|
||||
|
||||
"howett.net/plist"
|
||||
)
|
||||
|
||||
type macBattery struct {
|
||||
CurrentCapacity int `plist:"CurrentCapacity"`
|
||||
MaxCapacity int `plist:"MaxCapacity"`
|
||||
FullyCharged bool `plist:"FullyCharged"`
|
||||
IsCharging bool `plist:"IsCharging"`
|
||||
ExternalConnected bool `plist:"ExternalConnected"`
|
||||
}
|
||||
|
||||
func readMacBatteries() ([]macBattery, error) {
|
||||
out, err := exec.Command("ioreg", "-n", "AppleSmartBattery", "-r", "-a").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
var batteries []macBattery
|
||||
if _, err := plist.Unmarshal(out, &batteries); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return batteries, nil
|
||||
}
|
||||
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
// GetBatteryStats returns every readable battery reported by macOS.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
batteries, err := readMacBatteries()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(batteries) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
result := make([]Battery, 0, len(batteries))
|
||||
for _, bat := range batteries {
|
||||
if bat.MaxCapacity <= 0 {
|
||||
// skip ghost batteries with 0 capacity
|
||||
// https://github.com/distatus/battery/issues/34
|
||||
continue
|
||||
}
|
||||
percent := min(max(float64(bat.CurrentCapacity)/float64(bat.MaxCapacity)*100, 0), 100)
|
||||
state := stateUnknown
|
||||
switch {
|
||||
case !bat.ExternalConnected:
|
||||
state = stateDischarging
|
||||
case bat.IsCharging:
|
||||
state = stateCharging
|
||||
case bat.CurrentCapacity == 0:
|
||||
state = stateEmpty
|
||||
case !bat.FullyCharged:
|
||||
state = stateIdle
|
||||
default:
|
||||
state = stateFull
|
||||
}
|
||||
result = append(result, Battery{Name: "Primary", Percent: uint8(percent), State: state,
|
||||
FullChargeCapacity: uint64(bat.MaxCapacity), HasFullChargeCapacity: true, System: true})
|
||||
}
|
||||
if len(result) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
return normalizeBatteries(result), nil
|
||||
}
|
||||
@@ -1,13 +0,0 @@
|
||||
//go:build freebsd
|
||||
|
||||
package battery
|
||||
|
||||
import "errors"
|
||||
|
||||
func HasReadableBattery() bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func GetBatteryStats() (uint8, uint8, error) {
|
||||
return 0, 0, errors.ErrUnsupported
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
//go:build linux
|
||||
|
||||
package battery
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
)
|
||||
|
||||
var batteryRoot = "/sys/class/power_supply"
|
||||
|
||||
// HasReadableBattery reports whether collection currently finds a readable battery.
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
func parseSysfsState(status string) uint8 {
|
||||
switch status {
|
||||
case "Empty":
|
||||
return stateEmpty
|
||||
case "Full":
|
||||
return stateFull
|
||||
case "Charging":
|
||||
return stateCharging
|
||||
case "Discharging":
|
||||
return stateDischarging
|
||||
case "Not charging":
|
||||
return stateIdle
|
||||
default:
|
||||
return stateUnknown
|
||||
}
|
||||
}
|
||||
|
||||
// GetBatteryStats re-enumerates power supplies and returns every readable battery.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
entries, err := os.ReadDir(batteryRoot)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
batteries := make([]Battery, 0, len(entries))
|
||||
for _, entry := range entries {
|
||||
path := filepath.Join(batteryRoot, entry.Name())
|
||||
if utils.ReadStringFile(filepath.Join(path, "type")) != "Battery" {
|
||||
continue
|
||||
}
|
||||
capStr, ok := utils.ReadStringFileOK(filepath.Join(path, "capacity"))
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
cap, parseErr := strconv.Atoi(capStr)
|
||||
if parseErr != nil {
|
||||
continue
|
||||
}
|
||||
cap = min(max(cap, 0), 100)
|
||||
name := utils.ReadStringFile(filepath.Join(path, "model_name"))
|
||||
if name == "" {
|
||||
name = utils.ReadStringFile(filepath.Join(path, "model"))
|
||||
}
|
||||
if name == "" {
|
||||
name = entry.Name()
|
||||
}
|
||||
battery := Battery{
|
||||
Name: name,
|
||||
Percent: uint8(cap),
|
||||
State: parseSysfsState(utils.ReadStringFile(filepath.Join(path, "status"))),
|
||||
System: utils.ReadStringFile(filepath.Join(path, "scope")) != "Device",
|
||||
}
|
||||
for _, fullName := range []string{"charge_full", "energy_full"} {
|
||||
if parsed, ok := utils.ReadUintFile(filepath.Join(path, fullName)); ok && parsed > 0 {
|
||||
battery.FullChargeCapacity = parsed
|
||||
battery.HasFullChargeCapacity = true
|
||||
break
|
||||
}
|
||||
}
|
||||
batteries = append(batteries, battery)
|
||||
}
|
||||
if len(batteries) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
return normalizeBatteries(batteries), nil
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
//go:build testing && linux
|
||||
|
||||
package battery
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
type fakeBattery struct{ id, name, capacity, status, full, scope string }
|
||||
|
||||
func setupFakeSysfs(t *testing.T) (string, func(fakeBattery)) {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
previousRoot := batteryRoot
|
||||
batteryRoot = root
|
||||
t.Cleanup(func() { batteryRoot = previousRoot })
|
||||
write := func(path, value string) {
|
||||
t.Helper()
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(value), 0o644))
|
||||
}
|
||||
add := func(b fakeBattery) {
|
||||
t.Helper()
|
||||
dir := filepath.Join(root, b.id)
|
||||
write(filepath.Join(dir, "type"), "Battery")
|
||||
if b.capacity != "" {
|
||||
write(filepath.Join(dir, "capacity"), b.capacity)
|
||||
}
|
||||
write(filepath.Join(dir, "status"), b.status)
|
||||
if b.name != "" {
|
||||
write(filepath.Join(dir, "model_name"), b.name)
|
||||
}
|
||||
if b.full != "" {
|
||||
write(filepath.Join(dir, "energy_full"), b.full)
|
||||
}
|
||||
if b.scope != "" {
|
||||
write(filepath.Join(dir, "scope"), b.scope)
|
||||
}
|
||||
}
|
||||
return root, add
|
||||
}
|
||||
|
||||
func TestParseSysfsState(t *testing.T) {
|
||||
assert.Equal(t, stateEmpty, parseSysfsState("Empty"))
|
||||
assert.Equal(t, stateFull, parseSysfsState("Full"))
|
||||
assert.Equal(t, stateCharging, parseSysfsState("Charging"))
|
||||
assert.Equal(t, stateDischarging, parseSysfsState("Discharging"))
|
||||
assert.Equal(t, stateIdle, parseSysfsState("Not charging"))
|
||||
assert.Equal(t, stateUnknown, parseSysfsState("SomethingElse"))
|
||||
}
|
||||
|
||||
func TestGetBatteryStatsMultipleNamedAndPrimary(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", name: "Primary", capacity: "105", status: "Charging", full: "5000", scope: "System"})
|
||||
add(fakeBattery{id: "hidpp_battery_0", name: "MX Keys S", capacity: "55", status: "Unknown", full: "900", scope: "Device"})
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, batteries, 2)
|
||||
assert.Equal(t, "Primary", batteries[0].Name)
|
||||
assert.Equal(t, uint8(100), batteries[0].Percent)
|
||||
assert.Equal(t, stateUnknown, batteries[1].State)
|
||||
primary, ok := Primary(batteries)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, "Primary", primary.Name)
|
||||
}
|
||||
|
||||
func TestGetBatteryStatsFallbackDuplicatesAndUnreadable(t *testing.T) {
|
||||
root, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", name: "Keyboard", capacity: "80", status: "Discharging"})
|
||||
add(fakeBattery{id: "BAT1", name: "Keyboard", capacity: "-4", status: "SomethingWeird"})
|
||||
add(fakeBattery{id: "BAT2", capacity: "not-a-number", status: "Charging"})
|
||||
add(fakeBattery{id: "BAT3", capacity: "42", status: "Full"})
|
||||
ac := filepath.Join(root, "AC0")
|
||||
require.NoError(t, os.MkdirAll(ac, 0o755))
|
||||
require.NoError(t, os.WriteFile(filepath.Join(ac, "type"), []byte("Mains"), 0o644))
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, batteries, 3)
|
||||
assert.Equal(t, "Keyboard", batteries[0].Name)
|
||||
assert.Equal(t, "Keyboard (2)", batteries[1].Name)
|
||||
assert.Equal(t, uint8(0), batteries[1].Percent)
|
||||
assert.Equal(t, "BAT3", batteries[2].Name)
|
||||
}
|
||||
|
||||
func TestGetBatteryStatsHotPlugReenumerates(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
_, err := GetBatteryStats()
|
||||
assert.Error(t, err)
|
||||
assert.False(t, HasReadableBattery())
|
||||
add(fakeBattery{id: "BAT0", capacity: "64", status: "Discharging"})
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
assert.True(t, HasReadableBattery())
|
||||
require.Len(t, batteries, 1)
|
||||
assert.Equal(t, uint8(64), batteries[0].Percent)
|
||||
}
|
||||
|
||||
func TestGetBatteryStatsNoReadableCapacity(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", status: "Charging"})
|
||||
_, err := GetBatteryStats()
|
||||
assert.Error(t, err)
|
||||
assert.False(t, HasReadableBattery())
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
//go:build !darwin && !linux && !windows
|
||||
|
||||
package battery
|
||||
|
||||
import "errors"
|
||||
|
||||
func HasReadableBattery() bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
return nil, errors.ErrUnsupported
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package battery
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestPrimarySelection(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
bats []Battery
|
||||
want string
|
||||
}{
|
||||
{"largest reported capacity", []Battery{{Name: "Small", FullChargeCapacity: 20, HasFullChargeCapacity: true, System: true}, {Name: "Large", FullChargeCapacity: 80, HasFullChargeCapacity: true}}, "Large"},
|
||||
{"reported ranks over missing", []Battery{{Name: "Unknown", System: true}, {Name: "Known", FullChargeCapacity: 1, HasFullChargeCapacity: true}}, "Known"},
|
||||
{"system wins capacity tie", []Battery{{Name: "Peripheral", FullChargeCapacity: 50, HasFullChargeCapacity: true}, {Name: "System", FullChargeCapacity: 50, HasFullChargeCapacity: true, System: true}}, "System"},
|
||||
{"name resolves final tie", []Battery{{Name: "Zed"}, {Name: "Alpha"}}, "Alpha"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, ok := Primary(tt.bats)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, tt.want, got.Name)
|
||||
})
|
||||
}
|
||||
_, ok := Primary(nil)
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
func TestNormalizeBatteriesFallbackNames(t *testing.T) {
|
||||
bats := normalizeBatteries([]Battery{{}, {}, {Name: "Mouse"}, {Name: "Mouse"}})
|
||||
assert.Equal(t, []string{"Battery 1", "Battery 2", "Mouse", "Mouse (2)"}, []string{bats[0].Name, bats[1].Name, bats[2].Name, bats[3].Name})
|
||||
}
|
||||
|
||||
func TestNormalizeBatteriesStripsInvalidUTF8(t *testing.T) {
|
||||
// Firmware occasionally reports names that are not valid UTF-8 (a ThinkPad
|
||||
// reporting "LNV-5B11K63024@\xd0" in model_name is a real example).
|
||||
bats := normalizeBatteries([]Battery{{Name: "LNV-5B11K63024@\xd0"}, {Name: "\xff\xfe"}})
|
||||
assert.Equal(t, "LNV-5B11K63024@", bats[0].Name)
|
||||
// A name made up entirely of invalid bytes falls back to the generic name.
|
||||
assert.Equal(t, "Battery 2", bats[1].Name)
|
||||
for _, b := range bats {
|
||||
assert.True(t, utf8.ValidString(b.Name))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,291 @@
|
||||
//go:build windows
|
||||
|
||||
// Most of the Windows battery code is based on
|
||||
// distatus/battery by Karol 'Kenji Takahashi' Woźniak
|
||||
|
||||
package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
|
||||
"golang.org/x/sys/windows"
|
||||
)
|
||||
|
||||
type batteryQueryInformation struct {
|
||||
BatteryTag uint32
|
||||
InformationLevel int32
|
||||
AtRate int32
|
||||
}
|
||||
|
||||
type batteryInformation struct {
|
||||
Capabilities uint32
|
||||
Technology uint8
|
||||
Reserved [3]uint8
|
||||
Chemistry [4]uint8
|
||||
DesignedCapacity uint32
|
||||
FullChargedCapacity uint32
|
||||
DefaultAlert1 uint32
|
||||
DefaultAlert2 uint32
|
||||
CriticalBias uint32
|
||||
CycleCount uint32
|
||||
}
|
||||
|
||||
type batteryWaitStatus struct {
|
||||
BatteryTag uint32
|
||||
Timeout uint32
|
||||
PowerState uint32
|
||||
LowCapacity uint32
|
||||
HighCapacity uint32
|
||||
}
|
||||
|
||||
type batteryStatus struct {
|
||||
PowerState uint32
|
||||
Capacity uint32
|
||||
Voltage uint32
|
||||
Rate int32
|
||||
}
|
||||
|
||||
type winGUID struct {
|
||||
Data1 uint32
|
||||
Data2 uint16
|
||||
Data3 uint16
|
||||
Data4 [8]byte
|
||||
}
|
||||
|
||||
type spDeviceInterfaceData struct {
|
||||
cbSize uint32
|
||||
InterfaceClassGuid winGUID
|
||||
Flags uint32
|
||||
Reserved uint
|
||||
}
|
||||
|
||||
var guidDeviceBattery = winGUID{
|
||||
0x72631e54,
|
||||
0x78A4,
|
||||
0x11d0,
|
||||
[8]byte{0xbc, 0xf7, 0x00, 0xaa, 0x00, 0xb7, 0xb3, 0x2a},
|
||||
}
|
||||
|
||||
var (
|
||||
setupapi = &windows.LazyDLL{Name: "setupapi.dll", System: true}
|
||||
setupDiGetClassDevsW = setupapi.NewProc("SetupDiGetClassDevsW")
|
||||
setupDiEnumDeviceInterfaces = setupapi.NewProc("SetupDiEnumDeviceInterfaces")
|
||||
setupDiGetDeviceInterfaceDetailW = setupapi.NewProc("SetupDiGetDeviceInterfaceDetailW")
|
||||
setupDiDestroyDeviceInfoList = setupapi.NewProc("SetupDiDestroyDeviceInfoList")
|
||||
)
|
||||
|
||||
// winBatteryGet reads one battery by index.
|
||||
// Returns error == errNotFound when there are no more batteries.
|
||||
var errNotFound = errors.New("no more batteries")
|
||||
|
||||
func setupDiSetup(proc *windows.LazyProc, nargs, a1, a2, a3, a4, a5, a6 uintptr) (uintptr, error) {
|
||||
_ = nargs
|
||||
r1, _, errno := syscall.SyscallN(proc.Addr(), a1, a2, a3, a4, a5, a6)
|
||||
if windows.Handle(r1) == windows.InvalidHandle {
|
||||
if errno != 0 {
|
||||
return 0, error(errno)
|
||||
}
|
||||
return 0, syscall.EINVAL
|
||||
}
|
||||
return r1, nil
|
||||
}
|
||||
|
||||
func setupDiCall(proc *windows.LazyProc, nargs, a1, a2, a3, a4, a5, a6 uintptr) syscall.Errno {
|
||||
_ = nargs
|
||||
r1, _, errno := syscall.SyscallN(proc.Addr(), a1, a2, a3, a4, a5, a6)
|
||||
if r1 == 0 {
|
||||
if errno != 0 {
|
||||
return errno
|
||||
}
|
||||
return syscall.EINVAL
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func readWinBatteryState(powerState uint32) uint8 {
|
||||
switch {
|
||||
case powerState&0x00000004 != 0:
|
||||
return stateCharging
|
||||
case powerState&0x00000008 != 0:
|
||||
return stateEmpty
|
||||
case powerState&0x00000002 != 0:
|
||||
return stateDischarging
|
||||
case powerState&0x00000001 != 0:
|
||||
return stateFull
|
||||
default:
|
||||
return stateUnknown
|
||||
}
|
||||
}
|
||||
|
||||
func winBatteryGet(idx int) (Battery, error) {
|
||||
hdev, err := setupDiSetup(
|
||||
setupDiGetClassDevsW,
|
||||
4,
|
||||
uintptr(unsafe.Pointer(&guidDeviceBattery)),
|
||||
0, 0,
|
||||
2|16, // DIGCF_PRESENT|DIGCF_DEVICEINTERFACE
|
||||
0, 0,
|
||||
)
|
||||
if err != nil {
|
||||
return Battery{}, err
|
||||
}
|
||||
defer syscall.SyscallN(setupDiDestroyDeviceInfoList.Addr(), hdev)
|
||||
|
||||
var did spDeviceInterfaceData
|
||||
did.cbSize = uint32(unsafe.Sizeof(did))
|
||||
errno := setupDiCall(
|
||||
setupDiEnumDeviceInterfaces,
|
||||
5,
|
||||
hdev, 0,
|
||||
uintptr(unsafe.Pointer(&guidDeviceBattery)),
|
||||
uintptr(idx),
|
||||
uintptr(unsafe.Pointer(&did)),
|
||||
0,
|
||||
)
|
||||
if errno == 259 { // ERROR_NO_MORE_ITEMS
|
||||
return Battery{}, errNotFound
|
||||
}
|
||||
if errno != 0 {
|
||||
return Battery{}, errno
|
||||
}
|
||||
|
||||
var cbRequired uint32
|
||||
errno = setupDiCall(
|
||||
setupDiGetDeviceInterfaceDetailW,
|
||||
6,
|
||||
hdev,
|
||||
uintptr(unsafe.Pointer(&did)),
|
||||
0, 0,
|
||||
uintptr(unsafe.Pointer(&cbRequired)),
|
||||
0,
|
||||
)
|
||||
if errno != 0 && errno != 122 { // ERROR_INSUFFICIENT_BUFFER
|
||||
return Battery{}, errno
|
||||
}
|
||||
didd := make([]uint16, cbRequired/2)
|
||||
cbSize := (*uint32)(unsafe.Pointer(&didd[0]))
|
||||
if unsafe.Sizeof(uint(0)) == 8 {
|
||||
*cbSize = 8
|
||||
} else {
|
||||
*cbSize = 6
|
||||
}
|
||||
errno = setupDiCall(
|
||||
setupDiGetDeviceInterfaceDetailW,
|
||||
6,
|
||||
hdev,
|
||||
uintptr(unsafe.Pointer(&did)),
|
||||
uintptr(unsafe.Pointer(&didd[0])),
|
||||
uintptr(cbRequired),
|
||||
uintptr(unsafe.Pointer(&cbRequired)),
|
||||
0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return Battery{}, errno
|
||||
}
|
||||
devicePath := &didd[2:][0]
|
||||
|
||||
handle, err := windows.CreateFile(
|
||||
devicePath,
|
||||
windows.GENERIC_READ|windows.GENERIC_WRITE,
|
||||
windows.FILE_SHARE_READ|windows.FILE_SHARE_WRITE,
|
||||
nil,
|
||||
windows.OPEN_EXISTING,
|
||||
windows.FILE_ATTRIBUTE_NORMAL,
|
||||
0,
|
||||
)
|
||||
if err != nil {
|
||||
return Battery{}, err
|
||||
}
|
||||
defer windows.CloseHandle(handle)
|
||||
|
||||
var dwOut uint32
|
||||
var dwWait uint32
|
||||
var bqi batteryQueryInformation
|
||||
err = windows.DeviceIoControl(
|
||||
handle,
|
||||
2703424, // IOCTL_BATTERY_QUERY_TAG
|
||||
(*byte)(unsafe.Pointer(&dwWait)),
|
||||
uint32(unsafe.Sizeof(dwWait)),
|
||||
(*byte)(unsafe.Pointer(&bqi.BatteryTag)),
|
||||
uint32(unsafe.Sizeof(bqi.BatteryTag)),
|
||||
&dwOut, nil,
|
||||
)
|
||||
if err != nil || bqi.BatteryTag == 0 {
|
||||
return Battery{}, errors.New("battery tag not returned")
|
||||
}
|
||||
|
||||
var bi batteryInformation
|
||||
if err = windows.DeviceIoControl(
|
||||
handle,
|
||||
2703428, // IOCTL_BATTERY_QUERY_INFORMATION
|
||||
(*byte)(unsafe.Pointer(&bqi)),
|
||||
uint32(unsafe.Sizeof(bqi)),
|
||||
(*byte)(unsafe.Pointer(&bi)),
|
||||
uint32(unsafe.Sizeof(bi)),
|
||||
&dwOut, nil,
|
||||
); err != nil {
|
||||
return Battery{}, err
|
||||
}
|
||||
|
||||
// BatteryDeviceName is optional, so retain the deterministic fallback on error.
|
||||
name := ""
|
||||
nameQuery := bqi
|
||||
nameQuery.InformationLevel = 4 // BatteryDeviceName
|
||||
nameBuffer := make([]uint16, 128)
|
||||
if err := windows.DeviceIoControl(
|
||||
handle, 2703428,
|
||||
(*byte)(unsafe.Pointer(&nameQuery)), uint32(unsafe.Sizeof(nameQuery)),
|
||||
(*byte)(unsafe.Pointer(&nameBuffer[0])), uint32(len(nameBuffer)*2),
|
||||
&dwOut, nil,
|
||||
); err == nil {
|
||||
name = windows.UTF16ToString(nameBuffer)
|
||||
}
|
||||
|
||||
bws := batteryWaitStatus{BatteryTag: bqi.BatteryTag}
|
||||
var bs batteryStatus
|
||||
if err = windows.DeviceIoControl(
|
||||
handle,
|
||||
2703436, // IOCTL_BATTERY_QUERY_STATUS
|
||||
(*byte)(unsafe.Pointer(&bws)),
|
||||
uint32(unsafe.Sizeof(bws)),
|
||||
(*byte)(unsafe.Pointer(&bs)),
|
||||
uint32(unsafe.Sizeof(bs)),
|
||||
&dwOut, nil,
|
||||
); err != nil {
|
||||
return Battery{}, err
|
||||
}
|
||||
|
||||
if bs.Capacity == 0xffffffff || bi.FullChargedCapacity == 0 || bi.FullChargedCapacity == 0xffffffff {
|
||||
return Battery{}, errors.New("battery capacity unknown")
|
||||
}
|
||||
percent := min(float64(bs.Capacity)/float64(bi.FullChargedCapacity)*100, 100)
|
||||
return Battery{Name: name, Percent: uint8(percent), State: readWinBatteryState(bs.PowerState),
|
||||
FullChargeCapacity: uint64(bi.FullChargedCapacity), HasFullChargeCapacity: true, System: true}, nil
|
||||
}
|
||||
|
||||
// HasReadableBattery checks if the system has a battery and returns true if it does.
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
// GetBatteryStats returns every readable battery reported by Windows.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
batteries := make([]Battery, 0, 2)
|
||||
for i := 0; ; i++ {
|
||||
battery, bErr := winBatteryGet(i)
|
||||
if errors.Is(bErr, errNotFound) {
|
||||
break
|
||||
}
|
||||
if bErr != nil {
|
||||
continue
|
||||
}
|
||||
batteries = append(batteries, battery)
|
||||
}
|
||||
if len(batteries) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
return normalizeBatteries(batteries), nil
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
// Package btrfs reads btrfs filesystem state from sysfs.
|
||||
package btrfs
|
||||
|
||||
// Filesystem is a mounted btrfs filesystem read from /sys/fs/btrfs/<uuid>.
|
||||
type Filesystem struct {
|
||||
UUID string // stable filesystem UUID from sysfs
|
||||
MountID string // kernel filesystem identity for matching monitored mounts
|
||||
IODevice string // sole member block-device name, empty for multi-device/unknown pools
|
||||
Name string // label, else first mountpoint, else UUID
|
||||
Size uint64 // effective usable capacity, or raw member capacity when Raw
|
||||
Raw bool // capacity and usage are physical bytes, unsuitable for disk alerts
|
||||
Alloc uint64 // raw bytes allocated to data, metadata and system chunks
|
||||
Health string // ONLINE, or DEGRADED when a device is missing
|
||||
NRead uint64 // cumulative bytes read across member devices
|
||||
NWrite uint64 // cumulative bytes written across member devices
|
||||
Devices []Device
|
||||
}
|
||||
|
||||
// Device is one member device (devinfo/<devid>) with its error counters.
|
||||
type Device struct {
|
||||
Name string // "devid N"; sysfs does not expose the block device path
|
||||
State string // ONLINE or MISSING
|
||||
ReadErrs uint64
|
||||
WriteErrs uint64
|
||||
CorruptionErrs uint64
|
||||
}
|
||||
@@ -0,0 +1,285 @@
|
||||
//go:build linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unsafe"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
var (
|
||||
sysfsPath = "/sys/fs/btrfs"
|
||||
mountsPath = "/proc/self/mounts"
|
||||
mountinfoPath = "/proc/self/mountinfo"
|
||||
mountUUID = MountID
|
||||
deviceSize = ioctlDeviceSize
|
||||
filesystemUsage = statfsUsage
|
||||
)
|
||||
|
||||
// Filesystems returns all mounted btrfs filesystems, or nil when there are none.
|
||||
func Filesystems() ([]Filesystem, error) {
|
||||
entries, err := os.ReadDir(sysfsPath)
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
mounts := mountpointsByDevice()
|
||||
var filesystems []Filesystem
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() || entry.Name() == "features" {
|
||||
continue
|
||||
}
|
||||
fs, err := readFilesystem(filepath.Join(sysfsPath, entry.Name()), mounts)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("btrfs %s: %w", entry.Name(), err)
|
||||
}
|
||||
filesystems = append(filesystems, fs)
|
||||
}
|
||||
return filesystems, nil
|
||||
}
|
||||
|
||||
func readFilesystem(dir string, mounts map[string]string) (Filesystem, error) {
|
||||
fs := Filesystem{UUID: filepath.Base(dir), Name: utils.ReadStringFile(filepath.Join(dir, "label")), Health: "UNKNOWN"}
|
||||
for _, kind := range []string{"data", "metadata", "system"} {
|
||||
if value, ok := utils.ReadUintFile(filepath.Join(dir, "allocation", kind, "disk_used")); ok {
|
||||
fs.Alloc += value
|
||||
}
|
||||
}
|
||||
// devices/<name> links to the block device's sysfs directory.
|
||||
devices, err := os.ReadDir(filepath.Join(dir, "devices"))
|
||||
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fs, err
|
||||
}
|
||||
mountpoint := mounts["uuid:"+fs.UUID]
|
||||
if fs.Name == "" {
|
||||
fs.Name = mountpoint
|
||||
}
|
||||
var backingSize uint64
|
||||
for _, dev := range devices {
|
||||
if mountpoint == "" {
|
||||
mountpoint = mounts[dev.Name()]
|
||||
}
|
||||
if fs.Name == "" {
|
||||
fs.Name = mountpoint
|
||||
}
|
||||
devDir := filepath.Join(dir, "devices", dev.Name())
|
||||
if size, ok := utils.ReadUintFile(filepath.Join(devDir, "size")); ok {
|
||||
backingSize += size * 512
|
||||
}
|
||||
if stat := strings.Fields(utils.ReadStringFile(filepath.Join(devDir, "stat"))); len(stat) >= 7 {
|
||||
fs.NRead += parseUint(stat[2]) * 512
|
||||
fs.NWrite += parseUint(stat[6]) * 512
|
||||
}
|
||||
}
|
||||
devids, err := os.ReadDir(filepath.Join(dir, "devinfo"))
|
||||
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fs, err
|
||||
}
|
||||
capacityAvailable := len(devids) > 0
|
||||
healthKnown := len(devids) > 0
|
||||
for _, devid := range devids {
|
||||
devDir := filepath.Join(dir, "devinfo", devid.Name())
|
||||
// Replacement targets do not add filesystem capacity.
|
||||
replaceTarget, _ := utils.ReadUintFile(filepath.Join(devDir, "replace_target"))
|
||||
if replaceTarget != 1 {
|
||||
devid, err := strconv.ParseUint(devid.Name(), 10, 64)
|
||||
if err != nil {
|
||||
return fs, err
|
||||
}
|
||||
size, err := deviceSize(mountpoint, devid)
|
||||
if err != nil {
|
||||
capacityAvailable = false
|
||||
}
|
||||
fs.Size += size
|
||||
}
|
||||
dev := Device{Name: "devid " + devid.Name(), State: "ONLINE"}
|
||||
missing := utils.ReadStringFile(filepath.Join(devDir, "missing"))
|
||||
if missing != "0" && missing != "1" {
|
||||
healthKnown = false
|
||||
dev.State = "UNKNOWN"
|
||||
}
|
||||
if missing == "1" {
|
||||
dev.State = "MISSING"
|
||||
fs.Health = "DEGRADED"
|
||||
}
|
||||
for line := range strings.Lines(utils.ReadStringFile(filepath.Join(devDir, "error_stats"))) {
|
||||
if fields := strings.Fields(line); len(fields) == 2 {
|
||||
switch fields[0] {
|
||||
case "read_errs":
|
||||
dev.ReadErrs = parseUint(fields[1])
|
||||
case "write_errs":
|
||||
dev.WriteErrs = parseUint(fields[1])
|
||||
case "corruption_errs":
|
||||
dev.CorruptionErrs = parseUint(fields[1])
|
||||
}
|
||||
}
|
||||
}
|
||||
fs.Devices = append(fs.Devices, dev)
|
||||
}
|
||||
// Use one capacity source for the whole filesystem: device IDs cannot be
|
||||
// reliably matched to block-device names in sysfs. A partial ioctl result
|
||||
// must not be added to the complete backing-device total.
|
||||
if !capacityAvailable {
|
||||
fs.Size = backingSize
|
||||
}
|
||||
if fs.Health != "DEGRADED" && healthKnown {
|
||||
fs.Health = "ONLINE"
|
||||
}
|
||||
fs.MountID = mountUUID(mountpoint)
|
||||
if len(devices) == 1 && len(devids) == 1 && fs.Health == "ONLINE" {
|
||||
fs.IODevice = devices[0].Name()
|
||||
}
|
||||
fs.Raw = true
|
||||
if used, available, err := filesystemUsage(mountpoint); err == nil {
|
||||
// Effective capacity excludes reserved/unavailable space, so Size-Alloc
|
||||
// is available to applications and the usage ratio matches df.
|
||||
fs.Size, fs.Alloc, fs.Raw = used+available, used, false
|
||||
}
|
||||
if fs.Name == "" {
|
||||
fs.Name = filepath.Base(dir)
|
||||
}
|
||||
return fs, nil
|
||||
}
|
||||
|
||||
// mountpointsByDevice prefers UUID matches from mountinfo and retains source
|
||||
// device names as a fallback for environments where FS_INFO is unavailable.
|
||||
func mountpointsByDevice() map[string]string {
|
||||
mounts := mountpointsByUUID(utils.ReadStringFile(mountinfoPath), mountUUID)
|
||||
for line := range strings.Lines(utils.ReadStringFile(mountsPath)) {
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 3 || fields[2] != "btrfs" {
|
||||
continue
|
||||
}
|
||||
device := fields[0]
|
||||
if resolved, err := filepath.EvalSymlinks(device); err == nil {
|
||||
device = resolved
|
||||
}
|
||||
if _, seen := mounts[filepath.Base(device)]; !seen {
|
||||
mounts[filepath.Base(device)] = unescapeMountPath(fields[1])
|
||||
}
|
||||
}
|
||||
return mounts
|
||||
}
|
||||
|
||||
func parseUint(s string) uint64 {
|
||||
n, _ := strconv.ParseUint(s, 10, 64)
|
||||
return n
|
||||
}
|
||||
|
||||
// ioctlDeviceSize reads Btrfs's recorded device size, which can be smaller
|
||||
// than the block device after a filesystem resize. BTRFS_IOC_DEV_INFO is
|
||||
// _IOWR(0x94, 30, struct btrfs_ioctl_dev_info_args), a 4096-byte ABI structure.
|
||||
func ioctlDeviceSize(mountpoint string, devid uint64) (uint64, error) {
|
||||
if mountpoint == "" {
|
||||
return 0, errors.New("no accessible mountpoint")
|
||||
}
|
||||
f, err := os.Open(mountpoint)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer f.Close()
|
||||
args := struct {
|
||||
Devid uint64
|
||||
UUID [16]byte
|
||||
BytesUsed uint64
|
||||
TotalBytes uint64
|
||||
Reserved [4096 - 40]byte
|
||||
}{Devid: devid}
|
||||
_, _, errno := unix.Syscall(unix.SYS_IOCTL, f.Fd(), 0xd000941e, uintptr(unsafe.Pointer(&args)))
|
||||
if errno != 0 {
|
||||
return 0, errno
|
||||
}
|
||||
return args.TotalBytes, nil
|
||||
}
|
||||
|
||||
// The filesystem magic is unsigned even when Statfs_t.Type is int32.
|
||||
func isBtrfs(stat *unix.Statfs_t) bool {
|
||||
return uint32(stat.Type) == unix.BTRFS_SUPER_MAGIC
|
||||
}
|
||||
|
||||
func statfsUsage(path string) (used, available uint64, err error) {
|
||||
if path == "" {
|
||||
return 0, 0, errors.New("no accessible mountpoint")
|
||||
}
|
||||
var stat unix.Statfs_t
|
||||
if err = unix.Statfs(path, &stat); err != nil {
|
||||
return
|
||||
}
|
||||
if !isBtrfs(&stat) {
|
||||
return 0, 0, errors.New("mountpoint is not Btrfs")
|
||||
}
|
||||
blockSize := uint64(stat.Bsize)
|
||||
return (stat.Blocks - min(stat.Blocks, stat.Bfree)) * blockSize, min(stat.Blocks, stat.Bavail) * blockSize, nil
|
||||
}
|
||||
|
||||
// MountID returns the filesystem UUID via BTRFS_IOC_FS_INFO. Unlike statfs
|
||||
// f_fsid, this identity is shared by all subvolumes and bind mounts.
|
||||
func MountID(path string) string {
|
||||
if path == "" {
|
||||
return ""
|
||||
}
|
||||
var stat unix.Statfs_t
|
||||
if unix.Statfs(path, &stat) != nil || !isBtrfs(&stat) {
|
||||
return ""
|
||||
}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer f.Close()
|
||||
args := struct {
|
||||
MaxID uint64
|
||||
NumDevices uint64
|
||||
FSID [16]byte
|
||||
Reserved [992]byte
|
||||
}{}
|
||||
// _IOR(0x94, 31, 1024). Reuse the platform's read-direction bits;
|
||||
// MIPS/PowerPC use a different encoding than asm-generic.
|
||||
request := uintptr(unix.FS_IOC_GETFLAGS&0xe0000000) | 0x0400941f
|
||||
_, _, errno := unix.Syscall(unix.SYS_IOCTL, f.Fd(), request, uintptr(unsafe.Pointer(&args)))
|
||||
if errno != 0 {
|
||||
return ""
|
||||
}
|
||||
id := args.FSID
|
||||
return fmt.Sprintf("%x-%x-%x-%x-%x", id[:4], id[4:6], id[6:8], id[8:10], id[10:])
|
||||
}
|
||||
|
||||
// Btrfs mountinfo device numbers can be virtual (0:N), so query the UUID
|
||||
// through the mount instead of comparing those numbers with sysfs block devs.
|
||||
// Retry another path when a bind mount is inaccessible. Once resolved, reuse
|
||||
// the result for that mount device to avoid opening every Docker bind mount.
|
||||
func mountpointsByUUID(mountinfo string, identify func(string) string) map[string]string {
|
||||
mounts := make(map[string]string)
|
||||
resolved := make(map[string]bool)
|
||||
for line := range strings.Lines(mountinfo) {
|
||||
before, after, ok := strings.Cut(line, " - ")
|
||||
fields, fs := strings.Fields(before), strings.Fields(after)
|
||||
if !ok || len(fields) < 6 || len(fs) < 3 || fs[0] != "btrfs" || resolved[fields[2]] {
|
||||
continue
|
||||
}
|
||||
path := unescapeMountPath(fields[4])
|
||||
uuid := identify(path)
|
||||
if uuid == "" {
|
||||
continue
|
||||
}
|
||||
resolved[fields[2]] = true
|
||||
if mounts["uuid:"+uuid] == "" {
|
||||
mounts["uuid:"+uuid] = path
|
||||
}
|
||||
}
|
||||
return mounts
|
||||
}
|
||||
|
||||
func unescapeMountPath(path string) string {
|
||||
return strings.NewReplacer(`\040`, " ", `\011`, "\t", `\012`, "\n", `\134`, `\`).Replace(path)
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
//go:build testing && linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
func TestFilesystems(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
oldSysfs, oldMounts := sysfsPath, mountsPath
|
||||
sysfsPath, mountsPath = root, filepath.Join(root, "mounts")
|
||||
t.Cleanup(func() { sysfsPath, mountsPath = oldSysfs, oldMounts })
|
||||
|
||||
fsDir := filepath.Join(root, "1b2c3d4e-0000-0000-0000-000000000000")
|
||||
write := func(rel, content string) {
|
||||
path := filepath.Join(fsDir, rel)
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(content), 0o644))
|
||||
}
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "features"), 0o755))
|
||||
oldUsage := filesystemUsage
|
||||
filesystemUsage = func(string) (uint64, uint64, error) { return 0, 0, os.ErrNotExist }
|
||||
t.Cleanup(func() { filesystemUsage = oldUsage })
|
||||
oldDeviceSize := deviceSize
|
||||
t.Cleanup(func() { deviceSize = oldDeviceSize })
|
||||
deviceSize = func(_ string, devid uint64) (uint64, error) {
|
||||
value, _ := utils.ReadUintFile(filepath.Join(fsDir, "recorded-size", strconv.FormatUint(devid, 10)))
|
||||
return value, nil
|
||||
}
|
||||
// Recorded member capacities differ from the unchanged backing devices.
|
||||
write("recorded-size/1", "256000\n")
|
||||
write("recorded-size/2", "128000\n")
|
||||
write("label", "tank\n")
|
||||
write("allocation/data/disk_used", "4096\n")
|
||||
write("allocation/metadata/disk_used", "2048\n")
|
||||
write("allocation/system/disk_used", "1024\n")
|
||||
write("devices/sda/size", "1000\n")
|
||||
write("devices/sda/stat", "10 0 200 0 20 0 400 0 0 0 0\n")
|
||||
write("devices/sdb/size", "1000\n")
|
||||
write("devices/sdb/stat", "10 0 100 0 20 0 100 0 0 0 0\n")
|
||||
write("devinfo/1/missing", "0\n")
|
||||
write("devinfo/1/error_stats", "write_errs 1\nread_errs 2\nflush_errs 0\ncorruption_errs 3\ngeneration_errs 0\n")
|
||||
write("devinfo/2/missing", "1\n")
|
||||
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, Filesystem{
|
||||
UUID: "1b2c3d4e-0000-0000-0000-000000000000", Raw: true, Name: "tank", Size: 384000, Alloc: 7168, Health: "DEGRADED", NRead: 153600, NWrite: 256000,
|
||||
Devices: []Device{
|
||||
{Name: "devid 1", State: "ONLINE", ReadErrs: 2, WriteErrs: 1, CorruptionErrs: 3},
|
||||
{Name: "devid 2", State: "MISSING"},
|
||||
},
|
||||
}, filesystems[0])
|
||||
|
||||
// Unlabeled filesystems fall back to the first mountpoint, then the UUID.
|
||||
write("label", "\n")
|
||||
require.NoError(t, os.WriteFile(mountsPath, []byte(
|
||||
"/dev/sdz1 /other btrfs rw 0 0\n/dev/sdb /mnt/storage btrfs rw 0 0\n/dev/sdb /mnt/storage/sub btrfs rw,subvol=/sub 0 0\n",
|
||||
), 0o644))
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "/mnt/storage", filesystems[0].Name)
|
||||
|
||||
require.NoError(t, os.Remove(mountsPath))
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "1b2c3d4e-0000-0000-0000-000000000000", filesystems[0].Name)
|
||||
write("devinfo/3/replace_target", "1\n")
|
||||
write("recorded-size/3", "512000\n")
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(384000), filesystems[0].Size, "replacement target must not inflate capacity")
|
||||
|
||||
deviceSize = func(string, uint64) (uint64, error) { return 0, os.ErrPermission }
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
assert.Equal(t, "DEGRADED", filesystems[0].Health)
|
||||
assert.Equal(t, uint64(153600), filesystems[0].NRead)
|
||||
|
||||
// A partial ioctl result must not be mixed with the backing-device total.
|
||||
deviceSize = func(_ string, devid uint64) (uint64, error) {
|
||||
if devid == 2 {
|
||||
return 0, os.ErrPermission
|
||||
}
|
||||
return 256000, nil
|
||||
}
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
|
||||
// With no mount visible (e.g. Docker), the real lookup falls back too.
|
||||
deviceSize = ioctlDeviceSize
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
|
||||
filesystemUsage = func(string) (uint64, uint64, error) { return 100, 900, nil }
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1000), filesystems[0].Size)
|
||||
assert.Equal(t, uint64(100), filesystems[0].Alloc)
|
||||
assert.False(t, filesystems[0].Raw)
|
||||
}
|
||||
|
||||
func TestFilesystemsNoBtrfs(t *testing.T) {
|
||||
oldPath := sysfsPath
|
||||
sysfsPath = filepath.Join(t.TempDir(), "missing")
|
||||
t.Cleanup(func() { sysfsPath = oldPath })
|
||||
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Nil(t, filesystems)
|
||||
}
|
||||
|
||||
func TestIoctlDeviceSizeFailure(t *testing.T) {
|
||||
_, err := ioctlDeviceSize("", 1)
|
||||
require.Error(t, err)
|
||||
_, err = ioctlDeviceSize(t.TempDir(), 1)
|
||||
require.Error(t, err)
|
||||
assert.ErrorIs(t, err, unix.ENOTTY)
|
||||
}
|
||||
|
||||
func TestMountpointsDecodeEscapes(t *testing.T) {
|
||||
oldMounts := mountsPath
|
||||
mountsPath = filepath.Join(t.TempDir(), "mounts")
|
||||
t.Cleanup(func() { mountsPath = oldMounts })
|
||||
require.NoError(t, os.WriteFile(mountsPath, []byte("/dev/test-btrfs /mnt/my\\040data btrfs rw 0 0\n"), 0o644))
|
||||
assert.Equal(t, "/mnt/my data", mountpointsByDevice()["test-btrfs"])
|
||||
}
|
||||
|
||||
func TestFilesystemWithoutDevinfo(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "devices", "sda"), 0755))
|
||||
require.NoError(t, os.WriteFile(filepath.Join(root, "devices", "sda", "size"), []byte("1000"), 0644))
|
||||
fs, err := readFilesystem(root, nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(512000), fs.Size)
|
||||
assert.True(t, fs.Raw)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
assert.Empty(t, fs.Devices)
|
||||
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "devinfo", "1"), 0755))
|
||||
fs, err = readFilesystem(root, nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
require.Len(t, fs.Devices, 1)
|
||||
assert.Equal(t, "UNKNOWN", fs.Devices[0].State)
|
||||
|
||||
// Some older interfaces lack the devices directory too.
|
||||
fs, err = readFilesystem(t.TempDir(), nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
}
|
||||
|
||||
func TestLocalBtrfsUsage(t *testing.T) {
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for read-only live validation")
|
||||
}
|
||||
used, available, err := statfsUsage(path)
|
||||
require.NoError(t, err)
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
for _, fs := range filesystems {
|
||||
if !fs.Raw && fs.Alloc == used && fs.Size == used+available {
|
||||
t.Logf("pool=%s used=%d available=%d effective_capacity=%d", fs.Name, used, available, fs.Size)
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatal("collector did not report the mounted filesystem's usable capacity")
|
||||
}
|
||||
|
||||
func TestMountID(t *testing.T) {
|
||||
assert.Empty(t, MountID(""))
|
||||
assert.Empty(t, MountID(filepath.Join(t.TempDir(), "missing")))
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for live identity validation")
|
||||
}
|
||||
id := MountID(path)
|
||||
require.NotEmpty(t, id)
|
||||
assert.Equal(t, id, MountID(filepath.Join(path, ".")))
|
||||
}
|
||||
|
||||
func TestMountinfoUUIDLookup(t *testing.T) {
|
||||
info := `1 0 0:40 /@ /inaccessible ro shared:1 - btrfs /dev/mapper/unavailable rw
|
||||
2 0 0:40 /@/docker/hosts /etc/hosts ro - btrfs /dev/mapper/unavailable rw
|
||||
3 0 0:40 /@/docker/hostname /etc/hostname ro - btrfs /dev/mapper/unavailable rw
|
||||
4 0 0:41 /subvol /extra-filesystems/my\040disk ro master:2 - btrfs /dev/missing rw
|
||||
5 0 0:42 / /ext4 ro - ext4 /dev/mapper/unavailable rw
|
||||
malformed
|
||||
6 0 0:43 / /bad ro - btrfs
|
||||
`
|
||||
var calls []string
|
||||
mounts := mountpointsByUUID(info, func(path string) string {
|
||||
calls = append(calls, path)
|
||||
switch path {
|
||||
case "/etc/hosts":
|
||||
return "root-uuid"
|
||||
case "/extra-filesystems/my disk":
|
||||
return "extra-uuid"
|
||||
}
|
||||
return ""
|
||||
})
|
||||
assert.Equal(t, map[string]string{"uuid:root-uuid": "/etc/hosts", "uuid:extra-uuid": "/extra-filesystems/my disk"}, mounts)
|
||||
assert.Equal(t, []string{"/inaccessible", "/etc/hosts", "/extra-filesystems/my disk"}, calls)
|
||||
}
|
||||
|
||||
func TestDockerFilesystemWithoutDeviceNodes(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
oldSysfs, oldMounts, oldInfo, oldUUID, oldUsage := sysfsPath, mountsPath, mountinfoPath, mountUUID, filesystemUsage
|
||||
t.Cleanup(func() {
|
||||
sysfsPath, mountsPath, mountinfoPath, mountUUID, filesystemUsage = oldSysfs, oldMounts, oldInfo, oldUUID, oldUsage
|
||||
})
|
||||
sysfsPath = filepath.Join(root, "sysfs")
|
||||
mountsPath = filepath.Join(root, "missing-mounts")
|
||||
mountinfoPath = filepath.Join(root, "mountinfo")
|
||||
uuid := "11111111-1111-4111-8111-111111111111"
|
||||
dir := filepath.Join(sysfsPath, uuid)
|
||||
for path, content := range map[string]string{"devices/dm-0/size": "1000", "devinfo/1/missing": "0"} {
|
||||
target := filepath.Join(dir, path)
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(target), 0755))
|
||||
require.NoError(t, os.WriteFile(target, []byte(content), 0644))
|
||||
}
|
||||
require.NoError(t, os.WriteFile(mountinfoPath, []byte("2 1 0:40 /@/docker/hosts /etc/hosts ro - btrfs /dev/mapper/not-in-container rw\n"), 0644))
|
||||
mountUUID = func(path string) string {
|
||||
if path == "/etc/hosts" {
|
||||
return uuid
|
||||
}
|
||||
return ""
|
||||
}
|
||||
filesystemUsage = func(path string) (uint64, uint64, error) { require.Equal(t, "/etc/hosts", path); return 100, 900, nil }
|
||||
fs, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, fs, 1)
|
||||
assert.Equal(t, uuid, fs[0].MountID)
|
||||
assert.Equal(t, "dm-0", fs[0].IODevice)
|
||||
assert.False(t, fs[0].Raw)
|
||||
assert.Equal(t, uint64(1000), fs[0].Size)
|
||||
}
|
||||
|
||||
func TestLivePoolMountIdentity(t *testing.T) {
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for live validation")
|
||||
}
|
||||
id := MountID(path)
|
||||
require.NotEmpty(t, id)
|
||||
pools, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
for _, pool := range pools {
|
||||
if pool.UUID != id {
|
||||
continue
|
||||
}
|
||||
assert.Equal(t, id, pool.MountID)
|
||||
assert.False(t, pool.Raw)
|
||||
t.Logf("uuid=%s mount_identity=%s io_device=%s raw=%v", pool.UUID, pool.MountID, pool.IODevice, pool.Raw)
|
||||
return
|
||||
}
|
||||
t.Fatal("mounted Btrfs filesystem was not discovered")
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
//go:build !linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import "errors"
|
||||
|
||||
func Filesystems() ([]Filesystem, error) {
|
||||
return nil, errors.ErrUnsupported
|
||||
}
|
||||
|
||||
func MountID(string) string { return "" }
|
||||
+90
-10
@@ -2,6 +2,7 @@ package agent
|
||||
|
||||
import (
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
@@ -14,17 +15,38 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/lxzan/gws"
|
||||
"golang.org/x/crypto/ssh"
|
||||
"golang.org/x/net/proxy"
|
||||
)
|
||||
|
||||
const (
|
||||
wsDeadline = 70 * time.Second
|
||||
// Keep the connection alive long enough for a slow collection cycle to
|
||||
// finish before the hub considers the agent disconnected.
|
||||
wsDeadline = 120 * time.Second
|
||||
)
|
||||
|
||||
// errNoHubURL is returned when HUB_URL is unset. This is not a failure
|
||||
// condition: an agent configured with only a public key runs in SSH-only mode,
|
||||
// where the hub dials the agent and no outbound WebSocket client is expected.
|
||||
var errNoHubURL = errors.New("HUB_URL environment variable not set")
|
||||
|
||||
type caCertFileError struct {
|
||||
err error
|
||||
}
|
||||
|
||||
func (e *caCertFileError) Error() string {
|
||||
return e.err.Error()
|
||||
}
|
||||
|
||||
func (e *caCertFileError) Unwrap() error {
|
||||
return e.err
|
||||
}
|
||||
|
||||
// WebSocketClient manages the WebSocket connection between the agent and hub.
|
||||
// It handles authentication, message routing, and connection lifecycle management.
|
||||
type WebSocketClient struct {
|
||||
@@ -38,27 +60,32 @@ type WebSocketClient struct {
|
||||
hubRequest *common.HubRequest[cbor.RawMessage] // Reusable request structure for message parsing
|
||||
lastConnectAttempt time.Time // Timestamp of last connection attempt
|
||||
hubVerified bool // Whether the hub has been cryptographically verified
|
||||
tlsConfig *tls.Config // Optional TLS configuration with custom CA certificates
|
||||
}
|
||||
|
||||
// newWebSocketClient creates a new WebSocket client for the given agent.
|
||||
// It reads configuration from environment variables and validates the hub URL.
|
||||
func newWebSocketClient(agent *Agent) (client *WebSocketClient, err error) {
|
||||
hubURLStr, exists := GetEnv("HUB_URL")
|
||||
hubURLStr, exists := utils.GetEnv("HUB_URL")
|
||||
if !exists {
|
||||
return nil, errors.New("HUB_URL environment variable not set")
|
||||
return nil, errNoHubURL
|
||||
}
|
||||
|
||||
client = &WebSocketClient{}
|
||||
|
||||
client.hubURL, err = url.Parse(hubURLStr)
|
||||
if err != nil {
|
||||
return nil, errors.New("invalid hub URL")
|
||||
if err != nil || client.hubURL.Host == "" {
|
||||
return nil, fmt.Errorf("invalid HUB_URL %q: must include scheme and host (e.g. http://hub.example.com:8090)", hubURLStr)
|
||||
}
|
||||
// get registration token
|
||||
client.token, err = getToken()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
client.tlsConfig, err = getTLSConfig()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
client.agent = agent
|
||||
client.hubRequest = &common.HubRequest[cbor.RawMessage]{}
|
||||
@@ -72,12 +99,12 @@ func newWebSocketClient(agent *Agent) (client *WebSocketClient, err error) {
|
||||
// If neither is set, it returns an error.
|
||||
func getToken() (string, error) {
|
||||
// get token from env var
|
||||
token, _ := GetEnv("TOKEN")
|
||||
token, _ := utils.GetEnv("TOKEN")
|
||||
if token != "" {
|
||||
return token, nil
|
||||
}
|
||||
// get token from file
|
||||
tokenFile, _ := GetEnv("TOKEN_FILE")
|
||||
tokenFile, _ := utils.GetEnv("TOKEN_FILE")
|
||||
if tokenFile == "" {
|
||||
return "", errors.New("must set TOKEN or TOKEN_FILE")
|
||||
}
|
||||
@@ -85,7 +112,52 @@ func getToken() (string, error) {
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return strings.TrimSpace(string(tokenBytes)), nil
|
||||
return parseTokenFile(string(tokenBytes), tokenFile)
|
||||
}
|
||||
|
||||
// parseTokenFile reads a single token from TOKEN_FILE.
|
||||
// Blank lines and comments are ignored. Multiple tokens are rejected because
|
||||
// the agent supports only one outbound hub connection.
|
||||
func parseTokenFile(contents, path string) (string, error) {
|
||||
var token string
|
||||
for line := range strings.Lines(contents) {
|
||||
line = strings.TrimSpace(line)
|
||||
if len(line) == 0 || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
if token != "" {
|
||||
return "", fmt.Errorf("%s must contain a single token", path)
|
||||
}
|
||||
token = line
|
||||
}
|
||||
// An empty file keeps returning an empty token, as before: the caller decides
|
||||
// what to do about it.
|
||||
return token, nil
|
||||
}
|
||||
|
||||
// getTLSConfig returns a TLS configuration containing the system certificate
|
||||
// pool plus any certificates configured through CA_CERT_FILE. A nil config lets
|
||||
// gws use Go's default TLS configuration and system roots.
|
||||
func getTLSConfig() (*tls.Config, error) {
|
||||
caCertFile, _ := utils.GetEnv("CA_CERT_FILE")
|
||||
if caCertFile == "" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
caCertPEM, err := os.ReadFile(caCertFile)
|
||||
if err != nil {
|
||||
return nil, &caCertFileError{fmt.Errorf("read CA_CERT_FILE %q: %w", caCertFile, err)}
|
||||
}
|
||||
|
||||
rootCAs, err := x509.SystemCertPool()
|
||||
if err != nil {
|
||||
return nil, &caCertFileError{fmt.Errorf("load system CA certificate pool: %w", err)}
|
||||
}
|
||||
if !rootCAs.AppendCertsFromPEM(caCertPEM) {
|
||||
return nil, &caCertFileError{fmt.Errorf("CA_CERT_FILE %q does not contain any valid PEM certificates", caCertFile)}
|
||||
}
|
||||
|
||||
return &tls.Config{RootCAs: rootCAs}, nil
|
||||
}
|
||||
|
||||
// getOptions returns the WebSocket client options, creating them if necessary.
|
||||
@@ -103,14 +175,22 @@ func (client *WebSocketClient) getOptions() *gws.ClientOption {
|
||||
}
|
||||
client.hubURL.Path = path.Join(client.hubURL.Path, "api/beszel/agent-connect")
|
||||
|
||||
// make sure BESZEL_AGENT_ALL_PROXY works (GWS only checks ALL_PROXY)
|
||||
if val := os.Getenv("BESZEL_AGENT_ALL_PROXY"); val != "" {
|
||||
os.Setenv("ALL_PROXY", val)
|
||||
}
|
||||
|
||||
client.options = &gws.ClientOption{
|
||||
Addr: client.hubURL.String(),
|
||||
TlsConfig: &tls.Config{InsecureSkipVerify: true},
|
||||
TlsConfig: client.tlsConfig,
|
||||
RequestHeader: http.Header{
|
||||
"User-Agent": []string{getUserAgent()},
|
||||
"X-Token": []string{client.token},
|
||||
"X-Beszel": []string{beszel.Version},
|
||||
},
|
||||
NewDialer: func() (gws.Dialer, error) {
|
||||
return proxy.FromEnvironment(), nil
|
||||
},
|
||||
}
|
||||
return client.options
|
||||
}
|
||||
@@ -197,7 +277,7 @@ func (client *WebSocketClient) handleAuthChallenge(msg *common.HubRequest[cbor.R
|
||||
}
|
||||
|
||||
if authRequest.NeedSysInfo {
|
||||
response.Name, _ = GetEnv("SYSTEM_NAME")
|
||||
response.Name, _ = utils.GetEnv("SYSTEM_NAME")
|
||||
response.Hostname = client.agent.systemDetails.Hostname
|
||||
serverAddr := client.agent.connectionManager.serverOptions.Addr
|
||||
_, response.Port, _ = net.SplitHostPort(serverAddr)
|
||||
|
||||
+260
-89
@@ -1,12 +1,22 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"crypto/ed25519"
|
||||
"crypto/rand"
|
||||
"crypto/rsa"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"crypto/x509/pkix"
|
||||
"encoding/pem"
|
||||
"math/big"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -16,11 +26,34 @@ import (
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/lxzan/gws"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/crypto/ssh"
|
||||
)
|
||||
|
||||
// TestNewWebSocketClientNoHubURL verifies that an unset HUB_URL returns the
|
||||
// errNoHubURL sentinel rather than an opaque error. Callers rely on this to
|
||||
// distinguish SSH-only mode -- a supported configuration in which the hub dials
|
||||
// the agent -- from an actual misconfiguration.
|
||||
func TestNewWebSocketClientNoHubURL(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
|
||||
// t.Setenv registers restoration of the original value; unset afterwards so
|
||||
// GetEnv's LookupEnv reports the variable as absent rather than empty.
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "")
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
t.Setenv("HUB_URL", "")
|
||||
os.Unsetenv("HUB_URL")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, client)
|
||||
assert.ErrorIs(t, err, errNoHubURL)
|
||||
}
|
||||
|
||||
// TestNewWebSocketClient tests WebSocket client creation
|
||||
func TestNewWebSocketClient(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -52,11 +85,18 @@ func TestNewWebSocketClient(t *testing.T) {
|
||||
errorMsg: "HUB_URL environment variable not set",
|
||||
},
|
||||
{
|
||||
name: "invalid URL",
|
||||
name: "malformed URL",
|
||||
hubURL: "ht\ttp://invalid",
|
||||
token: "test-token",
|
||||
expectError: true,
|
||||
errorMsg: "invalid hub URL",
|
||||
errorMsg: "invalid HUB_URL",
|
||||
},
|
||||
{
|
||||
name: "URL without host",
|
||||
hubURL: "http:/api",
|
||||
token: "test-token",
|
||||
expectError: true,
|
||||
errorMsg: "invalid HUB_URL",
|
||||
},
|
||||
{
|
||||
name: "missing token",
|
||||
@@ -71,19 +111,11 @@ func TestNewWebSocketClient(t *testing.T) {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// Set up environment
|
||||
if tc.hubURL != "" {
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", tc.hubURL)
|
||||
} else {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", tc.hubURL)
|
||||
}
|
||||
if tc.token != "" {
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", tc.token)
|
||||
} else {
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", tc.token)
|
||||
}
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
|
||||
@@ -139,12 +171,8 @@ func TestWebSocketClient_GetOptions(t *testing.T) {
|
||||
for _, tc := range testCases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// Set up environment
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", tc.inputURL)
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", tc.inputURL)
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -170,6 +198,155 @@ func TestWebSocketClient_GetOptions(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestWebSocketClient_TLSVerification(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
serverCert, serverCertPEM := newSelfSignedServerCertificate(t)
|
||||
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
|
||||
server := httptest.NewUnstartedServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
conn, err := upgrader.Upgrade(w, r)
|
||||
if err == nil {
|
||||
go conn.ReadLoop()
|
||||
}
|
||||
}))
|
||||
server.TLS = &tls.Config{Certificates: []tls.Certificate{serverCert}}
|
||||
server.StartTLS()
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
caCertFile := filepath.Join(t.TempDir(), "hub-ca.crt")
|
||||
require.NoError(t, os.WriteFile(caCertFile, serverCertPEM, 0600))
|
||||
|
||||
newClient := func(t *testing.T, caCertFile string) *WebSocketClient {
|
||||
t.Helper()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", caCertFile)
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
return client
|
||||
}
|
||||
|
||||
t.Run("system roots are used by default", func(t *testing.T) {
|
||||
client := newClient(t, "")
|
||||
assert.Nil(t, client.getOptions().TlsConfig)
|
||||
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("custom CA trusts self-signed certificate", func(t *testing.T) {
|
||||
systemRoots, err := x509.SystemCertPool()
|
||||
require.NoError(t, err)
|
||||
client := newClient(t, caCertFile)
|
||||
assert.Greater(t, len(client.getOptions().TlsConfig.RootCAs.Subjects()), len(systemRoots.Subjects()))
|
||||
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, conn.NetConn().Close())
|
||||
})
|
||||
|
||||
t.Run("custom CA does not bypass hostname verification", func(t *testing.T) {
|
||||
client := newClient(t, caCertFile)
|
||||
client.getOptions().TlsConfig.ServerName = "wrong.example.com"
|
||||
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
func TestWebSocketClient_NonTLSConnection(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
conn, err := upgrader.Upgrade(w, r)
|
||||
if err == nil {
|
||||
go conn.ReadLoop()
|
||||
}
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", "")
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
assert.Nil(t, client.getOptions().TlsConfig)
|
||||
|
||||
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, conn.NetConn().Close())
|
||||
}
|
||||
|
||||
func TestGetTLSConfigErrors(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
testCases := []struct {
|
||||
name string
|
||||
path string
|
||||
contents []byte
|
||||
errorMatch string
|
||||
}{
|
||||
{
|
||||
name: "missing file",
|
||||
path: filepath.Join(tempDir, "missing.pem"),
|
||||
errorMatch: "read CA_CERT_FILE",
|
||||
},
|
||||
{
|
||||
name: "unreadable path",
|
||||
path: tempDir,
|
||||
errorMatch: "read CA_CERT_FILE",
|
||||
},
|
||||
{
|
||||
name: "empty file",
|
||||
path: filepath.Join(tempDir, "empty.pem"),
|
||||
contents: []byte{},
|
||||
errorMatch: "does not contain any valid PEM certificates",
|
||||
},
|
||||
{
|
||||
name: "malformed file",
|
||||
path: filepath.Join(tempDir, "malformed.pem"),
|
||||
contents: []byte("not a PEM certificate"),
|
||||
errorMatch: "does not contain any valid PEM certificates",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if tc.contents != nil {
|
||||
require.NoError(t, os.WriteFile(tc.path, tc.contents, 0600))
|
||||
}
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", tc.path)
|
||||
|
||||
tlsConfig, err := getTLSConfig()
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, tlsConfig)
|
||||
assert.Contains(t, err.Error(), tc.errorMatch)
|
||||
assert.Contains(t, err.Error(), tc.path)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func newSelfSignedServerCertificate(t *testing.T) (tls.Certificate, []byte) {
|
||||
t.Helper()
|
||||
privateKey, err := rsa.GenerateKey(rand.Reader, 2048)
|
||||
require.NoError(t, err)
|
||||
|
||||
template := &x509.Certificate{
|
||||
SerialNumber: big.NewInt(1),
|
||||
Subject: pkix.Name{CommonName: "127.0.0.1"},
|
||||
NotBefore: time.Now().Add(-time.Hour),
|
||||
NotAfter: time.Now().Add(time.Hour),
|
||||
IPAddresses: []net.IP{net.ParseIP("127.0.0.1")},
|
||||
KeyUsage: x509.KeyUsageDigitalSignature | x509.KeyUsageKeyEncipherment | x509.KeyUsageCertSign,
|
||||
ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageServerAuth},
|
||||
BasicConstraintsValid: true,
|
||||
IsCA: true,
|
||||
}
|
||||
certDER, err := x509.CreateCertificate(rand.Reader, template, template, &privateKey.PublicKey, privateKey)
|
||||
require.NoError(t, err)
|
||||
|
||||
certPEM := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: certDER})
|
||||
keyPEM := pem.EncodeToMemory(&pem.Block{Type: "RSA PRIVATE KEY", Bytes: x509.MarshalPKCS1PrivateKey(privateKey)})
|
||||
certificate, err := tls.X509KeyPair(certPEM, keyPEM)
|
||||
require.NoError(t, err)
|
||||
return certificate, certPEM
|
||||
}
|
||||
|
||||
// TestWebSocketClient_VerifySignature tests signature verification
|
||||
func TestWebSocketClient_VerifySignature(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -186,12 +363,8 @@ func TestWebSocketClient_VerifySignature(t *testing.T) {
|
||||
require.NoError(t, err)
|
||||
|
||||
// Set up environment
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -259,12 +432,8 @@ func TestWebSocketClient_HandleHubRequest(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
|
||||
// Set up environment
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -351,13 +520,8 @@ func TestGetUserAgent(t *testing.T) {
|
||||
func TestWebSocketClient_Close(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
|
||||
// Set up environment
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -372,13 +536,8 @@ func TestWebSocketClient_Close(t *testing.T) {
|
||||
func TestWebSocketClient_ConnectRateLimit(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
|
||||
// Set up environment
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -394,20 +553,10 @@ func TestWebSocketClient_ConnectRateLimit(t *testing.T) {
|
||||
|
||||
// TestGetToken tests the getToken function with various scenarios
|
||||
func TestGetToken(t *testing.T) {
|
||||
unsetEnvVars := func() {
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
os.Unsetenv("TOKEN")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN_FILE")
|
||||
os.Unsetenv("TOKEN_FILE")
|
||||
}
|
||||
|
||||
t.Run("token from TOKEN environment variable", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Set TOKEN env var
|
||||
expectedToken := "test-token-from-env"
|
||||
os.Setenv("TOKEN", expectedToken)
|
||||
defer os.Unsetenv("TOKEN")
|
||||
t.Setenv("TOKEN", expectedToken)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
@@ -415,12 +564,9 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("token from BESZEL_AGENT_TOKEN environment variable", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Set BESZEL_AGENT_TOKEN env var (should take precedence)
|
||||
expectedToken := "test-token-from-beszel-env"
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", expectedToken)
|
||||
defer os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", expectedToken)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
@@ -428,8 +574,6 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("token from TOKEN_FILE", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Create a temporary token file
|
||||
expectedToken := "test-token-from-file"
|
||||
tokenFile, err := os.CreateTemp("", "token-test-*.txt")
|
||||
@@ -441,17 +585,49 @@ func TestGetToken(t *testing.T) {
|
||||
tokenFile.Close()
|
||||
|
||||
// Set TOKEN_FILE env var
|
||||
os.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
defer os.Unsetenv("TOKEN_FILE")
|
||||
t.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, expectedToken, token)
|
||||
})
|
||||
|
||||
t.Run("token from BESZEL_AGENT_TOKEN_FILE", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
t.Run("TOKEN_FILE with surrounding blank lines and comments", func(t *testing.T) {
|
||||
expectedToken := "test-token-with-noise"
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("# hub token\n\n"+expectedToken+"\n\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, expectedToken, token)
|
||||
})
|
||||
|
||||
t.Run("TOKEN_FILE with multiple tokens is rejected", func(t *testing.T) {
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("11111111-1111-1111-1111-111111111111\n22222222-2222-2222-2222-222222222222\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
require.Error(t, err)
|
||||
assert.Empty(t, token)
|
||||
assert.Contains(t, err.Error(), "must contain a single token")
|
||||
})
|
||||
|
||||
t.Run("TOKEN_FILE holding only comments behaves like an empty file", func(t *testing.T) {
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("\n# only a comment\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, "", token)
|
||||
})
|
||||
|
||||
t.Run("token from BESZEL_AGENT_TOKEN_FILE", func(t *testing.T) {
|
||||
// Create a temporary token file
|
||||
expectedToken := "test-token-from-beszel-file"
|
||||
tokenFile, err := os.CreateTemp("", "token-test-*.txt")
|
||||
@@ -463,8 +639,7 @@ func TestGetToken(t *testing.T) {
|
||||
tokenFile.Close()
|
||||
|
||||
// Set BESZEL_AGENT_TOKEN_FILE env var (should take precedence)
|
||||
os.Setenv("BESZEL_AGENT_TOKEN_FILE", tokenFile.Name())
|
||||
defer os.Unsetenv("BESZEL_AGENT_TOKEN_FILE")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN_FILE", tokenFile.Name())
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
@@ -472,8 +647,6 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("TOKEN takes precedence over TOKEN_FILE", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Create a temporary token file
|
||||
fileToken := "token-from-file"
|
||||
tokenFile, err := os.CreateTemp("", "token-test-*.txt")
|
||||
@@ -486,12 +659,8 @@ func TestGetToken(t *testing.T) {
|
||||
|
||||
// Set both TOKEN and TOKEN_FILE
|
||||
envToken := "token-from-env"
|
||||
os.Setenv("TOKEN", envToken)
|
||||
os.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
defer func() {
|
||||
os.Unsetenv("TOKEN")
|
||||
os.Unsetenv("TOKEN_FILE")
|
||||
}()
|
||||
t.Setenv("TOKEN", envToken)
|
||||
t.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
@@ -499,7 +668,10 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("error when neither TOKEN nor TOKEN_FILE is set", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "")
|
||||
t.Setenv("TOKEN", "")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN_FILE", "")
|
||||
t.Setenv("TOKEN_FILE", "")
|
||||
|
||||
token, err := getToken()
|
||||
assert.Error(t, err)
|
||||
@@ -508,11 +680,8 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("error when TOKEN_FILE points to non-existent file", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Set TOKEN_FILE to a non-existent file
|
||||
os.Setenv("TOKEN_FILE", "/non/existent/file.txt")
|
||||
defer os.Unsetenv("TOKEN_FILE")
|
||||
t.Setenv("TOKEN_FILE", "/non/existent/file.txt")
|
||||
|
||||
token, err := getToken()
|
||||
assert.Error(t, err)
|
||||
@@ -521,8 +690,6 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("handles empty token file", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
// Create an empty token file
|
||||
tokenFile, err := os.CreateTemp("", "token-test-*.txt")
|
||||
require.NoError(t, err)
|
||||
@@ -530,8 +697,7 @@ func TestGetToken(t *testing.T) {
|
||||
tokenFile.Close()
|
||||
|
||||
// Set TOKEN_FILE env var
|
||||
os.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
defer os.Unsetenv("TOKEN_FILE")
|
||||
t.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
@@ -539,8 +705,6 @@ func TestGetToken(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("strips whitespace from TOKEN_FILE", func(t *testing.T) {
|
||||
unsetEnvVars()
|
||||
|
||||
tokenWithWhitespace := " test-token-with-whitespace \n\t"
|
||||
expectedToken := "test-token-with-whitespace"
|
||||
tokenFile, err := os.CreateTemp("", "token-test-*.txt")
|
||||
@@ -551,11 +715,18 @@ func TestGetToken(t *testing.T) {
|
||||
require.NoError(t, err)
|
||||
tokenFile.Close()
|
||||
|
||||
os.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
defer os.Unsetenv("TOKEN_FILE")
|
||||
t.Setenv("TOKEN_FILE", tokenFile.Name())
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, expectedToken, token, "Whitespace should be stripped from token file content")
|
||||
})
|
||||
}
|
||||
|
||||
func TestWebSocketDeadlineCoversSlowCollection(t *testing.T) {
|
||||
const minimumDeadline = 120 * time.Second
|
||||
|
||||
if wsDeadline < minimumDeadline {
|
||||
t.Fatalf("WebSocket deadline %s is shorter than the slow-collection window of %s", wsDeadline, minimumDeadline)
|
||||
}
|
||||
}
|
||||
|
||||
+71
-12
@@ -1,14 +1,18 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"net"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/health"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -83,7 +87,19 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
|
||||
wsClient, err := newWebSocketClient(c.agent)
|
||||
if err != nil {
|
||||
slog.Warn("Error creating WebSocket client", "err", err)
|
||||
var caCertErr *caCertFileError
|
||||
if errors.As(err, &caCertErr) {
|
||||
return err
|
||||
}
|
||||
disableSSH, _ := utils.GetEnv("DISABLE_SSH")
|
||||
if errors.Is(err, errNoHubURL) && disableSSH != "true" {
|
||||
// SSH-only mode: the hub dials the agent, so there is nothing to warn
|
||||
// about. With SSH also disabled there is no connection method at all,
|
||||
// so that case still warns.
|
||||
slog.Debug("WebSocket client not configured", "err", err)
|
||||
} else {
|
||||
slog.Warn("Error creating WebSocket client", "err", err)
|
||||
}
|
||||
}
|
||||
c.wsClient = wsClient
|
||||
|
||||
@@ -91,8 +107,8 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
c.eventChan = make(chan ConnectionEvent, 1)
|
||||
|
||||
// signal handling for shutdown
|
||||
sigChan := make(chan os.Signal, 1)
|
||||
signal.Notify(sigChan, syscall.SIGINT, syscall.SIGTERM)
|
||||
sigCtx, stopSignals := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
|
||||
defer stopSignals()
|
||||
|
||||
c.startWsTicker()
|
||||
c.connect()
|
||||
@@ -109,22 +125,47 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
_ = c.startWebSocketConnection()
|
||||
case <-healthTicker:
|
||||
_ = health.Update()
|
||||
case <-sigChan:
|
||||
slog.Info("Shutting down")
|
||||
_ = c.agent.StopServer()
|
||||
c.closeWebSocket()
|
||||
return health.CleanUp()
|
||||
case <-sigCtx.Done():
|
||||
slog.Info("Shutting down", "cause", context.Cause(sigCtx))
|
||||
return c.stop()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stop does not stop the connection manager itself, just any active connections. The manager will attempt to reconnect after stopping, so this should only be called immediately before shutting down the entire agent.
|
||||
//
|
||||
// If we need or want to expose a graceful Stop method in the future, do something like this to actually stop the manager:
|
||||
//
|
||||
// func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
// ctx, cancel := context.WithCancel(context.Background())
|
||||
// c.cancel = cancel
|
||||
//
|
||||
// for {
|
||||
// select {
|
||||
// case <-ctx.Done():
|
||||
// return c.stop()
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
//
|
||||
// func (c *ConnectionManager) Stop() {
|
||||
// c.cancel()
|
||||
// }
|
||||
func (c *ConnectionManager) stop() error {
|
||||
_ = c.agent.StopServer()
|
||||
c.closeWebSocket()
|
||||
return health.CleanUp()
|
||||
}
|
||||
|
||||
// handleEvent processes connection events and updates the connection state accordingly.
|
||||
func (c *ConnectionManager) handleEvent(event ConnectionEvent) {
|
||||
switch event {
|
||||
case WebSocketConnect:
|
||||
c.handleStateChange(WebSocketConnected)
|
||||
case SSHConnect:
|
||||
c.handleStateChange(SSHConnected)
|
||||
if c.State == Disconnected {
|
||||
c.handleStateChange(SSHConnected)
|
||||
}
|
||||
case WebSocketDisconnect:
|
||||
if c.State == WebSocketConnected {
|
||||
c.handleStateChange(Disconnected)
|
||||
@@ -185,9 +226,16 @@ func (c *ConnectionManager) connect() {
|
||||
|
||||
// Try WebSocket first, if it fails, start SSH server
|
||||
err := c.startWebSocketConnection()
|
||||
if err != nil && c.State == Disconnected {
|
||||
c.startSSHServer()
|
||||
c.startWsTicker()
|
||||
if err != nil {
|
||||
if shouldExitOnErr(err) {
|
||||
time.Sleep(2 * time.Second) // prevent tight restart loop
|
||||
_ = c.stop()
|
||||
os.Exit(1)
|
||||
}
|
||||
if c.State == Disconnected {
|
||||
c.startSSHServer()
|
||||
c.startWsTicker()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,3 +272,14 @@ func (c *ConnectionManager) closeWebSocket() {
|
||||
c.wsClient.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// shouldExitOnErr checks if the error is a DNS resolution failure and if the
|
||||
// EXIT_ON_DNS_ERROR env var is set. https://github.com/henrygd/beszel/issues/1924.
|
||||
func shouldExitOnErr(err error) bool {
|
||||
if val, _ := utils.GetEnv("EXIT_ON_DNS_ERROR"); val == "true" {
|
||||
if opErr, ok := errors.AsType[*net.OpError](err); ok {
|
||||
return strings.Contains(opErr.Err.Error(), "lookup")
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -1,14 +1,13 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"crypto/ed25519"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/url"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -115,6 +114,12 @@ func TestConnectionManager_EventHandling(t *testing.T) {
|
||||
event: SSHConnect,
|
||||
expectedState: SSHConnected,
|
||||
},
|
||||
{
|
||||
name: "SSH connect from WebSocket connected (no change)",
|
||||
initialState: WebSocketConnected,
|
||||
event: SSHConnect,
|
||||
expectedState: WebSocketConnected,
|
||||
},
|
||||
{
|
||||
name: "WebSocket disconnect from connected",
|
||||
initialState: WebSocketConnected,
|
||||
@@ -184,10 +189,6 @@ func TestConnectionManager_TickerManagement(t *testing.T) {
|
||||
|
||||
// TestConnectionManager_WebSocketConnectionFlow tests WebSocket connection logic
|
||||
func TestConnectionManager_WebSocketConnectionFlow(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping WebSocket connection test in short mode")
|
||||
}
|
||||
|
||||
agent := createTestAgent(t)
|
||||
cm := agent.connectionManager
|
||||
|
||||
@@ -197,19 +198,18 @@ func TestConnectionManager_WebSocketConnectionFlow(t *testing.T) {
|
||||
assert.Equal(t, Disconnected, cm.State, "State should remain Disconnected after failed connection")
|
||||
|
||||
// Test with invalid URL
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "invalid-url")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
|
||||
// Test with missing token
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "1,33%")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
_, err2 := newWebSocketClient(agent)
|
||||
assert.Error(t, err2, "WebSocket client creation should fail without token")
|
||||
assert.Error(t, err2, "WebSocket client creation should fail with invalid URL")
|
||||
|
||||
// Test with missing token
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "http://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "")
|
||||
|
||||
_, err3 := newWebSocketClient(agent)
|
||||
assert.Error(t, err3, "WebSocket client creation should fail without token")
|
||||
}
|
||||
|
||||
// TestConnectionManager_ReconnectionLogic tests reconnection prevention logic
|
||||
@@ -235,12 +235,8 @@ func TestConnectionManager_ConnectWithRateLimit(t *testing.T) {
|
||||
cm := agent.connectionManager
|
||||
|
||||
// Set up environment for WebSocket client creation
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "ws://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "ws://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
// Create WebSocket client
|
||||
wsClient, err := newWebSocketClient(agent)
|
||||
@@ -275,6 +271,19 @@ func TestConnectionManager_StartWithInvalidConfig(t *testing.T) {
|
||||
assert.Error(t, err, "Should error when starting already started connection manager")
|
||||
}
|
||||
|
||||
func TestConnectionManager_StartRejectsInvalidCACertFile(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
cm := agent.connectionManager
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "https://hub.example.com")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", t.TempDir())
|
||||
|
||||
err := cm.Start(ServerOptions{})
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "read CA_CERT_FILE")
|
||||
assert.Nil(t, cm.eventChan)
|
||||
}
|
||||
|
||||
// TestConnectionManager_CloseWebSocket tests WebSocket closing
|
||||
func TestConnectionManager_CloseWebSocket(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -286,12 +295,8 @@ func TestConnectionManager_CloseWebSocket(t *testing.T) {
|
||||
}, "Should not panic when closing nil WebSocket client")
|
||||
|
||||
// Set up environment and create WebSocket client
|
||||
os.Setenv("BESZEL_AGENT_HUB_URL", "ws://localhost:8080")
|
||||
os.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
defer func() {
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
os.Unsetenv("BESZEL_AGENT_TOKEN")
|
||||
}()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "ws://localhost:8080")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
wsClient, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
@@ -313,3 +318,65 @@ func TestConnectionManager_ConnectFlow(t *testing.T) {
|
||||
cm.connect()
|
||||
}, "Connect should not panic without WebSocket client")
|
||||
}
|
||||
|
||||
func TestShouldExitOnErr(t *testing.T) {
|
||||
createDialErr := func(msg string) error {
|
||||
return &net.OpError{
|
||||
Op: "dial",
|
||||
Net: "tcp",
|
||||
Err: errors.New(msg),
|
||||
}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
envValue string
|
||||
expected bool
|
||||
}{
|
||||
{
|
||||
name: "no env var",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var false",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "false",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var true, matching error",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "true",
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
name: "env var true, matching error with extra context",
|
||||
err: createDialErr("lookup beszel.server.lan on [::1]:53: read udp [::1]:44557->[::1]:53: read: connection refused"),
|
||||
envValue: "true",
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
name: "env var true, non-matching error",
|
||||
err: errors.New("connection refused"),
|
||||
envValue: "true",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var true, dial but not lookup",
|
||||
err: createDialErr("connection timeout"),
|
||||
envValue: "true",
|
||||
expected: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
t.Setenv("EXIT_ON_DNS_ERROR", tt.envValue)
|
||||
result := shouldExitOnErr(tt.err)
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
+3
-6
@@ -14,10 +14,10 @@ var lastPerCoreCpuTimes = make(map[uint16][]cpu.TimesStat)
|
||||
// init initializes the CPU monitoring by storing the initial CPU times
|
||||
// for the default 60-second cache interval.
|
||||
func init() {
|
||||
if times, err := cpu.Times(false); err == nil {
|
||||
if times, err := cpu.Times(false); err == nil && len(times) > 0 {
|
||||
lastCpuTimes[60000] = times[0]
|
||||
}
|
||||
if perCoreTimes, err := cpu.Times(true); err == nil {
|
||||
if perCoreTimes, err := cpu.Times(true); err == nil && len(perCoreTimes) > 0 {
|
||||
lastPerCoreCpuTimes[60000] = perCoreTimes
|
||||
}
|
||||
}
|
||||
@@ -89,10 +89,7 @@ func getPerCoreCpuUsage(cacheTimeMs uint16) (system.Uint8Slice, error) {
|
||||
lastTimes := lastPerCoreCpuTimes[cacheTimeMs]
|
||||
|
||||
// Limit to the number of cores available in both samples
|
||||
length := len(perCoreTimes)
|
||||
if len(lastTimes) < length {
|
||||
length = len(lastTimes)
|
||||
}
|
||||
length := min(len(lastTimes), len(perCoreTimes))
|
||||
|
||||
usage := make([]uint8, length)
|
||||
for i := 0; i < length; i++ {
|
||||
|
||||
+3
-1
@@ -6,6 +6,8 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
)
|
||||
|
||||
// GetDataDir returns the path to the data directory for the agent and an error
|
||||
@@ -16,7 +18,7 @@ func GetDataDir(dataDirs ...string) (string, error) {
|
||||
return testDataDirs(dataDirs)
|
||||
}
|
||||
|
||||
dataDir, _ := GetEnv("DATA_DIR")
|
||||
dataDir, _ := utils.GetEnv("DATA_DIR")
|
||||
if dataDir != "" {
|
||||
dataDirs = append(dataDirs, dataDir)
|
||||
}
|
||||
|
||||
+13
-27
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
@@ -13,6 +12,14 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func invalidDataDir(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
filePath := filepath.Join(t.TempDir(), "file")
|
||||
require.NoError(t, os.WriteFile(filePath, nil, 0644))
|
||||
return filepath.Join(filePath, "data")
|
||||
}
|
||||
|
||||
func TestGetDataDir(t *testing.T) {
|
||||
// Test with explicit dataDir parameter
|
||||
t.Run("explicit data dir", func(t *testing.T) {
|
||||
@@ -40,17 +47,7 @@ func TestGetDataDir(t *testing.T) {
|
||||
t.Run("DATA_DIR environment variable", func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
|
||||
// Set environment variable
|
||||
oldValue := os.Getenv("DATA_DIR")
|
||||
defer func() {
|
||||
if oldValue == "" {
|
||||
os.Unsetenv("BESZEL_AGENT_DATA_DIR")
|
||||
} else {
|
||||
os.Setenv("BESZEL_AGENT_DATA_DIR", oldValue)
|
||||
}
|
||||
}()
|
||||
|
||||
os.Setenv("BESZEL_AGENT_DATA_DIR", tempDir)
|
||||
t.Setenv("BESZEL_AGENT_DATA_DIR", tempDir)
|
||||
|
||||
result, err := GetDataDir()
|
||||
require.NoError(t, err)
|
||||
@@ -59,24 +56,13 @@ func TestGetDataDir(t *testing.T) {
|
||||
|
||||
// Test with invalid explicit dataDir
|
||||
t.Run("invalid explicit data dir", func(t *testing.T) {
|
||||
invalidPath := "/invalid/path/that/cannot/be/created"
|
||||
invalidPath := invalidDataDir(t)
|
||||
_, err := GetDataDir(invalidPath)
|
||||
assert.Error(t, err)
|
||||
})
|
||||
|
||||
// Test fallback behavior (empty dataDir, no env var)
|
||||
t.Run("fallback to default directories", func(t *testing.T) {
|
||||
// Clear DATA_DIR environment variable
|
||||
oldValue := os.Getenv("DATA_DIR")
|
||||
defer func() {
|
||||
if oldValue == "" {
|
||||
os.Unsetenv("DATA_DIR")
|
||||
} else {
|
||||
os.Setenv("DATA_DIR", oldValue)
|
||||
}
|
||||
}()
|
||||
os.Unsetenv("DATA_DIR")
|
||||
|
||||
// This will try platform-specific defaults, which may or may not work
|
||||
// We're mainly testing that it doesn't panic and returns some result
|
||||
result, err := GetDataDir()
|
||||
@@ -100,7 +86,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
// Test with multiple directories, first one valid
|
||||
t.Run("multiple dirs - first valid", func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
invalidDir := "/invalid/path"
|
||||
invalidDir := invalidDataDir(t)
|
||||
result, err := testDataDirs([]string{tempDir, invalidDir})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, tempDir, result)
|
||||
@@ -109,7 +95,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
// Test with multiple directories, second one valid
|
||||
t.Run("multiple dirs - second valid", func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
invalidDir := "/invalid/path"
|
||||
invalidDir := invalidDataDir(t)
|
||||
result, err := testDataDirs([]string{invalidDir, tempDir})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, tempDir, result)
|
||||
@@ -131,7 +117,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
|
||||
// Test with no valid directories
|
||||
t.Run("no valid directories", func(t *testing.T) {
|
||||
invalidPaths := []string{"/invalid/path1", "/invalid/path2"}
|
||||
invalidPaths := []string{invalidDataDir(t), invalidDataDir(t)}
|
||||
_, err := testDataDirs(invalidPaths)
|
||||
assert.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "data directory not found")
|
||||
|
||||
+554
-144
@@ -1,6 +1,7 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -8,11 +9,60 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/disk"
|
||||
)
|
||||
|
||||
// fsRegistrationContext holds the shared lookup state needed to resolve a
|
||||
// filesystem into the tracked fsStats key and metadata.
|
||||
type fsRegistrationContext struct {
|
||||
filesystem string // device part of optional FILESYSTEM env var
|
||||
filesystemName string // optional custom name from FILESYSTEM=device__name
|
||||
isWindows bool
|
||||
efPath string // path to extra filesystems (default "/extra-filesystems")
|
||||
diskIoCounters map[string]disk.IOCountersStat
|
||||
}
|
||||
|
||||
// diskDiscovery groups the transient state for a single initializeDiskInfo run so
|
||||
// helper methods can share the same partitions, mount paths, and lookup functions
|
||||
type diskDiscovery struct {
|
||||
agent *Agent
|
||||
rootMountPoint string
|
||||
partitions []disk.PartitionStat
|
||||
usageFn func(string) (*disk.UsageStat, error)
|
||||
ctx fsRegistrationContext
|
||||
}
|
||||
|
||||
// prevDisk stores previous per-device disk counters for a given cache interval
|
||||
type prevDisk struct {
|
||||
readBytes uint64
|
||||
writeBytes uint64
|
||||
readTime uint64 // cumulative ms spent on reads (from ReadTime)
|
||||
writeTime uint64 // cumulative ms spent on writes (from WriteTime)
|
||||
ioTime uint64 // cumulative ms spent doing I/O (from IoTime)
|
||||
weightedIO uint64 // cumulative weighted ms (queue-depth × ms, from WeightedIO)
|
||||
readCount uint64 // cumulative read operation count
|
||||
writeCount uint64 // cumulative write operation count
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// prevDiskFromCounter creates a prevDisk snapshot from a disk.IOCountersStat at time t.
|
||||
func prevDiskFromCounter(d disk.IOCountersStat, t time.Time) prevDisk {
|
||||
return prevDisk{
|
||||
readBytes: d.ReadBytes,
|
||||
writeBytes: d.WriteBytes,
|
||||
readTime: d.ReadTime,
|
||||
writeTime: d.WriteTime,
|
||||
ioTime: d.IoTime,
|
||||
weightedIO: d.WeightedIO,
|
||||
readCount: d.ReadCount,
|
||||
writeCount: d.WriteCount,
|
||||
at: t,
|
||||
}
|
||||
}
|
||||
|
||||
// parseFilesystemEntry parses a filesystem entry in the format "device__customname"
|
||||
// Returns the device/filesystem part and the custom name part
|
||||
func parseFilesystemEntry(entry string) (device, customName string) {
|
||||
@@ -26,23 +76,237 @@ func parseFilesystemEntry(entry string) (device, customName string) {
|
||||
return device, customName
|
||||
}
|
||||
|
||||
// extraFilesystemPartitionInfo derives the I/O device and optional display name
|
||||
// for a mounted /extra-filesystems partition. Prefer the partition device reported
|
||||
// by the system and only use the folder name for custom naming metadata.
|
||||
func extraFilesystemPartitionInfo(p disk.PartitionStat) (device, customName string) {
|
||||
device = strings.TrimSpace(p.Device)
|
||||
folderDevice, customName := parseFilesystemEntry(filepath.Base(p.Mountpoint))
|
||||
if device == "" {
|
||||
device = folderDevice
|
||||
}
|
||||
return device, customName
|
||||
}
|
||||
|
||||
func isDockerSpecialMountpoint(mountpoint string) bool {
|
||||
switch mountpoint {
|
||||
case "/etc/hosts", "/etc/resolv.conf", "/etc/hostname":
|
||||
return true
|
||||
default:
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// registerFilesystemStats resolves the tracked key and stats payload for a
|
||||
// filesystem before it is inserted into fsStats.
|
||||
func registerFilesystemStats(existing map[string]*system.FsStats, device, mountpoint string, root bool, customName string, ctx fsRegistrationContext) (string, *system.FsStats, bool) {
|
||||
key := device
|
||||
if !ctx.isWindows {
|
||||
key = filepath.Base(device)
|
||||
}
|
||||
|
||||
if root {
|
||||
// Try to map root device to a diskIoCounters entry. First checks for an
|
||||
// exact key match, then uses findIoDevice for normalized / prefix-based
|
||||
// matching (e.g. nda0p2 -> nda0), and finally falls back to FILESYSTEM.
|
||||
if _, ioMatch := ctx.diskIoCounters[key]; !ioMatch {
|
||||
if matchedKey, match := findIoDevice(key, ctx.diskIoCounters); match {
|
||||
key = matchedKey
|
||||
} else if ctx.filesystem != "" {
|
||||
if matchedKey, match := findIoDevice(ctx.filesystem, ctx.diskIoCounters); match {
|
||||
key = matchedKey
|
||||
}
|
||||
}
|
||||
if _, ioMatch = ctx.diskIoCounters[key]; !ioMatch {
|
||||
slog.Warn("Root I/O unmapped; set FILESYSTEM", "device", device, "mountpoint", mountpoint)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Check if non-root has diskstats and prefer the folder device for
|
||||
// /extra-filesystems mounts when the discovered partition device is a
|
||||
// mapper path (e.g. luks UUID) that obscures the underlying block device.
|
||||
if _, ioMatch := ctx.diskIoCounters[key]; !ioMatch {
|
||||
if strings.HasPrefix(mountpoint, ctx.efPath) {
|
||||
folderDevice, _ := parseFilesystemEntry(filepath.Base(mountpoint))
|
||||
if folderDevice != "" {
|
||||
if matchedKey, match := findIoDevice(folderDevice, ctx.diskIoCounters); match {
|
||||
key = matchedKey
|
||||
}
|
||||
}
|
||||
}
|
||||
if _, ioMatch = ctx.diskIoCounters[key]; !ioMatch {
|
||||
if matchedKey, match := findIoDevice(key, ctx.diskIoCounters); match {
|
||||
key = matchedKey
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if _, exists := existing[key]; exists {
|
||||
return "", nil, false
|
||||
}
|
||||
|
||||
fsStats := &system.FsStats{Root: root, Mountpoint: mountpoint}
|
||||
if customName != "" {
|
||||
fsStats.Name = customName
|
||||
}
|
||||
return key, fsStats, true
|
||||
}
|
||||
|
||||
// addFsStat inserts a discovered filesystem if it resolves to a new tracking
|
||||
// key. The key selection itself lives in buildFsStatRegistration so that logic
|
||||
// can stay directly unit-tested.
|
||||
func (d *diskDiscovery) addFsStat(device, mountpoint string, root bool, customName string) {
|
||||
key, fsStats, ok := registerFilesystemStats(d.agent.fsStats, device, mountpoint, root, customName, d.ctx)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
d.agent.fsStats[key] = fsStats
|
||||
name := key
|
||||
if customName != "" {
|
||||
name = customName
|
||||
}
|
||||
slog.Info("Detected disk", "name", name, "device", device, "mount", mountpoint, "io", key, "root", root)
|
||||
}
|
||||
|
||||
// addConfiguredRootFs resolves FILESYSTEM against partitions first, then falls
|
||||
// back to direct diskstats matching for setups like ZFS where partitions do not
|
||||
// expose the physical device name.
|
||||
func (d *diskDiscovery) addConfiguredRootFs() bool {
|
||||
if d.ctx.filesystem == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
for _, p := range d.partitions {
|
||||
if filesystemMatchesPartitionSetting(d.ctx.filesystem, p) {
|
||||
d.addFsStat(p.Device, p.Mountpoint, true, d.ctx.filesystemName)
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// FILESYSTEM may name a physical disk absent from partitions (e.g. ZFS lists
|
||||
// dataset paths like zroot/ROOT/default, not block devices).
|
||||
if ioKey, match := findIoDevice(d.ctx.filesystem, d.ctx.diskIoCounters); match {
|
||||
d.agent.fsStats[ioKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint, Name: d.ctx.filesystemName}
|
||||
return true
|
||||
}
|
||||
|
||||
slog.Warn("Partition details not found", "filesystem", d.ctx.filesystem)
|
||||
return false
|
||||
}
|
||||
|
||||
func isRootFallbackPartition(p disk.PartitionStat, rootMountPoint string) bool {
|
||||
return p.Mountpoint == rootMountPoint ||
|
||||
(isDockerSpecialMountpoint(p.Mountpoint) && strings.HasPrefix(p.Device, "/dev"))
|
||||
}
|
||||
|
||||
// addPartitionRootFs handles the non-configured root fallback path when a
|
||||
// partition looks like the active root mount but still needs translating to an
|
||||
// I/O device key.
|
||||
func (d *diskDiscovery) addPartitionRootFs(device, mountpoint string) bool {
|
||||
fs, match := findIoDevice(filepath.Base(device), d.ctx.diskIoCounters)
|
||||
if !match {
|
||||
return false
|
||||
}
|
||||
// The resolved I/O device is already known here, so use it directly to avoid
|
||||
// a second fallback search inside buildFsStatRegistration.
|
||||
d.addFsStat(fs, mountpoint, true, "")
|
||||
return true
|
||||
}
|
||||
|
||||
// addLastResortRootFs is only used when neither FILESYSTEM nor partition-based
|
||||
// heuristics can identify root, so it picks the busiest I/O device as a final
|
||||
// fallback and preserves the root mountpoint for usage collection.
|
||||
func (d *diskDiscovery) addLastResortRootFs() {
|
||||
rootKey := mostActiveIoDevice(d.ctx.diskIoCounters)
|
||||
if rootKey != "" {
|
||||
slog.Warn("Using most active device for root I/O; set FILESYSTEM to override", "device", rootKey)
|
||||
} else {
|
||||
rootKey = filepath.Base(d.rootMountPoint)
|
||||
if _, exists := d.agent.fsStats[rootKey]; exists {
|
||||
rootKey = "root"
|
||||
}
|
||||
slog.Warn("Root I/O device not detected; set FILESYSTEM to override")
|
||||
}
|
||||
d.agent.fsStats[rootKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint}
|
||||
}
|
||||
|
||||
// findPartitionByFilesystemSetting matches an EXTRA_FILESYSTEMS entry against a
|
||||
// discovered partition either by mountpoint or by device suffix.
|
||||
func findPartitionByFilesystemSetting(filesystem string, partitions []disk.PartitionStat) (disk.PartitionStat, bool) {
|
||||
for _, p := range partitions {
|
||||
if strings.HasSuffix(p.Device, filesystem) || p.Mountpoint == filesystem {
|
||||
return p, true
|
||||
}
|
||||
}
|
||||
return disk.PartitionStat{}, false
|
||||
}
|
||||
|
||||
// addConfiguredExtraFsEntry resolves one EXTRA_FILESYSTEMS entry, preferring a
|
||||
// discovered partition and falling back to any path that disk.Usage accepts.
|
||||
func (d *diskDiscovery) addConfiguredExtraFsEntry(filesystem, customName string) {
|
||||
if p, found := findPartitionByFilesystemSetting(filesystem, d.partitions); found {
|
||||
d.addFsStat(p.Device, p.Mountpoint, false, customName)
|
||||
return
|
||||
}
|
||||
|
||||
if _, err := d.usageFn(filesystem); err == nil {
|
||||
d.addFsStat(filepath.Base(filesystem), filesystem, false, customName)
|
||||
return
|
||||
} else {
|
||||
slog.Error("Invalid filesystem", "name", filesystem, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// addConfiguredExtraFilesystems parses and registers the comma-separated
|
||||
// EXTRA_FILESYSTEMS env var entries.
|
||||
func (d *diskDiscovery) addConfiguredExtraFilesystems(extraFilesystems string) {
|
||||
for fsEntry := range strings.SplitSeq(extraFilesystems, ",") {
|
||||
filesystem, customName := parseFilesystemEntry(fsEntry)
|
||||
d.addConfiguredExtraFsEntry(filesystem, customName)
|
||||
}
|
||||
}
|
||||
|
||||
// addPartitionExtraFs registers partitions mounted under /extra-filesystems so
|
||||
// their display names can come from the folder name while their I/O keys still
|
||||
// prefer the underlying partition device. Only direct children are matched to
|
||||
// avoid registering nested virtual mounts (e.g. /proc, /sys) that are returned by
|
||||
// disk.Partitions(true) when the host root is bind-mounted in /extra-filesystems.
|
||||
func (d *diskDiscovery) addPartitionExtraFs(p disk.PartitionStat) {
|
||||
if filepath.Dir(p.Mountpoint) != d.ctx.efPath {
|
||||
return
|
||||
}
|
||||
device, customName := extraFilesystemPartitionInfo(p)
|
||||
d.addFsStat(device, p.Mountpoint, false, customName)
|
||||
}
|
||||
|
||||
// addExtraFilesystemFolders handles bare directories under /extra-filesystems
|
||||
// that may not appear in partition discovery, while skipping mountpoints that
|
||||
// were already registered from higher-fidelity sources.
|
||||
func (d *diskDiscovery) addExtraFilesystemFolders(folderNames []string) {
|
||||
existingMountpoints := make(map[string]bool, len(d.agent.fsStats))
|
||||
for _, stats := range d.agent.fsStats {
|
||||
existingMountpoints[stats.Mountpoint] = true
|
||||
}
|
||||
|
||||
for _, folderName := range folderNames {
|
||||
mountpoint := filepath.Join(d.ctx.efPath, folderName)
|
||||
slog.Debug("/extra-filesystems", "mountpoint", mountpoint)
|
||||
if existingMountpoints[mountpoint] {
|
||||
continue
|
||||
}
|
||||
device, customName := parseFilesystemEntry(folderName)
|
||||
d.addFsStat(device, mountpoint, false, customName)
|
||||
}
|
||||
}
|
||||
|
||||
// Sets up the filesystems to monitor for disk usage and I/O.
|
||||
func (a *Agent) initializeDiskInfo() {
|
||||
filesystem, _ := GetEnv("FILESYSTEM")
|
||||
efPath := "/extra-filesystems"
|
||||
filesystemRaw, _ := utils.GetEnv("FILESYSTEM")
|
||||
filesystem, filesystemName := parseFilesystemEntry(filesystemRaw)
|
||||
hasRoot := false
|
||||
isWindows := runtime.GOOS == "windows"
|
||||
|
||||
partitions, err := disk.Partitions(false)
|
||||
partitions, err := disk.PartitionsWithContext(context.Background(), true)
|
||||
if err != nil {
|
||||
slog.Error("Error getting disk partitions", "err", err)
|
||||
}
|
||||
@@ -55,165 +319,234 @@ func (a *Agent) initializeDiskInfo() {
|
||||
}
|
||||
}
|
||||
|
||||
// ioContext := context.WithValue(a.sensorsContext,
|
||||
// common.EnvKey, common.EnvMap{common.HostProcEnvKey: "/tmp/testproc"},
|
||||
// )
|
||||
// diskIoCounters, err := disk.IOCountersWithContext(ioContext)
|
||||
|
||||
diskIoCounters, err := disk.IOCounters()
|
||||
if err != nil {
|
||||
slog.Error("Error getting diskstats", "err", err)
|
||||
}
|
||||
slog.Debug("Disk I/O", "diskstats", diskIoCounters)
|
||||
|
||||
// Helper function to add a filesystem to fsStats if it doesn't exist
|
||||
addFsStat := func(device, mountpoint string, root bool, customName ...string) {
|
||||
var key string
|
||||
if isWindows {
|
||||
key = device
|
||||
} else {
|
||||
key = filepath.Base(device)
|
||||
}
|
||||
var ioMatch bool
|
||||
if _, exists := a.fsStats[key]; !exists {
|
||||
if root {
|
||||
slog.Info("Detected root device", "name", key)
|
||||
// Check if root device is in /proc/diskstats. Do not guess a
|
||||
// fallback device for root: that can misattribute root I/O to a
|
||||
// different disk while usage remains tied to root mountpoint.
|
||||
if _, ioMatch = diskIoCounters[key]; !ioMatch {
|
||||
if matchedKey, match := findIoDevice(filesystem, diskIoCounters); match {
|
||||
key = matchedKey
|
||||
ioMatch = true
|
||||
} else {
|
||||
slog.Warn("Root I/O unmapped; set FILESYSTEM", "device", device, "mountpoint", mountpoint)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Check if non-root has diskstats and fall back to folder name if not
|
||||
// Scenario: device is encrypted and named luks-2bcb02be-999d-4417-8d18-5c61e660fb6e - not in /proc/diskstats.
|
||||
// However, the device can be specified by mounting folder from luks device at /extra-filesystems/sda1
|
||||
if _, ioMatch = diskIoCounters[key]; !ioMatch {
|
||||
efBase := filepath.Base(mountpoint)
|
||||
if _, ioMatch = diskIoCounters[efBase]; ioMatch {
|
||||
key = efBase
|
||||
}
|
||||
}
|
||||
}
|
||||
fsStats := &system.FsStats{Root: root, Mountpoint: mountpoint}
|
||||
if len(customName) > 0 && customName[0] != "" {
|
||||
fsStats.Name = customName[0]
|
||||
}
|
||||
a.fsStats[key] = fsStats
|
||||
}
|
||||
ctx := fsRegistrationContext{
|
||||
filesystem: filesystem,
|
||||
filesystemName: filesystemName,
|
||||
isWindows: isWindows,
|
||||
diskIoCounters: diskIoCounters,
|
||||
efPath: "/extra-filesystems",
|
||||
}
|
||||
|
||||
// Get the appropriate root mount point for this system
|
||||
rootMountPoint := a.getRootMountPoint()
|
||||
|
||||
// Use FILESYSTEM env var to find root filesystem
|
||||
if filesystem != "" {
|
||||
for _, p := range partitions {
|
||||
if strings.HasSuffix(p.Device, filesystem) || p.Mountpoint == filesystem {
|
||||
addFsStat(p.Device, p.Mountpoint, true)
|
||||
hasRoot = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasRoot {
|
||||
slog.Warn("Partition details not found", "filesystem", filesystem)
|
||||
}
|
||||
discovery := diskDiscovery{
|
||||
agent: a,
|
||||
rootMountPoint: a.getRootMountPoint(),
|
||||
partitions: partitions,
|
||||
usageFn: disk.Usage,
|
||||
ctx: ctx,
|
||||
}
|
||||
|
||||
// Add EXTRA_FILESYSTEMS env var values to fsStats
|
||||
if extraFilesystems, exists := GetEnv("EXTRA_FILESYSTEMS"); exists {
|
||||
for _, fsEntry := range strings.Split(extraFilesystems, ",") {
|
||||
// Parse custom name from format: device__customname
|
||||
fs, customName := parseFilesystemEntry(fsEntry)
|
||||
hasRoot = discovery.addConfiguredRootFs()
|
||||
|
||||
found := false
|
||||
for _, p := range partitions {
|
||||
if strings.HasSuffix(p.Device, fs) || p.Mountpoint == fs {
|
||||
addFsStat(p.Device, p.Mountpoint, false, customName)
|
||||
found = true
|
||||
break
|
||||
}
|
||||
}
|
||||
// if not in partitions, test if we can get disk usage
|
||||
if !found {
|
||||
if _, err := disk.Usage(fs); err == nil {
|
||||
addFsStat(filepath.Base(fs), fs, false, customName)
|
||||
} else {
|
||||
slog.Error("Invalid filesystem", "name", fs, "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
// Add EXTRA_FILESYSTEMS env var values to fsStats
|
||||
if extraFilesystems, exists := utils.GetEnv("EXTRA_FILESYSTEMS"); exists {
|
||||
discovery.addConfiguredExtraFilesystems(extraFilesystems)
|
||||
}
|
||||
|
||||
// Process partitions for various mount points
|
||||
for _, p := range partitions {
|
||||
// fmt.Println(p.Device, p.Mountpoint)
|
||||
// Binary root fallback or docker root fallback
|
||||
if !hasRoot && (p.Mountpoint == rootMountPoint || (isDockerSpecialMountpoint(p.Mountpoint) && strings.HasPrefix(p.Device, "/dev"))) {
|
||||
fs, match := findIoDevice(filepath.Base(p.Device), diskIoCounters)
|
||||
if match {
|
||||
addFsStat(fs, p.Mountpoint, true)
|
||||
hasRoot = true
|
||||
}
|
||||
}
|
||||
|
||||
// Check if device is in /extra-filesystems
|
||||
if strings.HasPrefix(p.Mountpoint, efPath) {
|
||||
device, customName := parseFilesystemEntry(p.Mountpoint)
|
||||
addFsStat(device, p.Mountpoint, false, customName)
|
||||
if !hasRoot && isRootFallbackPartition(p, discovery.rootMountPoint) {
|
||||
hasRoot = discovery.addPartitionRootFs(p.Device, p.Mountpoint)
|
||||
}
|
||||
discovery.addPartitionExtraFs(p)
|
||||
}
|
||||
|
||||
// Check all folders in /extra-filesystems and add them if not already present
|
||||
if folders, err := os.ReadDir(efPath); err == nil {
|
||||
existingMountpoints := make(map[string]bool)
|
||||
for _, stats := range a.fsStats {
|
||||
existingMountpoints[stats.Mountpoint] = true
|
||||
}
|
||||
if folders, err := os.ReadDir(discovery.ctx.efPath); err == nil {
|
||||
folderNames := make([]string, 0, len(folders))
|
||||
for _, folder := range folders {
|
||||
if folder.IsDir() {
|
||||
mountpoint := filepath.Join(efPath, folder.Name())
|
||||
slog.Debug("/extra-filesystems", "mountpoint", mountpoint)
|
||||
if !existingMountpoints[mountpoint] {
|
||||
device, customName := parseFilesystemEntry(folder.Name())
|
||||
addFsStat(device, mountpoint, false, customName)
|
||||
}
|
||||
folderNames = append(folderNames, folder.Name())
|
||||
}
|
||||
}
|
||||
discovery.addExtraFilesystemFolders(folderNames)
|
||||
}
|
||||
|
||||
// If no root filesystem set, use fallback
|
||||
// If no root filesystem set, try the most active I/O device as a last
|
||||
// resort (e.g. ZFS where dataset names are unrelated to disk names).
|
||||
if !hasRoot {
|
||||
rootKey := filepath.Base(rootMountPoint)
|
||||
if _, exists := a.fsStats[rootKey]; exists {
|
||||
rootKey = "root"
|
||||
}
|
||||
slog.Warn("Root device not detected; root I/O disabled", "mountpoint", rootMountPoint)
|
||||
a.fsStats[rootKey] = &system.FsStats{Root: true, Mountpoint: rootMountPoint}
|
||||
discovery.addLastResortRootFs()
|
||||
}
|
||||
|
||||
a.pruneDuplicateRootExtraFilesystems()
|
||||
a.initializeDiskIoStats(diskIoCounters)
|
||||
}
|
||||
|
||||
// Returns matching device from /proc/diskstats.
|
||||
// bool is true if a match was found.
|
||||
func findIoDevice(filesystem string, diskIoCounters map[string]disk.IOCountersStat) (string, bool) {
|
||||
for _, d := range diskIoCounters {
|
||||
if d.Name == filesystem || (d.Label != "" && d.Label == filesystem) {
|
||||
return d.Name, true
|
||||
// Removes extra filesystems that mirror root usage (https://github.com/henrygd/beszel/issues/1428).
|
||||
func (a *Agent) pruneDuplicateRootExtraFilesystems() {
|
||||
var rootMountpoint string
|
||||
for _, stats := range a.fsStats {
|
||||
if stats != nil && stats.Root {
|
||||
rootMountpoint = stats.Mountpoint
|
||||
break
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
if rootMountpoint == "" {
|
||||
return
|
||||
}
|
||||
rootUsage, err := disk.Usage(rootMountpoint)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
for name, stats := range a.fsStats {
|
||||
if stats == nil || stats.Root {
|
||||
continue
|
||||
}
|
||||
extraUsage, err := disk.Usage(stats.Mountpoint)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if hasSameDiskUsage(rootUsage, extraUsage) {
|
||||
slog.Info("Ignoring duplicate FS", "name", name, "mount", stats.Mountpoint)
|
||||
delete(a.fsStats, name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// hasSameDiskUsage compares root/extra usage with a small byte tolerance.
|
||||
func hasSameDiskUsage(a, b *disk.UsageStat) bool {
|
||||
if a == nil || b == nil || a.Total == 0 || b.Total == 0 {
|
||||
return false
|
||||
}
|
||||
// Allow minor drift between sequential disk usage calls.
|
||||
const toleranceBytes uint64 = 16 * 1024 * 1024
|
||||
return withinUsageTolerance(a.Total, b.Total, toleranceBytes) &&
|
||||
withinUsageTolerance(a.Used, b.Used, toleranceBytes)
|
||||
}
|
||||
|
||||
// withinUsageTolerance reports whether two byte values differ by at most tolerance.
|
||||
func withinUsageTolerance(a, b, tolerance uint64) bool {
|
||||
if a >= b {
|
||||
return a-b <= tolerance
|
||||
}
|
||||
return b-a <= tolerance
|
||||
}
|
||||
|
||||
type ioMatchCandidate struct {
|
||||
name string
|
||||
bytes uint64
|
||||
ops uint64
|
||||
}
|
||||
|
||||
// findIoDevice prefers exact device/label matches, then falls back to a
|
||||
// prefix-related candidate with the highest recent activity.
|
||||
func findIoDevice(filesystem string, diskIoCounters map[string]disk.IOCountersStat) (string, bool) {
|
||||
filesystem = normalizeDeviceName(filesystem)
|
||||
if filesystem == "" {
|
||||
return "", false
|
||||
}
|
||||
|
||||
candidates := []ioMatchCandidate{}
|
||||
|
||||
for _, d := range diskIoCounters {
|
||||
if normalizeDeviceName(d.Name) == filesystem || (d.Label != "" && normalizeDeviceName(d.Label) == filesystem) {
|
||||
return d.Name, true
|
||||
}
|
||||
if prefixRelated(normalizeDeviceName(d.Name), filesystem) ||
|
||||
(d.Label != "" && prefixRelated(normalizeDeviceName(d.Label), filesystem)) {
|
||||
candidates = append(candidates, ioMatchCandidate{
|
||||
name: d.Name,
|
||||
bytes: d.ReadBytes + d.WriteBytes,
|
||||
ops: d.ReadCount + d.WriteCount,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
if len(candidates) == 0 {
|
||||
return "", false
|
||||
}
|
||||
|
||||
best := candidates[0]
|
||||
for _, c := range candidates[1:] {
|
||||
if c.bytes > best.bytes ||
|
||||
(c.bytes == best.bytes && c.ops > best.ops) ||
|
||||
(c.bytes == best.bytes && c.ops == best.ops && c.name < best.name) {
|
||||
best = c
|
||||
}
|
||||
}
|
||||
|
||||
slog.Info("Using disk I/O fallback", "requested", filesystem, "selected", best.name)
|
||||
return best.name, true
|
||||
}
|
||||
|
||||
// mostActiveIoDevice returns the device with the highest I/O activity,
|
||||
// or "" if diskIoCounters is empty.
|
||||
func mostActiveIoDevice(diskIoCounters map[string]disk.IOCountersStat) string {
|
||||
var best ioMatchCandidate
|
||||
for _, d := range diskIoCounters {
|
||||
c := ioMatchCandidate{
|
||||
name: d.Name,
|
||||
bytes: d.ReadBytes + d.WriteBytes,
|
||||
ops: d.ReadCount + d.WriteCount,
|
||||
}
|
||||
if best.name == "" || c.bytes > best.bytes ||
|
||||
(c.bytes == best.bytes && c.ops > best.ops) ||
|
||||
(c.bytes == best.bytes && c.ops == best.ops && c.name < best.name) {
|
||||
best = c
|
||||
}
|
||||
}
|
||||
return best.name
|
||||
}
|
||||
|
||||
// prefixRelated reports whether either identifier is a prefix of the other.
|
||||
func prefixRelated(a, b string) bool {
|
||||
if a == "" || b == "" || a == b {
|
||||
return false
|
||||
}
|
||||
return strings.HasPrefix(a, b) || strings.HasPrefix(b, a)
|
||||
}
|
||||
|
||||
// filesystemMatchesPartitionSetting checks whether a FILESYSTEM env var value
|
||||
// matches a partition by mountpoint, exact device name, or prefix relationship
|
||||
// (e.g. FILESYSTEM=ada0 matches partition /dev/ada0p2).
|
||||
func filesystemMatchesPartitionSetting(filesystem string, p disk.PartitionStat) bool {
|
||||
filesystem = strings.TrimSpace(filesystem)
|
||||
if filesystem == "" {
|
||||
return false
|
||||
}
|
||||
if p.Mountpoint == filesystem {
|
||||
return true
|
||||
}
|
||||
|
||||
fsName := normalizeDeviceName(filesystem)
|
||||
partName := normalizeDeviceName(p.Device)
|
||||
if fsName == "" || partName == "" {
|
||||
return false
|
||||
}
|
||||
if fsName == partName {
|
||||
return true
|
||||
}
|
||||
return prefixRelated(partName, fsName)
|
||||
}
|
||||
|
||||
// normalizeDeviceName canonicalizes device strings for comparisons.
|
||||
func normalizeDeviceName(value string) string {
|
||||
name := filepath.Base(strings.TrimSpace(value))
|
||||
if name == "." {
|
||||
return ""
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// Sets start values for disk I/O stats.
|
||||
func (a *Agent) initializeDiskIoStats(diskIoCounters map[string]disk.IOCountersStat) {
|
||||
a.fsNames = a.fsNames[:0]
|
||||
now := time.Now()
|
||||
// ZFS datasets have no /proc/diskstats entry, so they are excluded from
|
||||
// I/O tracking instead of warning about a missing device (#1541).
|
||||
var zfsMountpoints map[string]bool
|
||||
if a.storagePoolManager != nil {
|
||||
zfsMountpoints = a.storagePoolManager.ZfsMountpoints()
|
||||
}
|
||||
for device, stats := range a.fsStats {
|
||||
if zfsMountpoints[stats.Mountpoint] {
|
||||
continue
|
||||
}
|
||||
// skip if not in diskIoCounters
|
||||
d, exists := diskIoCounters[device]
|
||||
if !exists {
|
||||
@@ -221,7 +554,7 @@ func (a *Agent) initializeDiskIoStats(diskIoCounters map[string]disk.IOCountersS
|
||||
continue
|
||||
}
|
||||
// populate initial values
|
||||
stats.Time = time.Now()
|
||||
stats.Time = now
|
||||
stats.TotalRead = d.ReadBytes
|
||||
stats.TotalWrite = d.WriteBytes
|
||||
// add to list of valid io device names
|
||||
@@ -238,20 +571,31 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
|
||||
!a.lastDiskUsageUpdate.IsZero() &&
|
||||
time.Since(a.lastDiskUsageUpdate) < a.diskUsageCacheDuration
|
||||
|
||||
// ZFS dataset mountpoints use `zfs list` values because statfs(2) reports
|
||||
// dataset-level usage that excludes child datasets (#1541).
|
||||
var zfsUsage map[string]zfsDatasetUsage
|
||||
if a.storagePoolManager != nil {
|
||||
zfsUsage = a.storagePoolManager.DatasetUsage()
|
||||
}
|
||||
|
||||
// disk usage
|
||||
for _, stats := range a.fsStats {
|
||||
// Skip non-root filesystems if caching is active
|
||||
if cacheExtraFs && !stats.Root {
|
||||
continue
|
||||
}
|
||||
if d, err := disk.Usage(stats.Mountpoint); err == nil {
|
||||
stats.DiskTotal = bytesToGigabytes(d.Total)
|
||||
stats.DiskUsed = bytesToGigabytes(d.Used)
|
||||
if stats.Root {
|
||||
systemStats.DiskTotal = bytesToGigabytes(d.Total)
|
||||
systemStats.DiskUsed = bytesToGigabytes(d.Used)
|
||||
systemStats.DiskPct = twoDecimals(d.UsedPercent)
|
||||
var total, used uint64
|
||||
var usedPct float64
|
||||
if u, ok := zfsUsage[stats.Mountpoint]; ok {
|
||||
total = u.used + u.avail
|
||||
used = u.used
|
||||
if total > 0 {
|
||||
usedPct = float64(used) / float64(total) * 100
|
||||
}
|
||||
} else if d, err := disk.Usage(stats.Mountpoint); err == nil {
|
||||
total = d.Total
|
||||
used = d.Used
|
||||
usedPct = d.UsedPercent
|
||||
} else {
|
||||
// reset stats if error (likely unmounted)
|
||||
slog.Error("Error getting disk stats", "name", stats.Mountpoint, "err", err)
|
||||
@@ -259,6 +603,14 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
|
||||
stats.DiskUsed = 0
|
||||
stats.TotalRead = 0
|
||||
stats.TotalWrite = 0
|
||||
continue
|
||||
}
|
||||
stats.DiskTotal = utils.BytesToGigabytes(total)
|
||||
stats.DiskUsed = utils.BytesToGigabytes(used)
|
||||
if stats.Root {
|
||||
systemStats.DiskTotal = stats.DiskTotal
|
||||
systemStats.DiskUsed = stats.DiskUsed
|
||||
systemStats.DiskPct = utils.TwoDecimals(usedPct)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -288,36 +640,72 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
prev, hasPrev := a.diskPrev[cacheTimeMs][name]
|
||||
if !hasPrev {
|
||||
// Seed from agent-level fsStats if present, else seed from current
|
||||
prev = prevDisk{readBytes: stats.TotalRead, writeBytes: stats.TotalWrite, at: stats.Time}
|
||||
prev = prevDisk{
|
||||
readBytes: stats.TotalRead,
|
||||
writeBytes: stats.TotalWrite,
|
||||
readTime: d.ReadTime,
|
||||
writeTime: d.WriteTime,
|
||||
ioTime: d.IoTime,
|
||||
weightedIO: d.WeightedIO,
|
||||
readCount: d.ReadCount,
|
||||
writeCount: d.WriteCount,
|
||||
at: stats.Time,
|
||||
}
|
||||
if prev.at.IsZero() {
|
||||
prev = prevDisk{readBytes: d.ReadBytes, writeBytes: d.WriteBytes, at: now}
|
||||
prev = prevDiskFromCounter(d, now)
|
||||
}
|
||||
}
|
||||
|
||||
msElapsed := uint64(now.Sub(prev.at).Milliseconds())
|
||||
|
||||
// Update per-interval snapshot
|
||||
a.diskPrev[cacheTimeMs][name] = prevDiskFromCounter(d, now)
|
||||
|
||||
// Avoid division by zero or clock issues
|
||||
if msElapsed < 100 {
|
||||
// Avoid division by zero or clock issues; update snapshot and continue
|
||||
a.diskPrev[cacheTimeMs][name] = prevDisk{readBytes: d.ReadBytes, writeBytes: d.WriteBytes, at: now}
|
||||
continue
|
||||
}
|
||||
|
||||
diskIORead := (d.ReadBytes - prev.readBytes) * 1000 / msElapsed
|
||||
diskIOWrite := (d.WriteBytes - prev.writeBytes) * 1000 / msElapsed
|
||||
readMbPerSecond := bytesToMegabytes(float64(diskIORead))
|
||||
writeMbPerSecond := bytesToMegabytes(float64(diskIOWrite))
|
||||
readMbPerSecond := utils.BytesToMegabytes(float64(diskIORead))
|
||||
writeMbPerSecond := utils.BytesToMegabytes(float64(diskIOWrite))
|
||||
|
||||
// validate values
|
||||
if readMbPerSecond > 50_000 || writeMbPerSecond > 50_000 {
|
||||
slog.Warn("Invalid disk I/O. Resetting.", "name", d.Name, "read", readMbPerSecond, "write", writeMbPerSecond)
|
||||
// Reset interval snapshot and seed from current
|
||||
a.diskPrev[cacheTimeMs][name] = prevDisk{readBytes: d.ReadBytes, writeBytes: d.WriteBytes, at: now}
|
||||
// also refresh agent baseline to avoid future negatives
|
||||
a.initializeDiskIoStats(ioCounters)
|
||||
continue
|
||||
}
|
||||
|
||||
// Update per-interval snapshot
|
||||
a.diskPrev[cacheTimeMs][name] = prevDisk{readBytes: d.ReadBytes, writeBytes: d.WriteBytes, at: now}
|
||||
// These properties are calculated differently on different platforms,
|
||||
// but generally represent cumulative time spent doing reads/writes on the device.
|
||||
// This can surpass 100% if there are multiple concurrent I/O operations.
|
||||
// Linux kernel docs:
|
||||
// This is the total number of milliseconds spent by all reads (as
|
||||
// measured from __make_request() to end_that_request_last()).
|
||||
// https://www.kernel.org/doc/Documentation/iostats.txt (fields 4, 8)
|
||||
diskReadTime := utils.TwoDecimals(float64(d.ReadTime-prev.readTime) / float64(msElapsed) * 100)
|
||||
diskWriteTime := utils.TwoDecimals(float64(d.WriteTime-prev.writeTime) / float64(msElapsed) * 100)
|
||||
|
||||
// I/O utilization %: fraction of wall time the device had any I/O in progress (0-100).
|
||||
diskIoUtilPct := utils.TwoDecimals(float64(d.IoTime-prev.ioTime) / float64(msElapsed) * 100)
|
||||
|
||||
// Weighted I/O: queue-depth weighted I/O time, normalized to interval (can exceed 100%).
|
||||
// Linux kernel field 11: incremented by iops_in_progress × ms_since_last_update.
|
||||
// Used to display queue depth. Multipled by 100 to increase accuracy of digit truncation (divided by 100 in UI).
|
||||
diskWeightedIO := utils.TwoDecimals(float64(d.WeightedIO-prev.weightedIO) / float64(msElapsed) * 100)
|
||||
|
||||
// r_await / w_await: average time per read/write operation in milliseconds.
|
||||
// Equivalent to r_await and w_await in iostat.
|
||||
var rAwait, wAwait float64
|
||||
if deltaReadCount := d.ReadCount - prev.readCount; deltaReadCount > 0 {
|
||||
rAwait = utils.TwoDecimals(float64(d.ReadTime-prev.readTime) / float64(deltaReadCount))
|
||||
}
|
||||
if deltaWriteCount := d.WriteCount - prev.writeCount; deltaWriteCount > 0 {
|
||||
wAwait = utils.TwoDecimals(float64(d.WriteTime-prev.writeTime) / float64(deltaWriteCount))
|
||||
}
|
||||
|
||||
// Update global fsStats baseline for cross-interval correctness
|
||||
stats.Time = now
|
||||
@@ -327,20 +715,42 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
stats.DiskWritePs = writeMbPerSecond
|
||||
stats.DiskReadBytes = diskIORead
|
||||
stats.DiskWriteBytes = diskIOWrite
|
||||
stats.DiskIoStats[0] = diskReadTime
|
||||
stats.DiskIoStats[1] = diskWriteTime
|
||||
stats.DiskIoStats[2] = diskIoUtilPct
|
||||
stats.DiskIoStats[3] = rAwait
|
||||
stats.DiskIoStats[4] = wAwait
|
||||
stats.DiskIoStats[5] = diskWeightedIO
|
||||
|
||||
if stats.Root {
|
||||
systemStats.DiskReadPs = stats.DiskReadPs
|
||||
systemStats.DiskWritePs = stats.DiskWritePs
|
||||
systemStats.DiskIO[0] = diskIORead
|
||||
systemStats.DiskIO[1] = diskIOWrite
|
||||
systemStats.DiskIOTotal[0] = d.ReadBytes
|
||||
systemStats.DiskIOTotal[1] = d.WriteBytes
|
||||
systemStats.DiskIoStats[0] = diskReadTime
|
||||
systemStats.DiskIoStats[1] = diskWriteTime
|
||||
systemStats.DiskIoStats[2] = diskIoUtilPct
|
||||
systemStats.DiskIoStats[3] = rAwait
|
||||
systemStats.DiskIoStats[4] = wAwait
|
||||
systemStats.DiskIoStats[5] = diskWeightedIO
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// getRootMountPoint returns the appropriate root mount point for the system
|
||||
// getRootMountPoint returns the appropriate root mount point for the system.
|
||||
// On Windows it returns the system drive (e.g. "C:").
|
||||
// For immutable systems like Fedora Silverblue, it returns /sysroot instead of /
|
||||
func (a *Agent) getRootMountPoint() string {
|
||||
if runtime.GOOS == "windows" {
|
||||
if sd := os.Getenv("SystemDrive"); sd != "" {
|
||||
return sd
|
||||
}
|
||||
return "C:"
|
||||
}
|
||||
|
||||
// 1. Check if /etc/os-release contains indicators of an immutable system
|
||||
if osReleaseContent, err := os.ReadFile("/etc/os-release"); err == nil {
|
||||
content := string(osReleaseContent)
|
||||
|
||||
+691
-23
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
@@ -79,14 +78,7 @@ func TestParseFilesystemEntry(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
fsEntry := strings.TrimSpace(tt.input)
|
||||
var fs, customName string
|
||||
if parts := strings.SplitN(fsEntry, "__", 2); len(parts) == 2 {
|
||||
fs = strings.TrimSpace(parts[0])
|
||||
customName = strings.TrimSpace(parts[1])
|
||||
} else {
|
||||
fs = fsEntry
|
||||
}
|
||||
fs, customName := parseFilesystemEntry(tt.input)
|
||||
|
||||
assert.Equal(t, tt.expectedFs, fs)
|
||||
assert.Equal(t, tt.expectedName, customName)
|
||||
@@ -94,6 +86,528 @@ func TestParseFilesystemEntry(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtraFilesystemPartitionInfo(t *testing.T) {
|
||||
t.Run("uses partition device for label-only mountpoint", func(t *testing.T) {
|
||||
device, customName := extraFilesystemPartitionInfo(disk.PartitionStat{
|
||||
Device: "/dev/sdc",
|
||||
Mountpoint: "/extra-filesystems/Share",
|
||||
})
|
||||
|
||||
assert.Equal(t, "/dev/sdc", device)
|
||||
assert.Equal(t, "", customName)
|
||||
})
|
||||
|
||||
t.Run("uses custom name from mountpoint suffix", func(t *testing.T) {
|
||||
device, customName := extraFilesystemPartitionInfo(disk.PartitionStat{
|
||||
Device: "/dev/sdc",
|
||||
Mountpoint: "/extra-filesystems/sdc__Share",
|
||||
})
|
||||
|
||||
assert.Equal(t, "/dev/sdc", device)
|
||||
assert.Equal(t, "Share", customName)
|
||||
})
|
||||
|
||||
t.Run("falls back to folder device when partition device is unavailable", func(t *testing.T) {
|
||||
device, customName := extraFilesystemPartitionInfo(disk.PartitionStat{
|
||||
Mountpoint: "/extra-filesystems/sdc__Share",
|
||||
})
|
||||
|
||||
assert.Equal(t, "sdc", device)
|
||||
assert.Equal(t, "Share", customName)
|
||||
})
|
||||
|
||||
t.Run("supports custom name without folder device prefix", func(t *testing.T) {
|
||||
device, customName := extraFilesystemPartitionInfo(disk.PartitionStat{
|
||||
Device: "/dev/sdc",
|
||||
Mountpoint: "/extra-filesystems/__Share",
|
||||
})
|
||||
|
||||
assert.Equal(t, "/dev/sdc", device)
|
||||
assert.Equal(t, "Share", customName)
|
||||
})
|
||||
}
|
||||
|
||||
func TestBuildFsStatRegistration(t *testing.T) {
|
||||
t.Run("uses basename for non-windows exact io match", func(t *testing.T) {
|
||||
key, stats, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
"/dev/sda1",
|
||||
"/mnt/data",
|
||||
false,
|
||||
"archive",
|
||||
fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"sda1": {Name: "sda1"},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "sda1", key)
|
||||
assert.Equal(t, "/mnt/data", stats.Mountpoint)
|
||||
assert.Equal(t, "archive", stats.Name)
|
||||
assert.False(t, stats.Root)
|
||||
})
|
||||
|
||||
t.Run("maps root partition to io device by prefix", func(t *testing.T) {
|
||||
key, stats, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
"/dev/ada0p2",
|
||||
"/",
|
||||
true,
|
||||
"",
|
||||
fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"ada0": {Name: "ada0", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "ada0", key)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/", stats.Mountpoint)
|
||||
})
|
||||
|
||||
t.Run("uses filesystem setting as root fallback", func(t *testing.T) {
|
||||
key, _, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
"overlay",
|
||||
"/",
|
||||
true,
|
||||
"",
|
||||
fsRegistrationContext{
|
||||
filesystem: "nvme0n1p2",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"nvme0n1": {Name: "nvme0n1", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "nvme0n1", key)
|
||||
})
|
||||
|
||||
t.Run("prefers parsed extra-filesystems device over mapper device", func(t *testing.T) {
|
||||
key, stats, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
"/dev/mapper/luks-2bcb02be-999d-4417-8d18-5c61e660fb6e",
|
||||
"/extra-filesystems/nvme0n1p2__Archive",
|
||||
false,
|
||||
"Archive",
|
||||
fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"dm-1": {Name: "dm-1", Label: "luks-2bcb02be-999d-4417-8d18-5c61e660fb6e"},
|
||||
"nvme0n1p2": {Name: "nvme0n1p2"},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "nvme0n1p2", key)
|
||||
assert.Equal(t, "Archive", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("falls back to mapper io device when folder device cannot be resolved", func(t *testing.T) {
|
||||
key, stats, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
"/dev/mapper/luks-2bcb02be-999d-4417-8d18-5c61e660fb6e",
|
||||
"/extra-filesystems/Archive",
|
||||
false,
|
||||
"Archive",
|
||||
fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"dm-1": {Name: "dm-1", Label: "luks-2bcb02be-999d-4417-8d18-5c61e660fb6e"},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "dm-1", key)
|
||||
assert.Equal(t, "Archive", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("uses full device name on windows", func(t *testing.T) {
|
||||
key, _, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{},
|
||||
`C:`,
|
||||
`C:\\`,
|
||||
false,
|
||||
"",
|
||||
fsRegistrationContext{
|
||||
isWindows: true,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
`C:`: {Name: `C:`},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, `C:`, key)
|
||||
})
|
||||
|
||||
t.Run("skips existing key", func(t *testing.T) {
|
||||
key, stats, ok := registerFilesystemStats(
|
||||
map[string]*system.FsStats{"sda1": {Mountpoint: "/existing"}},
|
||||
"/dev/sda1",
|
||||
"/mnt/data",
|
||||
false,
|
||||
"",
|
||||
fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"sda1": {Name: "sda1"},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
assert.False(t, ok)
|
||||
assert.Empty(t, key)
|
||||
assert.Nil(t, stats)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddConfiguredRootFs(t *testing.T) {
|
||||
t.Run("adds root from matching partition", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
rootMountPoint: "/",
|
||||
partitions: []disk.PartitionStat{{Device: "/dev/ada0p2", Mountpoint: "/"}},
|
||||
ctx: fsRegistrationContext{
|
||||
filesystem: "/dev/ada0p2",
|
||||
filesystemName: "root disk",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"ada0": {Name: "ada0", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addConfiguredRootFs()
|
||||
|
||||
assert.True(t, ok)
|
||||
stats, exists := agent.fsStats["ada0"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/", stats.Mountpoint)
|
||||
assert.Equal(t, "root disk", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("adds root from io device when partition is missing", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
rootMountPoint: "/sysroot",
|
||||
ctx: fsRegistrationContext{
|
||||
filesystem: "zroot",
|
||||
filesystemName: "root pool",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"nda0": {Name: "nda0", Label: "zroot", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addConfiguredRootFs()
|
||||
|
||||
assert.True(t, ok)
|
||||
stats, exists := agent.fsStats["nda0"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/sysroot", stats.Mountpoint)
|
||||
assert.Equal(t, "root pool", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("returns false when filesystem cannot be resolved", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
rootMountPoint: "/",
|
||||
ctx: fsRegistrationContext{
|
||||
filesystem: "missing-disk",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addConfiguredRootFs()
|
||||
|
||||
assert.False(t, ok)
|
||||
assert.Empty(t, agent.fsStats)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddPartitionRootFs(t *testing.T) {
|
||||
t.Run("adds root from fallback partition candidate", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"nvme0n1": {Name: "nvme0n1", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addPartitionRootFs("/dev/nvme0n1p2", "/")
|
||||
|
||||
assert.True(t, ok)
|
||||
stats, exists := agent.fsStats["nvme0n1"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/", stats.Mountpoint)
|
||||
})
|
||||
|
||||
t.Run("returns false when no io device matches", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{agent: agent, ctx: fsRegistrationContext{diskIoCounters: map[string]disk.IOCountersStat{}}}
|
||||
|
||||
ok := discovery.addPartitionRootFs("/dev/mapper/root", "/")
|
||||
|
||||
assert.False(t, ok)
|
||||
assert.Empty(t, agent.fsStats)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddLastResortRootFs(t *testing.T) {
|
||||
t.Run("uses most active io device when available", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{agent: agent, rootMountPoint: "/", ctx: fsRegistrationContext{diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"sda": {Name: "sda", ReadBytes: 5000, WriteBytes: 5000},
|
||||
"sdb": {Name: "sdb", ReadBytes: 1000, WriteBytes: 1000},
|
||||
}}}
|
||||
|
||||
discovery.addLastResortRootFs()
|
||||
|
||||
stats, exists := agent.fsStats["sda"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
})
|
||||
|
||||
t.Run("falls back to root key when mountpoint basename collides", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: map[string]*system.FsStats{
|
||||
"sysroot": {Mountpoint: "/extra-filesystems/sysroot"},
|
||||
}}
|
||||
discovery := diskDiscovery{agent: agent, rootMountPoint: "/sysroot", ctx: fsRegistrationContext{diskIoCounters: map[string]disk.IOCountersStat{}}}
|
||||
|
||||
discovery.addLastResortRootFs()
|
||||
|
||||
stats, exists := agent.fsStats["root"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/sysroot", stats.Mountpoint)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddConfiguredExtraFsEntry(t *testing.T) {
|
||||
t.Run("uses matching partition when present", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
partitions: []disk.PartitionStat{{Device: "/dev/sdb1", Mountpoint: "/mnt/backup"}},
|
||||
usageFn: func(string) (*disk.UsageStat, error) {
|
||||
t.Fatal("usage fallback should not be called when partition matches")
|
||||
return nil, nil
|
||||
},
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"sdb1": {Name: "sdb1"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
discovery.addConfiguredExtraFsEntry("sdb1", "backup")
|
||||
|
||||
stats, exists := agent.fsStats["sdb1"]
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "/mnt/backup", stats.Mountpoint)
|
||||
assert.Equal(t, "backup", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("falls back to usage-validated path", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
usageFn: func(path string) (*disk.UsageStat, error) {
|
||||
assert.Equal(t, "/srv/archive", path)
|
||||
return &disk.UsageStat{}, nil
|
||||
},
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"archive": {Name: "archive"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
discovery.addConfiguredExtraFsEntry("/srv/archive", "archive")
|
||||
|
||||
stats, exists := agent.fsStats["archive"]
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "/srv/archive", stats.Mountpoint)
|
||||
assert.Equal(t, "archive", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("ignores invalid filesystem entry", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
usageFn: func(string) (*disk.UsageStat, error) {
|
||||
return nil, os.ErrNotExist
|
||||
},
|
||||
}
|
||||
|
||||
discovery.addConfiguredExtraFsEntry("/missing/archive", "")
|
||||
|
||||
assert.Empty(t, agent.fsStats)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddConfiguredExtraFilesystems(t *testing.T) {
|
||||
t.Run("parses and registers multiple configured filesystems", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
partitions: []disk.PartitionStat{{Device: "/dev/sda1", Mountpoint: "/mnt/fast"}},
|
||||
usageFn: func(path string) (*disk.UsageStat, error) {
|
||||
if path == "/srv/archive" {
|
||||
return &disk.UsageStat{}, nil
|
||||
}
|
||||
return nil, os.ErrNotExist
|
||||
},
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"sda1": {Name: "sda1"},
|
||||
"archive": {Name: "archive"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
discovery.addConfiguredExtraFilesystems("sda1__fast,/srv/archive__cold")
|
||||
|
||||
assert.Contains(t, agent.fsStats, "sda1")
|
||||
assert.Equal(t, "fast", agent.fsStats["sda1"].Name)
|
||||
assert.Contains(t, agent.fsStats, "archive")
|
||||
assert.Equal(t, "cold", agent.fsStats["archive"].Name)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddExtraFilesystemFolders(t *testing.T) {
|
||||
t.Run("adds missing folders and skips existing mountpoints", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: map[string]*system.FsStats{
|
||||
"existing": {Mountpoint: "/extra-filesystems/existing"},
|
||||
}}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
efPath: "/extra-filesystems",
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"newdisk": {Name: "newdisk"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
discovery.addExtraFilesystemFolders([]string{"existing", "newdisk__Archive"})
|
||||
|
||||
assert.Len(t, agent.fsStats, 2)
|
||||
stats, exists := agent.fsStats["newdisk"]
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "/extra-filesystems/newdisk__Archive", stats.Mountpoint)
|
||||
assert.Equal(t, "Archive", stats.Name)
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddPartitionExtraFs(t *testing.T) {
|
||||
makeDiscovery := func(agent *Agent) diskDiscovery {
|
||||
return diskDiscovery{
|
||||
agent: agent,
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: false,
|
||||
efPath: "/extra-filesystems",
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"nvme0n1p1": {Name: "nvme0n1p1"},
|
||||
"nvme1n1": {Name: "nvme1n1"},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
t.Run("registers direct child of extra-filesystems", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
d := makeDiscovery(agent)
|
||||
|
||||
d.addPartitionExtraFs(disk.PartitionStat{
|
||||
Device: "/dev/nvme0n1p1",
|
||||
Mountpoint: "/extra-filesystems/nvme0n1p1__caddy1-root",
|
||||
})
|
||||
|
||||
stats, exists := agent.fsStats["nvme0n1p1"]
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "/extra-filesystems/nvme0n1p1__caddy1-root", stats.Mountpoint)
|
||||
assert.Equal(t, "caddy1-root", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("skips nested mount under extra-filesystem bind mount", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
d := makeDiscovery(agent)
|
||||
|
||||
// These simulate the virtual mounts that appear when host / is bind-mounted
|
||||
// with disk.Partitions(all=true) — e.g. /proc, /sys, /dev visible under the mount.
|
||||
for _, nested := range []string{
|
||||
"/extra-filesystems/nvme0n1p1__caddy1-root/proc",
|
||||
"/extra-filesystems/nvme0n1p1__caddy1-root/sys",
|
||||
"/extra-filesystems/nvme0n1p1__caddy1-root/dev",
|
||||
"/extra-filesystems/nvme0n1p1__caddy1-root/run",
|
||||
} {
|
||||
d.addPartitionExtraFs(disk.PartitionStat{Device: "tmpfs", Mountpoint: nested})
|
||||
}
|
||||
|
||||
assert.Empty(t, agent.fsStats)
|
||||
})
|
||||
|
||||
t.Run("registers both direct children, skips their nested mounts", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
d := makeDiscovery(agent)
|
||||
|
||||
partitions := []disk.PartitionStat{
|
||||
{Device: "/dev/nvme0n1p1", Mountpoint: "/extra-filesystems/nvme0n1p1__caddy1-root"},
|
||||
{Device: "/dev/nvme1n1", Mountpoint: "/extra-filesystems/nvme1n1__caddy1-docker"},
|
||||
{Device: "proc", Mountpoint: "/extra-filesystems/nvme0n1p1__caddy1-root/proc"},
|
||||
{Device: "sysfs", Mountpoint: "/extra-filesystems/nvme0n1p1__caddy1-root/sys"},
|
||||
{Device: "overlay", Mountpoint: "/extra-filesystems/nvme0n1p1__caddy1-root/var/lib/docker"},
|
||||
}
|
||||
for _, p := range partitions {
|
||||
d.addPartitionExtraFs(p)
|
||||
}
|
||||
|
||||
assert.Len(t, agent.fsStats, 2)
|
||||
assert.Equal(t, "caddy1-root", agent.fsStats["nvme0n1p1"].Name)
|
||||
assert.Equal(t, "caddy1-docker", agent.fsStats["nvme1n1"].Name)
|
||||
})
|
||||
|
||||
t.Run("skips partition not under extra-filesystems", func(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
d := makeDiscovery(agent)
|
||||
|
||||
d.addPartitionExtraFs(disk.PartitionStat{
|
||||
Device: "/dev/nvme0n1p1",
|
||||
Mountpoint: "/",
|
||||
})
|
||||
|
||||
assert.Empty(t, agent.fsStats)
|
||||
})
|
||||
}
|
||||
|
||||
func TestFindIoDevice(t *testing.T) {
|
||||
t.Run("matches by device name", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
@@ -117,7 +631,7 @@ func TestFindIoDevice(t *testing.T) {
|
||||
assert.Equal(t, "sda", device)
|
||||
})
|
||||
|
||||
t.Run("returns no fallback when not found", func(t *testing.T) {
|
||||
t.Run("returns no match when not found", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"sda": {Name: "sda"},
|
||||
"sdb": {Name: "sdb"},
|
||||
@@ -127,6 +641,106 @@ func TestFindIoDevice(t *testing.T) {
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, "", device)
|
||||
})
|
||||
|
||||
t.Run("uses uncertain unique prefix fallback", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"nvme0n1": {Name: "nvme0n1"},
|
||||
"sda": {Name: "sda"},
|
||||
}
|
||||
|
||||
device, ok := findIoDevice("nvme0n1p2", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "nvme0n1", device)
|
||||
})
|
||||
|
||||
t.Run("uses dominant activity when prefix matches are ambiguous", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"sda": {Name: "sda", ReadBytes: 5000, WriteBytes: 5000, ReadCount: 100, WriteCount: 100},
|
||||
"sdb": {Name: "sdb", ReadBytes: 1000, WriteBytes: 1000, ReadCount: 50, WriteCount: 50},
|
||||
}
|
||||
|
||||
device, ok := findIoDevice("sd", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "sda", device)
|
||||
})
|
||||
|
||||
t.Run("uses highest activity when ambiguous without dominance", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"sda": {Name: "sda", ReadBytes: 3000, WriteBytes: 3000, ReadCount: 50, WriteCount: 50},
|
||||
"sdb": {Name: "sdb", ReadBytes: 2500, WriteBytes: 2500, ReadCount: 40, WriteCount: 40},
|
||||
}
|
||||
|
||||
device, ok := findIoDevice("sd", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "sda", device)
|
||||
})
|
||||
|
||||
t.Run("matches /dev/-prefixed partition to parent disk", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"nda0": {Name: "nda0", ReadBytes: 1000, WriteBytes: 1000},
|
||||
}
|
||||
|
||||
device, ok := findIoDevice("/dev/nda0p2", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "nda0", device)
|
||||
})
|
||||
|
||||
t.Run("uses deterministic name tie-breaker", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"sdb": {Name: "sdb", ReadBytes: 2000, WriteBytes: 2000, ReadCount: 10, WriteCount: 10},
|
||||
"sda": {Name: "sda", ReadBytes: 2000, WriteBytes: 2000, ReadCount: 10, WriteCount: 10},
|
||||
}
|
||||
|
||||
device, ok := findIoDevice("sd", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "sda", device)
|
||||
})
|
||||
}
|
||||
|
||||
func TestFilesystemMatchesPartitionSetting(t *testing.T) {
|
||||
p := disk.PartitionStat{Device: "/dev/ada0p2", Mountpoint: "/"}
|
||||
|
||||
t.Run("matches mountpoint setting", func(t *testing.T) {
|
||||
assert.True(t, filesystemMatchesPartitionSetting("/", p))
|
||||
})
|
||||
|
||||
t.Run("matches exact partition setting", func(t *testing.T) {
|
||||
assert.True(t, filesystemMatchesPartitionSetting("ada0p2", p))
|
||||
assert.True(t, filesystemMatchesPartitionSetting("/dev/ada0p2", p))
|
||||
})
|
||||
|
||||
t.Run("matches prefix-style parent setting", func(t *testing.T) {
|
||||
assert.True(t, filesystemMatchesPartitionSetting("ada0", p))
|
||||
assert.True(t, filesystemMatchesPartitionSetting("/dev/ada0", p))
|
||||
})
|
||||
|
||||
t.Run("does not match unrelated device", func(t *testing.T) {
|
||||
assert.False(t, filesystemMatchesPartitionSetting("sda", p))
|
||||
assert.False(t, filesystemMatchesPartitionSetting("nvme0n1", p))
|
||||
assert.False(t, filesystemMatchesPartitionSetting("", p))
|
||||
})
|
||||
}
|
||||
|
||||
func TestMostActiveIoDevice(t *testing.T) {
|
||||
t.Run("returns most active device", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"nda0": {Name: "nda0", ReadBytes: 5000, WriteBytes: 5000, ReadCount: 100, WriteCount: 100},
|
||||
"nda1": {Name: "nda1", ReadBytes: 1000, WriteBytes: 1000, ReadCount: 50, WriteCount: 50},
|
||||
}
|
||||
assert.Equal(t, "nda0", mostActiveIoDevice(ioCounters))
|
||||
})
|
||||
|
||||
t.Run("uses deterministic tie-breaker", func(t *testing.T) {
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"sdb": {Name: "sdb", ReadBytes: 1000, WriteBytes: 1000, ReadCount: 10, WriteCount: 10},
|
||||
"sda": {Name: "sda", ReadBytes: 1000, WriteBytes: 1000, ReadCount: 10, WriteCount: 10},
|
||||
}
|
||||
assert.Equal(t, "sda", mostActiveIoDevice(ioCounters))
|
||||
})
|
||||
|
||||
t.Run("returns empty for empty map", func(t *testing.T) {
|
||||
assert.Equal(t, "", mostActiveIoDevice(map[string]disk.IOCountersStat{}))
|
||||
})
|
||||
}
|
||||
|
||||
func TestIsDockerSpecialMountpoint(t *testing.T) {
|
||||
@@ -151,18 +765,8 @@ func TestIsDockerSpecialMountpoint(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestInitializeDiskInfoWithCustomNames(t *testing.T) {
|
||||
// Set up environment variables
|
||||
oldEnv := os.Getenv("EXTRA_FILESYSTEMS")
|
||||
defer func() {
|
||||
if oldEnv != "" {
|
||||
os.Setenv("EXTRA_FILESYSTEMS", oldEnv)
|
||||
} else {
|
||||
os.Unsetenv("EXTRA_FILESYSTEMS")
|
||||
}
|
||||
}()
|
||||
|
||||
// Test with custom names
|
||||
os.Setenv("EXTRA_FILESYSTEMS", "sda1__my-storage,/dev/sdb1__backup-drive,nvme0n1p2")
|
||||
t.Setenv("EXTRA_FILESYSTEMS", "sda1__my-storage,/dev/sdb1__backup-drive,nvme0n1p2")
|
||||
|
||||
// Mock disk partitions (we'll just test the parsing logic)
|
||||
// Since the actual disk operations are system-dependent, we'll focus on the parsing
|
||||
@@ -190,7 +794,7 @@ func TestInitializeDiskInfoWithCustomNames(t *testing.T) {
|
||||
|
||||
for _, tc := range testCases {
|
||||
t.Run("env_"+tc.envValue, func(t *testing.T) {
|
||||
os.Setenv("EXTRA_FILESYSTEMS", tc.envValue)
|
||||
t.Setenv("EXTRA_FILESYSTEMS", tc.envValue)
|
||||
|
||||
// Create mock partitions that would match our test cases
|
||||
partitions := []disk.PartitionStat{}
|
||||
@@ -211,7 +815,7 @@ func TestInitializeDiskInfoWithCustomNames(t *testing.T) {
|
||||
// Test the parsing logic by calling the relevant part
|
||||
// We'll create a simplified version to test just the parsing
|
||||
extraFilesystems := tc.envValue
|
||||
for _, fsEntry := range strings.Split(extraFilesystems, ",") {
|
||||
for fsEntry := range strings.SplitSeq(extraFilesystems, ",") {
|
||||
// Parse the entry
|
||||
fsEntry = strings.TrimSpace(fsEntry)
|
||||
var fs, customName string
|
||||
@@ -373,3 +977,67 @@ func TestDiskUsageCaching(t *testing.T) {
|
||||
"lastDiskUsageUpdate should be refreshed when cache expires")
|
||||
})
|
||||
}
|
||||
|
||||
func TestHasSameDiskUsage(t *testing.T) {
|
||||
const toleranceBytes uint64 = 16 * 1024 * 1024
|
||||
|
||||
t.Run("returns true when totals and usage are equal", func(t *testing.T) {
|
||||
a := &disk.UsageStat{Total: 100 * 1024 * 1024 * 1024, Used: 42 * 1024 * 1024 * 1024}
|
||||
b := &disk.UsageStat{Total: 100 * 1024 * 1024 * 1024, Used: 42 * 1024 * 1024 * 1024}
|
||||
assert.True(t, hasSameDiskUsage(a, b))
|
||||
})
|
||||
|
||||
t.Run("returns true within tolerance", func(t *testing.T) {
|
||||
a := &disk.UsageStat{Total: 100 * 1024 * 1024 * 1024, Used: 42 * 1024 * 1024 * 1024}
|
||||
b := &disk.UsageStat{
|
||||
Total: a.Total + toleranceBytes - 1,
|
||||
Used: a.Used - toleranceBytes + 1,
|
||||
}
|
||||
assert.True(t, hasSameDiskUsage(a, b))
|
||||
})
|
||||
|
||||
t.Run("returns false when total exceeds tolerance", func(t *testing.T) {
|
||||
a := &disk.UsageStat{Total: 100 * 1024 * 1024 * 1024, Used: 42 * 1024 * 1024 * 1024}
|
||||
b := &disk.UsageStat{
|
||||
Total: a.Total + toleranceBytes + 1,
|
||||
Used: a.Used,
|
||||
}
|
||||
assert.False(t, hasSameDiskUsage(a, b))
|
||||
})
|
||||
|
||||
t.Run("returns false for nil or zero total", func(t *testing.T) {
|
||||
assert.False(t, hasSameDiskUsage(nil, &disk.UsageStat{Total: 1, Used: 1}))
|
||||
assert.False(t, hasSameDiskUsage(&disk.UsageStat{Total: 1, Used: 1}, nil))
|
||||
assert.False(t, hasSameDiskUsage(&disk.UsageStat{Total: 0, Used: 0}, &disk.UsageStat{Total: 1, Used: 1}))
|
||||
})
|
||||
}
|
||||
|
||||
func TestInitializeDiskIoStatsResetsTrackedDevices(t *testing.T) {
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"sda": {},
|
||||
"sdb": {},
|
||||
},
|
||||
fsNames: []string{"stale", "sda"},
|
||||
}
|
||||
|
||||
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
|
||||
"sda": {Name: "sda", ReadBytes: 10, WriteBytes: 20},
|
||||
"sdb": {Name: "sdb", ReadBytes: 30, WriteBytes: 40},
|
||||
})
|
||||
|
||||
assert.ElementsMatch(t, []string{"sda", "sdb"}, agent.fsNames)
|
||||
assert.Len(t, agent.fsNames, 2)
|
||||
assert.Equal(t, uint64(10), agent.fsStats["sda"].TotalRead)
|
||||
assert.Equal(t, uint64(20), agent.fsStats["sda"].TotalWrite)
|
||||
assert.False(t, agent.fsStats["sda"].Time.IsZero())
|
||||
assert.False(t, agent.fsStats["sdb"].Time.IsZero())
|
||||
|
||||
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
|
||||
"sdb": {Name: "sdb", ReadBytes: 50, WriteBytes: 60},
|
||||
})
|
||||
|
||||
assert.Equal(t, []string{"sdb"}, agent.fsNames)
|
||||
assert.Equal(t, uint64(50), agent.fsStats["sdb"].TotalRead)
|
||||
assert.Equal(t, uint64(60), agent.fsStats["sdb"].TotalWrite)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/shirou/gopsutil/v4/disk"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// TestUpdateDiskUsageZfsMountpoint verifies that a filesystem whose mountpoint
|
||||
// is a ZFS dataset reports `zfs list` usage (which includes child datasets)
|
||||
// instead of the dataset-scoped statfs values (#1541).
|
||||
func TestUpdateDiskUsageZfsMountpoint(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
|
||||
}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"tank": {Root: false, Mountpoint: "/tank"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
fs := agent.fsStats["tank"]
|
||||
require.NotNil(t, fs)
|
||||
assert.Equal(t, 22350.81, fs.DiskTotal) // (used + avail) in GiB
|
||||
assert.Equal(t, 11175.87, fs.DiskUsed)
|
||||
// Non-root filesystems do not populate system-level stats.
|
||||
assert.Equal(t, float64(0), stats.DiskTotal)
|
||||
}
|
||||
|
||||
// TestUpdateDiskUsageZfsRootPopulatesSystemStats verifies the root disk values
|
||||
// are derived from ZFS usage when the root mountpoint is a ZFS dataset.
|
||||
func TestUpdateDiskUsageZfsRootPopulatesSystemStats(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "rpool/ROOT/pve-1", Used: 900000000000, Avail: 300000000000, Mountpoint: "/"},
|
||||
}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"rpool/ROOT/pve-1": {Root: true, Mountpoint: "/"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
assert.Equal(t, 1117.59, agent.fsStats["rpool/ROOT/pve-1"].DiskTotal)
|
||||
assert.Equal(t, 838.19, agent.fsStats["rpool/ROOT/pve-1"].DiskUsed)
|
||||
assert.Equal(t, 75.0, stats.DiskPct)
|
||||
assert.Equal(t, 1117.59, stats.DiskTotal)
|
||||
assert.Equal(t, 838.19, stats.DiskUsed)
|
||||
}
|
||||
|
||||
// TestUpdateDiskUsageWithoutZfsManager falls back to statfs when no manager is
|
||||
// present (e.g. tests constructing bare Agent values).
|
||||
func TestUpdateDiskUsageWithoutZfsManager(t *testing.T) {
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"root": {Root: true, Mountpoint: "/"},
|
||||
},
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
assert.True(t, agent.fsStats["root"].DiskTotal > 0, "root usage should come from statfs")
|
||||
assert.True(t, stats.DiskTotal > 0)
|
||||
}
|
||||
|
||||
// TestInitializeDiskIoStatsSkipsZfsMountpoints verifies ZFS filesystems are
|
||||
// excluded from diskstats I/O tracking instead of warning about a missing device.
|
||||
func TestInitializeDiskIoStatsSkipsZfsMountpoints(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank", Mountpoint: "/tank"}}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"tank": {Root: false, Mountpoint: "/tank"},
|
||||
"sda1": {Root: false, Mountpoint: "/mnt/data"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
diskPrev: make(map[uint16]map[string]prevDisk),
|
||||
}
|
||||
|
||||
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
|
||||
"sda1": {Name: "sda1", ReadBytes: 100, WriteBytes: 100},
|
||||
})
|
||||
|
||||
assert.Equal(t, []string{"sda1"}, agent.fsNames)
|
||||
assert.Equal(t, uint64(100), agent.fsStats["sda1"].TotalRead)
|
||||
// ZFS entry is present but untouched by diskstats initialization.
|
||||
assert.Equal(t, uint64(0), agent.fsStats["tank"].TotalRead)
|
||||
}
|
||||
+340
-84
@@ -1,6 +1,7 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/binary"
|
||||
@@ -15,12 +16,16 @@ import (
|
||||
"os"
|
||||
"path"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/deltatracker"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
"github.com/blang/semver"
|
||||
)
|
||||
@@ -28,6 +33,7 @@ import (
|
||||
// ansiEscapePattern matches ANSI escape sequences (colors, cursor movement, etc.)
|
||||
// This includes CSI sequences like \x1b[...m and simple escapes like \x1b[K
|
||||
var ansiEscapePattern = regexp.MustCompile(`\x1b\[[0-9;]*[a-zA-Z]|\x1b\][^\x07]*\x07|\x1b[@-Z\\-_]`)
|
||||
var dockerContainerIDPattern = regexp.MustCompile(`^[a-fA-F0-9]{12,64}$`)
|
||||
|
||||
const (
|
||||
// Docker API timeout in milliseconds
|
||||
@@ -47,20 +53,25 @@ const (
|
||||
)
|
||||
|
||||
type dockerManager struct {
|
||||
client *http.Client // Client to query Docker API
|
||||
wg sync.WaitGroup // WaitGroup to wait for all goroutines to finish
|
||||
sem chan struct{} // Semaphore to limit concurrent container requests
|
||||
containerStatsMutex sync.RWMutex // Mutex to prevent concurrent access to containerStatsMap
|
||||
apiContainerList []*container.ApiInfo // List of containers from Docker API
|
||||
containerStatsMap map[string]*container.Stats // Keeps track of container stats
|
||||
validIds map[string]struct{} // Map of valid container ids, used to prune invalid containers from containerStatsMap
|
||||
goodDockerVersion bool // Whether docker version is at least 25.0.0 (one-shot works correctly)
|
||||
isWindows bool // Whether the Docker Engine API is running on Windows
|
||||
buf *bytes.Buffer // Buffer to store and read response bodies
|
||||
decoder *json.Decoder // Reusable JSON decoder that reads from buf
|
||||
apiStats *container.ApiStats // Reusable API stats object
|
||||
excludeContainers []string // Patterns to exclude containers by name
|
||||
usingPodman bool // Whether the Docker Engine API is running on Podman
|
||||
agent *Agent // Used to propagate system detail changes back to the agent
|
||||
client *http.Client // Client to query Docker API
|
||||
wg sync.WaitGroup // WaitGroup to wait for all goroutines to finish
|
||||
sem chan struct{} // Semaphore to limit concurrent container requests
|
||||
containerStatsMutex sync.RWMutex // Mutex to prevent concurrent access to containerStatsMap
|
||||
apiContainerList []*container.ApiInfo // List of containers from Docker API
|
||||
containerStatsMap map[string]*container.Stats // Keeps track of container stats
|
||||
validIds map[string]struct{} // Map of valid container ids, used to prune invalid containers from containerStatsMap
|
||||
goodDockerVersion bool // Whether docker version is at least 25.0.0 (one-shot works correctly)
|
||||
dockerVersionChecked bool // Whether a version probe has completed successfully
|
||||
isWindows bool // Whether the Docker Engine API is running on Windows
|
||||
buf *bytes.Buffer // Buffer to store and read response bodies
|
||||
excludeContainers []string // Patterns to exclude containers by name
|
||||
usingPodman bool // Whether the Docker Engine API is running on Podman
|
||||
|
||||
registryClient *http.Client // Client for registry requests; nil uses a client with a 10-second timeout
|
||||
imageUpdatesMutex sync.RWMutex // Protects imageUpdates, its entries, and imageUpdatesRunning
|
||||
imageUpdates map[string]*imageUpdateStatus // Shared update status keyed by normalized image reference
|
||||
imageUpdatesRunning bool // Whether a background image-update batch is in progress
|
||||
|
||||
// Cache-time-aware tracking for CPU stats (similar to cpu.go)
|
||||
// Maps cache time intervals to container-specific CPU usage tracking
|
||||
@@ -72,7 +83,7 @@ type dockerManager struct {
|
||||
// cacheTimeMs -> DeltaTracker for network bytes sent/received
|
||||
networkSentTrackers map[uint16]*deltatracker.DeltaTracker[string, uint64]
|
||||
networkRecvTrackers map[uint16]*deltatracker.DeltaTracker[string, uint64]
|
||||
retrySleep func(time.Duration)
|
||||
lastNetworkReadTime map[uint16]map[string]time.Time // cacheTimeMs -> containerId -> last network read time
|
||||
}
|
||||
|
||||
// userAgentRoundTripper is a custom http.RoundTripper that adds a User-Agent header to all requests
|
||||
@@ -81,6 +92,14 @@ type userAgentRoundTripper struct {
|
||||
userAgent string
|
||||
}
|
||||
|
||||
// dockerVersionResponse contains the /version fields used for engine checks.
|
||||
type dockerVersionResponse struct {
|
||||
Version string `json:"Version"`
|
||||
Components []struct {
|
||||
Name string `json:"Name"`
|
||||
} `json:"Components"`
|
||||
}
|
||||
|
||||
// RoundTrip implements the http.RoundTripper interface
|
||||
func (u *userAgentRoundTripper) RoundTrip(req *http.Request) (*http.Response, error) {
|
||||
req.Header.Set("User-Agent", u.userAgent)
|
||||
@@ -128,7 +147,14 @@ func (dm *dockerManager) getDockerStats(cacheTimeMs uint16) ([]*container.Stats,
|
||||
return nil, err
|
||||
}
|
||||
|
||||
dm.isWindows = strings.Contains(resp.Header.Get("Server"), "windows")
|
||||
// Detect Podman and Windows from Server header
|
||||
serverHeader := resp.Header.Get("Server")
|
||||
if !dm.usingPodman && detectPodmanFromHeader(serverHeader) {
|
||||
dm.setIsPodman()
|
||||
}
|
||||
dm.isWindows = strings.Contains(serverHeader, "windows")
|
||||
|
||||
dm.ensureDockerVersionChecked()
|
||||
|
||||
containersLength := len(dm.apiContainerList)
|
||||
|
||||
@@ -139,6 +165,9 @@ func (dm *dockerManager) getDockerStats(cacheTimeMs uint16) ([]*container.Stats,
|
||||
clear(dm.validIds)
|
||||
}
|
||||
|
||||
// Only schedule auxiliary work here; metrics never wait for image discovery.
|
||||
dm.refreshImageUpdates(dm.apiContainerList, time.Now())
|
||||
|
||||
var failedContainers []*container.ApiInfo
|
||||
|
||||
for _, ctr := range dm.apiContainerList {
|
||||
@@ -280,7 +309,7 @@ func (dm *dockerManager) cycleNetworkDeltasForCacheTime(cacheTimeMs uint16) {
|
||||
}
|
||||
|
||||
// calculateNetworkStats calculates network sent/receive deltas using DeltaTracker
|
||||
func (dm *dockerManager) calculateNetworkStats(ctr *container.ApiInfo, apiStats *container.ApiStats, stats *container.Stats, initialized bool, name string, cacheTimeMs uint16) (uint64, uint64) {
|
||||
func (dm *dockerManager) calculateNetworkStats(ctr *container.ApiInfo, apiStats *container.ApiStats, name string, cacheTimeMs uint16) (uint64, uint64) {
|
||||
var total_sent, total_recv uint64
|
||||
for _, v := range apiStats.Networks {
|
||||
total_sent += v.TxBytes
|
||||
@@ -299,10 +328,11 @@ func (dm *dockerManager) calculateNetworkStats(ctr *container.ApiInfo, apiStats
|
||||
sent_delta_raw := sentTracker.Delta(ctr.IdShort)
|
||||
recv_delta_raw := recvTracker.Delta(ctr.IdShort)
|
||||
|
||||
// Calculate bytes per second independently for Tx and Rx if we have previous data
|
||||
// Calculate bytes per second using per-cache-time read time to avoid
|
||||
// interference between different cache intervals (e.g. 1000ms vs 60000ms)
|
||||
var sent_delta, recv_delta uint64
|
||||
if initialized {
|
||||
millisecondsElapsed := uint64(time.Since(stats.PrevReadTime).Milliseconds())
|
||||
if prevReadTime, ok := dm.lastNetworkReadTime[cacheTimeMs][ctr.IdShort]; ok {
|
||||
millisecondsElapsed := uint64(time.Since(prevReadTime).Milliseconds())
|
||||
if millisecondsElapsed > 0 {
|
||||
if sent_delta_raw > 0 {
|
||||
sent_delta = sent_delta_raw * 1000 / millisecondsElapsed
|
||||
@@ -334,15 +364,58 @@ func validateCpuPercentage(cpuPct float64, containerName string) error {
|
||||
|
||||
// updateContainerStatsValues updates the final stats values
|
||||
func updateContainerStatsValues(stats *container.Stats, cpuPct float64, usedMemory uint64, sent_delta, recv_delta uint64, readTime time.Time) {
|
||||
stats.Cpu = twoDecimals(cpuPct)
|
||||
stats.Mem = bytesToMegabytes(float64(usedMemory))
|
||||
stats.Cpu = utils.TwoDecimals(cpuPct)
|
||||
stats.Mem = utils.BytesToMegabytes(float64(usedMemory))
|
||||
stats.Bandwidth = [2]uint64{sent_delta, recv_delta}
|
||||
// TODO(0.19+): stop populating NetworkSent/NetworkRecv (deprecated in 0.18.3)
|
||||
stats.NetworkSent = bytesToMegabytes(float64(sent_delta))
|
||||
stats.NetworkRecv = bytesToMegabytes(float64(recv_delta))
|
||||
stats.NetworkSent = utils.BytesToMegabytes(float64(sent_delta))
|
||||
stats.NetworkRecv = utils.BytesToMegabytes(float64(recv_delta))
|
||||
stats.PrevReadTime = readTime
|
||||
}
|
||||
|
||||
// convertContainerPortsToString formats the ports of a container into a sorted, deduplicated string.
|
||||
// ctr.Ports is nilled out after processing so the slice is not accidentally reused.
|
||||
func convertContainerPortsToString(ctr *container.ApiInfo) string {
|
||||
if len(ctr.Ports) == 0 {
|
||||
return ""
|
||||
}
|
||||
sort.Slice(ctr.Ports, func(i, j int) bool {
|
||||
if ctr.Ports[i].PublicPort != ctr.Ports[j].PublicPort {
|
||||
return ctr.Ports[i].PublicPort < ctr.Ports[j].PublicPort
|
||||
}
|
||||
return ctr.Ports[i].IP < ctr.Ports[j].IP
|
||||
})
|
||||
var builder strings.Builder
|
||||
seen := make(map[string]struct{})
|
||||
for _, p := range ctr.Ports {
|
||||
if p.PublicPort == 0 {
|
||||
continue
|
||||
}
|
||||
keyIP := p.IP
|
||||
if keyIP == "0.0.0.0" || keyIP == "::" {
|
||||
keyIP = ""
|
||||
}
|
||||
key := keyIP + ":" + strconv.Itoa(int(p.PublicPort))
|
||||
if _, ok := seen[key]; ok {
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
if builder.Len() > 0 {
|
||||
builder.WriteString(", ")
|
||||
}
|
||||
switch p.IP {
|
||||
case "0.0.0.0", "::":
|
||||
default:
|
||||
builder.WriteString(p.IP)
|
||||
builder.WriteByte(':')
|
||||
}
|
||||
builder.WriteString(strconv.Itoa(int(p.PublicPort)))
|
||||
}
|
||||
// clear ports slice so it doesn't get reused and blend into next response
|
||||
ctr.Ports = nil
|
||||
return builder.String()
|
||||
}
|
||||
|
||||
func parseDockerStatus(status string) (string, container.DockerHealth) {
|
||||
trimmed := strings.TrimSpace(status)
|
||||
if trimmed == "" {
|
||||
@@ -362,22 +435,60 @@ func parseDockerStatus(status string) (string, container.DockerHealth) {
|
||||
statusText = trimmed
|
||||
}
|
||||
|
||||
healthText := strings.ToLower(strings.TrimSpace(strings.TrimSuffix(trimmed[openIdx+1:], ")")))
|
||||
healthText := strings.TrimSpace(strings.TrimSuffix(trimmed[openIdx+1:], ")"))
|
||||
// Some Docker statuses include a "health:" prefix inside the parentheses.
|
||||
// Strip it so it maps correctly to the known health states.
|
||||
if colonIdx := strings.IndexRune(healthText, ':'); colonIdx != -1 {
|
||||
prefix := strings.TrimSpace(healthText[:colonIdx])
|
||||
prefix := strings.ToLower(strings.TrimSpace(healthText[:colonIdx]))
|
||||
if prefix == "health" || prefix == "health status" {
|
||||
healthText = strings.TrimSpace(healthText[colonIdx+1:])
|
||||
}
|
||||
}
|
||||
if health, ok := container.DockerHealthStrings[healthText]; ok {
|
||||
if health, ok := parseDockerHealthStatus(healthText); ok {
|
||||
return statusText, health
|
||||
}
|
||||
|
||||
return trimmed, container.DockerHealthNone
|
||||
}
|
||||
|
||||
// parseDockerHealthStatus maps Docker health status strings to container.DockerHealth values
|
||||
func parseDockerHealthStatus(status string) (container.DockerHealth, bool) {
|
||||
health, ok := container.DockerHealthStrings[strings.ToLower(strings.TrimSpace(status))]
|
||||
return health, ok
|
||||
}
|
||||
|
||||
// getPodmanContainerHealth fetches container health status from the container inspect endpoint.
|
||||
// Used for Podman which doesn't provide health status in the /containers/json endpoint as of March 2026.
|
||||
// https://github.com/containers/podman/issues/27786
|
||||
func (dm *dockerManager) getPodmanContainerHealth(containerID string) (container.DockerHealth, error) {
|
||||
resp, err := dm.client.Get(fmt.Sprintf("http://localhost/containers/%s/json", url.PathEscape(containerID)))
|
||||
if err != nil {
|
||||
return container.DockerHealthNone, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return container.DockerHealthNone, fmt.Errorf("container inspect request failed: %s", resp.Status)
|
||||
}
|
||||
|
||||
var inspectInfo struct {
|
||||
State struct {
|
||||
Health struct {
|
||||
Status string
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&inspectInfo); err != nil {
|
||||
return container.DockerHealthNone, err
|
||||
}
|
||||
|
||||
if health, ok := parseDockerHealthStatus(inspectInfo.State.Health.Status); ok {
|
||||
return health, nil
|
||||
}
|
||||
|
||||
return container.DockerHealthNone, nil
|
||||
}
|
||||
|
||||
// Updates stats for individual container with cache-time-aware delta tracking
|
||||
func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeMs uint16) error {
|
||||
name := ctr.Names[0][1:]
|
||||
@@ -387,6 +498,32 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
return err
|
||||
}
|
||||
|
||||
statusText, health := parseDockerStatus(ctr.Status)
|
||||
|
||||
// Docker exposes Health.Status on /containers/json in API 1.52+.
|
||||
// Podman currently requires falling back to the inspect endpoint as of March 2026.
|
||||
// https://github.com/containers/podman/issues/27786
|
||||
if ctr.Health.Status != "" {
|
||||
if h, ok := parseDockerHealthStatus(ctr.Health.Status); ok {
|
||||
health = h
|
||||
}
|
||||
} else if dm.usingPodman {
|
||||
if podmanHealth, err := dm.getPodmanContainerHealth(ctr.IdShort); err == nil {
|
||||
health = podmanHealth
|
||||
}
|
||||
}
|
||||
|
||||
// Read and decode the response before locking shared stats to avoid blocking
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return fmt.Errorf("container stats request failed: %s", resp.Status)
|
||||
}
|
||||
res := &container.ApiStats{}
|
||||
if err := json.NewDecoder(resp.Body).Decode(res); err != nil {
|
||||
return err
|
||||
}
|
||||
updateAvailable := dm.cachedImageUpdate(ctr.Image)
|
||||
|
||||
dm.containerStatsMutex.Lock()
|
||||
defer dm.containerStatsMutex.Unlock()
|
||||
|
||||
@@ -398,11 +535,16 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
}
|
||||
|
||||
stats.Id = ctr.IdShort
|
||||
|
||||
statusText, health := parseDockerStatus(ctr.Status)
|
||||
stats.Status = statusText
|
||||
stats.Health = health
|
||||
|
||||
stats.Image = ctr.Image
|
||||
stats.UpdateAvailable = updateAvailable
|
||||
|
||||
if len(ctr.Ports) > 0 {
|
||||
stats.Ports = convertContainerPortsToString(ctr)
|
||||
}
|
||||
|
||||
// reset current stats
|
||||
stats.Cpu = 0
|
||||
stats.Mem = 0
|
||||
@@ -411,23 +553,24 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
stats.NetworkSent = 0
|
||||
stats.NetworkRecv = 0
|
||||
|
||||
res := dm.apiStats
|
||||
res.Networks = nil
|
||||
if err := dm.decode(resp, res); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Initialize CPU tracking for this cache time interval
|
||||
dm.initializeCpuTracking(cacheTimeMs)
|
||||
|
||||
// Get previous CPU values
|
||||
prevCpuContainer, prevCpuSystem := dm.getCpuPreviousValues(cacheTimeMs, ctr.IdShort)
|
||||
|
||||
// Calculate CPU percentage based on platform
|
||||
// Calculate CPU percentage based on platform.
|
||||
// Podman reports system_cpu_usage from cgroup cpu.stat (not /proc/stat), so it reflects
|
||||
// only cgroup-tracked activity rather than total host capacity. Use a time-based method
|
||||
// instead so the result is comparable to host CPU utilization. See:
|
||||
// https://github.com/henrygd/beszel/issues/2049
|
||||
var cpuPct float64
|
||||
if dm.isWindows {
|
||||
prevRead := dm.lastCpuReadTime[cacheTimeMs][ctr.IdShort]
|
||||
cpuPct = res.CalculateCpuPercentWindows(prevCpuContainer, prevRead)
|
||||
} else if dm.usingPodman && res.CPUStats.OnlineCPUs > 0 {
|
||||
prevRead := dm.lastCpuReadTime[cacheTimeMs][ctr.IdShort]
|
||||
cpuPct = res.CalculateCpuPercentPodman(prevCpuContainer, prevRead)
|
||||
} else {
|
||||
cpuPct = res.CalculateCpuPercentLinux(prevCpuContainer, prevCpuSystem)
|
||||
}
|
||||
@@ -449,7 +592,13 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
}
|
||||
|
||||
// Calculate network stats using DeltaTracker
|
||||
sent_delta, recv_delta := dm.calculateNetworkStats(ctr, res, stats, initialized, name, cacheTimeMs)
|
||||
sent_delta, recv_delta := dm.calculateNetworkStats(ctr, res, name, cacheTimeMs)
|
||||
|
||||
// Store per-cache-time network read time for next rate calculation
|
||||
if dm.lastNetworkReadTime[cacheTimeMs] == nil {
|
||||
dm.lastNetworkReadTime[cacheTimeMs] = make(map[string]time.Time)
|
||||
}
|
||||
dm.lastNetworkReadTime[cacheTimeMs][ctr.IdShort] = time.Now()
|
||||
|
||||
// Store current network values for legacy compatibility
|
||||
var total_sent, total_recv uint64
|
||||
@@ -481,11 +630,14 @@ func (dm *dockerManager) deleteContainerStatsSync(id string) {
|
||||
for ct := range dm.lastCpuReadTime {
|
||||
delete(dm.lastCpuReadTime[ct], id)
|
||||
}
|
||||
for ct := range dm.lastNetworkReadTime {
|
||||
delete(dm.lastNetworkReadTime[ct], id)
|
||||
}
|
||||
}
|
||||
|
||||
// Creates a new http client for Docker or Podman API
|
||||
func newDockerManager() *dockerManager {
|
||||
dockerHost, exists := GetEnv("DOCKER_HOST")
|
||||
func newDockerManager(agent *Agent) *dockerManager {
|
||||
dockerHost, exists := utils.GetEnv("DOCKER_HOST")
|
||||
if exists {
|
||||
// return nil if set to empty string
|
||||
if dockerHost == "" {
|
||||
@@ -521,7 +673,7 @@ func newDockerManager() *dockerManager {
|
||||
|
||||
// configurable timeout
|
||||
timeout := time.Millisecond * time.Duration(dockerTimeoutMs)
|
||||
if t, set := GetEnv("DOCKER_TIMEOUT"); set {
|
||||
if t, set := utils.GetEnv("DOCKER_TIMEOUT"); set {
|
||||
timeout, err = time.ParseDuration(t)
|
||||
if err != nil {
|
||||
slog.Error(err.Error())
|
||||
@@ -538,7 +690,7 @@ func newDockerManager() *dockerManager {
|
||||
|
||||
// Read container exclusion patterns from environment variable
|
||||
var excludeContainers []string
|
||||
if excludeStr, set := GetEnv("EXCLUDE_CONTAINERS"); set && excludeStr != "" {
|
||||
if excludeStr, set := utils.GetEnv("EXCLUDE_CONTAINERS"); set && excludeStr != "" {
|
||||
parts := strings.SplitSeq(excludeStr, ",")
|
||||
for part := range parts {
|
||||
trimmed := strings.TrimSpace(part)
|
||||
@@ -550,6 +702,7 @@ func newDockerManager() *dockerManager {
|
||||
}
|
||||
|
||||
manager := &dockerManager{
|
||||
agent: agent,
|
||||
client: &http.Client{
|
||||
Timeout: timeout,
|
||||
Transport: userAgentTransport,
|
||||
@@ -557,7 +710,6 @@ func newDockerManager() *dockerManager {
|
||||
containerStatsMap: make(map[string]*container.Stats),
|
||||
sem: make(chan struct{}, 5),
|
||||
apiContainerList: []*container.ApiInfo{},
|
||||
apiStats: &container.ApiStats{},
|
||||
excludeContainers: excludeContainers,
|
||||
|
||||
// Initialize cache-time-aware tracking structures
|
||||
@@ -566,51 +718,55 @@ func newDockerManager() *dockerManager {
|
||||
lastCpuReadTime: make(map[uint16]map[string]time.Time),
|
||||
networkSentTrackers: make(map[uint16]*deltatracker.DeltaTracker[string, uint64]),
|
||||
networkRecvTrackers: make(map[uint16]*deltatracker.DeltaTracker[string, uint64]),
|
||||
retrySleep: time.Sleep,
|
||||
lastNetworkReadTime: make(map[uint16]map[string]time.Time),
|
||||
}
|
||||
|
||||
// If using podman, return client
|
||||
if strings.Contains(dockerHost, "podman") {
|
||||
manager.usingPodman = true
|
||||
manager.goodDockerVersion = true
|
||||
return manager
|
||||
}
|
||||
|
||||
// run version check in goroutine to avoid blocking (server may not be ready and requires retries)
|
||||
go manager.checkDockerVersion()
|
||||
|
||||
// give version check a chance to complete before returning
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
// Best-effort startup probe. If the engine is not ready yet, getDockerStats will
|
||||
// retry after the first successful /containers/json request.
|
||||
_, _ = manager.checkDockerVersion()
|
||||
|
||||
return manager
|
||||
}
|
||||
|
||||
// checkDockerVersion checks Docker version and sets goodDockerVersion if at least 25.0.0.
|
||||
// Versions before 25.0.0 have a bug with one-shot which requires all requests to be made in one batch.
|
||||
func (dm *dockerManager) checkDockerVersion() {
|
||||
var err error
|
||||
var resp *http.Response
|
||||
var versionInfo struct {
|
||||
Version string `json:"Version"`
|
||||
func (dm *dockerManager) checkDockerVersion() (bool, error) {
|
||||
resp, err := dm.client.Get("http://localhost/version")
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
const versionMaxTries = 2
|
||||
for i := 1; i <= versionMaxTries; i++ {
|
||||
resp, err = dm.client.Get("http://localhost/version")
|
||||
if err == nil && resp.StatusCode == http.StatusOK {
|
||||
break
|
||||
}
|
||||
if resp != nil {
|
||||
resp.Body.Close()
|
||||
}
|
||||
if i < versionMaxTries {
|
||||
slog.Debug("Failed to get Docker version; retrying", "attempt", i, "err", err, "response", resp)
|
||||
dm.retrySleep(5 * time.Second)
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
status := resp.Status
|
||||
resp.Body.Close()
|
||||
return false, fmt.Errorf("docker version request failed: %s", status)
|
||||
}
|
||||
if err != nil || resp.StatusCode != http.StatusOK {
|
||||
|
||||
var versionInfo dockerVersionResponse
|
||||
serverHeader := resp.Header.Get("Server")
|
||||
if err := dm.decode(resp, &versionInfo); err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
dm.applyDockerVersionInfo(serverHeader, &versionInfo)
|
||||
dm.dockerVersionChecked = true
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// ensureDockerVersionChecked retries the version probe after a successful
|
||||
// container list request.
|
||||
func (dm *dockerManager) ensureDockerVersionChecked() {
|
||||
if dm.dockerVersionChecked {
|
||||
return
|
||||
}
|
||||
if err := dm.decode(resp, &versionInfo); err != nil {
|
||||
if _, err := dm.checkDockerVersion(); err != nil {
|
||||
slog.Debug("Failed to get Docker version", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// applyDockerVersionInfo updates version-dependent behavior from engine metadata.
|
||||
func (dm *dockerManager) applyDockerVersionInfo(serverHeader string, versionInfo *dockerVersionResponse) {
|
||||
if detectPodmanEngine(serverHeader, versionInfo) {
|
||||
dm.setIsPodman()
|
||||
return
|
||||
}
|
||||
// if version > 24, one-shot works correctly and we can limit concurrent operations
|
||||
@@ -621,20 +777,18 @@ func (dm *dockerManager) checkDockerVersion() {
|
||||
}
|
||||
}
|
||||
|
||||
// Decodes Docker API JSON response using a reusable buffer and decoder. Not thread safe.
|
||||
// Decodes a Docker API JSON response using a reusable buffer. Not thread safe.
|
||||
func (dm *dockerManager) decode(resp *http.Response, d any) error {
|
||||
if dm.buf == nil {
|
||||
// initialize buffer with 256kb starting size
|
||||
dm.buf = bytes.NewBuffer(make([]byte, 0, 1024*256))
|
||||
dm.decoder = json.NewDecoder(dm.buf)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
defer dm.buf.Reset()
|
||||
_, err := dm.buf.ReadFrom(resp.Body)
|
||||
if err != nil {
|
||||
if _, err := dm.buf.ReadFrom(resp.Body); err != nil {
|
||||
return err
|
||||
}
|
||||
return dm.decoder.Decode(d)
|
||||
return json.Unmarshal(dm.buf.Bytes(), d)
|
||||
}
|
||||
|
||||
// Test docker / podman sockets and return if one exists
|
||||
@@ -649,9 +803,34 @@ func getDockerHost() string {
|
||||
return scheme + socks[0]
|
||||
}
|
||||
|
||||
func validateContainerID(containerID string) error {
|
||||
if !dockerContainerIDPattern.MatchString(containerID) {
|
||||
return fmt.Errorf("invalid container id")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func buildDockerContainerEndpoint(containerID, action string, query url.Values) (string, error) {
|
||||
if err := validateContainerID(containerID); err != nil {
|
||||
return "", err
|
||||
}
|
||||
u := &url.URL{
|
||||
Scheme: "http",
|
||||
Host: "localhost",
|
||||
Path: fmt.Sprintf("/containers/%s/%s", url.PathEscape(containerID), action),
|
||||
}
|
||||
if len(query) > 0 {
|
||||
u.RawQuery = query.Encode()
|
||||
}
|
||||
return u.String(), nil
|
||||
}
|
||||
|
||||
// getContainerInfo fetches the inspection data for a container
|
||||
func (dm *dockerManager) getContainerInfo(ctx context.Context, containerID string) ([]byte, error) {
|
||||
endpoint := fmt.Sprintf("http://localhost/containers/%s/json", containerID)
|
||||
endpoint, err := buildDockerContainerEndpoint(containerID, "json", nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -682,7 +861,15 @@ func (dm *dockerManager) getContainerInfo(ctx context.Context, containerID strin
|
||||
|
||||
// getLogs fetches the logs for a container
|
||||
func (dm *dockerManager) getLogs(ctx context.Context, containerID string) (string, error) {
|
||||
endpoint := fmt.Sprintf("http://localhost/containers/%s/logs?stdout=1&stderr=1&tail=%d", containerID, dockerLogsTail)
|
||||
query := url.Values{
|
||||
"stdout": []string{"1"},
|
||||
"stderr": []string{"1"},
|
||||
"tail": []string{fmt.Sprintf("%d", dockerLogsTail)},
|
||||
}
|
||||
endpoint, err := buildDockerContainerEndpoint(containerID, "logs", query)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil)
|
||||
if err != nil {
|
||||
return "", err
|
||||
@@ -700,8 +887,17 @@ func (dm *dockerManager) getLogs(ctx context.Context, containerID string) (strin
|
||||
}
|
||||
|
||||
var builder strings.Builder
|
||||
multiplexed := resp.Header.Get("Content-Type") == "application/vnd.docker.multiplexed-stream"
|
||||
if err := decodeDockerLogStream(resp.Body, &builder, multiplexed); err != nil {
|
||||
contentType := resp.Header.Get("Content-Type")
|
||||
multiplexed := strings.HasSuffix(contentType, "multiplexed-stream")
|
||||
logReader := io.Reader(resp.Body)
|
||||
if !multiplexed {
|
||||
// Podman may return multiplexed logs without Content-Type. Sniff the first frame header
|
||||
// with a small buffered reader only when the header check fails.
|
||||
bufferedReader := bufio.NewReaderSize(resp.Body, 8)
|
||||
multiplexed = detectDockerMultiplexedStream(bufferedReader)
|
||||
logReader = bufferedReader
|
||||
}
|
||||
if err := decodeDockerLogStream(logReader, &builder, multiplexed); err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
@@ -713,6 +909,23 @@ func (dm *dockerManager) getLogs(ctx context.Context, containerID string) (strin
|
||||
return logs, nil
|
||||
}
|
||||
|
||||
func detectDockerMultiplexedStream(reader *bufio.Reader) bool {
|
||||
const headerSize = 8
|
||||
header, err := reader.Peek(headerSize)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if header[0] != 0x01 && header[0] != 0x02 {
|
||||
return false
|
||||
}
|
||||
// Docker's stream framing header reserves bytes 1-3 as zero.
|
||||
if header[1] != 0 || header[2] != 0 || header[3] != 0 {
|
||||
return false
|
||||
}
|
||||
frameLen := binary.BigEndian.Uint32(header[4:])
|
||||
return frameLen <= maxLogFrameSize
|
||||
}
|
||||
|
||||
func decodeDockerLogStream(reader io.Reader, builder *strings.Builder, multiplexed bool) error {
|
||||
if !multiplexed {
|
||||
_, err := io.Copy(builder, io.LimitReader(reader, maxTotalLogSize))
|
||||
@@ -777,3 +990,46 @@ func (dm *dockerManager) GetHostInfo() (info container.HostInfo, err error) {
|
||||
func (dm *dockerManager) IsPodman() bool {
|
||||
return dm.usingPodman
|
||||
}
|
||||
|
||||
// setIsPodman sets the manager to Podman mode and updates system details accordingly.
|
||||
func (dm *dockerManager) setIsPodman() {
|
||||
if dm.usingPodman {
|
||||
return
|
||||
}
|
||||
dm.usingPodman = true
|
||||
dm.goodDockerVersion = true
|
||||
dm.dockerVersionChecked = true
|
||||
// keep system details updated - this may be detected late if server isn't ready when
|
||||
// agent starts, so make sure we notify the hub if this happens later.
|
||||
if dm.agent != nil {
|
||||
dm.agent.updateSystemDetails(func(details *system.Details) {
|
||||
details.Podman = true
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// detectPodmanFromHeader identifies Podman from the Docker API server header.
|
||||
func detectPodmanFromHeader(server string) bool {
|
||||
return strings.HasPrefix(server, "Libpod")
|
||||
}
|
||||
|
||||
// detectPodmanFromVersion identifies Podman from the version payload.
|
||||
func detectPodmanFromVersion(versionInfo *dockerVersionResponse) bool {
|
||||
if versionInfo == nil {
|
||||
return false
|
||||
}
|
||||
for _, component := range versionInfo.Components {
|
||||
if strings.HasPrefix(component.Name, "Podman") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// detectPodmanEngine checks both header and version metadata for Podman.
|
||||
func detectPodmanEngine(serverHeader string, versionInfo *dockerVersionResponse) bool {
|
||||
if detectPodmanFromHeader(serverHeader) {
|
||||
return true
|
||||
}
|
||||
return detectPodmanFromVersion(versionInfo)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/distribution/reference"
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
)
|
||||
|
||||
const imageUpdateInterval = time.Hour
|
||||
|
||||
type imageUpdateStatus struct {
|
||||
available bool
|
||||
checkedAt time.Time
|
||||
}
|
||||
|
||||
func normalizedImageReference(image string) string {
|
||||
named, err := reference.ParseNormalizedNamed(image)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
// Digest-pinned references cannot move to a new version.
|
||||
if _, pinned := named.(reference.Digested); pinned {
|
||||
return ""
|
||||
}
|
||||
return reference.TagNameOnly(named).String()
|
||||
}
|
||||
|
||||
// refreshImageUpdates starts at most one background batch. Neither its network
|
||||
// work nor its completion is part of the container metrics wait group.
|
||||
func (dm *dockerManager) refreshImageUpdates(containers []*container.ApiInfo, now time.Time) {
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
defer dm.imageUpdatesMutex.Unlock()
|
||||
if dm.imageUpdatesRunning {
|
||||
return
|
||||
}
|
||||
if dm.imageUpdates == nil {
|
||||
dm.imageUpdates = make(map[string]*imageUpdateStatus)
|
||||
}
|
||||
active := make(map[string]struct{}, len(containers))
|
||||
pending := make(map[string]*imageUpdateStatus)
|
||||
for _, ctr := range containers {
|
||||
if len(ctr.Names) > 0 && dm.shouldExcludeContainer(ctr.Names[0][1:]) {
|
||||
continue
|
||||
}
|
||||
key := normalizedImageReference(ctr.Image)
|
||||
if key == "" {
|
||||
continue
|
||||
}
|
||||
active[key] = struct{}{}
|
||||
entry := dm.imageUpdates[key]
|
||||
if entry == nil {
|
||||
entry = &imageUpdateStatus{}
|
||||
dm.imageUpdates[key] = entry
|
||||
}
|
||||
if entry.checkedAt.IsZero() || now.Sub(entry.checkedAt) >= imageUpdateInterval {
|
||||
pending[key] = entry
|
||||
}
|
||||
}
|
||||
for key := range dm.imageUpdates {
|
||||
if _, ok := active[key]; !ok {
|
||||
delete(dm.imageUpdates, key)
|
||||
}
|
||||
}
|
||||
if len(pending) == 0 {
|
||||
return
|
||||
}
|
||||
dm.imageUpdatesRunning = true
|
||||
go func() {
|
||||
// Limit auxiliary requests even on hosts running many different images.
|
||||
sem := make(chan struct{}, 2)
|
||||
var wg sync.WaitGroup
|
||||
for key, entry := range pending {
|
||||
sem <- struct{}{}
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
available, err := dm.checkImageUpdate(key)
|
||||
if err != nil {
|
||||
available = false
|
||||
slog.Debug("Image update check failed", "image", key, "err", err)
|
||||
}
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
entry.available = available
|
||||
entry.checkedAt = time.Now()
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdatesRunning = false
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
func (dm *dockerManager) cachedImageUpdate(image string) bool {
|
||||
key := normalizedImageReference(image)
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
defer dm.imageUpdatesMutex.RUnlock()
|
||||
entry := dm.imageUpdates[key]
|
||||
return entry != nil && entry.available
|
||||
}
|
||||
@@ -0,0 +1,225 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func waitForImageUpdates(t *testing.T, dm *dockerManager) {
|
||||
t.Helper()
|
||||
require.Eventually(t, func() bool {
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
defer dm.imageUpdatesMutex.RUnlock()
|
||||
return !dm.imageUpdatesRunning
|
||||
}, time.Second*3, time.Millisecond)
|
||||
}
|
||||
|
||||
func TestImageUpdateCacheAndStats(t *testing.T) {
|
||||
local := "sha256:" + strings.Repeat("a", 64)
|
||||
remote := "sha256:" + strings.Repeat("b", 64)
|
||||
var inspections, lookups atomic.Int32
|
||||
var fail atomic.Bool
|
||||
var upToDate atomic.Bool
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch {
|
||||
case strings.HasPrefix(r.URL.Path, "/images/"):
|
||||
inspections.Add(1)
|
||||
fmt.Fprintf(w, `{"RepoDigests":["docker.io/library/nginx@%s"]}`, local)
|
||||
case r.URL.Path == "/containers/json":
|
||||
fmt.Fprint(w, `[{"Id":"aaaaaaaaaaaa","Names":["/one"],"Image":"nginx","Status":"Up 2 hours"},{"Id":"bbbbbbbbbbbb","Names":["/two"],"Image":"docker.io/library/nginx:latest","Status":"Up 2 hours"}]`)
|
||||
case strings.Contains(r.URL.Path, "/stats"):
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576},"cpu_stats":{},"networks":{}}`)
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
dm.dockerVersionChecked = true
|
||||
dm.registryClient = &http.Client{Timeout: time.Second, Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
|
||||
if fail.Load() {
|
||||
return nil, fmt.Errorf("registry unavailable")
|
||||
}
|
||||
response := &http.Response{StatusCode: 200, Header: make(http.Header), Body: io.NopCloser(strings.NewReader(`{"token":"test"}`))}
|
||||
if r.Method == http.MethodHead {
|
||||
lookups.Add(1)
|
||||
digest := remote
|
||||
if upToDate.Load() {
|
||||
digest = local
|
||||
}
|
||||
response.Header.Set("Docker-Content-Digest", digest)
|
||||
}
|
||||
return response, nil
|
||||
})}
|
||||
stats, err := dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, stats, 2)
|
||||
waitForImageUpdates(t, dm)
|
||||
require.EqualValues(t, 1, lookups.Load())
|
||||
require.EqualValues(t, 1, inspections.Load())
|
||||
stats, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
for _, stat := range stats {
|
||||
require.True(t, stat.UpdateAvailable)
|
||||
if stat.Id == "aaaaaaaaaaaa" {
|
||||
require.Equal(t, "nginx", stat.Image)
|
||||
} else {
|
||||
require.Equal(t, "docker.io/library/nginx:latest", stat.Image)
|
||||
}
|
||||
}
|
||||
require.EqualValues(t, 1, lookups.Load())
|
||||
|
||||
expire := func() {
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdates["docker.io/library/nginx:latest"].checkedAt = time.Now().Add(-imageUpdateInterval)
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}
|
||||
upToDate.Store(true)
|
||||
expire()
|
||||
_, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
waitForImageUpdates(t, dm)
|
||||
require.EqualValues(t, 2, lookups.Load())
|
||||
require.False(t, dm.cachedImageUpdate("nginx:latest"))
|
||||
|
||||
// An expired positive result is cleared on failure, and the failure itself
|
||||
// is cached so realtime stats do not retry a broken registry every second.
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdates["docker.io/library/nginx:latest"].available = true
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
fail.Store(true)
|
||||
expire()
|
||||
_, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
waitForImageUpdates(t, dm)
|
||||
failedInspections := inspections.Load()
|
||||
stats, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, stats, 2)
|
||||
require.Equal(t, failedInspections, inspections.Load())
|
||||
for _, stat := range stats {
|
||||
require.False(t, stat.UpdateAvailable)
|
||||
require.Equal(t, 1.0, stat.Mem)
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageDiscoveryDoesNotBlockStats(t *testing.T) {
|
||||
started := make(chan struct{}, 1)
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
fmt.Fprintf(w, `{"RepoDigests":["example.com/app@sha256:%s"]}`, strings.Repeat("a", 64))
|
||||
} else {
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
defer func() { close(release); waitForImageUpdates(t, dm) }()
|
||||
dm.registryClient = &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
|
||||
started <- struct{}{}
|
||||
<-release
|
||||
return nil, fmt.Errorf("timeout")
|
||||
})}
|
||||
ctr := &container.ApiInfo{IdShort: "aaaaaaaaaaaa", Image: "example.com/app", Names: []string{"/one"}}
|
||||
dm.refreshImageUpdates([]*container.ApiInfo{ctr}, time.Now())
|
||||
select {
|
||||
case <-started:
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("check did not start")
|
||||
}
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- dm.updateContainerStats(ctr, defaultCacheTimeMs) }()
|
||||
select {
|
||||
case err := <-done:
|
||||
require.NoError(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("registry blocked stats")
|
||||
}
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
require.True(t, dm.imageUpdatesRunning)
|
||||
dm.imageUpdatesMutex.RUnlock()
|
||||
}
|
||||
|
||||
func TestNormalizeImageUpdateReferences(t *testing.T) {
|
||||
require.Equal(t, normalizedImageReference("nginx"), normalizedImageReference("docker.io/library/nginx:latest"))
|
||||
require.Empty(t, normalizedImageReference("bad reference"))
|
||||
require.Empty(t, normalizedImageReference("nginx@sha256:"+strings.Repeat("a", 64)))
|
||||
}
|
||||
|
||||
// A stats request can return headers promptly and then stall while reading its
|
||||
// body. The stats-map mutex must remain available during that read.
|
||||
func TestStatsResponseBodyDoesNotHoldStatsLock(t *testing.T) {
|
||||
started := make(chan struct{})
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
w.(http.Flusher).Flush()
|
||||
close(started)
|
||||
<-release
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- dm.updateContainerStats(&container.ApiInfo{IdShort: "aaaaaaaaaaaa", Names: []string{"/one"}, Image: "nginx"}, defaultCacheTimeMs)
|
||||
}()
|
||||
<-started
|
||||
locked := make(chan struct{})
|
||||
go func() { dm.containerStatsMutex.Lock(); dm.containerStatsMutex.Unlock(); close(locked) }()
|
||||
select {
|
||||
case <-locked:
|
||||
case <-time.After(time.Second):
|
||||
close(release)
|
||||
<-done
|
||||
t.Fatal("Docker response body held the stats mutex")
|
||||
}
|
||||
close(release)
|
||||
require.NoError(t, <-done)
|
||||
}
|
||||
|
||||
func TestImageUpdateStatsEncoding(t *testing.T) {
|
||||
original := container.Stats{Image: "nginx:latest", UpdateAvailable: true}
|
||||
encoded, err := cbor.Marshal(original)
|
||||
require.NoError(t, err)
|
||||
var fields map[int]any
|
||||
require.NoError(t, cbor.Unmarshal(encoded, &fields))
|
||||
require.Equal(t, true, fields[11])
|
||||
require.Equal(t, "nginx:latest", fields[8])
|
||||
var decoded container.Stats
|
||||
require.NoError(t, cbor.Unmarshal(encoded, &decoded))
|
||||
require.True(t, decoded.UpdateAvailable)
|
||||
require.Equal(t, original.Image, decoded.Image)
|
||||
encoded, err = json.Marshal(original)
|
||||
require.NoError(t, err)
|
||||
require.Contains(t, string(encoded), `"u":true`)
|
||||
}
|
||||
|
||||
func TestImageUpdateCacheExpiryBoundaryAndPruning(t *testing.T) {
|
||||
now := time.Now()
|
||||
key := normalizedImageReference("nginx")
|
||||
dm := &dockerManager{imageUpdates: map[string]*imageUpdateStatus{
|
||||
key: {available: true, checkedAt: now},
|
||||
"unused.example/image:latest": {checkedAt: now},
|
||||
}}
|
||||
dm.refreshImageUpdates([]*container.ApiInfo{{Image: "nginx"}}, now.Add(imageUpdateInterval-time.Nanosecond))
|
||||
require.False(t, dm.imageUpdatesRunning)
|
||||
require.Len(t, dm.imageUpdates, 1)
|
||||
require.True(t, dm.cachedImageUpdate("nginx:latest"))
|
||||
dm.refreshImageUpdates(nil, now)
|
||||
require.Empty(t, dm.imageUpdates)
|
||||
}
|
||||
@@ -0,0 +1,222 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
_ "crypto/sha256"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/distribution/reference"
|
||||
"github.com/opencontainers/go-digest"
|
||||
)
|
||||
|
||||
const imageRegistryTimeout = 10 * time.Second
|
||||
|
||||
const imageManifestAccept = "application/vnd.docker.distribution.manifest.list.v2+json, " +
|
||||
"application/vnd.docker.distribution.manifest.v2+json, " +
|
||||
"application/vnd.oci.image.manifest.v1+json, " +
|
||||
"application/vnd.oci.image.index.v1+json"
|
||||
|
||||
// checkImageUpdate compares the digest recorded by Docker for image with the
|
||||
// digest currently advertised by its registry. A digest-pinned reference is
|
||||
// immutable and therefore never has an update available.
|
||||
func (dm *dockerManager) checkImageUpdate(image string) (bool, error) {
|
||||
named, err := reference.ParseNormalizedNamed(image)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("parse image reference %q: %w", image, err)
|
||||
}
|
||||
if _, pinned := named.(reference.Digested); pinned {
|
||||
return false, nil
|
||||
}
|
||||
named = reference.TagNameOnly(named)
|
||||
|
||||
registry := reference.Domain(named)
|
||||
repository := reference.Path(named)
|
||||
tag := named.(reference.Tagged).Tag()
|
||||
|
||||
localDigest, err := dm.inspectImageDigest(image, registry, repository)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
remoteDigest, err := dm.registryImageDigest(registry, repository, tag)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
return remoteDigest != localDigest, nil
|
||||
}
|
||||
|
||||
// inspectImageDigest reads Docker's image metadata without using dm.decode.
|
||||
// The checker runs in the image-discovery goroutine, so it must not hold any
|
||||
// of the container statistics locks while waiting on the Docker API.
|
||||
func (dm *dockerManager) inspectImageDigest(image, registry, repository string) (string, error) {
|
||||
if dm.client == nil {
|
||||
return "", fmt.Errorf("inspect image %q: Docker client is unavailable", image)
|
||||
}
|
||||
|
||||
endpoint := "http://localhost/images/" + url.PathEscape(image) + "/json"
|
||||
resp, err := dm.client.Get(endpoint)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("inspect image %q: %w", image, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("inspect image %q failed: %s", image, responseStatus(resp))
|
||||
}
|
||||
|
||||
var inspect struct {
|
||||
RepoDigests []string `json:"RepoDigests"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&inspect); err != nil {
|
||||
return "", fmt.Errorf("decode image inspect %q: %w", image, err)
|
||||
}
|
||||
if len(inspect.RepoDigests) == 0 {
|
||||
return "", fmt.Errorf("inspect image %q returned no repository digests", image)
|
||||
}
|
||||
|
||||
localDigest, ok := matchingRepositoryDigest(inspect.RepoDigests, registry, repository)
|
||||
if !ok {
|
||||
return "", fmt.Errorf("inspect image %q returned no valid digest for %s/%s", image, registry, repository)
|
||||
}
|
||||
return localDigest, nil
|
||||
}
|
||||
|
||||
// matchingRepositoryDigest returns a valid digest belonging to the requested
|
||||
// repository. Docker can return multiple RepoDigests for one local image; an
|
||||
// unrelated first entry must never be used for the comparison.
|
||||
func matchingRepositoryDigest(repoDigests []string, registry, repository string) (string, bool) {
|
||||
for _, repoDigest := range repoDigests {
|
||||
repoDigest = strings.TrimSpace(repoDigest)
|
||||
at := strings.LastIndexByte(repoDigest, '@')
|
||||
if at <= 0 || at == len(repoDigest)-1 || strings.Contains(repoDigest[:at], "@") {
|
||||
continue
|
||||
}
|
||||
|
||||
repoRef, err := reference.ParseNormalizedNamed(repoDigest[:at])
|
||||
if err != nil || reference.Path(repoRef) != repository || !sameRegistry(reference.Domain(repoRef), registry) {
|
||||
continue
|
||||
}
|
||||
if _, hasTag := repoRef.(reference.Tagged); hasTag {
|
||||
continue
|
||||
}
|
||||
|
||||
d, err := digest.Parse(repoDigest[at+1:])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
return d.String(), true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
func sameRegistry(left, right string) bool {
|
||||
left = canonicalRegistry(left)
|
||||
right = canonicalRegistry(right)
|
||||
return left == right ||
|
||||
(left == "ghcr.io" && right == "lscr.io") ||
|
||||
(left == "lscr.io" && right == "ghcr.io")
|
||||
}
|
||||
|
||||
func canonicalRegistry(registry string) string {
|
||||
if registry == "index.docker.io" {
|
||||
return "docker.io"
|
||||
}
|
||||
return registry
|
||||
}
|
||||
|
||||
func (dm *dockerManager) registryImageDigest(registry, repository, tag string) (string, error) {
|
||||
client := dm.registryClient
|
||||
if client == nil {
|
||||
client = &http.Client{Timeout: imageRegistryTimeout}
|
||||
}
|
||||
|
||||
token, err := dm.registryToken(client, registry, repository)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
host := registry
|
||||
if registry == "docker.io" {
|
||||
host = "registry-1.docker.io"
|
||||
}
|
||||
manifestURL := "https://" + host + "/v2/" + repository + "/manifests/" + url.PathEscape(tag)
|
||||
req, err := http.NewRequest(http.MethodHead, manifestURL, nil)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("create manifest request: %w", err)
|
||||
}
|
||||
req.Header.Set("Accept", imageManifestAccept)
|
||||
if token != "" {
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
}
|
||||
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("fetch manifest %s:%s: %w", registry, repository, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("manifest request for %s:%s failed: %s", repository, tag, responseStatus(resp))
|
||||
}
|
||||
|
||||
remote := strings.TrimSpace(resp.Header.Get("Docker-Content-Digest"))
|
||||
d, err := digest.Parse(remote)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("manifest request for %s:%s returned invalid digest: %w", repository, tag, err)
|
||||
}
|
||||
return d.String(), nil
|
||||
}
|
||||
|
||||
func (dm *dockerManager) registryToken(client *http.Client, registry, repository string) (string, error) {
|
||||
var authURL string
|
||||
switch registry {
|
||||
case "docker.io":
|
||||
authURL = "https://auth.docker.io/token?service=registry.docker.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
|
||||
case "ghcr.io", "lscr.io":
|
||||
// lscr.io is the LinuxServer alias for its GHCR-backed images.
|
||||
authURL = "https://ghcr.io/token?service=ghcr.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
|
||||
default:
|
||||
// Anonymous registries remain supported, as they were before the
|
||||
// authenticated Docker Hub and GHCR paths were added.
|
||||
return "", nil
|
||||
}
|
||||
|
||||
req, err := http.NewRequest(http.MethodGet, authURL, nil)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("create registry auth request: %w", err)
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("fetch registry auth token for %s: %w", repository, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("registry auth request for %s failed: %s", repository, responseStatus(resp))
|
||||
}
|
||||
|
||||
var tokenResponse struct {
|
||||
Token string `json:"token"`
|
||||
AccessToken string `json:"access_token"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&tokenResponse); err != nil {
|
||||
return "", fmt.Errorf("decode registry auth response for %s: %w", repository, err)
|
||||
}
|
||||
token := strings.TrimSpace(tokenResponse.Token)
|
||||
if token == "" {
|
||||
token = strings.TrimSpace(tokenResponse.AccessToken)
|
||||
}
|
||||
if token == "" {
|
||||
return "", fmt.Errorf("registry auth response for %s contained no token", repository)
|
||||
}
|
||||
return token, nil
|
||||
}
|
||||
|
||||
func responseStatus(resp *http.Response) string {
|
||||
if resp.Status != "" {
|
||||
return resp.Status
|
||||
}
|
||||
return http.StatusText(resp.StatusCode)
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
type registryTransportFunc func(*http.Request) (*http.Response, error)
|
||||
|
||||
func (fn registryTransportFunc) RoundTrip(req *http.Request) (*http.Response, error) {
|
||||
return fn(req)
|
||||
}
|
||||
|
||||
func registryResponse(status int, body string) *http.Response {
|
||||
return &http.Response{
|
||||
StatusCode: status,
|
||||
Status: fmt.Sprintf("%d %s", status, http.StatusText(status)),
|
||||
Header: make(http.Header),
|
||||
Body: io.NopCloser(strings.NewReader(body)),
|
||||
}
|
||||
}
|
||||
|
||||
func registryDigest(fill byte) string {
|
||||
return "sha256:" + strings.Repeat(string(fill), 64)
|
||||
}
|
||||
|
||||
func newRegistryChecker(t *testing.T, inspectBody string, transport http.RoundTripper) *dockerManager {
|
||||
t.Helper()
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = io.WriteString(w, inspectBody)
|
||||
return
|
||||
}
|
||||
http.NotFound(w, r)
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
return &dockerManager{
|
||||
client: newDockerManagerForVersionTest(server).client,
|
||||
registryClient: &http.Client{Transport: transport},
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateUsesInspectAndManifestDigests(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
remote := registryDigest('b')
|
||||
var authCalls, manifestCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
switch {
|
||||
case req.Method == http.MethodGet && req.URL.Host == "auth.docker.io":
|
||||
authCalls.Add(1)
|
||||
require.Equal(t, "/token", req.URL.Path)
|
||||
return registryResponse(http.StatusOK, `{"token":"test-token"}`), nil
|
||||
case req.Method == http.MethodHead && req.URL.Host == "registry-1.docker.io":
|
||||
manifestCalls.Add(1)
|
||||
require.Equal(t, "/v2/library/alpine/manifests/latest", req.URL.Path)
|
||||
require.Equal(t, "Bearer test-token", req.Header.Get("Authorization"))
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", remote)
|
||||
return resp, nil
|
||||
default:
|
||||
return registryResponse(http.StatusNotFound, ""), nil
|
||||
}
|
||||
}))
|
||||
|
||||
available, err := dm.checkImageUpdate("alpine")
|
||||
require.NoError(t, err)
|
||||
require.True(t, available)
|
||||
require.EqualValues(t, 1, authCalls.Load())
|
||||
require.EqualValues(t, 1, manifestCalls.Load())
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateReportsUnknownInspectState(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
body string
|
||||
}{
|
||||
{name: "missing field", body: `{}`},
|
||||
{name: "empty field", body: `{"RepoDigests":[]}`},
|
||||
{name: "malformed reference", body: `{"RepoDigests":["not-a-repo-digest"]}`},
|
||||
{name: "wrong repository", body: `{"RepoDigests":["docker.io/library/busybox@` + registryDigest('a') + `"]}`},
|
||||
{name: "malformed digest", body: `{"RepoDigests":["docker.io/library/alpine@sha256:not-a-digest"]}`},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
var registryCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, test.body, registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
registryCalls.Add(1)
|
||||
return registryResponse(http.StatusOK, `{"token":"unexpected"}`), nil
|
||||
}))
|
||||
|
||||
available, err := dm.checkImageUpdate("alpine")
|
||||
require.Error(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 0, registryCalls.Load(), "invalid local state must not query a registry")
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateChecksInspectAuthAndManifestStatuses(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
validInspect := fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
inspectCode int
|
||||
authCode int
|
||||
manifestCode int
|
||||
remote string
|
||||
want string
|
||||
}{
|
||||
{name: "inspect status", inspectCode: http.StatusNotFound, want: "inspect image"},
|
||||
{name: "auth status", inspectCode: http.StatusOK, authCode: http.StatusUnauthorized, want: "registry auth"},
|
||||
{name: "manifest status", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusNotFound, remote: local, want: "manifest request"},
|
||||
{name: "missing digest", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusOK, want: "invalid digest"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if test.inspectCode != http.StatusOK && strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
w.WriteHeader(test.inspectCode)
|
||||
return
|
||||
}
|
||||
_, _ = io.WriteString(w, validInspect)
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
calls := 0
|
||||
dm := &dockerManager{client: newDockerManagerForVersionTest(server).client, registryClient: &http.Client{Transport: registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
calls++
|
||||
if req.Method == http.MethodGet {
|
||||
return registryResponse(test.authCode, `{"token":"test"}`), nil
|
||||
}
|
||||
response := registryResponse(test.manifestCode, "")
|
||||
response.Header.Set("Docker-Content-Digest", test.remote)
|
||||
return response, nil
|
||||
})}}
|
||||
|
||||
_, err := dm.checkImageUpdate("alpine")
|
||||
require.Error(t, err)
|
||||
require.Contains(t, err.Error(), test.want)
|
||||
if test.inspectCode != http.StatusOK {
|
||||
require.Zero(t, calls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateSupportsAnonymousAndLSCRRegistries(t *testing.T) {
|
||||
t.Run("anonymous registry", func(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
var calls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["example.com/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
calls.Add(1)
|
||||
require.Equal(t, http.MethodHead, req.Method)
|
||||
require.Equal(t, "example.com", req.URL.Host)
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", local)
|
||||
return resp, nil
|
||||
}))
|
||||
available, err := dm.checkImageUpdate("example.com/app")
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 1, calls.Load())
|
||||
})
|
||||
|
||||
t.Run("lscr ghcr alias", func(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
var authCalls, manifestCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["ghcr.io/linuxserver/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
if req.Method == http.MethodGet {
|
||||
authCalls.Add(1)
|
||||
return registryResponse(http.StatusOK, `{"token":"test"}`), nil
|
||||
}
|
||||
manifestCalls.Add(1)
|
||||
require.Equal(t, "lscr.io", req.URL.Host)
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", local)
|
||||
return resp, nil
|
||||
}))
|
||||
available, err := dm.checkImageUpdate("lscr.io/linuxserver/app")
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 1, authCalls.Load())
|
||||
require.EqualValues(t, 1, manifestCalls.Load())
|
||||
})
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateSkipsPinnedDigest(t *testing.T) {
|
||||
image := "docker.io/library/alpine@" + registryDigest('a')
|
||||
dm := &dockerManager{}
|
||||
available, err := dm.checkImageUpdate(image)
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
}
|
||||
+898
-128
File diff suppressed because it is too large
Load Diff
+8
-20
@@ -8,6 +8,7 @@ import (
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
)
|
||||
|
||||
@@ -141,9 +142,9 @@ func readEmmcHealth(blockName string) (emmcHealth, bool) {
|
||||
out.lifeA = lifeA
|
||||
out.lifeB = lifeB
|
||||
|
||||
out.model = readStringFile(filepath.Join(deviceDir, "name"))
|
||||
out.serial = readStringFile(filepath.Join(deviceDir, "serial"))
|
||||
out.revision = readStringFile(filepath.Join(deviceDir, "prv"))
|
||||
out.model = utils.ReadStringFile(filepath.Join(deviceDir, "name"))
|
||||
out.serial = utils.ReadStringFile(filepath.Join(deviceDir, "serial"))
|
||||
out.revision = utils.ReadStringFile(filepath.Join(deviceDir, "prv"))
|
||||
|
||||
if capBytes, ok := readBlockCapacityBytes(blockName); ok {
|
||||
out.capacity = capBytes
|
||||
@@ -153,7 +154,7 @@ func readEmmcHealth(blockName string) (emmcHealth, bool) {
|
||||
}
|
||||
|
||||
func readLifeTime(deviceDir string) (uint8, uint8, bool) {
|
||||
if content, ok := readStringFileOK(filepath.Join(deviceDir, "life_time")); ok {
|
||||
if content, ok := utils.ReadStringFileOK(filepath.Join(deviceDir, "life_time")); ok {
|
||||
a, b, ok := parseHexBytePair(content)
|
||||
return a, b, ok
|
||||
}
|
||||
@@ -170,7 +171,7 @@ func readBlockCapacityBytes(blockName string) (uint64, bool) {
|
||||
sizePath := filepath.Join(emmcSysfsRoot, "class", "block", blockName, "size")
|
||||
lbsPath := filepath.Join(emmcSysfsRoot, "class", "block", blockName, "queue", "logical_block_size")
|
||||
|
||||
sizeStr, ok := readStringFileOK(sizePath)
|
||||
sizeStr, ok := utils.ReadStringFileOK(sizePath)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
@@ -179,7 +180,7 @@ func readBlockCapacityBytes(blockName string) (uint64, bool) {
|
||||
return 0, false
|
||||
}
|
||||
|
||||
lbsStr, ok := readStringFileOK(lbsPath)
|
||||
lbsStr, ok := utils.ReadStringFileOK(lbsPath)
|
||||
logicalBlockSize := uint64(512)
|
||||
if ok {
|
||||
if parsed, err := strconv.ParseUint(lbsStr, 10, 64); err == nil && parsed > 0 {
|
||||
@@ -191,7 +192,7 @@ func readBlockCapacityBytes(blockName string) (uint64, bool) {
|
||||
}
|
||||
|
||||
func readHexByteFile(path string) (uint8, bool) {
|
||||
content, ok := readStringFileOK(path)
|
||||
content, ok := utils.ReadStringFileOK(path)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
@@ -199,19 +200,6 @@ func readHexByteFile(path string) (uint8, bool) {
|
||||
return b, ok
|
||||
}
|
||||
|
||||
func readStringFile(path string) string {
|
||||
content, _ := readStringFileOK(path)
|
||||
return content
|
||||
}
|
||||
|
||||
func readStringFileOK(path string) (string, bool) {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", false
|
||||
}
|
||||
return strings.TrimSpace(string(b)), true
|
||||
}
|
||||
|
||||
func hasEmmcHealthFiles(deviceDir string) bool {
|
||||
entries, err := os.ReadDir(deviceDir)
|
||||
if err != nil {
|
||||
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
type fanSensor struct {
|
||||
key, path string
|
||||
}
|
||||
|
||||
var getFanSensors = newFanSensorCache(hwmonRoot)
|
||||
|
||||
func newFanSensorCache(root string) func() ([]fanSensor, error) {
|
||||
return sync.OnceValues(func() ([]fanSensor, error) {
|
||||
return discoverHwmonFans(root)
|
||||
})
|
||||
}
|
||||
|
||||
// updateFans populates systemStats.Fans from the host's hwmon sysfs tree.
|
||||
// No-op on platforms where hwmon isn't available (see fans_other.go).
|
||||
func (a *Agent) updateFans(systemStats *system.Stats) {
|
||||
if hwmonRoot == "" {
|
||||
return
|
||||
}
|
||||
sensors, err := getFanSensors()
|
||||
if err != nil {
|
||||
slog.Debug("Error reading fans", "err", err)
|
||||
return
|
||||
}
|
||||
fans := readFanSensors(sensors)
|
||||
if len(fans) == 0 {
|
||||
return
|
||||
}
|
||||
systemStats.Fans = fans
|
||||
// Note: Commented out because we don't currently use this value in the UI.
|
||||
// Compute the single "dashboard" value used by the FanSpeed alert.
|
||||
// Per-sensor RPMs live in Stats.Fans and drive the multi-line FanChart
|
||||
// in the UI; the alert path only needs one number to compare against
|
||||
// the user's threshold, so we use the highest RPM across all fans
|
||||
// a.systemInfo.DashboardFan = 0
|
||||
// for _, rpm := range fans {
|
||||
// if rpm > a.systemInfo.DashboardFan {
|
||||
// a.systemInfo.DashboardFan = rpm
|
||||
// }
|
||||
// }
|
||||
}
|
||||
|
||||
// readHwmonFans walks the given hwmon root (typically /sys/class/hwmon) and
|
||||
// returns a map of "<chip>_<label-or-fan-idx>" → RPM for every fan*_input
|
||||
// file it finds. Zero RPM is retained because it can represent a real fan that
|
||||
// has stopped; negative and malformed readings are ignored.
|
||||
func readHwmonFans(root string) (map[string]uint16, error) {
|
||||
sensors, err := discoverHwmonFans(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return readFanSensors(sensors), nil
|
||||
}
|
||||
|
||||
func discoverHwmonFans(root string) ([]fanSensor, error) {
|
||||
entries, err := os.ReadDir(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var sensors []fanSensor
|
||||
for _, entry := range entries {
|
||||
chipDir := filepath.Join(root, entry.Name())
|
||||
sensorDir := chipDir
|
||||
inputs, _ := filepath.Glob(filepath.Join(sensorDir, "fan*_input"))
|
||||
|
||||
// Some legacy hwmon drivers (notably applesmc) register a hwmon class
|
||||
// device but create fan attributes on the parent platform device. In
|
||||
// sysfs that parent is exposed through hwmonN/device.
|
||||
if len(inputs) == 0 {
|
||||
deviceDir := filepath.Join(chipDir, "device")
|
||||
if deviceInputs, _ := filepath.Glob(filepath.Join(deviceDir, "fan*_input")); len(deviceInputs) > 0 {
|
||||
sensorDir = deviceDir
|
||||
inputs = deviceInputs
|
||||
}
|
||||
}
|
||||
|
||||
chipName := utils.ReadStringFile(filepath.Join(sensorDir, "name"))
|
||||
if chipName == "" {
|
||||
chipName = utils.ReadStringFile(filepath.Join(chipDir, "name"))
|
||||
}
|
||||
if chipName == "" {
|
||||
chipName = entry.Name()
|
||||
}
|
||||
for _, inputPath := range inputs {
|
||||
base := strings.TrimSuffix(filepath.Base(inputPath), "_input")
|
||||
label := utils.ReadStringFile(filepath.Join(sensorDir, base+"_label"))
|
||||
key := chipName + "_" + base
|
||||
if label != "" {
|
||||
key = chipName + "_" + label
|
||||
}
|
||||
sensors = append(sensors, fanSensor{key, inputPath})
|
||||
}
|
||||
}
|
||||
return sensors, nil
|
||||
}
|
||||
|
||||
func readFanSensors(sensors []fanSensor) map[string]uint16 {
|
||||
fans := make(map[string]uint16, len(sensors))
|
||||
for _, sensor := range sensors {
|
||||
if rpm, ok := utils.ReadUintFile(sensor.path); ok {
|
||||
fans[sensor.key] = uint16(rpm)
|
||||
}
|
||||
}
|
||||
return fans
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
// hwmonRoot is the sysfs entry point for hardware monitor chips. Each
|
||||
// subdirectory (hwmon0, hwmon1, …) is one chip; fan*_input files inside it
|
||||
// expose RPM readings.
|
||||
const hwmonRoot = "/sys/class/hwmon"
|
||||
@@ -0,0 +1,7 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
// hwmonRoot is empty on non-Linux platforms — fan RPM reporting via sysfs
|
||||
// hwmon is Linux-specific. updateFans() short-circuits when this is empty.
|
||||
const hwmonRoot = ""
|
||||
@@ -0,0 +1,105 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// writeFile creates path with parents and writes contents.
|
||||
func writeFile(t *testing.T, path, contents string) {
|
||||
t.Helper()
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(contents), 0o644))
|
||||
}
|
||||
|
||||
// TestReadHwmonFans verifies the /sys/class/hwmon walker:
|
||||
// - picks up fan*_input from every chip,
|
||||
// - keys entries by chip name + sensor label (or fan idx if no label),
|
||||
// - retains 0 RPM for stopped fans,
|
||||
// - tolerates chips with no fan files at all.
|
||||
func TestReadHwmonFans(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
|
||||
// hwmon0: Raspberry Pi 5 active cooler — one fan, no label.
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "name"), "pwmfan\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "fan1_input"), "6500\n")
|
||||
|
||||
// hwmon1: a thermal-only chip, no fan files. Must not error.
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "name"), "cpu_thermal\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "temp1_input"), "55000\n")
|
||||
|
||||
// hwmon2: two fans — one stopped (0 RPM) and one labeled "chassis".
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "name"), "nct6798\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan1_input"), "0\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan2_input"), "1200\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan2_label"), "chassis\n")
|
||||
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
|
||||
assert.Equal(t, map[string]uint16{
|
||||
"pwmfan_fan1": 6500,
|
||||
"nct6798_fan1": 0,
|
||||
"nct6798_chassis": 1200,
|
||||
}, fans)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansLegacyParent verifies legacy hwmon layouts such as applesmc,
|
||||
// where the hwmon class node exists but fan attributes live on hwmonN/device.
|
||||
func TestReadHwmonFansLegacyParent(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
deviceDir := filepath.Join(root, "devices", "applesmc.768")
|
||||
writeFile(t, filepath.Join(deviceDir, "name"), "applesmc\n")
|
||||
writeFile(t, filepath.Join(deviceDir, "fan1_input"), "1202\n")
|
||||
writeFile(t, filepath.Join(deviceDir, "fan1_label"), "Exhaust\n")
|
||||
|
||||
chipDir := filepath.Join(root, "hwmon1")
|
||||
require.NoError(t, os.MkdirAll(chipDir, 0o755))
|
||||
require.NoError(t, os.Symlink(deviceDir, filepath.Join(chipDir, "device")))
|
||||
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, map[string]uint16{"applesmc_Exhaust": 1202}, fans)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansMissingRoot returns an error rather than panicking when the
|
||||
// hwmon root doesn't exist (e.g. running on a kernel without hwmon support).
|
||||
func TestReadHwmonFansMissingRoot(t *testing.T) {
|
||||
_, err := readHwmonFans(filepath.Join(t.TempDir(), "does-not-exist"))
|
||||
assert.Error(t, err)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansEmpty returns an empty map (not nil error) when the root
|
||||
// exists but contains no chips at all.
|
||||
func TestReadHwmonFansEmpty(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, fans)
|
||||
}
|
||||
|
||||
func TestFanDiscoveryCache(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
input := filepath.Join(root, "hwmon0", "fan1_input")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "name"), "chip\n")
|
||||
writeFile(t, input, "1000\n")
|
||||
|
||||
getSensors := newFanSensorCache(root)
|
||||
sensors, err := getSensors()
|
||||
require.NoError(t, err)
|
||||
fans := readFanSensors(sensors)
|
||||
assert.Equal(t, uint16(1000), fans["chip_fan1"])
|
||||
|
||||
writeFile(t, input, "1200\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "fan1_label"), "case\n")
|
||||
sensors, err = getSensors()
|
||||
require.NoError(t, err)
|
||||
fans = readFanSensors(sensors)
|
||||
assert.Equal(t, map[string]uint16{"chip_fan1": 1200}, fans)
|
||||
}
|
||||
@@ -50,6 +50,9 @@ func generateFingerprint(hostname, cpuModel string) string {
|
||||
if info, err := cpu.Info(); err == nil && len(info) > 0 {
|
||||
cpuModel = info[0].ModelName
|
||||
}
|
||||
if cpuModel == "" {
|
||||
cpuModel = getCpuModelFromCpuinfo()
|
||||
}
|
||||
}
|
||||
fingerprint = hostname + cpuModel
|
||||
}
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
|
||||
+60
-27
@@ -15,6 +15,7 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -47,6 +48,8 @@ type GPUManager struct {
|
||||
// Per-cache-key tracking for delta calculations
|
||||
// cacheKey -> gpuId -> snapshot of last count/usage/power values
|
||||
lastSnapshots map[uint16]map[string]*gpuSnapshot
|
||||
// Per-card energy snapshots for Intel sysfs power calculation.
|
||||
intelSysfsEnergySnapshots map[string]intelSysfsEnergySnapshot
|
||||
}
|
||||
|
||||
// gpuSnapshot stores the last observed incremental values for delta tracking
|
||||
@@ -89,6 +92,7 @@ const (
|
||||
collectorSourceNVML collectorSource = "nvml"
|
||||
collectorSourceNvidiaSMI collectorSource = collectorSource(nvidiaSmiCmd)
|
||||
collectorSourceIntelGpuTop collectorSource = collectorSource(intelGpuStatsCmd)
|
||||
collectorSourceIntelSysfs collectorSource = "intel_sysfs"
|
||||
collectorSourceAmdSysfs collectorSource = "amd_sysfs"
|
||||
collectorSourceRocmSMI collectorSource = collectorSource(rocmSmiCmd)
|
||||
collectorSourceMacmon collectorSource = collectorSource(macmonCmd)
|
||||
@@ -105,6 +109,7 @@ func isValidCollectorSource(source collectorSource) bool {
|
||||
collectorSourceNVML,
|
||||
collectorSourceNvidiaSMI,
|
||||
collectorSourceIntelGpuTop,
|
||||
collectorSourceIntelSysfs,
|
||||
collectorSourceAmdSysfs,
|
||||
collectorSourceRocmSMI,
|
||||
collectorSourceMacmon,
|
||||
@@ -121,6 +126,8 @@ type gpuCapabilities struct {
|
||||
hasAmdSysfs bool
|
||||
hasTegrastats bool
|
||||
hasIntelGpuTop bool
|
||||
hasXe bool
|
||||
hasIntelSysfs bool
|
||||
hasNvtop bool
|
||||
hasMacmon bool
|
||||
hasPowermetrics bool
|
||||
@@ -291,8 +298,8 @@ func (gm *GPUManager) parseAmdData(output []byte) bool {
|
||||
}
|
||||
gpu := gm.GpuDataMap[id]
|
||||
gpu.Temperature, _ = strconv.ParseFloat(v.Temperature, 64)
|
||||
gpu.MemoryUsed = bytesToMegabytes(memoryUsage)
|
||||
gpu.MemoryTotal = bytesToMegabytes(totalMemory)
|
||||
gpu.MemoryUsed = utils.BytesToMegabytes(memoryUsage)
|
||||
gpu.MemoryTotal = utils.BytesToMegabytes(totalMemory)
|
||||
gpu.Usage += usage
|
||||
gpu.Power += power
|
||||
gpu.Count++
|
||||
@@ -354,28 +361,33 @@ func (gm *GPUManager) calculateGPUAverage(id string, gpu *system.GPUData, cacheK
|
||||
|
||||
// If no new data arrived
|
||||
if deltaCount == 0 {
|
||||
// If GPU appears suspended (instantaneous values are 0), return zero values
|
||||
// Otherwise return last known average for temporary collection gaps
|
||||
if gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
|
||||
// Only discrete GPUs report temp/memory, so treat all-zero as suspended (return zeros).
|
||||
// Engine-based (Intel) GPUs don't, so carry the last average forward across sample gaps.
|
||||
if gpu.Engines == nil && gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
|
||||
return system.GPUData{Name: gpu.Name}
|
||||
}
|
||||
return gm.lastAvgData[id] // zero value if not found
|
||||
lastAvg := gm.lastAvgData[id] // zero value if not found
|
||||
if lastAvg.Name == "" {
|
||||
lastAvg.Name = gpu.Name
|
||||
}
|
||||
return lastAvg
|
||||
}
|
||||
|
||||
// Calculate new average
|
||||
gpuAvg := *gpu
|
||||
deltaUsage, deltaPower, deltaPowerPkg := gm.calculateDeltas(gpu, lastSnapshot)
|
||||
|
||||
gpuAvg.Power = twoDecimals(deltaPower / float64(deltaCount))
|
||||
gpuAvg.Power = utils.TwoDecimals(deltaPower / float64(deltaCount))
|
||||
|
||||
gpuAvg.PowerPkg = utils.TwoDecimals(deltaPowerPkg / float64(deltaCount))
|
||||
|
||||
if gpu.Engines != nil {
|
||||
// make fresh map for averaged engine metrics to avoid mutating
|
||||
// the accumulator map stored in gm.GpuDataMap
|
||||
gpuAvg.Engines = make(map[string]float64, len(gpu.Engines))
|
||||
gpuAvg.Usage = gm.calculateIntelGPUUsage(&gpuAvg, gpu, lastSnapshot, deltaCount)
|
||||
gpuAvg.PowerPkg = twoDecimals(deltaPowerPkg / float64(deltaCount))
|
||||
} else {
|
||||
gpuAvg.Usage = twoDecimals(deltaUsage / float64(deltaCount))
|
||||
gpuAvg.Usage = utils.TwoDecimals(deltaUsage / float64(deltaCount))
|
||||
}
|
||||
|
||||
gm.lastAvgData[id] = gpuAvg
|
||||
@@ -410,17 +422,17 @@ func (gm *GPUManager) calculateIntelGPUUsage(gpuAvg, gpu *system.GPUData, lastSn
|
||||
} else {
|
||||
deltaEngine = engine
|
||||
}
|
||||
gpuAvg.Engines[name] = twoDecimals(deltaEngine / float64(deltaCount))
|
||||
gpuAvg.Engines[name] = utils.TwoDecimals(deltaEngine / float64(deltaCount))
|
||||
maxEngineUsage = max(maxEngineUsage, deltaEngine/float64(deltaCount))
|
||||
}
|
||||
return twoDecimals(maxEngineUsage)
|
||||
return utils.TwoDecimals(maxEngineUsage)
|
||||
}
|
||||
|
||||
// updateInstantaneousValues updates values that should reflect current state, not averages
|
||||
func (gm *GPUManager) updateInstantaneousValues(gpuAvg *system.GPUData, gpu *system.GPUData) {
|
||||
gpuAvg.Temperature = twoDecimals(gpu.Temperature)
|
||||
gpuAvg.MemoryUsed = twoDecimals(gpu.MemoryUsed)
|
||||
gpuAvg.MemoryTotal = twoDecimals(gpu.MemoryTotal)
|
||||
gpuAvg.Temperature = utils.TwoDecimals(gpu.Temperature)
|
||||
gpuAvg.MemoryUsed = utils.TwoDecimals(gpu.MemoryUsed)
|
||||
gpuAvg.MemoryTotal = utils.TwoDecimals(gpu.MemoryTotal)
|
||||
}
|
||||
|
||||
// storeSnapshot saves the current GPU state for this cache key
|
||||
@@ -443,6 +455,8 @@ func (gm *GPUManager) storeSnapshot(id string, gpu *system.GPUData, cacheKey uin
|
||||
func (gm *GPUManager) discoverGpuCapabilities() gpuCapabilities {
|
||||
caps := gpuCapabilities{
|
||||
hasAmdSysfs: gm.hasAmdSysfs(),
|
||||
hasXe: gm.hasXe(),
|
||||
hasIntelSysfs: gm.hasIntelSysfs(),
|
||||
}
|
||||
if _, err := exec.LookPath(nvidiaSmiCmd); err == nil {
|
||||
caps.hasNvidiaSmi = true
|
||||
@@ -460,7 +474,7 @@ func (gm *GPUManager) discoverGpuCapabilities() gpuCapabilities {
|
||||
caps.hasNvtop = true
|
||||
}
|
||||
if runtime.GOOS == "darwin" {
|
||||
if _, err := exec.LookPath(macmonCmd); err == nil {
|
||||
if _, err := utils.LookPathHomebrew(macmonCmd); err == nil {
|
||||
caps.hasMacmon = true
|
||||
}
|
||||
if _, err := exec.LookPath(powermetricsCmd); err == nil {
|
||||
@@ -471,7 +485,7 @@ func (gm *GPUManager) discoverGpuCapabilities() gpuCapabilities {
|
||||
}
|
||||
|
||||
func hasAnyGpuCollector(caps gpuCapabilities) bool {
|
||||
return caps.hasNvidiaSmi || caps.hasRocmSmi || caps.hasAmdSysfs || caps.hasTegrastats || caps.hasIntelGpuTop || caps.hasNvtop || caps.hasMacmon || caps.hasPowermetrics
|
||||
return caps.hasNvidiaSmi || caps.hasRocmSmi || caps.hasAmdSysfs || caps.hasTegrastats || caps.hasIntelGpuTop || caps.hasIntelSysfs || caps.hasNvtop || caps.hasMacmon || caps.hasPowermetrics
|
||||
}
|
||||
|
||||
func (gm *GPUManager) startIntelCollector() {
|
||||
@@ -541,7 +555,7 @@ func (gm *GPUManager) collectorDefinitions(caps gpuCapabilities) map[collectorSo
|
||||
return map[collectorSource]collectorDefinition{
|
||||
collectorSourceNVML: {
|
||||
group: collectorGroupNvidia,
|
||||
available: caps.hasNvidiaSmi,
|
||||
available: true,
|
||||
start: func(_ func()) bool {
|
||||
return gm.startNvmlCollector()
|
||||
},
|
||||
@@ -562,6 +576,13 @@ func (gm *GPUManager) collectorDefinitions(caps gpuCapabilities) map[collectorSo
|
||||
return true
|
||||
},
|
||||
},
|
||||
collectorSourceIntelSysfs: {
|
||||
group: collectorGroupIntel,
|
||||
available: caps.hasIntelSysfs,
|
||||
start: func(_ func()) bool {
|
||||
return gm.startIntelSysfsCollector()
|
||||
},
|
||||
},
|
||||
collectorSourceAmdSysfs: {
|
||||
group: collectorGroupAmd,
|
||||
available: caps.hasAmdSysfs,
|
||||
@@ -687,7 +708,7 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
priorities := make([]collectorSource, 0, 4)
|
||||
|
||||
if caps.hasNvidiaSmi && !caps.hasTegrastats {
|
||||
if nvml, _ := GetEnv("NVML"); nvml == "true" {
|
||||
if nvml, _ := utils.GetEnv("NVML"); nvml == "true" {
|
||||
priorities = append(priorities, collectorSourceNVML, collectorSourceNvidiaSMI)
|
||||
} else {
|
||||
priorities = append(priorities, collectorSourceNvidiaSMI)
|
||||
@@ -695,7 +716,7 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
}
|
||||
|
||||
if caps.hasRocmSmi {
|
||||
if val, _ := GetEnv("AMD_SYSFS"); val == "true" {
|
||||
if val, _ := utils.GetEnv("AMD_SYSFS"); val == "true" {
|
||||
priorities = append(priorities, collectorSourceAmdSysfs)
|
||||
} else {
|
||||
priorities = append(priorities, collectorSourceRocmSMI)
|
||||
@@ -704,12 +725,23 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
priorities = append(priorities, collectorSourceAmdSysfs)
|
||||
}
|
||||
|
||||
if caps.hasIntelGpuTop {
|
||||
if caps.hasIntelGpuTop && !caps.hasXe {
|
||||
priorities = append(priorities, collectorSourceIntelGpuTop)
|
||||
}
|
||||
if caps.hasIntelSysfs {
|
||||
priorities = append(priorities, collectorSourceIntelSysfs)
|
||||
}
|
||||
|
||||
// Apple collectors are currently opt-in only.
|
||||
// Apple collectors are currently opt-in only for testing.
|
||||
// Enable them with GPU_COLLECTOR=macmon or GPU_COLLECTOR=powermetrics.
|
||||
// TODO: uncomment below when Apple collectors are confirmed to be working.
|
||||
//
|
||||
// Prefer macmon on macOS (no sudo). Fall back to powermetrics if present.
|
||||
// if caps.hasMacmon {
|
||||
// priorities = append(priorities, collectorSourceMacmon)
|
||||
// } else if caps.hasPowermetrics {
|
||||
// priorities = append(priorities, collectorSourcePowermetrics)
|
||||
// }
|
||||
|
||||
// Keep nvtop as a last resort only when no vendor collector exists.
|
||||
if len(priorities) == 0 && caps.hasNvtop {
|
||||
@@ -720,14 +752,11 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
|
||||
// NewGPUManager creates and initializes a new GPUManager
|
||||
func NewGPUManager() (*GPUManager, error) {
|
||||
if skipGPU, _ := GetEnv("SKIP_GPU"); skipGPU == "true" {
|
||||
if skipGPU, _ := utils.GetEnv("SKIP_GPU"); skipGPU == "true" {
|
||||
return nil, nil
|
||||
}
|
||||
var gm GPUManager
|
||||
caps := gm.discoverGpuCapabilities()
|
||||
if !hasAnyGpuCollector(caps) {
|
||||
return nil, fmt.Errorf(noGPUFoundMsg)
|
||||
}
|
||||
gm.GpuDataMap = make(map[string]*system.GPUData)
|
||||
|
||||
// Jetson devices should always use tegrastats (ignore GPU_COLLECTOR).
|
||||
@@ -736,8 +765,8 @@ func NewGPUManager() (*GPUManager, error) {
|
||||
return &gm, nil
|
||||
}
|
||||
|
||||
// if GPU_COLLECTOR is set, start user-defined collectors.
|
||||
if collectorConfig, ok := GetEnv("GPU_COLLECTOR"); ok && strings.TrimSpace(collectorConfig) != "" {
|
||||
// Respect explicit collector selection before capability auto-detection.
|
||||
if collectorConfig, ok := utils.GetEnv("GPU_COLLECTOR"); ok && strings.TrimSpace(collectorConfig) != "" {
|
||||
priorities := parseCollectorPriority(collectorConfig)
|
||||
if gm.startCollectorsByPriority(priorities, caps) == 0 {
|
||||
return nil, fmt.Errorf("no configured GPU collectors are available")
|
||||
@@ -745,6 +774,10 @@ func NewGPUManager() (*GPUManager, error) {
|
||||
return &gm, nil
|
||||
}
|
||||
|
||||
if !hasAnyGpuCollector(caps) {
|
||||
return nil, fmt.Errorf(noGPUFoundMsg)
|
||||
}
|
||||
|
||||
// auto-detect and start collectors when GPU_COLLECTOR is unset.
|
||||
if gm.startCollectorsByPriority(gm.resolveLegacyCollectorPriority(caps), caps) == 0 {
|
||||
return nil, fmt.Errorf(noGPUFoundMsg)
|
||||
|
||||
+22
-20
@@ -13,6 +13,7 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -32,8 +33,8 @@ func (gm *GPUManager) hasAmdSysfs() bool {
|
||||
return false
|
||||
}
|
||||
for _, vendorPath := range cards {
|
||||
vendor, err := os.ReadFile(vendorPath)
|
||||
if err == nil && strings.TrimSpace(string(vendor)) == "0x1002" {
|
||||
vendor, err := utils.ReadStringFileLimited(vendorPath, 64)
|
||||
if err == nil && vendor == "0x1002" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
@@ -87,12 +88,11 @@ func (gm *GPUManager) collectAmdStats() error {
|
||||
|
||||
// isAmdGpu checks whether a DRM card path belongs to AMD vendor ID 0x1002.
|
||||
func isAmdGpu(cardPath string) bool {
|
||||
vendorPath := filepath.Join(cardPath, "device/vendor")
|
||||
vendor, err := os.ReadFile(vendorPath)
|
||||
vendor, err := utils.ReadStringFileLimited(filepath.Join(cardPath, "device/vendor"), 64)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.TrimSpace(string(vendor)) == "0x1002"
|
||||
return vendor == "0x1002"
|
||||
}
|
||||
|
||||
// updateAmdGpuData reads GPU metrics from sysfs and updates the GPU data map.
|
||||
@@ -103,10 +103,8 @@ func (gm *GPUManager) updateAmdGpuData(cardPath string) bool {
|
||||
|
||||
// Read all sysfs values first (no lock needed - these can be slow)
|
||||
usage, usageErr := readSysfsFloat(filepath.Join(devicePath, "gpu_busy_percent"))
|
||||
vramUsed, memUsedErr := readSysfsFloat(filepath.Join(devicePath, "mem_info_vram_used"))
|
||||
vramTotal, _ := readSysfsFloat(filepath.Join(devicePath, "mem_info_vram_total"))
|
||||
memUsed := vramUsed
|
||||
memTotal := vramTotal
|
||||
memUsed, memUsedErr := readSysfsFloat(filepath.Join(devicePath, "mem_info_vram_used"))
|
||||
memTotal, _ := readSysfsFloat(filepath.Join(devicePath, "mem_info_vram_total"))
|
||||
// if gtt is present, add it to the memory used and total (https://github.com/henrygd/beszel/issues/1569#issuecomment-3837640484)
|
||||
if gttUsed, err := readSysfsFloat(filepath.Join(devicePath, "mem_info_gtt_used")); err == nil && gttUsed > 0 {
|
||||
if gttTotal, err := readSysfsFloat(filepath.Join(devicePath, "mem_info_gtt_total")); err == nil {
|
||||
@@ -146,8 +144,8 @@ func (gm *GPUManager) updateAmdGpuData(cardPath string) bool {
|
||||
if usageErr == nil {
|
||||
gpu.Usage += usage
|
||||
}
|
||||
gpu.MemoryUsed = bytesToMegabytes(memUsed)
|
||||
gpu.MemoryTotal = bytesToMegabytes(memTotal)
|
||||
gpu.MemoryUsed = utils.BytesToMegabytes(memUsed)
|
||||
gpu.MemoryTotal = utils.BytesToMegabytes(memTotal)
|
||||
gpu.Temperature = temp
|
||||
gpu.Power += power
|
||||
gpu.Count++
|
||||
@@ -156,11 +154,12 @@ func (gm *GPUManager) updateAmdGpuData(cardPath string) bool {
|
||||
|
||||
// readSysfsFloat reads and parses a numeric value from a sysfs file.
|
||||
func readSysfsFloat(path string) (float64, error) {
|
||||
val, err := os.ReadFile(path)
|
||||
val, err := utils.ReadStringFileLimited(path, 64)
|
||||
if err != nil {
|
||||
slog.Debug("Failed to read sysfs value", "path", path, "error", err)
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseFloat(strings.TrimSpace(string(val)), 64)
|
||||
return strconv.ParseFloat(val, 64)
|
||||
}
|
||||
|
||||
// normalizeHexID normalizes hex IDs by trimming spaces, lowercasing, and dropping 0x.
|
||||
@@ -243,7 +242,10 @@ func getCachedAmdgpuName(deviceID, revisionID string) (name string, found bool,
|
||||
|
||||
// normalizeAmdgpuName trims standard suffixes from AMDGPU product names.
|
||||
func normalizeAmdgpuName(name string) string {
|
||||
return strings.TrimSuffix(strings.TrimSpace(name), " Graphics")
|
||||
for _, suffix := range []string{" Graphics", " Series"} {
|
||||
name = strings.TrimSuffix(name, suffix)
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// cacheAmdgpuName stores a resolved AMDGPU name in the lookup cache.
|
||||
@@ -272,16 +274,16 @@ func cacheMissingAmdgpuName(deviceID, revisionID string) {
|
||||
// Falls back to showing the raw device ID if not found in the lookup table.
|
||||
func getAmdGpuName(devicePath string) string {
|
||||
// Try product_name first (works for some enterprise GPUs)
|
||||
if prod, err := os.ReadFile(filepath.Join(devicePath, "product_name")); err == nil {
|
||||
return strings.TrimSpace(string(prod))
|
||||
if prod, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "product_name"), 128); err == nil {
|
||||
return prod
|
||||
}
|
||||
|
||||
// Read PCI device ID and look it up
|
||||
if deviceID, err := os.ReadFile(filepath.Join(devicePath, "device")); err == nil {
|
||||
id := normalizeHexID(string(deviceID))
|
||||
if deviceID, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "device"), 64); err == nil {
|
||||
id := normalizeHexID(deviceID)
|
||||
revision := ""
|
||||
if revBytes, revErr := os.ReadFile(filepath.Join(devicePath, "revision")); revErr == nil {
|
||||
revision = normalizeHexID(string(revBytes))
|
||||
if rev, revErr := utils.ReadStringFileLimited(filepath.Join(devicePath, "revision"), 64); revErr == nil {
|
||||
revision = normalizeHexID(rev)
|
||||
}
|
||||
|
||||
if name, found, done := getCachedAmdgpuName(id, revision); found {
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -128,14 +129,14 @@ func TestUpdateAmdGpuDataWithFakeSysfs(t *testing.T) {
|
||||
{
|
||||
name: "sums vram and gtt when gtt is present",
|
||||
writeGTT: true,
|
||||
wantMemoryUsed: bytesToMegabytes(1073741824 + 536870912),
|
||||
wantMemoryTotal: bytesToMegabytes(2147483648 + 4294967296),
|
||||
wantMemoryUsed: utils.BytesToMegabytes(1073741824 + 536870912),
|
||||
wantMemoryTotal: utils.BytesToMegabytes(2147483648 + 4294967296),
|
||||
},
|
||||
{
|
||||
name: "falls back to vram when gtt is missing",
|
||||
writeGTT: false,
|
||||
wantMemoryUsed: bytesToMegabytes(1073741824),
|
||||
wantMemoryTotal: bytesToMegabytes(2147483648),
|
||||
wantMemoryUsed: utils.BytesToMegabytes(1073741824),
|
||||
wantMemoryTotal: utils.BytesToMegabytes(2147483648),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
+6
-1
@@ -13,6 +13,7 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -171,7 +172,11 @@ type macmonSample struct {
|
||||
}
|
||||
|
||||
func (gm *GPUManager) collectMacmonPipe() (err error) {
|
||||
cmd := exec.Command(macmonCmd, "pipe", "-i", strconv.Itoa(macmonIntervalMs))
|
||||
macmonPath, err := utils.LookPathHomebrew(macmonCmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
cmd := exec.Command(macmonPath, "pipe", "-i", strconv.Itoa(macmonIntervalMs))
|
||||
// Avoid blocking if macmon writes to stderr.
|
||||
cmd.Stderr = io.Discard
|
||||
stdout, err := cmd.StdoutPipe()
|
||||
|
||||
+2
-1
@@ -7,6 +7,7 @@ import (
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -52,7 +53,7 @@ func (gm *GPUManager) updateIntelFromStats(sample *intelGpuStats) bool {
|
||||
func (gm *GPUManager) collectIntelStats() (err error) {
|
||||
// Build command arguments, optionally selecting a device via -d
|
||||
args := []string{"-s", intelGpuStatsInterval, "-l"}
|
||||
if dev, ok := GetEnv("INTEL_GPU_DEVICE"); ok && dev != "" {
|
||||
if dev, ok := utils.GetEnv("INTEL_GPU_DEVICE"); ok && dev != "" {
|
||||
args = append(args, "-d", dev)
|
||||
}
|
||||
cmd := exec.Command(intelGpuStatsCmd, args...)
|
||||
|
||||
@@ -0,0 +1,280 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
var (
|
||||
drmSysfsRoot = "/sys/class/drm"
|
||||
intelSysfsNow = time.Now
|
||||
)
|
||||
|
||||
type intelSysfsEnergySnapshot struct {
|
||||
microjoules uint64
|
||||
timestamp time.Time
|
||||
}
|
||||
|
||||
type intelSysfsCard struct {
|
||||
cardPath string
|
||||
hwmonDir string
|
||||
}
|
||||
|
||||
// hasIntelSysfs returns true if any Intel DRM card exposes an hwmon energy counter.
|
||||
func (gm *GPUManager) hasIntelSysfs() bool {
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
return err == nil && len(cards) > 0
|
||||
}
|
||||
|
||||
// startIntelSysfsCollector starts Intel GPU collection via sysfs.
|
||||
func (gm *GPUManager) startIntelSysfsCollector() bool {
|
||||
go func() {
|
||||
if err := gm.collectIntelSysfsStats(); err != nil {
|
||||
slog.Warn("Error collecting Intel GPU data via sysfs", "err", err)
|
||||
}
|
||||
}()
|
||||
return true
|
||||
}
|
||||
|
||||
// collectIntelSysfsStats collects Intel GPU metrics directly from DRM sysfs / hwmon.
|
||||
func (gm *GPUManager) collectIntelSysfsStats() error {
|
||||
sysfsPollInterval := 3000 * time.Millisecond
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(cards) == 0 {
|
||||
return errNoValidData
|
||||
}
|
||||
|
||||
slog.Debug("Using sysfs for Intel GPU data collection", "cards", len(cards))
|
||||
for _, card := range cards {
|
||||
slog.Debug("Intel sysfs card detected", "card", filepath.Base(card.cardPath), "hwmon", card.hwmonDir)
|
||||
}
|
||||
|
||||
failures := 0
|
||||
for {
|
||||
hasData := false
|
||||
for _, card := range cards {
|
||||
if gm.updateIntelSysfsGpuData(card.cardPath, card.hwmonDir) {
|
||||
hasData = true
|
||||
}
|
||||
}
|
||||
if !hasData {
|
||||
failures++
|
||||
if failures > maxFailureRetries {
|
||||
return errNoValidData
|
||||
}
|
||||
slog.Warn("No Intel GPU data from sysfs", "failures", failures)
|
||||
time.Sleep(retryWaitTime)
|
||||
continue
|
||||
}
|
||||
failures = 0
|
||||
time.Sleep(sysfsPollInterval)
|
||||
}
|
||||
}
|
||||
|
||||
func discoverIntelSysfsCards() ([]intelSysfsCard, error) {
|
||||
paths, err := filepath.Glob(filepath.Join(drmSysfsRoot, "card*"))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var cards []intelSysfsCard
|
||||
for _, cardPath := range paths {
|
||||
if strings.Contains(filepath.Base(cardPath), "-") || !isIntelGpu(cardPath) {
|
||||
continue
|
||||
}
|
||||
hwmonDir := findIntelEnergyHwmon(filepath.Join(cardPath, "device"))
|
||||
if hwmonDir == "" {
|
||||
continue
|
||||
}
|
||||
cards = append(cards, intelSysfsCard{cardPath: cardPath, hwmonDir: hwmonDir})
|
||||
}
|
||||
return cards, nil
|
||||
}
|
||||
|
||||
func isIntelGpu(cardPath string) bool {
|
||||
vendor, err := utils.ReadStringFileLimited(filepath.Join(cardPath, "device/vendor"), 64)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.EqualFold(strings.TrimSpace(vendor), "0x8086")
|
||||
}
|
||||
|
||||
func findIntelEnergyHwmon(devicePath string) string {
|
||||
hwmons, _ := filepath.Glob(filepath.Join(devicePath, "hwmon/hwmon*"))
|
||||
var fallback string
|
||||
for _, hwmonDir := range hwmons {
|
||||
if !sysfsFileExists(filepath.Join(hwmonDir, "energy1_input")) {
|
||||
continue
|
||||
}
|
||||
if name, err := utils.ReadStringFileLimited(filepath.Join(hwmonDir, "name"), 64); err == nil && strings.EqualFold(strings.TrimSpace(name), "xe") {
|
||||
return hwmonDir
|
||||
}
|
||||
if fallback == "" {
|
||||
fallback = hwmonDir
|
||||
}
|
||||
}
|
||||
return fallback
|
||||
}
|
||||
|
||||
func sysfsFileExists(path string) bool {
|
||||
_, err := utils.ReadStringFileLimited(path, 1)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// updateIntelSysfsGpuData reads GPU metrics from sysfs and updates the GPU data map.
|
||||
// Returns true if the required energy counter was read successfully.
|
||||
func (gm *GPUManager) updateIntelSysfsGpuData(cardPath, hwmonDir string) bool {
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
id := filepath.Base(cardPath)
|
||||
|
||||
energy, err := readSysfsUint(filepath.Join(hwmonDir, "energy1_input"))
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
now := intelSysfsNow()
|
||||
power, hasPower := gm.calculateIntelSysfsPower(id, energy, now)
|
||||
powerPkg, hasPowerPkg := gm.readIntelSysfsPowerPkg(id, hwmonDir, now)
|
||||
temp := readIntelSysfsTemperature(hwmonDir)
|
||||
usage, usageErr := readOptionalSysfsFloat(filepath.Join(devicePath, "gpu_busy_percent"))
|
||||
memUsed, memUsedErr := readFirstOptionalSysfsFloat(
|
||||
filepath.Join(devicePath, "mem_info_vram_used"),
|
||||
filepath.Join(devicePath, "mem_info_lmem_used"),
|
||||
filepath.Join(devicePath, "mem_info_local_mem_used"),
|
||||
)
|
||||
memTotal, memTotalErr := readFirstOptionalSysfsFloat(
|
||||
filepath.Join(devicePath, "mem_info_vram_total"),
|
||||
filepath.Join(devicePath, "mem_info_lmem_total"),
|
||||
filepath.Join(devicePath, "mem_info_local_mem_total"),
|
||||
)
|
||||
|
||||
gm.Lock()
|
||||
defer gm.Unlock()
|
||||
|
||||
gpu, ok := gm.GpuDataMap[id]
|
||||
if !ok {
|
||||
gpu = &system.GPUData{Name: getIntelSysfsGpuName(cardPath)}
|
||||
gm.GpuDataMap[id] = gpu
|
||||
}
|
||||
|
||||
if usageErr == nil {
|
||||
gpu.Usage += usage
|
||||
}
|
||||
if memUsedErr == nil {
|
||||
gpu.MemoryUsed = utils.BytesToMegabytes(memUsed)
|
||||
}
|
||||
if memTotalErr == nil {
|
||||
gpu.MemoryTotal = utils.BytesToMegabytes(memTotal)
|
||||
}
|
||||
if temp > 0 {
|
||||
gpu.Temperature = temp
|
||||
}
|
||||
if hasPower {
|
||||
gpu.Power += power
|
||||
slog.Debug("Computed Intel sysfs GPU power", "card", id, "watts", power)
|
||||
}
|
||||
if hasPowerPkg {
|
||||
gpu.PowerPkg += powerPkg
|
||||
}
|
||||
gpu.Count++
|
||||
return true
|
||||
}
|
||||
|
||||
func (gm *GPUManager) calculateIntelSysfsPower(cardID string, microjoules uint64, timestamp time.Time) (float64, bool) {
|
||||
if gm.intelSysfsEnergySnapshots == nil {
|
||||
gm.intelSysfsEnergySnapshots = make(map[string]intelSysfsEnergySnapshot)
|
||||
}
|
||||
|
||||
last, ok := gm.intelSysfsEnergySnapshots[cardID]
|
||||
gm.intelSysfsEnergySnapshots[cardID] = intelSysfsEnergySnapshot{microjoules: microjoules, timestamp: timestamp}
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
if microjoules < last.microjoules {
|
||||
slog.Debug("Intel sysfs energy counter reset", "card", cardID)
|
||||
return 0, false
|
||||
}
|
||||
elapsed := timestamp.Sub(last.timestamp).Seconds()
|
||||
if elapsed <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
delta := microjoules - last.microjoules
|
||||
return float64(delta) / 1_000_000.0 / elapsed, true
|
||||
}
|
||||
|
||||
func (gm *GPUManager) readIntelSysfsPowerPkg(cardID, hwmonDir string, timestamp time.Time) (float64, bool) {
|
||||
energyPaths, _ := filepath.Glob(filepath.Join(hwmonDir, "energy*_input"))
|
||||
for _, path := range energyPaths {
|
||||
if filepath.Base(path) == "energy1_input" {
|
||||
continue
|
||||
}
|
||||
energy, err := readSysfsUint(path)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
return gm.calculateIntelSysfsPower(cardID+":"+filepath.Base(path), energy, timestamp)
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
func readIntelSysfsTemperature(hwmonDir string) float64 {
|
||||
tempPaths, _ := filepath.Glob(filepath.Join(hwmonDir, "temp*_input"))
|
||||
for _, path := range tempPaths {
|
||||
temp, err := readSysfsFloat(path)
|
||||
if err == nil && temp > 0 {
|
||||
return temp / 1000.0
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func readSysfsUint(path string) (uint64, error) {
|
||||
val, err := utils.ReadStringFileLimited(path, 64)
|
||||
if err != nil {
|
||||
slog.Debug("Failed to read sysfs value", "path", path, "error", err)
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseUint(strings.TrimSpace(val), 10, 64)
|
||||
}
|
||||
|
||||
func readOptionalSysfsFloat(path string) (float64, error) {
|
||||
val, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseFloat(strings.TrimSpace(string(val)), 64)
|
||||
}
|
||||
|
||||
func readFirstOptionalSysfsFloat(paths ...string) (float64, error) {
|
||||
for _, path := range paths {
|
||||
val, err := readOptionalSysfsFloat(path)
|
||||
if err == nil {
|
||||
return val, nil
|
||||
}
|
||||
}
|
||||
return 0, fmt.Errorf("no sysfs values found")
|
||||
}
|
||||
|
||||
func getIntelSysfsGpuName(cardPath string) string {
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
if product, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "product_name"), 128); err == nil && strings.TrimSpace(product) != "" {
|
||||
return strings.TrimSpace(product)
|
||||
}
|
||||
if name, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "name"), 128); err == nil && strings.TrimSpace(name) != "" {
|
||||
return strings.TrimSpace(name)
|
||||
}
|
||||
return fmt.Sprintf("Intel GPU %s", filepath.Base(cardPath))
|
||||
}
|
||||
@@ -0,0 +1,217 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func setupIntelSysfsTest(t *testing.T) (root, cardPath, hwmonPath string) {
|
||||
t.Helper()
|
||||
root = t.TempDir()
|
||||
oldRoot := drmSysfsRoot
|
||||
drmSysfsRoot = root
|
||||
t.Cleanup(func() {
|
||||
drmSysfsRoot = oldRoot
|
||||
})
|
||||
|
||||
cardPath = filepath.Join(root, "card0")
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
hwmonPath = filepath.Join(devicePath, "hwmon", "hwmon0")
|
||||
require.NoError(t, os.MkdirAll(hwmonPath, 0o755))
|
||||
return root, cardPath, hwmonPath
|
||||
}
|
||||
|
||||
func writeIntelSysfsFile(t *testing.T, basePath, name, content string) {
|
||||
t.Helper()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(basePath, name), []byte(content), 0o644))
|
||||
}
|
||||
|
||||
func setIntelSysfsTime(t *testing.T, now time.Time) {
|
||||
t.Helper()
|
||||
oldNow := intelSysfsNow
|
||||
intelSysfsNow = func() time.Time { return now }
|
||||
t.Cleanup(func() {
|
||||
intelSysfsNow = oldNow
|
||||
})
|
||||
}
|
||||
|
||||
func TestIntelSysfsDetectsIntelCardWithEnergy(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.True(t, gm.hasIntelSysfs())
|
||||
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, cards, 1)
|
||||
assert.Equal(t, cardPath, cards[0].cardPath)
|
||||
assert.Equal(t, hwmonPath, cards[0].hwmonDir)
|
||||
}
|
||||
|
||||
func TestIntelSysfsRejectsNonIntelCard(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x1002\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.False(t, gm.hasIntelSysfs())
|
||||
}
|
||||
|
||||
func TestIntelSysfsRequiresEnergyInput(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.False(t, gm.hasIntelSysfs())
|
||||
}
|
||||
|
||||
func TestIntelSysfsFirstSampleInitializesWithoutBogusPower(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
ok := gm.updateIntelSysfsGpuData(cardPath, hwmonPath)
|
||||
require.True(t, ok)
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, "Intel GPU card0", gpu.Name)
|
||||
assert.Equal(t, 0.0, gpu.Power)
|
||||
assert.Equal(t, 1.0, gpu.Count)
|
||||
}
|
||||
|
||||
func TestIntelSysfsSecondSampleComputesWatts(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
oldNow := intelSysfsNow
|
||||
intelSysfsNow = func() time.Time { return time.Unix(100, 0) }
|
||||
t.Cleanup(func() { intelSysfsNow = oldNow })
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "6000000\n")
|
||||
intelSysfsNow = func() time.Time { return time.Unix(102, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 2.5, gpu.Power)
|
||||
assert.Equal(t, 2.0, gpu.Count)
|
||||
}
|
||||
|
||||
func TestIntelSysfsSecondEnergyCounterMapsToPowerPkg(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy2_input", "2000000\n")
|
||||
|
||||
oldNow := intelSysfsNow
|
||||
t.Cleanup(func() { intelSysfsNow = oldNow })
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
intelSysfsNow = func() time.Time { return time.Unix(100, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "2000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy2_input", "8000000\n")
|
||||
intelSysfsNow = func() time.Time { return time.Unix(102, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 0.5, gpu.Power)
|
||||
assert.Equal(t, 3.0, gpu.PowerPkg)
|
||||
}
|
||||
|
||||
func TestIntelSysfsCounterResetSkipsOneSample(t *testing.T) {
|
||||
gm := &GPUManager{}
|
||||
power, ok := gm.calculateIntelSysfsPower("card0", 5000000, time.Unix(100, 0))
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, 0.0, power)
|
||||
|
||||
power, ok = gm.calculateIntelSysfsPower("card0", 1000000, time.Unix(101, 0))
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, 0.0, power)
|
||||
|
||||
power, ok = gm.calculateIntelSysfsPower("card0", 3000000, time.Unix(103, 0))
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, 1.0, power)
|
||||
}
|
||||
|
||||
func TestIntelSysfsTempInputMapsToCelsius(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "temp1_input", "43500\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 43.5, gpu.Temperature)
|
||||
}
|
||||
|
||||
func TestIntelSysfsMissingOptionalFilesDoNotFail(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 0.0, gpu.Usage)
|
||||
assert.Equal(t, 0.0, gpu.MemoryUsed)
|
||||
assert.Equal(t, 0.0, gpu.MemoryTotal)
|
||||
assert.Equal(t, 0.0, gpu.Temperature)
|
||||
}
|
||||
|
||||
func TestIntelSysfsMapsOpportunisticMemoryAndUsage(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, devicePath, "gpu_busy_percent", "37\n")
|
||||
writeIntelSysfsFile(t, devicePath, "mem_info_lmem_used", "1073741824\n")
|
||||
writeIntelSysfsFile(t, devicePath, "mem_info_lmem_total", "2147483648\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 37.0, gpu.Usage)
|
||||
assert.Equal(t, utils.BytesToMegabytes(1073741824), gpu.MemoryUsed)
|
||||
assert.Equal(t, utils.BytesToMegabytes(2147483648), gpu.MemoryTotal)
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
type intelSysfsEnergySnapshot struct{}
|
||||
|
||||
func (gm *GPUManager) hasIntelSysfs() bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func (gm *GPUManager) startIntelSysfsCollector() bool {
|
||||
return false
|
||||
}
|
||||
+45
-3
@@ -5,10 +5,12 @@ import (
|
||||
"io"
|
||||
"log/slog"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -47,9 +49,14 @@ func (gm *GPUManager) updateNvtopSnapshots(snapshots []nvtopSnapshot) bool {
|
||||
|
||||
valid := false
|
||||
usedIDs := make(map[string]struct{}, len(snapshots))
|
||||
var xeName string
|
||||
for i, sample := range snapshots {
|
||||
// nvtop leaves device_name unset on xe devices.
|
||||
if sample.DeviceName == "" {
|
||||
continue
|
||||
if xeName == "" {
|
||||
xeName = xeGpuName()
|
||||
}
|
||||
sample.DeviceName = xeName
|
||||
}
|
||||
indexID := "n" + strconv.Itoa(i)
|
||||
id := indexID
|
||||
@@ -80,10 +87,10 @@ func (gm *GPUManager) updateNvtopSnapshots(snapshots []nvtopSnapshot) bool {
|
||||
gpu.Temperature = parseNvtopNumber(*sample.Temp)
|
||||
}
|
||||
if sample.MemUsed != nil {
|
||||
gpu.MemoryUsed = bytesToMegabytes(parseNvtopNumber(*sample.MemUsed))
|
||||
gpu.MemoryUsed = utils.BytesToMegabytes(parseNvtopNumber(*sample.MemUsed))
|
||||
}
|
||||
if sample.MemTotal != nil {
|
||||
gpu.MemoryTotal = bytesToMegabytes(parseNvtopNumber(*sample.MemTotal))
|
||||
gpu.MemoryTotal = utils.BytesToMegabytes(parseNvtopNumber(*sample.MemTotal))
|
||||
}
|
||||
if sample.GpuUtil != nil {
|
||||
gpu.Usage += parseNvtopNumber(*sample.GpuUtil)
|
||||
@@ -157,3 +164,38 @@ func (gm *GPUManager) startNvtopCollector(interval string, onFailure func()) {
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// xeDevicePath returns the sysfs device path of the first xe GPU, or "".
|
||||
func xeDevicePath() string {
|
||||
cards, err := filepath.Glob("/sys/class/drm/card*")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
for _, card := range cards {
|
||||
if strings.Contains(filepath.Base(card), "-") {
|
||||
continue
|
||||
}
|
||||
if uevent, err := utils.ReadStringFileLimited(filepath.Join(card, "device", "uevent"), 4096); err == nil && strings.Contains(uevent, "DRIVER=xe") {
|
||||
return filepath.Join(card, "device")
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func (gm *GPUManager) hasXe() bool {
|
||||
return xeDevicePath() != ""
|
||||
}
|
||||
|
||||
// xeGpuName names an xe GPU from its PCI device id; nvtop leaves device_name unset on xe.
|
||||
func xeGpuName() string {
|
||||
devicePath := xeDevicePath()
|
||||
if devicePath == "" {
|
||||
return "GPU"
|
||||
}
|
||||
id, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "device"), 64)
|
||||
if err != nil {
|
||||
return "GPU"
|
||||
}
|
||||
id = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(id, "0x")))
|
||||
return "Intel GPU (" + id + ")"
|
||||
}
|
||||
|
||||
+71
-40
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
@@ -11,6 +10,7 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
@@ -266,8 +266,8 @@ func TestParseNvtopData(t *testing.T) {
|
||||
assert.Equal(t, 48.0, g0.Temperature)
|
||||
assert.Equal(t, 5.0, g0.Usage)
|
||||
assert.Equal(t, 13.0, g0.Power)
|
||||
assert.Equal(t, bytesToMegabytes(349372416), g0.MemoryUsed)
|
||||
assert.Equal(t, bytesToMegabytes(4294967296), g0.MemoryTotal)
|
||||
assert.Equal(t, utils.BytesToMegabytes(349372416), g0.MemoryUsed)
|
||||
assert.Equal(t, utils.BytesToMegabytes(4294967296), g0.MemoryTotal)
|
||||
assert.Equal(t, 1.0, g0.Count)
|
||||
|
||||
g1, ok := gm.GpuDataMap["n1"]
|
||||
@@ -276,8 +276,8 @@ func TestParseNvtopData(t *testing.T) {
|
||||
assert.Equal(t, 48.0, g1.Temperature)
|
||||
assert.Equal(t, 12.0, g1.Usage)
|
||||
assert.Equal(t, 9.0, g1.Power)
|
||||
assert.Equal(t, bytesToMegabytes(1213784064), g1.MemoryUsed)
|
||||
assert.Equal(t, bytesToMegabytes(16929173504), g1.MemoryTotal)
|
||||
assert.Equal(t, utils.BytesToMegabytes(1213784064), g1.MemoryUsed)
|
||||
assert.Equal(t, utils.BytesToMegabytes(16929173504), g1.MemoryTotal)
|
||||
assert.Equal(t, 1.0, g1.Count)
|
||||
}
|
||||
|
||||
@@ -332,11 +332,12 @@ func TestUpdateNvtopSnapshotsKeepsDeviceAssociationWhenOrderChanges(t *testing.T
|
||||
}
|
||||
|
||||
func TestParseCollectorPriority(t *testing.T) {
|
||||
got := parseCollectorPriority(" nvml, nvidia-smi, intel_gpu_top, amd_sysfs, nvtop, rocm-smi, bad ")
|
||||
got := parseCollectorPriority(" nvml, nvidia-smi, intel_gpu_top, intel_sysfs, amd_sysfs, nvtop, rocm-smi, bad ")
|
||||
want := []collectorSource{
|
||||
collectorSourceNVML,
|
||||
collectorSourceNvidiaSMI,
|
||||
collectorSourceIntelGpuTop,
|
||||
collectorSourceIntelSysfs,
|
||||
collectorSourceAmdSysfs,
|
||||
collectorSourceNVTop,
|
||||
collectorSourceRocmSMI,
|
||||
@@ -565,6 +566,42 @@ func TestGetCurrentData(t *testing.T) {
|
||||
assert.EqualValues(t, 2, gm.GpuDataMap["0"].Count, "Count should still be 2")
|
||||
})
|
||||
|
||||
t.Run("carries Intel GPU average forward between samples", func(t *testing.T) {
|
||||
// Intel GPUs report no temp/memory, so between-sample gaps (delta 0) must
|
||||
// reuse the last average instead of returning zeros and blanking the chart.
|
||||
gm := &GPUManager{
|
||||
GpuDataMap: map[string]*system.GPUData{
|
||||
"0": {
|
||||
Name: "GPU",
|
||||
Usage: 0, // derived from engines for Intel
|
||||
Power: 200, // averages to 100 over 2 counts
|
||||
PowerPkg: 60, // averages to 30 over 2 counts
|
||||
Count: 2,
|
||||
Engines: map[string]float64{
|
||||
"Render/3D": 80, // averages to 40
|
||||
"Video": 20, // averages to 10
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
cacheKey := uint16(1000) // realtime cache key
|
||||
|
||||
// First collection - computes and stores averages
|
||||
result1 := gm.GetCurrentData(cacheKey)
|
||||
assert.InDelta(t, 100.0, result1["0"].Power, 0.01)
|
||||
assert.InDelta(t, 30.0, result1["0"].PowerPkg, 0.01)
|
||||
assert.InDelta(t, 40.0, result1["0"].Engines["Render/3D"], 0.01)
|
||||
|
||||
// Second collection with no new sample (count unchanged, temp/mem still 0).
|
||||
// Must carry the last average forward rather than blanking to zero.
|
||||
result2 := gm.GetCurrentData(cacheKey)
|
||||
assert.Equal(t, "GPU", result2["0"].Name, "Name should be preserved")
|
||||
assert.InDelta(t, 100.0, result2["0"].Power, 0.01, "Should reuse last average power, not 0")
|
||||
assert.InDelta(t, 30.0, result2["0"].PowerPkg, 0.01, "Should reuse last average package power, not 0")
|
||||
assert.InDelta(t, 40.0, result2["0"].Engines["Render/3D"], 0.01, "Should reuse last average engine usage")
|
||||
})
|
||||
|
||||
t.Run("tracks separate averages per cache key", func(t *testing.T) {
|
||||
gm := &GPUManager{
|
||||
GpuDataMap: map[string]*system.GPUData{
|
||||
@@ -1083,8 +1120,6 @@ func TestCalculateGPUAverage(t *testing.T) {
|
||||
|
||||
func TestGPUCapabilitiesAndLegacyPriority(t *testing.T) {
|
||||
// Save original PATH
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
hasAmdSysfs := (&GPUManager{}).hasAmdSysfs()
|
||||
|
||||
tests := []struct {
|
||||
@@ -1178,7 +1213,7 @@ echo "[]"`
|
||||
{
|
||||
name: "no gpu tools available",
|
||||
setupCommands: func(_ string) error {
|
||||
os.Setenv("PATH", "")
|
||||
t.Setenv("PATH", "")
|
||||
return nil
|
||||
},
|
||||
wantErr: true,
|
||||
@@ -1188,7 +1223,7 @@ echo "[]"`
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
os.Setenv("PATH", tempDir)
|
||||
t.Setenv("PATH", tempDir)
|
||||
if err := tt.setupCommands(tempDir); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -1234,13 +1269,9 @@ echo "[]"`
|
||||
}
|
||||
|
||||
func TestCollectorStartHelpers(t *testing.T) {
|
||||
// Save original PATH
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
// Set up temp dir with the commands
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -1370,11 +1401,8 @@ echo '[{"device_name":"NVIDIA Test GPU","temp":"52C","power_draw":"31W","gpu_uti
|
||||
}
|
||||
|
||||
func TestNewGPUManagerPriorityNvtopFallback(t *testing.T) {
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "nvtop,nvidia-smi")
|
||||
|
||||
nvtopPath := filepath.Join(dir, "nvtop")
|
||||
@@ -1399,11 +1427,8 @@ echo "0, NVIDIA Priority GPU, 45, 512, 2048, 12, 25"`
|
||||
}
|
||||
|
||||
func TestNewGPUManagerPriorityMixedCollectors(t *testing.T) {
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "intel_gpu_top,rocm-smi")
|
||||
|
||||
intelPath := filepath.Join(dir, "intel_gpu_top")
|
||||
@@ -1433,11 +1458,8 @@ echo '{"card0": {"Temperature (Sensor edge) (C)": "49.0", "Current Socket Graphi
|
||||
}
|
||||
|
||||
func TestNewGPUManagerPriorityNvmlFallbackToNvidiaSmi(t *testing.T) {
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "nvml,nvidia-smi")
|
||||
|
||||
nvidiaPath := filepath.Join(dir, "nvidia-smi")
|
||||
@@ -1456,11 +1478,8 @@ echo "0, NVIDIA Fallback GPU, 41, 256, 1024, 8, 14"`
|
||||
}
|
||||
|
||||
func TestNewGPUManagerConfiguredCollectorsMustStart(t *testing.T) {
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
|
||||
t.Run("configured valid collector unavailable", func(t *testing.T) {
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "nvidia-smi")
|
||||
@@ -1479,12 +1498,28 @@ func TestNewGPUManagerConfiguredCollectorsMustStart(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
func TestNewGPUManagerJetsonIgnoresCollectorConfig(t *testing.T) {
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
func TestCollectorDefinitionsNvmlDoesNotRequireNvidiaSmi(t *testing.T) {
|
||||
gm := &GPUManager{}
|
||||
definitions := gm.collectorDefinitions(gpuCapabilities{})
|
||||
require.Contains(t, definitions, collectorSourceNVML)
|
||||
assert.True(t, definitions[collectorSourceNVML].available)
|
||||
}
|
||||
|
||||
func TestNewGPUManagerConfiguredNvmlBypassesCapabilityGate(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "nvml")
|
||||
|
||||
gm, err := NewGPUManager()
|
||||
require.Nil(t, gm)
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "no configured GPU collectors are available")
|
||||
assert.NotContains(t, err.Error(), noGPUFoundMsg)
|
||||
}
|
||||
|
||||
func TestNewGPUManagerJetsonIgnoresCollectorConfig(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("PATH", dir)
|
||||
t.Setenv("BESZEL_AGENT_GPU_COLLECTOR", "nvidia-smi")
|
||||
|
||||
tegraPath := filepath.Join(dir, "tegrastats")
|
||||
@@ -1719,12 +1754,8 @@ func TestIntelUpdateFromStats(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestIntelCollectorStreaming(t *testing.T) {
|
||||
// Save and override PATH
|
||||
origPath := os.Getenv("PATH")
|
||||
defer os.Setenv("PATH", origPath)
|
||||
|
||||
dir := t.TempDir()
|
||||
os.Setenv("PATH", dir)
|
||||
t.Setenv("PATH", dir)
|
||||
|
||||
// Create a fake intel_gpu_top that prints -l format with four samples (first will be skipped) and exits
|
||||
scriptPath := filepath.Join(dir, "intel_gpu_top")
|
||||
|
||||
+25
-5
@@ -51,6 +51,7 @@ func NewHandlerRegistry() *HandlerRegistry {
|
||||
registry.Register(common.GetContainerInfo, &GetContainerInfoHandler{})
|
||||
registry.Register(common.GetSmartData, &GetSmartDataHandler{})
|
||||
registry.Register(common.GetSystemdInfo, &GetSystemdInfoHandler{})
|
||||
registry.Register(common.GetZfsData, &GetZfsDataHandler{})
|
||||
|
||||
return registry
|
||||
}
|
||||
@@ -166,14 +167,33 @@ type GetSmartDataHandler struct{}
|
||||
|
||||
func (h *GetSmartDataHandler) Handle(hctx *HandlerContext) error {
|
||||
if hctx.Agent.smartManager == nil {
|
||||
// return empty map to indicate no data
|
||||
return hctx.SendResponse(map[string]smart.SmartData{}, hctx.RequestID)
|
||||
return hctx.SendResponse(smart.SmartDataResponse{Data: map[string]smart.SmartData{}}, hctx.RequestID)
|
||||
}
|
||||
if err := hctx.Agent.smartManager.Refresh(false); err != nil {
|
||||
complete, err := hctx.Agent.smartManager.Refresh(false)
|
||||
if err != nil {
|
||||
slog.Debug("smart refresh failed", "err", err)
|
||||
}
|
||||
data := hctx.Agent.smartManager.GetCurrentData()
|
||||
return hctx.SendResponse(data, hctx.RequestID)
|
||||
return hctx.SendResponse(smart.SmartDataResponse{
|
||||
Data: hctx.Agent.smartManager.GetCurrentData(),
|
||||
Complete: complete,
|
||||
}, hctx.RequestID)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// GetZfsDataHandler handles ZFS detail data requests
|
||||
type GetZfsDataHandler struct{}
|
||||
|
||||
func (h *GetZfsDataHandler) Handle(hctx *HandlerContext) error {
|
||||
if hctx.Agent.storagePoolManager == nil {
|
||||
return hctx.SendResponse(nil, hctx.RequestID)
|
||||
}
|
||||
var req common.ZfsDataRequest
|
||||
if err := cbor.Unmarshal(hctx.Request.Data, &req); err != nil {
|
||||
return err
|
||||
}
|
||||
return hctx.SendResponse(hctx.Agent.storagePoolManager.GetDetail(req.Force), hctx.RequestID)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
+41
-1
@@ -1,13 +1,15 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
@@ -18,6 +20,44 @@ type MockHandler struct {
|
||||
handleFunc func(ctx *HandlerContext) error
|
||||
}
|
||||
|
||||
func TestNewAgentResponseSmartData(t *testing.T) {
|
||||
response := newAgentResponse(smart.SmartDataResponse{
|
||||
Data: map[string]smart.SmartData{
|
||||
"AAA": {SerialNumber: "AAA"},
|
||||
},
|
||||
Complete: true,
|
||||
}, nil)
|
||||
|
||||
assert.Equal(t, "AAA", response.SmartData["AAA"].SerialNumber)
|
||||
assert.True(t, response.SmartComplete)
|
||||
}
|
||||
|
||||
func TestGetZfsDataHandlerForceRefresh(t *testing.T) {
|
||||
poolCalls := 0
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
poolCalls++
|
||||
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.GetDetail(false)
|
||||
|
||||
requestData, err := cbor.Marshal(common.ZfsDataRequest{Force: true})
|
||||
assert.NoError(t, err)
|
||||
ctx := &HandlerContext{
|
||||
Agent: &Agent{storagePoolManager: zm},
|
||||
Request: &common.HubRequest[cbor.RawMessage]{
|
||||
Action: common.GetZfsData,
|
||||
Data: requestData,
|
||||
},
|
||||
SendResponse: func(any, *uint32) error { return nil },
|
||||
}
|
||||
|
||||
assert.NoError(t, (&GetZfsDataHandler{}).Handle(ctx))
|
||||
assert.Equal(t, 2, poolCalls)
|
||||
}
|
||||
|
||||
func (m *MockHandler) Handle(ctx *HandlerContext) error {
|
||||
if m.handleFunc != nil {
|
||||
return m.handleFunc(ctx)
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package health
|
||||
|
||||
@@ -37,7 +36,6 @@ func TestHealth(t *testing.T) {
|
||||
})
|
||||
|
||||
// This test uses synctest to simulate time passing.
|
||||
// NOTE: This test requires GOEXPERIMENT=synctest to run.
|
||||
t.Run("check with simulated time", func(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
// Update the file to set the initial timestamp.
|
||||
|
||||
@@ -8,6 +8,6 @@
|
||||
</PropertyGroup>
|
||||
|
||||
<ItemGroup>
|
||||
<PackageReference Include="LibreHardwareMonitorLib" Version="0.9.5" />
|
||||
<PackageReference Include="LibreHardwareMonitorLib" Version="0.9.6" />
|
||||
</ItemGroup>
|
||||
</Project>
|
||||
|
||||
@@ -0,0 +1,293 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
)
|
||||
|
||||
// mdraidSysfsRoot is a test hook; production value is "/sys".
|
||||
var mdraidSysfsRoot = "/sys"
|
||||
|
||||
type mdraidHealth struct {
|
||||
level string
|
||||
arrayState string
|
||||
degraded uint64
|
||||
faultyDisks uint64
|
||||
populatedDisks uint64
|
||||
raidDisks uint64
|
||||
syncAction string
|
||||
syncCompleted string
|
||||
syncSpeed string
|
||||
mismatchCnt uint64
|
||||
capacity uint64
|
||||
}
|
||||
|
||||
// scanMdraidDevices discovers Linux md arrays exposed in sysfs.
|
||||
func scanMdraidDevices() []*DeviceInfo {
|
||||
blockDir := filepath.Join(mdraidSysfsRoot, "block")
|
||||
entries, err := os.ReadDir(blockDir)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
devices := make([]*DeviceInfo, 0, 2)
|
||||
for _, ent := range entries {
|
||||
name := ent.Name()
|
||||
if !isMdraidBlockName(name) {
|
||||
continue
|
||||
}
|
||||
mdDir := filepath.Join(blockDir, name, "md")
|
||||
if !utils.FileExists(filepath.Join(mdDir, "array_state")) {
|
||||
continue
|
||||
}
|
||||
|
||||
devPath := filepath.Join("/dev", name)
|
||||
devices = append(devices, &DeviceInfo{
|
||||
Name: devPath,
|
||||
Type: "mdraid",
|
||||
InfoName: devPath + " [mdraid]",
|
||||
Protocol: "MD",
|
||||
})
|
||||
}
|
||||
|
||||
return devices
|
||||
}
|
||||
|
||||
// collectMdraidHealth reads mdraid health and stores it in SmartDataMap.
|
||||
func (sm *SmartManager) collectMdraidHealth(deviceInfo *DeviceInfo) (bool, error) {
|
||||
if deviceInfo == nil || deviceInfo.Name == "" {
|
||||
return false, nil
|
||||
}
|
||||
|
||||
base := filepath.Base(deviceInfo.Name)
|
||||
if !isMdraidBlockName(base) && !strings.EqualFold(deviceInfo.Type, "mdraid") {
|
||||
return false, nil
|
||||
}
|
||||
|
||||
health, ok := readMdraidHealth(base)
|
||||
if !ok {
|
||||
return false, nil
|
||||
}
|
||||
|
||||
deviceInfo.Type = "mdraid"
|
||||
key := fmt.Sprintf("mdraid:%s", base)
|
||||
status := mdraidSmartStatus(health)
|
||||
|
||||
attrs := make([]*smart.SmartAttribute, 0, 10)
|
||||
if health.arrayState != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "ArrayState", RawString: health.arrayState})
|
||||
}
|
||||
if health.level != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "RaidLevel", RawString: health.level})
|
||||
}
|
||||
if health.raidDisks > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "RaidDisks", RawValue: health.raidDisks})
|
||||
}
|
||||
if health.degraded > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "Degraded", RawValue: health.degraded})
|
||||
}
|
||||
if health.faultyDisks > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "FaultyDisks", RawValue: health.faultyDisks})
|
||||
}
|
||||
if health.syncAction != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "SyncAction", RawString: health.syncAction})
|
||||
}
|
||||
if health.syncCompleted != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "SyncCompleted", RawString: health.syncCompleted})
|
||||
}
|
||||
if health.syncSpeed != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "SyncSpeed", RawString: health.syncSpeed})
|
||||
}
|
||||
if health.mismatchCnt > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "MismatchCount", RawValue: health.mismatchCnt})
|
||||
}
|
||||
|
||||
sm.Lock()
|
||||
defer sm.Unlock()
|
||||
|
||||
if _, exists := sm.SmartDataMap[key]; !exists {
|
||||
sm.SmartDataMap[key] = &smart.SmartData{}
|
||||
}
|
||||
|
||||
data := sm.SmartDataMap[key]
|
||||
data.ModelName = "Linux MD RAID"
|
||||
if health.level != "" {
|
||||
data.ModelName = "Linux MD RAID (" + health.level + ")"
|
||||
}
|
||||
data.Capacity = health.capacity
|
||||
data.SmartStatus = status
|
||||
data.DiskName = filepath.Join("/dev", base)
|
||||
data.DiskType = "mdraid"
|
||||
data.Attributes = attrs
|
||||
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// readMdraidHealth reads md array health fields from sysfs.
|
||||
func readMdraidHealth(blockName string) (mdraidHealth, bool) {
|
||||
var out mdraidHealth
|
||||
|
||||
if !isMdraidBlockName(blockName) {
|
||||
return out, false
|
||||
}
|
||||
|
||||
mdDir := filepath.Join(mdraidSysfsRoot, "block", blockName, "md")
|
||||
arrayState, okState := utils.ReadStringFileOK(filepath.Join(mdDir, "array_state"))
|
||||
if !okState {
|
||||
return out, false
|
||||
}
|
||||
|
||||
out.arrayState = arrayState
|
||||
out.level = utils.ReadStringFile(filepath.Join(mdDir, "level"))
|
||||
out.syncAction = utils.ReadStringFile(filepath.Join(mdDir, "sync_action"))
|
||||
out.syncCompleted = utils.ReadStringFile(filepath.Join(mdDir, "sync_completed"))
|
||||
out.syncSpeed = utils.ReadStringFile(filepath.Join(mdDir, "sync_speed"))
|
||||
|
||||
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "raid_disks")); ok {
|
||||
out.raidDisks = val
|
||||
}
|
||||
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "degraded")); ok {
|
||||
out.degraded = val
|
||||
}
|
||||
out.faultyDisks, out.populatedDisks = countMdraidMemberStates(blockName, mdraidSysfsRoot)
|
||||
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "mismatch_cnt")); ok {
|
||||
out.mismatchCnt = val
|
||||
}
|
||||
|
||||
if capBytes, ok := readMdraidBlockCapacityBytes(blockName, mdraidSysfsRoot); ok {
|
||||
out.capacity = capBytes
|
||||
}
|
||||
|
||||
return out, true
|
||||
}
|
||||
|
||||
// mdraidSmartStatus maps md state/sync signals to a SMART-like status.
|
||||
func mdraidSmartStatus(health mdraidHealth) string {
|
||||
state := strings.ToLower(strings.TrimSpace(health.arrayState))
|
||||
switch state {
|
||||
case "inactive", "faulty", "broken", "stopped":
|
||||
return "FAILED"
|
||||
}
|
||||
// During rebuild/recovery, arrays are often temporarily degraded; report as
|
||||
// warning instead of hard failure while synchronization is in progress.
|
||||
syncAction := strings.ToLower(strings.TrimSpace(health.syncAction))
|
||||
switch syncAction {
|
||||
case "resync", "recover", "reshape":
|
||||
return "WARNING"
|
||||
}
|
||||
// Use actual faulty member count rather than the degraded counter, which
|
||||
// equals raid_disks minus active_disks. On QNAP systems raid_disks may be
|
||||
// set to a large value (e.g. 32) while only a few slots are ever used,
|
||||
// making degraded misleadingly large despite zero failed disks.
|
||||
if health.faultyDisks > 0 {
|
||||
return "FAILED"
|
||||
}
|
||||
if health.degraded > 0 {
|
||||
if isSparseSlotDegraded(health) {
|
||||
// A sysfs snapshot cannot distinguish reserved slots from a removed
|
||||
// member on sparse arrays, so report the ambiguity as a warning.
|
||||
return "WARNING"
|
||||
}
|
||||
return "FAILED"
|
||||
}
|
||||
if health.mismatchCnt > 0 {
|
||||
return "WARNING"
|
||||
}
|
||||
// "check" scans for consistency problems without repairing mismatches.
|
||||
// With no mismatches, keep it green while reporting progress attributes.
|
||||
switch syncAction {
|
||||
case "repair":
|
||||
return "WARNING"
|
||||
}
|
||||
switch state {
|
||||
case "clean", "active", "active-idle", "write-pending", "read-auto", "readonly":
|
||||
return "PASSED"
|
||||
}
|
||||
return "UNKNOWN"
|
||||
}
|
||||
|
||||
// countMdraidMemberStates reads member device directories under
|
||||
// block/<name>/md and returns how many are explicitly marked "faulty", plus
|
||||
// how many are populated at all (regardless of state). populatedDisks lets
|
||||
// callers distinguish RAID slots that were never used (QNAP reserves far
|
||||
// more raid_disks than it ever populates) from members that went missing.
|
||||
func countMdraidMemberStates(blockName, root string) (faultyDisks, populatedDisks uint64) {
|
||||
devDir := filepath.Join(root, "block", blockName, "md")
|
||||
entries, err := os.ReadDir(devDir)
|
||||
if err != nil {
|
||||
return 0, 0
|
||||
}
|
||||
for _, ent := range entries {
|
||||
if !strings.HasPrefix(ent.Name(), "dev-") {
|
||||
continue
|
||||
}
|
||||
populatedDisks++
|
||||
statePath := filepath.Join(devDir, ent.Name(), "state")
|
||||
state := utils.ReadStringFile(statePath)
|
||||
if strings.Contains(state, "faulty") {
|
||||
faultyDisks++
|
||||
}
|
||||
}
|
||||
return faultyDisks, populatedDisks
|
||||
}
|
||||
|
||||
// isSparseSlotDegraded reports whether a non-zero "degraded" count may be
|
||||
// explained by RAID slots that were never populated. QNAP configures system
|
||||
// arrays with raid_disks set to a large fixed maximum (e.g. 32) far beyond the
|
||||
// handful of slots it ever populates, so sparse slots outnumber populated ones.
|
||||
func isSparseSlotDegraded(health mdraidHealth) bool {
|
||||
if health.populatedDisks == 0 || health.raidDisks <= health.populatedDisks {
|
||||
return false
|
||||
}
|
||||
sparseSlots := health.raidDisks - health.populatedDisks
|
||||
return sparseSlots > health.populatedDisks
|
||||
}
|
||||
|
||||
// isMdraidBlockName matches /dev/mdN-style block device names.
|
||||
func isMdraidBlockName(name string) bool {
|
||||
if !strings.HasPrefix(name, "md") {
|
||||
return false
|
||||
}
|
||||
suffix := strings.TrimPrefix(name, "md")
|
||||
if suffix == "" {
|
||||
return false
|
||||
}
|
||||
for _, c := range suffix {
|
||||
if c < '0' || c > '9' {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// readMdraidBlockCapacityBytes converts block size metadata into bytes.
|
||||
func readMdraidBlockCapacityBytes(blockName, root string) (uint64, bool) {
|
||||
sizePath := filepath.Join(root, "block", blockName, "size")
|
||||
lbsPath := filepath.Join(root, "block", blockName, "queue", "logical_block_size")
|
||||
|
||||
sizeStr, ok := utils.ReadStringFileOK(sizePath)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
sectors, err := strconv.ParseUint(sizeStr, 10, 64)
|
||||
if err != nil || sectors == 0 {
|
||||
return 0, false
|
||||
}
|
||||
|
||||
logicalBlockSize := uint64(512)
|
||||
if lbsStr, ok := utils.ReadStringFileOK(lbsPath); ok {
|
||||
if parsed, err := strconv.ParseUint(lbsStr, 10, 64); err == nil && parsed > 0 {
|
||||
logicalBlockSize = parsed
|
||||
}
|
||||
}
|
||||
|
||||
return sectors * logicalBlockSize, true
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
)
|
||||
|
||||
func TestMdraidMockSysfsScanAndCollect(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
prev := mdraidSysfsRoot
|
||||
mdraidSysfsRoot = tmp
|
||||
t.Cleanup(func() { mdraidSysfsRoot = prev })
|
||||
|
||||
mdDir := filepath.Join(tmp, "block", "md0", "md")
|
||||
queueDir := filepath.Join(tmp, "block", "md0", "queue")
|
||||
if err := os.MkdirAll(mdDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.MkdirAll(queueDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
write := func(path, content string) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
write(filepath.Join(mdDir, "array_state"), "active\n")
|
||||
write(filepath.Join(mdDir, "level"), "raid1\n")
|
||||
write(filepath.Join(mdDir, "raid_disks"), "2\n")
|
||||
write(filepath.Join(mdDir, "degraded"), "0\n")
|
||||
write(filepath.Join(mdDir, "sync_action"), "resync\n")
|
||||
write(filepath.Join(mdDir, "sync_completed"), "10%\n")
|
||||
write(filepath.Join(mdDir, "sync_speed"), "100M\n")
|
||||
write(filepath.Join(mdDir, "mismatch_cnt"), "0\n")
|
||||
|
||||
// Simulate two healthy member devices (no faulty state).
|
||||
for _, dev := range []string{"dev-sda", "dev-sdb"} {
|
||||
devPath := filepath.Join(mdDir, dev)
|
||||
if err := os.MkdirAll(devPath, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
write(filepath.Join(devPath, "state"), "in_sync\n")
|
||||
}
|
||||
write(filepath.Join(queueDir, "logical_block_size"), "512\n")
|
||||
write(filepath.Join(tmp, "block", "md0", "size"), "2048\n")
|
||||
|
||||
devs := scanMdraidDevices()
|
||||
if len(devs) != 1 {
|
||||
t.Fatalf("scanMdraidDevices() = %d devices, want 1", len(devs))
|
||||
}
|
||||
if devs[0].Name != "/dev/md0" || devs[0].Type != "mdraid" {
|
||||
t.Fatalf("scanMdraidDevices()[0] = %+v, want Name=/dev/md0 Type=mdraid", devs[0])
|
||||
}
|
||||
|
||||
sm := &SmartManager{SmartDataMap: map[string]*smart.SmartData{}}
|
||||
ok, err := sm.collectMdraidHealth(devs[0])
|
||||
if err != nil || !ok {
|
||||
t.Fatalf("collectMdraidHealth() = (ok=%v, err=%v), want (true,nil)", ok, err)
|
||||
}
|
||||
if len(sm.SmartDataMap) != 1 {
|
||||
t.Fatalf("SmartDataMap len=%d, want 1", len(sm.SmartDataMap))
|
||||
}
|
||||
var got *smart.SmartData
|
||||
for _, v := range sm.SmartDataMap {
|
||||
got = v
|
||||
break
|
||||
}
|
||||
if got == nil {
|
||||
t.Fatalf("SmartDataMap value nil")
|
||||
}
|
||||
if got.DiskType != "mdraid" || got.DiskName != "/dev/md0" {
|
||||
t.Fatalf("disk fields = (type=%q name=%q), want (mdraid,/dev/md0)", got.DiskType, got.DiskName)
|
||||
}
|
||||
if got.SmartStatus != "WARNING" {
|
||||
t.Fatalf("SmartStatus=%q, want WARNING", got.SmartStatus)
|
||||
}
|
||||
if got.ModelName == "" || got.Capacity == 0 {
|
||||
t.Fatalf("identity fields = (model=%q cap=%d), want non-empty model and cap>0", got.ModelName, got.Capacity)
|
||||
}
|
||||
if len(got.Attributes) < 5 {
|
||||
t.Fatalf("attributes len=%d, want >= 5", len(got.Attributes))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCountMdraidMemberStates(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
|
||||
write := func(path, content string) {
|
||||
t.Helper()
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
mdDir := filepath.Join(tmp, "block", "md0", "md")
|
||||
|
||||
// No dev-* entries: zero faulty, zero populated.
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 0 {
|
||||
t.Fatalf("no members: got (faulty=%d populated=%d), want (0,0)", faulty, populated)
|
||||
}
|
||||
|
||||
// Two healthy members.
|
||||
write(filepath.Join(mdDir, "dev-sda", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 2 {
|
||||
t.Fatalf("all in_sync: got (faulty=%d populated=%d), want (0,2)", faulty, populated)
|
||||
}
|
||||
|
||||
// One faulty member.
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "faulty\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 1 || populated != 2 {
|
||||
t.Fatalf("one faulty: got (faulty=%d populated=%d), want (1,2)", faulty, populated)
|
||||
}
|
||||
|
||||
// QNAP-style: 28 degraded slots but no dev-* entries for them, 4 in_sync.
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdc", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdd", "state"), "in_sync\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 4 {
|
||||
t.Fatalf("qnap sparse: got (faulty=%d populated=%d), want (0,4)", faulty, populated)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMdraidSmartStatus(t *testing.T) {
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "inactive"}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(inactive) = %q, want FAILED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1, syncAction: "recover"}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded+recover) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded+faulty) = %q, want FAILED", got)
|
||||
}
|
||||
// QNAP-style: raid_disks=32 but only 4 populated; degraded=28 but no faulty devices.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 28, faultyDisks: 0, raidDisks: 32, populatedDisks: 4}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(qnap sparse) = %q, want WARNING", got)
|
||||
}
|
||||
// A member disappearing from the same sparse array is indistinguishable
|
||||
// from another reserved slot, so it must not be reported as healthy.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 29, faultyDisks: 0, raidDisks: 32, populatedDisks: 3}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(qnap sparse missing member) = %q, want WARNING", got)
|
||||
}
|
||||
// A genuinely missing member (removed dev-* entry, not just an unpopulated
|
||||
// QNAP reserve slot) must still fail: raid_disks=4, only 3 populated, all
|
||||
// of them in_sync, so faultyDisks==0 but degraded==1.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 3}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(missing member) = %q, want FAILED", got)
|
||||
}
|
||||
// Degraded with no member-state info at all (e.g. sysfs read failed) must
|
||||
// still fail rather than being silently treated as a sparse QNAP array.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 0}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded, no member info) = %q, want FAILED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", syncAction: "recover"}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(recover) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "check"}); got != "PASSED" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+check) = %q, want PASSED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "check", mismatchCnt: 1}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+check+mismatch) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", mismatchCnt: 1}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+mismatch) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "repair"}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(repair) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean"}); got != "PASSED" {
|
||||
t.Fatalf("mdraidSmartStatus(clean) = %q, want PASSED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "unknown"}); got != "UNKNOWN" {
|
||||
t.Fatalf("mdraidSmartStatus(unknown) = %q, want UNKNOWN", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
func scanMdraidDevices() []*DeviceInfo {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (sm *SmartManager) collectMdraidHealth(deviceInfo *DeviceInfo) (bool, error) {
|
||||
return false, nil
|
||||
}
|
||||
+18
-14
@@ -8,6 +8,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/deltatracker"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
psutilNet "github.com/shirou/gopsutil/v4/net"
|
||||
)
|
||||
@@ -94,7 +95,7 @@ func (a *Agent) initializeNetIoStats() {
|
||||
a.netInterfaces = make(map[string]struct{}, 0)
|
||||
|
||||
// parse NICS env var for whitelist / blacklist
|
||||
nicsEnvVal, nicsEnvExists := GetEnv("NICS")
|
||||
nicsEnvVal, nicsEnvExists := utils.GetEnv("NICS")
|
||||
var nicCfg *NicConfig
|
||||
if nicsEnvExists {
|
||||
nicCfg = newNicConfig(nicsEnvVal)
|
||||
@@ -103,10 +104,7 @@ func (a *Agent) initializeNetIoStats() {
|
||||
// get current network I/O stats and record valid interfaces
|
||||
if netIO, err := psutilNet.IOCounters(true); err == nil {
|
||||
for _, v := range netIO {
|
||||
if nicsEnvExists && !isValidNic(v.Name, nicCfg) {
|
||||
continue
|
||||
}
|
||||
if a.skipNetworkInterface(v) {
|
||||
if skipNetworkInterface(v, nicCfg) {
|
||||
continue
|
||||
}
|
||||
slog.Info("Detected network interface", "name", v.Name, "sent", v.BytesSent, "recv", v.BytesRecv)
|
||||
@@ -215,10 +213,8 @@ func (a *Agent) applyNetworkTotals(
|
||||
totalBytesSent, totalBytesRecv uint64,
|
||||
bytesSentPerSecond, bytesRecvPerSecond uint64,
|
||||
) {
|
||||
networkSentPs := bytesToMegabytes(float64(bytesSentPerSecond))
|
||||
networkRecvPs := bytesToMegabytes(float64(bytesRecvPerSecond))
|
||||
if networkSentPs > 10_000 || networkRecvPs > 10_000 {
|
||||
slog.Warn("Invalid net stats. Resetting.", "sent", networkSentPs, "recv", networkRecvPs)
|
||||
if bytesSentPerSecond > 10_000_000_000 || bytesRecvPerSecond > 10_000_000_000 {
|
||||
slog.Warn("Invalid net stats. Resetting.", "sent", bytesSentPerSecond, "recv", bytesRecvPerSecond)
|
||||
for _, v := range netIO {
|
||||
if _, exists := a.netInterfaces[v.Name]; !exists {
|
||||
continue
|
||||
@@ -228,21 +224,29 @@ func (a *Agent) applyNetworkTotals(
|
||||
a.initializeNetIoStats()
|
||||
delete(a.netIoStats, cacheTimeMs)
|
||||
delete(a.netInterfaceDeltaTrackers, cacheTimeMs)
|
||||
systemStats.NetworkSent = 0
|
||||
systemStats.NetworkRecv = 0
|
||||
systemStats.Bandwidth[0], systemStats.Bandwidth[1] = 0, 0
|
||||
return
|
||||
}
|
||||
|
||||
systemStats.NetworkSent = networkSentPs
|
||||
systemStats.NetworkRecv = networkRecvPs
|
||||
systemStats.Bandwidth[0], systemStats.Bandwidth[1] = bytesSentPerSecond, bytesRecvPerSecond
|
||||
nis.BytesSent = totalBytesSent
|
||||
nis.BytesRecv = totalBytesRecv
|
||||
a.netIoStats[cacheTimeMs] = nis
|
||||
}
|
||||
|
||||
func (a *Agent) skipNetworkInterface(v psutilNet.IOCountersStat) bool {
|
||||
// skipNetworkInterface returns true if the network interface should be ignored.
|
||||
func skipNetworkInterface(v psutilNet.IOCountersStat, nicCfg *NicConfig) bool {
|
||||
if nicCfg != nil {
|
||||
if !isValidNic(v.Name, nicCfg) {
|
||||
return true
|
||||
}
|
||||
// In whitelist mode, we honor explicit inclusion without auto-filtering.
|
||||
if !nicCfg.isBlacklist {
|
||||
return false
|
||||
}
|
||||
// In blacklist mode, still apply the auto-filter below.
|
||||
}
|
||||
|
||||
switch {
|
||||
case strings.HasPrefix(v.Name, "lo"),
|
||||
strings.HasPrefix(v.Name, "docker"),
|
||||
|
||||
+33
-22
@@ -261,6 +261,39 @@ func TestNewNicConfig(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
func TestSkipNetworkInterface(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
nic psutilNet.IOCountersStat
|
||||
nicCfg *NicConfig
|
||||
expectSkip bool
|
||||
}{
|
||||
{"loopback lo", psutilNet.IOCountersStat{Name: "lo", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"loopback lo0", psutilNet.IOCountersStat{Name: "lo0", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"docker prefix", psutilNet.IOCountersStat{Name: "docker0", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"br- prefix", psutilNet.IOCountersStat{Name: "br-lan", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"veth prefix", psutilNet.IOCountersStat{Name: "veth0abc", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"bond prefix", psutilNet.IOCountersStat{Name: "bond0", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"cali prefix", psutilNet.IOCountersStat{Name: "cali1234", BytesSent: 100, BytesRecv: 100}, nil, true},
|
||||
{"zero BytesRecv", psutilNet.IOCountersStat{Name: "eth0", BytesSent: 100, BytesRecv: 0}, nil, true},
|
||||
{"zero BytesSent", psutilNet.IOCountersStat{Name: "eth0", BytesSent: 0, BytesRecv: 100}, nil, true},
|
||||
{"both zero", psutilNet.IOCountersStat{Name: "eth0", BytesSent: 0, BytesRecv: 0}, nil, true},
|
||||
{"normal eth0", psutilNet.IOCountersStat{Name: "eth0", BytesSent: 100, BytesRecv: 200}, nil, false},
|
||||
{"normal wlan0", psutilNet.IOCountersStat{Name: "wlan0", BytesSent: 1, BytesRecv: 1}, nil, false},
|
||||
{"whitelist overrides skip (docker)", psutilNet.IOCountersStat{Name: "docker0", BytesSent: 100, BytesRecv: 100}, newNicConfig("docker0"), false},
|
||||
{"whitelist overrides skip (lo)", psutilNet.IOCountersStat{Name: "lo", BytesSent: 100, BytesRecv: 100}, newNicConfig("lo"), false},
|
||||
{"whitelist exclusion", psutilNet.IOCountersStat{Name: "eth1", BytesSent: 100, BytesRecv: 100}, newNicConfig("eth0"), true},
|
||||
{"blacklist skip lo", psutilNet.IOCountersStat{Name: "lo", BytesSent: 100, BytesRecv: 100}, newNicConfig("-eth0"), true},
|
||||
{"blacklist explicit eth0", psutilNet.IOCountersStat{Name: "eth0", BytesSent: 100, BytesRecv: 100}, newNicConfig("-eth0"), true},
|
||||
{"blacklist allow eth1", psutilNet.IOCountersStat{Name: "eth1", BytesSent: 100, BytesRecv: 100}, newNicConfig("-eth0"), false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
assert.Equal(t, tt.expectSkip, skipNetworkInterface(tt.nic, tt.nicCfg))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnsureNetworkInterfacesMap(t *testing.T) {
|
||||
var a Agent
|
||||
var stats system.Stats
|
||||
@@ -383,8 +416,6 @@ func TestApplyNetworkTotals(t *testing.T) {
|
||||
totalBytesSent uint64
|
||||
totalBytesRecv uint64
|
||||
expectReset bool
|
||||
expectedNetworkSent float64
|
||||
expectedNetworkRecv float64
|
||||
expectedBandwidthSent uint64
|
||||
expectedBandwidthRecv uint64
|
||||
}{
|
||||
@@ -395,8 +426,6 @@ func TestApplyNetworkTotals(t *testing.T) {
|
||||
totalBytesSent: 10000000,
|
||||
totalBytesRecv: 20000000,
|
||||
expectReset: false,
|
||||
expectedNetworkSent: 0.95, // ~1 MB/s rounded to 2 decimals
|
||||
expectedNetworkRecv: 1.91, // ~2 MB/s rounded to 2 decimals
|
||||
expectedBandwidthSent: 1000000,
|
||||
expectedBandwidthRecv: 2000000,
|
||||
},
|
||||
@@ -424,18 +453,6 @@ func TestApplyNetworkTotals(t *testing.T) {
|
||||
totalBytesRecv: 20000000,
|
||||
expectReset: true,
|
||||
},
|
||||
{
|
||||
name: "Valid network stats - at threshold boundary",
|
||||
bytesSentPerSecond: 10485750000, // ~9999.99 MB/s (rounds to 9999.99)
|
||||
bytesRecvPerSecond: 10485750000, // ~9999.99 MB/s (rounds to 9999.99)
|
||||
totalBytesSent: 10000000,
|
||||
totalBytesRecv: 20000000,
|
||||
expectReset: false,
|
||||
expectedNetworkSent: 9999.99,
|
||||
expectedNetworkRecv: 9999.99,
|
||||
expectedBandwidthSent: 10485750000,
|
||||
expectedBandwidthRecv: 10485750000,
|
||||
},
|
||||
{
|
||||
name: "Zero values",
|
||||
bytesSentPerSecond: 0,
|
||||
@@ -443,8 +460,6 @@ func TestApplyNetworkTotals(t *testing.T) {
|
||||
totalBytesSent: 0,
|
||||
totalBytesRecv: 0,
|
||||
expectReset: false,
|
||||
expectedNetworkSent: 0.0,
|
||||
expectedNetworkRecv: 0.0,
|
||||
expectedBandwidthSent: 0,
|
||||
expectedBandwidthRecv: 0,
|
||||
},
|
||||
@@ -481,14 +496,10 @@ func TestApplyNetworkTotals(t *testing.T) {
|
||||
// Should have reset network tracking state - maps cleared and stats zeroed
|
||||
assert.NotContains(t, a.netIoStats, cacheTimeMs, "cache entry should be cleared after reset")
|
||||
assert.NotContains(t, a.netInterfaceDeltaTrackers, cacheTimeMs, "tracker should be cleared on reset")
|
||||
assert.Zero(t, systemStats.NetworkSent)
|
||||
assert.Zero(t, systemStats.NetworkRecv)
|
||||
assert.Zero(t, systemStats.Bandwidth[0])
|
||||
assert.Zero(t, systemStats.Bandwidth[1])
|
||||
} else {
|
||||
// Should have applied stats
|
||||
assert.Equal(t, tt.expectedNetworkSent, systemStats.NetworkSent)
|
||||
assert.Equal(t, tt.expectedNetworkRecv, systemStats.NetworkRecv)
|
||||
assert.Equal(t, tt.expectedBandwidthSent, systemStats.Bandwidth[0])
|
||||
assert.Equal(t, tt.expectedBandwidthRecv, systemStats.Bandwidth[1])
|
||||
|
||||
|
||||
@@ -21,6 +21,9 @@ func newAgentResponse(data any, requestID *uint32) common.AgentResponse {
|
||||
response.String = &v
|
||||
case map[string]smart.SmartData:
|
||||
response.SmartData = v
|
||||
case smart.SmartDataResponse:
|
||||
response.SmartData = v.Data
|
||||
response.SmartComplete = v.Complete
|
||||
case systemd.ServiceDetails:
|
||||
response.ServiceInfo = v
|
||||
default:
|
||||
|
||||
+70
-21
@@ -2,48 +2,67 @@ package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"path"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/common"
|
||||
"github.com/shirou/gopsutil/v4/sensors"
|
||||
)
|
||||
|
||||
type SensorConfig struct {
|
||||
context context.Context
|
||||
sensors map[string]struct{}
|
||||
primarySensor string
|
||||
isBlacklist bool
|
||||
hasWildcards bool
|
||||
skipCollection bool
|
||||
}
|
||||
|
||||
func (a *Agent) newSensorConfig() *SensorConfig {
|
||||
primarySensor, _ := GetEnv("PRIMARY_SENSOR")
|
||||
sysSensors, _ := GetEnv("SYS_SENSORS")
|
||||
sensorsEnvVal, sensorsSet := GetEnv("SENSORS")
|
||||
skipCollection := sensorsSet && sensorsEnvVal == ""
|
||||
|
||||
return a.newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, skipCollection)
|
||||
}
|
||||
var errTemperatureFetchTimeout = errors.New("temperature collection timed out")
|
||||
|
||||
// Matches sensors.TemperaturesWithContext to allow for panic recovery (gopsutil/issues/1832)
|
||||
type getTempsFn func(ctx context.Context) ([]sensors.TemperatureStat, error)
|
||||
|
||||
type SensorConfig struct {
|
||||
context context.Context
|
||||
sensors map[string]struct{}
|
||||
primarySensor string
|
||||
timeout time.Duration
|
||||
isBlacklist bool
|
||||
hasWildcards bool
|
||||
skipCollection bool
|
||||
firstRun bool
|
||||
}
|
||||
|
||||
func (a *Agent) newSensorConfig() *SensorConfig {
|
||||
primarySensor, _ := utils.GetEnv("PRIMARY_SENSOR")
|
||||
sysSensors, _ := utils.GetEnv("SYS_SENSORS")
|
||||
sensorsEnvVal, sensorsSet := utils.GetEnv("SENSORS")
|
||||
skipCollection := sensorsSet && sensorsEnvVal == ""
|
||||
sensorsTimeout, _ := utils.GetEnv("SENSORS_TIMEOUT")
|
||||
|
||||
return a.newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout, skipCollection)
|
||||
}
|
||||
|
||||
// newSensorConfigWithEnv creates a SensorConfig with the provided environment variables
|
||||
// sensorsSet indicates if the SENSORS environment variable was explicitly set (even to empty string)
|
||||
func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal string, skipCollection bool) *SensorConfig {
|
||||
func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout string, skipCollection bool) *SensorConfig {
|
||||
timeout := 2 * time.Second
|
||||
if sensorsTimeout != "" {
|
||||
if d, err := time.ParseDuration(sensorsTimeout); err == nil {
|
||||
timeout = d
|
||||
} else {
|
||||
slog.Warn("Invalid SENSORS_TIMEOUT", "value", sensorsTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
config := &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: primarySensor,
|
||||
timeout: timeout,
|
||||
skipCollection: skipCollection,
|
||||
firstRun: true,
|
||||
sensors: make(map[string]struct{}),
|
||||
}
|
||||
|
||||
@@ -85,10 +104,12 @@ func (a *Agent) updateTemperatures(systemStats *system.Stats) {
|
||||
// reset high temp
|
||||
a.systemInfo.DashboardTemp = 0
|
||||
|
||||
temps, err := a.getTempsWithPanicRecovery(getSensorTemps)
|
||||
temps, err := a.getTempsWithTimeout(getSensorTemps)
|
||||
if err != nil {
|
||||
// retry once on panic (gopsutil/issues/1832)
|
||||
temps, err = a.getTempsWithPanicRecovery(getSensorTemps)
|
||||
if !errors.Is(err, errTemperatureFetchTimeout) {
|
||||
temps, err = a.getTempsWithTimeout(getSensorTemps)
|
||||
}
|
||||
if err != nil {
|
||||
slog.Warn("Error updating temperatures", "err", err)
|
||||
if len(systemStats.Temperatures) > 0 {
|
||||
@@ -135,7 +156,7 @@ func (a *Agent) updateTemperatures(systemStats *system.Stats) {
|
||||
case sensorName:
|
||||
a.systemInfo.DashboardTemp = sensor.Temperature
|
||||
}
|
||||
systemStats.Temperatures[sensorName] = twoDecimals(sensor.Temperature)
|
||||
systemStats.Temperatures[sensorName] = utils.TwoDecimals(sensor.Temperature)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -151,6 +172,34 @@ func (a *Agent) getTempsWithPanicRecovery(getTemps getTempsFn) (temps []sensors.
|
||||
return
|
||||
}
|
||||
|
||||
func (a *Agent) getTempsWithTimeout(getTemps getTempsFn) ([]sensors.TemperatureStat, error) {
|
||||
type result struct {
|
||||
temps []sensors.TemperatureStat
|
||||
err error
|
||||
}
|
||||
|
||||
// Use a longer timeout on the first run to allow for initialization
|
||||
// (e.g. Windows LHM subprocess startup)
|
||||
timeout := a.sensorConfig.timeout
|
||||
if a.sensorConfig.firstRun {
|
||||
a.sensorConfig.firstRun = false
|
||||
timeout = 10 * time.Second
|
||||
}
|
||||
|
||||
resultCh := make(chan result, 1)
|
||||
go func() {
|
||||
temps, err := a.getTempsWithPanicRecovery(getTemps)
|
||||
resultCh <- result{temps: temps, err: err}
|
||||
}()
|
||||
|
||||
select {
|
||||
case res := <-resultCh:
|
||||
return res.temps, res.err
|
||||
case <-time.After(timeout):
|
||||
return nil, errTemperatureFetchTimeout
|
||||
}
|
||||
}
|
||||
|
||||
// isValidSensor checks if a sensor is valid based on the sensor name and the sensor config
|
||||
func isValidSensor(sensorName string, config *SensorConfig) bool {
|
||||
// if no sensors configured, everything is valid
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build !windows
|
||||
//go:build !windows && !freebsd
|
||||
|
||||
package agent
|
||||
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
//go:build freebsd
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/sensors"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
var getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
return getFreeBSDSensorTemps(ctx, unix.SysctlUint32)
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
//go:build freebsd || testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/sensors"
|
||||
)
|
||||
|
||||
const (
|
||||
freebsdZeroCelsiusDeciKelvin = 2731
|
||||
freebsdAcpiThermalZoneCount = 16
|
||||
)
|
||||
|
||||
type freebsdSysctlUintReader func(name string) (uint32, error)
|
||||
|
||||
func getFreeBSDSensorTemps(ctx context.Context, readSysctl freebsdSysctlUintReader) ([]sensors.TemperatureStat, error) {
|
||||
cpuCount, err := readSysctl("hw.ncpu")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
temps := make([]sensors.TemperatureStat, 0, int(cpuCount)+freebsdAcpiThermalZoneCount)
|
||||
for cpu := range cpuCount {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return temps, ctx.Err()
|
||||
default:
|
||||
}
|
||||
|
||||
sysctlName := fmt.Sprintf("dev.cpu.%d.temperature", cpu)
|
||||
value, err := readSysctl(sysctlName)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
temp, ok := freebsdDeciKelvinToCelsius(value)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
temps = append(temps, sensors.TemperatureStat{
|
||||
SensorKey: fmt.Sprintf("cpu.%d", cpu),
|
||||
Temperature: temp,
|
||||
})
|
||||
}
|
||||
|
||||
for zone := 0; zone < freebsdAcpiThermalZoneCount; zone++ {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return temps, ctx.Err()
|
||||
default:
|
||||
}
|
||||
|
||||
sysctlName := fmt.Sprintf("hw.acpi.thermal.tz%d.temperature", zone)
|
||||
value, err := readSysctl(sysctlName)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
temp, ok := freebsdDeciKelvinToCelsius(value)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
temps = append(temps, sensors.TemperatureStat{
|
||||
SensorKey: fmt.Sprintf("acpi.thermal.tz%d", zone),
|
||||
Temperature: temp,
|
||||
})
|
||||
}
|
||||
|
||||
return temps, nil
|
||||
}
|
||||
|
||||
func freebsdDeciKelvinToCelsius(value uint32) (float64, bool) {
|
||||
if value <= freebsdZeroCelsiusDeciKelvin {
|
||||
return 0, false
|
||||
}
|
||||
temp := float64(int64(value)-freebsdZeroCelsiusDeciKelvin) / 10
|
||||
if temp <= 0 || temp >= 200 {
|
||||
return 0, false
|
||||
}
|
||||
return temp, true
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
var errFakeFreeBSDSysctlNotFound = errors.New("sysctl not found")
|
||||
|
||||
type fakeFreeBSDSysctls struct {
|
||||
values map[string]uint32
|
||||
errs map[string]error
|
||||
}
|
||||
|
||||
func (f fakeFreeBSDSysctls) read(name string) (uint32, error) {
|
||||
if err, ok := f.errs[name]; ok {
|
||||
return 0, err
|
||||
}
|
||||
if value, ok := f.values[name]; ok {
|
||||
return value, nil
|
||||
}
|
||||
return 0, errFakeFreeBSDSysctlNotFound
|
||||
}
|
||||
|
||||
func TestFreeBSDDeciKelvinToCelsius(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
value uint32
|
||||
expected float64
|
||||
ok bool
|
||||
}{
|
||||
{
|
||||
name: "45 Celsius",
|
||||
value: 3181,
|
||||
expected: 45,
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "fractional Celsius",
|
||||
value: 3186,
|
||||
expected: 45.5,
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "zero deci-Kelvin",
|
||||
value: 0,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "zero Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "below zero Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin - 1,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "invalid signed integer",
|
||||
value: 1<<32 - 1,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "unreasonably high Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin + 2000,
|
||||
ok: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result, ok := freebsdDeciKelvinToCelsius(tt.value)
|
||||
assert.Equal(t, tt.ok, ok)
|
||||
assert.InDelta(t, tt.expected, result, 0.001)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTemps(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{
|
||||
"hw.ncpu": 4,
|
||||
"dev.cpu.0.temperature": 3231,
|
||||
"dev.cpu.1.temperature": 3242,
|
||||
"dev.cpu.3.temperature": freebsdZeroCelsiusDeciKelvin,
|
||||
"hw.acpi.thermal.tz0.temperature": 3101,
|
||||
"hw.acpi.thermal.tz2.temperature": 3116,
|
||||
"hw.acpi.thermal.tz3.temperature": freebsdZeroCelsiusDeciKelvin,
|
||||
"unrelated.sensor.value": 9999,
|
||||
"dev.cpu.99.temperature": 9999,
|
||||
"dev.amdtemp.0.core0.foo": 9999,
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
require.Len(t, temps, 4)
|
||||
assert.Equal(t, "cpu.0", temps[0].SensorKey)
|
||||
assert.InDelta(t, 50.0, temps[0].Temperature, 0.001)
|
||||
assert.Equal(t, "cpu.1", temps[1].SensorKey)
|
||||
assert.InDelta(t, 51.1, temps[1].Temperature, 0.001)
|
||||
assert.Equal(t, "acpi.thermal.tz0", temps[2].SensorKey)
|
||||
assert.InDelta(t, 37.0, temps[2].Temperature, 0.001)
|
||||
assert.Equal(t, "acpi.thermal.tz2", temps[3].SensorKey)
|
||||
assert.InDelta(t, 38.5, temps[3].Temperature, 0.001)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsCpuCountError(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
errs: map[string]error{
|
||||
"hw.ncpu": errors.New("permission denied"),
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
assert.Nil(t, temps)
|
||||
assert.EqualError(t, err, "permission denied")
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsNoTemperatureSysctls(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{"hw.ncpu": 2},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, temps)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsAcpiOnly(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{
|
||||
"hw.ncpu": 0,
|
||||
"hw.acpi.thermal.tz0.temperature": 3081,
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
require.Len(t, temps, 1)
|
||||
assert.Equal(t, "acpi.thermal.tz0", temps[0].SensorKey)
|
||||
assert.InDelta(t, 35.0, temps[0].Temperature, 0.001)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsContextCancelled(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{"hw.ncpu": 2},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(ctx, reader.read)
|
||||
|
||||
assert.Empty(t, temps)
|
||||
assert.ErrorIs(t, err, context.Canceled)
|
||||
}
|
||||
+98
-30
@@ -1,13 +1,12 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
@@ -169,6 +168,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
primarySensor string
|
||||
sysSensors string
|
||||
sensors string
|
||||
sensorsTimeout string
|
||||
skipCollection bool
|
||||
expectedConfig *SensorConfig
|
||||
}{
|
||||
@@ -180,12 +180,37 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
isBlacklist: false,
|
||||
hasWildcards: false,
|
||||
skipCollection: false,
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Custom timeout",
|
||||
primarySensor: "",
|
||||
sysSensors: "",
|
||||
sensors: "",
|
||||
sensorsTimeout: "5s",
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
timeout: 5 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Invalid timeout falls back to default",
|
||||
primarySensor: "",
|
||||
sysSensors: "",
|
||||
sensors: "",
|
||||
sensorsTimeout: "notaduration",
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Explicitly set to empty string",
|
||||
primarySensor: "",
|
||||
@@ -195,6 +220,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
isBlacklist: false,
|
||||
hasWildcards: false,
|
||||
@@ -209,6 +235,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
isBlacklist: false,
|
||||
hasWildcards: false,
|
||||
@@ -222,6 +249,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{
|
||||
"cpu_temp": {},
|
||||
"gpu_temp": {},
|
||||
@@ -238,6 +266,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{
|
||||
"cpu_temp": {},
|
||||
"gpu_temp": {},
|
||||
@@ -254,6 +283,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{
|
||||
"cpu_*": {},
|
||||
"gpu_temp": {},
|
||||
@@ -270,6 +300,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
expectedConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{
|
||||
"cpu_*": {},
|
||||
"gpu_temp": {},
|
||||
@@ -285,6 +316,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
sensors: "cpu_temp",
|
||||
expectedConfig: &SensorConfig{
|
||||
primarySensor: "cpu_temp",
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{
|
||||
"cpu_temp": {},
|
||||
},
|
||||
@@ -296,7 +328,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := agent.newSensorConfigWithEnv(tt.primarySensor, tt.sysSensors, tt.sensors, tt.skipCollection)
|
||||
result := agent.newSensorConfigWithEnv(tt.primarySensor, tt.sysSensors, tt.sensors, tt.sensorsTimeout, tt.skipCollection)
|
||||
|
||||
// Check primary sensor
|
||||
assert.Equal(t, tt.expectedConfig.primarySensor, result.primarySensor)
|
||||
@@ -315,6 +347,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
// Check flags
|
||||
assert.Equal(t, tt.expectedConfig.isBlacklist, result.isBlacklist)
|
||||
assert.Equal(t, tt.expectedConfig.hasWildcards, result.hasWildcards)
|
||||
assert.Equal(t, tt.expectedConfig.timeout, result.timeout)
|
||||
|
||||
// Check context
|
||||
if tt.sysSensors != "" {
|
||||
@@ -330,40 +363,18 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestNewSensorConfig(t *testing.T) {
|
||||
// Save original environment variables
|
||||
originalPrimary, hasPrimary := os.LookupEnv("BESZEL_AGENT_PRIMARY_SENSOR")
|
||||
originalSys, hasSys := os.LookupEnv("BESZEL_AGENT_SYS_SENSORS")
|
||||
originalSensors, hasSensors := os.LookupEnv("BESZEL_AGENT_SENSORS")
|
||||
|
||||
// Restore environment variables after the test
|
||||
defer func() {
|
||||
// Clean up test environment variables
|
||||
os.Unsetenv("BESZEL_AGENT_PRIMARY_SENSOR")
|
||||
os.Unsetenv("BESZEL_AGENT_SYS_SENSORS")
|
||||
os.Unsetenv("BESZEL_AGENT_SENSORS")
|
||||
|
||||
// Restore original values if they existed
|
||||
if hasPrimary {
|
||||
os.Setenv("BESZEL_AGENT_PRIMARY_SENSOR", originalPrimary)
|
||||
}
|
||||
if hasSys {
|
||||
os.Setenv("BESZEL_AGENT_SYS_SENSORS", originalSys)
|
||||
}
|
||||
if hasSensors {
|
||||
os.Setenv("BESZEL_AGENT_SENSORS", originalSensors)
|
||||
}
|
||||
}()
|
||||
|
||||
// Set test environment variables
|
||||
os.Setenv("BESZEL_AGENT_PRIMARY_SENSOR", "test_primary")
|
||||
os.Setenv("BESZEL_AGENT_SYS_SENSORS", "/test/path")
|
||||
os.Setenv("BESZEL_AGENT_SENSORS", "test_sensor1,test_*,test_sensor3")
|
||||
t.Setenv("BESZEL_AGENT_PRIMARY_SENSOR", "test_primary")
|
||||
t.Setenv("BESZEL_AGENT_SYS_SENSORS", "/test/path")
|
||||
t.Setenv("BESZEL_AGENT_SENSORS", "test_sensor1,test_*,test_sensor3")
|
||||
t.Setenv("BESZEL_AGENT_SENSORS_TIMEOUT", "7s")
|
||||
|
||||
agent := &Agent{}
|
||||
result := agent.newSensorConfig()
|
||||
|
||||
// Verify results
|
||||
assert.Equal(t, "test_primary", result.primarySensor)
|
||||
assert.Equal(t, 7*time.Second, result.timeout)
|
||||
assert.NotNil(t, result.sensors)
|
||||
assert.Equal(t, 3, len(result.sensors))
|
||||
assert.True(t, result.hasWildcards)
|
||||
@@ -552,3 +563,60 @@ func TestGetTempsWithPanicRecovery(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGetTempsWithTimeout(t *testing.T) {
|
||||
agent := &Agent{
|
||||
sensorConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
timeout: 10 * time.Millisecond,
|
||||
},
|
||||
}
|
||||
|
||||
t.Run("returns temperatures before timeout", func(t *testing.T) {
|
||||
temps, err := agent.getTempsWithTimeout(func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
return []sensors.TemperatureStat{{SensorKey: "cpu_temp", Temperature: 42}}, nil
|
||||
})
|
||||
|
||||
require.NoError(t, err)
|
||||
require.Len(t, temps, 1)
|
||||
assert.Equal(t, "cpu_temp", temps[0].SensorKey)
|
||||
})
|
||||
|
||||
t.Run("returns timeout error when collector hangs", func(t *testing.T) {
|
||||
temps, err := agent.getTempsWithTimeout(func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
return []sensors.TemperatureStat{{SensorKey: "cpu_temp", Temperature: 42}}, nil
|
||||
})
|
||||
|
||||
assert.Nil(t, temps)
|
||||
assert.ErrorIs(t, err, errTemperatureFetchTimeout)
|
||||
})
|
||||
}
|
||||
|
||||
func TestUpdateTemperaturesSkipsOnTimeout(t *testing.T) {
|
||||
agent := &Agent{
|
||||
systemInfo: system.Info{DashboardTemp: 99},
|
||||
sensorConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
timeout: 10 * time.Millisecond,
|
||||
},
|
||||
}
|
||||
|
||||
originalGetSensorTemps := getSensorTemps
|
||||
t.Cleanup(func() {
|
||||
getSensorTemps = originalGetSensorTemps
|
||||
})
|
||||
getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
stats := &system.Stats{
|
||||
Temperatures: map[string]float64{"stale": 50},
|
||||
}
|
||||
|
||||
agent.updateTemperatures(stats)
|
||||
|
||||
assert.Equal(t, 0.0, agent.systemInfo.DashboardTemp)
|
||||
assert.Equal(t, map[string]float64{}, stats.Temperatures)
|
||||
}
|
||||
|
||||
@@ -214,9 +214,12 @@ func (lhm *lhmProcess) getTemps(ctx context.Context) (temps []sensors.Temperatur
|
||||
return temps, nil
|
||||
}
|
||||
|
||||
// getSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
|
||||
// getSensorTemps is a variable so tests can replace the platform sensor collector.
|
||||
var getSensorTemps = getWindowsSensorTemps
|
||||
|
||||
// getWindowsSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
|
||||
// NB: LibreHardwareMonitorLib requires admin privileges to access all available sensors.
|
||||
func getSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
|
||||
func getWindowsSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
|
||||
defer func() {
|
||||
if err != nil {
|
||||
slog.Debug("Error reading sensors", "err", err)
|
||||
|
||||
+13
-26
@@ -12,6 +12,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
@@ -28,15 +29,12 @@ type ServerOptions struct {
|
||||
Keys []gossh.PublicKey // SSH public keys for authentication
|
||||
}
|
||||
|
||||
// hubVersions caches hub versions by session ID to avoid repeated parsing.
|
||||
var hubVersions map[string]semver.Version
|
||||
|
||||
// StartServer starts the SSH server with the provided options.
|
||||
// It configures the server with secure defaults, sets up authentication,
|
||||
// and begins listening for connections. Returns an error if the server
|
||||
// is already running or if there's an issue starting the server.
|
||||
func (a *Agent) StartServer(opts ServerOptions) error {
|
||||
if disableSSH, _ := GetEnv("DISABLE_SSH"); disableSSH == "true" {
|
||||
if disableSSH, _ := utils.GetEnv("DISABLE_SSH"); disableSSH == "true" {
|
||||
return errors.New("SSH disabled")
|
||||
}
|
||||
if a.server != nil {
|
||||
@@ -98,24 +96,15 @@ func (a *Agent) StartServer(opts ServerOptions) error {
|
||||
return a.server.Serve(ln)
|
||||
}
|
||||
|
||||
// getHubVersion retrieves and caches the hub version for a given session.
|
||||
// It extracts the version from the SSH client version string and caches
|
||||
// it to avoid repeated parsing. Returns a zero version if parsing fails.
|
||||
func (a *Agent) getHubVersion(sessionId string, sessionCtx ssh.Context) semver.Version {
|
||||
if hubVersions == nil {
|
||||
hubVersions = make(map[string]semver.Version, 1)
|
||||
}
|
||||
hubVersion, ok := hubVersions[sessionId]
|
||||
if ok {
|
||||
return hubVersion
|
||||
}
|
||||
// Extract hub version from SSH client version
|
||||
// getHubVersion extracts the hub version from the SSH client version string
|
||||
// for a given session. Returns a zero version if parsing fails.
|
||||
func (a *Agent) getHubVersion(sessionCtx ssh.Context) semver.Version {
|
||||
clientVersion := sessionCtx.Value(ssh.ContextKeyClientVersion)
|
||||
if versionStr, ok := clientVersion.(string); ok {
|
||||
hubVersion, _ = extractHubVersion(versionStr)
|
||||
hubVersion, _ := extractHubVersion(versionStr)
|
||||
return hubVersion
|
||||
}
|
||||
hubVersions[sessionId] = hubVersion
|
||||
return hubVersion
|
||||
return semver.Version{}
|
||||
}
|
||||
|
||||
// handleSession handles an incoming SSH session by gathering system statistics
|
||||
@@ -126,9 +115,8 @@ func (a *Agent) handleSession(s ssh.Session) {
|
||||
a.connectionManager.eventChan <- SSHConnect
|
||||
|
||||
sessionCtx := s.Context()
|
||||
sessionID := sessionCtx.SessionID()
|
||||
|
||||
hubVersion := a.getHubVersion(sessionID, sessionCtx)
|
||||
hubVersion := a.getHubVersion(sessionCtx)
|
||||
|
||||
// Legacy one-shot behavior for older hubs
|
||||
if hubVersion.LT(beszel.MinVersionAgentResponse) {
|
||||
@@ -192,7 +180,7 @@ func (a *Agent) handleSSHRequest(w io.Writer, req *common.HubRequest[cbor.RawMes
|
||||
|
||||
// handleLegacyStats serves the legacy one-shot stats payload for older hubs
|
||||
func (a *Agent) handleLegacyStats(w io.Writer, hubVersion semver.Version) error {
|
||||
stats := a.gatherStats(common.DataRequestOptions{CacheTimeMs: 60_000})
|
||||
stats := a.gatherStats(common.DataRequestOptions{CacheTimeMs: defaultDataCacheTimeMs})
|
||||
return a.writeToSession(w, stats, hubVersion)
|
||||
}
|
||||
|
||||
@@ -238,11 +226,11 @@ func ParseKeys(input string) ([]gossh.PublicKey, error) {
|
||||
// and finally defaults to ":45876".
|
||||
func GetAddress(addr string) string {
|
||||
if addr == "" {
|
||||
addr, _ = GetEnv("LISTEN")
|
||||
addr, _ = utils.GetEnv("LISTEN")
|
||||
}
|
||||
if addr == "" {
|
||||
// Legacy PORT environment variable support
|
||||
addr, _ = GetEnv("PORT")
|
||||
addr, _ = utils.GetEnv("PORT")
|
||||
}
|
||||
if addr == "" {
|
||||
return ":45876"
|
||||
@@ -258,7 +246,7 @@ func GetAddress(addr string) string {
|
||||
// It checks the NETWORK environment variable first, then infers from
|
||||
// the address format: addresses starting with "/" are "unix", others are "tcp".
|
||||
func GetNetwork(addr string) string {
|
||||
if network, ok := GetEnv("NETWORK"); ok && network != "" {
|
||||
if network, ok := utils.GetEnv("NETWORK"); ok && network != "" {
|
||||
return network
|
||||
}
|
||||
if strings.HasPrefix(addr, "/") {
|
||||
@@ -277,6 +265,5 @@ func (a *Agent) StopServer() error {
|
||||
slog.Info("Stopping SSH server")
|
||||
_ = a.server.Close()
|
||||
a.server = nil
|
||||
a.connectionManager.eventChan <- SSHDisconnect
|
||||
return nil
|
||||
}
|
||||
|
||||
+50
-48
@@ -1,5 +1,4 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
@@ -184,8 +183,7 @@ func TestStartServer(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestStartServerDisableSSH(t *testing.T) {
|
||||
os.Setenv("BESZEL_AGENT_DISABLE_SSH", "true")
|
||||
defer os.Unsetenv("BESZEL_AGENT_DISABLE_SSH")
|
||||
t.Setenv("BESZEL_AGENT_DISABLE_SSH", "true")
|
||||
|
||||
agent, err := NewAgent("")
|
||||
require.NoError(t, err)
|
||||
@@ -200,6 +198,28 @@ func TestStartServerDisableSSH(t *testing.T) {
|
||||
assert.Contains(t, err.Error(), "SSH disabled")
|
||||
}
|
||||
|
||||
func TestStopServerDoesNotBlockWhenEventQueueFull(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
agent.server = &ssh.Server{}
|
||||
agent.connectionManager.eventChan = make(chan ConnectionEvent, 1)
|
||||
agent.connectionManager.eventChan <- WebSocketConnect
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- agent.StopServer()
|
||||
}()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
require.NoError(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("StopServer blocked on the connection event queue")
|
||||
}
|
||||
|
||||
assert.Nil(t, agent.server)
|
||||
assert.Equal(t, WebSocketConnect, <-agent.connectionManager.eventChan)
|
||||
}
|
||||
|
||||
/////////////////////////////////////////////////////////////////
|
||||
//////////////////// ParseKeys Tests ////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////
|
||||
@@ -406,27 +426,23 @@ func TestGetHubVersion(t *testing.T) {
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
}
|
||||
|
||||
// Test first call - should extract and cache version
|
||||
version := agent.getHubVersion("test-session-123", mockCtx)
|
||||
// Test first call - should extract version
|
||||
version := agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.12.0", version.String())
|
||||
|
||||
// Test second call - should return cached version
|
||||
mockCtx.clientVersion = "SSH-2.0-beszel_0.11.0" // Change version but should still return cached
|
||||
version = agent.getHubVersion("test-session-123", mockCtx)
|
||||
assert.Equal(t, "0.12.0", version.String()) // Should still be cached version
|
||||
|
||||
// Test different session - should extract new version
|
||||
version = agent.getHubVersion("different-session", mockCtx)
|
||||
// Test that version reflects the current client version (no stale caching)
|
||||
mockCtx.clientVersion = "SSH-2.0-beszel_0.11.0"
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.11.0", version.String())
|
||||
|
||||
// Test with invalid version string (non-beszel client)
|
||||
mockCtx.clientVersion = "SSH-2.0-OpenSSH_8.0"
|
||||
version = agent.getHubVersion("invalid-session", mockCtx)
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.0.0", version.String()) // Should be empty version for non-beszel clients
|
||||
|
||||
// Test with no client version
|
||||
mockCtx.clientVersion = ""
|
||||
version = agent.getHubVersion("no-version-session", mockCtx)
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.True(t, version.EQ(semver.Version{})) // Should be empty version
|
||||
}
|
||||
|
||||
@@ -503,9 +519,6 @@ func TestWriteToSessionEncoding(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
// Reset the global hubVersions map to ensure clean state for each test
|
||||
hubVersions = nil
|
||||
|
||||
agent, err := NewAgent("")
|
||||
require.NoError(t, err)
|
||||
|
||||
@@ -587,39 +600,28 @@ func createTestCombinedData() *system.CombinedData {
|
||||
}
|
||||
}
|
||||
|
||||
func TestHubVersionCaching(t *testing.T) {
|
||||
// Reset the global hubVersions map to ensure clean state
|
||||
hubVersions = nil
|
||||
|
||||
// TestGetHubVersionConcurrent guards against a regression of the
|
||||
// "concurrent map writes" panic previously caused by a shared, unsynchronized
|
||||
// hubVersions cache (see https://github.com/henrygd/beszel/issues/2128).
|
||||
// getHubVersion no longer shares mutable state between sessions, so calling
|
||||
// it concurrently from many goroutines must be safe under `go test -race`.
|
||||
func TestGetHubVersionConcurrent(t *testing.T) {
|
||||
agent, err := NewAgent("")
|
||||
require.NoError(t, err)
|
||||
|
||||
ctx1 := &mockSSHContext{
|
||||
sessionID: "session1",
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
const goroutines = 50
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(goroutines)
|
||||
for i := 0; i < goroutines; i++ {
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
ctx := &mockSSHContext{
|
||||
sessionID: fmt.Sprintf("session-%d", i),
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
}
|
||||
version := agent.getHubVersion(ctx)
|
||||
assert.Equal(t, "0.12.0", version.String())
|
||||
}(i)
|
||||
}
|
||||
ctx2 := &mockSSHContext{
|
||||
sessionID: "session2",
|
||||
clientVersion: "SSH-2.0-beszel_0.11.0",
|
||||
}
|
||||
|
||||
// First calls should cache the versions
|
||||
v1 := agent.getHubVersion("session1", ctx1)
|
||||
v2 := agent.getHubVersion("session2", ctx2)
|
||||
|
||||
assert.Equal(t, "0.12.0", v1.String())
|
||||
assert.Equal(t, "0.11.0", v2.String())
|
||||
|
||||
// Verify caching by changing context but keeping same session ID
|
||||
ctx1.clientVersion = "SSH-2.0-beszel_0.10.0"
|
||||
v1Cached := agent.getHubVersion("session1", ctx1)
|
||||
assert.Equal(t, "0.12.0", v1Cached.String()) // Should still be cached version
|
||||
|
||||
// New session should get new version
|
||||
ctx3 := &mockSSHContext{
|
||||
sessionID: "session3",
|
||||
clientVersion: "SSH-2.0-beszel_0.13.0",
|
||||
}
|
||||
v3 := agent.getHubVersion("session3", ctx3)
|
||||
assert.Equal(t, "0.13.0", v3.String())
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
+188
-76
@@ -18,18 +18,22 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
)
|
||||
|
||||
// SmartManager manages data collection for SMART devices
|
||||
type SmartManager struct {
|
||||
sync.Mutex
|
||||
SmartDataMap map[string]*smart.SmartData
|
||||
SmartDevices []*DeviceInfo
|
||||
refreshMutex sync.Mutex
|
||||
lastScanTime time.Time
|
||||
smartctlPath string
|
||||
excludedDevices map[string]struct{}
|
||||
SmartDataMap map[string]*smart.SmartData
|
||||
SmartDevices []*DeviceInfo
|
||||
refreshMutex sync.Mutex
|
||||
lastScanTime time.Time
|
||||
smartctlPath string
|
||||
excludedDevices map[string]struct{}
|
||||
darwinNvmeOnce sync.Once
|
||||
darwinNvmeCapacity map[string]uint64 // serial → bytes cache, written once via darwinNvmeOnce
|
||||
darwinNvmeProvider func() ([]byte, error) // overridable for testing
|
||||
}
|
||||
|
||||
type scanOutput struct {
|
||||
@@ -51,6 +55,11 @@ type DeviceInfo struct {
|
||||
typeVerified bool
|
||||
// parserType holds the parser type (nvme, sat, scsi) that last succeeded.
|
||||
parserType string
|
||||
// explicitType reports whether Type came from an explicit ":type" hint in
|
||||
// SMART_DEVICES. Such a type is a deliberate user override and must always be
|
||||
// passed to smartctl via -d, even for scsi/ata where a scan-detected type is
|
||||
// otherwise left off (see smartctlArgs and issue #1345).
|
||||
explicitType bool
|
||||
}
|
||||
|
||||
// deviceKey is a composite key for a device, used to identify a device uniquely.
|
||||
@@ -61,8 +70,9 @@ type deviceKey struct {
|
||||
|
||||
var errNoValidSmartData = fmt.Errorf("no valid SMART data found") // Error for missing data
|
||||
|
||||
// Refresh updates SMART data for all known devices
|
||||
func (sm *SmartManager) Refresh(forceScan bool) error {
|
||||
// Refresh updates SMART data for all known devices and reports whether every
|
||||
// discovered device was collected successfully.
|
||||
func (sm *SmartManager) Refresh(forceScan bool) (bool, error) {
|
||||
sm.refreshMutex.Lock()
|
||||
defer sm.refreshMutex.Unlock()
|
||||
|
||||
@@ -83,7 +93,7 @@ func (sm *SmartManager) Refresh(forceScan bool) error {
|
||||
}
|
||||
}
|
||||
|
||||
return sm.resolveRefreshError(scanErr, collectErr)
|
||||
return scanErr == nil && collectErr == nil, sm.resolveRefreshError(scanErr, collectErr)
|
||||
}
|
||||
|
||||
// devicesSnapshot returns a copy of the current device slice to avoid iterating
|
||||
@@ -156,7 +166,7 @@ func (sm *SmartManager) ScanDevices(force bool) error {
|
||||
currentDevices := sm.devicesSnapshot()
|
||||
|
||||
var configuredDevices []*DeviceInfo
|
||||
if configuredRaw, ok := GetEnv("SMART_DEVICES"); ok {
|
||||
if configuredRaw, ok := utils.GetEnv("SMART_DEVICES"); ok {
|
||||
slog.Info("SMART_DEVICES", "value", configuredRaw)
|
||||
config := strings.TrimSpace(configuredRaw)
|
||||
if config == "" {
|
||||
@@ -199,6 +209,13 @@ func (sm *SmartManager) ScanDevices(force bool) error {
|
||||
hasValidScan = true
|
||||
}
|
||||
|
||||
// Add Linux mdraid arrays by reading sysfs health fields. This does not
|
||||
// require smartctl and does not scan the whole device.
|
||||
if raidDevices := scanMdraidDevices(); len(raidDevices) > 0 {
|
||||
scannedDevices = append(scannedDevices, raidDevices...)
|
||||
hasValidScan = true
|
||||
}
|
||||
|
||||
finalDevices := mergeDeviceLists(currentDevices, scannedDevices, configuredDevices)
|
||||
finalDevices = sm.filterExcludedDevices(finalDevices)
|
||||
sm.updateSmartDevices(finalDevices)
|
||||
@@ -215,7 +232,7 @@ func (sm *SmartManager) ScanDevices(force bool) error {
|
||||
}
|
||||
|
||||
func (sm *SmartManager) parseConfiguredDevices(config string) ([]*DeviceInfo, error) {
|
||||
splitChar := os.Getenv("SMART_DEVICES_SEPARATOR")
|
||||
splitChar, _ := utils.GetEnv("SMART_DEVICES_SEPARATOR")
|
||||
if splitChar == "" {
|
||||
splitChar = ","
|
||||
}
|
||||
@@ -240,8 +257,9 @@ func (sm *SmartManager) parseConfiguredDevices(config string) ([]*DeviceInfo, er
|
||||
}
|
||||
|
||||
devices = append(devices, &DeviceInfo{
|
||||
Name: name,
|
||||
Type: devType,
|
||||
Name: name,
|
||||
Type: devType,
|
||||
explicitType: devType != "",
|
||||
})
|
||||
}
|
||||
|
||||
@@ -253,7 +271,7 @@ func (sm *SmartManager) parseConfiguredDevices(config string) ([]*DeviceInfo, er
|
||||
}
|
||||
|
||||
func (sm *SmartManager) refreshExcludedDevices() {
|
||||
rawValue, _ := GetEnv("EXCLUDE_SMART")
|
||||
rawValue, _ := utils.GetEnv("EXCLUDE_SMART")
|
||||
sm.excludedDevices = make(map[string]struct{})
|
||||
|
||||
for entry := range strings.SplitSeq(rawValue, ",") {
|
||||
@@ -357,9 +375,15 @@ func (sm *SmartManager) parseSmartOutput(deviceInfo *DeviceInfo, output []byte)
|
||||
Type string
|
||||
Parse func([]byte) (bool, int)
|
||||
}{
|
||||
{Type: "nvme", Parse: sm.parseSmartForNvme},
|
||||
{Type: "sat", Parse: sm.parseSmartForSata},
|
||||
{Type: "scsi", Parse: sm.parseSmartForScsi},
|
||||
{Type: "nvme", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForNvme(output, deviceInfo.Type)
|
||||
}},
|
||||
{Type: "sat", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForSata(output, deviceInfo.Type)
|
||||
}},
|
||||
{Type: "scsi", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForScsi(output, deviceInfo.Type)
|
||||
}},
|
||||
}
|
||||
|
||||
deviceType := normalizeParserType(deviceInfo.parserType)
|
||||
@@ -450,6 +474,12 @@ func (sm *SmartManager) CollectSmart(deviceInfo *DeviceInfo) error {
|
||||
return errNoValidSmartData
|
||||
}
|
||||
|
||||
// mdraid health is not exposed via SMART; Linux exposes array state in sysfs.
|
||||
if deviceInfo != nil {
|
||||
if ok, err := sm.collectMdraidHealth(deviceInfo); ok {
|
||||
return err
|
||||
}
|
||||
}
|
||||
// eMMC health is not exposed via SMART on Linux, but the kernel provides
|
||||
// wear / EOL indicators via sysfs. Prefer that path when available.
|
||||
if deviceInfo != nil {
|
||||
@@ -462,10 +492,11 @@ func (sm *SmartManager) CollectSmart(deviceInfo *DeviceInfo) error {
|
||||
return errNoValidSmartData
|
||||
}
|
||||
|
||||
// slog.Info("collecting SMART data", "device", deviceInfo.Name, "type", deviceInfo.Type, "has_existing_data", sm.hasDataForDevice(deviceInfo.Name))
|
||||
// slog.Info("collecting SMART data", "device", deviceInfo.Name, "type", deviceInfo.Type, "has_existing_data", sm.hasDataForDevice(deviceInfo))
|
||||
|
||||
// Check if we have any existing data for this device
|
||||
hasExistingData := sm.hasDataForDevice(deviceInfo.Name)
|
||||
// Check if we have existing data for this exact device identity. Multiple
|
||||
// bridge slots can share a path, so a name-only match is not sufficient.
|
||||
hasExistingData := sm.hasDataForDevice(deviceInfo)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
@@ -476,7 +507,7 @@ func (sm *SmartManager) CollectSmart(deviceInfo *DeviceInfo) error {
|
||||
output, err := cmd.CombinedOutput()
|
||||
|
||||
// Check if device is in standby (exit status 2)
|
||||
if exitErr, ok := err.(*exec.ExitError); ok && exitErr.ExitCode() == 2 {
|
||||
if exitErr, ok := errors.AsType[*exec.ExitError](err); ok && exitErr.ExitCode() == 2 {
|
||||
if hasExistingData {
|
||||
// Device is in standby and we have cached data, keep using cache
|
||||
return nil
|
||||
@@ -541,7 +572,9 @@ func (sm *SmartManager) smartctlArgs(deviceInfo *DeviceInfo, includeStandby bool
|
||||
deviceType = strings.ToLower(deviceInfo.Type)
|
||||
parserType = strings.ToLower(deviceInfo.parserType)
|
||||
// types sometimes misidentified in scan; see github.com/henrygd/beszel/issues/1345
|
||||
if deviceType != "" && deviceType != "scsi" && deviceType != "ata" {
|
||||
// An explicit SMART_DEVICES ":type" hint is a deliberate override, so always
|
||||
// pass it through; otherwise scsi/ata are left off so smartctl can auto-detect.
|
||||
if deviceType != "" && (deviceInfo.explicitType || (deviceType != "scsi" && deviceType != "ata")) {
|
||||
args = append(args, "-d", deviceInfo.Type)
|
||||
}
|
||||
}
|
||||
@@ -566,14 +599,18 @@ func (sm *SmartManager) smartctlArgs(deviceInfo *DeviceInfo, includeStandby bool
|
||||
return args
|
||||
}
|
||||
|
||||
// hasDataForDevice checks if we have cached SMART data for a specific device
|
||||
func (sm *SmartManager) hasDataForDevice(deviceName string) bool {
|
||||
// hasDataForDevice checks if we have cached SMART data for a specific device identity.
|
||||
func (sm *SmartManager) hasDataForDevice(deviceInfo *DeviceInfo) bool {
|
||||
if deviceInfo == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
sm.Lock()
|
||||
defer sm.Unlock()
|
||||
|
||||
// Check if any cached data has this device name
|
||||
deviceKey := makeDeviceKey(deviceInfo.Name, deviceInfo.Type)
|
||||
for _, data := range sm.SmartDataMap {
|
||||
if data != nil && data.DiskName == deviceName {
|
||||
if data != nil && makeDeviceKey(data.DiskName, data.DiskType) == deviceKey {
|
||||
return true
|
||||
}
|
||||
}
|
||||
@@ -646,6 +683,9 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
target.Type = prev.Type
|
||||
target.typeVerified = true
|
||||
target.parserType = prev.parserType
|
||||
if prev.explicitType {
|
||||
target.explicitType = true
|
||||
}
|
||||
}
|
||||
|
||||
// applyConfiguredMetadata updates a matched device with any configured
|
||||
@@ -659,6 +699,9 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
existingDev.typeVerified = false
|
||||
existingDev.parserType = normalizeParserType(newType)
|
||||
}
|
||||
if configuredDev.explicitType {
|
||||
existingDev.explicitType = true
|
||||
}
|
||||
if configuredDev.InfoName != "" {
|
||||
existingDev.InfoName = configuredDev.InfoName
|
||||
}
|
||||
@@ -715,7 +758,14 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
continue
|
||||
}
|
||||
if existingDev := deviceIndexByName[configuredDevice.Name]; existingDev != nil {
|
||||
oldKey := makeDeviceKey(existingDev.Name, existingDev.Type)
|
||||
if prev := existingIndex[key]; prev != nil {
|
||||
preserveVerifiedType(existingDev, prev)
|
||||
}
|
||||
applyConfiguredMetadata(existingDev, configuredDevice)
|
||||
delete(deviceIndex, oldKey)
|
||||
deviceIndex[makeDeviceKey(existingDev.Name, existingDev.Type)] = existingDev
|
||||
delete(deviceIndexByName, configuredDevice.Name)
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -819,9 +869,11 @@ func (sm *SmartManager) isVirtualDeviceFromStrings(fields ...string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// parseSmartForSata parses the output of smartctl --all -j for SATA/ATA devices and updates the SmartDataMap
|
||||
// parseSmartForSata parses the output of smartctl --all -j for SATA/ATA devices and updates the SmartDataMap.
|
||||
// deviceType is the exact type used to identify and query the device; when set,
|
||||
// it takes precedence over the generic type reported by smartctl.
|
||||
// Returns hasValidData and exitStatus
|
||||
func (sm *SmartManager) parseSmartForSata(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForSata(output []byte, deviceType string) (bool, int) {
|
||||
var data smart.SmartInfoForSata
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -857,14 +909,20 @@ func (sm *SmartManager) parseSmartForSata(output []byte) (bool, int) {
|
||||
smartData.FirmwareVersion = data.FirmwareVersion
|
||||
smartData.Capacity = data.UserCapacity.Bytes
|
||||
smartData.Temperature = data.Temperature.Current
|
||||
if smartData.Temperature == 0 {
|
||||
if temp, ok := temperatureFromAtaDeviceStatistics(data.AtaDeviceStatistics); ok {
|
||||
smartData.Temperature = temp
|
||||
}
|
||||
}
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
// get values from ata_device_statistics if necessary
|
||||
var ataDeviceStats smart.AtaDeviceStatistics
|
||||
if smartData.Temperature == 0 {
|
||||
if temp := findAtaDeviceStatisticsValue(&data, &ataDeviceStats, 5, "Current Temperature", 0, 255); temp != nil {
|
||||
smartData.Temperature = uint8(*temp)
|
||||
}
|
||||
}
|
||||
|
||||
// update SmartAttributes
|
||||
smartData.Attributes = make([]*smart.SmartAttribute, 0, len(data.AtaSmartAttributes.Table))
|
||||
@@ -873,6 +931,9 @@ func (sm *SmartManager) parseSmartForSata(output []byte) (bool, int) {
|
||||
if parsed, ok := smart.ParseSmartRawValueString(attr.Raw.String); ok {
|
||||
rawValue = parsed
|
||||
}
|
||||
if smartData.SmartStatus == "PASSED" && rawValue > 0 && (attr.ID == 5 || attr.ID == 197 || attr.ID == 198) {
|
||||
smartData.SmartStatus = "WARNING"
|
||||
}
|
||||
smartAttr := &smart.SmartAttribute{
|
||||
ID: attr.ID,
|
||||
Name: attr.Name,
|
||||
@@ -900,23 +961,20 @@ func getSmartStatus(temperature uint8, passed bool) string {
|
||||
}
|
||||
}
|
||||
|
||||
func temperatureFromAtaDeviceStatistics(stats smart.AtaDeviceStatistics) (uint8, bool) {
|
||||
entry := findAtaDeviceStatisticsEntry(stats, 5, "Current Temperature")
|
||||
if entry == nil || entry.Value == nil {
|
||||
return 0, false
|
||||
}
|
||||
if *entry.Value > 255 {
|
||||
return 0, false
|
||||
}
|
||||
return uint8(*entry.Value), true
|
||||
}
|
||||
|
||||
// findAtaDeviceStatisticsEntry centralizes ATA devstat lookups so additional
|
||||
// metrics can be pulled from the same structure in the future.
|
||||
func findAtaDeviceStatisticsEntry(stats smart.AtaDeviceStatistics, pageNumber uint8, entryName string) *smart.AtaDeviceStatisticsEntry {
|
||||
for pageIdx := range stats.Pages {
|
||||
page := &stats.Pages[pageIdx]
|
||||
if page.Number != pageNumber {
|
||||
func findAtaDeviceStatisticsValue(data *smart.SmartInfoForSata, ataDeviceStats *smart.AtaDeviceStatistics, entryNumber uint8, entryName string, minValue, maxValue int64) *int64 {
|
||||
if len(ataDeviceStats.Pages) == 0 {
|
||||
if len(data.AtaDeviceStatistics) == 0 {
|
||||
return nil
|
||||
}
|
||||
if err := json.Unmarshal(data.AtaDeviceStatistics, ataDeviceStats); err != nil {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
for pageIdx := range ataDeviceStats.Pages {
|
||||
page := &ataDeviceStats.Pages[pageIdx]
|
||||
if page.Number != entryNumber {
|
||||
continue
|
||||
}
|
||||
for entryIdx := range page.Table {
|
||||
@@ -924,13 +982,16 @@ func findAtaDeviceStatisticsEntry(stats smart.AtaDeviceStatistics, pageNumber ui
|
||||
if !strings.EqualFold(entry.Name, entryName) {
|
||||
continue
|
||||
}
|
||||
return entry
|
||||
if entry.Value == nil || *entry.Value < minValue || *entry.Value > maxValue {
|
||||
return nil
|
||||
}
|
||||
return entry.Value
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (sm *SmartManager) parseSmartForScsi(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForScsi(output []byte, deviceType string) (bool, int) {
|
||||
var data smart.SmartInfoForScsi
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -965,6 +1026,9 @@ func (sm *SmartManager) parseSmartForScsi(output []byte) (bool, int) {
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
attributes := make([]*smart.SmartAttribute, 0, 10)
|
||||
attributes = append(attributes, &smart.SmartAttribute{Name: "PowerOnHours", RawValue: data.PowerOnTime.Hours})
|
||||
@@ -1016,9 +1080,57 @@ func parseScsiGigabytesProcessed(value string) int64 {
|
||||
return parsed
|
||||
}
|
||||
|
||||
// parseSmartForNvme parses the output of smartctl --all -j /dev/nvmeX and updates the SmartDataMap
|
||||
// lookupDarwinNvmeCapacity returns the capacity in bytes for a given NVMe serial number on Darwin.
|
||||
// It uses system_profiler SPNVMeDataType to get capacity since Apple SSDs don't report user_capacity
|
||||
// via smartctl. Results are cached after the first call via sync.Once.
|
||||
func (sm *SmartManager) lookupDarwinNvmeCapacity(serial string) uint64 {
|
||||
sm.darwinNvmeOnce.Do(func() {
|
||||
sm.darwinNvmeCapacity = make(map[string]uint64)
|
||||
|
||||
provider := sm.darwinNvmeProvider
|
||||
if provider == nil {
|
||||
provider = func() ([]byte, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
return exec.CommandContext(ctx, "system_profiler", "SPNVMeDataType", "-json").Output()
|
||||
}
|
||||
}
|
||||
|
||||
out, err := provider()
|
||||
if err != nil {
|
||||
slog.Debug("system_profiler NVMe lookup failed", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
var result struct {
|
||||
SPNVMeDataType []struct {
|
||||
Items []struct {
|
||||
DeviceSerial string `json:"device_serial"`
|
||||
SizeInBytes uint64 `json:"size_in_bytes"`
|
||||
} `json:"_items"`
|
||||
} `json:"SPNVMeDataType"`
|
||||
}
|
||||
if err := json.Unmarshal(out, &result); err != nil {
|
||||
slog.Debug("system_profiler NVMe parse failed", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
for _, controller := range result.SPNVMeDataType {
|
||||
for _, item := range controller.Items {
|
||||
if item.DeviceSerial != "" && item.SizeInBytes > 0 {
|
||||
sm.darwinNvmeCapacity[item.DeviceSerial] = item.SizeInBytes
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
return sm.darwinNvmeCapacity[serial]
|
||||
}
|
||||
|
||||
// parseSmartForNvme parses the output of smartctl --all -j /dev/nvmeX and updates the SmartDataMap.
|
||||
// deviceType is the exact type used to identify and query the device; when set,
|
||||
// it takes precedence over the generic type reported by smartctl.
|
||||
// Returns hasValidData and exitStatus
|
||||
func (sm *SmartManager) parseSmartForNvme(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForNvme(output []byte, deviceType string) (bool, int) {
|
||||
data := &smart.SmartInfoForNvme{}
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -1052,10 +1164,19 @@ func (sm *SmartManager) parseSmartForNvme(output []byte) (bool, int) {
|
||||
smartData.SerialNumber = data.SerialNumber
|
||||
smartData.FirmwareVersion = data.FirmwareVersion
|
||||
smartData.Capacity = data.UserCapacity.Bytes
|
||||
if smartData.Capacity == 0 {
|
||||
smartData.Capacity = data.NVMeTotalCapacity
|
||||
}
|
||||
if smartData.Capacity == 0 && (runtime.GOOS == "darwin" || sm.darwinNvmeProvider != nil) {
|
||||
smartData.Capacity = sm.lookupDarwinNvmeCapacity(data.SerialNumber)
|
||||
}
|
||||
smartData.Temperature = data.NVMeSmartHealthInformationLog.Temperature
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
// nvme attributes does not follow the same format as ata attributes,
|
||||
// so we manually map each field to SmartAttributes
|
||||
@@ -1087,32 +1208,21 @@ func (sm *SmartManager) parseSmartForNvme(output []byte) (bool, int) {
|
||||
|
||||
// detectSmartctl checks if smartctl is installed, returns an error if not
|
||||
func (sm *SmartManager) detectSmartctl() (string, error) {
|
||||
isWindows := runtime.GOOS == "windows"
|
||||
|
||||
// Load embedded smartctl.exe for Windows amd64 builds.
|
||||
if isWindows && runtime.GOARCH == "amd64" {
|
||||
if path, err := ensureEmbeddedSmartctl(); err == nil {
|
||||
return path, nil
|
||||
if runtime.GOOS == "windows" {
|
||||
// Load embedded smartctl.exe for Windows amd64 builds.
|
||||
if runtime.GOARCH == "amd64" {
|
||||
if path, err := ensureEmbeddedSmartctl(); err == nil {
|
||||
return path, nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if path, err := exec.LookPath("smartctl"); err == nil {
|
||||
return path, nil
|
||||
}
|
||||
locations := []string{}
|
||||
if isWindows {
|
||||
locations = append(locations,
|
||||
"C:\\Program Files\\smartmontools\\bin\\smartctl.exe",
|
||||
)
|
||||
} else {
|
||||
locations = append(locations, "/opt/homebrew/bin/smartctl")
|
||||
}
|
||||
for _, location := range locations {
|
||||
// Try to find smartctl in the default installation location
|
||||
const location = "C:\\Program Files\\smartmontools\\bin\\smartctl.exe"
|
||||
if _, err := os.Stat(location); err == nil {
|
||||
return location, nil
|
||||
}
|
||||
}
|
||||
return "", errors.New("smartctl not found")
|
||||
|
||||
return utils.LookPathHomebrew("smartctl")
|
||||
}
|
||||
|
||||
// isNvmeControllerPath checks if the path matches an NVMe controller pattern
|
||||
@@ -1146,9 +1256,11 @@ func NewSmartManager() (*SmartManager, error) {
|
||||
slog.Debug("smartctl", "path", path, "err", err)
|
||||
if err != nil {
|
||||
// Keep the previous fail-fast behavior unless this Linux host exposes
|
||||
// eMMC health via sysfs, in which case smartctl is optional.
|
||||
if runtime.GOOS == "linux" && len(scanEmmcDevices()) > 0 {
|
||||
return sm, nil
|
||||
// eMMC or mdraid health via sysfs, in which case smartctl is optional.
|
||||
if runtime.GOOS == "linux" {
|
||||
if len(scanEmmcDevices()) > 0 || len(scanMdraidDevices()) > 0 {
|
||||
return sm, nil
|
||||
}
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
|
||||
+542
-10
@@ -1,12 +1,13 @@
|
||||
//go:build testing
|
||||
// +build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
@@ -25,7 +26,7 @@ func TestParseSmartForScsi(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForScsi(data)
|
||||
hasData, exitStatus := sm.parseSmartForScsi(data, "")
|
||||
if !hasData {
|
||||
t.Fatalf("expected SCSI data to parse successfully")
|
||||
}
|
||||
@@ -70,7 +71,7 @@ func TestParseSmartForSata(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForSata(data)
|
||||
hasData, exitStatus := sm.parseSmartForSata(data, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 64, exitStatus)
|
||||
|
||||
@@ -89,6 +90,52 @@ func TestParseSmartForSata(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForSataWarnsForCriticalAttributes(t *testing.T) {
|
||||
for _, attrID := range []int{5, 197, 198} {
|
||||
t.Run("attribute "+strconv.Itoa(attrID), func(t *testing.T) {
|
||||
jsonPayload := []byte(fmt.Sprintf(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
"device": {"name": "/dev/sda", "type": "sat"},
|
||||
"model_name": "Example",
|
||||
"serial_number": "WARNING%d",
|
||||
"smart_status": {"passed": true},
|
||||
"temperature": {"current": 30},
|
||||
"ata_smart_attributes": {"table": [{"id": %d, "raw": {"value": 1, "string": "1"}}]}
|
||||
}`, attrID, attrID))
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, "WARNING", sm.SmartDataMap[fmt.Sprintf("WARNING%d", attrID)].SmartStatus)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForSataPreservesFailedAndUnknownStatus(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
temperature int
|
||||
want string
|
||||
}{
|
||||
{name: "failed", temperature: 30, want: "FAILED"},
|
||||
{name: "unknown", want: "UNKNOWN"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
jsonPayload := []byte(fmt.Sprintf(`{
|
||||
"device": {"name": "/dev/sda", "type": "sat"},
|
||||
"serial_number": "PRESERVE%s",
|
||||
"temperature": {"current": %d},
|
||||
"ata_smart_attributes": {"table": [{"id": 197, "raw": {"value": 1, "string": "1"}}]}
|
||||
}`, test.name, test.temperature))
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, test.want, sm.SmartDataMap["PRESERVE"+test.name].SmartStatus)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
|
||||
jsonPayload := []byte(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
@@ -113,7 +160,7 @@ func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -122,6 +169,78 @@ func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
|
||||
assert.Equal(t, uint8(22), deviceData.Temperature)
|
||||
}
|
||||
|
||||
func TestParseSmartForSataAtaDeviceStatistics(t *testing.T) {
|
||||
// tests that ata_device_statistics values are parsed correctly
|
||||
jsonPayload := []byte(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
"device": {"name": "/dev/sdb", "type": "sat"},
|
||||
"model_name": "SanDisk SSD U110 16GB",
|
||||
"serial_number": "lksjfh23lhj",
|
||||
"firmware_version": "U21B001",
|
||||
"user_capacity": {"bytes": 16013942784},
|
||||
"smart_status": {"passed": true},
|
||||
"ata_smart_attributes": {"table": []},
|
||||
"ata_device_statistics": {
|
||||
"pages": [
|
||||
{
|
||||
"number": 5,
|
||||
"name": "Temperature Statistics",
|
||||
"table": [
|
||||
{"name": "Current Temperature", "value": 43, "flags": {"valid": true}},
|
||||
{"name": "Specified Minimum Operating Temperature", "value": -20, "flags": {"valid": true}}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
deviceData, ok := sm.SmartDataMap["lksjfh23lhj"]
|
||||
require.True(t, ok, "expected smart data entry for serial lksjfh23lhj")
|
||||
assert.Equal(t, uint8(43), deviceData.Temperature)
|
||||
}
|
||||
|
||||
func TestParseSmartForSataNegativeDeviceStatistics(t *testing.T) {
|
||||
// Tests that negative values in ata_device_statistics (e.g. min operating temp)
|
||||
// do not cause the entire SAT parser to fail.
|
||||
jsonPayload := []byte(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
"device": {"name": "/dev/sdb", "type": "sat"},
|
||||
"model_name": "SanDisk SSD U110 16GB",
|
||||
"serial_number": "NEGATIVE123",
|
||||
"firmware_version": "U21B001",
|
||||
"user_capacity": {"bytes": 16013942784},
|
||||
"smart_status": {"passed": true},
|
||||
"temperature": {"current": 38},
|
||||
"ata_smart_attributes": {"table": []},
|
||||
"ata_device_statistics": {
|
||||
"pages": [
|
||||
{
|
||||
"number": 5,
|
||||
"name": "Temperature Statistics",
|
||||
"table": [
|
||||
{"name": "Current Temperature", "value": 38, "flags": {"valid": true}},
|
||||
{"name": "Specified Minimum Operating Temperature", "value": -20, "flags": {"valid": true}}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
deviceData, ok := sm.SmartDataMap["NEGATIVE123"]
|
||||
require.True(t, ok, "expected smart data entry for serial NEGATIVE123")
|
||||
assert.Equal(t, uint8(38), deviceData.Temperature)
|
||||
}
|
||||
|
||||
func TestParseSmartForSataParentheticalRawValue(t *testing.T) {
|
||||
jsonPayload := []byte(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
@@ -152,7 +271,7 @@ func TestParseSmartForSataParentheticalRawValue(t *testing.T) {
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -174,7 +293,7 @@ func TestParseSmartForNvme(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForNvme(data)
|
||||
hasData, exitStatus := sm.parseSmartForNvme(data, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -197,13 +316,15 @@ func TestParseSmartForNvme(t *testing.T) {
|
||||
func TestHasDataForDevice(t *testing.T) {
|
||||
sm := &SmartManager{
|
||||
SmartDataMap: map[string]*smart.SmartData{
|
||||
"serial-1": {DiskName: "/dev/sda"},
|
||||
"serial-1": {DiskName: "/dev/sda", DiskType: "jms56x,0"},
|
||||
"serial-2": nil,
|
||||
},
|
||||
}
|
||||
|
||||
assert.True(t, sm.hasDataForDevice("/dev/sda"))
|
||||
assert.False(t, sm.hasDataForDevice("/dev/sdb"))
|
||||
assert.True(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sda", Type: "jms56x,0"}))
|
||||
assert.False(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sda", Type: "jms56x,1"}))
|
||||
assert.False(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sdb", Type: "jms56x,0"}))
|
||||
assert.False(t, sm.hasDataForDevice(nil))
|
||||
}
|
||||
|
||||
func TestDevicesSnapshotReturnsCopy(t *testing.T) {
|
||||
@@ -321,6 +442,81 @@ func TestSmartctlArgs(t *testing.T) {
|
||||
)
|
||||
}
|
||||
|
||||
// TestSmartctlArgsExplicitType verifies that an explicit SMART_DEVICES type hint
|
||||
// is always passed to smartctl via -d, while a scan-detected scsi/ata type is
|
||||
// still left off so smartctl can auto-detect it (see issue #1345).
|
||||
func TestSmartctlArgsExplicitType(t *testing.T) {
|
||||
sm := &SmartManager{}
|
||||
|
||||
// Scan-detected scsi: -d is intentionally omitted.
|
||||
scanScsi := &DeviceInfo{Name: "/dev/sda", Type: "scsi"}
|
||||
assert.Equal(t,
|
||||
[]string{"-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(scanScsi, false),
|
||||
)
|
||||
|
||||
// Explicit scsi from SMART_DEVICES: -d scsi must be passed.
|
||||
explicitScsi := &DeviceInfo{Name: "/dev/sda", Type: "scsi", explicitType: true}
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "scsi", "-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(explicitScsi, false),
|
||||
)
|
||||
|
||||
// Explicit ata from SMART_DEVICES: -d ata must be passed (devstat still added).
|
||||
explicitAta := &DeviceInfo{Name: "/dev/sdb", Type: "ata", explicitType: true}
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "ata", "-a", "--json=c", "-l", "devstat", "/dev/sdb"},
|
||||
sm.smartctlArgs(explicitAta, false),
|
||||
)
|
||||
}
|
||||
|
||||
// TestSmartDevicesExplicitTypeFlowsToSmartctlArgs is a regression test for
|
||||
// issue #2072: an explicit SMART_DEVICES type (e.g. /dev/sda:scsi) must win over
|
||||
// a wrong scan-detected type (sat) and be handed to smartctl as -d scsi.
|
||||
func TestSmartDevicesExplicitTypeFlowsToSmartctlArgs(t *testing.T) {
|
||||
sm := &SmartManager{}
|
||||
|
||||
configured, err := sm.parseConfiguredDevices("/dev/sda:scsi")
|
||||
require.NoError(t, err)
|
||||
require.Len(t, configured, 1)
|
||||
assert.True(t, configured[0].explicitType)
|
||||
|
||||
// smartctl --scan misreports this USB drive as sat, which fails on it.
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat", Protocol: "ATA"},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 1)
|
||||
|
||||
device := merged[0]
|
||||
assert.Equal(t, "scsi", device.Type, "configured type should win over scan-detected sat")
|
||||
assert.True(t, device.explicitType, "explicit hint must survive the merge")
|
||||
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "scsi", "-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(device, false),
|
||||
"explicit scsi type must be passed to smartctl, not dropped",
|
||||
)
|
||||
}
|
||||
|
||||
// TestMergeDeviceListsPreservesExplicitTypeAcrossRescan ensures a verified,
|
||||
// explicitly-typed device keeps its explicit flag when a later scan re-reports
|
||||
// it with a different auto-detected type.
|
||||
func TestMergeDeviceListsPreservesExplicitTypeAcrossRescan(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "scsi", parserType: "scsi", typeVerified: true, explicitType: true},
|
||||
}
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat"},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(existing, scanned, nil)
|
||||
require.Len(t, merged, 1)
|
||||
assert.Equal(t, "scsi", merged[0].Type)
|
||||
assert.True(t, merged[0].explicitType, "explicit type flag should survive a rescan")
|
||||
}
|
||||
|
||||
func TestResolveRefreshError(t *testing.T) {
|
||||
scanErr := errors.New("scan failed")
|
||||
collectErr := errors.New("collect failed")
|
||||
@@ -463,6 +659,74 @@ func TestMergeDeviceListsPrefersConfigured(t *testing.T) {
|
||||
assert.Equal(t, "sat", byName["/dev/sdb"].Type)
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsExpandsConfiguredDevicesWithSamePath(t *testing.T) {
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "sat", InfoName: "scan-info", Protocol: "ATA"},
|
||||
}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 2)
|
||||
|
||||
byKey := make(map[deviceKey]*DeviceInfo, len(merged))
|
||||
for _, device := range merged {
|
||||
byKey[makeDeviceKey(device.Name, device.Type)] = device
|
||||
}
|
||||
|
||||
first := byKey[makeDeviceKey("/dev/sdb", "jms56x,0")]
|
||||
require.NotNil(t, first)
|
||||
assert.Equal(t, "scan-info", first.InfoName)
|
||||
assert.Equal(t, "ATA", first.Protocol)
|
||||
assert.True(t, first.explicitType)
|
||||
|
||||
second := byKey[makeDeviceKey("/dev/sdb", "jms56x,1")]
|
||||
require.NotNil(t, second)
|
||||
assert.True(t, second.explicitType)
|
||||
assert.NotContains(t, byKey, makeDeviceKey("/dev/sdb", "sat"))
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsPreservesSamePathVerificationAcrossRescan(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", parserType: "sat", typeVerified: true, explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", parserType: "sat", typeVerified: true, explicitType: true},
|
||||
}
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "sat", Protocol: "ATA"},
|
||||
}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(existing, scanned, configured)
|
||||
require.Len(t, merged, 2)
|
||||
byKey := make(map[deviceKey]*DeviceInfo, len(merged))
|
||||
for _, device := range merged {
|
||||
byKey[makeDeviceKey(device.Name, device.Type)] = device
|
||||
assert.True(t, device.typeVerified, device.Type)
|
||||
assert.Equal(t, "sat", device.parserType, device.Type)
|
||||
assert.True(t, device.explicitType, device.Type)
|
||||
}
|
||||
assert.Contains(t, byKey, makeDeviceKey("/dev/sdb", "jms56x,0"))
|
||||
assert.Contains(t, byKey, makeDeviceKey("/dev/sdb", "jms56x,1"))
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsDeduplicatesConfiguredIdentityAfterRekey(t *testing.T) {
|
||||
scanned := []*DeviceInfo{{Name: "/dev/sdb", Type: "sat"}}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 1)
|
||||
assert.Equal(t, "/dev/sdb", merged[0].Name)
|
||||
assert.Equal(t, "jms56x,0", merged[0].Type)
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsPreservesVerification(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat+megaraid", parserType: "sat", typeVerified: true},
|
||||
@@ -607,6 +871,20 @@ func TestParseSmartOutputKeepsCustomType(t *testing.T) {
|
||||
assert.Equal(t, "sat+megaraid", device.Type)
|
||||
assert.Equal(t, "sat", device.parserType)
|
||||
assert.True(t, device.typeVerified)
|
||||
assert.Equal(t, "sat+megaraid", sm.SmartDataMap["9C40918040082"].DiskType)
|
||||
}
|
||||
|
||||
func TestParseSmartOutputDoesNotNormalizeDeviceIdentity(t *testing.T) {
|
||||
fixturePath := filepath.Join("test-data", "smart", "sda.json")
|
||||
data, err := os.ReadFile(fixturePath)
|
||||
require.NoError(t, err)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
device := &DeviceInfo{Name: "/dev/sda", Type: "ata", explicitType: true}
|
||||
|
||||
require.True(t, sm.parseSmartOutput(device, data))
|
||||
assert.Equal(t, "sat", device.parserType)
|
||||
assert.Equal(t, "ata", sm.SmartDataMap["9C40918040082"].DiskType)
|
||||
}
|
||||
|
||||
func TestParseSmartOutputResetsVerificationOnFailure(t *testing.T) {
|
||||
@@ -728,6 +1006,182 @@ func TestIsVirtualDeviceScsi(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindAtaDeviceStatisticsValue(t *testing.T) {
|
||||
val42 := int64(42)
|
||||
val100 := int64(100)
|
||||
valMinus20 := int64(-20)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
data smart.SmartInfoForSata
|
||||
ataDeviceStats smart.AtaDeviceStatistics
|
||||
entryNumber uint8
|
||||
entryName string
|
||||
minValue int64
|
||||
maxValue int64
|
||||
expectedValue *int64
|
||||
}{
|
||||
{
|
||||
name: "value in ataDeviceStats",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Current Temperature", Value: &val42},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: &val42,
|
||||
},
|
||||
{
|
||||
name: "value unmarshaled from data",
|
||||
data: smart.SmartInfoForSata{
|
||||
AtaDeviceStatistics: []byte(`{"pages":[{"number":5,"table":[{"name":"Current Temperature","value":100}]}]}`),
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 255,
|
||||
expectedValue: &val100,
|
||||
},
|
||||
{
|
||||
name: "value out of range (too high)",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Current Temperature", Value: &val100},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 50,
|
||||
expectedValue: nil,
|
||||
},
|
||||
{
|
||||
name: "value out of range (too low)",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Min Temp", Value: &valMinus20},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Min Temp",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: nil,
|
||||
},
|
||||
{
|
||||
name: "no statistics available",
|
||||
data: smart.SmartInfoForSata{},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 255,
|
||||
expectedValue: nil,
|
||||
},
|
||||
{
|
||||
name: "wrong page number",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 1,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Current Temperature", Value: &val42},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: nil,
|
||||
},
|
||||
{
|
||||
name: "wrong entry name",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Other Stat", Value: &val42},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: nil,
|
||||
},
|
||||
{
|
||||
name: "case insensitive name match",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "CURRENT TEMPERATURE", Value: &val42},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: &val42,
|
||||
},
|
||||
{
|
||||
name: "entry value is nil",
|
||||
ataDeviceStats: smart.AtaDeviceStatistics{
|
||||
Pages: []smart.AtaDeviceStatisticsPage{
|
||||
{
|
||||
Number: 5,
|
||||
Table: []smart.AtaDeviceStatisticsEntry{
|
||||
{Name: "Current Temperature", Value: nil},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
entryNumber: 5,
|
||||
entryName: "Current Temperature",
|
||||
minValue: 0,
|
||||
maxValue: 100,
|
||||
expectedValue: nil,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := findAtaDeviceStatisticsValue(&tt.data, &tt.ataDeviceStats, tt.entryNumber, tt.entryName, tt.minValue, tt.maxValue)
|
||||
if tt.expectedValue == nil {
|
||||
assert.Nil(t, result)
|
||||
} else {
|
||||
require.NotNil(t, result)
|
||||
assert.Equal(t, *tt.expectedValue, *result)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRefreshExcludedDevices(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -788,7 +1242,7 @@ func TestRefreshExcludedDevices(t *testing.T) {
|
||||
t.Setenv("EXCLUDE_SMART", tt.envValue)
|
||||
} else {
|
||||
// Ensure env var is not set for empty test
|
||||
os.Unsetenv("EXCLUDE_SMART")
|
||||
t.Setenv("EXCLUDE_SMART", "")
|
||||
}
|
||||
|
||||
sm := &SmartManager{}
|
||||
@@ -952,3 +1406,81 @@ func TestIsNvmeControllerPath(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForNvmeAppleSSD(t *testing.T) {
|
||||
// Apple SSDs don't report user_capacity via smartctl; capacity should be fetched
|
||||
// from system_profiler via the darwinNvmeProvider fallback.
|
||||
fixturePath := filepath.Join("test-data", "smart", "apple_nvme.json")
|
||||
data, err := os.ReadFile(fixturePath)
|
||||
require.NoError(t, err)
|
||||
|
||||
providerCalls := 0
|
||||
fakeProvider := func() ([]byte, error) {
|
||||
providerCalls++
|
||||
return []byte(`{
|
||||
"SPNVMeDataType": [{
|
||||
"_items": [{
|
||||
"device_serial": "0ba0147940253c15",
|
||||
"size_in_bytes": 251000193024
|
||||
}]
|
||||
}]
|
||||
}`), nil
|
||||
}
|
||||
|
||||
sm := &SmartManager{
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
darwinNvmeProvider: fakeProvider,
|
||||
}
|
||||
|
||||
hasData, _ := sm.parseSmartForNvme(data, "")
|
||||
require.True(t, hasData)
|
||||
|
||||
deviceData, ok := sm.SmartDataMap["0ba0147940253c15"]
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, "APPLE SSD AP0256Q", deviceData.ModelName)
|
||||
assert.Equal(t, uint64(251000193024), deviceData.Capacity)
|
||||
assert.Equal(t, uint8(42), deviceData.Temperature)
|
||||
assert.Equal(t, "PASSED", deviceData.SmartStatus)
|
||||
assert.Equal(t, 1, providerCalls, "system_profiler should be called once")
|
||||
|
||||
// Second parse: provider should NOT be called again (cache hit)
|
||||
_, _ = sm.parseSmartForNvme(data, "")
|
||||
assert.Equal(t, 1, providerCalls, "system_profiler should not be called again after caching")
|
||||
}
|
||||
|
||||
func TestLookupDarwinNvmeCapacityMultipleDisks(t *testing.T) {
|
||||
fakeProvider := func() ([]byte, error) {
|
||||
return []byte(`{
|
||||
"SPNVMeDataType": [
|
||||
{
|
||||
"_items": [
|
||||
{"device_serial": "serial-disk0", "size_in_bytes": 251000193024},
|
||||
{"device_serial": "serial-disk1", "size_in_bytes": 1000204886016}
|
||||
]
|
||||
},
|
||||
{
|
||||
"_items": [
|
||||
{"device_serial": "serial-disk2", "size_in_bytes": 512110190592}
|
||||
]
|
||||
}
|
||||
]
|
||||
}`), nil
|
||||
}
|
||||
|
||||
sm := &SmartManager{darwinNvmeProvider: fakeProvider}
|
||||
assert.Equal(t, uint64(251000193024), sm.lookupDarwinNvmeCapacity("serial-disk0"))
|
||||
assert.Equal(t, uint64(1000204886016), sm.lookupDarwinNvmeCapacity("serial-disk1"))
|
||||
assert.Equal(t, uint64(512110190592), sm.lookupDarwinNvmeCapacity("serial-disk2"))
|
||||
assert.Equal(t, uint64(0), sm.lookupDarwinNvmeCapacity("unknown-serial"))
|
||||
}
|
||||
|
||||
func TestLookupDarwinNvmeCapacityProviderError(t *testing.T) {
|
||||
fakeProvider := func() ([]byte, error) {
|
||||
return nil, errors.New("system_profiler not found")
|
||||
}
|
||||
|
||||
sm := &SmartManager{darwinNvmeProvider: fakeProvider}
|
||||
assert.Equal(t, uint64(0), sm.lookupDarwinNvmeCapacity("any-serial"))
|
||||
// Cache should be initialized even on error so we don't retry (Once already fired)
|
||||
assert.NotNil(t, sm.darwinNvmeCapacity)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,462 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
zfsentity "github.com/henrygd/beszel/internal/entities/zfs"
|
||||
)
|
||||
|
||||
// zfsDatasetUsage holds usage values for a ZFS dataset mountpoint.
|
||||
type zfsDatasetUsage struct {
|
||||
used uint64
|
||||
avail uint64
|
||||
}
|
||||
|
||||
// datasetUsageRefreshInterval controls how often `zfs list` is re-run for the
|
||||
// mountpoint usage map. Dataset inventory changes rarely.
|
||||
const datasetUsageRefreshInterval = 5 * time.Minute
|
||||
|
||||
// poolStatsRefreshInterval controls how often `zpool list` is re-run for pool
|
||||
// capacity. Health and I/O are read from procfs on Linux, so the utility only
|
||||
// needs to refresh slow-moving space accounting.
|
||||
const poolStatsRefreshInterval = time.Minute
|
||||
|
||||
// btrfsFilesystems is the btrfs source; overridable in tests.
|
||||
var btrfsFilesystems = btrfs.Filesystems
|
||||
|
||||
type poolKernelSample struct {
|
||||
nread uint64
|
||||
nwrite uint64
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// StoragePoolManager combines independent backend inventories. Metrics and
|
||||
// dataset usage require the agent lock; GetDetail is safe for concurrent calls.
|
||||
type StoragePoolManager struct {
|
||||
backends []*poolBackend
|
||||
detailInterval time.Duration
|
||||
}
|
||||
|
||||
// poolBackend owns one backend's collectors and caches. Collector functions
|
||||
// are immutable after construction and may run concurrently for metrics/details.
|
||||
type poolBackend struct {
|
||||
name string
|
||||
poolStatsFn func() ([]zfs.PoolStat, error) // capacity/health source
|
||||
datasetsFn func() ([]zfs.Dataset, error) // dataset inventory source
|
||||
kernelStatsFn func() ([]zfs.PoolKernelStat, error) // procfs pool state/I/O source
|
||||
poolStatusesFn func() ([]zfs.PoolStatus, error) // scrub/vdev detail source
|
||||
|
||||
poolData []zfs.PoolStat // cached pool inventory (TTL below)
|
||||
lastPoolStats time.Time
|
||||
kernelSamples map[string]poolKernelSample
|
||||
|
||||
datasetUsage map[string]zfsDatasetUsage // mountpoint -> usage
|
||||
lastUsageRefresh time.Time
|
||||
|
||||
// Detail data (pools, vdevs, scrub, datasets) is cached and refreshed on
|
||||
// an interval. Accessed from handler goroutines, so it is mutex-protected.
|
||||
detailMu sync.Mutex
|
||||
detail *zfsentity.ZfsData
|
||||
lastDetailRefresh time.Time
|
||||
detailFailed bool
|
||||
}
|
||||
|
||||
func newStoragePoolManager() *StoragePoolManager {
|
||||
return &StoragePoolManager{
|
||||
backends: []*poolBackend{newZfsBackend(), newBtrfsBackend()},
|
||||
detailInterval: time.Hour,
|
||||
}
|
||||
}
|
||||
|
||||
func newZfsBackend() *poolBackend {
|
||||
return &poolBackend{
|
||||
name: "zfs",
|
||||
poolStatsFn: optionalPoolSource(zfs.PoolStats),
|
||||
datasetsFn: zfs.Datasets,
|
||||
kernelStatsFn: optionalPoolSource(zfs.PoolKernelStats),
|
||||
poolStatusesFn: optionalPoolSource(zfs.PoolStatuses),
|
||||
}
|
||||
}
|
||||
|
||||
func newBtrfsBackend() *poolBackend {
|
||||
return &poolBackend{
|
||||
name: "btrfs",
|
||||
poolStatsFn: btrfsSource(btrfsPoolStats),
|
||||
kernelStatsFn: btrfsSource(btrfsKernelStats),
|
||||
poolStatusesFn: btrfsSource(btrfsPoolStatuses),
|
||||
}
|
||||
}
|
||||
|
||||
// datasets is optional: only backends that expose datasets provide a collector.
|
||||
func (b *poolBackend) datasets() ([]zfs.Dataset, error) {
|
||||
if b.datasetsFn == nil {
|
||||
return nil, nil
|
||||
}
|
||||
return b.datasetsFn()
|
||||
}
|
||||
|
||||
// A missing utility/interface is a successfully observed absent backend.
|
||||
func optionalPoolSource[T any](source func() ([]T, error)) func() ([]T, error) {
|
||||
return func() ([]T, error) {
|
||||
items, err := source()
|
||||
if errors.Is(err, zfs.ErrNoZfs) || errors.Is(err, exec.ErrNotFound) || errors.Is(err, errors.ErrUnsupported) {
|
||||
return nil, nil
|
||||
}
|
||||
return items, err
|
||||
}
|
||||
}
|
||||
|
||||
func btrfsSource[T any](convert func(btrfs.Filesystem) T) func() ([]T, error) {
|
||||
return func() ([]T, error) {
|
||||
filesystems, err := optionalPoolSource(btrfsFilesystems)()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
items := make([]T, 0, len(filesystems))
|
||||
for _, fs := range filesystems {
|
||||
items = append(items, convert(fs))
|
||||
}
|
||||
return items, nil
|
||||
}
|
||||
}
|
||||
|
||||
// Update refreshes systemStats.ZfsPools with the latest pool data. I/O
|
||||
// throughput and health come from inexpensive kernel kstats on Linux. Pool
|
||||
// capacity and dataset usage come from separately cached utility calls. The
|
||||
// pool map is empty when both backends are absent.
|
||||
func (m *StoragePoolManager) Update(systemStats *system.Stats) {
|
||||
// Rebuild the combined map so successful pool removals clear old samples.
|
||||
systemStats.ZfsPools = nil
|
||||
for _, backend := range m.backends {
|
||||
backend.updateBackendStats(systemStats)
|
||||
}
|
||||
}
|
||||
|
||||
func (b *poolBackend) updateBackendStats(systemStats *system.Stats) {
|
||||
pools := b.poolStats()
|
||||
if len(pools) == 0 {
|
||||
b.kernelSamples = nil
|
||||
return
|
||||
}
|
||||
|
||||
kernelStats, ioRates := b.kernelStats()
|
||||
|
||||
if systemStats.ZfsPools == nil {
|
||||
systemStats.ZfsPools = make(map[string]*system.ZfsPool, len(pools))
|
||||
}
|
||||
for i := range pools {
|
||||
pool := &pools[i]
|
||||
// Full precision, matching the dataset values below; the frontend
|
||||
// formats any magnitude.
|
||||
stats := &system.ZfsPool{
|
||||
DisplayName: pool.DisplayName,
|
||||
Raw: pool.Raw,
|
||||
Total: float64(pool.Size) / (1024 * 1024 * 1024),
|
||||
Used: float64(pool.Alloc) / (1024 * 1024 * 1024),
|
||||
Health: pool.Health,
|
||||
}
|
||||
if kernel, exists := kernelStats[pool.Name]; exists && kernel.Health != "" {
|
||||
stats.Health = kernel.Health
|
||||
}
|
||||
if io, exists := ioRates[pool.Name]; exists {
|
||||
stats.ReadBytes = io.NRead
|
||||
stats.WriteBytes = io.NWrite
|
||||
}
|
||||
slog.Debug("Storage pool sample", "backend", b.name, "pool", pool.Name, "health", stats.Health, "used_gb", stats.Used, "read_bps", stats.ReadBytes, "write_bps", stats.WriteBytes)
|
||||
systemStats.ZfsPools[pool.Name] = stats
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// poolStats returns the cached pool inventory, calling its collector at most
|
||||
// every poolStatsRefreshInterval. On failure the previous inventory is
|
||||
// retained and the refresh is retried on the next cadence.
|
||||
func (b *poolBackend) poolStats() []zfs.PoolStat {
|
||||
if b.lastPoolStats.IsZero() || time.Since(b.lastPoolStats) >= poolStatsRefreshInterval {
|
||||
pools, err := b.poolStatsFn()
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool stats unavailable", "backend", b.name, "err", err)
|
||||
} else {
|
||||
b.poolData = pools
|
||||
}
|
||||
b.lastPoolStats = time.Now()
|
||||
}
|
||||
return b.poolData
|
||||
}
|
||||
|
||||
// kernelStats reads cumulative pool counters and converts them to per-second
|
||||
// rates. Counter decreases indicate a pool export/import and reset the
|
||||
// baseline instead of producing an underflow spike.
|
||||
func (b *poolBackend) kernelStats() (map[string]zfs.PoolKernelStat, map[string]zfs.PoolIoStats) {
|
||||
if b.kernelStatsFn == nil {
|
||||
return nil, nil
|
||||
}
|
||||
stats, err := b.kernelStatsFn()
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool kernel stats unavailable", "backend", b.name, "err", err)
|
||||
return nil, nil
|
||||
}
|
||||
now := time.Now()
|
||||
byName := make(map[string]zfs.PoolKernelStat, len(stats))
|
||||
rates := make(map[string]zfs.PoolIoStats, len(stats))
|
||||
nextSamples := make(map[string]poolKernelSample, len(stats))
|
||||
for _, stat := range stats {
|
||||
byName[stat.Name] = stat
|
||||
if previous, ok := b.kernelSamples[stat.Name]; ok && now.After(previous.at) &&
|
||||
stat.NRead >= previous.nread && stat.NWrite >= previous.nwrite {
|
||||
seconds := now.Sub(previous.at).Seconds()
|
||||
rates[stat.Name] = zfs.PoolIoStats{
|
||||
NRead: uint64(float64(stat.NRead-previous.nread) / seconds),
|
||||
NWrite: uint64(float64(stat.NWrite-previous.nwrite) / seconds),
|
||||
}
|
||||
}
|
||||
nextSamples[stat.Name] = poolKernelSample{nread: stat.NRead, nwrite: stat.NWrite, at: now}
|
||||
}
|
||||
b.kernelSamples = nextSamples
|
||||
return byName, rates
|
||||
}
|
||||
|
||||
// refreshDatasetUsage re-runs `zfs list` when the refresh window has elapsed
|
||||
// and rebuilds the mountpoint-keyed usage map.
|
||||
func (b *poolBackend) refreshDatasetUsage() {
|
||||
if !b.lastUsageRefresh.IsZero() && time.Since(b.lastUsageRefresh) < datasetUsageRefreshInterval {
|
||||
return
|
||||
}
|
||||
datasets, err := b.datasets()
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool dataset usage unavailable", "backend", b.name, "err", err)
|
||||
} else {
|
||||
usage := make(map[string]zfsDatasetUsage, len(datasets))
|
||||
for _, ds := range datasets {
|
||||
if ds.Mountpoint != "" && ds.Mountpoint != "-" {
|
||||
usage[ds.Mountpoint] = zfsDatasetUsage{used: ds.Used, avail: ds.Avail}
|
||||
}
|
||||
}
|
||||
b.datasetUsage = usage
|
||||
}
|
||||
b.lastUsageRefresh = time.Now()
|
||||
}
|
||||
|
||||
// DatasetUsage returns ZFS dataset usage keyed by mountpoint, refreshed at
|
||||
// most every datasetUsageRefreshInterval. On failure the previous map is
|
||||
// retained and a debug log is emitted.
|
||||
func (m *StoragePoolManager) DatasetUsage() map[string]zfsDatasetUsage {
|
||||
for _, backend := range m.backends {
|
||||
if backend.name == "zfs" {
|
||||
backend.refreshDatasetUsage()
|
||||
return backend.datasetUsage
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// GetDetail combines backend snapshots, identifying successful inventories so
|
||||
// the hub can accept partial updates without deleting failed backend records.
|
||||
func (m *StoragePoolManager) GetDetail(force bool) *zfsentity.ZfsData {
|
||||
data := &zfsentity.ZfsData{Complete: true}
|
||||
for _, backend := range m.backends {
|
||||
snapshot := backend.getBackendDetail(force, m.detailInterval)
|
||||
data.Pools = append(data.Pools, snapshot.Pools...)
|
||||
if snapshot.Complete {
|
||||
data.CompleteBackends = append(data.CompleteBackends, backend.name)
|
||||
} else {
|
||||
data.Complete = false
|
||||
}
|
||||
}
|
||||
return data
|
||||
}
|
||||
|
||||
func (b *poolBackend) getBackendDetail(force bool, interval time.Duration) *zfsentity.ZfsData {
|
||||
b.detailMu.Lock()
|
||||
defer b.detailMu.Unlock()
|
||||
|
||||
if force || b.detailFailed || b.detail == nil || time.Since(b.lastDetailRefresh) >= interval {
|
||||
if data, err := b.collectDetail(b.detail); err != nil {
|
||||
b.detailFailed = true
|
||||
slog.Debug("Storage pool detail collection failed", "backend", b.name, "err", err)
|
||||
if b.detail == nil {
|
||||
return &zfsentity.ZfsData{}
|
||||
}
|
||||
return &zfsentity.ZfsData{Pools: b.detail.Pools}
|
||||
} else {
|
||||
b.detailFailed = false
|
||||
b.detail = data
|
||||
b.lastDetailRefresh = time.Now()
|
||||
}
|
||||
}
|
||||
if b.detail == nil {
|
||||
return &zfsentity.ZfsData{}
|
||||
}
|
||||
return b.detail
|
||||
}
|
||||
|
||||
// collectDetail builds a ZfsData payload from the current system state.
|
||||
func (b *poolBackend) collectDetail(previous *zfsentity.ZfsData) (*zfsentity.ZfsData, error) {
|
||||
pools, err := b.poolStatsFn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(pools) == 0 {
|
||||
return &zfsentity.ZfsData{Pools: []*zfsentity.PoolDetail{}, Complete: true}, nil
|
||||
}
|
||||
|
||||
statuses, statusErr := b.poolStatusesFn()
|
||||
if statusErr != nil {
|
||||
slog.Debug("Storage pool status unavailable", "backend", b.name, "err", statusErr)
|
||||
}
|
||||
datasets, datasetsErr := b.datasets()
|
||||
if datasetsErr != nil {
|
||||
slog.Debug("Storage pool datasets unavailable", "backend", b.name, "err", datasetsErr)
|
||||
}
|
||||
|
||||
statusByPool := make(map[string]zfs.PoolStatus, len(statuses))
|
||||
for _, st := range statuses {
|
||||
statusByPool[st.Name] = st
|
||||
}
|
||||
|
||||
previousByPool := make(map[string]*zfsentity.PoolDetail)
|
||||
if previous != nil {
|
||||
for _, pool := range previous.Pools {
|
||||
if pool != nil {
|
||||
previousByPool[pool.Name] = pool
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
data := &zfsentity.ZfsData{Pools: make([]*zfsentity.PoolDetail, 0, len(pools)), Complete: true}
|
||||
for i := range pools {
|
||||
p := &pools[i]
|
||||
detail := &zfsentity.PoolDetail{
|
||||
DisplayName: p.DisplayName,
|
||||
Raw: p.Raw,
|
||||
Name: p.Name,
|
||||
Health: p.Health,
|
||||
Size: p.Size,
|
||||
Alloc: p.Alloc,
|
||||
Free: p.Free,
|
||||
}
|
||||
if st, ok := statusByPool[p.Name]; statusErr == nil && ok {
|
||||
if st.Scrub.State != "" && st.Scrub.State != "NONE" {
|
||||
detail.Scrub = &zfsentity.Scrub{
|
||||
State: st.Scrub.State,
|
||||
Progress: st.Scrub.Progress,
|
||||
Errors: st.Scrub.Errors,
|
||||
}
|
||||
}
|
||||
for _, v := range st.Vdevs {
|
||||
detail.Vdevs = append(detail.Vdevs, &zfsentity.Vdev{
|
||||
Name: v.Name,
|
||||
State: v.State,
|
||||
ReadErrs: v.ReadErrs,
|
||||
WriteErrs: v.WriteErrs,
|
||||
ChecksumErrs: v.ChecksumErrs,
|
||||
})
|
||||
}
|
||||
} else {
|
||||
if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Scrub = cached.Scrub
|
||||
detail.Vdevs = cached.Vdevs
|
||||
}
|
||||
}
|
||||
if datasetsErr == nil {
|
||||
foundDataset := false
|
||||
for _, ds := range datasets {
|
||||
if poolOfDataset(ds.Name) == p.Name {
|
||||
foundDataset = true
|
||||
detail.Datasets = append(detail.Datasets, &zfsentity.Dataset{
|
||||
Name: ds.Name,
|
||||
Used: ds.Used,
|
||||
Avail: ds.Avail,
|
||||
Mountpoint: ds.Mountpoint,
|
||||
})
|
||||
}
|
||||
}
|
||||
if !foundDataset {
|
||||
if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Datasets = cached.Datasets
|
||||
}
|
||||
}
|
||||
} else if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Datasets = cached.Datasets
|
||||
}
|
||||
data.Pools = append(data.Pools, detail)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
// poolOfDataset returns the pool name for a dataset name (everything before
|
||||
// the first '/'). Datasets without a separator belong to a pool of the same
|
||||
// name.
|
||||
func poolOfDataset(name string) string {
|
||||
if idx := strings.IndexByte(name, '/'); idx >= 0 {
|
||||
return name[:idx]
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// ZfsMountpoints returns the set of mountpoints backed by ZFS datasets.
|
||||
func (m *StoragePoolManager) ZfsMountpoints() map[string]bool {
|
||||
usage := m.DatasetUsage()
|
||||
mountpoints := make(map[string]bool, len(usage))
|
||||
for mountpoint := range usage {
|
||||
mountpoints[mountpoint] = true
|
||||
}
|
||||
return mountpoints
|
||||
}
|
||||
|
||||
func btrfsPoolStats(fs btrfs.Filesystem) zfs.PoolStat {
|
||||
return zfs.PoolStat{MountID: fs.MountID, IODevice: fs.IODevice, Raw: fs.Raw, DisplayName: fs.Name, Name: "b:" + fs.UUID, Size: fs.Size, Alloc: fs.Alloc, Free: fs.Size - min(fs.Alloc, fs.Size), Health: fs.Health}
|
||||
}
|
||||
|
||||
func btrfsKernelStats(fs btrfs.Filesystem) zfs.PoolKernelStat {
|
||||
return zfs.PoolKernelStat{Name: "b:" + fs.UUID, Health: fs.Health, NRead: fs.NRead, NWrite: fs.NWrite}
|
||||
}
|
||||
|
||||
func btrfsPoolStatuses(fs btrfs.Filesystem) zfs.PoolStatus {
|
||||
status := zfs.PoolStatus{Name: "b:" + fs.UUID, State: fs.Health, Scrub: zfs.ScrubStatus{State: "NONE"}}
|
||||
for _, dev := range fs.Devices {
|
||||
status.Vdevs = append(status.Vdevs, zfs.VdevStatus{
|
||||
Name: dev.Name, State: dev.State,
|
||||
ReadErrs: dev.ReadErrs, WriteErrs: dev.WriteErrs, ChecksumErrs: dev.CorruptionErrs,
|
||||
})
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
// markDuplicateCharts leaves pool telemetry and detail intact, but tells the
|
||||
// hub which charts already have a filesystem equivalent. Only exact kernel
|
||||
// filesystem and I/O-device matches qualify; labels are never used.
|
||||
func (m *StoragePoolManager) markDuplicateCharts(stats *system.Stats, filesystems map[string]*system.FsStats, mountID func(string) string) {
|
||||
identities := make(map[string]string, len(filesystems))
|
||||
for device, fs := range filesystems {
|
||||
if fs.DiskTotal > 0 {
|
||||
identities[device] = mountID(fs.Mountpoint)
|
||||
}
|
||||
}
|
||||
for _, backend := range m.backends {
|
||||
for _, pool := range backend.poolData {
|
||||
sample := stats.ZfsPools[pool.Name]
|
||||
if sample == nil || pool.MountID == "" {
|
||||
continue
|
||||
}
|
||||
for device, identity := range identities {
|
||||
if identity != pool.MountID {
|
||||
continue
|
||||
}
|
||||
// Raw physical usage is not equivalent to a filesystem usage chart.
|
||||
sample.HideUsage = !pool.Raw
|
||||
if pool.IODevice != "" && pool.IODevice == device {
|
||||
sample.HideIO = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,505 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestOptionalPoolSource(t *testing.T) {
|
||||
failure := errors.New("timeout")
|
||||
for _, err := range []error{nil, zfs.ErrNoZfs, fmt.Errorf("zpool: %w", exec.ErrNotFound), errors.ErrUnsupported, failure} {
|
||||
_, got := optionalPoolSource(func() ([]zfs.PoolStat, error) { return nil, err })()
|
||||
if err == failure {
|
||||
assert.ErrorIs(t, got, failure)
|
||||
} else {
|
||||
assert.NoError(t, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type poolTestBackend struct {
|
||||
name string
|
||||
err error
|
||||
alloc uint64
|
||||
read uint64
|
||||
empty bool
|
||||
}
|
||||
|
||||
func (state *poolTestBackend) backend() *poolBackend {
|
||||
name := "zfs"
|
||||
if strings.HasPrefix(state.name, "b:") {
|
||||
name = "btrfs"
|
||||
}
|
||||
return &poolBackend{
|
||||
name: name,
|
||||
poolStatsFn: func() ([]zfs.PoolStat, error) {
|
||||
if state.empty {
|
||||
return nil, state.err
|
||||
}
|
||||
return []zfs.PoolStat{{Name: state.name, Size: 100, Alloc: state.alloc}}, state.err
|
||||
},
|
||||
kernelStatsFn: func() ([]zfs.PoolKernelStat, error) {
|
||||
return []zfs.PoolKernelStat{{Name: state.name, NRead: state.read}}, state.err
|
||||
},
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) { return nil, nil },
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return nil, nil },
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndependentPoolBackendCaches(t *testing.T) {
|
||||
for _, failed := range []int{0, 1} {
|
||||
t.Run([]string{"zfs", "btrfs"}[failed], func(t *testing.T) {
|
||||
states := []*poolTestBackend{{name: "tank", alloc: 10}, {name: "b:uuid", alloc: 10}}
|
||||
managers := []*poolBackend{states[0].backend(), states[1].backend()}
|
||||
zm := &StoragePoolManager{backends: managers, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 2)
|
||||
require.True(t, zm.GetDetail(true).Complete)
|
||||
baseline := poolKernelSample{at: time.Now().Add(-time.Second)}
|
||||
for i, m := range managers {
|
||||
m.lastPoolStats = time.Time{}
|
||||
m.kernelSamples[states[i].name] = baseline
|
||||
states[i].alloc = 20
|
||||
states[i].read = 100
|
||||
}
|
||||
states[failed].err = errors.New("collection failed")
|
||||
zm.Update(&stats)
|
||||
healthy := 1 - failed
|
||||
assert.Equal(t, uint64(10), managers[failed].poolData[0].Alloc)
|
||||
assert.Equal(t, uint64(20), managers[healthy].poolData[0].Alloc)
|
||||
assert.Equal(t, baseline, managers[failed].kernelSamples[states[failed].name])
|
||||
assert.Zero(t, stats.ZfsPools[states[failed].name].ReadBytes)
|
||||
assert.Positive(t, stats.ZfsPools[states[healthy].name].ReadBytes)
|
||||
partial := zm.GetDetail(true)
|
||||
assert.False(t, partial.Complete)
|
||||
assert.False(t, partial.CanRefreshPool(states[failed].name))
|
||||
assert.True(t, partial.CanRefreshPool(states[healthy].name))
|
||||
assert.Equal(t, uint64(10), partial.Pools[failed].Alloc)
|
||||
assert.Equal(t, uint64(20), partial.Pools[healthy].Alloc)
|
||||
assert.False(t, zm.GetDetail(false).Complete, "a failed forced refresh must not become complete from cache")
|
||||
|
||||
// Successful empty inventory removes only the healthy backend's pool.
|
||||
states[healthy].empty = true
|
||||
managers[healthy].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 1)
|
||||
assert.Contains(t, stats.ZfsPools, states[failed].name)
|
||||
partial = zm.GetDetail(true)
|
||||
require.Len(t, partial.Pools, 1)
|
||||
assert.True(t, partial.CanRefreshPool(states[healthy].name))
|
||||
|
||||
// Recovery uses the retained I/O baseline, then normal removal works.
|
||||
states[failed].err = nil
|
||||
managers[failed].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
assert.Positive(t, stats.ZfsPools[states[failed].name].ReadBytes)
|
||||
assert.True(t, zm.GetDetail(true).Complete)
|
||||
states[failed].empty = true
|
||||
managers[failed].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
assert.Empty(t, stats.ZfsPools)
|
||||
assert.Empty(t, zm.GetDetail(true).Pools)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndependentBackendsWithoutCache(t *testing.T) {
|
||||
z := &poolTestBackend{name: "tank", err: errors.New("ZFS failure")}
|
||||
b := &poolTestBackend{name: "b:uuid", alloc: 20}
|
||||
zm := &StoragePoolManager{backends: []*poolBackend{z.backend(), b.backend()}, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 1)
|
||||
assert.Contains(t, stats.ZfsPools, "b:uuid")
|
||||
detail := zm.GetDetail(true)
|
||||
require.Len(t, detail.Pools, 1)
|
||||
assert.False(t, detail.Complete)
|
||||
assert.Equal(t, []string{"btrfs"}, detail.CompleteBackends)
|
||||
}
|
||||
|
||||
func TestConcurrentBackendDetailsAndMetrics(t *testing.T) {
|
||||
zm := &StoragePoolManager{backends: []*poolBackend{(&poolTestBackend{name: "tank"}).backend(), (&poolTestBackend{name: "b:uuid"}).backend()}, detailInterval: time.Hour}
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < 3; i++ {
|
||||
wg.Add(1)
|
||||
go func(metrics bool) {
|
||||
defer wg.Done()
|
||||
for j := 0; j < 10; j++ {
|
||||
if metrics {
|
||||
zm.Update(&system.Stats{})
|
||||
} else {
|
||||
zm.GetDetail(true)
|
||||
}
|
||||
}
|
||||
}(i == 0)
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
func TestStoragePoolBackendOrder(t *testing.T) {
|
||||
z := (&poolTestBackend{name: "tank"}).backend()
|
||||
b := (&poolTestBackend{name: "b:uuid"}).backend()
|
||||
b.datasetsFn = nil // Btrfs does not expose datasets.
|
||||
z.datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank/data", Mountpoint: "/tank", Used: 10}}, nil
|
||||
}
|
||||
m := &StoragePoolManager{backends: []*poolBackend{b, z}, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
m.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 2)
|
||||
detail := m.GetDetail(true)
|
||||
require.True(t, detail.Complete)
|
||||
assert.Equal(t, []string{"btrfs", "zfs"}, detail.CompleteBackends)
|
||||
assert.Empty(t, detail.Pools[0].Datasets)
|
||||
assert.Len(t, detail.Pools[1].Datasets, 1)
|
||||
assert.Equal(t, uint64(10), m.DatasetUsage()["/tank"].used)
|
||||
|
||||
b.poolData[0].MountID = "uuid"
|
||||
b.poolData[0].IODevice = "sda"
|
||||
calls := 0
|
||||
m.markDuplicateCharts(&stats, map[string]*system.FsStats{
|
||||
"sda": {Mountpoint: "/", DiskTotal: 100},
|
||||
}, func(string) string { calls++; return "uuid" })
|
||||
assert.Equal(t, 1, calls, "resolve each filesystem once across all backends")
|
||||
assert.True(t, stats.ZfsPools["b:uuid"].HideUsage)
|
||||
assert.True(t, stats.ZfsPools["b:uuid"].HideIO)
|
||||
}
|
||||
|
||||
func TestUpdatePopulatesZfsPools(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Size: 23999000000000, Alloc: 12000000000000, Free: 11999000000000, Health: "DEGRADED"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank/apps", Used: 5000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
|
||||
{Name: "tank/backup", Used: 6000000000000, Avail: 11999000000000, Mountpoint: "/tank/backup"},
|
||||
// Small zvol (Proxmox VM EFI disk): must not round to zero.
|
||||
{Name: "rpool/vm-100-disk-2", Used: 4194304, Avail: 0, Mountpoint: "-"},
|
||||
}, nil
|
||||
}
|
||||
var kernelCalls int
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
kernelCalls++
|
||||
return []zfs.PoolKernelStat{{
|
||||
Name: "tank", Health: "ONLINE",
|
||||
NRead: uint64(kernelCalls-1) * 1250, NWrite: uint64(kernelCalls-1) * 5120,
|
||||
}}, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
// The first kernel sample establishes the cumulative-counter baseline.
|
||||
zm.Update(&stats)
|
||||
zm.backends[0].kernelSamples["tank"] = poolKernelSample{at: time.Now().Add(-time.Second)}
|
||||
zm.Update(&stats)
|
||||
require.NotNil(t, stats.ZfsPools)
|
||||
require.Contains(t, stats.ZfsPools, "tank")
|
||||
assert.InDelta(t, 22350.8105, stats.ZfsPools["tank"].Total, 0.0001) // Size in GiB
|
||||
assert.InDelta(t, 11175.8709, stats.ZfsPools["tank"].Used, 0.0001) // Alloc in GiB
|
||||
assert.Equal(t, "ONLINE", stats.ZfsPools["tank"].Health)
|
||||
assert.InDelta(t, 1250, stats.ZfsPools["tank"].ReadBytes, 5)
|
||||
assert.InDelta(t, 5120, stats.ZfsPools["tank"].WriteBytes, 5)
|
||||
|
||||
}
|
||||
|
||||
// TestUpdateKernelStatsMissing verifies pools without a kernel sample report zero
|
||||
// I/O instead of erroring.
|
||||
func TestUpdateKernelStatsMissing(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Size: 1, Alloc: 1, Health: "ONLINE"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.NotNil(t, stats.ZfsPools)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
|
||||
}
|
||||
|
||||
func TestUpdateKernelCounterReset(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Health: "ONLINE"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.backends[0].kernelSamples = map[string]poolKernelSample{
|
||||
"tank": {nread: 100, nwrite: 200, at: time.Now().Add(-time.Second)},
|
||||
}
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
return []zfs.PoolKernelStat{{Name: "tank", Health: "ONLINE", NRead: 10, NWrite: 20}}, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
|
||||
}
|
||||
|
||||
func TestUpdateNoZfs(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
calls++
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
zm.Update(&stats)
|
||||
assert.Nil(t, stats.ZfsPools)
|
||||
assert.Equal(t, 1, calls, "failed pool discovery should be cached until the next refresh interval")
|
||||
}
|
||||
|
||||
func TestUpdateEmptyPools(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
calls++
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
zm.Update(&stats)
|
||||
assert.Nil(t, stats.ZfsPools)
|
||||
assert.Equal(t, 1, calls, "an empty pool inventory should be cached until the next refresh interval")
|
||||
}
|
||||
|
||||
func TestDatasetUsage(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
calls++
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
|
||||
{Name: "tank/apps", Used: 1000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
|
||||
{Name: "rpool", Used: 900000000000, Avail: 300000000000, Mountpoint: "-"}, // zvol/unmounted: excluded
|
||||
}, nil
|
||||
}
|
||||
|
||||
usage := zm.DatasetUsage()
|
||||
require.Len(t, usage, 2)
|
||||
assert.Equal(t, zfsDatasetUsage{used: 12000000000000, avail: 11999000000000}, usage["/tank"])
|
||||
assert.Equal(t, zfsDatasetUsage{used: 1000000000000, avail: 11999000000000}, usage["/tank/apps"])
|
||||
assert.Equal(t, 1, calls)
|
||||
|
||||
// Second call within the refresh window must not re-run the collector.
|
||||
zm.DatasetUsage()
|
||||
assert.Equal(t, 1, calls)
|
||||
}
|
||||
|
||||
func TestDatasetUsageRefreshOnErrorKeepsPrevious(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank", Used: 1, Avail: 1, Mountpoint: "/tank"}}, nil
|
||||
}
|
||||
assert.Len(t, zm.DatasetUsage(), 1)
|
||||
|
||||
// Force refresh window expiry, then a failing collector.
|
||||
zm.backends[0].lastUsageRefresh = time.Now().Add(-10 * time.Minute)
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
usage := zm.DatasetUsage()
|
||||
assert.Len(t, usage, 1, "previous usage should be retained on error")
|
||||
}
|
||||
|
||||
func TestGetDetailForceRefresh(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
poolCalls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
poolCalls++
|
||||
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
|
||||
first := zm.GetDetail(false)
|
||||
assert.True(t, first.Complete)
|
||||
require.Len(t, first.Pools, 1)
|
||||
assert.Equal(t, uint64(1), first.Pools[0].Alloc)
|
||||
|
||||
cached := zm.GetDetail(false)
|
||||
require.Len(t, cached.Pools, 1)
|
||||
assert.Equal(t, uint64(1), cached.Pools[0].Alloc)
|
||||
assert.Equal(t, 1, poolCalls)
|
||||
|
||||
refreshed := zm.GetDetail(true)
|
||||
assert.True(t, refreshed.Complete)
|
||||
require.Len(t, refreshed.Pools, 1)
|
||||
assert.Equal(t, uint64(2), refreshed.Pools[0].Alloc)
|
||||
assert.Equal(t, 2, poolCalls)
|
||||
}
|
||||
|
||||
func TestGetDetailSuccessfulEmptyInventoryClearsCache(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank"}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
|
||||
require.Len(t, zm.GetDetail(false).Pools, 1)
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, nil }
|
||||
empty := zm.GetDetail(true)
|
||||
assert.True(t, empty.Complete)
|
||||
assert.Empty(t, empty.Pools)
|
||||
}
|
||||
|
||||
func TestGetDetailFailureReturnsIncompleteCachedInventory(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank"}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) {
|
||||
return []zfs.PoolStatus{{Name: "tank", Vdevs: []zfs.VdevStatus{{Name: "mirror-0"}}}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank/data"}}, nil
|
||||
}
|
||||
first := zm.GetDetail(false)
|
||||
require.True(t, first.Complete)
|
||||
require.Len(t, first.Pools[0].Vdevs, 1)
|
||||
require.Len(t, first.Pools[0].Datasets, 1)
|
||||
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, zfs.ErrNoZfs }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, zfs.ErrNoZfs }
|
||||
partial := zm.GetDetail(true)
|
||||
require.True(t, partial.Complete)
|
||||
require.Len(t, partial.Pools[0].Vdevs, 1)
|
||||
require.Len(t, partial.Pools[0].Datasets, 1)
|
||||
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, zfs.ErrNoZfs }
|
||||
lastSuccessfulRefresh := zm.backends[0].lastDetailRefresh
|
||||
failed := zm.GetDetail(true)
|
||||
assert.False(t, failed.Complete)
|
||||
require.Len(t, failed.Pools, 1)
|
||||
assert.Equal(t, "tank", failed.Pools[0].Name)
|
||||
assert.Equal(t, lastSuccessfulRefresh, zm.backends[0].lastDetailRefresh)
|
||||
}
|
||||
|
||||
func TestZfsMountpoints(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Mountpoint: "/tank"},
|
||||
{Name: "rpool/ROOT/pve-1", Mountpoint: "/"},
|
||||
}, nil
|
||||
}
|
||||
mountpoints := zm.ZfsMountpoints()
|
||||
assert.Len(t, mountpoints, 2)
|
||||
assert.True(t, mountpoints["/tank"])
|
||||
assert.True(t, mountpoints["/"])
|
||||
}
|
||||
|
||||
func TestBtrfsRawCapacityPropagates(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolStatsFn: func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{btrfsPoolStats(btrfs.Filesystem{UUID: "raw", Name: "raw", Size: 200, Alloc: 100, Raw: true})}, nil
|
||||
},
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) { return nil, nil },
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return nil, nil }}}}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.True(t, stats.ZfsPools["b:raw"].Raw)
|
||||
detail := zm.GetDetail(true)
|
||||
require.True(t, detail.Complete)
|
||||
require.Len(t, detail.Pools, 1)
|
||||
assert.True(t, detail.Pools[0].Raw)
|
||||
}
|
||||
|
||||
func TestMarkDuplicatePoolCharts(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name, poolID, device string
|
||||
raw bool
|
||||
diskTotal float64
|
||||
wantUsage, wantIO bool
|
||||
}{
|
||||
{"single device root", "fs1", "dm-0", false, 100, true, true},
|
||||
{"multi device", "fs1", "", false, 100, true, false},
|
||||
{"different IO device", "fs1", "nvme0n1", false, 100, true, false},
|
||||
{"different filesystem", "fs2", "dm-0", false, 100, false, false},
|
||||
{"unknown identity", "", "dm-0", false, 100, false, false},
|
||||
{"raw usage", "fs1", "dm-0", true, 100, false, true},
|
||||
{"failed disk collection", "fs1", "dm-0", false, 0, false, false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolData: []zfs.PoolStat{{Name: "arbitrary label", MountID: tc.poolID, IODevice: tc.device, Raw: tc.raw}}}}}
|
||||
stats := &system.Stats{ZfsPools: map[string]*system.ZfsPool{"arbitrary label": {}}}
|
||||
fs := map[string]*system.FsStats{"dm-0": {Root: true, Mountpoint: "/", DiskTotal: tc.diskTotal}}
|
||||
zm.markDuplicateCharts(stats, fs, func(string) string { return "fs1" })
|
||||
assert.Equal(t, tc.wantUsage, stats.ZfsPools["arbitrary label"].HideUsage)
|
||||
assert.Equal(t, tc.wantIO, stats.ZfsPools["arbitrary label"].HideIO)
|
||||
// Bind mounts and custom extra-filesystem names have the same identity.
|
||||
fs["dm-0"].Root = false
|
||||
fs["dm-0"].Mountpoint = "/extra-filesystems/storage"
|
||||
fs["dm-0"].Name = "custom name"
|
||||
stats.ZfsPools["arbitrary label"] = &system.ZfsPool{}
|
||||
zm.markDuplicateCharts(stats, fs, func(string) string { return "fs1" })
|
||||
assert.Equal(t, tc.wantUsage, stats.ZfsPools["arbitrary label"].HideUsage)
|
||||
assert.Equal(t, tc.wantIO, stats.ZfsPools["arbitrary label"].HideIO)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBtrfsPoolIdentities(t *testing.T) {
|
||||
old := btrfsFilesystems
|
||||
t.Cleanup(func() { btrfsFilesystems = old })
|
||||
label := "tank"
|
||||
btrfsFilesystems = func() ([]btrfs.Filesystem, error) {
|
||||
return []btrfs.Filesystem{
|
||||
{UUID: "11111111-1111-4111-8111-111111111111", Name: label, Size: 100, Health: "ONLINE", NRead: 100, Devices: []btrfs.Device{{Name: "first"}}},
|
||||
{UUID: "22222222-2222-4222-8222-222222222222", Name: "tank", Size: 200, Health: "DEGRADED", NRead: 200, Devices: []btrfs.Device{{Name: "second"}}},
|
||||
}, nil
|
||||
}
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolStatsFn: func() ([]zfs.PoolStat, error) { return []zfs.PoolStat{{Name: "tank", Size: 300}}, nil },
|
||||
kernelStatsFn: func() ([]zfs.PoolKernelStat, error) { return []zfs.PoolKernelStat{{Name: "tank", NRead: 300}}, nil },
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) {
|
||||
return []zfs.PoolStatus{{Name: "tank", Vdevs: []zfs.VdevStatus{{Name: "zfs-device"}}}}, nil
|
||||
},
|
||||
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return []zfs.Dataset{{Name: "tank/data"}}, nil }}, newBtrfsBackend()}}
|
||||
first := "b:11111111-1111-4111-8111-111111111111"
|
||||
second := "b:22222222-2222-4222-8222-222222222222"
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 3)
|
||||
assert.Contains(t, stats.ZfsPools, "tank")
|
||||
assert.Equal(t, "ONLINE", stats.ZfsPools[first].Health)
|
||||
assert.Equal(t, "DEGRADED", stats.ZfsPools[second].Health)
|
||||
assert.Equal(t, uint64(100), zm.backends[1].kernelSamples[first].nread)
|
||||
assert.Equal(t, uint64(200), zm.backends[1].kernelSamples[second].nread)
|
||||
detail := zm.GetDetail(true)
|
||||
require.Len(t, detail.Pools, 3)
|
||||
assert.Equal(t, "zfs-device", detail.Pools[0].Vdevs[0].Name)
|
||||
assert.Len(t, detail.Pools[0].Datasets, 1)
|
||||
assert.Equal(t, "first", detail.Pools[1].Vdevs[0].Name)
|
||||
assert.Empty(t, detail.Pools[1].Datasets)
|
||||
assert.Equal(t, "second", detail.Pools[2].Vdevs[0].Name)
|
||||
label = "renamed"
|
||||
zm.backends[0].lastPoolStats = time.Time{}
|
||||
zm.backends[1].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 3)
|
||||
assert.Equal(t, "renamed", stats.ZfsPools[first].DisplayName)
|
||||
assert.Equal(t, first, zm.GetDetail(true).Pools[1].Name)
|
||||
assert.Equal(t, "renamed", zm.GetDetail(true).Pools[1].DisplayName)
|
||||
}
|
||||
+152
-61
@@ -4,15 +4,17 @@ import (
|
||||
"bufio"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"os"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/agent/battery"
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
|
||||
@@ -22,13 +24,6 @@ import (
|
||||
"github.com/shirou/gopsutil/v4/mem"
|
||||
)
|
||||
|
||||
// prevDisk stores previous per-device disk counters for a given cache interval
|
||||
type prevDisk struct {
|
||||
readBytes uint64
|
||||
writeBytes uint64
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// Sets initial / non-changing values about the host system
|
||||
func (a *Agent) refreshSystemDetails() {
|
||||
a.systemInfo.AgentVersion = beszel.Version
|
||||
@@ -38,7 +33,11 @@ func (a *Agent) refreshSystemDetails() {
|
||||
|
||||
if a.dockerManager != nil {
|
||||
a.systemDetails.Podman = a.dockerManager.IsPodman()
|
||||
hostInfo, _ = a.dockerManager.GetHostInfo()
|
||||
// Docker's host info describes the machine its daemon runs on. On macOS and
|
||||
// Windows that is a Linux VM, so its CPU and memory totals are not this host's.
|
||||
if runtime.GOOS != "darwin" && runtime.GOOS != "windows" {
|
||||
hostInfo, _ = a.dockerManager.GetHostInfo()
|
||||
}
|
||||
}
|
||||
|
||||
a.systemDetails.Hostname, _ = os.Hostname()
|
||||
@@ -85,6 +84,12 @@ func (a *Agent) refreshSystemDetails() {
|
||||
if info, err := cpu.Info(); err == nil && len(info) > 0 {
|
||||
a.systemDetails.CpuModel = info[0].ModelName
|
||||
}
|
||||
// gopsutil doesn't parse the "cpu model" field from /proc/cpuinfo, which
|
||||
// is the only source of the CPU model name on MIPS. Fall back to reading
|
||||
// it directly when ModelName is empty.
|
||||
if a.systemDetails.CpuModel == "" {
|
||||
a.systemDetails.CpuModel = getCpuModelFromCpuinfo()
|
||||
}
|
||||
// cores / threads
|
||||
cores, _ := cpu.Counts(false)
|
||||
threads := hostInfo.NCPU
|
||||
@@ -107,33 +112,58 @@ func (a *Agent) refreshSystemDetails() {
|
||||
}
|
||||
|
||||
// zfs
|
||||
if _, err := getARCSize(); err != nil {
|
||||
if _, err := zfs.ARCSize(); err != nil {
|
||||
slog.Debug("Not monitoring ZFS ARC", "err", err)
|
||||
} else {
|
||||
a.zfs = true
|
||||
}
|
||||
}
|
||||
|
||||
// attachSystemDetails returns details only for fresh default-interval responses.
|
||||
func (a *Agent) attachSystemDetails(data *system.CombinedData, cacheTimeMs uint16, includeRequested bool) *system.CombinedData {
|
||||
if cacheTimeMs != defaultDataCacheTimeMs || (!includeRequested && !a.detailsDirty) {
|
||||
return data
|
||||
}
|
||||
|
||||
// copy data to avoid adding details to the original cached struct
|
||||
response := *data
|
||||
response.Details = &a.systemDetails
|
||||
a.detailsDirty = false
|
||||
return &response
|
||||
}
|
||||
|
||||
// updateSystemDetails applies a mutation to the static details payload and marks
|
||||
// it for inclusion on the next fresh default-interval response.
|
||||
func (a *Agent) updateSystemDetails(updateFunc func(details *system.Details)) {
|
||||
updateFunc(&a.systemDetails)
|
||||
a.detailsDirty = true
|
||||
}
|
||||
|
||||
// Returns current info, stats about the host system
|
||||
func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
var systemStats system.Stats
|
||||
|
||||
// battery
|
||||
if batteryPercent, batteryState, err := battery.GetBatteryStats(); err == nil {
|
||||
systemStats.Battery[0] = batteryPercent
|
||||
systemStats.Battery[1] = batteryState
|
||||
if batteries, err := battery.GetBatteryStats(); err == nil {
|
||||
systemStats.Batteries = make(map[string]uint8, len(batteries))
|
||||
for _, device := range batteries {
|
||||
systemStats.Batteries[device.Name] = device.Percent
|
||||
}
|
||||
if primary, ok := battery.Primary(batteries); ok {
|
||||
systemStats.Battery = [2]uint8{primary.Percent, primary.State}
|
||||
}
|
||||
}
|
||||
|
||||
// cpu metrics
|
||||
cpuMetrics, err := getCpuMetrics(cacheTimeMs)
|
||||
if err == nil {
|
||||
systemStats.Cpu = twoDecimals(cpuMetrics.Total)
|
||||
systemStats.Cpu = utils.TwoDecimals(cpuMetrics.Total)
|
||||
systemStats.CpuBreakdown = []float64{
|
||||
twoDecimals(cpuMetrics.User),
|
||||
twoDecimals(cpuMetrics.System),
|
||||
twoDecimals(cpuMetrics.Iowait),
|
||||
twoDecimals(cpuMetrics.Steal),
|
||||
twoDecimals(cpuMetrics.Idle),
|
||||
utils.TwoDecimals(cpuMetrics.User),
|
||||
utils.TwoDecimals(cpuMetrics.System),
|
||||
utils.TwoDecimals(cpuMetrics.Iowait),
|
||||
utils.TwoDecimals(cpuMetrics.Steal),
|
||||
utils.TwoDecimals(cpuMetrics.Idle),
|
||||
}
|
||||
} else {
|
||||
slog.Error("Error getting cpu metrics", "err", err)
|
||||
@@ -146,9 +176,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
|
||||
// load average
|
||||
if avgstat, err := load.Avg(); err == nil {
|
||||
systemStats.LoadAvg[0] = avgstat.Load1
|
||||
systemStats.LoadAvg[1] = avgstat.Load5
|
||||
systemStats.LoadAvg[2] = avgstat.Load15
|
||||
systemStats.LoadAvg[0] = utils.TwoDecimals(avgstat.Load1)
|
||||
systemStats.LoadAvg[1] = utils.TwoDecimals(avgstat.Load5)
|
||||
systemStats.LoadAvg[2] = utils.TwoDecimals(avgstat.Load15)
|
||||
slog.Debug("Load average", "5m", avgstat.Load5, "15m", avgstat.Load15)
|
||||
} else {
|
||||
slog.Error("Error getting load average", "err", err)
|
||||
@@ -156,21 +186,11 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
|
||||
// memory
|
||||
if v, err := mem.VirtualMemory(); err == nil {
|
||||
used, cacheBuff, swapUsed := calculateHostMemoryUsage(v, a.memCalc == "htop")
|
||||
// swap
|
||||
systemStats.Swap = bytesToGigabytes(v.SwapTotal)
|
||||
systemStats.SwapUsed = bytesToGigabytes(v.SwapTotal - v.SwapFree - v.SwapCached)
|
||||
// cache + buffers value for default mem calculation
|
||||
// note: gopsutil automatically adds SReclaimable to v.Cached
|
||||
cacheBuff := v.Cached + v.Buffers - v.Shared
|
||||
if cacheBuff <= 0 {
|
||||
cacheBuff = max(v.Total-v.Free-v.Used, 0)
|
||||
}
|
||||
// htop memory calculation overrides (likely outdated as of mid 2025)
|
||||
if a.memCalc == "htop" {
|
||||
// cacheBuff = v.Cached + v.Buffers - v.Shared
|
||||
v.Used = v.Total - (v.Free + cacheBuff)
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
}
|
||||
systemStats.Swap = utils.BytesToGigabytes(v.SwapTotal)
|
||||
systemStats.SwapUsed = utils.BytesToGigabytes(swapUsed)
|
||||
v.Used = used
|
||||
// if a.memCalc == "legacy" {
|
||||
// v.Used = v.Total - v.Free - v.Buffers - v.Cached
|
||||
// cacheBuff = v.Total - v.Free - v.Used
|
||||
@@ -178,16 +198,20 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
// }
|
||||
// subtract ZFS ARC size from used memory and add as its own category
|
||||
if a.zfs {
|
||||
if arcSize, _ := getARCSize(); arcSize > 0 && arcSize < v.Used {
|
||||
if arcSize, _ := zfs.ARCSize(); arcSize > 0 && arcSize < v.Used {
|
||||
v.Used = v.Used - arcSize
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
systemStats.MemZfsArc = bytesToGigabytes(arcSize)
|
||||
systemStats.MemZfsArc = utils.BytesToGigabytes(arcSize)
|
||||
}
|
||||
}
|
||||
systemStats.Mem = bytesToGigabytes(v.Total)
|
||||
systemStats.MemBuffCache = bytesToGigabytes(cacheBuff)
|
||||
systemStats.MemUsed = bytesToGigabytes(v.Used)
|
||||
systemStats.MemPct = twoDecimals(v.UsedPercent)
|
||||
if v.Total > 0 {
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
} else {
|
||||
v.UsedPercent = 0
|
||||
}
|
||||
systemStats.Mem = utils.BytesToGigabytes(v.Total)
|
||||
systemStats.MemBuffCache = utils.BytesToGigabytes(cacheBuff)
|
||||
systemStats.MemUsed = utils.BytesToGigabytes(v.Used)
|
||||
systemStats.MemPct = utils.TwoDecimals(v.UsedPercent)
|
||||
}
|
||||
|
||||
// disk usage
|
||||
@@ -196,6 +220,10 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
// disk i/o (cache-aware per interval)
|
||||
a.updateDiskIo(cacheTimeMs, &systemStats)
|
||||
|
||||
// storage pool stats
|
||||
a.storagePoolManager.Update(&systemStats)
|
||||
a.storagePoolManager.markDuplicateCharts(&systemStats, a.fsStats, btrfs.MountID)
|
||||
|
||||
// network stats (per cache interval)
|
||||
a.updateNetworkStats(cacheTimeMs, &systemStats)
|
||||
|
||||
@@ -203,6 +231,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
// TODO: maybe refactor to methods on systemStats
|
||||
a.updateTemperatures(&systemStats)
|
||||
|
||||
// fan speeds (Linux-only; sysfs hwmon)
|
||||
a.updateFans(&systemStats)
|
||||
|
||||
// GPU data
|
||||
if a.gpuManager != nil {
|
||||
// reset high gpu percent
|
||||
@@ -243,37 +274,97 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
a.systemInfo.MemPct = systemStats.MemPct
|
||||
a.systemInfo.DiskPct = systemStats.DiskPct
|
||||
a.systemInfo.Battery = systemStats.Battery
|
||||
a.systemInfo.Uptime, _ = host.Uptime()
|
||||
a.systemInfo.Uptime, _ = getUptime()
|
||||
a.systemInfo.BandwidthBytes = systemStats.Bandwidth[0] + systemStats.Bandwidth[1]
|
||||
a.systemInfo.Threads = a.systemDetails.Threads
|
||||
|
||||
return systemStats
|
||||
}
|
||||
|
||||
// Returns the size of the ZFS ARC memory cache in bytes
|
||||
func getARCSize() (uint64, error) {
|
||||
file, err := os.Open("/proc/spl/kstat/zfs/arcstats")
|
||||
// cpuModelFallbackKeys are the field names to look for in /proc/cpuinfo when
|
||||
// gopsutil fails to return a ModelName. The "cpu model" key is used on MIPS
|
||||
// (e.g. "MIPS 1004Kc V2.15"), while "system type" provides SoC information
|
||||
// on various embedded architectures.
|
||||
var cpuModelFallbackKeys = []string{"cpu model", "system type"}
|
||||
|
||||
// getCpuModelFromCpuinfo reads /proc/cpuinfo and returns a CPU model string.
|
||||
// This is a fallback for architectures where gopsutil's cpu.Info() does not
|
||||
// populate ModelName, most notably MIPS.
|
||||
func getCpuModelFromCpuinfo() string {
|
||||
file, err := os.Open("/proc/cpuinfo")
|
||||
if err != nil {
|
||||
return 0, err
|
||||
return ""
|
||||
}
|
||||
defer file.Close()
|
||||
return parseCpuModel(file)
|
||||
}
|
||||
|
||||
// Scan the lines
|
||||
scanner := bufio.NewScanner(file)
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if strings.HasPrefix(line, "size") {
|
||||
// Example line: size 4 15032385536
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 3 {
|
||||
return 0, err
|
||||
// parseCpuModel scans r (expected to be /proc/cpuinfo content) and returns
|
||||
// a combined CPU model string. It collects values from all matching keys
|
||||
// and joins them with " / " when multiple are found.
|
||||
func parseCpuModel(r io.Reader) string {
|
||||
lines := readLines(r)
|
||||
var parts []string
|
||||
for _, key := range cpuModelFallbackKeys {
|
||||
for _, line := range lines {
|
||||
after, found := strings.CutPrefix(line, key)
|
||||
if !found {
|
||||
continue
|
||||
}
|
||||
after = strings.TrimSpace(after)
|
||||
if len(after) < 2 || after[0] != ':' {
|
||||
continue
|
||||
}
|
||||
if value := strings.TrimSpace(after[1:]); value != "" {
|
||||
parts = append(parts, value)
|
||||
break
|
||||
}
|
||||
// Return the size as uint64
|
||||
return strconv.ParseUint(fields[2], 10, 64)
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, " / ")
|
||||
}
|
||||
|
||||
return 0, fmt.Errorf("failed to parse size field")
|
||||
// readLines reads all lines from r into a slice.
|
||||
func readLines(r io.Reader) []string {
|
||||
scanner := bufio.NewScanner(r)
|
||||
var lines []string
|
||||
for scanner.Scan() {
|
||||
lines = append(lines, scanner.Text())
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
// calculateHostMemoryUsage derives counters defensively because /proc/meminfo may
|
||||
// change while gopsutil reads it. Invalid unsigned subtractions saturate at zero.
|
||||
func calculateHostMemoryUsage(v *mem.VirtualMemoryStat, htop bool) (used, cacheBuff, swapUsed uint64) {
|
||||
used = v.Used
|
||||
if used > v.Total {
|
||||
used = saturatingSub(v.Total, v.Available)
|
||||
}
|
||||
|
||||
// gopsutil automatically adds SReclaimable to Cached.
|
||||
cacheBuff = min(v.Cached, v.Total)
|
||||
cacheBuff += min(v.Buffers, v.Total-cacheBuff)
|
||||
cacheBuff = saturatingSub(cacheBuff, min(v.Shared, v.Total))
|
||||
if v.Cached == 0 && v.Buffers == 0 {
|
||||
cacheBuff = saturatingSub(v.Total, v.Free, used)
|
||||
}
|
||||
if htop {
|
||||
used = saturatingSub(v.Total, v.Free, cacheBuff)
|
||||
}
|
||||
// Cached swap pages still occupy swap slots and are included in `free`'s used value.
|
||||
return used, cacheBuff, saturatingSub(v.SwapTotal, v.SwapFree)
|
||||
}
|
||||
|
||||
// saturatingSub subtracts each value, returning zero on underflow.
|
||||
func saturatingSub(value uint64, subtrahends ...uint64) uint64 {
|
||||
for _, subtrahend := range subtrahends {
|
||||
if subtrahend > value {
|
||||
return 0
|
||||
}
|
||||
value -= subtrahend
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
// getOsPrettyName attempts to get the pretty OS name from /etc/os-release on Linux systems
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/shirou/gopsutil/v4/mem"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestGatherStatsDoesNotAttachDetailsToCachedRequests(t *testing.T) {
|
||||
agent := &Agent{
|
||||
cache: NewSystemDataCache(),
|
||||
systemDetails: system.Details{Hostname: "updated-host", Podman: true},
|
||||
detailsDirty: true,
|
||||
}
|
||||
cached := &system.CombinedData{
|
||||
Info: system.Info{Hostname: "cached-host"},
|
||||
}
|
||||
agent.cache.Set(cached, defaultDataCacheTimeMs)
|
||||
|
||||
response := agent.gatherStats(common.DataRequestOptions{CacheTimeMs: defaultDataCacheTimeMs})
|
||||
|
||||
assert.Same(t, cached, response)
|
||||
assert.Nil(t, response.Details)
|
||||
assert.True(t, agent.detailsDirty)
|
||||
assert.Equal(t, "cached-host", response.Info.Hostname)
|
||||
assert.Nil(t, cached.Details)
|
||||
|
||||
secondResponse := agent.gatherStats(common.DataRequestOptions{CacheTimeMs: defaultDataCacheTimeMs})
|
||||
assert.Same(t, cached, secondResponse)
|
||||
assert.Nil(t, secondResponse.Details)
|
||||
}
|
||||
|
||||
func TestCalculateHostMemoryUsage(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
memory mem.VirtualMemoryStat
|
||||
htop bool
|
||||
used, cacheBuff, swapUsed uint64
|
||||
}{
|
||||
{
|
||||
name: "normal",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 40, Used: 60, Free: 20, Cached: 25, Buffers: 10, Shared: 5, SwapTotal: 20, SwapFree: 8, SwapCached: 2},
|
||||
used: 60,
|
||||
cacheBuff: 30,
|
||||
swapUsed: 12,
|
||||
},
|
||||
{
|
||||
name: "inconsistent counters saturate",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 110, Used: ^uint64(0) - 9, Free: 90, Cached: 5, Buffers: 10, Shared: 20, SwapTotal: 10, SwapFree: 9, SwapCached: 2},
|
||||
used: 0,
|
||||
cacheBuff: 0,
|
||||
swapUsed: 1,
|
||||
},
|
||||
{
|
||||
name: "htop subtraction saturates",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 20, Used: 80, Free: 90, Cached: 20, Buffers: 5, SwapTotal: 30, SwapFree: 10, SwapCached: 5},
|
||||
htop: true,
|
||||
used: 0,
|
||||
cacheBuff: 25,
|
||||
swapUsed: 20,
|
||||
},
|
||||
{
|
||||
name: "zero cache from shared cancellation does not fall back",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Used: 60, Free: 10, Cached: 20, Buffers: 10, Shared: 30},
|
||||
used: 60,
|
||||
cacheBuff: 0,
|
||||
},
|
||||
{
|
||||
name: "absent cache counters use fallback",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Used: 60, Free: 10},
|
||||
used: 60,
|
||||
cacheBuff: 30,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
used, cacheBuff, swapUsed := calculateHostMemoryUsage(&tt.memory, tt.htop)
|
||||
assert.Equal(t, tt.used, used)
|
||||
assert.Equal(t, tt.cacheBuff, cacheBuff)
|
||||
assert.Equal(t, tt.swapUsed, swapUsed)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUpdateSystemDetailsMarksDetailsDirty(t *testing.T) {
|
||||
agent := &Agent{}
|
||||
|
||||
agent.updateSystemDetails(func(details *system.Details) {
|
||||
details.Hostname = "updated-host"
|
||||
details.Podman = true
|
||||
})
|
||||
|
||||
assert.True(t, agent.detailsDirty)
|
||||
assert.Equal(t, "updated-host", agent.systemDetails.Hostname)
|
||||
assert.True(t, agent.systemDetails.Podman)
|
||||
|
||||
original := &system.CombinedData{}
|
||||
realTimeResponse := agent.attachSystemDetails(original, 1000, true)
|
||||
assert.Same(t, original, realTimeResponse)
|
||||
assert.Nil(t, realTimeResponse.Details)
|
||||
assert.True(t, agent.detailsDirty)
|
||||
|
||||
response := agent.attachSystemDetails(original, defaultDataCacheTimeMs, false)
|
||||
require.NotNil(t, response.Details)
|
||||
assert.NotSame(t, original, response)
|
||||
assert.Equal(t, "updated-host", response.Details.Hostname)
|
||||
assert.True(t, response.Details.Podman)
|
||||
assert.False(t, agent.detailsDirty)
|
||||
assert.Nil(t, original.Details)
|
||||
}
|
||||
|
||||
func TestParseCpuModel(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
expected string
|
||||
}{
|
||||
{
|
||||
name: "MIPS with both cpu model and system type",
|
||||
input: `system type : MediaTek MT7621 ver:1 eco:3
|
||||
machine : ASUS RT-AX53U
|
||||
processor : 0
|
||||
cpu model : MIPS 1004Kc V2.15
|
||||
BogoMIPS : 586.13
|
||||
wait instruction : yes`,
|
||||
expected: "MIPS 1004Kc V2.15 / MediaTek MT7621 ver:1 eco:3",
|
||||
},
|
||||
{
|
||||
name: "MIPS with different SoC",
|
||||
input: `system type : Atheros AR7161 rev 2
|
||||
machine : NETGEAR WNDR3700
|
||||
processor : 0
|
||||
cpu model : MIPS 24Kc V7.4
|
||||
BogoMIPS : 452.19`,
|
||||
expected: "MIPS 24Kc V7.4 / Atheros AR7161 rev 2",
|
||||
},
|
||||
{
|
||||
name: "only system type when cpu model missing",
|
||||
input: `system type : Broadcom BCM47xx
|
||||
processor : 0
|
||||
BogoMIPS : 296.11`,
|
||||
expected: "Broadcom BCM47xx",
|
||||
},
|
||||
{
|
||||
name: "only cpu model when system type missing",
|
||||
input: `processor : 0
|
||||
cpu model : MIPS 34Kc V2.15
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "MIPS 34Kc V2.15",
|
||||
},
|
||||
{
|
||||
name: "x86 cpuinfo returns empty",
|
||||
input: `processor : 0
|
||||
vendor_id : GenuineIntel
|
||||
cpu family : 6
|
||||
model : 142
|
||||
model name : Intel(R) Core(TM) i5-8250U CPU @ 1.60GHz
|
||||
stepping : 10`,
|
||||
expected: "",
|
||||
},
|
||||
{
|
||||
name: "empty input",
|
||||
input: "",
|
||||
expected: "",
|
||||
},
|
||||
{
|
||||
name: "cpu model with extra whitespace",
|
||||
input: `processor : 0
|
||||
cpu model : MIPS 34Kc V2.15
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "MIPS 34Kc V2.15",
|
||||
},
|
||||
{
|
||||
name: "cpu model without value",
|
||||
input: `processor : 0
|
||||
cpu model :
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := parseCpuModel(strings.NewReader(tt.input))
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
+4
-3
@@ -15,6 +15,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/coreos/go-systemd/v22/dbus"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/systemd"
|
||||
)
|
||||
|
||||
@@ -49,7 +50,7 @@ func isSystemdAvailable() bool {
|
||||
|
||||
// newSystemdManager creates a new systemdManager.
|
||||
func newSystemdManager() (*systemdManager, error) {
|
||||
if skipSystemd, _ := GetEnv("SKIP_SYSTEMD"); skipSystemd == "true" {
|
||||
if skipSystemd, _ := utils.GetEnv("SKIP_SYSTEMD"); skipSystemd == "true" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
@@ -294,13 +295,13 @@ func unescapeServiceName(name string) string {
|
||||
// otherwise defaults to "*service".
|
||||
func getServicePatterns() []string {
|
||||
patterns := []string{}
|
||||
if envPatterns, _ := GetEnv("SERVICE_PATTERNS"); envPatterns != "" {
|
||||
if envPatterns, _ := utils.GetEnv("SERVICE_PATTERNS"); envPatterns != "" {
|
||||
for pattern := range strings.SplitSeq(envPatterns, ",") {
|
||||
pattern = strings.TrimSpace(pattern)
|
||||
if pattern == "" {
|
||||
continue
|
||||
}
|
||||
if !strings.HasSuffix(pattern, ".service") {
|
||||
if !strings.HasSuffix(pattern, "timer") && !strings.HasSuffix(pattern, ".service") {
|
||||
pattern += ".service"
|
||||
}
|
||||
patterns = append(patterns, pattern)
|
||||
|
||||
+9
-12
@@ -156,20 +156,23 @@ func TestGetServicePatterns(t *testing.T) {
|
||||
expected: []string{"*nginx*.service", "*apache*.service"},
|
||||
cleanupEnvVars: true,
|
||||
},
|
||||
{
|
||||
name: "opt into timer monitoring",
|
||||
prefixedEnv: "nginx.service,docker,apache.timer",
|
||||
unprefixedEnv: "",
|
||||
expected: []string{"nginx.service", "docker.service", "apache.timer"},
|
||||
cleanupEnvVars: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
// Clean up any existing env vars
|
||||
os.Unsetenv("BESZEL_AGENT_SERVICE_PATTERNS")
|
||||
os.Unsetenv("SERVICE_PATTERNS")
|
||||
|
||||
// Set up environment variables
|
||||
if tt.prefixedEnv != "" {
|
||||
os.Setenv("BESZEL_AGENT_SERVICE_PATTERNS", tt.prefixedEnv)
|
||||
t.Setenv("BESZEL_AGENT_SERVICE_PATTERNS", tt.prefixedEnv)
|
||||
}
|
||||
if tt.unprefixedEnv != "" {
|
||||
os.Setenv("SERVICE_PATTERNS", tt.unprefixedEnv)
|
||||
t.Setenv("SERVICE_PATTERNS", tt.unprefixedEnv)
|
||||
}
|
||||
|
||||
// Run the function
|
||||
@@ -177,12 +180,6 @@ func TestGetServicePatterns(t *testing.T) {
|
||||
|
||||
// Verify results
|
||||
assert.Equal(t, tt.expected, result, "Patterns should match expected values")
|
||||
|
||||
// Cleanup
|
||||
if tt.cleanupEnvVars {
|
||||
os.Unsetenv("BESZEL_AGENT_SERVICE_PATTERNS")
|
||||
os.Unsetenv("SERVICE_PATTERNS")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"json_format_version": [1, 0],
|
||||
"smartctl": {
|
||||
"version": [7, 4],
|
||||
"argv": ["smartctl", "-aix", "-j", "IOService:/AppleARMPE/arm-io@10F00000/AppleT810xIO/ans@77400000/AppleASCWrapV4/iop-ans-nub/RTBuddy(ANS2)/RTBuddyService/AppleANS3NVMeController/NS_01@1"],
|
||||
"exit_status": 4
|
||||
},
|
||||
"device": {
|
||||
"name": "IOService:/AppleARMPE/arm-io@10F00000/AppleT810xIO/ans@77400000/AppleASCWrapV4/iop-ans-nub/RTBuddy(ANS2)/RTBuddyService/AppleANS3NVMeController/NS_01@1",
|
||||
"info_name": "IOService:/AppleARMPE/arm-io@10F00000/AppleT810xIO/ans@77400000/AppleASCWrapV4/iop-ans-nub/RTBuddy(ANS2)/RTBuddyService/AppleANS3NVMeController/NS_01@1",
|
||||
"type": "nvme",
|
||||
"protocol": "NVMe"
|
||||
},
|
||||
"model_name": "APPLE SSD AP0256Q",
|
||||
"serial_number": "0ba0147940253c15",
|
||||
"firmware_version": "555",
|
||||
"smart_support": {
|
||||
"available": true,
|
||||
"enabled": true
|
||||
},
|
||||
"smart_status": {
|
||||
"passed": true,
|
||||
"nvme": {
|
||||
"value": 0
|
||||
}
|
||||
},
|
||||
"nvme_smart_health_information_log": {
|
||||
"critical_warning": 0,
|
||||
"temperature": 42,
|
||||
"available_spare": 100,
|
||||
"available_spare_threshold": 99,
|
||||
"percentage_used": 1,
|
||||
"data_units_read": 270189386,
|
||||
"data_units_written": 166753862,
|
||||
"host_reads": 7543766995,
|
||||
"host_writes": 3761621926,
|
||||
"controller_busy_time": 0,
|
||||
"power_cycles": 366,
|
||||
"power_on_hours": 2850,
|
||||
"unsafe_shutdowns": 195,
|
||||
"media_errors": 0,
|
||||
"num_err_log_entries": 0
|
||||
},
|
||||
"temperature": {
|
||||
"current": 42
|
||||
},
|
||||
"power_cycle_count": 366,
|
||||
"power_on_time": {
|
||||
"hours": 2850
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
tank 12000000000000 11999000000000 /tank
|
||||
tank/apps 1000000000000 11999000000000 /tank/apps
|
||||
tank/backup 2000000000000 11999000000000 /tank/backup
|
||||
tank/media 1000000000000 11999000000000 /tank/my media
|
||||
rpool 900000000000 300000000000 -
|
||||
rpool/ROOT 1000000000 300000000000 -
|
||||
rpool/ROOT/pve-1 890000000000 300000000000 /
|
||||
rpool/data 9000000000 300000000000 -
|
||||
rpool/data/subvol-100-disk-0 400000000000 300000000000 /subvol-100-disk-0
|
||||
@@ -0,0 +1,2 @@
|
||||
tank 23999000000000 12000000000000 11999000000000 ONLINE
|
||||
rpool 1200000000000 900000000000 300000000000 DEGRADED
|
||||
@@ -0,0 +1,29 @@
|
||||
pool: tank
|
||||
state: ONLINE
|
||||
scan: scrub repaired 0B in 00:05:12 with 0 errors on Sun Jun 1 02:00:12 2025
|
||||
config:
|
||||
|
||||
NAME STATE READ WRITE CKSUM
|
||||
tank ONLINE 0 0 0
|
||||
mirror-0 ONLINE 0 0 0
|
||||
sda ONLINE 0 0 0
|
||||
sdb ONLINE 0 0 0
|
||||
|
||||
errors: No known data errors
|
||||
|
||||
pool: rpool
|
||||
state: DEGRADED
|
||||
status: One or more devices could not be used because the label is missing or
|
||||
invalid. Sufficient replicas exist for the pool to continue functioning in a
|
||||
degraded state.
|
||||
scan: scrub in progress since Sun Jun 8 01:00:00 2025
|
||||
10.00% done, 01:30:00 to go, 0.00/s
|
||||
config:
|
||||
|
||||
NAME STATE READ WRITE CKSUM
|
||||
rpool DEGRADED 0 0 0
|
||||
mirror-0 DEGRADED 0 0 0
|
||||
sda ONLINE 0 0 0
|
||||
sdb FAULTED 1 2 3
|
||||
|
||||
errors: 1 data errors, use '-v' for a list
|
||||
@@ -0,0 +1,44 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"math"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/host"
|
||||
)
|
||||
|
||||
// uptimeFilePath is a variable so tests can point it at a fixture.
|
||||
var uptimeFilePath = "/proc/uptime"
|
||||
|
||||
// getUptime returns the system uptime in seconds.
|
||||
//
|
||||
// This reads /proc/uptime instead of using host.Uptime(), which calls the
|
||||
// sysinfo(2) syscall. Inside an LXC container lxcfs virtualizes /proc/uptime
|
||||
// but cannot intercept a syscall, so sysinfo(2) reports the host's uptime
|
||||
// rather than the container's.
|
||||
//
|
||||
// Falls back to host.Uptime() if /proc/uptime is missing or unparseable, so
|
||||
// behavior is unchanged anywhere the file isn't available.
|
||||
func getUptime() (uint64, error) {
|
||||
data, err := os.ReadFile(uptimeFilePath)
|
||||
if err != nil {
|
||||
return host.Uptime()
|
||||
}
|
||||
fields := strings.Fields(string(data))
|
||||
if len(fields) == 0 {
|
||||
return host.Uptime()
|
||||
}
|
||||
seconds, err := strconv.ParseFloat(fields[0], 64)
|
||||
if err != nil ||
|
||||
math.IsNaN(seconds) ||
|
||||
math.IsInf(seconds, 0) ||
|
||||
seconds < 0 ||
|
||||
seconds >= 1<<64 {
|
||||
return host.Uptime()
|
||||
}
|
||||
return uint64(seconds), nil
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGetUptimeFromProc(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
contents string
|
||||
want uint64
|
||||
}{
|
||||
{"typical", "12345.67 98765.43\n", 12345},
|
||||
{"zero", "0.00 0.00\n", 0},
|
||||
{"no trailing newline", "42.99 7.00", 42},
|
||||
{"single field", "600.5", 600},
|
||||
{"large value", "266030.12 1000000.00\n", 266030},
|
||||
}
|
||||
|
||||
prev := uptimeFilePath
|
||||
t.Cleanup(func() { uptimeFilePath = prev })
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "uptime")
|
||||
if err := os.WriteFile(path, []byte(tt.contents), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
uptimeFilePath = path
|
||||
|
||||
got, err := getUptime()
|
||||
if err != nil {
|
||||
t.Fatalf("getUptime() returned error: %v", err)
|
||||
}
|
||||
if got != tt.want {
|
||||
t.Errorf("getUptime() = %d, want %d", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func writeUptime(contents string) func(t *testing.T) string {
|
||||
return func(t *testing.T) string {
|
||||
path := filepath.Join(t.TempDir(), "uptime")
|
||||
if err := os.WriteFile(path, []byte(contents), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
}
|
||||
|
||||
// Malformed, missing, or out-of-range input must fall back to host.Uptime()
|
||||
// rather than returning a bogus value, so the agent still reports something sane.
|
||||
func TestGetUptimeFallsBack(t *testing.T) {
|
||||
prev := uptimeFilePath
|
||||
t.Cleanup(func() { uptimeFilePath = prev })
|
||||
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
prepare func(t *testing.T) string
|
||||
}{
|
||||
{"missing file", func(t *testing.T) string {
|
||||
return filepath.Join(t.TempDir(), "does-not-exist")
|
||||
}},
|
||||
{"empty file", func(t *testing.T) string {
|
||||
path := filepath.Join(t.TempDir(), "uptime")
|
||||
if err := os.WriteFile(path, nil, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}},
|
||||
{"unparseable", func(t *testing.T) string {
|
||||
path := filepath.Join(t.TempDir(), "uptime")
|
||||
if err := os.WriteFile(path, []byte("not-a-number 1.0\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}},
|
||||
{"NaN", writeUptime("NaN 1.0\n")},
|
||||
{"positive infinity", writeUptime("+Inf 1.0\n")},
|
||||
{"negative infinity", writeUptime("-Inf 1.0\n")},
|
||||
{"negative", writeUptime("-42.5 1.0\n")},
|
||||
{"exceeds uint64 range", writeUptime("1e20 1.0\n")},
|
||||
} {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
uptimeFilePath = tt.prepare(t)
|
||||
|
||||
got, err := getUptime()
|
||||
if err != nil {
|
||||
t.Fatalf("getUptime() returned error: %v", err)
|
||||
}
|
||||
if got == 0 {
|
||||
t.Error("getUptime() = 0, expected fallback to host.Uptime()")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
import "github.com/shirou/gopsutil/v4/host"
|
||||
|
||||
// getUptime returns the system uptime in seconds.
|
||||
func getUptime() (uint64, error) {
|
||||
return host.Uptime()
|
||||
}
|
||||
@@ -1,15 +0,0 @@
|
||||
package agent
|
||||
|
||||
import "math"
|
||||
|
||||
func bytesToMegabytes(b float64) float64 {
|
||||
return twoDecimals(b / 1048576)
|
||||
}
|
||||
|
||||
func bytesToGigabytes(b uint64) float64 {
|
||||
return twoDecimals(float64(b) / 1073741824)
|
||||
}
|
||||
|
||||
func twoDecimals(value float64) float64 {
|
||||
return math.Round(value*100) / 100
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
// Package utils provides utility functions for the agent.
|
||||
package utils
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// GetEnv retrieves an environment variable with a "BESZEL_AGENT_" prefix, or falls back to the unprefixed key.
|
||||
func GetEnv(key string) (value string, exists bool) {
|
||||
if value, exists = os.LookupEnv("BESZEL_AGENT_" + key); exists {
|
||||
return value, exists
|
||||
}
|
||||
return os.LookupEnv(key)
|
||||
}
|
||||
|
||||
// BytesToMegabytes converts bytes to megabytes and rounds to two decimal places.
|
||||
func BytesToMegabytes(b float64) float64 {
|
||||
return TwoDecimals(b / 1048576)
|
||||
}
|
||||
|
||||
// BytesToGigabytes converts bytes to gigabytes and rounds to two decimal places.
|
||||
func BytesToGigabytes(b uint64) float64 {
|
||||
return TwoDecimals(float64(b) / 1073741824)
|
||||
}
|
||||
|
||||
// TwoDecimals rounds a float64 value to two decimal places.
|
||||
func TwoDecimals(value float64) float64 {
|
||||
return math.Round(value*100) / 100
|
||||
}
|
||||
|
||||
// func RoundFloat(val float64, precision uint) float64 {
|
||||
// ratio := math.Pow(10, float64(precision))
|
||||
// return math.Round(val*ratio) / ratio
|
||||
// }
|
||||
|
||||
// ReadStringFile returns trimmed file contents or empty string on error.
|
||||
func ReadStringFile(path string) string {
|
||||
content, _ := ReadStringFileOK(path)
|
||||
return content
|
||||
}
|
||||
|
||||
// ReadStringFileOK returns trimmed file contents and read success.
|
||||
func ReadStringFileOK(path string) (string, bool) {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", false
|
||||
}
|
||||
return strings.TrimSpace(string(b)), true
|
||||
}
|
||||
|
||||
// ReadStringFileLimited reads a file into a string with a maximum size (in bytes) to avoid
|
||||
// allocating large buffers and potential panics with pseudo-files when the size is misreported.
|
||||
func ReadStringFileLimited(path string, maxSize int) (string, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
buf := make([]byte, maxSize)
|
||||
n, err := f.Read(buf)
|
||||
if err != nil && err != io.EOF {
|
||||
return "", err
|
||||
}
|
||||
if n < 0 {
|
||||
return "", fmt.Errorf("%s returned negative bytes: %d", path, n)
|
||||
}
|
||||
return strings.TrimSpace(string(buf[:n])), nil
|
||||
}
|
||||
|
||||
// FileExists reports whether the given path exists.
|
||||
func FileExists(path string) bool {
|
||||
_, err := os.Stat(path)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// ReadUintFile parses a decimal uint64 value from a file.
|
||||
func ReadUintFile(path string) (uint64, bool) {
|
||||
raw, ok := ReadStringFileOK(path)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
parsed, err := strconv.ParseUint(raw, 10, 64)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
return parsed, true
|
||||
}
|
||||
|
||||
// LookPathHomebrew is like exec.LookPath but also checks Homebrew paths.
|
||||
func LookPathHomebrew(file string) (string, error) {
|
||||
foundPath, lookPathErr := exec.LookPath(file)
|
||||
if lookPathErr == nil {
|
||||
return foundPath, nil
|
||||
}
|
||||
var homebrewPath string
|
||||
switch runtime.GOOS {
|
||||
case "darwin":
|
||||
homebrewPath = filepath.Join("/opt", "homebrew", "bin", file)
|
||||
case "linux":
|
||||
homebrewPath = filepath.Join("/home", "linuxbrew", ".linuxbrew", "bin", file)
|
||||
}
|
||||
if homebrewPath != "" {
|
||||
if _, err := os.Stat(homebrewPath); err == nil {
|
||||
return homebrewPath, nil
|
||||
}
|
||||
}
|
||||
return "", lookPathErr
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
package utils
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
func TestTwoDecimals(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input float64
|
||||
expected float64
|
||||
}{
|
||||
{"round down", 1.234, 1.23},
|
||||
{"round half up", 1.235, 1.24}, // math.Round rounds half up
|
||||
{"no rounding needed", 1.23, 1.23},
|
||||
{"negative number", -1.235, -1.24}, // math.Round rounds half up (more negative)
|
||||
{"zero", 0.0, 0.0},
|
||||
{"large number", 123.456, 123.46}, // rounds 5 up
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := TwoDecimals(tt.input)
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBytesToMegabytes(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input float64
|
||||
expected float64
|
||||
}{
|
||||
{"1 MB", 1048576, 1.0},
|
||||
{"512 KB", 524288, 0.5},
|
||||
{"zero", 0, 0},
|
||||
{"large value", 1073741824, 1024}, // 1 GB = 1024 MB
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := BytesToMegabytes(tt.input)
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBytesToGigabytes(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input uint64
|
||||
expected float64
|
||||
}{
|
||||
{"1 GB", 1073741824, 1.0},
|
||||
{"512 MB", 536870912, 0.5},
|
||||
{"0 GB", 0, 0},
|
||||
{"2 GB", 2147483648, 2.0},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := BytesToGigabytes(tt.input)
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileFunctions(t *testing.T) {
|
||||
tmpDir := t.TempDir()
|
||||
testFilePath := filepath.Join(tmpDir, "test.txt")
|
||||
testContent := "hello world"
|
||||
|
||||
// Test FileExists (false)
|
||||
assert.False(t, FileExists(testFilePath))
|
||||
|
||||
// Test ReadStringFileOK (false)
|
||||
content, ok := ReadStringFileOK(testFilePath)
|
||||
assert.False(t, ok)
|
||||
assert.Empty(t, content)
|
||||
|
||||
// Test ReadStringFile (empty)
|
||||
assert.Empty(t, ReadStringFile(testFilePath))
|
||||
|
||||
// Write file
|
||||
err := os.WriteFile(testFilePath, []byte(testContent+"\n "), 0644)
|
||||
assert.NoError(t, err)
|
||||
|
||||
// Test FileExists (true)
|
||||
assert.True(t, FileExists(testFilePath))
|
||||
|
||||
// Test ReadStringFileOK (true)
|
||||
content, ok = ReadStringFileOK(testFilePath)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, testContent, content)
|
||||
|
||||
// Test ReadStringFile (content)
|
||||
assert.Equal(t, testContent, ReadStringFile(testFilePath))
|
||||
}
|
||||
|
||||
func TestReadUintFile(t *testing.T) {
|
||||
tmpDir := t.TempDir()
|
||||
|
||||
t.Run("valid uint", func(t *testing.T) {
|
||||
path := filepath.Join(tmpDir, "uint.txt")
|
||||
os.WriteFile(path, []byte(" 12345\n"), 0644)
|
||||
val, ok := ReadUintFile(path)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, uint64(12345), val)
|
||||
})
|
||||
|
||||
t.Run("invalid uint", func(t *testing.T) {
|
||||
path := filepath.Join(tmpDir, "invalid.txt")
|
||||
os.WriteFile(path, []byte("abc"), 0644)
|
||||
val, ok := ReadUintFile(path)
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, uint64(0), val)
|
||||
})
|
||||
|
||||
t.Run("missing file", func(t *testing.T) {
|
||||
path := filepath.Join(tmpDir, "missing.txt")
|
||||
val, ok := ReadUintFile(path)
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, uint64(0), val)
|
||||
})
|
||||
}
|
||||
|
||||
func TestGetEnv(t *testing.T) {
|
||||
key := "TEST_VAR"
|
||||
prefixedKey := "BESZEL_AGENT_" + key
|
||||
|
||||
t.Run("prefixed variable exists", func(t *testing.T) {
|
||||
t.Setenv(prefixedKey, "prefixed_val")
|
||||
t.Setenv(key, "unprefixed_val")
|
||||
|
||||
val, exists := GetEnv(key)
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "prefixed_val", val)
|
||||
})
|
||||
|
||||
t.Run("only unprefixed variable exists", func(t *testing.T) {
|
||||
t.Setenv(key, "unprefixed_val")
|
||||
|
||||
val, exists := GetEnv(key)
|
||||
assert.True(t, exists)
|
||||
assert.Equal(t, "unprefixed_val", val)
|
||||
})
|
||||
|
||||
t.Run("neither variable exists", func(t *testing.T) {
|
||||
val, exists := GetEnv(key)
|
||||
assert.False(t, exists)
|
||||
assert.Empty(t, val)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,164 @@
|
||||
// Package zfs provides functions to read ZFS statistics.
|
||||
package zfs
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
var commandTimeout = 10 * time.Second
|
||||
|
||||
var commandOutput = func(name string, args ...string) ([]byte, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), commandTimeout)
|
||||
defer cancel()
|
||||
cmd := exec.CommandContext(ctx, name, args...)
|
||||
cmd.Env = append(os.Environ(), "LC_ALL=C", "LANG=C")
|
||||
out, err := cmd.Output()
|
||||
if ctx.Err() != nil {
|
||||
return nil, fmt.Errorf("%s timed out after %s: %w", name, commandTimeout, ctx.Err())
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// ErrNoZfs is returned when the ZFS utilities or kernel interfaces are unavailable.
|
||||
var ErrNoZfs = errors.New("zfs utilities unavailable")
|
||||
|
||||
// PoolStat is a snapshot of a ZFS pool's capacity and health.
|
||||
type PoolStat struct {
|
||||
DisplayName string // optional friendly name; Name remains the stable key
|
||||
MountID string // Btrfs filesystem identity, empty for other backends
|
||||
IODevice string // sole Btrfs member device, if known
|
||||
Raw bool // physical accounting rather than usable filesystem space
|
||||
Name string
|
||||
Size uint64 // total capacity in bytes
|
||||
Alloc uint64 // allocated bytes
|
||||
Free uint64 // free bytes
|
||||
Health string // ONLINE, DEGRADED, FAULTED, ...
|
||||
}
|
||||
|
||||
// PoolKernelStat is the inexpensive pool telemetry exposed by the ZFS kernel.
|
||||
// NRead and NWrite are cumulative byte counters since the pool was imported.
|
||||
type PoolKernelStat struct {
|
||||
Name string
|
||||
Health string
|
||||
NRead uint64
|
||||
NWrite uint64
|
||||
}
|
||||
|
||||
// PoolIoStats holds calculated per-second I/O rates for a pool.
|
||||
type PoolIoStats struct {
|
||||
NRead uint64
|
||||
NWrite uint64
|
||||
}
|
||||
|
||||
// Dataset is a single ZFS dataset with usage information.
|
||||
type Dataset struct {
|
||||
Name string
|
||||
Used uint64
|
||||
Avail uint64
|
||||
Mountpoint string
|
||||
}
|
||||
|
||||
// PoolStats returns capacity and health for all pools on the system using
|
||||
// `zpool list`. Frequent health and I/O sampling uses PoolKernelStats instead.
|
||||
func PoolStats() ([]PoolStat, error) {
|
||||
out, err := commandOutput("zpool", "list", "-Hp", "-o", "name,size,alloc,free,health")
|
||||
if err != nil {
|
||||
var exitErr *exec.ExitError
|
||||
if errors.As(err, &exitErr) && strings.Contains(string(exitErr.Stderr), "no pools available") {
|
||||
return nil, nil
|
||||
}
|
||||
return nil, fmt.Errorf("zpool list: %w", err)
|
||||
}
|
||||
return parseZpoolListOutput(out)
|
||||
}
|
||||
|
||||
// Datasets returns all datasets on the system with usage and mountpoint
|
||||
// information using `zfs list` (recursive by default).
|
||||
func Datasets() ([]Dataset, error) {
|
||||
out, err := commandOutput("zfs", "list", "-Hp", "-o", "name,used,avail,mountpoint")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("zfs list: %w", err)
|
||||
}
|
||||
return parseZfsListOutput(out)
|
||||
}
|
||||
|
||||
// parseZpoolListOutput parses `zpool list -Hp -o name,size,alloc,free,health` output.
|
||||
// Columns are tab-separated; numeric columns are raw bytes.
|
||||
func parseZpoolListOutput(out []byte) ([]PoolStat, error) {
|
||||
var pools []PoolStat
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := strings.TrimSpace(scanner.Text())
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
if line == "no pools available" && len(pools) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
fields := strings.Split(line, "\t")
|
||||
if len(fields) < 5 {
|
||||
return nil, fmt.Errorf("unexpected zpool list line: %q", line)
|
||||
}
|
||||
size, err := strconv.ParseUint(fields[1], 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing size for pool %q: %w", fields[0], err)
|
||||
}
|
||||
alloc, err := strconv.ParseUint(fields[2], 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing alloc for pool %q: %w", fields[0], err)
|
||||
}
|
||||
free, err := strconv.ParseUint(fields[3], 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing free for pool %q: %w", fields[0], err)
|
||||
}
|
||||
pools = append(pools, PoolStat{
|
||||
Name: fields[0],
|
||||
Size: size,
|
||||
Alloc: alloc,
|
||||
Free: free,
|
||||
Health: fields[4],
|
||||
})
|
||||
}
|
||||
return pools, scanner.Err()
|
||||
}
|
||||
|
||||
// parseZfsListOutput parses `zfs list -Hp -o name,used,avail,mountpoint` output.
|
||||
// The mountpoint column may contain spaces, so it is split on tabs only.
|
||||
func parseZfsListOutput(out []byte) ([]Dataset, error) {
|
||||
var datasets []Dataset
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := strings.TrimSpace(scanner.Text())
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
fields := strings.SplitN(line, "\t", 4)
|
||||
if len(fields) < 4 {
|
||||
return nil, fmt.Errorf("unexpected zfs list line: %q", line)
|
||||
}
|
||||
used, err := strconv.ParseUint(fields[1], 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing used for dataset %q: %w", fields[0], err)
|
||||
}
|
||||
avail, err := strconv.ParseUint(fields[2], 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing avail for dataset %q: %w", fields[0], err)
|
||||
}
|
||||
datasets = append(datasets, Dataset{
|
||||
Name: fields[0],
|
||||
Used: used,
|
||||
Avail: avail,
|
||||
Mountpoint: fields[3],
|
||||
})
|
||||
}
|
||||
return datasets, scanner.Err()
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
//go:build freebsd
|
||||
|
||||
package zfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
func ARCSize() (uint64, error) {
|
||||
return unix.SysctlUint64("kstat.zfs.misc.arcstats.size")
|
||||
}
|
||||
|
||||
// FreeBSD does not expose Linux's per-pool procfs kstats. Capacity, health,
|
||||
// and detail collection still work through the cached utilities.
|
||||
func PoolKernelStats() ([]PoolKernelStat, error) {
|
||||
return nil, errors.ErrUnsupported
|
||||
}
|
||||
@@ -0,0 +1,199 @@
|
||||
//go:build linux
|
||||
|
||||
// Package zfs provides functions to read ZFS statistics.
|
||||
package zfs
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
var procZfsPath = "/proc/spl/kstat/zfs"
|
||||
|
||||
func ARCSize() (uint64, error) {
|
||||
file, err := os.Open(filepath.Join(procZfsPath, "arcstats"))
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
scanner := bufio.NewScanner(file)
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if strings.HasPrefix(line, "size") {
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 3 {
|
||||
return 0, fmt.Errorf("unexpected arcstats size format: %s", line)
|
||||
}
|
||||
return strconv.ParseUint(fields[2], 10, 64)
|
||||
}
|
||||
}
|
||||
if err := scanner.Err(); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
return 0, fmt.Errorf("size field not found in arcstats")
|
||||
}
|
||||
|
||||
// PoolKernelStats reads pool state and cumulative I/O counters directly from
|
||||
// procfs. These kstats are the same interfaces used by node_exporter's Linux
|
||||
// ZFS collector and avoid keeping a `zpool iostat` subprocess alive.
|
||||
func PoolKernelStats() ([]PoolKernelStat, error) {
|
||||
poolDirs := make(map[string]struct{})
|
||||
for _, filename := range []string{"state", "io", "objset-*"} {
|
||||
paths, err := filepath.Glob(filepath.Join(procZfsPath, "*", filename))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, path := range paths {
|
||||
poolDirs[filepath.Dir(path)] = struct{}{}
|
||||
}
|
||||
}
|
||||
if len(poolDirs) == 0 {
|
||||
return nil, ErrNoZfs
|
||||
}
|
||||
pools := make([]PoolKernelStat, 0, len(poolDirs))
|
||||
for poolDir := range poolDirs {
|
||||
nread, nwrite, err := readPoolCounters(poolDir)
|
||||
if err != nil {
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
continue // pool may have been exported after the glob
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
state, err := os.ReadFile(filepath.Join(poolDir, "state"))
|
||||
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return nil, err
|
||||
}
|
||||
pools = append(pools, PoolKernelStat{
|
||||
Name: filepath.Base(poolDir), Health: strings.ToUpper(strings.TrimSpace(string(state))),
|
||||
NRead: nread, NWrite: nwrite,
|
||||
})
|
||||
}
|
||||
if len(pools) == 0 {
|
||||
return nil, ErrNoZfs
|
||||
}
|
||||
return pools, nil
|
||||
}
|
||||
|
||||
// readPoolCounters supports both ZFS kernel interfaces. OpenZFS through 2.3
|
||||
// exposes aggregate vdev counters in "io". When that file is unavailable, sum
|
||||
// the logical I/O counters exposed for each dataset in the pool.
|
||||
func readPoolCounters(poolDir string) (uint64, uint64, error) {
|
||||
nread, nwrite, err := readPoolIO(filepath.Join(poolDir, "io"))
|
||||
if err == nil || !errors.Is(err, os.ErrNotExist) {
|
||||
return nread, nwrite, err
|
||||
}
|
||||
return readPoolObjsets(poolDir)
|
||||
}
|
||||
|
||||
func readPoolIO(path string) (uint64, uint64, error) {
|
||||
file, err := os.Open(path)
|
||||
if err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
scanner := bufio.NewScanner(file)
|
||||
for scanner.Scan() {
|
||||
fields := strings.Fields(scanner.Text())
|
||||
if len(fields) < 2 || fields[0] != "nread" {
|
||||
continue
|
||||
}
|
||||
if !scanner.Scan() {
|
||||
break
|
||||
}
|
||||
values := strings.Fields(scanner.Text())
|
||||
if len(values) < 2 {
|
||||
break
|
||||
}
|
||||
nread, err := strconv.ParseUint(values[0], 10, 64)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("parsing nread in %s: %w", path, err)
|
||||
}
|
||||
nwrite, err := strconv.ParseUint(values[1], 10, 64)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("parsing nwritten in %s: %w", path, err)
|
||||
}
|
||||
return nread, nwrite, nil
|
||||
}
|
||||
if err := scanner.Err(); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
return 0, 0, fmt.Errorf("I/O counters not found in %s", path)
|
||||
}
|
||||
|
||||
func readPoolObjsets(poolDir string) (uint64, uint64, error) {
|
||||
paths, err := filepath.Glob(filepath.Join(poolDir, "objset-*"))
|
||||
if err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
if len(paths) == 0 {
|
||||
return 0, 0, fmt.Errorf("dataset I/O counters not found in %s", poolDir)
|
||||
}
|
||||
|
||||
var totalRead, totalWrite uint64
|
||||
objsetsRead := 0
|
||||
for _, path := range paths {
|
||||
nread, nwrite, err := readObjsetIO(path)
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
continue // dataset may have been destroyed after the glob
|
||||
}
|
||||
if err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
totalRead += nread
|
||||
totalWrite += nwrite
|
||||
objsetsRead++
|
||||
}
|
||||
if objsetsRead == 0 {
|
||||
return 0, 0, fmt.Errorf("dataset I/O counters not found in %s", poolDir)
|
||||
}
|
||||
return totalRead, totalWrite, nil
|
||||
}
|
||||
|
||||
func readObjsetIO(path string) (uint64, uint64, error) {
|
||||
file, err := os.Open(path)
|
||||
if err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
var nread, nwrite uint64
|
||||
var foundRead, foundWrite bool
|
||||
scanner := bufio.NewScanner(file)
|
||||
for scanner.Scan() {
|
||||
fields := strings.Fields(scanner.Text())
|
||||
if len(fields) < 3 {
|
||||
continue
|
||||
}
|
||||
var target *uint64
|
||||
switch fields[0] {
|
||||
case "nread":
|
||||
target = &nread
|
||||
foundRead = true
|
||||
case "nwritten":
|
||||
target = &nwrite
|
||||
foundWrite = true
|
||||
default:
|
||||
continue
|
||||
}
|
||||
value, err := strconv.ParseUint(fields[2], 10, 64)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("parsing %s in %s: %w", fields[0], path, err)
|
||||
}
|
||||
*target = value
|
||||
}
|
||||
if err := scanner.Err(); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
if !foundRead || !foundWrite {
|
||||
return 0, 0, fmt.Errorf("incomplete I/O counters in %s", path)
|
||||
}
|
||||
return nread, nwrite, nil
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user