mirror of
https://github.com/henrygd/beszel.git
synced 2026-09-26 11:27:47 +02:00
Compare commits
233 Commits
451f474016
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c7f2177b08 | ||
|
|
a042e19549 | ||
|
|
fe83f5b831 | ||
|
|
46fa7c581e | ||
|
|
24792aa24f | ||
|
|
16e3fbadce | ||
|
|
86ab0fae8b | ||
|
|
badd4c8245 | ||
|
|
7bea20e3b6 | ||
|
|
433b83800f | ||
|
|
3dfe062ee4 | ||
|
|
3fb97b800c | ||
|
|
d708def38f | ||
|
|
f50fb4f8e5 | ||
|
|
c25408651f | ||
|
|
d591da46f3 | ||
|
|
d80a2f49f9 | ||
|
|
21b648a005 | ||
|
|
151423ac63 | ||
|
|
f7528a0208 | ||
|
|
a20a7d2edc | ||
|
|
f2adb9cf94 | ||
|
|
4d10ea2e03 | ||
|
|
b5ef015451 | ||
|
|
6141b15f03 | ||
|
|
c21412f45d | ||
|
|
367d2f39da | ||
|
|
eabd9a950a | ||
|
|
0be9882b34 | ||
|
|
e4b84b72ab | ||
|
|
fc33e62736 | ||
|
|
a99fe5e997 | ||
|
|
0870716052 | ||
|
|
b1270e341c | ||
|
|
9042a8c5c8 | ||
|
|
8bf6917fe0 | ||
|
|
2d5ea3fa08 | ||
|
|
627d364071 | ||
|
|
8047f005d4 | ||
|
|
4a4610bbc3 | ||
|
|
1aaabfc255 | ||
|
|
97ea3c16cb | ||
|
|
c9de35fad2 | ||
|
|
a5f216f425 | ||
|
|
cbe4824ac3 | ||
|
|
2c69197d2d | ||
|
|
97e6f64bdc | ||
|
|
4a5915b141 | ||
|
|
e68372dce4 | ||
|
|
c52f3acb94 | ||
|
|
c09eb8c6df | ||
|
|
a0dc19eacf | ||
|
|
912bc50874 | ||
|
|
c54dbfba7c | ||
|
|
dd3f7d58b5 | ||
|
|
0509053a69 | ||
|
|
187dc886a9 | ||
|
|
2784460621 | ||
|
|
b0bc727411 | ||
|
|
6937453282 | ||
|
|
97de1471d7 | ||
|
|
bd7e359dcd | ||
|
|
d2352e8882 | ||
|
|
f7fd3ef403 | ||
|
|
db3afeabd9 | ||
|
|
b347599928 | ||
|
|
6e5440c21f | ||
|
|
bb1b39928e | ||
|
|
4bf70700f2 | ||
|
|
18f7a4bbc0 | ||
|
|
f0f1f7985c | ||
|
|
50f6fc075d | ||
|
|
a0bf338796 | ||
|
|
982101743e | ||
|
|
6a7b2772d9 | ||
|
|
086091a0fe | ||
|
|
f204dc17e6 | ||
|
|
5fe1583655 | ||
|
|
6d82ee70b1 | ||
|
|
bb270e02a8 | ||
|
|
312c109138 | ||
|
|
c938368089 | ||
|
|
8d6a5d5f6e | ||
|
|
98687be2f2 | ||
|
|
e39e153ca0 | ||
|
|
5b87f7d7cb | ||
|
|
9a0aa5a89e | ||
|
|
997adc19bb | ||
|
|
08d813620c | ||
|
|
6cb302fcf6 | ||
|
|
59eed073c3 | ||
|
|
266a74bab8 | ||
|
|
ad24484caa | ||
|
|
027d0c204d | ||
|
|
46d94a9804 | ||
|
|
82fc772882 | ||
|
|
c157c2026d | ||
|
|
e1d9ebc61d | ||
|
|
5af6b6b184 | ||
|
|
ffcdb04167 | ||
|
|
bc21da9cb3 | ||
|
|
71af06b31c | ||
|
|
a8def47018 | ||
|
|
d2a253082f | ||
|
|
3d8fc39e94 | ||
|
|
7d347cfd6a | ||
|
|
5790fbecce | ||
|
|
f9309da9f0 | ||
|
|
7d97b0d23a | ||
|
|
f104f31ee3 | ||
|
|
a1ca51608a | ||
|
|
a4de2e87c4 | ||
|
|
5969d36856 | ||
|
|
b1895247ba | ||
|
|
097180e8d7 | ||
|
|
ed88e6efae | ||
|
|
b670224ed8 | ||
|
|
917d069ab3 | ||
|
|
b38fb7dafa | ||
|
|
8675199e20 | ||
|
|
3af6512514 | ||
|
|
87620f3251 | ||
|
|
fa9de55433 | ||
|
|
e235c9935c | ||
|
|
7c60f02802 | ||
|
|
6fe268e463 | ||
|
|
467f176713 | ||
|
|
4c8e3c69ba | ||
|
|
8dfdacb8f5 | ||
|
|
d61b75ffdf | ||
|
|
d7256c7af7 | ||
|
|
0ad707288a | ||
|
|
0bc5470f08 | ||
|
|
4c48fe0c41 | ||
|
|
6efe4be648 | ||
|
|
f1e5797c76 | ||
|
|
6f92b9396d | ||
|
|
946f2e6be1 | ||
|
|
ba90daf4d6 | ||
|
|
aa1d67a122 | ||
|
|
68a3f8962a | ||
|
|
0eb3426619 | ||
|
|
96beadc8c9 | ||
|
|
65a6f60304 | ||
|
|
54dae08631 | ||
|
|
2df1f722e4 | ||
|
|
2054b276a7 | ||
|
|
19f250c7de | ||
|
|
e07f91b920 | ||
|
|
218aa8478a | ||
|
|
90b789aa72 | ||
|
|
2d01d71f46 | ||
|
|
f894d188cf | ||
|
|
46cc602d37 | ||
|
|
89ad51d4ce | ||
|
|
d5f41af3a6 | ||
|
|
1074503af1 | ||
|
|
90f1bdef1e | ||
|
|
5f383c0eb1 | ||
|
|
ae037b278e | ||
|
|
9f1128933f | ||
|
|
dbe10e3a8f | ||
|
|
7e7bcb3b35 | ||
|
|
418e3f0892 | ||
|
|
7556a63378 | ||
|
|
260b082c5e | ||
|
|
55054a75ab | ||
|
|
ccd1735a8e | ||
|
|
c67b69d17e | ||
|
|
da3ab62d4e | ||
|
|
ec4ec01a39 | ||
|
|
ca5497324c | ||
|
|
adaf6f338d | ||
|
|
1ab1229a61 | ||
|
|
9a54d844ba | ||
|
|
87405c5f10 | ||
|
|
bfa6a1e361 | ||
|
|
3688b2d033 | ||
|
|
b4e1f3fafe | ||
|
|
eebcd56462 | ||
|
|
eb5dd230cf | ||
|
|
3337dff64b | ||
|
|
b68acea5a8 | ||
|
|
66ac62a125 | ||
|
|
e2a18ec636 | ||
|
|
ffd4fc2c45 | ||
|
|
052489cada | ||
|
|
d50c09176f | ||
|
|
fe84cfa16d | ||
|
|
6d0b83f6de | ||
|
|
cf90249519 | ||
|
|
6607d4c0d6 | ||
|
|
bc55e249c4 | ||
|
|
977826e8f3 | ||
|
|
8450b40e0c | ||
|
|
ac4436bea3 | ||
|
|
fa5cda83c2 | ||
|
|
35af36fbd2 | ||
|
|
e380ab6917 | ||
|
|
c3a432101b | ||
|
|
98e86b4c9c | ||
|
|
d40372842b | ||
|
|
c9b6279e61 | ||
|
|
9887b662ad | ||
|
|
71bc6f9b9f | ||
|
|
87468965dc | ||
|
|
db0da58ac8 | ||
|
|
b9b3a23063 | ||
|
|
ac34e58113 | ||
|
|
3c8703d1c0 | ||
|
|
01efba50a5 | ||
|
|
d0453f1ca6 | ||
|
|
7ffc6e81ce | ||
|
|
1aa9fcd31d | ||
|
|
bd52134558 | ||
|
|
3fac02f0c0 | ||
|
|
e4c0522cae | ||
|
|
f2ccaacecb | ||
|
|
a81ce14046 | ||
|
|
6e4a2c8a3d | ||
|
|
8cf39be6ef | ||
|
|
146ca4284c | ||
|
|
6eae195d9c | ||
|
|
30bf65991b | ||
|
|
f57f5883ce | ||
|
|
0472343730 | ||
|
|
d3a1d61955 | ||
|
|
9708e24fd2 | ||
|
|
17e910a246 | ||
|
|
0e65a2373f | ||
|
|
c1c1cd1bcb | ||
|
|
cd9ea51039 | ||
|
|
a71617e058 |
@@ -1,6 +1,6 @@
|
||||
# Node.js dependencies
|
||||
node_modules
|
||||
internalsite/node_modules
|
||||
node_modules/
|
||||
**/node_modules/
|
||||
|
||||
# Go build artifacts and binaries
|
||||
build
|
||||
|
||||
12
.github/dependabot.yml
vendored
Normal file
12
.github/dependabot.yml
vendored
Normal file
@@ -0,0 +1,12 @@
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: gomod
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
|
||||
54
.github/workflows/docker-images.yml
vendored
54
.github/workflows/docker-images.yml
vendored
@@ -29,6 +29,7 @@ jobs:
|
||||
# henrygd/beszel-agent:alpine
|
||||
- image: henrygd/beszel-agent
|
||||
dockerfile: ./internal/dockerfile_agent_alpine
|
||||
flavor: latest=false
|
||||
registry: docker.io
|
||||
username_secret: DOCKERHUB_USERNAME
|
||||
password_secret: DOCKERHUB_TOKEN
|
||||
@@ -41,7 +42,7 @@ jobs:
|
||||
# henrygd/beszel-agent-nvidia
|
||||
- image: henrygd/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia
|
||||
platforms: linux/amd64
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: docker.io
|
||||
username_secret: DOCKERHUB_USERNAME
|
||||
password_secret: DOCKERHUB_TOKEN
|
||||
@@ -52,6 +53,20 @@ jobs:
|
||||
type=semver,pattern={{major}}
|
||||
type=raw,value={{sha}},enable=${{ github.ref_type != 'tag' }}
|
||||
|
||||
# henrygd/beszel-agent-nvidia:slim
|
||||
- image: henrygd/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia_slim
|
||||
flavor: latest=false
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: docker.io
|
||||
username_secret: DOCKERHUB_USERNAME
|
||||
password_secret: DOCKERHUB_TOKEN
|
||||
tags: |
|
||||
type=raw,value=slim
|
||||
type=semver,pattern={{version}}-slim
|
||||
type=semver,pattern={{major}}.{{minor}}-slim
|
||||
type=semver,pattern={{major}}-slim
|
||||
|
||||
# henrygd/beszel-agent-intel
|
||||
- image: henrygd/beszel-agent-intel
|
||||
dockerfile: ./internal/dockerfile_agent_intel
|
||||
@@ -96,7 +111,7 @@ jobs:
|
||||
# ghcr.io/henrygd/beszel-agent-nvidia
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia
|
||||
platforms: linux/amd64
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password_secret: GITHUB_TOKEN
|
||||
@@ -107,6 +122,20 @@ jobs:
|
||||
type=semver,pattern={{major}}
|
||||
type=raw,value={{sha}},enable=${{ github.ref_type != 'tag' }}
|
||||
|
||||
# ghcr.io/henrygd/beszel-agent-nvidia:slim
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-nvidia
|
||||
dockerfile: ./internal/dockerfile_agent_nvidia_slim
|
||||
flavor: latest=false
|
||||
platforms: linux/amd64,linux/arm64
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password_secret: GITHUB_TOKEN
|
||||
tags: |
|
||||
type=raw,value=slim
|
||||
type=semver,pattern={{version}}-slim
|
||||
type=semver,pattern={{major}}.{{minor}}-slim
|
||||
type=semver,pattern={{major}}-slim
|
||||
|
||||
# ghcr.io/henrygd/beszel-agent-intel
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent-intel
|
||||
dockerfile: ./internal/dockerfile_agent_intel
|
||||
@@ -124,6 +153,7 @@ jobs:
|
||||
# ghcr.io/henrygd/beszel-agent:alpine
|
||||
- image: ghcr.io/${{ github.repository }}/beszel-agent
|
||||
dockerfile: ./internal/dockerfile_agent_alpine
|
||||
flavor: latest=false
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password_secret: GITHUB_TOKEN
|
||||
@@ -133,7 +163,7 @@ jobs:
|
||||
type=semver,pattern={{major}}.{{minor}}-alpine
|
||||
type=semver,pattern={{major}}-alpine
|
||||
|
||||
# henrygd/beszel-agent (keep at bottom so it gets built after :alpine and gets the latest tag)
|
||||
# henrygd/beszel-agent
|
||||
- image: henrygd/beszel-agent
|
||||
dockerfile: ./internal/dockerfile_agent
|
||||
registry: docker.io
|
||||
@@ -152,7 +182,7 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
@@ -164,16 +194,18 @@ jobs:
|
||||
run: bun run --cwd ./internal/site build
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
uses: docker/setup-qemu-action@v4
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
- name: Docker metadata
|
||||
id: metadata
|
||||
uses: docker/metadata-action@v5
|
||||
uses: docker/metadata-action@v6
|
||||
with:
|
||||
images: ${{ matrix.image }}
|
||||
# Variant images must not overwrite the standard image's latest tag.
|
||||
flavor: ${{ matrix.flavor || 'latest=auto' }}
|
||||
tags: ${{ matrix.tags }}
|
||||
|
||||
# https://github.com/docker/login-action
|
||||
@@ -181,7 +213,7 @@ jobs:
|
||||
env:
|
||||
password_secret_exists: ${{ secrets[matrix.password_secret] != '' && 'true' || 'false' }}
|
||||
if: github.event_name != 'pull_request' && env.password_secret_exists == 'true'
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: ${{ matrix.username || secrets[matrix.username_secret] }}
|
||||
password: ${{ secrets[matrix.password_secret] }}
|
||||
@@ -190,11 +222,13 @@ jobs:
|
||||
# Build and push Docker image with Buildx (don't push on PR)
|
||||
# https://github.com/docker/build-push-action
|
||||
- name: Build and push Docker image
|
||||
uses: docker/build-push-action@v5
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: ./
|
||||
file: ${{ matrix.dockerfile }}
|
||||
platforms: ${{ matrix.platforms || 'linux/amd64,linux/arm64,linux/arm/v7' }}
|
||||
platforms: ${{ matrix.platforms || 'linux/amd64,linux/arm64,linux/arm/v6,linux/arm/v7' }}
|
||||
push: ${{ github.ref_type == 'tag' && secrets[matrix.password_secret] != '' }}
|
||||
provenance: mode=max
|
||||
sbom: true
|
||||
tags: ${{ steps.metadata.outputs.tags }}
|
||||
labels: ${{ steps.metadata.outputs.labels }}
|
||||
|
||||
109
.github/workflows/helm-charts.yml
vendored
Normal file
109
.github/workflows/helm-charts.yml
vendored
Normal file
@@ -0,0 +1,109 @@
|
||||
name: Helm charts
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- "supplemental/helm/**"
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "supplemental/helm/**"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
env:
|
||||
OCI_REGISTRY: ghcr.io/henrygd/beszel-charts
|
||||
|
||||
jobs:
|
||||
changes:
|
||||
name: Detect changed charts
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
charts: ${{ steps.changes.outputs.charts }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Detect changed charts
|
||||
id: changes
|
||||
env:
|
||||
BASE_SHA: ${{ github.event_name == 'pull_request' && github.event.pull_request.base.sha || github.event.before }}
|
||||
run: |
|
||||
charts=()
|
||||
|
||||
for name in beszel-agent beszel-hub; do
|
||||
path="supplemental/helm/$name"
|
||||
if ! git diff --quiet "$BASE_SHA" "$GITHUB_SHA" -- "$path"; then
|
||||
charts+=("$name|$path")
|
||||
fi
|
||||
done
|
||||
|
||||
printf '%s\n' "${charts[@]}" \
|
||||
| jq -Rsc 'split("\n") | map(select(length > 0) | split("|") | {name: .[0], path: .[1]})' \
|
||||
| xargs -0 printf 'charts=%s\n' >> "$GITHUB_OUTPUT"
|
||||
|
||||
validate-and-publish:
|
||||
name: ${{ github.event_name == 'push' && 'Publish' || 'Validate' }} ${{ matrix.chart.name }}
|
||||
needs: changes
|
||||
if: needs.changes.outputs.charts != '[]'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
chart: ${{ fromJSON(needs.changes.outputs.charts) }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Helm
|
||||
uses: azure/setup-helm@v5
|
||||
|
||||
- name: Lint chart
|
||||
run: helm lint "${{ matrix.chart.path }}" --set env.KEY=ci-placeholder
|
||||
|
||||
- name: Render chart
|
||||
run: helm template "${{ matrix.chart.name }}" "${{ matrix.chart.path }}" --set env.KEY=ci-placeholder > /dev/null
|
||||
|
||||
- name: Package chart
|
||||
id: package
|
||||
env:
|
||||
CHART_NAME: ${{ matrix.chart.name }}
|
||||
CHART_PATH: ${{ matrix.chart.path }}
|
||||
run: |
|
||||
version=$(awk '/^version:/ { print $2 }' "$CHART_PATH/Chart.yaml")
|
||||
test -n "$version"
|
||||
|
||||
mkdir -p .helm-packages
|
||||
helm package "$CHART_PATH" --destination .helm-packages
|
||||
|
||||
package=".helm-packages/${CHART_NAME}-${version}.tgz"
|
||||
test -f "$package"
|
||||
echo "version=$version" >> "$GITHUB_OUTPUT"
|
||||
echo "package=$package" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Log in to GHCR
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: echo "$GITHUB_TOKEN" | helm registry login ghcr.io --username "$GITHUB_ACTOR" --password-stdin
|
||||
|
||||
- name: Check chart version is unpublished
|
||||
env:
|
||||
CHART_NAME: ${{ matrix.chart.name }}
|
||||
CHART_VERSION: ${{ steps.package.outputs.version }}
|
||||
run: |
|
||||
chart="oci://${OCI_REGISTRY}/${CHART_NAME}"
|
||||
if helm show chart "$chart" --version "$CHART_VERSION" > /dev/null 2>&1; then
|
||||
echo "${CHART_NAME} ${CHART_VERSION} is already published. Bump version in Chart.yaml." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Publish chart
|
||||
if: github.event_name == 'push'
|
||||
run: helm push "${{ steps.package.outputs.package }}" "oci://${OCI_REGISTRY}"
|
||||
4
.github/workflows/inactivity-actions.yml
vendored
4
.github/workflows/inactivity-actions.yml
vendored
@@ -15,7 +15,7 @@ jobs:
|
||||
name: Lock Inactive Issues
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- uses: klaasnicolaas/action-inactivity-lock@v1.1.3
|
||||
- uses: klaasnicolaas/action-inactivity-lock@v2.0.1
|
||||
id: lock
|
||||
with:
|
||||
days-inactive-issues: 14
|
||||
@@ -29,7 +29,7 @@ jobs:
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- name: Close Stale Issues
|
||||
uses: actions/stale@v10
|
||||
uses: actions/stale@v11
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
|
||||
10
.github/workflows/release.yml
vendored
10
.github/workflows/release.yml
vendored
@@ -13,7 +13,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
@@ -27,12 +27,12 @@ jobs:
|
||||
run: bun run --cwd ./internal/site build
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version: "^1.22.1"
|
||||
go-version: stable
|
||||
|
||||
- name: Set up .NET
|
||||
uses: actions/setup-dotnet@v4
|
||||
uses: actions/setup-dotnet@v6
|
||||
with:
|
||||
dotnet-version: "9.0.x"
|
||||
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: GoReleaser beszel
|
||||
uses: goreleaser/goreleaser-action@v6
|
||||
uses: goreleaser/goreleaser-action@v7
|
||||
with:
|
||||
workdir: ./
|
||||
distribution: goreleaser
|
||||
|
||||
101
.github/workflows/update-helm-charts.yml
vendored
Normal file
101
.github/workflows/update-helm-charts.yml
vendored
Normal file
@@ -0,0 +1,101 @@
|
||||
name: Update Helm charts
|
||||
|
||||
on:
|
||||
release:
|
||||
types:
|
||||
- published
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: update-helm-charts
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
update:
|
||||
name: Propose chart update
|
||||
if: ${{ github.repository_owner == 'henrygd' && startsWith(github.event.release.tag_name, 'v') && !github.event.release.prerelease }}
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
BRANCH: automation/update-helm-app-version
|
||||
RELEASE_TAG: ${{ github.event.release.tag_name }}
|
||||
AUTOMATION_TOKEN: ${{ secrets.CR_TOKEN || github.token }}
|
||||
|
||||
steps:
|
||||
- name: Checkout main
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
ref: main
|
||||
token: ${{ env.AUTOMATION_TOKEN }}
|
||||
|
||||
- name: Update chart versions
|
||||
id: update
|
||||
run: |
|
||||
version="${RELEASE_TAG#v}"
|
||||
if [[ ! "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
|
||||
echo "Unsupported software release version: $version" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
changed=false
|
||||
for chart in supplemental/helm/beszel-agent supplemental/helm/beszel-hub; do
|
||||
current_app_version=$(awk -F '"' '/^appVersion:/ { print $2 }' "$chart/Chart.yaml")
|
||||
if [[ "$current_app_version" == "$version" ]]; then
|
||||
echo "$chart already uses appVersion $version"
|
||||
continue
|
||||
fi
|
||||
|
||||
newest_version=$(printf '%s\n' "$current_app_version" "$version" | sort -V | tail -n 1)
|
||||
if [[ "$newest_version" != "$version" ]]; then
|
||||
echo "Skipping stale update of $chart from $current_app_version to $version"
|
||||
continue
|
||||
fi
|
||||
|
||||
chart_version=$(awk '/^version:/ { print $2 }' "$chart/Chart.yaml")
|
||||
if [[ ! "$chart_version" =~ ^([0-9]+)\.([0-9]+)\.([0-9]+)$ ]]; then
|
||||
echo "Unsupported chart version in $chart/Chart.yaml: $chart_version" >&2
|
||||
exit 1
|
||||
fi
|
||||
next_chart_version="${BASH_REMATCH[1]}.${BASH_REMATCH[2]}.$((BASH_REMATCH[3] + 1))"
|
||||
|
||||
NEW_APP_VERSION="$version" NEW_CHART_VERSION="$next_chart_version" \
|
||||
perl -pi -e 's/^appVersion:.*$/appVersion: "$ENV{NEW_APP_VERSION}"/; s/^version:.*$/version: $ENV{NEW_CHART_VERSION}/' \
|
||||
"$chart/Chart.yaml"
|
||||
OLD_APP_VERSION="$current_app_version" NEW_APP_VERSION="$version" \
|
||||
perl -pi -e 's/\Q$ENV{OLD_APP_VERSION}\E/$ENV{NEW_APP_VERSION}/g' "$chart/README.md"
|
||||
|
||||
echo "$chart: appVersion $current_app_version -> $version, chart $chart_version -> $next_chart_version"
|
||||
changed=true
|
||||
done
|
||||
|
||||
echo "changed=$changed" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Open or update pull request
|
||||
if: steps.update.outputs.changed == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ env.AUTOMATION_TOKEN }}
|
||||
run: |
|
||||
version="${RELEASE_TAG#v}"
|
||||
title="chore(helm): update app version to ${version}"
|
||||
body="Updates the Helm charts for [Beszel ${version}](${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/releases/tag/${RELEASE_TAG}) and bumps their chart patch versions. Merging this pull request publishes the updated charts to GHCR."
|
||||
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git checkout -B "$BRANCH"
|
||||
git add supplemental/helm/beszel-agent/Chart.yaml \
|
||||
supplemental/helm/beszel-agent/README.md \
|
||||
supplemental/helm/beszel-hub/Chart.yaml \
|
||||
supplemental/helm/beszel-hub/README.md
|
||||
git commit -m "$title"
|
||||
|
||||
git fetch origin "$BRANCH" || true
|
||||
git push --force-with-lease origin "HEAD:refs/heads/${BRANCH}"
|
||||
|
||||
pr_number=$(gh pr list --head "$BRANCH" --base main --state open --json number --jq '.[0].number')
|
||||
if [[ -n "$pr_number" ]]; then
|
||||
gh pr edit "$pr_number" --title "$title" --body "$body"
|
||||
else
|
||||
gh pr create --base main --head "$BRANCH" --title "$title" --body "$body"
|
||||
fi
|
||||
10
.github/workflows/vulncheck.yml
vendored
10
.github/workflows/vulncheck.yml
vendored
@@ -2,10 +2,6 @@
|
||||
|
||||
name: VulnCheck
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
@@ -19,11 +15,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check out code into the Go module directory
|
||||
uses: actions/checkout@v6
|
||||
uses: actions/checkout@v7
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version: 1.26.x
|
||||
go-version: stable
|
||||
# cached: false
|
||||
- name: Get official govulncheck
|
||||
run: go install golang.org/x/vuln/cmd/govulncheck@latest
|
||||
|
||||
3
.gitignore
vendored
3
.gitignore
vendored
@@ -3,7 +3,6 @@ pb_data
|
||||
data
|
||||
temp
|
||||
.vscode
|
||||
beszel-agent
|
||||
beszel_data
|
||||
beszel_data*
|
||||
dist
|
||||
@@ -21,3 +20,5 @@ __debug_*
|
||||
agent/lhm/obj
|
||||
agent/lhm/bin
|
||||
dockerfile_agent_dev
|
||||
.cr-release-packages
|
||||
.tmp
|
||||
|
||||
@@ -31,12 +31,16 @@ builds:
|
||||
goarch: arm64
|
||||
- goos: freebsd
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: arm
|
||||
|
||||
- id: beszel-agent
|
||||
binary: beszel-agent
|
||||
main: internal/cmd/agent/agent.go
|
||||
env:
|
||||
- CGO_ENABLED=0
|
||||
ldflags:
|
||||
- -s -w -X github.com/henrygd/beszel/internal/ghupdate.buildGOARM={{ .Arm }}
|
||||
goos:
|
||||
- linux
|
||||
- darwin
|
||||
@@ -52,6 +56,10 @@ builds:
|
||||
- mipsle
|
||||
- mips
|
||||
- ppc64le
|
||||
goarm:
|
||||
- "5"
|
||||
- "6"
|
||||
- "7"
|
||||
gomips:
|
||||
- hardfloat
|
||||
- softfloat
|
||||
@@ -71,6 +79,8 @@ builds:
|
||||
gomips: hardfloat
|
||||
- goos: windows
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: arm
|
||||
- goos: darwin
|
||||
goarch: riscv64
|
||||
- goos: windows
|
||||
@@ -97,6 +107,7 @@ archives:
|
||||
{{ .Binary }}_
|
||||
{{- .Os }}_
|
||||
{{- .Arch }}
|
||||
{{- if ne .Arm "6" }}{{ with .Arm }}v{{ . }}{{ end }}{{ end }}
|
||||
format_overrides:
|
||||
- goos: windows
|
||||
formats: [zip]
|
||||
|
||||
2
Makefile
2
Makefile
@@ -52,7 +52,7 @@ lint:
|
||||
golangci-lint run
|
||||
|
||||
test:
|
||||
go test -tags=testing ./...
|
||||
go test -tags='testing no_ui' ./...
|
||||
|
||||
tidy:
|
||||
go mod tidy
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
|
||||
## Reporting a Vulnerability
|
||||
|
||||
If you find a vulnerability in the latest version, please [submit a private advisory](https://github.com/henrygd/beszel/security/advisories/new).
|
||||
**PLEASE ONLY USE SECURITY ADVISORIES FOR REAL HIGH SEVERITY VULNERABILITIES.**
|
||||
|
||||
If it's low severity (use best judgement) you may open an issue instead of an advisory.
|
||||
If you find a vulnerability in the latest version, and it is not high severity, open an issue instead of an advisory.
|
||||
|
||||
I am overwhelmed with advisories, often erroneous, which are clearly found and written by AI. I don't have the capacity to review all of them.
|
||||
|
||||
@@ -29,6 +29,7 @@ type Agent struct {
|
||||
fsNames []string // List of filesystem device names being monitored
|
||||
fsStats map[string]*system.FsStats // Keeps track of disk stats for each filesystem
|
||||
diskPrev map[uint16]map[string]prevDisk // Previous disk I/O counters per cache interval
|
||||
diskBaseline map[string]prevDisk // Latest disk I/O counters of any interval, seeds a new interval
|
||||
diskUsageCacheDuration time.Duration // How long to cache disk usage (to avoid waking sleeping disks)
|
||||
lastDiskUsageUpdate time.Time // Last time disk usage was collected
|
||||
netInterfaces map[string]struct{} // Stores all valid network interfaces
|
||||
@@ -48,6 +49,9 @@ type Agent struct {
|
||||
keys []gossh.PublicKey // SSH public keys
|
||||
smartManager *SmartManager // Manages SMART data
|
||||
systemdManager *systemdManager // Manages systemd services
|
||||
monitorManager *MonitorManager // Manages network monitors
|
||||
storagePoolManager *StoragePoolManager // Manages storage pool and dataset data
|
||||
packageUpdates *packageUpdatesManager // Checks for pending package updates
|
||||
}
|
||||
|
||||
// NewAgent creates a new agent with the given data directory for persisting data.
|
||||
@@ -121,6 +125,22 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
// initialize handler registry
|
||||
agent.handlerRegistry = NewHandlerRegistry()
|
||||
|
||||
// initialize monitor manager
|
||||
agent.monitorManager = newMonitorManager()
|
||||
|
||||
agent.storagePoolManager = newStoragePoolManager()
|
||||
|
||||
// Retain ZFS_INTERVAL for the shared storage pool detail refresh interval.
|
||||
if zfsIntervalEnv, exists := utils.GetEnv("ZFS_INTERVAL"); exists {
|
||||
if duration, err := time.ParseDuration(zfsIntervalEnv); err == nil && duration > 0 {
|
||||
agent.storagePoolManager.detailInterval = duration
|
||||
agent.systemDetails.ZfsInterval = duration
|
||||
slog.Info("ZFS_INTERVAL", "duration", duration)
|
||||
} else {
|
||||
slog.Warn("Invalid ZFS_INTERVAL", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// initialize disk info
|
||||
agent.initializeDiskInfo()
|
||||
|
||||
@@ -137,6 +157,8 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
|
||||
slog.Debug("SMART", "err", err)
|
||||
}
|
||||
|
||||
agent.packageUpdates = newPackageUpdatesManager(agent.dataDir)
|
||||
|
||||
// initialize GPU manager
|
||||
agent.gpuManager, err = NewGPUManager()
|
||||
if err != nil {
|
||||
@@ -178,6 +200,11 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
}
|
||||
}
|
||||
|
||||
if a.monitorManager != nil {
|
||||
data.Monitors = a.monitorManager.GetResults(cacheTimeMs)
|
||||
slog.Debug("Monitors", "data", data.Monitors)
|
||||
}
|
||||
|
||||
// skip updating systemd services if cache time is not the default 60sec interval
|
||||
if a.systemdManager != nil && cacheTimeMs == defaultDataCacheTimeMs {
|
||||
totalCount := uint16(a.systemdManager.getServiceStatsCount())
|
||||
@@ -187,13 +214,29 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
}
|
||||
if a.systemdManager.hasFreshStats {
|
||||
data.SystemdServices = a.systemdManager.getServiceStats(nil, false)
|
||||
data.SystemdServicesUpdated = true
|
||||
// Preserve an explicit zero count so the hub can distinguish a fresh
|
||||
// empty snapshot from a response that omitted systemd data.
|
||||
if totalCount == 0 {
|
||||
data.Info.Services = []uint16{0, 0}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if a.packageUpdates != nil {
|
||||
data.Info.PackageUpdates = a.packageUpdates.get(time.Now())
|
||||
}
|
||||
|
||||
data.Stats.ExtraFs = make(map[string]*system.FsStats)
|
||||
data.Info.ExtraFsPct = make(map[string]float64)
|
||||
for name, stats := range a.fsStats {
|
||||
if !stats.Root && stats.DiskTotal > 0 {
|
||||
if stats.Root {
|
||||
if stats.Name != "" {
|
||||
data.Info.RootDiskName = stats.Name
|
||||
}
|
||||
continue
|
||||
}
|
||||
if stats.DiskTotal > 0 {
|
||||
// Use custom name if available, otherwise use device name
|
||||
key := name
|
||||
if stats.Name != "" {
|
||||
@@ -217,7 +260,11 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
|
||||
// Start initializes and starts the agent with optional WebSocket connection
|
||||
func (a *Agent) Start(serverOptions ServerOptions) error {
|
||||
a.keys = serverOptions.Keys
|
||||
return a.connectionManager.Start(serverOptions)
|
||||
err := a.connectionManager.Start(serverOptions)
|
||||
if err != nil {
|
||||
a.cleanupSensorShadow()
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (a *Agent) getFingerprint() string {
|
||||
|
||||
@@ -1,6 +1,13 @@
|
||||
// Package battery provides functions to check if the system has a battery and return the charge state and percentage.
|
||||
// Package battery provides battery information for the host and connected devices.
|
||||
package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
const (
|
||||
stateUnknown uint8 = iota
|
||||
stateEmpty
|
||||
@@ -9,3 +16,58 @@ const (
|
||||
stateDischarging
|
||||
stateIdle
|
||||
)
|
||||
|
||||
// Battery is a readable battery reported by the operating system.
|
||||
type Battery struct {
|
||||
Name string
|
||||
Percent uint8
|
||||
State uint8
|
||||
FullChargeCapacity uint64
|
||||
HasFullChargeCapacity bool
|
||||
System bool
|
||||
}
|
||||
|
||||
var errNoBatteries = errors.New("no readable batteries")
|
||||
|
||||
// normalizeBatteries supplies stable fallback names and disambiguates duplicates.
|
||||
func normalizeBatteries(batteries []Battery) []Battery {
|
||||
nameCounts := make(map[string]int, len(batteries))
|
||||
for i := range batteries {
|
||||
// Names come from firmware (e.g. sysfs model_name) and are not guaranteed to
|
||||
// be valid UTF-8. Invalid bytes are rejected when the hub decodes the CBOR
|
||||
// payload, which drops every metric for the system, so strip them here.
|
||||
name := strings.TrimSpace(strings.ToValidUTF8(batteries[i].Name, ""))
|
||||
if name == "" {
|
||||
name = "Battery " + strconv.Itoa(i+1)
|
||||
}
|
||||
nameCounts[name]++
|
||||
if nameCounts[name] > 1 {
|
||||
name += " (" + strconv.Itoa(nameCounts[name]) + ")"
|
||||
}
|
||||
batteries[i].Name = name
|
||||
}
|
||||
return batteries
|
||||
}
|
||||
|
||||
// Primary returns the representative battery. Reported full-charge capacity wins,
|
||||
// then system-scoped devices, then name for deterministic ties.
|
||||
func Primary(batteries []Battery) (Battery, bool) {
|
||||
if len(batteries) == 0 {
|
||||
return Battery{}, false
|
||||
}
|
||||
ordered := append([]Battery(nil), batteries...)
|
||||
sort.SliceStable(ordered, func(i, j int) bool {
|
||||
a, b := ordered[i], ordered[j]
|
||||
if a.HasFullChargeCapacity != b.HasFullChargeCapacity {
|
||||
return a.HasFullChargeCapacity
|
||||
}
|
||||
if a.HasFullChargeCapacity && a.FullChargeCapacity != b.FullChargeCapacity {
|
||||
return a.FullChargeCapacity > b.FullChargeCapacity
|
||||
}
|
||||
if a.System != b.System {
|
||||
return a.System
|
||||
}
|
||||
return a.Name < b.Name
|
||||
})
|
||||
return ordered[0], true
|
||||
}
|
||||
|
||||
@@ -3,11 +3,7 @@
|
||||
package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"math"
|
||||
"os/exec"
|
||||
"sync"
|
||||
|
||||
"howett.net/plist"
|
||||
)
|
||||
@@ -35,62 +31,46 @@ func readMacBatteries() ([]macBattery, error) {
|
||||
return batteries, nil
|
||||
}
|
||||
|
||||
// HasReadableBattery checks if the system has a battery and returns true if it does.
|
||||
var HasReadableBattery = sync.OnceValue(func() bool {
|
||||
systemHasBattery := false
|
||||
batteries, err := readMacBatteries()
|
||||
slog.Debug("Batteries", "batteries", batteries, "err", err)
|
||||
for _, bat := range batteries {
|
||||
if bat.MaxCapacity > 0 {
|
||||
systemHasBattery = true
|
||||
break
|
||||
}
|
||||
}
|
||||
return systemHasBattery
|
||||
})
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
// GetBatteryStats returns the current battery percent and charge state.
|
||||
// Uses CurrentCapacity/MaxCapacity to match the value macOS displays.
|
||||
func GetBatteryStats() (batteryPercent uint8, batteryState uint8, err error) {
|
||||
if !HasReadableBattery() {
|
||||
return batteryPercent, batteryState, errors.ErrUnsupported
|
||||
}
|
||||
// GetBatteryStats returns every readable battery reported by macOS.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
batteries, err := readMacBatteries()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(batteries) == 0 {
|
||||
return batteryPercent, batteryState, errors.New("no batteries")
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
|
||||
totalCapacity := 0
|
||||
totalCharge := 0
|
||||
batteryState = math.MaxUint8
|
||||
|
||||
result := make([]Battery, 0, len(batteries))
|
||||
for _, bat := range batteries {
|
||||
if bat.MaxCapacity == 0 {
|
||||
if bat.MaxCapacity <= 0 {
|
||||
// skip ghost batteries with 0 capacity
|
||||
// https://github.com/distatus/battery/issues/34
|
||||
continue
|
||||
}
|
||||
totalCapacity += bat.MaxCapacity
|
||||
totalCharge += min(bat.CurrentCapacity, bat.MaxCapacity)
|
||||
|
||||
percent := min(max(float64(bat.CurrentCapacity)/float64(bat.MaxCapacity)*100, 0), 100)
|
||||
state := stateUnknown
|
||||
switch {
|
||||
case !bat.ExternalConnected:
|
||||
batteryState = stateDischarging
|
||||
state = stateDischarging
|
||||
case bat.IsCharging:
|
||||
batteryState = stateCharging
|
||||
state = stateCharging
|
||||
case bat.CurrentCapacity == 0:
|
||||
batteryState = stateEmpty
|
||||
state = stateEmpty
|
||||
case !bat.FullyCharged:
|
||||
batteryState = stateIdle
|
||||
state = stateIdle
|
||||
default:
|
||||
batteryState = stateFull
|
||||
state = stateFull
|
||||
}
|
||||
result = append(result, Battery{Name: "Primary", Percent: uint8(percent), State: state,
|
||||
FullChargeCapacity: uint64(bat.MaxCapacity), HasFullChargeCapacity: true, System: true})
|
||||
}
|
||||
|
||||
if totalCapacity == 0 || batteryState == math.MaxUint8 {
|
||||
return batteryPercent, batteryState, errors.New("no battery capacity")
|
||||
if len(result) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
|
||||
batteryPercent = uint8(float64(totalCharge) / float64(totalCapacity) * 100)
|
||||
return batteryPercent, batteryState, nil
|
||||
return normalizeBatteries(result), nil
|
||||
}
|
||||
|
||||
@@ -3,58 +3,19 @@
|
||||
package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"sync"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
)
|
||||
|
||||
// getBatteryPaths returns the paths of all batteries in /sys/class/power_supply
|
||||
var getBatteryPaths func() ([]string, error)
|
||||
var batteryRoot = "/sys/class/power_supply"
|
||||
|
||||
// HasReadableBattery checks if the system has a battery and returns true if it does.
|
||||
var HasReadableBattery func() bool
|
||||
|
||||
func init() {
|
||||
resetBatteryState("/sys/class/power_supply")
|
||||
}
|
||||
|
||||
// resetBatteryState resets the sync.Once functions to a fresh state.
|
||||
// Tests call this after swapping sysfsPowerSupply so the new path is picked up.
|
||||
func resetBatteryState(sysfsPowerSupplyPath string) {
|
||||
getBatteryPaths = sync.OnceValues(func() ([]string, error) {
|
||||
entries, err := os.ReadDir(sysfsPowerSupplyPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var paths []string
|
||||
for _, e := range entries {
|
||||
path := filepath.Join(sysfsPowerSupplyPath, e.Name())
|
||||
if utils.ReadStringFile(filepath.Join(path, "type")) == "Battery" {
|
||||
paths = append(paths, path)
|
||||
}
|
||||
}
|
||||
return paths, nil
|
||||
})
|
||||
HasReadableBattery = sync.OnceValue(func() bool {
|
||||
systemHasBattery := false
|
||||
paths, err := getBatteryPaths()
|
||||
for _, path := range paths {
|
||||
if _, ok := utils.ReadStringFileOK(filepath.Join(path, "capacity")); ok {
|
||||
systemHasBattery = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !systemHasBattery {
|
||||
slog.Debug("No battery found", "err", err)
|
||||
}
|
||||
return systemHasBattery
|
||||
})
|
||||
// HasReadableBattery reports whether collection currently finds a readable battery.
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
func parseSysfsState(status string) uint8 {
|
||||
@@ -74,26 +35,18 @@ func parseSysfsState(status string) uint8 {
|
||||
}
|
||||
}
|
||||
|
||||
// GetBatteryStats returns the current battery percent and charge state.
|
||||
// Reads /sys/class/power_supply/*/capacity directly so the kernel-reported
|
||||
// value is used, which is always 0-100 and matches what the OS displays.
|
||||
func GetBatteryStats() (batteryPercent uint8, batteryState uint8, err error) {
|
||||
if !HasReadableBattery() {
|
||||
return batteryPercent, batteryState, errors.ErrUnsupported
|
||||
}
|
||||
paths, err := getBatteryPaths()
|
||||
// GetBatteryStats re-enumerates power supplies and returns every readable battery.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
entries, err := os.ReadDir(batteryRoot)
|
||||
if err != nil {
|
||||
return batteryPercent, batteryState, err
|
||||
return nil, err
|
||||
}
|
||||
if len(paths) == 0 {
|
||||
return batteryPercent, batteryState, errors.New("no batteries")
|
||||
}
|
||||
|
||||
batteryState = math.MaxUint8
|
||||
totalPercent := 0
|
||||
count := 0
|
||||
|
||||
for _, path := range paths {
|
||||
batteries := make([]Battery, 0, len(entries))
|
||||
for _, entry := range entries {
|
||||
path := filepath.Join(batteryRoot, entry.Name())
|
||||
if utils.ReadStringFile(filepath.Join(path, "type")) != "Battery" {
|
||||
continue
|
||||
}
|
||||
capStr, ok := utils.ReadStringFileOK(filepath.Join(path, "capacity"))
|
||||
if !ok {
|
||||
continue
|
||||
@@ -102,19 +55,31 @@ func GetBatteryStats() (batteryPercent uint8, batteryState uint8, err error) {
|
||||
if parseErr != nil {
|
||||
continue
|
||||
}
|
||||
totalPercent += cap
|
||||
count++
|
||||
|
||||
state := parseSysfsState(utils.ReadStringFile(filepath.Join(path, "status")))
|
||||
if state != stateUnknown {
|
||||
batteryState = state
|
||||
cap = min(max(cap, 0), 100)
|
||||
name := utils.ReadStringFile(filepath.Join(path, "model_name"))
|
||||
if name == "" {
|
||||
name = utils.ReadStringFile(filepath.Join(path, "model"))
|
||||
}
|
||||
if name == "" {
|
||||
name = entry.Name()
|
||||
}
|
||||
battery := Battery{
|
||||
Name: name,
|
||||
Percent: uint8(cap),
|
||||
State: parseSysfsState(utils.ReadStringFile(filepath.Join(path, "status"))),
|
||||
System: utils.ReadStringFile(filepath.Join(path, "scope")) != "Device",
|
||||
}
|
||||
for _, fullName := range []string{"charge_full", "energy_full"} {
|
||||
if parsed, ok := utils.ReadUintFile(filepath.Join(path, fullName)); ok && parsed > 0 {
|
||||
battery.FullChargeCapacity = parsed
|
||||
battery.HasFullChargeCapacity = true
|
||||
break
|
||||
}
|
||||
}
|
||||
batteries = append(batteries, battery)
|
||||
}
|
||||
|
||||
if count == 0 || batteryState == math.MaxUint8 {
|
||||
return batteryPercent, batteryState, errors.New("no battery capacity")
|
||||
if len(batteries) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
|
||||
batteryPercent = uint8(totalPercent / count)
|
||||
return batteryPercent, batteryState, nil
|
||||
return normalizeBatteries(batteries), nil
|
||||
}
|
||||
|
||||
@@ -8,194 +8,102 @@ import (
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// setupFakeSysfs creates a temporary sysfs-like tree under t.TempDir(),
|
||||
// swaps sysfsPowerSupply, resets the sync.Once caches, and restores
|
||||
// everything on cleanup. Returns a helper to create battery directories.
|
||||
func setupFakeSysfs(t *testing.T) (tmpDir string, addBattery func(name, capacity, status string)) {
|
||||
type fakeBattery struct{ id, name, capacity, status, full, scope string }
|
||||
|
||||
func setupFakeSysfs(t *testing.T) (string, func(fakeBattery)) {
|
||||
t.Helper()
|
||||
|
||||
tmp := t.TempDir()
|
||||
resetBatteryState(tmp)
|
||||
|
||||
write := func(path, content string) {
|
||||
root := t.TempDir()
|
||||
previousRoot := batteryRoot
|
||||
batteryRoot = root
|
||||
t.Cleanup(func() { batteryRoot = previousRoot })
|
||||
write := func(path, value string) {
|
||||
t.Helper()
|
||||
dir := filepath.Dir(path)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(value), 0o644))
|
||||
}
|
||||
add := func(b fakeBattery) {
|
||||
t.Helper()
|
||||
dir := filepath.Join(root, b.id)
|
||||
write(filepath.Join(dir, "type"), "Battery")
|
||||
if b.capacity != "" {
|
||||
write(filepath.Join(dir, "capacity"), b.capacity)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
write(filepath.Join(dir, "status"), b.status)
|
||||
if b.name != "" {
|
||||
write(filepath.Join(dir, "model_name"), b.name)
|
||||
}
|
||||
if b.full != "" {
|
||||
write(filepath.Join(dir, "energy_full"), b.full)
|
||||
}
|
||||
if b.scope != "" {
|
||||
write(filepath.Join(dir, "scope"), b.scope)
|
||||
}
|
||||
}
|
||||
|
||||
addBattery = func(name, capacity, status string) {
|
||||
t.Helper()
|
||||
batDir := filepath.Join(tmp, name)
|
||||
write(filepath.Join(batDir, "type"), "Battery")
|
||||
write(filepath.Join(batDir, "capacity"), capacity)
|
||||
write(filepath.Join(batDir, "status"), status)
|
||||
}
|
||||
|
||||
return tmp, addBattery
|
||||
return root, add
|
||||
}
|
||||
|
||||
func TestParseSysfsState(t *testing.T) {
|
||||
tests := []struct {
|
||||
input string
|
||||
want uint8
|
||||
}{
|
||||
{"Empty", stateEmpty},
|
||||
{"Full", stateFull},
|
||||
{"Charging", stateCharging},
|
||||
{"Discharging", stateDischarging},
|
||||
{"Not charging", stateIdle},
|
||||
{"", stateUnknown},
|
||||
{"SomethingElse", stateUnknown},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
assert.Equal(t, tt.want, parseSysfsState(tt.input), "parseSysfsState(%q)", tt.input)
|
||||
}
|
||||
assert.Equal(t, stateEmpty, parseSysfsState("Empty"))
|
||||
assert.Equal(t, stateFull, parseSysfsState("Full"))
|
||||
assert.Equal(t, stateCharging, parseSysfsState("Charging"))
|
||||
assert.Equal(t, stateDischarging, parseSysfsState("Discharging"))
|
||||
assert.Equal(t, stateIdle, parseSysfsState("Not charging"))
|
||||
assert.Equal(t, stateUnknown, parseSysfsState("SomethingElse"))
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_SingleBattery(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "72", "Discharging")
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, uint8(72), pct)
|
||||
assert.Equal(t, stateDischarging, state)
|
||||
func TestGetBatteryStatsMultipleNamedAndPrimary(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", name: "Primary", capacity: "105", status: "Charging", full: "5000", scope: "System"})
|
||||
add(fakeBattery{id: "hidpp_battery_0", name: "MX Keys S", capacity: "55", status: "Unknown", full: "900", scope: "Device"})
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, batteries, 2)
|
||||
assert.Equal(t, "Primary", batteries[0].Name)
|
||||
assert.Equal(t, uint8(100), batteries[0].Percent)
|
||||
assert.Equal(t, stateUnknown, batteries[1].State)
|
||||
primary, ok := Primary(batteries)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, "Primary", primary.Name)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_MultipleBatteries(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "80", "Charging")
|
||||
addBattery("BAT1", "40", "Charging")
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
// average of 80 and 40 = 60
|
||||
assert.EqualValues(t, 60, pct)
|
||||
assert.Equal(t, stateCharging, state)
|
||||
func TestGetBatteryStatsFallbackDuplicatesAndUnreadable(t *testing.T) {
|
||||
root, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", name: "Keyboard", capacity: "80", status: "Discharging"})
|
||||
add(fakeBattery{id: "BAT1", name: "Keyboard", capacity: "-4", status: "SomethingWeird"})
|
||||
add(fakeBattery{id: "BAT2", capacity: "not-a-number", status: "Charging"})
|
||||
add(fakeBattery{id: "BAT3", capacity: "42", status: "Full"})
|
||||
ac := filepath.Join(root, "AC0")
|
||||
require.NoError(t, os.MkdirAll(ac, 0o755))
|
||||
require.NoError(t, os.WriteFile(filepath.Join(ac, "type"), []byte("Mains"), 0o644))
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, batteries, 3)
|
||||
assert.Equal(t, "Keyboard", batteries[0].Name)
|
||||
assert.Equal(t, "Keyboard (2)", batteries[1].Name)
|
||||
assert.Equal(t, uint8(0), batteries[1].Percent)
|
||||
assert.Equal(t, "BAT3", batteries[2].Name)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_FullBattery(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "100", "Full")
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, uint8(100), pct)
|
||||
assert.Equal(t, stateFull, state)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_EmptyBattery(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "0", "Empty")
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, uint8(0), pct)
|
||||
assert.Equal(t, stateEmpty, state)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_NotCharging(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "80", "Not charging")
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, uint8(80), pct)
|
||||
assert.Equal(t, stateIdle, state)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_NoBatteries(t *testing.T) {
|
||||
setupFakeSysfs(t) // empty directory, no batteries
|
||||
|
||||
_, _, err := GetBatteryStats()
|
||||
func TestGetBatteryStatsHotPlugReenumerates(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
_, err := GetBatteryStats()
|
||||
assert.Error(t, err)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_NonBatterySupplyIgnored(t *testing.T) {
|
||||
tmp, addBattery := setupFakeSysfs(t)
|
||||
|
||||
// Add a real battery
|
||||
addBattery("BAT0", "55", "Charging")
|
||||
|
||||
// Add an AC adapter (type != Battery) - should be ignored
|
||||
acDir := filepath.Join(tmp, "AC0")
|
||||
if err := os.MkdirAll(acDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(acDir, "type"), []byte("Mains"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
pct, state, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, uint8(55), pct)
|
||||
assert.Equal(t, stateCharging, state)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_InvalidCapacitySkipped(t *testing.T) {
|
||||
tmp, addBattery := setupFakeSysfs(t)
|
||||
|
||||
// One battery with valid capacity
|
||||
addBattery("BAT0", "90", "Discharging")
|
||||
|
||||
// Another with invalid capacity text
|
||||
badDir := filepath.Join(tmp, "BAT1")
|
||||
if err := os.MkdirAll(badDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(badDir, "type"), []byte("Battery"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(badDir, "capacity"), []byte("not-a-number"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(badDir, "status"), []byte("Discharging"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
pct, _, err := GetBatteryStats()
|
||||
assert.NoError(t, err)
|
||||
// Only BAT0 counted
|
||||
assert.Equal(t, uint8(90), pct)
|
||||
}
|
||||
|
||||
func TestGetBatteryStats_UnknownStatusOnly(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "50", "SomethingWeird")
|
||||
|
||||
_, _, err := GetBatteryStats()
|
||||
assert.Error(t, err)
|
||||
}
|
||||
|
||||
func TestHasReadableBattery_True(t *testing.T) {
|
||||
_, addBattery := setupFakeSysfs(t)
|
||||
addBattery("BAT0", "50", "Charging")
|
||||
|
||||
assert.False(t, HasReadableBattery())
|
||||
add(fakeBattery{id: "BAT0", capacity: "64", status: "Discharging"})
|
||||
batteries, err := GetBatteryStats()
|
||||
require.NoError(t, err)
|
||||
assert.True(t, HasReadableBattery())
|
||||
require.Len(t, batteries, 1)
|
||||
assert.Equal(t, uint8(64), batteries[0].Percent)
|
||||
}
|
||||
|
||||
func TestHasReadableBattery_False(t *testing.T) {
|
||||
setupFakeSysfs(t) // no batteries
|
||||
|
||||
assert.False(t, HasReadableBattery())
|
||||
}
|
||||
|
||||
func TestHasReadableBattery_NoCapacityFile(t *testing.T) {
|
||||
tmp, _ := setupFakeSysfs(t)
|
||||
|
||||
// Battery dir with type file but no capacity file
|
||||
batDir := filepath.Join(tmp, "BAT0")
|
||||
err := os.MkdirAll(batDir, 0o755)
|
||||
assert.NoError(t, err)
|
||||
err = os.WriteFile(filepath.Join(batDir, "type"), []byte("Battery"), 0o644)
|
||||
assert.NoError(t, err)
|
||||
|
||||
func TestGetBatteryStatsNoReadableCapacity(t *testing.T) {
|
||||
_, add := setupFakeSysfs(t)
|
||||
add(fakeBattery{id: "BAT0", status: "Charging"})
|
||||
_, err := GetBatteryStats()
|
||||
assert.Error(t, err)
|
||||
assert.False(t, HasReadableBattery())
|
||||
}
|
||||
|
||||
@@ -8,6 +8,6 @@ func HasReadableBattery() bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func GetBatteryStats() (uint8, uint8, error) {
|
||||
return 0, 0, errors.ErrUnsupported
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
return nil, errors.ErrUnsupported
|
||||
}
|
||||
|
||||
48
agent/battery/battery_test.go
Normal file
48
agent/battery/battery_test.go
Normal file
@@ -0,0 +1,48 @@
|
||||
package battery
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestPrimarySelection(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
bats []Battery
|
||||
want string
|
||||
}{
|
||||
{"largest reported capacity", []Battery{{Name: "Small", FullChargeCapacity: 20, HasFullChargeCapacity: true, System: true}, {Name: "Large", FullChargeCapacity: 80, HasFullChargeCapacity: true}}, "Large"},
|
||||
{"reported ranks over missing", []Battery{{Name: "Unknown", System: true}, {Name: "Known", FullChargeCapacity: 1, HasFullChargeCapacity: true}}, "Known"},
|
||||
{"system wins capacity tie", []Battery{{Name: "Peripheral", FullChargeCapacity: 50, HasFullChargeCapacity: true}, {Name: "System", FullChargeCapacity: 50, HasFullChargeCapacity: true, System: true}}, "System"},
|
||||
{"name resolves final tie", []Battery{{Name: "Zed"}, {Name: "Alpha"}}, "Alpha"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, ok := Primary(tt.bats)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, tt.want, got.Name)
|
||||
})
|
||||
}
|
||||
_, ok := Primary(nil)
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
func TestNormalizeBatteriesFallbackNames(t *testing.T) {
|
||||
bats := normalizeBatteries([]Battery{{}, {}, {Name: "Mouse"}, {Name: "Mouse"}})
|
||||
assert.Equal(t, []string{"Battery 1", "Battery 2", "Mouse", "Mouse (2)"}, []string{bats[0].Name, bats[1].Name, bats[2].Name, bats[3].Name})
|
||||
}
|
||||
|
||||
func TestNormalizeBatteriesStripsInvalidUTF8(t *testing.T) {
|
||||
// Firmware occasionally reports names that are not valid UTF-8 (a ThinkPad
|
||||
// reporting "LNV-5B11K63024@\xd0" in model_name is a real example).
|
||||
bats := normalizeBatteries([]Battery{{Name: "LNV-5B11K63024@\xd0"}, {Name: "\xff\xfe"}})
|
||||
assert.Equal(t, "LNV-5B11K63024@", bats[0].Name)
|
||||
// A name made up entirely of invalid bytes falls back to the generic name.
|
||||
assert.Equal(t, "Battery 2", bats[1].Name)
|
||||
for _, b := range bats {
|
||||
assert.True(t, utf8.ValidString(b.Name))
|
||||
}
|
||||
}
|
||||
@@ -7,9 +7,6 @@ package battery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"math"
|
||||
"sync"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
|
||||
@@ -79,7 +76,7 @@ var (
|
||||
setupDiDestroyDeviceInfoList = setupapi.NewProc("SetupDiDestroyDeviceInfoList")
|
||||
)
|
||||
|
||||
// winBatteryGet reads one battery by index. Returns (fullCapacity, currentCapacity, state, error).
|
||||
// winBatteryGet reads one battery by index.
|
||||
// Returns error == errNotFound when there are no more batteries.
|
||||
var errNotFound = errors.New("no more batteries")
|
||||
|
||||
@@ -122,7 +119,7 @@ func readWinBatteryState(powerState uint32) uint8 {
|
||||
}
|
||||
}
|
||||
|
||||
func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
func winBatteryGet(idx int) (Battery, error) {
|
||||
hdev, err := setupDiSetup(
|
||||
setupDiGetClassDevsW,
|
||||
4,
|
||||
@@ -132,7 +129,7 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
0, 0,
|
||||
)
|
||||
if err != nil {
|
||||
return 0, 0, stateUnknown, err
|
||||
return Battery{}, err
|
||||
}
|
||||
defer syscall.SyscallN(setupDiDestroyDeviceInfoList.Addr(), hdev)
|
||||
|
||||
@@ -148,10 +145,10 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
0,
|
||||
)
|
||||
if errno == 259 { // ERROR_NO_MORE_ITEMS
|
||||
return 0, 0, stateUnknown, errNotFound
|
||||
return Battery{}, errNotFound
|
||||
}
|
||||
if errno != 0 {
|
||||
return 0, 0, stateUnknown, errno
|
||||
return Battery{}, errno
|
||||
}
|
||||
|
||||
var cbRequired uint32
|
||||
@@ -165,7 +162,7 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
0,
|
||||
)
|
||||
if errno != 0 && errno != 122 { // ERROR_INSUFFICIENT_BUFFER
|
||||
return 0, 0, stateUnknown, errno
|
||||
return Battery{}, errno
|
||||
}
|
||||
didd := make([]uint16, cbRequired/2)
|
||||
cbSize := (*uint32)(unsafe.Pointer(&didd[0]))
|
||||
@@ -185,7 +182,7 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
0,
|
||||
)
|
||||
if errno != 0 {
|
||||
return 0, 0, stateUnknown, errno
|
||||
return Battery{}, errno
|
||||
}
|
||||
devicePath := &didd[2:][0]
|
||||
|
||||
@@ -199,7 +196,7 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
0,
|
||||
)
|
||||
if err != nil {
|
||||
return 0, 0, stateUnknown, err
|
||||
return Battery{}, err
|
||||
}
|
||||
defer windows.CloseHandle(handle)
|
||||
|
||||
@@ -216,7 +213,7 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
&dwOut, nil,
|
||||
)
|
||||
if err != nil || bqi.BatteryTag == 0 {
|
||||
return 0, 0, stateUnknown, errors.New("battery tag not returned")
|
||||
return Battery{}, errors.New("battery tag not returned")
|
||||
}
|
||||
|
||||
var bi batteryInformation
|
||||
@@ -229,7 +226,21 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
uint32(unsafe.Sizeof(bi)),
|
||||
&dwOut, nil,
|
||||
); err != nil {
|
||||
return 0, 0, stateUnknown, err
|
||||
return Battery{}, err
|
||||
}
|
||||
|
||||
// BatteryDeviceName is optional, so retain the deterministic fallback on error.
|
||||
name := ""
|
||||
nameQuery := bqi
|
||||
nameQuery.InformationLevel = 4 // BatteryDeviceName
|
||||
nameBuffer := make([]uint16, 128)
|
||||
if err := windows.DeviceIoControl(
|
||||
handle, 2703428,
|
||||
(*byte)(unsafe.Pointer(&nameQuery)), uint32(unsafe.Sizeof(nameQuery)),
|
||||
(*byte)(unsafe.Pointer(&nameBuffer[0])), uint32(len(nameBuffer)*2),
|
||||
&dwOut, nil,
|
||||
); err == nil {
|
||||
name = windows.UTF16ToString(nameBuffer)
|
||||
}
|
||||
|
||||
bws := batteryWaitStatus{BatteryTag: bqi.BatteryTag}
|
||||
@@ -243,56 +254,38 @@ func winBatteryGet(idx int) (full, current uint32, state uint8, err error) {
|
||||
uint32(unsafe.Sizeof(bs)),
|
||||
&dwOut, nil,
|
||||
); err != nil {
|
||||
return 0, 0, stateUnknown, err
|
||||
return Battery{}, err
|
||||
}
|
||||
|
||||
if bs.Capacity == 0xffffffff { // BATTERY_UNKNOWN_CAPACITY
|
||||
return 0, 0, stateUnknown, errors.New("battery capacity unknown")
|
||||
if bs.Capacity == 0xffffffff || bi.FullChargedCapacity == 0 || bi.FullChargedCapacity == 0xffffffff {
|
||||
return Battery{}, errors.New("battery capacity unknown")
|
||||
}
|
||||
|
||||
return bi.FullChargedCapacity, bs.Capacity, readWinBatteryState(bs.PowerState), nil
|
||||
percent := min(float64(bs.Capacity)/float64(bi.FullChargedCapacity)*100, 100)
|
||||
return Battery{Name: name, Percent: uint8(percent), State: readWinBatteryState(bs.PowerState),
|
||||
FullChargeCapacity: uint64(bi.FullChargedCapacity), HasFullChargeCapacity: true, System: true}, nil
|
||||
}
|
||||
|
||||
// HasReadableBattery checks if the system has a battery and returns true if it does.
|
||||
var HasReadableBattery = sync.OnceValue(func() bool {
|
||||
systemHasBattery := false
|
||||
full, _, _, err := winBatteryGet(0)
|
||||
if err == nil && full > 0 {
|
||||
systemHasBattery = true
|
||||
}
|
||||
if !systemHasBattery {
|
||||
slog.Debug("No battery found", "err", err)
|
||||
}
|
||||
return systemHasBattery
|
||||
})
|
||||
|
||||
// GetBatteryStats returns the current battery percent and charge state.
|
||||
func GetBatteryStats() (batteryPercent uint8, batteryState uint8, err error) {
|
||||
if !HasReadableBattery() {
|
||||
return batteryPercent, batteryState, errors.ErrUnsupported
|
||||
}
|
||||
|
||||
totalFull := uint32(0)
|
||||
totalCurrent := uint32(0)
|
||||
batteryState = math.MaxUint8
|
||||
func HasReadableBattery() bool {
|
||||
batteries, _ := GetBatteryStats()
|
||||
return len(batteries) > 0
|
||||
}
|
||||
|
||||
// GetBatteryStats returns every readable battery reported by Windows.
|
||||
func GetBatteryStats() ([]Battery, error) {
|
||||
batteries := make([]Battery, 0, 2)
|
||||
for i := 0; ; i++ {
|
||||
full, current, state, bErr := winBatteryGet(i)
|
||||
battery, bErr := winBatteryGet(i)
|
||||
if errors.Is(bErr, errNotFound) {
|
||||
break
|
||||
}
|
||||
if bErr != nil || full == 0 {
|
||||
if bErr != nil {
|
||||
continue
|
||||
}
|
||||
totalFull += full
|
||||
totalCurrent += min(current, full)
|
||||
batteryState = state
|
||||
batteries = append(batteries, battery)
|
||||
}
|
||||
|
||||
if totalFull == 0 || batteryState == math.MaxUint8 {
|
||||
return batteryPercent, batteryState, errors.New("no battery capacity")
|
||||
if len(batteries) == 0 {
|
||||
return nil, errNoBatteries
|
||||
}
|
||||
|
||||
batteryPercent = uint8(float64(totalCurrent) / float64(totalFull) * 100)
|
||||
return batteryPercent, batteryState, nil
|
||||
return normalizeBatteries(batteries), nil
|
||||
}
|
||||
|
||||
26
agent/btrfs/btrfs.go
Normal file
26
agent/btrfs/btrfs.go
Normal file
@@ -0,0 +1,26 @@
|
||||
// Package btrfs reads btrfs filesystem state from sysfs.
|
||||
package btrfs
|
||||
|
||||
// Filesystem is a mounted btrfs filesystem read from /sys/fs/btrfs/<uuid>.
|
||||
type Filesystem struct {
|
||||
UUID string // stable filesystem UUID from sysfs
|
||||
MountID string // kernel filesystem identity for matching monitored mounts
|
||||
IODevice string // sole member block-device name, empty for multi-device/unknown pools
|
||||
Name string // label, else first mountpoint, else UUID
|
||||
Size uint64 // effective usable capacity, or raw member capacity when Raw
|
||||
Raw bool // capacity and usage are physical bytes, unsuitable for disk alerts
|
||||
Alloc uint64 // raw bytes allocated to data, metadata and system chunks
|
||||
Health string // ONLINE, or DEGRADED when a device is missing
|
||||
NRead uint64 // cumulative bytes read across member devices
|
||||
NWrite uint64 // cumulative bytes written across member devices
|
||||
Devices []Device
|
||||
}
|
||||
|
||||
// Device is one member device (devinfo/<devid>) with its error counters.
|
||||
type Device struct {
|
||||
Name string // "devid N"; sysfs does not expose the block device path
|
||||
State string // ONLINE or MISSING
|
||||
ReadErrs uint64
|
||||
WriteErrs uint64
|
||||
CorruptionErrs uint64
|
||||
}
|
||||
285
agent/btrfs/btrfs_linux.go
Normal file
285
agent/btrfs/btrfs_linux.go
Normal file
@@ -0,0 +1,285 @@
|
||||
//go:build linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unsafe"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
var (
|
||||
sysfsPath = "/sys/fs/btrfs"
|
||||
mountsPath = "/proc/self/mounts"
|
||||
mountinfoPath = "/proc/self/mountinfo"
|
||||
mountUUID = MountID
|
||||
deviceSize = ioctlDeviceSize
|
||||
filesystemUsage = statfsUsage
|
||||
)
|
||||
|
||||
// Filesystems returns all mounted btrfs filesystems, or nil when there are none.
|
||||
func Filesystems() ([]Filesystem, error) {
|
||||
entries, err := os.ReadDir(sysfsPath)
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
mounts := mountpointsByDevice()
|
||||
var filesystems []Filesystem
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() || entry.Name() == "features" {
|
||||
continue
|
||||
}
|
||||
fs, err := readFilesystem(filepath.Join(sysfsPath, entry.Name()), mounts)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("btrfs %s: %w", entry.Name(), err)
|
||||
}
|
||||
filesystems = append(filesystems, fs)
|
||||
}
|
||||
return filesystems, nil
|
||||
}
|
||||
|
||||
func readFilesystem(dir string, mounts map[string]string) (Filesystem, error) {
|
||||
fs := Filesystem{UUID: filepath.Base(dir), Name: utils.ReadStringFile(filepath.Join(dir, "label")), Health: "UNKNOWN"}
|
||||
for _, kind := range []string{"data", "metadata", "system"} {
|
||||
if value, ok := utils.ReadUintFile(filepath.Join(dir, "allocation", kind, "disk_used")); ok {
|
||||
fs.Alloc += value
|
||||
}
|
||||
}
|
||||
// devices/<name> links to the block device's sysfs directory.
|
||||
devices, err := os.ReadDir(filepath.Join(dir, "devices"))
|
||||
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fs, err
|
||||
}
|
||||
mountpoint := mounts["uuid:"+fs.UUID]
|
||||
if fs.Name == "" {
|
||||
fs.Name = mountpoint
|
||||
}
|
||||
var backingSize uint64
|
||||
for _, dev := range devices {
|
||||
if mountpoint == "" {
|
||||
mountpoint = mounts[dev.Name()]
|
||||
}
|
||||
if fs.Name == "" {
|
||||
fs.Name = mountpoint
|
||||
}
|
||||
devDir := filepath.Join(dir, "devices", dev.Name())
|
||||
if size, ok := utils.ReadUintFile(filepath.Join(devDir, "size")); ok {
|
||||
backingSize += size * 512
|
||||
}
|
||||
if stat := strings.Fields(utils.ReadStringFile(filepath.Join(devDir, "stat"))); len(stat) >= 7 {
|
||||
fs.NRead += parseUint(stat[2]) * 512
|
||||
fs.NWrite += parseUint(stat[6]) * 512
|
||||
}
|
||||
}
|
||||
devids, err := os.ReadDir(filepath.Join(dir, "devinfo"))
|
||||
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fs, err
|
||||
}
|
||||
capacityAvailable := len(devids) > 0
|
||||
healthKnown := len(devids) > 0
|
||||
for _, devid := range devids {
|
||||
devDir := filepath.Join(dir, "devinfo", devid.Name())
|
||||
// Replacement targets do not add filesystem capacity.
|
||||
replaceTarget, _ := utils.ReadUintFile(filepath.Join(devDir, "replace_target"))
|
||||
if replaceTarget != 1 {
|
||||
devid, err := strconv.ParseUint(devid.Name(), 10, 64)
|
||||
if err != nil {
|
||||
return fs, err
|
||||
}
|
||||
size, err := deviceSize(mountpoint, devid)
|
||||
if err != nil {
|
||||
capacityAvailable = false
|
||||
}
|
||||
fs.Size += size
|
||||
}
|
||||
dev := Device{Name: "devid " + devid.Name(), State: "ONLINE"}
|
||||
missing := utils.ReadStringFile(filepath.Join(devDir, "missing"))
|
||||
if missing != "0" && missing != "1" {
|
||||
healthKnown = false
|
||||
dev.State = "UNKNOWN"
|
||||
}
|
||||
if missing == "1" {
|
||||
dev.State = "MISSING"
|
||||
fs.Health = "DEGRADED"
|
||||
}
|
||||
for line := range strings.Lines(utils.ReadStringFile(filepath.Join(devDir, "error_stats"))) {
|
||||
if fields := strings.Fields(line); len(fields) == 2 {
|
||||
switch fields[0] {
|
||||
case "read_errs":
|
||||
dev.ReadErrs = parseUint(fields[1])
|
||||
case "write_errs":
|
||||
dev.WriteErrs = parseUint(fields[1])
|
||||
case "corruption_errs":
|
||||
dev.CorruptionErrs = parseUint(fields[1])
|
||||
}
|
||||
}
|
||||
}
|
||||
fs.Devices = append(fs.Devices, dev)
|
||||
}
|
||||
// Use one capacity source for the whole filesystem: device IDs cannot be
|
||||
// reliably matched to block-device names in sysfs. A partial ioctl result
|
||||
// must not be added to the complete backing-device total.
|
||||
if !capacityAvailable {
|
||||
fs.Size = backingSize
|
||||
}
|
||||
if fs.Health != "DEGRADED" && healthKnown {
|
||||
fs.Health = "ONLINE"
|
||||
}
|
||||
fs.MountID = mountUUID(mountpoint)
|
||||
if len(devices) == 1 && len(devids) == 1 && fs.Health == "ONLINE" {
|
||||
fs.IODevice = devices[0].Name()
|
||||
}
|
||||
fs.Raw = true
|
||||
if used, available, err := filesystemUsage(mountpoint); err == nil {
|
||||
// Effective capacity excludes reserved/unavailable space, so Size-Alloc
|
||||
// is available to applications and the usage ratio matches df.
|
||||
fs.Size, fs.Alloc, fs.Raw = used+available, used, false
|
||||
}
|
||||
if fs.Name == "" {
|
||||
fs.Name = filepath.Base(dir)
|
||||
}
|
||||
return fs, nil
|
||||
}
|
||||
|
||||
// mountpointsByDevice prefers UUID matches from mountinfo and retains source
|
||||
// device names as a fallback for environments where FS_INFO is unavailable.
|
||||
func mountpointsByDevice() map[string]string {
|
||||
mounts := mountpointsByUUID(utils.ReadStringFile(mountinfoPath), mountUUID)
|
||||
for line := range strings.Lines(utils.ReadStringFile(mountsPath)) {
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 3 || fields[2] != "btrfs" {
|
||||
continue
|
||||
}
|
||||
device := fields[0]
|
||||
if resolved, err := filepath.EvalSymlinks(device); err == nil {
|
||||
device = resolved
|
||||
}
|
||||
if _, seen := mounts[filepath.Base(device)]; !seen {
|
||||
mounts[filepath.Base(device)] = unescapeMountPath(fields[1])
|
||||
}
|
||||
}
|
||||
return mounts
|
||||
}
|
||||
|
||||
func parseUint(s string) uint64 {
|
||||
n, _ := strconv.ParseUint(s, 10, 64)
|
||||
return n
|
||||
}
|
||||
|
||||
// ioctlDeviceSize reads Btrfs's recorded device size, which can be smaller
|
||||
// than the block device after a filesystem resize. BTRFS_IOC_DEV_INFO is
|
||||
// _IOWR(0x94, 30, struct btrfs_ioctl_dev_info_args), a 4096-byte ABI structure.
|
||||
func ioctlDeviceSize(mountpoint string, devid uint64) (uint64, error) {
|
||||
if mountpoint == "" {
|
||||
return 0, errors.New("no accessible mountpoint")
|
||||
}
|
||||
f, err := os.Open(mountpoint)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer f.Close()
|
||||
args := struct {
|
||||
Devid uint64
|
||||
UUID [16]byte
|
||||
BytesUsed uint64
|
||||
TotalBytes uint64
|
||||
Reserved [4096 - 40]byte
|
||||
}{Devid: devid}
|
||||
_, _, errno := unix.Syscall(unix.SYS_IOCTL, f.Fd(), 0xd000941e, uintptr(unsafe.Pointer(&args)))
|
||||
if errno != 0 {
|
||||
return 0, errno
|
||||
}
|
||||
return args.TotalBytes, nil
|
||||
}
|
||||
|
||||
// The filesystem magic is unsigned even when Statfs_t.Type is int32.
|
||||
func isBtrfs(stat *unix.Statfs_t) bool {
|
||||
return uint32(stat.Type) == unix.BTRFS_SUPER_MAGIC
|
||||
}
|
||||
|
||||
func statfsUsage(path string) (used, available uint64, err error) {
|
||||
if path == "" {
|
||||
return 0, 0, errors.New("no accessible mountpoint")
|
||||
}
|
||||
var stat unix.Statfs_t
|
||||
if err = unix.Statfs(path, &stat); err != nil {
|
||||
return
|
||||
}
|
||||
if !isBtrfs(&stat) {
|
||||
return 0, 0, errors.New("mountpoint is not Btrfs")
|
||||
}
|
||||
blockSize := uint64(stat.Bsize)
|
||||
return (stat.Blocks - min(stat.Blocks, stat.Bfree)) * blockSize, min(stat.Blocks, stat.Bavail) * blockSize, nil
|
||||
}
|
||||
|
||||
// MountID returns the filesystem UUID via BTRFS_IOC_FS_INFO. Unlike statfs
|
||||
// f_fsid, this identity is shared by all subvolumes and bind mounts.
|
||||
func MountID(path string) string {
|
||||
if path == "" {
|
||||
return ""
|
||||
}
|
||||
var stat unix.Statfs_t
|
||||
if unix.Statfs(path, &stat) != nil || !isBtrfs(&stat) {
|
||||
return ""
|
||||
}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer f.Close()
|
||||
args := struct {
|
||||
MaxID uint64
|
||||
NumDevices uint64
|
||||
FSID [16]byte
|
||||
Reserved [992]byte
|
||||
}{}
|
||||
// _IOR(0x94, 31, 1024). Reuse the platform's read-direction bits;
|
||||
// MIPS/PowerPC use a different encoding than asm-generic.
|
||||
request := uintptr(unix.FS_IOC_GETFLAGS&0xe0000000) | 0x0400941f
|
||||
_, _, errno := unix.Syscall(unix.SYS_IOCTL, f.Fd(), request, uintptr(unsafe.Pointer(&args)))
|
||||
if errno != 0 {
|
||||
return ""
|
||||
}
|
||||
id := args.FSID
|
||||
return fmt.Sprintf("%x-%x-%x-%x-%x", id[:4], id[4:6], id[6:8], id[8:10], id[10:])
|
||||
}
|
||||
|
||||
// Btrfs mountinfo device numbers can be virtual (0:N), so query the UUID
|
||||
// through the mount instead of comparing those numbers with sysfs block devs.
|
||||
// Retry another path when a bind mount is inaccessible. Once resolved, reuse
|
||||
// the result for that mount device to avoid opening every Docker bind mount.
|
||||
func mountpointsByUUID(mountinfo string, identify func(string) string) map[string]string {
|
||||
mounts := make(map[string]string)
|
||||
resolved := make(map[string]bool)
|
||||
for line := range strings.Lines(mountinfo) {
|
||||
before, after, ok := strings.Cut(line, " - ")
|
||||
fields, fs := strings.Fields(before), strings.Fields(after)
|
||||
if !ok || len(fields) < 6 || len(fs) < 3 || fs[0] != "btrfs" || resolved[fields[2]] {
|
||||
continue
|
||||
}
|
||||
path := unescapeMountPath(fields[4])
|
||||
uuid := identify(path)
|
||||
if uuid == "" {
|
||||
continue
|
||||
}
|
||||
resolved[fields[2]] = true
|
||||
if mounts["uuid:"+uuid] == "" {
|
||||
mounts["uuid:"+uuid] = path
|
||||
}
|
||||
}
|
||||
return mounts
|
||||
}
|
||||
|
||||
func unescapeMountPath(path string) string {
|
||||
return strings.NewReplacer(`\040`, " ", `\011`, "\t", `\012`, "\n", `\134`, `\`).Replace(path)
|
||||
}
|
||||
274
agent/btrfs/btrfs_linux_test.go
Normal file
274
agent/btrfs/btrfs_linux_test.go
Normal file
@@ -0,0 +1,274 @@
|
||||
//go:build testing && linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
func TestFilesystems(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
oldSysfs, oldMounts := sysfsPath, mountsPath
|
||||
sysfsPath, mountsPath = root, filepath.Join(root, "mounts")
|
||||
t.Cleanup(func() { sysfsPath, mountsPath = oldSysfs, oldMounts })
|
||||
|
||||
fsDir := filepath.Join(root, "1b2c3d4e-0000-0000-0000-000000000000")
|
||||
write := func(rel, content string) {
|
||||
path := filepath.Join(fsDir, rel)
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(content), 0o644))
|
||||
}
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "features"), 0o755))
|
||||
oldUsage := filesystemUsage
|
||||
filesystemUsage = func(string) (uint64, uint64, error) { return 0, 0, os.ErrNotExist }
|
||||
t.Cleanup(func() { filesystemUsage = oldUsage })
|
||||
oldDeviceSize := deviceSize
|
||||
t.Cleanup(func() { deviceSize = oldDeviceSize })
|
||||
deviceSize = func(_ string, devid uint64) (uint64, error) {
|
||||
value, _ := utils.ReadUintFile(filepath.Join(fsDir, "recorded-size", strconv.FormatUint(devid, 10)))
|
||||
return value, nil
|
||||
}
|
||||
// Recorded member capacities differ from the unchanged backing devices.
|
||||
write("recorded-size/1", "256000\n")
|
||||
write("recorded-size/2", "128000\n")
|
||||
write("label", "tank\n")
|
||||
write("allocation/data/disk_used", "4096\n")
|
||||
write("allocation/metadata/disk_used", "2048\n")
|
||||
write("allocation/system/disk_used", "1024\n")
|
||||
write("devices/sda/size", "1000\n")
|
||||
write("devices/sda/stat", "10 0 200 0 20 0 400 0 0 0 0\n")
|
||||
write("devices/sdb/size", "1000\n")
|
||||
write("devices/sdb/stat", "10 0 100 0 20 0 100 0 0 0 0\n")
|
||||
write("devinfo/1/missing", "0\n")
|
||||
write("devinfo/1/error_stats", "write_errs 1\nread_errs 2\nflush_errs 0\ncorruption_errs 3\ngeneration_errs 0\n")
|
||||
write("devinfo/2/missing", "1\n")
|
||||
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, Filesystem{
|
||||
UUID: "1b2c3d4e-0000-0000-0000-000000000000", Raw: true, Name: "tank", Size: 384000, Alloc: 7168, Health: "DEGRADED", NRead: 153600, NWrite: 256000,
|
||||
Devices: []Device{
|
||||
{Name: "devid 1", State: "ONLINE", ReadErrs: 2, WriteErrs: 1, CorruptionErrs: 3},
|
||||
{Name: "devid 2", State: "MISSING"},
|
||||
},
|
||||
}, filesystems[0])
|
||||
|
||||
// Unlabeled filesystems fall back to the first mountpoint, then the UUID.
|
||||
write("label", "\n")
|
||||
require.NoError(t, os.WriteFile(mountsPath, []byte(
|
||||
"/dev/sdz1 /other btrfs rw 0 0\n/dev/sdb /mnt/storage btrfs rw 0 0\n/dev/sdb /mnt/storage/sub btrfs rw,subvol=/sub 0 0\n",
|
||||
), 0o644))
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "/mnt/storage", filesystems[0].Name)
|
||||
|
||||
require.NoError(t, os.Remove(mountsPath))
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "1b2c3d4e-0000-0000-0000-000000000000", filesystems[0].Name)
|
||||
write("devinfo/3/replace_target", "1\n")
|
||||
write("recorded-size/3", "512000\n")
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(384000), filesystems[0].Size, "replacement target must not inflate capacity")
|
||||
|
||||
deviceSize = func(string, uint64) (uint64, error) { return 0, os.ErrPermission }
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
assert.Equal(t, "DEGRADED", filesystems[0].Health)
|
||||
assert.Equal(t, uint64(153600), filesystems[0].NRead)
|
||||
|
||||
// A partial ioctl result must not be mixed with the backing-device total.
|
||||
deviceSize = func(_ string, devid uint64) (uint64, error) {
|
||||
if devid == 2 {
|
||||
return 0, os.ErrPermission
|
||||
}
|
||||
return 256000, nil
|
||||
}
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
|
||||
// With no mount visible (e.g. Docker), the real lookup falls back too.
|
||||
deviceSize = ioctlDeviceSize
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, filesystems, 1)
|
||||
assert.Equal(t, uint64(1024000), filesystems[0].Size)
|
||||
|
||||
filesystemUsage = func(string) (uint64, uint64, error) { return 100, 900, nil }
|
||||
filesystems, err = Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1000), filesystems[0].Size)
|
||||
assert.Equal(t, uint64(100), filesystems[0].Alloc)
|
||||
assert.False(t, filesystems[0].Raw)
|
||||
}
|
||||
|
||||
func TestFilesystemsNoBtrfs(t *testing.T) {
|
||||
oldPath := sysfsPath
|
||||
sysfsPath = filepath.Join(t.TempDir(), "missing")
|
||||
t.Cleanup(func() { sysfsPath = oldPath })
|
||||
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
assert.Nil(t, filesystems)
|
||||
}
|
||||
|
||||
func TestIoctlDeviceSizeFailure(t *testing.T) {
|
||||
_, err := ioctlDeviceSize("", 1)
|
||||
require.Error(t, err)
|
||||
_, err = ioctlDeviceSize(t.TempDir(), 1)
|
||||
require.Error(t, err)
|
||||
assert.ErrorIs(t, err, unix.ENOTTY)
|
||||
}
|
||||
|
||||
func TestMountpointsDecodeEscapes(t *testing.T) {
|
||||
oldMounts := mountsPath
|
||||
mountsPath = filepath.Join(t.TempDir(), "mounts")
|
||||
t.Cleanup(func() { mountsPath = oldMounts })
|
||||
require.NoError(t, os.WriteFile(mountsPath, []byte("/dev/test-btrfs /mnt/my\\040data btrfs rw 0 0\n"), 0o644))
|
||||
assert.Equal(t, "/mnt/my data", mountpointsByDevice()["test-btrfs"])
|
||||
}
|
||||
|
||||
func TestFilesystemWithoutDevinfo(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "devices", "sda"), 0755))
|
||||
require.NoError(t, os.WriteFile(filepath.Join(root, "devices", "sda", "size"), []byte("1000"), 0644))
|
||||
fs, err := readFilesystem(root, nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(512000), fs.Size)
|
||||
assert.True(t, fs.Raw)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
assert.Empty(t, fs.Devices)
|
||||
|
||||
require.NoError(t, os.MkdirAll(filepath.Join(root, "devinfo", "1"), 0755))
|
||||
fs, err = readFilesystem(root, nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
require.Len(t, fs.Devices, 1)
|
||||
assert.Equal(t, "UNKNOWN", fs.Devices[0].State)
|
||||
|
||||
// Some older interfaces lack the devices directory too.
|
||||
fs, err = readFilesystem(t.TempDir(), nil)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "UNKNOWN", fs.Health)
|
||||
}
|
||||
|
||||
func TestLocalBtrfsUsage(t *testing.T) {
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for read-only live validation")
|
||||
}
|
||||
used, available, err := statfsUsage(path)
|
||||
require.NoError(t, err)
|
||||
filesystems, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
for _, fs := range filesystems {
|
||||
if !fs.Raw && fs.Alloc == used && fs.Size == used+available {
|
||||
t.Logf("pool=%s used=%d available=%d effective_capacity=%d", fs.Name, used, available, fs.Size)
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatal("collector did not report the mounted filesystem's usable capacity")
|
||||
}
|
||||
|
||||
func TestMountID(t *testing.T) {
|
||||
assert.Empty(t, MountID(""))
|
||||
assert.Empty(t, MountID(filepath.Join(t.TempDir(), "missing")))
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for live identity validation")
|
||||
}
|
||||
id := MountID(path)
|
||||
require.NotEmpty(t, id)
|
||||
assert.Equal(t, id, MountID(filepath.Join(path, ".")))
|
||||
}
|
||||
|
||||
func TestMountinfoUUIDLookup(t *testing.T) {
|
||||
info := `1 0 0:40 /@ /inaccessible ro shared:1 - btrfs /dev/mapper/unavailable rw
|
||||
2 0 0:40 /@/docker/hosts /etc/hosts ro - btrfs /dev/mapper/unavailable rw
|
||||
3 0 0:40 /@/docker/hostname /etc/hostname ro - btrfs /dev/mapper/unavailable rw
|
||||
4 0 0:41 /subvol /extra-filesystems/my\040disk ro master:2 - btrfs /dev/missing rw
|
||||
5 0 0:42 / /ext4 ro - ext4 /dev/mapper/unavailable rw
|
||||
malformed
|
||||
6 0 0:43 / /bad ro - btrfs
|
||||
`
|
||||
var calls []string
|
||||
mounts := mountpointsByUUID(info, func(path string) string {
|
||||
calls = append(calls, path)
|
||||
switch path {
|
||||
case "/etc/hosts":
|
||||
return "root-uuid"
|
||||
case "/extra-filesystems/my disk":
|
||||
return "extra-uuid"
|
||||
}
|
||||
return ""
|
||||
})
|
||||
assert.Equal(t, map[string]string{"uuid:root-uuid": "/etc/hosts", "uuid:extra-uuid": "/extra-filesystems/my disk"}, mounts)
|
||||
assert.Equal(t, []string{"/inaccessible", "/etc/hosts", "/extra-filesystems/my disk"}, calls)
|
||||
}
|
||||
|
||||
func TestDockerFilesystemWithoutDeviceNodes(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
oldSysfs, oldMounts, oldInfo, oldUUID, oldUsage := sysfsPath, mountsPath, mountinfoPath, mountUUID, filesystemUsage
|
||||
t.Cleanup(func() {
|
||||
sysfsPath, mountsPath, mountinfoPath, mountUUID, filesystemUsage = oldSysfs, oldMounts, oldInfo, oldUUID, oldUsage
|
||||
})
|
||||
sysfsPath = filepath.Join(root, "sysfs")
|
||||
mountsPath = filepath.Join(root, "missing-mounts")
|
||||
mountinfoPath = filepath.Join(root, "mountinfo")
|
||||
uuid := "11111111-1111-4111-8111-111111111111"
|
||||
dir := filepath.Join(sysfsPath, uuid)
|
||||
for path, content := range map[string]string{"devices/dm-0/size": "1000", "devinfo/1/missing": "0"} {
|
||||
target := filepath.Join(dir, path)
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(target), 0755))
|
||||
require.NoError(t, os.WriteFile(target, []byte(content), 0644))
|
||||
}
|
||||
require.NoError(t, os.WriteFile(mountinfoPath, []byte("2 1 0:40 /@/docker/hosts /etc/hosts ro - btrfs /dev/mapper/not-in-container rw\n"), 0644))
|
||||
mountUUID = func(path string) string {
|
||||
if path == "/etc/hosts" {
|
||||
return uuid
|
||||
}
|
||||
return ""
|
||||
}
|
||||
filesystemUsage = func(path string) (uint64, uint64, error) { require.Equal(t, "/etc/hosts", path); return 100, 900, nil }
|
||||
fs, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, fs, 1)
|
||||
assert.Equal(t, uuid, fs[0].MountID)
|
||||
assert.Equal(t, "dm-0", fs[0].IODevice)
|
||||
assert.False(t, fs[0].Raw)
|
||||
assert.Equal(t, uint64(1000), fs[0].Size)
|
||||
}
|
||||
|
||||
func TestLivePoolMountIdentity(t *testing.T) {
|
||||
path := os.Getenv("BESZEL_TEST_BTRFS_MOUNT")
|
||||
if path == "" {
|
||||
t.Skip("set BESZEL_TEST_BTRFS_MOUNT for live validation")
|
||||
}
|
||||
id := MountID(path)
|
||||
require.NotEmpty(t, id)
|
||||
pools, err := Filesystems()
|
||||
require.NoError(t, err)
|
||||
for _, pool := range pools {
|
||||
if pool.UUID != id {
|
||||
continue
|
||||
}
|
||||
assert.Equal(t, id, pool.MountID)
|
||||
assert.False(t, pool.Raw)
|
||||
t.Logf("uuid=%s mount_identity=%s io_device=%s raw=%v", pool.UUID, pool.MountID, pool.IODevice, pool.Raw)
|
||||
return
|
||||
}
|
||||
t.Fatal("mounted Btrfs filesystem was not discovered")
|
||||
}
|
||||
11
agent/btrfs/btrfs_unsupported.go
Normal file
11
agent/btrfs/btrfs_unsupported.go
Normal file
@@ -0,0 +1,11 @@
|
||||
//go:build !linux
|
||||
|
||||
package btrfs
|
||||
|
||||
import "errors"
|
||||
|
||||
func Filesystems() ([]Filesystem, error) {
|
||||
return nil, errors.ErrUnsupported
|
||||
}
|
||||
|
||||
func MountID(string) string { return "" }
|
||||
@@ -2,6 +2,7 @@ package agent
|
||||
|
||||
import (
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
@@ -24,9 +25,28 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
wsDeadline = 70 * time.Second
|
||||
// Keep the connection alive long enough for a slow collection cycle to
|
||||
// finish before the hub considers the agent disconnected.
|
||||
wsDeadline = 120 * time.Second
|
||||
)
|
||||
|
||||
// errNoHubURL is returned when HUB_URL is unset. This is not a failure
|
||||
// condition: an agent configured with only a public key runs in SSH-only mode,
|
||||
// where the hub dials the agent and no outbound WebSocket client is expected.
|
||||
var errNoHubURL = errors.New("HUB_URL environment variable not set")
|
||||
|
||||
type caCertFileError struct {
|
||||
err error
|
||||
}
|
||||
|
||||
func (e *caCertFileError) Error() string {
|
||||
return e.err.Error()
|
||||
}
|
||||
|
||||
func (e *caCertFileError) Unwrap() error {
|
||||
return e.err
|
||||
}
|
||||
|
||||
// WebSocketClient manages the WebSocket connection between the agent and hub.
|
||||
// It handles authentication, message routing, and connection lifecycle management.
|
||||
type WebSocketClient struct {
|
||||
@@ -40,6 +60,7 @@ type WebSocketClient struct {
|
||||
hubRequest *common.HubRequest[cbor.RawMessage] // Reusable request structure for message parsing
|
||||
lastConnectAttempt time.Time // Timestamp of last connection attempt
|
||||
hubVerified bool // Whether the hub has been cryptographically verified
|
||||
tlsConfig *tls.Config // Optional TLS configuration with custom CA certificates
|
||||
}
|
||||
|
||||
// newWebSocketClient creates a new WebSocket client for the given agent.
|
||||
@@ -47,20 +68,24 @@ type WebSocketClient struct {
|
||||
func newWebSocketClient(agent *Agent) (client *WebSocketClient, err error) {
|
||||
hubURLStr, exists := utils.GetEnv("HUB_URL")
|
||||
if !exists {
|
||||
return nil, errors.New("HUB_URL environment variable not set")
|
||||
return nil, errNoHubURL
|
||||
}
|
||||
|
||||
client = &WebSocketClient{}
|
||||
|
||||
client.hubURL, err = url.Parse(hubURLStr)
|
||||
if err != nil {
|
||||
return nil, errors.New("invalid hub URL")
|
||||
if err != nil || client.hubURL.Host == "" {
|
||||
return nil, fmt.Errorf("invalid HUB_URL %q: must include scheme and host (e.g. http://hub.example.com:8090)", hubURLStr)
|
||||
}
|
||||
// get registration token
|
||||
client.token, err = getToken()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
client.tlsConfig, err = getTLSConfig()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
client.agent = agent
|
||||
client.hubRequest = &common.HubRequest[cbor.RawMessage]{}
|
||||
@@ -87,7 +112,52 @@ func getToken() (string, error) {
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return strings.TrimSpace(string(tokenBytes)), nil
|
||||
return parseTokenFile(string(tokenBytes), tokenFile)
|
||||
}
|
||||
|
||||
// parseTokenFile reads a single token from TOKEN_FILE.
|
||||
// Blank lines and comments are ignored. Multiple tokens are rejected because
|
||||
// the agent supports only one outbound hub connection.
|
||||
func parseTokenFile(contents, path string) (string, error) {
|
||||
var token string
|
||||
for line := range strings.Lines(contents) {
|
||||
line = strings.TrimSpace(line)
|
||||
if len(line) == 0 || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
if token != "" {
|
||||
return "", fmt.Errorf("%s must contain a single token", path)
|
||||
}
|
||||
token = line
|
||||
}
|
||||
// An empty file keeps returning an empty token, as before: the caller decides
|
||||
// what to do about it.
|
||||
return token, nil
|
||||
}
|
||||
|
||||
// getTLSConfig returns a TLS configuration containing the system certificate
|
||||
// pool plus any certificates configured through CA_CERT_FILE. A nil config lets
|
||||
// gws use Go's default TLS configuration and system roots.
|
||||
func getTLSConfig() (*tls.Config, error) {
|
||||
caCertFile, _ := utils.GetEnv("CA_CERT_FILE")
|
||||
if caCertFile == "" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
caCertPEM, err := os.ReadFile(caCertFile)
|
||||
if err != nil {
|
||||
return nil, &caCertFileError{fmt.Errorf("read CA_CERT_FILE %q: %w", caCertFile, err)}
|
||||
}
|
||||
|
||||
rootCAs, err := x509.SystemCertPool()
|
||||
if err != nil {
|
||||
return nil, &caCertFileError{fmt.Errorf("load system CA certificate pool: %w", err)}
|
||||
}
|
||||
if !rootCAs.AppendCertsFromPEM(caCertPEM) {
|
||||
return nil, &caCertFileError{fmt.Errorf("CA_CERT_FILE %q does not contain any valid PEM certificates", caCertFile)}
|
||||
}
|
||||
|
||||
return &tls.Config{RootCAs: rootCAs}, nil
|
||||
}
|
||||
|
||||
// getOptions returns the WebSocket client options, creating them if necessary.
|
||||
@@ -112,7 +182,7 @@ func (client *WebSocketClient) getOptions() *gws.ClientOption {
|
||||
|
||||
client.options = &gws.ClientOption{
|
||||
Addr: client.hubURL.String(),
|
||||
TlsConfig: &tls.Config{InsecureSkipVerify: true},
|
||||
TlsConfig: client.tlsConfig,
|
||||
RequestHeader: http.Header{
|
||||
"User-Agent": []string{getUserAgent()},
|
||||
"X-Token": []string{client.token},
|
||||
|
||||
@@ -4,8 +4,19 @@ package agent
|
||||
|
||||
import (
|
||||
"crypto/ed25519"
|
||||
"crypto/rand"
|
||||
"crypto/rsa"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"crypto/x509/pkix"
|
||||
"encoding/pem"
|
||||
"math/big"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -15,11 +26,34 @@ import (
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/lxzan/gws"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/crypto/ssh"
|
||||
)
|
||||
|
||||
// TestNewWebSocketClientNoHubURL verifies that an unset HUB_URL returns the
|
||||
// errNoHubURL sentinel rather than an opaque error. Callers rely on this to
|
||||
// distinguish SSH-only mode -- a supported configuration in which the hub dials
|
||||
// the agent -- from an actual misconfiguration.
|
||||
func TestNewWebSocketClientNoHubURL(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
|
||||
// t.Setenv registers restoration of the original value; unset afterwards so
|
||||
// GetEnv's LookupEnv reports the variable as absent rather than empty.
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "")
|
||||
os.Unsetenv("BESZEL_AGENT_HUB_URL")
|
||||
t.Setenv("HUB_URL", "")
|
||||
os.Unsetenv("HUB_URL")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
|
||||
client, err := newWebSocketClient(agent)
|
||||
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, client)
|
||||
assert.ErrorIs(t, err, errNoHubURL)
|
||||
}
|
||||
|
||||
// TestNewWebSocketClient tests WebSocket client creation
|
||||
func TestNewWebSocketClient(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -51,11 +85,18 @@ func TestNewWebSocketClient(t *testing.T) {
|
||||
errorMsg: "HUB_URL environment variable not set",
|
||||
},
|
||||
{
|
||||
name: "invalid URL",
|
||||
name: "malformed URL",
|
||||
hubURL: "ht\ttp://invalid",
|
||||
token: "test-token",
|
||||
expectError: true,
|
||||
errorMsg: "invalid hub URL",
|
||||
errorMsg: "invalid HUB_URL",
|
||||
},
|
||||
{
|
||||
name: "URL without host",
|
||||
hubURL: "http:/api",
|
||||
token: "test-token",
|
||||
expectError: true,
|
||||
errorMsg: "invalid HUB_URL",
|
||||
},
|
||||
{
|
||||
name: "missing token",
|
||||
@@ -157,6 +198,155 @@ func TestWebSocketClient_GetOptions(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestWebSocketClient_TLSVerification(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
serverCert, serverCertPEM := newSelfSignedServerCertificate(t)
|
||||
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
|
||||
server := httptest.NewUnstartedServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
conn, err := upgrader.Upgrade(w, r)
|
||||
if err == nil {
|
||||
go conn.ReadLoop()
|
||||
}
|
||||
}))
|
||||
server.TLS = &tls.Config{Certificates: []tls.Certificate{serverCert}}
|
||||
server.StartTLS()
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
caCertFile := filepath.Join(t.TempDir(), "hub-ca.crt")
|
||||
require.NoError(t, os.WriteFile(caCertFile, serverCertPEM, 0600))
|
||||
|
||||
newClient := func(t *testing.T, caCertFile string) *WebSocketClient {
|
||||
t.Helper()
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", caCertFile)
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
return client
|
||||
}
|
||||
|
||||
t.Run("system roots are used by default", func(t *testing.T) {
|
||||
client := newClient(t, "")
|
||||
assert.Nil(t, client.getOptions().TlsConfig)
|
||||
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("custom CA trusts self-signed certificate", func(t *testing.T) {
|
||||
systemRoots, err := x509.SystemCertPool()
|
||||
require.NoError(t, err)
|
||||
client := newClient(t, caCertFile)
|
||||
assert.Greater(t, len(client.getOptions().TlsConfig.RootCAs.Subjects()), len(systemRoots.Subjects()))
|
||||
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, conn.NetConn().Close())
|
||||
})
|
||||
|
||||
t.Run("custom CA does not bypass hostname verification", func(t *testing.T) {
|
||||
client := newClient(t, caCertFile)
|
||||
client.getOptions().TlsConfig.ServerName = "wrong.example.com"
|
||||
_, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
func TestWebSocketClient_NonTLSConnection(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
upgrader := gws.NewUpgrader(&gws.BuiltinEventHandler{}, nil)
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
conn, err := upgrader.Upgrade(w, r)
|
||||
if err == nil {
|
||||
go conn.ReadLoop()
|
||||
}
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", server.URL)
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", "")
|
||||
client, err := newWebSocketClient(agent)
|
||||
require.NoError(t, err)
|
||||
assert.Nil(t, client.getOptions().TlsConfig)
|
||||
|
||||
conn, _, err := gws.NewClient(&gws.BuiltinEventHandler{}, client.getOptions())
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, conn.NetConn().Close())
|
||||
}
|
||||
|
||||
func TestGetTLSConfigErrors(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
testCases := []struct {
|
||||
name string
|
||||
path string
|
||||
contents []byte
|
||||
errorMatch string
|
||||
}{
|
||||
{
|
||||
name: "missing file",
|
||||
path: filepath.Join(tempDir, "missing.pem"),
|
||||
errorMatch: "read CA_CERT_FILE",
|
||||
},
|
||||
{
|
||||
name: "unreadable path",
|
||||
path: tempDir,
|
||||
errorMatch: "read CA_CERT_FILE",
|
||||
},
|
||||
{
|
||||
name: "empty file",
|
||||
path: filepath.Join(tempDir, "empty.pem"),
|
||||
contents: []byte{},
|
||||
errorMatch: "does not contain any valid PEM certificates",
|
||||
},
|
||||
{
|
||||
name: "malformed file",
|
||||
path: filepath.Join(tempDir, "malformed.pem"),
|
||||
contents: []byte("not a PEM certificate"),
|
||||
errorMatch: "does not contain any valid PEM certificates",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if tc.contents != nil {
|
||||
require.NoError(t, os.WriteFile(tc.path, tc.contents, 0600))
|
||||
}
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", tc.path)
|
||||
|
||||
tlsConfig, err := getTLSConfig()
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, tlsConfig)
|
||||
assert.Contains(t, err.Error(), tc.errorMatch)
|
||||
assert.Contains(t, err.Error(), tc.path)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func newSelfSignedServerCertificate(t *testing.T) (tls.Certificate, []byte) {
|
||||
t.Helper()
|
||||
privateKey, err := rsa.GenerateKey(rand.Reader, 2048)
|
||||
require.NoError(t, err)
|
||||
|
||||
template := &x509.Certificate{
|
||||
SerialNumber: big.NewInt(1),
|
||||
Subject: pkix.Name{CommonName: "127.0.0.1"},
|
||||
NotBefore: time.Now().Add(-time.Hour),
|
||||
NotAfter: time.Now().Add(time.Hour),
|
||||
IPAddresses: []net.IP{net.ParseIP("127.0.0.1")},
|
||||
KeyUsage: x509.KeyUsageDigitalSignature | x509.KeyUsageKeyEncipherment | x509.KeyUsageCertSign,
|
||||
ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageServerAuth},
|
||||
BasicConstraintsValid: true,
|
||||
IsCA: true,
|
||||
}
|
||||
certDER, err := x509.CreateCertificate(rand.Reader, template, template, &privateKey.PublicKey, privateKey)
|
||||
require.NoError(t, err)
|
||||
|
||||
certPEM := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: certDER})
|
||||
keyPEM := pem.EncodeToMemory(&pem.Block{Type: "RSA PRIVATE KEY", Bytes: x509.MarshalPKCS1PrivateKey(privateKey)})
|
||||
certificate, err := tls.X509KeyPair(certPEM, keyPEM)
|
||||
require.NoError(t, err)
|
||||
return certificate, certPEM
|
||||
}
|
||||
|
||||
// TestWebSocketClient_VerifySignature tests signature verification
|
||||
func TestWebSocketClient_VerifySignature(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -402,6 +592,41 @@ func TestGetToken(t *testing.T) {
|
||||
assert.Equal(t, expectedToken, token)
|
||||
})
|
||||
|
||||
t.Run("TOKEN_FILE with surrounding blank lines and comments", func(t *testing.T) {
|
||||
expectedToken := "test-token-with-noise"
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("# hub token\n\n"+expectedToken+"\n\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, expectedToken, token)
|
||||
})
|
||||
|
||||
t.Run("TOKEN_FILE with multiple tokens is rejected", func(t *testing.T) {
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("11111111-1111-1111-1111-111111111111\n22222222-2222-2222-2222-222222222222\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
require.Error(t, err)
|
||||
assert.Empty(t, token)
|
||||
assert.Contains(t, err.Error(), "must contain a single token")
|
||||
})
|
||||
|
||||
t.Run("TOKEN_FILE holding only comments behaves like an empty file", func(t *testing.T) {
|
||||
tokenFile := filepath.Join(t.TempDir(), "token")
|
||||
require.NoError(t, os.WriteFile(tokenFile, []byte("\n# only a comment\n"), 0o600))
|
||||
|
||||
t.Setenv("TOKEN_FILE", tokenFile)
|
||||
|
||||
token, err := getToken()
|
||||
assert.NoError(t, err)
|
||||
assert.Equal(t, "", token)
|
||||
})
|
||||
|
||||
t.Run("token from BESZEL_AGENT_TOKEN_FILE", func(t *testing.T) {
|
||||
// Create a temporary token file
|
||||
expectedToken := "test-token-from-beszel-file"
|
||||
@@ -497,3 +722,11 @@ func TestGetToken(t *testing.T) {
|
||||
assert.Equal(t, expectedToken, token, "Whitespace should be stripped from token file content")
|
||||
})
|
||||
}
|
||||
|
||||
func TestWebSocketDeadlineCoversSlowCollection(t *testing.T) {
|
||||
const minimumDeadline = 120 * time.Second
|
||||
|
||||
if wsDeadline < minimumDeadline {
|
||||
t.Fatalf("WebSocket deadline %s is shorter than the slow-collection window of %s", wsDeadline, minimumDeadline)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,11 +4,15 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"net"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/health"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
@@ -83,7 +87,19 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
|
||||
wsClient, err := newWebSocketClient(c.agent)
|
||||
if err != nil {
|
||||
slog.Warn("Error creating WebSocket client", "err", err)
|
||||
var caCertErr *caCertFileError
|
||||
if errors.As(err, &caCertErr) {
|
||||
return err
|
||||
}
|
||||
disableSSH, _ := utils.GetEnv("DISABLE_SSH")
|
||||
if errors.Is(err, errNoHubURL) && disableSSH != "true" {
|
||||
// SSH-only mode: the hub dials the agent, so there is nothing to warn
|
||||
// about. With SSH also disabled there is no connection method at all,
|
||||
// so that case still warns.
|
||||
slog.Debug("WebSocket client not configured", "err", err)
|
||||
} else {
|
||||
slog.Warn("Error creating WebSocket client", "err", err)
|
||||
}
|
||||
}
|
||||
c.wsClient = wsClient
|
||||
|
||||
@@ -111,20 +127,47 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
_ = health.Update()
|
||||
case <-sigCtx.Done():
|
||||
slog.Info("Shutting down", "cause", context.Cause(sigCtx))
|
||||
_ = c.agent.StopServer()
|
||||
c.closeWebSocket()
|
||||
return health.CleanUp()
|
||||
return c.stop()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stop does not stop the connection manager itself, just any active connections. The manager will attempt to reconnect after stopping, so this should only be called immediately before shutting down the entire agent.
|
||||
//
|
||||
// If we need or want to expose a graceful Stop method in the future, do something like this to actually stop the manager:
|
||||
//
|
||||
// func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
|
||||
// ctx, cancel := context.WithCancel(context.Background())
|
||||
// c.cancel = cancel
|
||||
//
|
||||
// for {
|
||||
// select {
|
||||
// case <-ctx.Done():
|
||||
// return c.stop()
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
//
|
||||
// func (c *ConnectionManager) Stop() {
|
||||
// c.cancel()
|
||||
// }
|
||||
func (c *ConnectionManager) stop() error {
|
||||
_ = c.agent.StopServer()
|
||||
c.agent.monitorManager.Stop()
|
||||
c.closeWebSocket()
|
||||
c.agent.cleanupSensorShadow()
|
||||
return health.CleanUp()
|
||||
}
|
||||
|
||||
// handleEvent processes connection events and updates the connection state accordingly.
|
||||
func (c *ConnectionManager) handleEvent(event ConnectionEvent) {
|
||||
switch event {
|
||||
case WebSocketConnect:
|
||||
c.handleStateChange(WebSocketConnected)
|
||||
case SSHConnect:
|
||||
c.handleStateChange(SSHConnected)
|
||||
if c.State == Disconnected {
|
||||
c.handleStateChange(SSHConnected)
|
||||
}
|
||||
case WebSocketDisconnect:
|
||||
if c.State == WebSocketConnected {
|
||||
c.handleStateChange(Disconnected)
|
||||
@@ -185,9 +228,16 @@ func (c *ConnectionManager) connect() {
|
||||
|
||||
// Try WebSocket first, if it fails, start SSH server
|
||||
err := c.startWebSocketConnection()
|
||||
if err != nil && c.State == Disconnected {
|
||||
c.startSSHServer()
|
||||
c.startWsTicker()
|
||||
if err != nil {
|
||||
if shouldExitOnErr(err) {
|
||||
time.Sleep(2 * time.Second) // prevent tight restart loop
|
||||
_ = c.stop()
|
||||
os.Exit(1)
|
||||
}
|
||||
if c.State == Disconnected {
|
||||
c.startSSHServer()
|
||||
c.startWsTicker()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,3 +274,14 @@ func (c *ConnectionManager) closeWebSocket() {
|
||||
c.wsClient.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// shouldExitOnErr checks if the error is a DNS resolution failure and if the
|
||||
// EXIT_ON_DNS_ERROR env var is set. https://github.com/henrygd/beszel/issues/1924.
|
||||
func shouldExitOnErr(err error) bool {
|
||||
if val, _ := utils.GetEnv("EXIT_ON_DNS_ERROR"); val == "true" {
|
||||
if opErr, ok := errors.AsType[*net.OpError](err); ok {
|
||||
return strings.Contains(opErr.Err.Error(), "lookup")
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ package agent
|
||||
|
||||
import (
|
||||
"crypto/ed25519"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/url"
|
||||
@@ -113,6 +114,12 @@ func TestConnectionManager_EventHandling(t *testing.T) {
|
||||
event: SSHConnect,
|
||||
expectedState: SSHConnected,
|
||||
},
|
||||
{
|
||||
name: "SSH connect from WebSocket connected (no change)",
|
||||
initialState: WebSocketConnected,
|
||||
event: SSHConnect,
|
||||
expectedState: WebSocketConnected,
|
||||
},
|
||||
{
|
||||
name: "WebSocket disconnect from connected",
|
||||
initialState: WebSocketConnected,
|
||||
@@ -264,6 +271,19 @@ func TestConnectionManager_StartWithInvalidConfig(t *testing.T) {
|
||||
assert.Error(t, err, "Should error when starting already started connection manager")
|
||||
}
|
||||
|
||||
func TestConnectionManager_StartRejectsInvalidCACertFile(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
cm := agent.connectionManager
|
||||
t.Setenv("BESZEL_AGENT_HUB_URL", "https://hub.example.com")
|
||||
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
|
||||
t.Setenv("BESZEL_AGENT_CA_CERT_FILE", t.TempDir())
|
||||
|
||||
err := cm.Start(ServerOptions{})
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "read CA_CERT_FILE")
|
||||
assert.Nil(t, cm.eventChan)
|
||||
}
|
||||
|
||||
// TestConnectionManager_CloseWebSocket tests WebSocket closing
|
||||
func TestConnectionManager_CloseWebSocket(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
@@ -298,3 +318,65 @@ func TestConnectionManager_ConnectFlow(t *testing.T) {
|
||||
cm.connect()
|
||||
}, "Connect should not panic without WebSocket client")
|
||||
}
|
||||
|
||||
func TestShouldExitOnErr(t *testing.T) {
|
||||
createDialErr := func(msg string) error {
|
||||
return &net.OpError{
|
||||
Op: "dial",
|
||||
Net: "tcp",
|
||||
Err: errors.New(msg),
|
||||
}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
envValue string
|
||||
expected bool
|
||||
}{
|
||||
{
|
||||
name: "no env var",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var false",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "false",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var true, matching error",
|
||||
err: createDialErr("lookup lkahsdfasdf: no such host"),
|
||||
envValue: "true",
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
name: "env var true, matching error with extra context",
|
||||
err: createDialErr("lookup beszel.server.lan on [::1]:53: read udp [::1]:44557->[::1]:53: read: connection refused"),
|
||||
envValue: "true",
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
name: "env var true, non-matching error",
|
||||
err: errors.New("connection refused"),
|
||||
envValue: "true",
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
name: "env var true, dial but not lookup",
|
||||
err: createDialErr("connection timeout"),
|
||||
envValue: "true",
|
||||
expected: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
t.Setenv("EXIT_ON_DNS_ERROR", tt.envValue)
|
||||
result := shouldExitOnErr(tt.err)
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,6 +12,14 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func invalidDataDir(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
filePath := filepath.Join(t.TempDir(), "file")
|
||||
require.NoError(t, os.WriteFile(filePath, nil, 0644))
|
||||
return filepath.Join(filePath, "data")
|
||||
}
|
||||
|
||||
func TestGetDataDir(t *testing.T) {
|
||||
// Test with explicit dataDir parameter
|
||||
t.Run("explicit data dir", func(t *testing.T) {
|
||||
@@ -48,7 +56,7 @@ func TestGetDataDir(t *testing.T) {
|
||||
|
||||
// Test with invalid explicit dataDir
|
||||
t.Run("invalid explicit data dir", func(t *testing.T) {
|
||||
invalidPath := "/invalid/path/that/cannot/be/created"
|
||||
invalidPath := invalidDataDir(t)
|
||||
_, err := GetDataDir(invalidPath)
|
||||
assert.Error(t, err)
|
||||
})
|
||||
@@ -78,7 +86,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
// Test with multiple directories, first one valid
|
||||
t.Run("multiple dirs - first valid", func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
invalidDir := "/invalid/path"
|
||||
invalidDir := invalidDataDir(t)
|
||||
result, err := testDataDirs([]string{tempDir, invalidDir})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, tempDir, result)
|
||||
@@ -87,7 +95,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
// Test with multiple directories, second one valid
|
||||
t.Run("multiple dirs - second valid", func(t *testing.T) {
|
||||
tempDir := t.TempDir()
|
||||
invalidDir := "/invalid/path"
|
||||
invalidDir := invalidDataDir(t)
|
||||
result, err := testDataDirs([]string{invalidDir, tempDir})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, tempDir, result)
|
||||
@@ -109,7 +117,7 @@ func TestTestDataDirs(t *testing.T) {
|
||||
|
||||
// Test with no valid directories
|
||||
t.Run("no valid directories", func(t *testing.T) {
|
||||
invalidPaths := []string{"/invalid/path1", "/invalid/path2"}
|
||||
invalidPaths := []string{invalidDataDir(t), invalidDataDir(t)}
|
||||
_, err := testDataDirs(invalidPaths)
|
||||
assert.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "data directory not found")
|
||||
|
||||
177
agent/disk.go
177
agent/disk.go
@@ -3,6 +3,7 @@ package agent
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
@@ -18,7 +19,8 @@ import (
|
||||
// fsRegistrationContext holds the shared lookup state needed to resolve a
|
||||
// filesystem into the tracked fsStats key and metadata.
|
||||
type fsRegistrationContext struct {
|
||||
filesystem string // value of optional FILESYSTEM env var
|
||||
filesystem string // device part of optional FILESYSTEM env var
|
||||
filesystemName string // optional custom name from FILESYSTEM=device__name
|
||||
isWindows bool
|
||||
efPath string // path to extra filesystems (default "/extra-filesystems")
|
||||
diskIoCounters map[string]disk.IOCountersStat
|
||||
@@ -152,12 +154,12 @@ func registerFilesystemStats(existing map[string]*system.FsStats, device, mountp
|
||||
}
|
||||
|
||||
// addFsStat inserts a discovered filesystem if it resolves to a new tracking
|
||||
// key. The key selection itself lives in buildFsStatRegistration so that logic
|
||||
// can stay directly unit-tested.
|
||||
func (d *diskDiscovery) addFsStat(device, mountpoint string, root bool, customName string) {
|
||||
// key and reports whether it was added. The key selection itself lives in
|
||||
// registerFilesystemStats so that logic can stay directly unit-tested.
|
||||
func (d *diskDiscovery) addFsStat(device, mountpoint string, root bool, customName string) bool {
|
||||
key, fsStats, ok := registerFilesystemStats(d.agent.fsStats, device, mountpoint, root, customName, d.ctx)
|
||||
if !ok {
|
||||
return
|
||||
return false
|
||||
}
|
||||
d.agent.fsStats[key] = fsStats
|
||||
name := key
|
||||
@@ -165,6 +167,7 @@ func (d *diskDiscovery) addFsStat(device, mountpoint string, root bool, customNa
|
||||
name = customName
|
||||
}
|
||||
slog.Info("Detected disk", "name", name, "device", device, "mount", mountpoint, "io", key, "root", root)
|
||||
return true
|
||||
}
|
||||
|
||||
// addConfiguredRootFs resolves FILESYSTEM against partitions first, then falls
|
||||
@@ -177,7 +180,7 @@ func (d *diskDiscovery) addConfiguredRootFs() bool {
|
||||
|
||||
for _, p := range d.partitions {
|
||||
if filesystemMatchesPartitionSetting(d.ctx.filesystem, p) {
|
||||
d.addFsStat(p.Device, p.Mountpoint, true, "")
|
||||
d.addFsStat(p.Device, p.Mountpoint, true, d.ctx.filesystemName)
|
||||
return true
|
||||
}
|
||||
}
|
||||
@@ -185,7 +188,7 @@ func (d *diskDiscovery) addConfiguredRootFs() bool {
|
||||
// FILESYSTEM may name a physical disk absent from partitions (e.g. ZFS lists
|
||||
// dataset paths like zroot/ROOT/default, not block devices).
|
||||
if ioKey, match := findIoDevice(d.ctx.filesystem, d.ctx.diskIoCounters); match {
|
||||
d.agent.fsStats[ioKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint}
|
||||
d.agent.fsStats[ioKey] = &system.FsStats{Root: true, Mountpoint: d.rootMountPoint, Name: d.ctx.filesystemName}
|
||||
return true
|
||||
}
|
||||
|
||||
@@ -202,14 +205,24 @@ func isRootFallbackPartition(p disk.PartitionStat, rootMountPoint string) bool {
|
||||
// partition looks like the active root mount but still needs translating to an
|
||||
// I/O device key.
|
||||
func (d *diskDiscovery) addPartitionRootFs(device, mountpoint string) bool {
|
||||
fs, match := findIoDevice(filepath.Base(device), d.ctx.diskIoCounters)
|
||||
// device is passed through as-is: findIoDevice normalizes it, and
|
||||
// filepath.Base would turn a Windows volume name such as "C:" into "\"
|
||||
// on the way in (#2417).
|
||||
fs, match := findIoDevice(device, d.ctx.diskIoCounters)
|
||||
if !match {
|
||||
return false
|
||||
}
|
||||
// The resolved I/O device is already known here, so use it directly to avoid
|
||||
// a second fallback search inside buildFsStatRegistration.
|
||||
d.addFsStat(fs, mountpoint, true, "")
|
||||
return true
|
||||
// The root device is already resolved, so if it was registered earlier as an
|
||||
// extra filesystem (e.g. root drive listed in EXTRA_FILESYSTEMS), promote that
|
||||
// entry rather than letting addLastResortRootFs guess a different device.
|
||||
if stats, exists := d.agent.fsStats[fs]; exists {
|
||||
stats.Root = true
|
||||
stats.Mountpoint = mountpoint
|
||||
return true
|
||||
}
|
||||
// Use the resolved I/O device directly to avoid a second fallback search
|
||||
// inside registerFilesystemStats.
|
||||
return d.addFsStat(fs, mountpoint, true, "")
|
||||
}
|
||||
|
||||
// addLastResortRootFs is only used when neither FILESYSTEM nor partition-based
|
||||
@@ -300,7 +313,8 @@ func (d *diskDiscovery) addExtraFilesystemFolders(folderNames []string) {
|
||||
|
||||
// Sets up the filesystems to monitor for disk usage and I/O.
|
||||
func (a *Agent) initializeDiskInfo() {
|
||||
filesystem, _ := utils.GetEnv("FILESYSTEM")
|
||||
filesystemRaw, _ := utils.GetEnv("FILESYSTEM")
|
||||
filesystem, filesystemName := parseFilesystemEntry(filesystemRaw)
|
||||
hasRoot := false
|
||||
isWindows := runtime.GOOS == "windows"
|
||||
|
||||
@@ -324,6 +338,7 @@ func (a *Agent) initializeDiskInfo() {
|
||||
slog.Debug("Disk I/O", "diskstats", diskIoCounters)
|
||||
ctx := fsRegistrationContext{
|
||||
filesystem: filesystem,
|
||||
filesystemName: filesystemName,
|
||||
isWindows: isWindows,
|
||||
diskIoCounters: diskIoCounters,
|
||||
efPath: "/extra-filesystems",
|
||||
@@ -523,18 +538,57 @@ func filesystemMatchesPartitionSetting(filesystem string, p disk.PartitionStat)
|
||||
|
||||
// normalizeDeviceName canonicalizes device strings for comparisons.
|
||||
func normalizeDeviceName(value string) string {
|
||||
name := filepath.Base(strings.TrimSpace(value))
|
||||
name := strings.TrimSpace(value)
|
||||
if volume, ok := windowsVolumeName(name); ok {
|
||||
return volume
|
||||
}
|
||||
name = filepath.Base(name)
|
||||
if name == "." {
|
||||
return ""
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// windowsVolumeName returns the canonical form of a bare Windows volume
|
||||
// specifier, so that "C:", "c:", `C:\` and "C:/" all name the same drive.
|
||||
// Drive letters are case-insensitive on Windows, so the letter is uppercased.
|
||||
//
|
||||
// filepath.Base cannot do this. On Windows it treats "C:" as a volume name
|
||||
// with no path element to take the base of and returns "\", so every drive
|
||||
// letter normalizes to the same key. findIoDevice then returns whichever
|
||||
// counter the map happened to yield first, which registers the root
|
||||
// filesystem under a random drive (#2417).
|
||||
func windowsVolumeName(value string) (string, bool) {
|
||||
if len(value) < 2 || value[1] != ':' {
|
||||
return "", false
|
||||
}
|
||||
if c := value[0]; !('a' <= c && c <= 'z' || 'A' <= c && c <= 'Z') {
|
||||
return "", false
|
||||
}
|
||||
// Only separators may follow the specifier. "C:data" is a drive-relative
|
||||
// path, not a volume.
|
||||
for i := 2; i < len(value); i++ {
|
||||
if value[i] != '\\' && value[i] != '/' {
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
return strings.ToUpper(value[:2]), true
|
||||
}
|
||||
|
||||
// Sets start values for disk I/O stats.
|
||||
func (a *Agent) initializeDiskIoStats(diskIoCounters map[string]disk.IOCountersStat) {
|
||||
a.fsNames = a.fsNames[:0]
|
||||
now := time.Now()
|
||||
// ZFS datasets have no /proc/diskstats entry, so they are excluded from
|
||||
// I/O tracking instead of warning about a missing device (#1541).
|
||||
var zfsMountpoints map[string]bool
|
||||
if a.storagePoolManager != nil {
|
||||
zfsMountpoints = a.storagePoolManager.ZfsMountpoints()
|
||||
}
|
||||
for device, stats := range a.fsStats {
|
||||
if zfsMountpoints[stats.Mountpoint] {
|
||||
continue
|
||||
}
|
||||
// skip if not in diskIoCounters
|
||||
d, exists := diskIoCounters[device]
|
||||
if !exists {
|
||||
@@ -542,9 +596,9 @@ func (a *Agent) initializeDiskIoStats(diskIoCounters map[string]disk.IOCountersS
|
||||
continue
|
||||
}
|
||||
// populate initial values
|
||||
stats.Time = now
|
||||
stats.TotalRead = d.ReadBytes
|
||||
stats.TotalWrite = d.WriteBytes
|
||||
a.setDiskBaseline(device, prevDiskFromCounter(d, now))
|
||||
// add to list of valid io device names
|
||||
a.fsNames = append(a.fsNames, device)
|
||||
}
|
||||
@@ -559,20 +613,31 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
|
||||
!a.lastDiskUsageUpdate.IsZero() &&
|
||||
time.Since(a.lastDiskUsageUpdate) < a.diskUsageCacheDuration
|
||||
|
||||
// ZFS dataset mountpoints use `zfs list` values because statfs(2) reports
|
||||
// dataset-level usage that excludes child datasets (#1541).
|
||||
var zfsUsage map[string]zfsDatasetUsage
|
||||
if a.storagePoolManager != nil {
|
||||
zfsUsage = a.storagePoolManager.DatasetUsage()
|
||||
}
|
||||
|
||||
// disk usage
|
||||
for _, stats := range a.fsStats {
|
||||
// Skip non-root filesystems if caching is active
|
||||
if cacheExtraFs && !stats.Root {
|
||||
continue
|
||||
}
|
||||
if d, err := disk.Usage(stats.Mountpoint); err == nil {
|
||||
stats.DiskTotal = utils.BytesToGigabytes(d.Total)
|
||||
stats.DiskUsed = utils.BytesToGigabytes(d.Used)
|
||||
if stats.Root {
|
||||
systemStats.DiskTotal = utils.BytesToGigabytes(d.Total)
|
||||
systemStats.DiskUsed = utils.BytesToGigabytes(d.Used)
|
||||
systemStats.DiskPct = utils.TwoDecimals(d.UsedPercent)
|
||||
var total, used uint64
|
||||
var usedPct float64
|
||||
if u, ok := zfsUsage[stats.Mountpoint]; ok {
|
||||
total = u.used + u.avail
|
||||
used = u.used
|
||||
if total > 0 {
|
||||
usedPct = float64(used) / float64(total) * 100
|
||||
}
|
||||
} else if d, err := disk.Usage(stats.Mountpoint); err == nil {
|
||||
total = d.Total
|
||||
used = d.Used
|
||||
usedPct = d.UsedPercent
|
||||
} else {
|
||||
// reset stats if error (likely unmounted)
|
||||
slog.Error("Error getting disk stats", "name", stats.Mountpoint, "err", err)
|
||||
@@ -580,6 +645,14 @@ func (a *Agent) updateDiskUsage(systemStats *system.Stats) {
|
||||
stats.DiskUsed = 0
|
||||
stats.TotalRead = 0
|
||||
stats.TotalWrite = 0
|
||||
continue
|
||||
}
|
||||
stats.DiskTotal = utils.BytesToGigabytes(total)
|
||||
stats.DiskUsed = utils.BytesToGigabytes(used)
|
||||
if stats.Root {
|
||||
systemStats.DiskTotal = stats.DiskTotal
|
||||
systemStats.DiskUsed = stats.DiskUsed
|
||||
systemStats.DiskPct = utils.TwoDecimals(usedPct)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -608,19 +681,9 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
// Previous snapshot for this interval and device
|
||||
prev, hasPrev := a.diskPrev[cacheTimeMs][name]
|
||||
if !hasPrev {
|
||||
// Seed from agent-level fsStats if present, else seed from current
|
||||
prev = prevDisk{
|
||||
readBytes: stats.TotalRead,
|
||||
writeBytes: stats.TotalWrite,
|
||||
readTime: d.ReadTime,
|
||||
writeTime: d.WriteTime,
|
||||
ioTime: d.IoTime,
|
||||
weightedIO: d.WeightedIO,
|
||||
readCount: d.ReadCount,
|
||||
writeCount: d.WriteCount,
|
||||
at: stats.Time,
|
||||
}
|
||||
if prev.at.IsZero() {
|
||||
// Seed from the latest counters of any interval, else seed from current
|
||||
prev, hasPrev = a.diskBaseline[name]
|
||||
if !hasPrev {
|
||||
prev = prevDiskFromCounter(d, now)
|
||||
}
|
||||
}
|
||||
@@ -655,29 +718,31 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
// This is the total number of milliseconds spent by all reads (as
|
||||
// measured from __make_request() to end_that_request_last()).
|
||||
// https://www.kernel.org/doc/Documentation/iostats.txt (fields 4, 8)
|
||||
diskReadTime := utils.TwoDecimals(float64(d.ReadTime-prev.readTime) / float64(msElapsed) * 100)
|
||||
diskWriteTime := utils.TwoDecimals(float64(d.WriteTime-prev.writeTime) / float64(msElapsed) * 100)
|
||||
deltaReadTime := ioTimeDelta(d.ReadTime, prev.readTime)
|
||||
deltaWriteTime := ioTimeDelta(d.WriteTime, prev.writeTime)
|
||||
diskReadTime := utils.TwoDecimals(float64(deltaReadTime) / float64(msElapsed) * 100)
|
||||
diskWriteTime := utils.TwoDecimals(float64(deltaWriteTime) / float64(msElapsed) * 100)
|
||||
|
||||
// I/O utilization %: fraction of wall time the device had any I/O in progress (0-100).
|
||||
diskIoUtilPct := utils.TwoDecimals(float64(d.IoTime-prev.ioTime) / float64(msElapsed) * 100)
|
||||
diskIoUtilPct := utils.TwoDecimals(float64(ioTimeDelta(d.IoTime, prev.ioTime)) / float64(msElapsed) * 100)
|
||||
|
||||
// Weighted I/O: queue-depth weighted I/O time, normalized to interval (can exceed 100%).
|
||||
// Linux kernel field 11: incremented by iops_in_progress × ms_since_last_update.
|
||||
// Used to display queue depth. Multipled by 100 to increase accuracy of digit truncation (divided by 100 in UI).
|
||||
diskWeightedIO := utils.TwoDecimals(float64(d.WeightedIO-prev.weightedIO) / float64(msElapsed) * 100)
|
||||
diskWeightedIO := utils.TwoDecimals(float64(ioTimeDelta(d.WeightedIO, prev.weightedIO)) / float64(msElapsed) * 100)
|
||||
|
||||
// r_await / w_await: average time per read/write operation in milliseconds.
|
||||
// Equivalent to r_await and w_await in iostat.
|
||||
var rAwait, wAwait float64
|
||||
if deltaReadCount := d.ReadCount - prev.readCount; deltaReadCount > 0 {
|
||||
rAwait = utils.TwoDecimals(float64(d.ReadTime-prev.readTime) / float64(deltaReadCount))
|
||||
rAwait = utils.TwoDecimals(float64(deltaReadTime) / float64(deltaReadCount))
|
||||
}
|
||||
if deltaWriteCount := d.WriteCount - prev.writeCount; deltaWriteCount > 0 {
|
||||
wAwait = utils.TwoDecimals(float64(d.WriteTime-prev.writeTime) / float64(deltaWriteCount))
|
||||
wAwait = utils.TwoDecimals(float64(deltaWriteTime) / float64(deltaWriteCount))
|
||||
}
|
||||
|
||||
// Update global fsStats baseline for cross-interval correctness
|
||||
stats.Time = now
|
||||
// Update the baseline that seeds new intervals
|
||||
a.setDiskBaseline(name, prevDiskFromCounter(d, now))
|
||||
stats.TotalRead = d.ReadBytes
|
||||
stats.TotalWrite = d.WriteBytes
|
||||
stats.DiskReadPs = readMbPerSecond
|
||||
@@ -696,6 +761,8 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
systemStats.DiskWritePs = stats.DiskWritePs
|
||||
systemStats.DiskIO[0] = diskIORead
|
||||
systemStats.DiskIO[1] = diskIOWrite
|
||||
systemStats.DiskIOTotal[0] = d.ReadBytes
|
||||
systemStats.DiskIOTotal[1] = d.WriteBytes
|
||||
systemStats.DiskIoStats[0] = diskReadTime
|
||||
systemStats.DiskIoStats[1] = diskWriteTime
|
||||
systemStats.DiskIoStats[2] = diskIoUtilPct
|
||||
@@ -707,6 +774,30 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
|
||||
}
|
||||
}
|
||||
|
||||
// setDiskBaseline stores the latest counters of a device. A cache interval
|
||||
// without its own snapshot measures its first sample from them.
|
||||
func (a *Agent) setDiskBaseline(name string, d prevDisk) {
|
||||
if a.diskBaseline == nil {
|
||||
a.diskBaseline = make(map[string]prevDisk)
|
||||
}
|
||||
a.diskBaseline[name] = d
|
||||
}
|
||||
|
||||
// ioTimeDelta returns the increase of a cumulative millisecond counter from
|
||||
// the disk I/O stats. Linux prints these fields of /proc/diskstats as 32-bit
|
||||
// unsigned ints, so they wrap to zero at 2^32. A busy disk reaches that in
|
||||
// days for the weighted I/O time. Other platforms report 64-bit counters,
|
||||
// so a lower value there is a reset.
|
||||
func ioTimeDelta(current, previous uint64) uint64 {
|
||||
if current >= previous {
|
||||
return current - previous
|
||||
}
|
||||
if runtime.GOOS == "linux" && previous <= math.MaxUint32 {
|
||||
return current + (math.MaxUint32 + 1 - previous)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// getRootMountPoint returns the appropriate root mount point for the system.
|
||||
// On Windows it returns the system drive (e.g. "C:").
|
||||
// For immutable systems like Fedora Silverblue, it returns /sysroot instead of /
|
||||
|
||||
125
agent/disk_io_linux_test.go
Normal file
125
agent/disk_io_linux_test.go
Normal file
@@ -0,0 +1,125 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/shirou/gopsutil/v4/disk"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// Linux prints four millisecond fields of /proc/diskstats as 32-bit unsigned ints:
|
||||
// read time, write time, io time and weighted io time. They wrap to zero at 2^32.
|
||||
func TestUpdateDiskIoTimeCounterWrap(t *testing.T) {
|
||||
const wrap = uint64(1) << 32
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
base uint64 // added to every previous time counter
|
||||
}{
|
||||
{"no wrap", 0},
|
||||
{"32-bit wrap", wrap - 1000},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
// Deltas over 60s: read 300ms / 10 ops, write 400ms / 20 ops,
|
||||
// io time 1200ms, weighted io 3000ms.
|
||||
prev := prevDisk{
|
||||
readBytes: 20000 * 512,
|
||||
writeBytes: 10000 * 512,
|
||||
readTime: tt.base + 900,
|
||||
writeTime: tt.base + 700,
|
||||
ioTime: tt.base + 400,
|
||||
weightedIO: tt.base,
|
||||
readCount: 1000,
|
||||
writeCount: 500,
|
||||
at: time.Now().Add(-60 * time.Second),
|
||||
}
|
||||
cur := func(v uint64) uint64 { return v % wrap }
|
||||
line := fmt.Sprintf(" 8 0 sda %d 0 %d %d %d 0 %d %d 0 %d %d\n",
|
||||
1010, 21200, cur(prev.readTime+300),
|
||||
520, 10400, cur(prev.writeTime+400),
|
||||
cur(prev.ioTime+1200), cur(prev.weightedIO+3000))
|
||||
|
||||
dir := t.TempDir()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "diskstats"), []byte(line), 0o644))
|
||||
t.Setenv("HOST_PROC", dir)
|
||||
t.Setenv("HOST_SYS", dir)
|
||||
t.Setenv("HOST_DEV", dir)
|
||||
t.Setenv("HOST_RUN", dir)
|
||||
|
||||
fs := &system.FsStats{Root: true}
|
||||
a := &Agent{
|
||||
fsNames: []string{"sda"},
|
||||
fsStats: map[string]*system.FsStats{"sda": fs},
|
||||
diskPrev: map[uint16]map[string]prevDisk{60000: {"sda": prev}},
|
||||
}
|
||||
var stats system.Stats
|
||||
a.updateDiskIo(60000, &stats)
|
||||
|
||||
// Same order as DiskIoStats in system.FsStats.
|
||||
want := [6]float64{0.5, 0.67, 2, 30, 20, 5}
|
||||
for i := range want {
|
||||
assert.InDelta(t, want[i], fs.DiskIoStats[i], 0.01, "DiskIoStats[%d]", i)
|
||||
assert.InDelta(t, want[i], stats.DiskIoStats[i], 0.01, "system DiskIoStats[%d]", i)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The first sample of a cache interval has no snapshot of its own. It must
|
||||
// measure the time counters from the same baseline as the byte counters.
|
||||
func TestUpdateDiskIoFirstSampleOfInterval(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("HOST_PROC", dir)
|
||||
t.Setenv("HOST_SYS", dir)
|
||||
t.Setenv("HOST_DEV", dir)
|
||||
t.Setenv("HOST_RUN", dir)
|
||||
writeDiskstats := func(line string) {
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "diskstats"), []byte(line), 0o644))
|
||||
}
|
||||
|
||||
writeDiskstats(" 8 0 sda 1000 0 20000 900 500 0 10000 700 0 400 0\n")
|
||||
counters, err := disk.IOCounters("sda")
|
||||
require.NoError(t, err)
|
||||
|
||||
fs := &system.FsStats{Root: true}
|
||||
a := &Agent{
|
||||
fsStats: map[string]*system.FsStats{"sda": fs},
|
||||
diskPrev: map[uint16]map[string]prevDisk{},
|
||||
}
|
||||
a.initializeDiskIoStats(counters)
|
||||
|
||||
// updateDiskIo skips samples less than 100ms apart.
|
||||
time.Sleep(150 * time.Millisecond)
|
||||
|
||||
// Deltas: read 300ms / 10 ops, write 400ms / 20 ops, io time 1200ms, weighted io 3000ms.
|
||||
writeDiskstats(" 8 0 sda 1010 0 21200 1200 520 0 10400 1100 0 1600 3000\n")
|
||||
var stats system.Stats
|
||||
a.updateDiskIo(60000, &stats)
|
||||
|
||||
require.NotZero(t, fs.DiskReadBytes, "bytes are measured from the baseline")
|
||||
for i := range 3 {
|
||||
assert.NotZero(t, fs.DiskIoStats[i], "DiskIoStats[%d]", i)
|
||||
}
|
||||
assert.InDelta(t, 30, fs.DiskIoStats[3], 0.01, "r_await")
|
||||
assert.InDelta(t, 20, fs.DiskIoStats[4], 0.01, "w_await")
|
||||
assert.NotZero(t, fs.DiskIoStats[5], "weighted io")
|
||||
|
||||
// A second interval starts from the latest counters, not from the ones at start.
|
||||
time.Sleep(150 * time.Millisecond)
|
||||
// Deltas: read 100ms / 10 ops, write 100ms / 20 ops.
|
||||
writeDiskstats(" 8 0 sda 1020 0 22400 1300 540 0 10800 1200 0 1800 3500\n")
|
||||
a.updateDiskIo(1000, &stats)
|
||||
|
||||
assert.InDelta(t, 10, fs.DiskIoStats[3], 0.01, "r_await")
|
||||
assert.InDelta(t, 5, fs.DiskIoStats[4], 0.01, "w_await")
|
||||
}
|
||||
@@ -3,7 +3,9 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"math"
|
||||
"os"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -78,14 +80,7 @@ func TestParseFilesystemEntry(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
fsEntry := strings.TrimSpace(tt.input)
|
||||
var fs, customName string
|
||||
if parts := strings.SplitN(fsEntry, "__", 2); len(parts) == 2 {
|
||||
fs = strings.TrimSpace(parts[0])
|
||||
customName = strings.TrimSpace(parts[1])
|
||||
} else {
|
||||
fs = fsEntry
|
||||
}
|
||||
fs, customName := parseFilesystemEntry(tt.input)
|
||||
|
||||
assert.Equal(t, tt.expectedFs, fs)
|
||||
assert.Equal(t, tt.expectedName, customName)
|
||||
@@ -287,8 +282,9 @@ func TestAddConfiguredRootFs(t *testing.T) {
|
||||
rootMountPoint: "/",
|
||||
partitions: []disk.PartitionStat{{Device: "/dev/ada0p2", Mountpoint: "/"}},
|
||||
ctx: fsRegistrationContext{
|
||||
filesystem: "/dev/ada0p2",
|
||||
isWindows: false,
|
||||
filesystem: "/dev/ada0p2",
|
||||
filesystemName: "root disk",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"ada0": {Name: "ada0", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
@@ -302,6 +298,7 @@ func TestAddConfiguredRootFs(t *testing.T) {
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/", stats.Mountpoint)
|
||||
assert.Equal(t, "root disk", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("adds root from io device when partition is missing", func(t *testing.T) {
|
||||
@@ -310,8 +307,9 @@ func TestAddConfiguredRootFs(t *testing.T) {
|
||||
agent: agent,
|
||||
rootMountPoint: "/sysroot",
|
||||
ctx: fsRegistrationContext{
|
||||
filesystem: "zroot",
|
||||
isWindows: false,
|
||||
filesystem: "zroot",
|
||||
filesystemName: "root pool",
|
||||
isWindows: false,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"nda0": {Name: "nda0", Label: "zroot", ReadBytes: 1000, WriteBytes: 1000},
|
||||
},
|
||||
@@ -325,6 +323,7 @@ func TestAddConfiguredRootFs(t *testing.T) {
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
assert.Equal(t, "/sysroot", stats.Mountpoint)
|
||||
assert.Equal(t, "root pool", stats.Name)
|
||||
})
|
||||
|
||||
t.Run("returns false when filesystem cannot be resolved", func(t *testing.T) {
|
||||
@@ -1033,8 +1032,10 @@ func TestInitializeDiskIoStatsResetsTrackedDevices(t *testing.T) {
|
||||
assert.Len(t, agent.fsNames, 2)
|
||||
assert.Equal(t, uint64(10), agent.fsStats["sda"].TotalRead)
|
||||
assert.Equal(t, uint64(20), agent.fsStats["sda"].TotalWrite)
|
||||
assert.False(t, agent.fsStats["sda"].Time.IsZero())
|
||||
assert.False(t, agent.fsStats["sdb"].Time.IsZero())
|
||||
assert.Equal(t, uint64(10), agent.diskBaseline["sda"].readBytes)
|
||||
assert.Equal(t, uint64(40), agent.diskBaseline["sdb"].writeBytes)
|
||||
assert.False(t, agent.diskBaseline["sda"].at.IsZero())
|
||||
assert.False(t, agent.diskBaseline["sdb"].at.IsZero())
|
||||
|
||||
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
|
||||
"sdb": {Name: "sdb", ReadBytes: 50, WriteBytes: 60},
|
||||
@@ -1044,3 +1045,114 @@ func TestInitializeDiskIoStatsResetsTrackedDevices(t *testing.T) {
|
||||
assert.Equal(t, uint64(50), agent.fsStats["sdb"].TotalRead)
|
||||
assert.Equal(t, uint64(60), agent.fsStats["sdb"].TotalWrite)
|
||||
}
|
||||
|
||||
func TestIoTimeDelta(t *testing.T) {
|
||||
assert.Equal(t, uint64(300), ioTimeDelta(1200, 900))
|
||||
|
||||
// A lower value is a 32-bit wrap only on Linux. Other platforms
|
||||
// report 64-bit counters, so there it is a reset.
|
||||
var want uint64
|
||||
if runtime.GOOS == "linux" {
|
||||
want = 1200
|
||||
}
|
||||
assert.Equal(t, want, ioTimeDelta(200, math.MaxUint32+1-1000))
|
||||
|
||||
assert.Equal(t, uint64(0), ioTimeDelta(200, math.MaxUint32+1000))
|
||||
}
|
||||
|
||||
func TestNormalizeDeviceName(t *testing.T) {
|
||||
// A Windows volume name is not a path element, so every spelling of the
|
||||
// same drive has to normalize to the same key. filepath.Base cannot do
|
||||
// this: on Windows it strips the "C:" specifier and returns "\", which
|
||||
// collapses every drive letter onto one key (#2417).
|
||||
for _, spelling := range []string{"C:", `C:\`, "C:/", `C:\\`} {
|
||||
assert.Equal(t, "C:", normalizeDeviceName(spelling), "spelling %q", spelling)
|
||||
}
|
||||
// Drive letters are case-insensitive, so the letter is uppercased.
|
||||
assert.Equal(t, "D:", normalizeDeviceName("d:"))
|
||||
assert.Equal(t, "C:", normalizeDeviceName(" c: "))
|
||||
assert.Equal(t, "C:", normalizeDeviceName(`c:\`))
|
||||
|
||||
// Non-volume inputs keep using filepath.Base.
|
||||
assert.Equal(t, "sda1", normalizeDeviceName("/dev/sda1"))
|
||||
assert.Equal(t, "sda1", normalizeDeviceName("/dev/sda1/"))
|
||||
assert.Equal(t, "nvme0n1p2", normalizeDeviceName(" /dev/nvme0n1p2 "))
|
||||
assert.Equal(t, "", normalizeDeviceName("."))
|
||||
assert.Equal(t, "", normalizeDeviceName(" "))
|
||||
|
||||
// A drive-relative path is a path, not a volume.
|
||||
assert.Equal(t, `C:data`, normalizeDeviceName(`C:data`))
|
||||
}
|
||||
|
||||
func TestFindIoDeviceWindowsVolumeNames(t *testing.T) {
|
||||
// Every drive normalizes to a distinct key, so the root drive resolves
|
||||
// exactly instead of to whichever counter the map yielded first (#2417).
|
||||
ioCounters := map[string]disk.IOCountersStat{
|
||||
"C:": {Name: "C:", ReadBytes: 10, WriteBytes: 10},
|
||||
"D:": {Name: "D:", ReadBytes: 20, WriteBytes: 20},
|
||||
"P:": {Name: "P:", ReadBytes: 30, WriteBytes: 30},
|
||||
}
|
||||
|
||||
for i := 0; i < 32; i++ {
|
||||
device, ok := findIoDevice("C:", ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "C:", device)
|
||||
}
|
||||
|
||||
// The drive may arrive with a trailing separator, as a mount point does.
|
||||
device, ok := findIoDevice(`C:\`, ioCounters)
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, "C:", device)
|
||||
}
|
||||
|
||||
func TestAddPartitionRootFsWindowsDrive(t *testing.T) {
|
||||
agent := &Agent{fsStats: make(map[string]*system.FsStats)}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: true,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"C:": {Name: "C:"},
|
||||
"D:": {Name: "D:"},
|
||||
"P:": {Name: "P:"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addPartitionRootFs("C:", `C:\`)
|
||||
|
||||
assert.True(t, ok)
|
||||
assert.Len(t, agent.fsStats, 1)
|
||||
stats, exists := agent.fsStats["C:"]
|
||||
assert.True(t, exists)
|
||||
assert.True(t, stats.Root)
|
||||
}
|
||||
|
||||
func TestAddPartitionRootFsKeyAlreadyRegistered(t *testing.T) {
|
||||
// The root drive is also listed in EXTRA_FILESYSTEMS, so its key is taken
|
||||
// before the root fallback runs. The existing entry must be promoted to root
|
||||
// rather than falling back to the most active device, which here is D:.
|
||||
agent := &Agent{fsStats: map[string]*system.FsStats{
|
||||
"C:": {Mountpoint: `C:\`, Name: "System"},
|
||||
"D:": {Mountpoint: `D:\`},
|
||||
}}
|
||||
discovery := diskDiscovery{
|
||||
agent: agent,
|
||||
rootMountPoint: `C:\`,
|
||||
ctx: fsRegistrationContext{
|
||||
isWindows: true,
|
||||
diskIoCounters: map[string]disk.IOCountersStat{
|
||||
"C:": {Name: "C:", ReadBytes: 10},
|
||||
"D:": {Name: "D:", ReadBytes: 100},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
ok := discovery.addPartitionRootFs("C:", `C:\`)
|
||||
assert.True(t, ok)
|
||||
assert.Len(t, agent.fsStats, 2)
|
||||
assert.True(t, agent.fsStats["C:"].Root)
|
||||
assert.Equal(t, `C:\`, agent.fsStats["C:"].Mountpoint)
|
||||
assert.Equal(t, "System", agent.fsStats["C:"].Name)
|
||||
assert.False(t, agent.fsStats["D:"].Root)
|
||||
}
|
||||
|
||||
110
agent/disk_zfs_test.go
Normal file
110
agent/disk_zfs_test.go
Normal file
@@ -0,0 +1,110 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/shirou/gopsutil/v4/disk"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// TestUpdateDiskUsageZfsMountpoint verifies that a filesystem whose mountpoint
|
||||
// is a ZFS dataset reports `zfs list` usage (which includes child datasets)
|
||||
// instead of the dataset-scoped statfs values (#1541).
|
||||
func TestUpdateDiskUsageZfsMountpoint(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
|
||||
}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"tank": {Root: false, Mountpoint: "/tank"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
fs := agent.fsStats["tank"]
|
||||
require.NotNil(t, fs)
|
||||
assert.Equal(t, 22350.81, fs.DiskTotal) // (used + avail) in GiB
|
||||
assert.Equal(t, 11175.87, fs.DiskUsed)
|
||||
// Non-root filesystems do not populate system-level stats.
|
||||
assert.Equal(t, float64(0), stats.DiskTotal)
|
||||
}
|
||||
|
||||
// TestUpdateDiskUsageZfsRootPopulatesSystemStats verifies the root disk values
|
||||
// are derived from ZFS usage when the root mountpoint is a ZFS dataset.
|
||||
func TestUpdateDiskUsageZfsRootPopulatesSystemStats(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "rpool/ROOT/pve-1", Used: 900000000000, Avail: 300000000000, Mountpoint: "/"},
|
||||
}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"rpool/ROOT/pve-1": {Root: true, Mountpoint: "/"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
assert.Equal(t, 1117.59, agent.fsStats["rpool/ROOT/pve-1"].DiskTotal)
|
||||
assert.Equal(t, 838.19, agent.fsStats["rpool/ROOT/pve-1"].DiskUsed)
|
||||
assert.Equal(t, 75.0, stats.DiskPct)
|
||||
assert.Equal(t, 1117.59, stats.DiskTotal)
|
||||
assert.Equal(t, 838.19, stats.DiskUsed)
|
||||
}
|
||||
|
||||
// TestUpdateDiskUsageWithoutZfsManager falls back to statfs when no manager is
|
||||
// present (e.g. tests constructing bare Agent values).
|
||||
func TestUpdateDiskUsageWithoutZfsManager(t *testing.T) {
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"root": {Root: true, Mountpoint: "/"},
|
||||
},
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
agent.updateDiskUsage(&stats)
|
||||
|
||||
assert.True(t, agent.fsStats["root"].DiskTotal > 0, "root usage should come from statfs")
|
||||
assert.True(t, stats.DiskTotal > 0)
|
||||
}
|
||||
|
||||
// TestInitializeDiskIoStatsSkipsZfsMountpoints verifies ZFS filesystems are
|
||||
// excluded from diskstats I/O tracking instead of warning about a missing device.
|
||||
func TestInitializeDiskIoStatsSkipsZfsMountpoints(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank", Mountpoint: "/tank"}}, nil
|
||||
}
|
||||
agent := &Agent{
|
||||
fsStats: map[string]*system.FsStats{
|
||||
"tank": {Root: false, Mountpoint: "/tank"},
|
||||
"sda1": {Root: false, Mountpoint: "/mnt/data"},
|
||||
},
|
||||
storagePoolManager: zm,
|
||||
diskPrev: make(map[uint16]map[string]prevDisk),
|
||||
}
|
||||
|
||||
agent.initializeDiskIoStats(map[string]disk.IOCountersStat{
|
||||
"sda1": {Name: "sda1", ReadBytes: 100, WriteBytes: 100},
|
||||
})
|
||||
|
||||
assert.Equal(t, []string{"sda1"}, agent.fsNames)
|
||||
assert.Equal(t, uint64(100), agent.fsStats["sda1"].TotalRead)
|
||||
// ZFS entry is present but untouched by diskstats initialization.
|
||||
assert.Equal(t, uint64(0), agent.fsStats["tank"].TotalRead)
|
||||
}
|
||||
@@ -65,11 +65,15 @@ type dockerManager struct {
|
||||
dockerVersionChecked bool // Whether a version probe has completed successfully
|
||||
isWindows bool // Whether the Docker Engine API is running on Windows
|
||||
buf *bytes.Buffer // Buffer to store and read response bodies
|
||||
decoder *json.Decoder // Reusable JSON decoder that reads from buf
|
||||
apiStats *container.ApiStats // Reusable API stats object
|
||||
excludeContainers []string // Patterns to exclude containers by name
|
||||
usingPodman bool // Whether the Docker Engine API is running on Podman
|
||||
|
||||
registryClient *http.Client // Client for registry requests; nil uses a client with a 10-second timeout
|
||||
imageUpdatesDisabled bool // Whether image update checks are disabled by configuration
|
||||
imageUpdatesMutex sync.RWMutex // Protects imageUpdates, its entries, and imageUpdatesRunning
|
||||
imageUpdates map[string]*imageUpdateStatus // Shared update status keyed by normalized image reference
|
||||
imageUpdatesRunning bool // Whether a background image-update batch is in progress
|
||||
|
||||
// Cache-time-aware tracking for CPU stats (similar to cpu.go)
|
||||
// Maps cache time intervals to container-specific CPU usage tracking
|
||||
lastCpuContainer map[uint16]map[string]uint64 // cacheTimeMs -> containerId -> last cpu container usage
|
||||
@@ -162,6 +166,9 @@ func (dm *dockerManager) getDockerStats(cacheTimeMs uint16) ([]*container.Stats,
|
||||
clear(dm.validIds)
|
||||
}
|
||||
|
||||
// Only schedule auxiliary work here; metrics never wait for image discovery.
|
||||
dm.refreshImageUpdates(dm.apiContainerList, time.Now())
|
||||
|
||||
var failedContainers []*container.ApiInfo
|
||||
|
||||
for _, ctr := range dm.apiContainerList {
|
||||
@@ -374,16 +381,26 @@ func convertContainerPortsToString(ctr *container.ApiInfo) string {
|
||||
return ""
|
||||
}
|
||||
sort.Slice(ctr.Ports, func(i, j int) bool {
|
||||
return ctr.Ports[i].PublicPort < ctr.Ports[j].PublicPort
|
||||
if ctr.Ports[i].PublicPort != ctr.Ports[j].PublicPort {
|
||||
return ctr.Ports[i].PublicPort < ctr.Ports[j].PublicPort
|
||||
}
|
||||
return ctr.Ports[i].IP < ctr.Ports[j].IP
|
||||
})
|
||||
var builder strings.Builder
|
||||
seenPorts := make(map[uint16]struct{})
|
||||
seen := make(map[string]struct{})
|
||||
for _, p := range ctr.Ports {
|
||||
_, ok := seenPorts[p.PublicPort]
|
||||
if p.PublicPort == 0 || ok {
|
||||
if p.PublicPort == 0 {
|
||||
continue
|
||||
}
|
||||
seenPorts[p.PublicPort] = struct{}{}
|
||||
keyIP := p.IP
|
||||
if keyIP == "0.0.0.0" || keyIP == "::" {
|
||||
keyIP = ""
|
||||
}
|
||||
key := keyIP + ":" + strconv.Itoa(int(p.PublicPort))
|
||||
if _, ok := seen[key]; ok {
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
if builder.Len() > 0 {
|
||||
builder.WriteString(", ")
|
||||
}
|
||||
@@ -497,6 +514,17 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
}
|
||||
}
|
||||
|
||||
// Read and decode the response before locking shared stats to avoid blocking
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return fmt.Errorf("container stats request failed: %s", resp.Status)
|
||||
}
|
||||
res := &container.ApiStats{}
|
||||
if err := json.NewDecoder(resp.Body).Decode(res); err != nil {
|
||||
return err
|
||||
}
|
||||
updateAvailable := dm.cachedImageUpdate(ctr.Image)
|
||||
|
||||
dm.containerStatsMutex.Lock()
|
||||
defer dm.containerStatsMutex.Unlock()
|
||||
|
||||
@@ -511,6 +539,9 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
stats.Status = statusText
|
||||
stats.Health = health
|
||||
|
||||
stats.Image = ctr.Image
|
||||
stats.UpdateAvailable = updateAvailable
|
||||
|
||||
if len(ctr.Ports) > 0 {
|
||||
stats.Ports = convertContainerPortsToString(ctr)
|
||||
}
|
||||
@@ -523,23 +554,24 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
|
||||
stats.NetworkSent = 0
|
||||
stats.NetworkRecv = 0
|
||||
|
||||
res := dm.apiStats
|
||||
res.Networks = nil
|
||||
if err := dm.decode(resp, res); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Initialize CPU tracking for this cache time interval
|
||||
dm.initializeCpuTracking(cacheTimeMs)
|
||||
|
||||
// Get previous CPU values
|
||||
prevCpuContainer, prevCpuSystem := dm.getCpuPreviousValues(cacheTimeMs, ctr.IdShort)
|
||||
|
||||
// Calculate CPU percentage based on platform
|
||||
// Calculate CPU percentage based on platform.
|
||||
// Podman reports system_cpu_usage from cgroup cpu.stat (not /proc/stat), so it reflects
|
||||
// only cgroup-tracked activity rather than total host capacity. Use a time-based method
|
||||
// instead so the result is comparable to host CPU utilization. See:
|
||||
// https://github.com/henrygd/beszel/issues/2049
|
||||
var cpuPct float64
|
||||
if dm.isWindows {
|
||||
prevRead := dm.lastCpuReadTime[cacheTimeMs][ctr.IdShort]
|
||||
cpuPct = res.CalculateCpuPercentWindows(prevCpuContainer, prevRead)
|
||||
} else if dm.usingPodman && res.CPUStats.OnlineCPUs > 0 {
|
||||
prevRead := dm.lastCpuReadTime[cacheTimeMs][ctr.IdShort]
|
||||
cpuPct = res.CalculateCpuPercentPodman(prevCpuContainer, prevRead)
|
||||
} else {
|
||||
cpuPct = res.CalculateCpuPercentLinux(prevCpuContainer, prevCpuSystem)
|
||||
}
|
||||
@@ -657,6 +689,8 @@ func newDockerManager(agent *Agent) *dockerManager {
|
||||
userAgent: "Docker-Client/",
|
||||
}
|
||||
|
||||
dockerImageCheck, _ := utils.GetEnv("DOCKER_IMAGE_CHECK")
|
||||
|
||||
// Read container exclusion patterns from environment variable
|
||||
var excludeContainers []string
|
||||
if excludeStr, set := utils.GetEnv("EXCLUDE_CONTAINERS"); set && excludeStr != "" {
|
||||
@@ -676,11 +710,11 @@ func newDockerManager(agent *Agent) *dockerManager {
|
||||
Timeout: timeout,
|
||||
Transport: userAgentTransport,
|
||||
},
|
||||
containerStatsMap: make(map[string]*container.Stats),
|
||||
sem: make(chan struct{}, 5),
|
||||
apiContainerList: []*container.ApiInfo{},
|
||||
apiStats: &container.ApiStats{},
|
||||
excludeContainers: excludeContainers,
|
||||
containerStatsMap: make(map[string]*container.Stats),
|
||||
sem: make(chan struct{}, 5),
|
||||
apiContainerList: []*container.ApiInfo{},
|
||||
excludeContainers: excludeContainers,
|
||||
imageUpdatesDisabled: dockerImageCheck == "false",
|
||||
|
||||
// Initialize cache-time-aware tracking structures
|
||||
lastCpuContainer: make(map[uint16]map[string]uint64),
|
||||
@@ -747,20 +781,18 @@ func (dm *dockerManager) applyDockerVersionInfo(serverHeader string, versionInfo
|
||||
}
|
||||
}
|
||||
|
||||
// Decodes Docker API JSON response using a reusable buffer and decoder. Not thread safe.
|
||||
// Decodes a Docker API JSON response using a reusable buffer. Not thread safe.
|
||||
func (dm *dockerManager) decode(resp *http.Response, d any) error {
|
||||
if dm.buf == nil {
|
||||
// initialize buffer with 256kb starting size
|
||||
dm.buf = bytes.NewBuffer(make([]byte, 0, 1024*256))
|
||||
dm.decoder = json.NewDecoder(dm.buf)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
defer dm.buf.Reset()
|
||||
_, err := dm.buf.ReadFrom(resp.Body)
|
||||
if err != nil {
|
||||
if _, err := dm.buf.ReadFrom(resp.Body); err != nil {
|
||||
return err
|
||||
}
|
||||
return dm.decoder.Decode(d)
|
||||
return json.Unmarshal(dm.buf.Bytes(), d)
|
||||
}
|
||||
|
||||
// Test docker / podman sockets and return if one exists
|
||||
|
||||
108
agent/docker_image_updates.go
Normal file
108
agent/docker_image_updates.go
Normal file
@@ -0,0 +1,108 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/distribution/reference"
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
)
|
||||
|
||||
const imageUpdateInterval = time.Hour
|
||||
|
||||
type imageUpdateStatus struct {
|
||||
available bool
|
||||
checkedAt time.Time
|
||||
}
|
||||
|
||||
func normalizedImageReference(image string) string {
|
||||
named, err := reference.ParseNormalizedNamed(image)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
// Digest-pinned references cannot move to a new version.
|
||||
if _, pinned := named.(reference.Digested); pinned {
|
||||
return ""
|
||||
}
|
||||
return reference.TagNameOnly(named).String()
|
||||
}
|
||||
|
||||
// refreshImageUpdates starts at most one background batch. Neither its network
|
||||
// work nor its completion is part of the container metrics wait group.
|
||||
func (dm *dockerManager) refreshImageUpdates(containers []*container.ApiInfo, now time.Time) {
|
||||
if dm.imageUpdatesDisabled {
|
||||
return
|
||||
}
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
defer dm.imageUpdatesMutex.Unlock()
|
||||
if dm.imageUpdatesRunning {
|
||||
return
|
||||
}
|
||||
if dm.imageUpdates == nil {
|
||||
dm.imageUpdates = make(map[string]*imageUpdateStatus)
|
||||
}
|
||||
active := make(map[string]struct{}, len(containers))
|
||||
pending := make(map[string]*imageUpdateStatus)
|
||||
for _, ctr := range containers {
|
||||
if len(ctr.Names) > 0 && dm.shouldExcludeContainer(ctr.Names[0][1:]) {
|
||||
continue
|
||||
}
|
||||
key := normalizedImageReference(ctr.Image)
|
||||
if key == "" {
|
||||
continue
|
||||
}
|
||||
active[key] = struct{}{}
|
||||
entry := dm.imageUpdates[key]
|
||||
if entry == nil {
|
||||
entry = &imageUpdateStatus{}
|
||||
dm.imageUpdates[key] = entry
|
||||
}
|
||||
if entry.checkedAt.IsZero() || now.Sub(entry.checkedAt) >= imageUpdateInterval {
|
||||
pending[key] = entry
|
||||
}
|
||||
}
|
||||
for key := range dm.imageUpdates {
|
||||
if _, ok := active[key]; !ok {
|
||||
delete(dm.imageUpdates, key)
|
||||
}
|
||||
}
|
||||
if len(pending) == 0 {
|
||||
return
|
||||
}
|
||||
dm.imageUpdatesRunning = true
|
||||
go func() {
|
||||
// Limit auxiliary requests even on hosts running many different images.
|
||||
sem := make(chan struct{}, 2)
|
||||
var wg sync.WaitGroup
|
||||
for key, entry := range pending {
|
||||
sem <- struct{}{}
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
available, err := dm.checkImageUpdate(key)
|
||||
if err != nil {
|
||||
available = false
|
||||
slog.Debug("Image update check failed", "image", key, "err", err)
|
||||
}
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
entry.available = available
|
||||
entry.checkedAt = time.Now()
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdatesRunning = false
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
func (dm *dockerManager) cachedImageUpdate(image string) bool {
|
||||
key := normalizedImageReference(image)
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
defer dm.imageUpdatesMutex.RUnlock()
|
||||
entry := dm.imageUpdates[key]
|
||||
return entry != nil && entry.available
|
||||
}
|
||||
248
agent/docker_image_updates_test.go
Normal file
248
agent/docker_image_updates_test.go
Normal file
@@ -0,0 +1,248 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func waitForImageUpdates(t *testing.T, dm *dockerManager) {
|
||||
t.Helper()
|
||||
require.Eventually(t, func() bool {
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
defer dm.imageUpdatesMutex.RUnlock()
|
||||
return !dm.imageUpdatesRunning
|
||||
}, time.Second*3, time.Millisecond)
|
||||
}
|
||||
|
||||
func TestDisableDockerImageUpdateCheck(t *testing.T) {
|
||||
t.Setenv("BESZEL_AGENT_DOCKER_IMAGE_CHECK", "false")
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path == "/version" {
|
||||
fmt.Fprint(w, `{"Version":"25.0.0"}`)
|
||||
return
|
||||
}
|
||||
http.NotFound(w, r)
|
||||
}))
|
||||
defer server.Close()
|
||||
t.Setenv("BESZEL_AGENT_DOCKER_HOST", server.URL)
|
||||
|
||||
dm := newDockerManager(nil)
|
||||
require.True(t, dm.imageUpdatesDisabled)
|
||||
dm.registryClient = &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
|
||||
t.Fatal("disabled image update check made a registry request")
|
||||
return nil, nil
|
||||
})}
|
||||
dm.refreshImageUpdates([]*container.ApiInfo{{Image: "nginx", Names: []string{"/nginx"}}}, time.Now())
|
||||
require.False(t, dm.imageUpdatesRunning)
|
||||
require.Nil(t, dm.imageUpdates)
|
||||
}
|
||||
|
||||
func TestImageUpdateCacheAndStats(t *testing.T) {
|
||||
local := "sha256:" + strings.Repeat("a", 64)
|
||||
remote := "sha256:" + strings.Repeat("b", 64)
|
||||
var inspections, lookups atomic.Int32
|
||||
var fail atomic.Bool
|
||||
var upToDate atomic.Bool
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch {
|
||||
case strings.HasPrefix(r.URL.Path, "/images/"):
|
||||
inspections.Add(1)
|
||||
fmt.Fprintf(w, `{"RepoDigests":["docker.io/library/nginx@%s"]}`, local)
|
||||
case r.URL.Path == "/containers/json":
|
||||
fmt.Fprint(w, `[{"Id":"aaaaaaaaaaaa","Names":["/one"],"Image":"nginx","Status":"Up 2 hours"},{"Id":"bbbbbbbbbbbb","Names":["/two"],"Image":"docker.io/library/nginx:latest","Status":"Up 2 hours"}]`)
|
||||
case strings.Contains(r.URL.Path, "/stats"):
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576},"cpu_stats":{},"networks":{}}`)
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
dm.dockerVersionChecked = true
|
||||
dm.registryClient = &http.Client{Timeout: time.Second, Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
|
||||
if fail.Load() {
|
||||
return nil, fmt.Errorf("registry unavailable")
|
||||
}
|
||||
response := &http.Response{StatusCode: 200, Header: make(http.Header), Body: io.NopCloser(strings.NewReader(`{"token":"test"}`))}
|
||||
if r.Method == http.MethodHead {
|
||||
lookups.Add(1)
|
||||
digest := remote
|
||||
if upToDate.Load() {
|
||||
digest = local
|
||||
}
|
||||
response.Header.Set("Docker-Content-Digest", digest)
|
||||
}
|
||||
return response, nil
|
||||
})}
|
||||
stats, err := dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, stats, 2)
|
||||
waitForImageUpdates(t, dm)
|
||||
require.EqualValues(t, 1, lookups.Load())
|
||||
require.EqualValues(t, 1, inspections.Load())
|
||||
stats, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
for _, stat := range stats {
|
||||
require.True(t, stat.UpdateAvailable)
|
||||
if stat.Id == "aaaaaaaaaaaa" {
|
||||
require.Equal(t, "nginx", stat.Image)
|
||||
} else {
|
||||
require.Equal(t, "docker.io/library/nginx:latest", stat.Image)
|
||||
}
|
||||
}
|
||||
require.EqualValues(t, 1, lookups.Load())
|
||||
|
||||
expire := func() {
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdates["docker.io/library/nginx:latest"].checkedAt = time.Now().Add(-imageUpdateInterval)
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
}
|
||||
upToDate.Store(true)
|
||||
expire()
|
||||
_, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
waitForImageUpdates(t, dm)
|
||||
require.EqualValues(t, 2, lookups.Load())
|
||||
require.False(t, dm.cachedImageUpdate("nginx:latest"))
|
||||
|
||||
// An expired positive result is cleared on failure, and the failure itself
|
||||
// is cached so realtime stats do not retry a broken registry every second.
|
||||
dm.imageUpdatesMutex.Lock()
|
||||
dm.imageUpdates["docker.io/library/nginx:latest"].available = true
|
||||
dm.imageUpdatesMutex.Unlock()
|
||||
fail.Store(true)
|
||||
expire()
|
||||
_, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
waitForImageUpdates(t, dm)
|
||||
failedInspections := inspections.Load()
|
||||
stats, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, stats, 2)
|
||||
require.Equal(t, failedInspections, inspections.Load())
|
||||
for _, stat := range stats {
|
||||
require.False(t, stat.UpdateAvailable)
|
||||
require.Equal(t, 1.0, stat.Mem)
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageDiscoveryDoesNotBlockStats(t *testing.T) {
|
||||
started := make(chan struct{}, 1)
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
fmt.Fprintf(w, `{"RepoDigests":["example.com/app@sha256:%s"]}`, strings.Repeat("a", 64))
|
||||
} else {
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
defer func() { close(release); waitForImageUpdates(t, dm) }()
|
||||
dm.registryClient = &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
|
||||
started <- struct{}{}
|
||||
<-release
|
||||
return nil, fmt.Errorf("timeout")
|
||||
})}
|
||||
ctr := &container.ApiInfo{IdShort: "aaaaaaaaaaaa", Image: "example.com/app", Names: []string{"/one"}}
|
||||
dm.refreshImageUpdates([]*container.ApiInfo{ctr}, time.Now())
|
||||
select {
|
||||
case <-started:
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("check did not start")
|
||||
}
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- dm.updateContainerStats(ctr, defaultCacheTimeMs) }()
|
||||
select {
|
||||
case err := <-done:
|
||||
require.NoError(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("registry blocked stats")
|
||||
}
|
||||
dm.imageUpdatesMutex.RLock()
|
||||
require.True(t, dm.imageUpdatesRunning)
|
||||
dm.imageUpdatesMutex.RUnlock()
|
||||
}
|
||||
|
||||
func TestNormalizeImageUpdateReferences(t *testing.T) {
|
||||
require.Equal(t, normalizedImageReference("nginx"), normalizedImageReference("docker.io/library/nginx:latest"))
|
||||
require.Empty(t, normalizedImageReference("bad reference"))
|
||||
require.Empty(t, normalizedImageReference("nginx@sha256:"+strings.Repeat("a", 64)))
|
||||
}
|
||||
|
||||
// A stats request can return headers promptly and then stall while reading its
|
||||
// body. The stats-map mutex must remain available during that read.
|
||||
func TestStatsResponseBodyDoesNotHoldStatsLock(t *testing.T) {
|
||||
started := make(chan struct{})
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
w.(http.Flusher).Flush()
|
||||
close(started)
|
||||
<-release
|
||||
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
|
||||
}))
|
||||
defer server.Close()
|
||||
dm := newDockerManagerForVersionTest(server)
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- dm.updateContainerStats(&container.ApiInfo{IdShort: "aaaaaaaaaaaa", Names: []string{"/one"}, Image: "nginx"}, defaultCacheTimeMs)
|
||||
}()
|
||||
<-started
|
||||
locked := make(chan struct{})
|
||||
go func() { dm.containerStatsMutex.Lock(); dm.containerStatsMutex.Unlock(); close(locked) }()
|
||||
select {
|
||||
case <-locked:
|
||||
case <-time.After(time.Second):
|
||||
close(release)
|
||||
<-done
|
||||
t.Fatal("Docker response body held the stats mutex")
|
||||
}
|
||||
close(release)
|
||||
require.NoError(t, <-done)
|
||||
}
|
||||
|
||||
func TestImageUpdateStatsEncoding(t *testing.T) {
|
||||
original := container.Stats{Image: "nginx:latest", UpdateAvailable: true}
|
||||
encoded, err := cbor.Marshal(original)
|
||||
require.NoError(t, err)
|
||||
var fields map[int]any
|
||||
require.NoError(t, cbor.Unmarshal(encoded, &fields))
|
||||
require.Equal(t, true, fields[11])
|
||||
require.Equal(t, "nginx:latest", fields[8])
|
||||
var decoded container.Stats
|
||||
require.NoError(t, cbor.Unmarshal(encoded, &decoded))
|
||||
require.True(t, decoded.UpdateAvailable)
|
||||
require.Equal(t, original.Image, decoded.Image)
|
||||
encoded, err = json.Marshal(original)
|
||||
require.NoError(t, err)
|
||||
require.Contains(t, string(encoded), `"u":true`)
|
||||
}
|
||||
|
||||
func TestImageUpdateCacheExpiryBoundaryAndPruning(t *testing.T) {
|
||||
now := time.Now()
|
||||
key := normalizedImageReference("nginx")
|
||||
dm := &dockerManager{imageUpdates: map[string]*imageUpdateStatus{
|
||||
key: {available: true, checkedAt: now},
|
||||
"unused.example/image:latest": {checkedAt: now},
|
||||
}}
|
||||
dm.refreshImageUpdates([]*container.ApiInfo{{Image: "nginx"}}, now.Add(imageUpdateInterval-time.Nanosecond))
|
||||
require.False(t, dm.imageUpdatesRunning)
|
||||
require.Len(t, dm.imageUpdates, 1)
|
||||
require.True(t, dm.cachedImageUpdate("nginx:latest"))
|
||||
dm.refreshImageUpdates(nil, now)
|
||||
require.Empty(t, dm.imageUpdates)
|
||||
}
|
||||
224
agent/docker_registry.go
Normal file
224
agent/docker_registry.go
Normal file
@@ -0,0 +1,224 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
_ "crypto/sha256"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"slices"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/distribution/reference"
|
||||
"github.com/opencontainers/go-digest"
|
||||
)
|
||||
|
||||
const imageRegistryTimeout = 10 * time.Second
|
||||
|
||||
const imageManifestAccept = "application/vnd.docker.distribution.manifest.list.v2+json, " +
|
||||
"application/vnd.docker.distribution.manifest.v2+json, " +
|
||||
"application/vnd.oci.image.manifest.v1+json, " +
|
||||
"application/vnd.oci.image.index.v1+json"
|
||||
|
||||
// checkImageUpdate compares the digest recorded by Docker for image with the
|
||||
// digest currently advertised by its registry. A digest-pinned reference is
|
||||
// immutable and therefore never has an update available.
|
||||
func (dm *dockerManager) checkImageUpdate(image string) (bool, error) {
|
||||
named, err := reference.ParseNormalizedNamed(image)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("parse image reference %q: %w", image, err)
|
||||
}
|
||||
if _, pinned := named.(reference.Digested); pinned {
|
||||
return false, nil
|
||||
}
|
||||
named = reference.TagNameOnly(named)
|
||||
|
||||
registry := reference.Domain(named)
|
||||
repository := reference.Path(named)
|
||||
tag := named.(reference.Tagged).Tag()
|
||||
|
||||
localDigests, err := dm.inspectImageDigests(image, registry, repository)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
remoteDigest, err := dm.registryImageDigest(registry, repository, tag)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
return !slices.Contains(localDigests, remoteDigest), nil
|
||||
}
|
||||
|
||||
// inspectImageDigests reads Docker's image metadata without using dm.decode.
|
||||
// The checker runs in the image-discovery goroutine, so it must not hold any
|
||||
// of the container statistics locks while waiting on the Docker API.
|
||||
func (dm *dockerManager) inspectImageDigests(image, registry, repository string) ([]string, error) {
|
||||
if dm.client == nil {
|
||||
return nil, fmt.Errorf("inspect image %q: Docker client is unavailable", image)
|
||||
}
|
||||
|
||||
endpoint := "http://localhost/images/" + url.PathEscape(image) + "/json"
|
||||
resp, err := dm.client.Get(endpoint)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("inspect image %q: %w", image, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, fmt.Errorf("inspect image %q failed: %s", image, responseStatus(resp))
|
||||
}
|
||||
|
||||
var inspect struct {
|
||||
RepoDigests []string `json:"RepoDigests"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&inspect); err != nil {
|
||||
return nil, fmt.Errorf("decode image inspect %q: %w", image, err)
|
||||
}
|
||||
if len(inspect.RepoDigests) == 0 {
|
||||
return nil, fmt.Errorf("inspect image %q returned no repository digests", image)
|
||||
}
|
||||
|
||||
localDigests := matchingRepositoryDigests(inspect.RepoDigests, registry, repository)
|
||||
if len(localDigests) == 0 {
|
||||
return nil, fmt.Errorf("inspect image %q returned no valid digest for %s/%s", image, registry, repository)
|
||||
}
|
||||
return localDigests, nil
|
||||
}
|
||||
|
||||
// matchingRepositoryDigests returns all valid digests belonging to the requested
|
||||
// repository. Container engines can return both index and platform manifest digests for one
|
||||
// local image, in either order.
|
||||
func matchingRepositoryDigests(repoDigests []string, registry, repository string) []string {
|
||||
var digests []string
|
||||
for _, repoDigest := range repoDigests {
|
||||
repoDigest = strings.TrimSpace(repoDigest)
|
||||
at := strings.LastIndexByte(repoDigest, '@')
|
||||
if at <= 0 || at == len(repoDigest)-1 || strings.Contains(repoDigest[:at], "@") {
|
||||
continue
|
||||
}
|
||||
|
||||
repoRef, err := reference.ParseNormalizedNamed(repoDigest[:at])
|
||||
if err != nil || reference.Path(repoRef) != repository || !sameRegistry(reference.Domain(repoRef), registry) {
|
||||
continue
|
||||
}
|
||||
if _, hasTag := repoRef.(reference.Tagged); hasTag {
|
||||
continue
|
||||
}
|
||||
|
||||
d, err := digest.Parse(repoDigest[at+1:])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
digests = append(digests, d.String())
|
||||
}
|
||||
return digests
|
||||
}
|
||||
|
||||
func sameRegistry(left, right string) bool {
|
||||
left = canonicalRegistry(left)
|
||||
right = canonicalRegistry(right)
|
||||
return left == right ||
|
||||
(left == "ghcr.io" && right == "lscr.io") ||
|
||||
(left == "lscr.io" && right == "ghcr.io")
|
||||
}
|
||||
|
||||
func canonicalRegistry(registry string) string {
|
||||
if registry == "index.docker.io" {
|
||||
return "docker.io"
|
||||
}
|
||||
return registry
|
||||
}
|
||||
|
||||
func (dm *dockerManager) registryImageDigest(registry, repository, tag string) (string, error) {
|
||||
client := dm.registryClient
|
||||
if client == nil {
|
||||
client = &http.Client{Timeout: imageRegistryTimeout}
|
||||
}
|
||||
|
||||
token, err := dm.registryToken(client, registry, repository)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
host := registry
|
||||
if registry == "docker.io" {
|
||||
host = "registry-1.docker.io"
|
||||
}
|
||||
manifestURL := "https://" + host + "/v2/" + repository + "/manifests/" + url.PathEscape(tag)
|
||||
req, err := http.NewRequest(http.MethodHead, manifestURL, nil)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("create manifest request: %w", err)
|
||||
}
|
||||
req.Header.Set("Accept", imageManifestAccept)
|
||||
if token != "" {
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
}
|
||||
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("fetch manifest %s:%s: %w", registry, repository, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("manifest request for %s:%s failed: %s", repository, tag, responseStatus(resp))
|
||||
}
|
||||
|
||||
remote := strings.TrimSpace(resp.Header.Get("Docker-Content-Digest"))
|
||||
d, err := digest.Parse(remote)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("manifest request for %s:%s returned invalid digest: %w", repository, tag, err)
|
||||
}
|
||||
return d.String(), nil
|
||||
}
|
||||
|
||||
func (dm *dockerManager) registryToken(client *http.Client, registry, repository string) (string, error) {
|
||||
var authURL string
|
||||
switch registry {
|
||||
case "docker.io":
|
||||
authURL = "https://auth.docker.io/token?service=registry.docker.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
|
||||
case "ghcr.io", "lscr.io":
|
||||
// lscr.io is the LinuxServer alias for its GHCR-backed images.
|
||||
authURL = "https://ghcr.io/token?service=ghcr.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
|
||||
default:
|
||||
// Anonymous registries remain supported, as they were before the
|
||||
// authenticated Docker Hub and GHCR paths were added.
|
||||
return "", nil
|
||||
}
|
||||
|
||||
req, err := http.NewRequest(http.MethodGet, authURL, nil)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("create registry auth request: %w", err)
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("fetch registry auth token for %s: %w", repository, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("registry auth request for %s failed: %s", repository, responseStatus(resp))
|
||||
}
|
||||
|
||||
var tokenResponse struct {
|
||||
Token string `json:"token"`
|
||||
AccessToken string `json:"access_token"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&tokenResponse); err != nil {
|
||||
return "", fmt.Errorf("decode registry auth response for %s: %w", repository, err)
|
||||
}
|
||||
token := strings.TrimSpace(tokenResponse.Token)
|
||||
if token == "" {
|
||||
token = strings.TrimSpace(tokenResponse.AccessToken)
|
||||
}
|
||||
if token == "" {
|
||||
return "", fmt.Errorf("registry auth response for %s contained no token", repository)
|
||||
}
|
||||
return token, nil
|
||||
}
|
||||
|
||||
func responseStatus(resp *http.Response) string {
|
||||
if resp.Status != "" {
|
||||
return resp.Status
|
||||
}
|
||||
return http.StatusText(resp.StatusCode)
|
||||
}
|
||||
243
agent/docker_registry_test.go
Normal file
243
agent/docker_registry_test.go
Normal file
@@ -0,0 +1,243 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
type registryTransportFunc func(*http.Request) (*http.Response, error)
|
||||
|
||||
func (fn registryTransportFunc) RoundTrip(req *http.Request) (*http.Response, error) {
|
||||
return fn(req)
|
||||
}
|
||||
|
||||
func registryResponse(status int, body string) *http.Response {
|
||||
return &http.Response{
|
||||
StatusCode: status,
|
||||
Status: fmt.Sprintf("%d %s", status, http.StatusText(status)),
|
||||
Header: make(http.Header),
|
||||
Body: io.NopCloser(strings.NewReader(body)),
|
||||
}
|
||||
}
|
||||
|
||||
func registryDigest(fill byte) string {
|
||||
return "sha256:" + strings.Repeat(string(fill), 64)
|
||||
}
|
||||
|
||||
func newRegistryChecker(t *testing.T, inspectBody string, transport http.RoundTripper) *dockerManager {
|
||||
t.Helper()
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = io.WriteString(w, inspectBody)
|
||||
return
|
||||
}
|
||||
http.NotFound(w, r)
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
return &dockerManager{
|
||||
client: newDockerManagerForVersionTest(server).client,
|
||||
registryClient: &http.Client{Transport: transport},
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateUsesInspectAndManifestDigests(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
remote := registryDigest('b')
|
||||
var authCalls, manifestCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
switch {
|
||||
case req.Method == http.MethodGet && req.URL.Host == "auth.docker.io":
|
||||
authCalls.Add(1)
|
||||
require.Equal(t, "/token", req.URL.Path)
|
||||
return registryResponse(http.StatusOK, `{"token":"test-token"}`), nil
|
||||
case req.Method == http.MethodHead && req.URL.Host == "registry-1.docker.io":
|
||||
manifestCalls.Add(1)
|
||||
require.Equal(t, "/v2/library/alpine/manifests/latest", req.URL.Path)
|
||||
require.Equal(t, "Bearer test-token", req.Header.Get("Authorization"))
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", remote)
|
||||
return resp, nil
|
||||
default:
|
||||
return registryResponse(http.StatusNotFound, ""), nil
|
||||
}
|
||||
}))
|
||||
|
||||
available, err := dm.checkImageUpdate("alpine")
|
||||
require.NoError(t, err)
|
||||
require.True(t, available)
|
||||
require.EqualValues(t, 1, authCalls.Load())
|
||||
require.EqualValues(t, 1, manifestCalls.Load())
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateMatchesAnyRepositoryDigest(t *testing.T) {
|
||||
platform := registryDigest('a')
|
||||
index := registryDigest('b')
|
||||
other := registryDigest('c')
|
||||
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
digests []string
|
||||
remote string
|
||||
available bool
|
||||
}{
|
||||
{name: "platform then index, remote index", digests: []string{platform, index}, remote: index},
|
||||
{name: "index then platform, remote index", digests: []string{index, platform}, remote: index},
|
||||
{name: "platform then index, remote platform", digests: []string{platform, index}, remote: platform},
|
||||
{name: "index then platform, remote platform", digests: []string{index, platform}, remote: platform},
|
||||
{name: "neither matches", digests: []string{platform, index}, remote: other, available: true},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
inspect := fmt.Sprintf(`{"RepoDigests":["docker.io/library/busybox@%s","docker.io/library/alpine@%s","docker.io/library/alpine@sha256:invalid","docker.io/library/alpine@%s"]}`, test.remote, test.digests[0], test.digests[1])
|
||||
var manifestCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, inspect, registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
if req.Method == http.MethodGet {
|
||||
return registryResponse(http.StatusOK, `{"token":"test"}`), nil
|
||||
}
|
||||
manifestCalls.Add(1)
|
||||
require.Equal(t, http.MethodHead, req.Method)
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", test.remote)
|
||||
return resp, nil
|
||||
}))
|
||||
|
||||
available, err := dm.checkImageUpdate("alpine")
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, test.available, available)
|
||||
require.EqualValues(t, 1, manifestCalls.Load())
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateReportsUnknownInspectState(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
body string
|
||||
}{
|
||||
{name: "missing field", body: `{}`},
|
||||
{name: "empty field", body: `{"RepoDigests":[]}`},
|
||||
{name: "malformed reference", body: `{"RepoDigests":["not-a-repo-digest"]}`},
|
||||
{name: "wrong repository", body: `{"RepoDigests":["docker.io/library/busybox@` + registryDigest('a') + `"]}`},
|
||||
{name: "malformed digest", body: `{"RepoDigests":["docker.io/library/alpine@sha256:not-a-digest"]}`},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
var registryCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, test.body, registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
registryCalls.Add(1)
|
||||
return registryResponse(http.StatusOK, `{"token":"unexpected"}`), nil
|
||||
}))
|
||||
|
||||
available, err := dm.checkImageUpdate("alpine")
|
||||
require.Error(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 0, registryCalls.Load(), "invalid local state must not query a registry")
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateChecksInspectAuthAndManifestStatuses(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
validInspect := fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
inspectCode int
|
||||
authCode int
|
||||
manifestCode int
|
||||
remote string
|
||||
want string
|
||||
}{
|
||||
{name: "inspect status", inspectCode: http.StatusNotFound, want: "inspect image"},
|
||||
{name: "auth status", inspectCode: http.StatusOK, authCode: http.StatusUnauthorized, want: "registry auth"},
|
||||
{name: "manifest status", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusNotFound, remote: local, want: "manifest request"},
|
||||
{name: "missing digest", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusOK, want: "invalid digest"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if test.inspectCode != http.StatusOK && strings.HasPrefix(r.URL.Path, "/images/") {
|
||||
w.WriteHeader(test.inspectCode)
|
||||
return
|
||||
}
|
||||
_, _ = io.WriteString(w, validInspect)
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
calls := 0
|
||||
dm := &dockerManager{client: newDockerManagerForVersionTest(server).client, registryClient: &http.Client{Transport: registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
calls++
|
||||
if req.Method == http.MethodGet {
|
||||
return registryResponse(test.authCode, `{"token":"test"}`), nil
|
||||
}
|
||||
response := registryResponse(test.manifestCode, "")
|
||||
response.Header.Set("Docker-Content-Digest", test.remote)
|
||||
return response, nil
|
||||
})}}
|
||||
|
||||
_, err := dm.checkImageUpdate("alpine")
|
||||
require.Error(t, err)
|
||||
require.Contains(t, err.Error(), test.want)
|
||||
if test.inspectCode != http.StatusOK {
|
||||
require.Zero(t, calls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateSupportsAnonymousAndLSCRRegistries(t *testing.T) {
|
||||
t.Run("anonymous registry", func(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
var calls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["example.com/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
calls.Add(1)
|
||||
require.Equal(t, http.MethodHead, req.Method)
|
||||
require.Equal(t, "example.com", req.URL.Host)
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", local)
|
||||
return resp, nil
|
||||
}))
|
||||
available, err := dm.checkImageUpdate("example.com/app")
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 1, calls.Load())
|
||||
})
|
||||
|
||||
t.Run("lscr ghcr alias", func(t *testing.T) {
|
||||
local := registryDigest('a')
|
||||
var authCalls, manifestCalls atomic.Int32
|
||||
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["ghcr.io/linuxserver/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
|
||||
if req.Method == http.MethodGet {
|
||||
authCalls.Add(1)
|
||||
return registryResponse(http.StatusOK, `{"token":"test"}`), nil
|
||||
}
|
||||
manifestCalls.Add(1)
|
||||
require.Equal(t, "lscr.io", req.URL.Host)
|
||||
resp := registryResponse(http.StatusOK, "")
|
||||
resp.Header.Set("Docker-Content-Digest", local)
|
||||
return resp, nil
|
||||
}))
|
||||
available, err := dm.checkImageUpdate("lscr.io/linuxserver/app")
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
require.EqualValues(t, 1, authCalls.Load())
|
||||
require.EqualValues(t, 1, manifestCalls.Load())
|
||||
})
|
||||
}
|
||||
|
||||
func TestCheckImageUpdateSkipsPinnedDigest(t *testing.T) {
|
||||
image := "docker.io/library/alpine@" + registryDigest('a')
|
||||
dm := &dockerManager{}
|
||||
available, err := dm.checkImageUpdate(image)
|
||||
require.NoError(t, err)
|
||||
require.False(t, available)
|
||||
}
|
||||
@@ -729,6 +729,7 @@ func TestGetDockerStatsChecksDockerVersionAfterContainerList(t *testing.T) {
|
||||
|
||||
stats, err := dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, stats, "A successful empty snapshot must remain distinguishable from a collection failure")
|
||||
assert.Empty(t, stats)
|
||||
assert.True(t, dm.dockerVersionChecked)
|
||||
assert.Equal(t, tt.expectedGood, dm.goodDockerVersion)
|
||||
@@ -742,6 +743,7 @@ func TestGetDockerStatsChecksDockerVersionAfterContainerList(t *testing.T) {
|
||||
|
||||
stats, err = dm.getDockerStats(defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, stats, "A successful empty snapshot must remain distinguishable from a collection failure")
|
||||
assert.Empty(t, stats)
|
||||
assert.Equal(t, tt.expectedGood, dm.goodDockerVersion)
|
||||
assert.Equal(t, tt.expectedPodman, dm.usingPodman)
|
||||
@@ -804,6 +806,24 @@ func TestGetDockerStatsRetriesVersionCheckUntilSuccess(t *testing.T) {
|
||||
assert.Equal(t, 2, requestCounts["/version"])
|
||||
}
|
||||
|
||||
// A failed decode must not break later decodes. Previously the reused json.Decoder
|
||||
// stayed desynced after one truncated response, breaking decode until restart.
|
||||
func TestDecodeRecoversFromError(t *testing.T) {
|
||||
dm := &dockerManager{}
|
||||
|
||||
// truncated JSON: body reads fine, decode fails
|
||||
var bad []container.ApiInfo
|
||||
err := dm.decode(&http.Response{Body: io.NopCloser(strings.NewReader(`[{"Id":"abc`))}, &bad)
|
||||
require.Error(t, err)
|
||||
|
||||
// the next decode must still succeed
|
||||
var good []container.ApiInfo
|
||||
err = dm.decode(&http.Response{Body: io.NopCloser(strings.NewReader(`[{"Id":"abcdef012345","Names":["/ok"]}]`))}, &good)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, good, 1)
|
||||
assert.Equal(t, "abcdef012345", good[0].Id)
|
||||
}
|
||||
|
||||
func TestCycleCpuDeltas(t *testing.T) {
|
||||
dm := &dockerManager{
|
||||
lastCpuContainer: map[uint16]map[string]uint64{
|
||||
@@ -1003,6 +1023,199 @@ func TestCpuPercentageCalculationWithRealData(t *testing.T) {
|
||||
assert.InDelta(t, expectedPct, actualPct, 0.01)
|
||||
}
|
||||
|
||||
func TestCpuPercentageHandlesCounterRollback(t *testing.T) {
|
||||
// If a stats response is processed after a newer one for the same container,
|
||||
// or an accounting counter resets, the current total can be lower than the
|
||||
// stored previous value. Unsigned subtraction wraps to ~2^64 instead of
|
||||
// going negative, so the percentage explodes, validateCpuPercentage rejects
|
||||
// the sample, and the whole collection is discarded - network stats too.
|
||||
stats := &container.ApiStats{
|
||||
CPUStats: container.CPUStats{
|
||||
CPUUsage: container.CPUUsage{TotalUsage: 1_000_000},
|
||||
SystemUsage: 20_000_000,
|
||||
},
|
||||
}
|
||||
|
||||
// Container counter went backwards.
|
||||
assert.Equal(t, 0.0, stats.CalculateCpuPercentLinux(2_000_000, 10_000_000))
|
||||
// System counter went backwards.
|
||||
assert.Equal(t, 0.0, stats.CalculateCpuPercentLinux(500_000, 30_000_000))
|
||||
// A normal forward sample is unaffected: 500000 / 10000000 * 100 = 5%.
|
||||
assert.InDelta(t, 5.0, stats.CalculateCpuPercentLinux(500_000, 10_000_000), 0.001)
|
||||
}
|
||||
|
||||
func TestCpuPercentageWindowsHandlesCounterRollback(t *testing.T) {
|
||||
now := time.Now()
|
||||
stats := &container.ApiStats{
|
||||
Read: now,
|
||||
NumProcs: 4,
|
||||
CPUStats: container.CPUStats{
|
||||
CPUUsage: container.CPUUsage{TotalUsage: 1_000_000},
|
||||
},
|
||||
}
|
||||
prevRead := now.Add(-time.Second)
|
||||
|
||||
// Container counter went backwards.
|
||||
assert.Equal(t, 0.0, stats.CalculateCpuPercentWindows(2_000_000, prevRead))
|
||||
// A normal forward sample is unaffected.
|
||||
assert.Greater(t, stats.CalculateCpuPercentWindows(500_000, prevRead), 0.0)
|
||||
}
|
||||
|
||||
func TestCalculateCpuPercentPodman(t *testing.T) {
|
||||
baseTime := time.Date(2026, 3, 15, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
prevCpuContainer uint64
|
||||
prevRead time.Time
|
||||
currentUsage uint64
|
||||
currentRead time.Time
|
||||
onlineCPUs uint32
|
||||
expectedPct float64
|
||||
}{
|
||||
{
|
||||
name: "normal calculation",
|
||||
// container used 2ms of CPU over 1s with 2 CPUs → 0.1%
|
||||
prevCpuContainer: 1_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 1_002_000_000, // +2ms CPU time
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 2,
|
||||
expectedPct: 0.1, // 2e6 / (1e9 * 2) * 100
|
||||
},
|
||||
{
|
||||
name: "first run returns zero",
|
||||
prevCpuContainer: 0,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 5_000_000,
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 4,
|
||||
expectedPct: 0.0,
|
||||
},
|
||||
{
|
||||
name: "zero online cpus returns zero",
|
||||
prevCpuContainer: 1_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 1_010_000_000,
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 0,
|
||||
expectedPct: 0.0,
|
||||
},
|
||||
{
|
||||
name: "same read time returns zero",
|
||||
prevCpuContainer: 1_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 1_010_000_000,
|
||||
currentRead: baseTime, // no elapsed time
|
||||
onlineCPUs: 2,
|
||||
expectedPct: 0.0,
|
||||
},
|
||||
{
|
||||
name: "counter rollback returns zero",
|
||||
prevCpuContainer: 2_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 1_000_000_000,
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 2,
|
||||
expectedPct: 0.0,
|
||||
},
|
||||
{
|
||||
name: "100% single cpu",
|
||||
// container consumed a full CPU-second over 1s on a 1-CPU host → 100%
|
||||
prevCpuContainer: 1_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 2_000_000_000, // +1s CPU time
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 1,
|
||||
expectedPct: 100.0, // 1e9 / (1e9 * 1) * 100
|
||||
},
|
||||
{
|
||||
name: "high utilization on multi-cpu host",
|
||||
// container used 800ms on a 4-CPU host over 1s → 20%
|
||||
prevCpuContainer: 10_000_000_000,
|
||||
prevRead: baseTime,
|
||||
currentUsage: 10_800_000_000,
|
||||
currentRead: baseTime.Add(time.Second),
|
||||
onlineCPUs: 4,
|
||||
expectedPct: 20.0, // 800e6 / (1e9 * 4) * 100
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
s := &container.ApiStats{
|
||||
Read: tt.currentRead,
|
||||
CPUStats: container.CPUStats{
|
||||
CPUUsage: container.CPUUsage{TotalUsage: tt.currentUsage},
|
||||
OnlineCPUs: tt.onlineCPUs,
|
||||
},
|
||||
}
|
||||
got := s.CalculateCpuPercentPodman(tt.prevCpuContainer, tt.prevRead)
|
||||
assert.InDelta(t, tt.expectedPct, got, 0.001, "test %q", tt.name)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUpdateContainerStatsPodmanCpuCalculation(t *testing.T) {
|
||||
// Verify that Podman containers use the time-based CPU calculation
|
||||
// when online_cpus is provided in the stats response.
|
||||
// container used 20ms CPU over 1s with 2 CPUs → 1%
|
||||
prevReadTime := time.Date(2026, 3, 15, 21, 26, 58, 0, time.UTC) // 1 second before stats read
|
||||
const prevCpuUsage = uint64(5_000_000_000)
|
||||
|
||||
dm := &dockerManager{
|
||||
client: &http.Client{Transport: roundTripFunc(func(req *http.Request) (*http.Response, error) {
|
||||
switch req.URL.EscapedPath() {
|
||||
case "/containers/0123456789ab/stats":
|
||||
return &http.Response{
|
||||
StatusCode: http.StatusOK,
|
||||
Status: "200 OK",
|
||||
Header: make(http.Header),
|
||||
Body: io.NopCloser(strings.NewReader(`{
|
||||
"read":"2026-03-15T21:26:59Z",
|
||||
"cpu_stats":{"cpu_usage":{"total_usage":5020000000},"system_cpu_usage":9999999,"online_cpus":2},
|
||||
"memory_stats":{"usage":1048576,"stats":{"inactive_file":262144}},
|
||||
"networks":{"eth0":{"rx_bytes":0,"tx_bytes":0}}
|
||||
}`)),
|
||||
Request: req,
|
||||
}, nil
|
||||
default:
|
||||
return nil, fmt.Errorf("unexpected path: %s", req.URL.EscapedPath())
|
||||
}
|
||||
})},
|
||||
containerStatsMap: make(map[string]*container.Stats),
|
||||
usingPodman: true,
|
||||
lastCpuContainer: map[uint16]map[string]uint64{
|
||||
defaultCacheTimeMs: {"0123456789ab": prevCpuUsage},
|
||||
},
|
||||
lastCpuSystem: map[uint16]map[string]uint64{
|
||||
defaultCacheTimeMs: {"0123456789ab": 1}, // intentionally tiny — should NOT be used
|
||||
},
|
||||
lastCpuReadTime: map[uint16]map[string]time.Time{
|
||||
defaultCacheTimeMs: {"0123456789ab": prevReadTime},
|
||||
},
|
||||
networkSentTrackers: make(map[uint16]*deltatracker.DeltaTracker[string, uint64]),
|
||||
networkRecvTrackers: make(map[uint16]*deltatracker.DeltaTracker[string, uint64]),
|
||||
lastNetworkReadTime: make(map[uint16]map[string]time.Time),
|
||||
}
|
||||
|
||||
ctr := &container.ApiInfo{
|
||||
IdShort: "0123456789ab",
|
||||
Names: []string{"/myapp"},
|
||||
Status: "Up 5 minutes",
|
||||
Image: "myapp:latest",
|
||||
}
|
||||
|
||||
err := dm.updateContainerStats(ctr, defaultCacheTimeMs)
|
||||
require.NoError(t, err)
|
||||
|
||||
// cpu delta = 5020000000 - 5000000000 = 20000000 ns (20ms)
|
||||
// elapsed = 1s = 1000000000 ns, online_cpus = 2
|
||||
// expected = 20000000 / (1000000000 * 2) * 100 = 1.0%
|
||||
expectedCpu := 1.0
|
||||
assert.InDelta(t, expectedCpu, dm.containerStatsMap[ctr.IdShort].Cpu, 0.01)
|
||||
}
|
||||
|
||||
func TestNetworkStatsCalculationWithRealData(t *testing.T) {
|
||||
// Create synthetic test data to avoid timing issues
|
||||
apiStats1 := &container.ApiStats{
|
||||
@@ -1462,7 +1675,6 @@ func TestUpdateContainerStatsUsesPodmanInspectHealthFallback(t *testing.T) {
|
||||
}
|
||||
})},
|
||||
containerStatsMap: make(map[string]*container.Stats),
|
||||
apiStats: &container.ApiStats{},
|
||||
usingPodman: true,
|
||||
lastCpuContainer: make(map[uint16]map[string]uint64),
|
||||
lastCpuSystem: make(map[uint16]map[string]uint64),
|
||||
@@ -1864,6 +2076,14 @@ func TestConvertContainerPortsToString(t *testing.T) {
|
||||
},
|
||||
expected: "80, 443",
|
||||
},
|
||||
{
|
||||
name: "ipv4 and ipv6 wildcard bindings are deduplicated",
|
||||
ports: []port{
|
||||
{PublicPort: 80, IP: "0.0.0.0"},
|
||||
{PublicPort: 80, IP: "::"},
|
||||
},
|
||||
expected: "80",
|
||||
},
|
||||
{
|
||||
name: "multiple ports with different IPs",
|
||||
ports: []port{
|
||||
@@ -1872,6 +2092,22 @@ func TestConvertContainerPortsToString(t *testing.T) {
|
||||
},
|
||||
expected: "80, 1.2.3.4:443",
|
||||
},
|
||||
{
|
||||
name: "same port bound to multiple IPs shows all entries",
|
||||
ports: []port{
|
||||
{PublicPort: 65533, IP: "172.16.151.72"},
|
||||
{PublicPort: 65533, IP: "172.16.156.25"},
|
||||
},
|
||||
expected: "172.16.151.72:65533, 172.16.156.25:65533",
|
||||
},
|
||||
{
|
||||
name: "same port bound to IPv4 and IPv6",
|
||||
ports: []port{
|
||||
{PublicPort: 65534, IP: "172.16.151.72"},
|
||||
{PublicPort: 65534, IP: "fd04:38e2:98c6:3fd::72"},
|
||||
},
|
||||
expected: "172.16.151.72:65534, fd04:38e2:98c6:3fd::72:65534",
|
||||
},
|
||||
{
|
||||
name: "ports slice is nilled after call",
|
||||
ports: []port{
|
||||
|
||||
133
agent/fans.go
Normal file
133
agent/fans.go
Normal file
@@ -0,0 +1,133 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
type fanSensor struct {
|
||||
key, path, chip string
|
||||
}
|
||||
|
||||
var getFanSensors = newFanSensorCache(hwmonRoot)
|
||||
|
||||
func newFanSensorCache(root string) func() ([]fanSensor, error) {
|
||||
return sync.OnceValues(func() ([]fanSensor, error) {
|
||||
return discoverHwmonFans(root)
|
||||
})
|
||||
}
|
||||
|
||||
// updateFans populates systemStats.Fans from the host's hwmon sysfs tree.
|
||||
// No-op on platforms where hwmon isn't available (see fans_other.go).
|
||||
func (a *Agent) updateFans(systemStats *system.Stats) {
|
||||
if hwmonRoot == "" {
|
||||
return
|
||||
}
|
||||
sensors, err := getFanSensors()
|
||||
if err != nil {
|
||||
slog.Debug("Error reading fans", "err", err)
|
||||
return
|
||||
}
|
||||
// Filter before reading fan*_input: each read can wake an idle GPU.
|
||||
if a.sensorConfig != nil && a.sensorConfig.skipGPU {
|
||||
sensors = filterGpuFans(sensors)
|
||||
}
|
||||
fans := readFanSensors(sensors)
|
||||
if len(fans) == 0 {
|
||||
return
|
||||
}
|
||||
systemStats.Fans = fans
|
||||
// Note: Commented out because we don't currently use this value in the UI.
|
||||
// Compute the single "dashboard" value used by the FanSpeed alert.
|
||||
// Per-sensor RPMs live in Stats.Fans and drive the multi-line FanChart
|
||||
// in the UI; the alert path only needs one number to compare against
|
||||
// the user's threshold, so we use the highest RPM across all fans
|
||||
// a.systemInfo.DashboardFan = 0
|
||||
// for _, rpm := range fans {
|
||||
// if rpm > a.systemInfo.DashboardFan {
|
||||
// a.systemInfo.DashboardFan = rpm
|
||||
// }
|
||||
// }
|
||||
}
|
||||
|
||||
// readHwmonFans walks the given hwmon root (typically /sys/class/hwmon) and
|
||||
// returns a map of "<chip>_<label-or-fan-idx>" → RPM for every fan*_input
|
||||
// file it finds. Zero RPM is retained because it can represent a real fan that
|
||||
// has stopped; negative and malformed readings are ignored.
|
||||
func readHwmonFans(root string) (map[string]uint16, error) {
|
||||
sensors, err := discoverHwmonFans(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return readFanSensors(sensors), nil
|
||||
}
|
||||
|
||||
func discoverHwmonFans(root string) ([]fanSensor, error) {
|
||||
entries, err := os.ReadDir(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var sensors []fanSensor
|
||||
for _, entry := range entries {
|
||||
chipDir := filepath.Join(root, entry.Name())
|
||||
sensorDir := chipDir
|
||||
inputs, _ := filepath.Glob(filepath.Join(sensorDir, "fan*_input"))
|
||||
|
||||
// Some legacy hwmon drivers (notably applesmc) register a hwmon class
|
||||
// device but create fan attributes on the parent platform device. In
|
||||
// sysfs that parent is exposed through hwmonN/device.
|
||||
if len(inputs) == 0 {
|
||||
deviceDir := filepath.Join(chipDir, "device")
|
||||
if deviceInputs, _ := filepath.Glob(filepath.Join(deviceDir, "fan*_input")); len(deviceInputs) > 0 {
|
||||
sensorDir = deviceDir
|
||||
inputs = deviceInputs
|
||||
}
|
||||
}
|
||||
|
||||
chipName := utils.ReadStringFile(filepath.Join(sensorDir, "name"))
|
||||
if chipName == "" {
|
||||
chipName = utils.ReadStringFile(filepath.Join(chipDir, "name"))
|
||||
}
|
||||
if chipName == "" {
|
||||
chipName = entry.Name()
|
||||
}
|
||||
for _, inputPath := range inputs {
|
||||
base := strings.TrimSuffix(filepath.Base(inputPath), "_input")
|
||||
label := utils.ReadStringFile(filepath.Join(sensorDir, base+"_label"))
|
||||
key := chipName + "_" + base
|
||||
if label != "" {
|
||||
key = chipName + "_" + label
|
||||
}
|
||||
sensors = append(sensors, fanSensor{key, inputPath, chipName})
|
||||
}
|
||||
}
|
||||
return sensors, nil
|
||||
}
|
||||
|
||||
func readFanSensors(sensors []fanSensor) map[string]uint16 {
|
||||
fans := make(map[string]uint16, len(sensors))
|
||||
for _, sensor := range sensors {
|
||||
if rpm, ok := utils.ReadUintFile(sensor.path); ok {
|
||||
fans[sensor.key] = uint16(rpm)
|
||||
}
|
||||
}
|
||||
return fans
|
||||
}
|
||||
|
||||
// filterGpuFans drops GPU chips without touching the shared cache backing array.
|
||||
func filterGpuFans(sensors []fanSensor) []fanSensor {
|
||||
kept := make([]fanSensor, 0, len(sensors))
|
||||
for _, sensor := range sensors {
|
||||
if isGpuChipName(sensor.chip) {
|
||||
continue
|
||||
}
|
||||
kept = append(kept, sensor)
|
||||
}
|
||||
return kept
|
||||
}
|
||||
8
agent/fans_linux.go
Normal file
8
agent/fans_linux.go
Normal file
@@ -0,0 +1,8 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
// hwmonRoot is the sysfs entry point for hardware monitor chips. Each
|
||||
// subdirectory (hwmon0, hwmon1, …) is one chip; fan*_input files inside it
|
||||
// expose RPM readings.
|
||||
const hwmonRoot = "/sys/class/hwmon"
|
||||
7
agent/fans_other.go
Normal file
7
agent/fans_other.go
Normal file
@@ -0,0 +1,7 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
// hwmonRoot is empty on non-Linux platforms — fan RPM reporting via sysfs
|
||||
// hwmon is Linux-specific. updateFans() short-circuits when this is empty.
|
||||
const hwmonRoot = ""
|
||||
122
agent/fans_test.go
Normal file
122
agent/fans_test.go
Normal file
@@ -0,0 +1,122 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// writeFile creates path with parents and writes contents.
|
||||
func writeFile(t *testing.T, path, contents string) {
|
||||
t.Helper()
|
||||
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, os.WriteFile(path, []byte(contents), 0o644))
|
||||
}
|
||||
|
||||
// TestReadHwmonFans verifies the /sys/class/hwmon walker:
|
||||
// - picks up fan*_input from every chip,
|
||||
// - keys entries by chip name + sensor label (or fan idx if no label),
|
||||
// - retains 0 RPM for stopped fans,
|
||||
// - tolerates chips with no fan files at all.
|
||||
func TestReadHwmonFans(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
|
||||
// hwmon0: Raspberry Pi 5 active cooler — one fan, no label.
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "name"), "pwmfan\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "fan1_input"), "6500\n")
|
||||
|
||||
// hwmon1: a thermal-only chip, no fan files. Must not error.
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "name"), "cpu_thermal\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "temp1_input"), "55000\n")
|
||||
|
||||
// hwmon2: two fans — one stopped (0 RPM) and one labeled "chassis".
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "name"), "nct6798\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan1_input"), "0\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan2_input"), "1200\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon2", "fan2_label"), "chassis\n")
|
||||
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
|
||||
assert.Equal(t, map[string]uint16{
|
||||
"pwmfan_fan1": 6500,
|
||||
"nct6798_fan1": 0,
|
||||
"nct6798_chassis": 1200,
|
||||
}, fans)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansLegacyParent verifies legacy hwmon layouts such as applesmc,
|
||||
// where the hwmon class node exists but fan attributes live on hwmonN/device.
|
||||
func TestReadHwmonFansLegacyParent(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
deviceDir := filepath.Join(root, "devices", "applesmc.768")
|
||||
writeFile(t, filepath.Join(deviceDir, "name"), "applesmc\n")
|
||||
writeFile(t, filepath.Join(deviceDir, "fan1_input"), "1202\n")
|
||||
writeFile(t, filepath.Join(deviceDir, "fan1_label"), "Exhaust\n")
|
||||
|
||||
chipDir := filepath.Join(root, "hwmon1")
|
||||
require.NoError(t, os.MkdirAll(chipDir, 0o755))
|
||||
require.NoError(t, os.Symlink(deviceDir, filepath.Join(chipDir, "device")))
|
||||
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, map[string]uint16{"applesmc_Exhaust": 1202}, fans)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansMissingRoot returns an error rather than panicking when the
|
||||
// hwmon root doesn't exist (e.g. running on a kernel without hwmon support).
|
||||
func TestReadHwmonFansMissingRoot(t *testing.T) {
|
||||
_, err := readHwmonFans(filepath.Join(t.TempDir(), "does-not-exist"))
|
||||
assert.Error(t, err)
|
||||
}
|
||||
|
||||
// TestReadHwmonFansEmpty returns an empty map (not nil error) when the root
|
||||
// exists but contains no chips at all.
|
||||
func TestReadHwmonFansEmpty(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
fans, err := readHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, fans)
|
||||
}
|
||||
|
||||
func TestFanDiscoveryCache(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
input := filepath.Join(root, "hwmon0", "fan1_input")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "name"), "chip\n")
|
||||
writeFile(t, input, "1000\n")
|
||||
|
||||
getSensors := newFanSensorCache(root)
|
||||
sensors, err := getSensors()
|
||||
require.NoError(t, err)
|
||||
fans := readFanSensors(sensors)
|
||||
assert.Equal(t, uint16(1000), fans["chip_fan1"])
|
||||
|
||||
writeFile(t, input, "1200\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "fan1_label"), "case\n")
|
||||
sensors, err = getSensors()
|
||||
require.NoError(t, err)
|
||||
fans = readFanSensors(sensors)
|
||||
assert.Equal(t, map[string]uint16{"chip_fan1": 1200}, fans)
|
||||
}
|
||||
|
||||
func TestFilterGpuFans(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "name"), "xe\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon0", "fan1_input"), "1200\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "name"), "nct6798\n")
|
||||
writeFile(t, filepath.Join(root, "hwmon1", "fan1_input"), "800\n")
|
||||
|
||||
discovered, err := discoverHwmonFans(root)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, discovered, 2)
|
||||
|
||||
filtered := filterGpuFans(discovered)
|
||||
require.Len(t, filtered, 1)
|
||||
assert.Equal(t, "nct6798_fan1", filtered[0].key)
|
||||
assert.Len(t, discovered, 2)
|
||||
}
|
||||
@@ -50,6 +50,9 @@ func generateFingerprint(hostname, cpuModel string) string {
|
||||
if info, err := cpu.Info(); err == nil && len(info) > 0 {
|
||||
cpuModel = info[0].ModelName
|
||||
}
|
||||
if cpuModel == "" {
|
||||
cpuModel = getCpuModelFromCpuinfo()
|
||||
}
|
||||
}
|
||||
fingerprint = hostname + cpuModel
|
||||
}
|
||||
|
||||
66
agent/gpu.go
66
agent/gpu.go
@@ -48,6 +48,8 @@ type GPUManager struct {
|
||||
// Per-cache-key tracking for delta calculations
|
||||
// cacheKey -> gpuId -> snapshot of last count/usage/power values
|
||||
lastSnapshots map[uint16]map[string]*gpuSnapshot
|
||||
// Per-card energy snapshots for Intel sysfs power calculation.
|
||||
intelSysfsEnergySnapshots map[string]intelSysfsEnergySnapshot
|
||||
}
|
||||
|
||||
// gpuSnapshot stores the last observed incremental values for delta tracking
|
||||
@@ -90,6 +92,7 @@ const (
|
||||
collectorSourceNVML collectorSource = "nvml"
|
||||
collectorSourceNvidiaSMI collectorSource = collectorSource(nvidiaSmiCmd)
|
||||
collectorSourceIntelGpuTop collectorSource = collectorSource(intelGpuStatsCmd)
|
||||
collectorSourceIntelSysfs collectorSource = "intel_sysfs"
|
||||
collectorSourceAmdSysfs collectorSource = "amd_sysfs"
|
||||
collectorSourceRocmSMI collectorSource = collectorSource(rocmSmiCmd)
|
||||
collectorSourceMacmon collectorSource = collectorSource(macmonCmd)
|
||||
@@ -106,6 +109,7 @@ func isValidCollectorSource(source collectorSource) bool {
|
||||
collectorSourceNVML,
|
||||
collectorSourceNvidiaSMI,
|
||||
collectorSourceIntelGpuTop,
|
||||
collectorSourceIntelSysfs,
|
||||
collectorSourceAmdSysfs,
|
||||
collectorSourceRocmSMI,
|
||||
collectorSourceMacmon,
|
||||
@@ -122,6 +126,8 @@ type gpuCapabilities struct {
|
||||
hasAmdSysfs bool
|
||||
hasTegrastats bool
|
||||
hasIntelGpuTop bool
|
||||
hasXe bool
|
||||
hasIntelSysfs bool
|
||||
hasNvtop bool
|
||||
hasMacmon bool
|
||||
hasPowermetrics bool
|
||||
@@ -355,12 +361,16 @@ func (gm *GPUManager) calculateGPUAverage(id string, gpu *system.GPUData, cacheK
|
||||
|
||||
// If no new data arrived
|
||||
if deltaCount == 0 {
|
||||
// If GPU appears suspended (instantaneous values are 0), return zero values
|
||||
// Otherwise return last known average for temporary collection gaps
|
||||
if gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
|
||||
// Only discrete GPUs report temp/memory, so treat all-zero as suspended (return zeros).
|
||||
// Engine-based (Intel) GPUs don't, so carry the last average forward across sample gaps.
|
||||
if gpu.Engines == nil && gpu.Temperature == 0 && gpu.MemoryUsed == 0 {
|
||||
return system.GPUData{Name: gpu.Name}
|
||||
}
|
||||
return gm.lastAvgData[id] // zero value if not found
|
||||
lastAvg := gm.lastAvgData[id] // zero value if not found
|
||||
if lastAvg.Name == "" {
|
||||
lastAvg.Name = gpu.Name
|
||||
}
|
||||
return lastAvg
|
||||
}
|
||||
|
||||
// Calculate new average
|
||||
@@ -369,12 +379,13 @@ func (gm *GPUManager) calculateGPUAverage(id string, gpu *system.GPUData, cacheK
|
||||
|
||||
gpuAvg.Power = utils.TwoDecimals(deltaPower / float64(deltaCount))
|
||||
|
||||
gpuAvg.PowerPkg = utils.TwoDecimals(deltaPowerPkg / float64(deltaCount))
|
||||
|
||||
if gpu.Engines != nil {
|
||||
// make fresh map for averaged engine metrics to avoid mutating
|
||||
// the accumulator map stored in gm.GpuDataMap
|
||||
gpuAvg.Engines = make(map[string]float64, len(gpu.Engines))
|
||||
gpuAvg.Usage = gm.calculateIntelGPUUsage(&gpuAvg, gpu, lastSnapshot, deltaCount)
|
||||
gpuAvg.PowerPkg = utils.TwoDecimals(deltaPowerPkg / float64(deltaCount))
|
||||
} else {
|
||||
gpuAvg.Usage = utils.TwoDecimals(deltaUsage / float64(deltaCount))
|
||||
}
|
||||
@@ -443,7 +454,9 @@ func (gm *GPUManager) storeSnapshot(id string, gpu *system.GPUData, cacheKey uin
|
||||
// It only reports capability presence and does not apply policy decisions.
|
||||
func (gm *GPUManager) discoverGpuCapabilities() gpuCapabilities {
|
||||
caps := gpuCapabilities{
|
||||
hasAmdSysfs: gm.hasAmdSysfs(),
|
||||
hasAmdSysfs: gm.hasAmdSysfs(),
|
||||
hasXe: gm.hasXe(),
|
||||
hasIntelSysfs: gm.hasIntelSysfs(),
|
||||
}
|
||||
if _, err := exec.LookPath(nvidiaSmiCmd); err == nil {
|
||||
caps.hasNvidiaSmi = true
|
||||
@@ -472,7 +485,7 @@ func (gm *GPUManager) discoverGpuCapabilities() gpuCapabilities {
|
||||
}
|
||||
|
||||
func hasAnyGpuCollector(caps gpuCapabilities) bool {
|
||||
return caps.hasNvidiaSmi || caps.hasRocmSmi || caps.hasAmdSysfs || caps.hasTegrastats || caps.hasIntelGpuTop || caps.hasNvtop || caps.hasMacmon || caps.hasPowermetrics
|
||||
return caps.hasNvidiaSmi || caps.hasRocmSmi || caps.hasAmdSysfs || caps.hasTegrastats || caps.hasIntelGpuTop || caps.hasIntelSysfs || caps.hasNvtop || caps.hasMacmon || caps.hasPowermetrics
|
||||
}
|
||||
|
||||
func (gm *GPUManager) startIntelCollector() {
|
||||
@@ -563,6 +576,13 @@ func (gm *GPUManager) collectorDefinitions(caps gpuCapabilities) map[collectorSo
|
||||
return true
|
||||
},
|
||||
},
|
||||
collectorSourceIntelSysfs: {
|
||||
group: collectorGroupIntel,
|
||||
available: caps.hasIntelSysfs,
|
||||
start: func(_ func()) bool {
|
||||
return gm.startIntelSysfsCollector()
|
||||
},
|
||||
},
|
||||
collectorSourceAmdSysfs: {
|
||||
group: collectorGroupAmd,
|
||||
available: caps.hasAmdSysfs,
|
||||
@@ -705,9 +725,12 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
priorities = append(priorities, collectorSourceAmdSysfs)
|
||||
}
|
||||
|
||||
if caps.hasIntelGpuTop {
|
||||
if caps.hasIntelGpuTop && !caps.hasXe {
|
||||
priorities = append(priorities, collectorSourceIntelGpuTop)
|
||||
}
|
||||
if caps.hasIntelSysfs {
|
||||
priorities = append(priorities, collectorSourceIntelSysfs)
|
||||
}
|
||||
|
||||
// Apple collectors are currently opt-in only for testing.
|
||||
// Enable them with GPU_COLLECTOR=macmon or GPU_COLLECTOR=powermetrics.
|
||||
@@ -727,9 +750,36 @@ func (gm *GPUManager) resolveLegacyCollectorPriority(caps gpuCapabilities) []col
|
||||
return priorities
|
||||
}
|
||||
|
||||
// gpuHwmonChips are hwmon chip names belonging to GPUs. Sensor reads on some
|
||||
// of these drivers (notably Intel Xe, where each read is a runtime PM resume)
|
||||
// wake the card, so SKIP_GPU must avoid touching them, not just hide them.
|
||||
var gpuHwmonChips = []string{"xe", "i915", "amdgpu", "radeon", "nvidia", "nouveau"}
|
||||
|
||||
func isGpuChipName(name string) bool {
|
||||
name = strings.ToLower(strings.TrimSpace(name))
|
||||
for _, chip := range gpuHwmonChips {
|
||||
if name == chip {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// SensorKeys are "<chip>" or "<chip>_<label>".
|
||||
func isGpuSensorKey(key string) bool {
|
||||
key = strings.ToLower(strings.TrimSpace(key))
|
||||
for _, chip := range gpuHwmonChips {
|
||||
if key == chip || strings.HasPrefix(key, chip+"_") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// NewGPUManager creates and initializes a new GPUManager
|
||||
func NewGPUManager() (*GPUManager, error) {
|
||||
if skipGPU, _ := utils.GetEnv("SKIP_GPU"); skipGPU == "true" {
|
||||
slog.Info("SKIP_GPU enabled, skipping GPU monitoring (collectors, temperatures, and fans)")
|
||||
return nil, nil
|
||||
}
|
||||
var gm GPUManager
|
||||
|
||||
280
agent/gpu_intel_sysfs_linux.go
Normal file
280
agent/gpu_intel_sysfs_linux.go
Normal file
@@ -0,0 +1,280 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
)
|
||||
|
||||
var (
|
||||
drmSysfsRoot = "/sys/class/drm"
|
||||
intelSysfsNow = time.Now
|
||||
)
|
||||
|
||||
type intelSysfsEnergySnapshot struct {
|
||||
microjoules uint64
|
||||
timestamp time.Time
|
||||
}
|
||||
|
||||
type intelSysfsCard struct {
|
||||
cardPath string
|
||||
hwmonDir string
|
||||
}
|
||||
|
||||
// hasIntelSysfs returns true if any Intel DRM card exposes an hwmon energy counter.
|
||||
func (gm *GPUManager) hasIntelSysfs() bool {
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
return err == nil && len(cards) > 0
|
||||
}
|
||||
|
||||
// startIntelSysfsCollector starts Intel GPU collection via sysfs.
|
||||
func (gm *GPUManager) startIntelSysfsCollector() bool {
|
||||
go func() {
|
||||
if err := gm.collectIntelSysfsStats(); err != nil {
|
||||
slog.Warn("Error collecting Intel GPU data via sysfs", "err", err)
|
||||
}
|
||||
}()
|
||||
return true
|
||||
}
|
||||
|
||||
// collectIntelSysfsStats collects Intel GPU metrics directly from DRM sysfs / hwmon.
|
||||
func (gm *GPUManager) collectIntelSysfsStats() error {
|
||||
sysfsPollInterval := 3000 * time.Millisecond
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(cards) == 0 {
|
||||
return errNoValidData
|
||||
}
|
||||
|
||||
slog.Debug("Using sysfs for Intel GPU data collection", "cards", len(cards))
|
||||
for _, card := range cards {
|
||||
slog.Debug("Intel sysfs card detected", "card", filepath.Base(card.cardPath), "hwmon", card.hwmonDir)
|
||||
}
|
||||
|
||||
failures := 0
|
||||
for {
|
||||
hasData := false
|
||||
for _, card := range cards {
|
||||
if gm.updateIntelSysfsGpuData(card.cardPath, card.hwmonDir) {
|
||||
hasData = true
|
||||
}
|
||||
}
|
||||
if !hasData {
|
||||
failures++
|
||||
if failures > maxFailureRetries {
|
||||
return errNoValidData
|
||||
}
|
||||
slog.Warn("No Intel GPU data from sysfs", "failures", failures)
|
||||
time.Sleep(retryWaitTime)
|
||||
continue
|
||||
}
|
||||
failures = 0
|
||||
time.Sleep(sysfsPollInterval)
|
||||
}
|
||||
}
|
||||
|
||||
func discoverIntelSysfsCards() ([]intelSysfsCard, error) {
|
||||
paths, err := filepath.Glob(filepath.Join(drmSysfsRoot, "card*"))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var cards []intelSysfsCard
|
||||
for _, cardPath := range paths {
|
||||
if strings.Contains(filepath.Base(cardPath), "-") || !isIntelGpu(cardPath) {
|
||||
continue
|
||||
}
|
||||
hwmonDir := findIntelEnergyHwmon(filepath.Join(cardPath, "device"))
|
||||
if hwmonDir == "" {
|
||||
continue
|
||||
}
|
||||
cards = append(cards, intelSysfsCard{cardPath: cardPath, hwmonDir: hwmonDir})
|
||||
}
|
||||
return cards, nil
|
||||
}
|
||||
|
||||
func isIntelGpu(cardPath string) bool {
|
||||
vendor, err := utils.ReadStringFileLimited(filepath.Join(cardPath, "device/vendor"), 64)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.EqualFold(strings.TrimSpace(vendor), "0x8086")
|
||||
}
|
||||
|
||||
func findIntelEnergyHwmon(devicePath string) string {
|
||||
hwmons, _ := filepath.Glob(filepath.Join(devicePath, "hwmon/hwmon*"))
|
||||
var fallback string
|
||||
for _, hwmonDir := range hwmons {
|
||||
if !sysfsFileExists(filepath.Join(hwmonDir, "energy1_input")) {
|
||||
continue
|
||||
}
|
||||
if name, err := utils.ReadStringFileLimited(filepath.Join(hwmonDir, "name"), 64); err == nil && strings.EqualFold(strings.TrimSpace(name), "xe") {
|
||||
return hwmonDir
|
||||
}
|
||||
if fallback == "" {
|
||||
fallback = hwmonDir
|
||||
}
|
||||
}
|
||||
return fallback
|
||||
}
|
||||
|
||||
func sysfsFileExists(path string) bool {
|
||||
_, err := utils.ReadStringFileLimited(path, 1)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// updateIntelSysfsGpuData reads GPU metrics from sysfs and updates the GPU data map.
|
||||
// Returns true if the required energy counter was read successfully.
|
||||
func (gm *GPUManager) updateIntelSysfsGpuData(cardPath, hwmonDir string) bool {
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
id := filepath.Base(cardPath)
|
||||
|
||||
energy, err := readSysfsUint(filepath.Join(hwmonDir, "energy1_input"))
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
now := intelSysfsNow()
|
||||
power, hasPower := gm.calculateIntelSysfsPower(id, energy, now)
|
||||
powerPkg, hasPowerPkg := gm.readIntelSysfsPowerPkg(id, hwmonDir, now)
|
||||
temp := readIntelSysfsTemperature(hwmonDir)
|
||||
usage, usageErr := readOptionalSysfsFloat(filepath.Join(devicePath, "gpu_busy_percent"))
|
||||
memUsed, memUsedErr := readFirstOptionalSysfsFloat(
|
||||
filepath.Join(devicePath, "mem_info_vram_used"),
|
||||
filepath.Join(devicePath, "mem_info_lmem_used"),
|
||||
filepath.Join(devicePath, "mem_info_local_mem_used"),
|
||||
)
|
||||
memTotal, memTotalErr := readFirstOptionalSysfsFloat(
|
||||
filepath.Join(devicePath, "mem_info_vram_total"),
|
||||
filepath.Join(devicePath, "mem_info_lmem_total"),
|
||||
filepath.Join(devicePath, "mem_info_local_mem_total"),
|
||||
)
|
||||
|
||||
gm.Lock()
|
||||
defer gm.Unlock()
|
||||
|
||||
gpu, ok := gm.GpuDataMap[id]
|
||||
if !ok {
|
||||
gpu = &system.GPUData{Name: getIntelSysfsGpuName(cardPath)}
|
||||
gm.GpuDataMap[id] = gpu
|
||||
}
|
||||
|
||||
if usageErr == nil {
|
||||
gpu.Usage += usage
|
||||
}
|
||||
if memUsedErr == nil {
|
||||
gpu.MemoryUsed = utils.BytesToMegabytes(memUsed)
|
||||
}
|
||||
if memTotalErr == nil {
|
||||
gpu.MemoryTotal = utils.BytesToMegabytes(memTotal)
|
||||
}
|
||||
if temp > 0 {
|
||||
gpu.Temperature = temp
|
||||
}
|
||||
if hasPower {
|
||||
gpu.Power += power
|
||||
slog.Debug("Computed Intel sysfs GPU power", "card", id, "watts", power)
|
||||
}
|
||||
if hasPowerPkg {
|
||||
gpu.PowerPkg += powerPkg
|
||||
}
|
||||
gpu.Count++
|
||||
return true
|
||||
}
|
||||
|
||||
func (gm *GPUManager) calculateIntelSysfsPower(cardID string, microjoules uint64, timestamp time.Time) (float64, bool) {
|
||||
if gm.intelSysfsEnergySnapshots == nil {
|
||||
gm.intelSysfsEnergySnapshots = make(map[string]intelSysfsEnergySnapshot)
|
||||
}
|
||||
|
||||
last, ok := gm.intelSysfsEnergySnapshots[cardID]
|
||||
gm.intelSysfsEnergySnapshots[cardID] = intelSysfsEnergySnapshot{microjoules: microjoules, timestamp: timestamp}
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
if microjoules < last.microjoules {
|
||||
slog.Debug("Intel sysfs energy counter reset", "card", cardID)
|
||||
return 0, false
|
||||
}
|
||||
elapsed := timestamp.Sub(last.timestamp).Seconds()
|
||||
if elapsed <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
delta := microjoules - last.microjoules
|
||||
return float64(delta) / 1_000_000.0 / elapsed, true
|
||||
}
|
||||
|
||||
func (gm *GPUManager) readIntelSysfsPowerPkg(cardID, hwmonDir string, timestamp time.Time) (float64, bool) {
|
||||
energyPaths, _ := filepath.Glob(filepath.Join(hwmonDir, "energy*_input"))
|
||||
for _, path := range energyPaths {
|
||||
if filepath.Base(path) == "energy1_input" {
|
||||
continue
|
||||
}
|
||||
energy, err := readSysfsUint(path)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
return gm.calculateIntelSysfsPower(cardID+":"+filepath.Base(path), energy, timestamp)
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
func readIntelSysfsTemperature(hwmonDir string) float64 {
|
||||
tempPaths, _ := filepath.Glob(filepath.Join(hwmonDir, "temp*_input"))
|
||||
for _, path := range tempPaths {
|
||||
temp, err := readSysfsFloat(path)
|
||||
if err == nil && temp > 0 {
|
||||
return temp / 1000.0
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func readSysfsUint(path string) (uint64, error) {
|
||||
val, err := utils.ReadStringFileLimited(path, 64)
|
||||
if err != nil {
|
||||
slog.Debug("Failed to read sysfs value", "path", path, "error", err)
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseUint(strings.TrimSpace(val), 10, 64)
|
||||
}
|
||||
|
||||
func readOptionalSysfsFloat(path string) (float64, error) {
|
||||
val, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseFloat(strings.TrimSpace(string(val)), 64)
|
||||
}
|
||||
|
||||
func readFirstOptionalSysfsFloat(paths ...string) (float64, error) {
|
||||
for _, path := range paths {
|
||||
val, err := readOptionalSysfsFloat(path)
|
||||
if err == nil {
|
||||
return val, nil
|
||||
}
|
||||
}
|
||||
return 0, fmt.Errorf("no sysfs values found")
|
||||
}
|
||||
|
||||
func getIntelSysfsGpuName(cardPath string) string {
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
if product, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "product_name"), 128); err == nil && strings.TrimSpace(product) != "" {
|
||||
return strings.TrimSpace(product)
|
||||
}
|
||||
if name, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "name"), 128); err == nil && strings.TrimSpace(name) != "" {
|
||||
return strings.TrimSpace(name)
|
||||
}
|
||||
return fmt.Sprintf("Intel GPU %s", filepath.Base(cardPath))
|
||||
}
|
||||
217
agent/gpu_intel_sysfs_linux_test.go
Normal file
217
agent/gpu_intel_sysfs_linux_test.go
Normal file
@@ -0,0 +1,217 @@
|
||||
//go:build linux
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func setupIntelSysfsTest(t *testing.T) (root, cardPath, hwmonPath string) {
|
||||
t.Helper()
|
||||
root = t.TempDir()
|
||||
oldRoot := drmSysfsRoot
|
||||
drmSysfsRoot = root
|
||||
t.Cleanup(func() {
|
||||
drmSysfsRoot = oldRoot
|
||||
})
|
||||
|
||||
cardPath = filepath.Join(root, "card0")
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
hwmonPath = filepath.Join(devicePath, "hwmon", "hwmon0")
|
||||
require.NoError(t, os.MkdirAll(hwmonPath, 0o755))
|
||||
return root, cardPath, hwmonPath
|
||||
}
|
||||
|
||||
func writeIntelSysfsFile(t *testing.T, basePath, name, content string) {
|
||||
t.Helper()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(basePath, name), []byte(content), 0o644))
|
||||
}
|
||||
|
||||
func setIntelSysfsTime(t *testing.T, now time.Time) {
|
||||
t.Helper()
|
||||
oldNow := intelSysfsNow
|
||||
intelSysfsNow = func() time.Time { return now }
|
||||
t.Cleanup(func() {
|
||||
intelSysfsNow = oldNow
|
||||
})
|
||||
}
|
||||
|
||||
func TestIntelSysfsDetectsIntelCardWithEnergy(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.True(t, gm.hasIntelSysfs())
|
||||
|
||||
cards, err := discoverIntelSysfsCards()
|
||||
require.NoError(t, err)
|
||||
require.Len(t, cards, 1)
|
||||
assert.Equal(t, cardPath, cards[0].cardPath)
|
||||
assert.Equal(t, hwmonPath, cards[0].hwmonDir)
|
||||
}
|
||||
|
||||
func TestIntelSysfsRejectsNonIntelCard(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x1002\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.False(t, gm.hasIntelSysfs())
|
||||
}
|
||||
|
||||
func TestIntelSysfsRequiresEnergyInput(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
|
||||
gm := &GPUManager{}
|
||||
assert.False(t, gm.hasIntelSysfs())
|
||||
}
|
||||
|
||||
func TestIntelSysfsFirstSampleInitializesWithoutBogusPower(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "name", "xe\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
ok := gm.updateIntelSysfsGpuData(cardPath, hwmonPath)
|
||||
require.True(t, ok)
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, "Intel GPU card0", gpu.Name)
|
||||
assert.Equal(t, 0.0, gpu.Power)
|
||||
assert.Equal(t, 1.0, gpu.Count)
|
||||
}
|
||||
|
||||
func TestIntelSysfsSecondSampleComputesWatts(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
oldNow := intelSysfsNow
|
||||
intelSysfsNow = func() time.Time { return time.Unix(100, 0) }
|
||||
t.Cleanup(func() { intelSysfsNow = oldNow })
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "6000000\n")
|
||||
intelSysfsNow = func() time.Time { return time.Unix(102, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 2.5, gpu.Power)
|
||||
assert.Equal(t, 2.0, gpu.Count)
|
||||
}
|
||||
|
||||
func TestIntelSysfsSecondEnergyCounterMapsToPowerPkg(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy2_input", "2000000\n")
|
||||
|
||||
oldNow := intelSysfsNow
|
||||
t.Cleanup(func() { intelSysfsNow = oldNow })
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
intelSysfsNow = func() time.Time { return time.Unix(100, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "2000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy2_input", "8000000\n")
|
||||
intelSysfsNow = func() time.Time { return time.Unix(102, 0) }
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 0.5, gpu.Power)
|
||||
assert.Equal(t, 3.0, gpu.PowerPkg)
|
||||
}
|
||||
|
||||
func TestIntelSysfsCounterResetSkipsOneSample(t *testing.T) {
|
||||
gm := &GPUManager{}
|
||||
power, ok := gm.calculateIntelSysfsPower("card0", 5000000, time.Unix(100, 0))
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, 0.0, power)
|
||||
|
||||
power, ok = gm.calculateIntelSysfsPower("card0", 1000000, time.Unix(101, 0))
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, 0.0, power)
|
||||
|
||||
power, ok = gm.calculateIntelSysfsPower("card0", 3000000, time.Unix(103, 0))
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, 1.0, power)
|
||||
}
|
||||
|
||||
func TestIntelSysfsTempInputMapsToCelsius(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "temp1_input", "43500\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 43.5, gpu.Temperature)
|
||||
}
|
||||
|
||||
func TestIntelSysfsMissingOptionalFilesDoNotFail(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 0.0, gpu.Usage)
|
||||
assert.Equal(t, 0.0, gpu.MemoryUsed)
|
||||
assert.Equal(t, 0.0, gpu.MemoryTotal)
|
||||
assert.Equal(t, 0.0, gpu.Temperature)
|
||||
}
|
||||
|
||||
func TestIntelSysfsMapsOpportunisticMemoryAndUsage(t *testing.T) {
|
||||
_, cardPath, hwmonPath := setupIntelSysfsTest(t)
|
||||
devicePath := filepath.Join(cardPath, "device")
|
||||
writeIntelSysfsFile(t, devicePath, "vendor", "0x8086\n")
|
||||
writeIntelSysfsFile(t, devicePath, "gpu_busy_percent", "37\n")
|
||||
writeIntelSysfsFile(t, devicePath, "mem_info_lmem_used", "1073741824\n")
|
||||
writeIntelSysfsFile(t, devicePath, "mem_info_lmem_total", "2147483648\n")
|
||||
writeIntelSysfsFile(t, hwmonPath, "energy1_input", "1000000\n")
|
||||
setIntelSysfsTime(t, time.Unix(100, 0))
|
||||
|
||||
gm := &GPUManager{GpuDataMap: make(map[string]*system.GPUData)}
|
||||
require.True(t, gm.updateIntelSysfsGpuData(cardPath, hwmonPath))
|
||||
|
||||
gpu := gm.GpuDataMap["card0"]
|
||||
require.NotNil(t, gpu)
|
||||
assert.Equal(t, 37.0, gpu.Usage)
|
||||
assert.Equal(t, utils.BytesToMegabytes(1073741824), gpu.MemoryUsed)
|
||||
assert.Equal(t, utils.BytesToMegabytes(2147483648), gpu.MemoryTotal)
|
||||
}
|
||||
13
agent/gpu_intel_sysfs_unsupported.go
Normal file
13
agent/gpu_intel_sysfs_unsupported.go
Normal file
@@ -0,0 +1,13 @@
|
||||
//go:build !linux
|
||||
|
||||
package agent
|
||||
|
||||
type intelSysfsEnergySnapshot struct{}
|
||||
|
||||
func (gm *GPUManager) hasIntelSysfs() bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func (gm *GPUManager) startIntelSysfsCollector() bool {
|
||||
return false
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"io"
|
||||
"log/slog"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -48,9 +49,14 @@ func (gm *GPUManager) updateNvtopSnapshots(snapshots []nvtopSnapshot) bool {
|
||||
|
||||
valid := false
|
||||
usedIDs := make(map[string]struct{}, len(snapshots))
|
||||
var xeName string
|
||||
for i, sample := range snapshots {
|
||||
// nvtop leaves device_name unset on xe devices.
|
||||
if sample.DeviceName == "" {
|
||||
continue
|
||||
if xeName == "" {
|
||||
xeName = xeGpuName()
|
||||
}
|
||||
sample.DeviceName = xeName
|
||||
}
|
||||
indexID := "n" + strconv.Itoa(i)
|
||||
id := indexID
|
||||
@@ -158,3 +164,38 @@ func (gm *GPUManager) startNvtopCollector(interval string, onFailure func()) {
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// xeDevicePath returns the sysfs device path of the first xe GPU, or "".
|
||||
func xeDevicePath() string {
|
||||
cards, err := filepath.Glob("/sys/class/drm/card*")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
for _, card := range cards {
|
||||
if strings.Contains(filepath.Base(card), "-") {
|
||||
continue
|
||||
}
|
||||
if uevent, err := utils.ReadStringFileLimited(filepath.Join(card, "device", "uevent"), 4096); err == nil && strings.Contains(uevent, "DRIVER=xe") {
|
||||
return filepath.Join(card, "device")
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func (gm *GPUManager) hasXe() bool {
|
||||
return xeDevicePath() != ""
|
||||
}
|
||||
|
||||
// xeGpuName names an xe GPU from its PCI device id; nvtop leaves device_name unset on xe.
|
||||
func xeGpuName() string {
|
||||
devicePath := xeDevicePath()
|
||||
if devicePath == "" {
|
||||
return "GPU"
|
||||
}
|
||||
id, err := utils.ReadStringFileLimited(filepath.Join(devicePath, "device"), 64)
|
||||
if err != nil {
|
||||
return "GPU"
|
||||
}
|
||||
id = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(id, "0x")))
|
||||
return "Intel GPU (" + id + ")"
|
||||
}
|
||||
|
||||
@@ -332,11 +332,12 @@ func TestUpdateNvtopSnapshotsKeepsDeviceAssociationWhenOrderChanges(t *testing.T
|
||||
}
|
||||
|
||||
func TestParseCollectorPriority(t *testing.T) {
|
||||
got := parseCollectorPriority(" nvml, nvidia-smi, intel_gpu_top, amd_sysfs, nvtop, rocm-smi, bad ")
|
||||
got := parseCollectorPriority(" nvml, nvidia-smi, intel_gpu_top, intel_sysfs, amd_sysfs, nvtop, rocm-smi, bad ")
|
||||
want := []collectorSource{
|
||||
collectorSourceNVML,
|
||||
collectorSourceNvidiaSMI,
|
||||
collectorSourceIntelGpuTop,
|
||||
collectorSourceIntelSysfs,
|
||||
collectorSourceAmdSysfs,
|
||||
collectorSourceNVTop,
|
||||
collectorSourceRocmSMI,
|
||||
@@ -565,6 +566,42 @@ func TestGetCurrentData(t *testing.T) {
|
||||
assert.EqualValues(t, 2, gm.GpuDataMap["0"].Count, "Count should still be 2")
|
||||
})
|
||||
|
||||
t.Run("carries Intel GPU average forward between samples", func(t *testing.T) {
|
||||
// Intel GPUs report no temp/memory, so between-sample gaps (delta 0) must
|
||||
// reuse the last average instead of returning zeros and blanking the chart.
|
||||
gm := &GPUManager{
|
||||
GpuDataMap: map[string]*system.GPUData{
|
||||
"0": {
|
||||
Name: "GPU",
|
||||
Usage: 0, // derived from engines for Intel
|
||||
Power: 200, // averages to 100 over 2 counts
|
||||
PowerPkg: 60, // averages to 30 over 2 counts
|
||||
Count: 2,
|
||||
Engines: map[string]float64{
|
||||
"Render/3D": 80, // averages to 40
|
||||
"Video": 20, // averages to 10
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
cacheKey := uint16(1000) // realtime cache key
|
||||
|
||||
// First collection - computes and stores averages
|
||||
result1 := gm.GetCurrentData(cacheKey)
|
||||
assert.InDelta(t, 100.0, result1["0"].Power, 0.01)
|
||||
assert.InDelta(t, 30.0, result1["0"].PowerPkg, 0.01)
|
||||
assert.InDelta(t, 40.0, result1["0"].Engines["Render/3D"], 0.01)
|
||||
|
||||
// Second collection with no new sample (count unchanged, temp/mem still 0).
|
||||
// Must carry the last average forward rather than blanking to zero.
|
||||
result2 := gm.GetCurrentData(cacheKey)
|
||||
assert.Equal(t, "GPU", result2["0"].Name, "Name should be preserved")
|
||||
assert.InDelta(t, 100.0, result2["0"].Power, 0.01, "Should reuse last average power, not 0")
|
||||
assert.InDelta(t, 30.0, result2["0"].PowerPkg, 0.01, "Should reuse last average package power, not 0")
|
||||
assert.InDelta(t, 40.0, result2["0"].Engines["Render/3D"], 0.01, "Should reuse last average engine usage")
|
||||
})
|
||||
|
||||
t.Run("tracks separate averages per cache key", func(t *testing.T) {
|
||||
gm := &GPUManager{
|
||||
GpuDataMap: map[string]*system.GPUData{
|
||||
@@ -1082,7 +1119,6 @@ func TestCalculateGPUAverage(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestGPUCapabilitiesAndLegacyPriority(t *testing.T) {
|
||||
// Save original PATH
|
||||
hasAmdSysfs := (&GPUManager{}).hasAmdSysfs()
|
||||
|
||||
tests := []struct {
|
||||
@@ -1176,7 +1212,7 @@ echo "[]"`
|
||||
{
|
||||
name: "no gpu tools available",
|
||||
setupCommands: func(_ string) error {
|
||||
t.Setenv("PATH", "")
|
||||
// The subtest already restricts PATH to its empty temporary directory.
|
||||
return nil
|
||||
},
|
||||
wantErr: true,
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
|
||||
"log/slog"
|
||||
@@ -51,6 +52,8 @@ func NewHandlerRegistry() *HandlerRegistry {
|
||||
registry.Register(common.GetContainerInfo, &GetContainerInfoHandler{})
|
||||
registry.Register(common.GetSmartData, &GetSmartDataHandler{})
|
||||
registry.Register(common.GetSystemdInfo, &GetSystemdInfoHandler{})
|
||||
registry.Register(common.SyncNetworkMonitors, &SyncNetworkMonitorsHandler{})
|
||||
registry.Register(common.GetZfsData, &GetZfsDataHandler{})
|
||||
|
||||
return registry
|
||||
}
|
||||
@@ -166,14 +169,33 @@ type GetSmartDataHandler struct{}
|
||||
|
||||
func (h *GetSmartDataHandler) Handle(hctx *HandlerContext) error {
|
||||
if hctx.Agent.smartManager == nil {
|
||||
// return empty map to indicate no data
|
||||
return hctx.SendResponse(map[string]smart.SmartData{}, hctx.RequestID)
|
||||
return hctx.SendResponse(smart.SmartDataResponse{Data: map[string]smart.SmartData{}}, hctx.RequestID)
|
||||
}
|
||||
if err := hctx.Agent.smartManager.Refresh(false); err != nil {
|
||||
complete, err := hctx.Agent.smartManager.Refresh(false)
|
||||
if err != nil {
|
||||
slog.Debug("smart refresh failed", "err", err)
|
||||
}
|
||||
data := hctx.Agent.smartManager.GetCurrentData()
|
||||
return hctx.SendResponse(data, hctx.RequestID)
|
||||
return hctx.SendResponse(smart.SmartDataResponse{
|
||||
Data: hctx.Agent.smartManager.GetCurrentData(),
|
||||
Complete: complete,
|
||||
}, hctx.RequestID)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// GetZfsDataHandler handles ZFS detail data requests
|
||||
type GetZfsDataHandler struct{}
|
||||
|
||||
func (h *GetZfsDataHandler) Handle(hctx *HandlerContext) error {
|
||||
if hctx.Agent.storagePoolManager == nil {
|
||||
return hctx.SendResponse(nil, hctx.RequestID)
|
||||
}
|
||||
var req common.ZfsDataRequest
|
||||
if err := cbor.Unmarshal(hctx.Request.Data, &req); err != nil {
|
||||
return err
|
||||
}
|
||||
return hctx.SendResponse(hctx.Agent.storagePoolManager.GetDetail(req.Force), hctx.RequestID)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
@@ -203,3 +225,21 @@ func (h *GetSystemdInfoHandler) Handle(hctx *HandlerContext) error {
|
||||
|
||||
return hctx.SendResponse(details, hctx.RequestID)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// SyncNetworkMonitorsHandler handles monitor configuration sync from hub
|
||||
type SyncNetworkMonitorsHandler struct{}
|
||||
|
||||
func (h *SyncNetworkMonitorsHandler) Handle(hctx *HandlerContext) error {
|
||||
var req monitor.SyncRequest
|
||||
if err := cbor.Unmarshal(hctx.Request.Data, &req); err != nil {
|
||||
return err
|
||||
}
|
||||
resp, err := hctx.Agent.monitorManager.HandleSyncRequest(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return hctx.SendResponse(resp, hctx.RequestID)
|
||||
}
|
||||
|
||||
@@ -4,9 +4,12 @@ package agent
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/smart"
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
@@ -17,6 +20,44 @@ type MockHandler struct {
|
||||
handleFunc func(ctx *HandlerContext) error
|
||||
}
|
||||
|
||||
func TestNewAgentResponseSmartData(t *testing.T) {
|
||||
response := newAgentResponse(smart.SmartDataResponse{
|
||||
Data: map[string]smart.SmartData{
|
||||
"AAA": {SerialNumber: "AAA"},
|
||||
},
|
||||
Complete: true,
|
||||
}, nil)
|
||||
|
||||
assert.Equal(t, "AAA", response.SmartData["AAA"].SerialNumber)
|
||||
assert.True(t, response.SmartComplete)
|
||||
}
|
||||
|
||||
func TestGetZfsDataHandlerForceRefresh(t *testing.T) {
|
||||
poolCalls := 0
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
poolCalls++
|
||||
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.GetDetail(false)
|
||||
|
||||
requestData, err := cbor.Marshal(common.ZfsDataRequest{Force: true})
|
||||
assert.NoError(t, err)
|
||||
ctx := &HandlerContext{
|
||||
Agent: &Agent{storagePoolManager: zm},
|
||||
Request: &common.HubRequest[cbor.RawMessage]{
|
||||
Action: common.GetZfsData,
|
||||
Data: requestData,
|
||||
},
|
||||
SendResponse: func(any, *uint32) error { return nil },
|
||||
}
|
||||
|
||||
assert.NoError(t, (&GetZfsDataHandler{}).Handle(ctx))
|
||||
assert.Equal(t, 2, poolCalls)
|
||||
}
|
||||
|
||||
func (m *MockHandler) Handle(ctx *HandlerContext) error {
|
||||
if m.handleFunc != nil {
|
||||
return m.handleFunc(ctx)
|
||||
|
||||
@@ -17,15 +17,17 @@ import (
|
||||
var mdraidSysfsRoot = "/sys"
|
||||
|
||||
type mdraidHealth struct {
|
||||
level string
|
||||
arrayState string
|
||||
degraded uint64
|
||||
raidDisks uint64
|
||||
syncAction string
|
||||
syncCompleted string
|
||||
syncSpeed string
|
||||
mismatchCnt uint64
|
||||
capacity uint64
|
||||
level string
|
||||
arrayState string
|
||||
degraded uint64
|
||||
faultyDisks uint64
|
||||
populatedDisks uint64
|
||||
raidDisks uint64
|
||||
syncAction string
|
||||
syncCompleted string
|
||||
syncSpeed string
|
||||
mismatchCnt uint64
|
||||
capacity uint64
|
||||
}
|
||||
|
||||
// scanMdraidDevices discovers Linux md arrays exposed in sysfs.
|
||||
@@ -92,6 +94,9 @@ func (sm *SmartManager) collectMdraidHealth(deviceInfo *DeviceInfo) (bool, error
|
||||
if health.degraded > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "Degraded", RawValue: health.degraded})
|
||||
}
|
||||
if health.faultyDisks > 0 {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "FaultyDisks", RawValue: health.faultyDisks})
|
||||
}
|
||||
if health.syncAction != "" {
|
||||
attrs = append(attrs, &smart.SmartAttribute{Name: "SyncAction", RawString: health.syncAction})
|
||||
}
|
||||
@@ -152,6 +157,7 @@ func readMdraidHealth(blockName string) (mdraidHealth, bool) {
|
||||
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "degraded")); ok {
|
||||
out.degraded = val
|
||||
}
|
||||
out.faultyDisks, out.populatedDisks = countMdraidMemberStates(blockName, mdraidSysfsRoot)
|
||||
if val, ok := utils.ReadUintFile(filepath.Join(mdDir, "mismatch_cnt")); ok {
|
||||
out.mismatchCnt = val
|
||||
}
|
||||
@@ -177,13 +183,27 @@ func mdraidSmartStatus(health mdraidHealth) string {
|
||||
case "resync", "recover", "reshape":
|
||||
return "WARNING"
|
||||
}
|
||||
if health.degraded > 0 {
|
||||
// Use actual faulty member count rather than the degraded counter, which
|
||||
// equals raid_disks minus active_disks. On QNAP systems raid_disks may be
|
||||
// set to a large value (e.g. 32) while only a few slots are ever used,
|
||||
// making degraded misleadingly large despite zero failed disks.
|
||||
if health.faultyDisks > 0 {
|
||||
return "FAILED"
|
||||
}
|
||||
switch syncAction {
|
||||
case "check", "repair":
|
||||
if health.degraded > 0 {
|
||||
if isSparseSlotDegraded(health) {
|
||||
// A sysfs snapshot cannot distinguish reserved slots from a removed
|
||||
// member on sparse arrays, so report the ambiguity as a warning.
|
||||
return "WARNING"
|
||||
}
|
||||
return "FAILED"
|
||||
}
|
||||
if health.mismatchCnt > 0 {
|
||||
return "WARNING"
|
||||
}
|
||||
// "check" and "repair" are requested consistency scans, not evidence of
|
||||
// array failure. With no health issues above, keep scrubbing green while
|
||||
// reporting the sync action and progress attributes.
|
||||
switch state {
|
||||
case "clean", "active", "active-idle", "write-pending", "read-auto", "readonly":
|
||||
return "PASSED"
|
||||
@@ -191,6 +211,43 @@ func mdraidSmartStatus(health mdraidHealth) string {
|
||||
return "UNKNOWN"
|
||||
}
|
||||
|
||||
// countMdraidMemberStates reads member device directories under
|
||||
// block/<name>/md and returns how many are explicitly marked "faulty", plus
|
||||
// how many are populated at all (regardless of state). populatedDisks lets
|
||||
// callers distinguish RAID slots that were never used (QNAP reserves far
|
||||
// more raid_disks than it ever populates) from members that went missing.
|
||||
func countMdraidMemberStates(blockName, root string) (faultyDisks, populatedDisks uint64) {
|
||||
devDir := filepath.Join(root, "block", blockName, "md")
|
||||
entries, err := os.ReadDir(devDir)
|
||||
if err != nil {
|
||||
return 0, 0
|
||||
}
|
||||
for _, ent := range entries {
|
||||
if !strings.HasPrefix(ent.Name(), "dev-") {
|
||||
continue
|
||||
}
|
||||
populatedDisks++
|
||||
statePath := filepath.Join(devDir, ent.Name(), "state")
|
||||
state := utils.ReadStringFile(statePath)
|
||||
if strings.Contains(state, "faulty") {
|
||||
faultyDisks++
|
||||
}
|
||||
}
|
||||
return faultyDisks, populatedDisks
|
||||
}
|
||||
|
||||
// isSparseSlotDegraded reports whether a non-zero "degraded" count may be
|
||||
// explained by RAID slots that were never populated. QNAP configures system
|
||||
// arrays with raid_disks set to a large fixed maximum (e.g. 32) far beyond the
|
||||
// handful of slots it ever populates, so sparse slots outnumber populated ones.
|
||||
func isSparseSlotDegraded(health mdraidHealth) bool {
|
||||
if health.populatedDisks == 0 || health.raidDisks <= health.populatedDisks {
|
||||
return false
|
||||
}
|
||||
sparseSlots := health.raidDisks - health.populatedDisks
|
||||
return sparseSlots > health.populatedDisks
|
||||
}
|
||||
|
||||
// isMdraidBlockName matches /dev/mdN-style block device names.
|
||||
func isMdraidBlockName(name string) bool {
|
||||
if !strings.HasPrefix(name, "md") {
|
||||
|
||||
@@ -40,6 +40,15 @@ func TestMdraidMockSysfsScanAndCollect(t *testing.T) {
|
||||
write(filepath.Join(mdDir, "sync_completed"), "10%\n")
|
||||
write(filepath.Join(mdDir, "sync_speed"), "100M\n")
|
||||
write(filepath.Join(mdDir, "mismatch_cnt"), "0\n")
|
||||
|
||||
// Simulate two healthy member devices (no faulty state).
|
||||
for _, dev := range []string{"dev-sda", "dev-sdb"} {
|
||||
devPath := filepath.Join(mdDir, dev)
|
||||
if err := os.MkdirAll(devPath, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
write(filepath.Join(devPath, "state"), "in_sync\n")
|
||||
}
|
||||
write(filepath.Join(queueDir, "logical_block_size"), "512\n")
|
||||
write(filepath.Join(tmp, "block", "md0", "size"), "2048\n")
|
||||
|
||||
@@ -81,19 +90,110 @@ func TestMdraidMockSysfsScanAndCollect(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCountMdraidMemberStates(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
|
||||
write := func(path, content string) {
|
||||
t.Helper()
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
mdDir := filepath.Join(tmp, "block", "md0", "md")
|
||||
|
||||
// No dev-* entries: zero faulty, zero populated.
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 0 {
|
||||
t.Fatalf("no members: got (faulty=%d populated=%d), want (0,0)", faulty, populated)
|
||||
}
|
||||
|
||||
// Two healthy members.
|
||||
write(filepath.Join(mdDir, "dev-sda", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 2 {
|
||||
t.Fatalf("all in_sync: got (faulty=%d populated=%d), want (0,2)", faulty, populated)
|
||||
}
|
||||
|
||||
// One faulty member.
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "faulty\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 1 || populated != 2 {
|
||||
t.Fatalf("one faulty: got (faulty=%d populated=%d), want (1,2)", faulty, populated)
|
||||
}
|
||||
|
||||
// QNAP-style: 28 degraded slots but no dev-* entries for them, 4 in_sync.
|
||||
write(filepath.Join(mdDir, "dev-sdb", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdc", "state"), "in_sync\n")
|
||||
write(filepath.Join(mdDir, "dev-sdd", "state"), "in_sync\n")
|
||||
if faulty, populated := countMdraidMemberStates("md0", tmp); faulty != 0 || populated != 4 {
|
||||
t.Fatalf("qnap sparse: got (faulty=%d populated=%d), want (0,4)", faulty, populated)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMdraidSmartStatus(t *testing.T) {
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "inactive"}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(inactive) = %q, want FAILED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, syncAction: "recover"}); got != "WARNING" {
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1, syncAction: "recover"}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded+recover) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded) = %q, want FAILED", got)
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", degraded: 1, faultyDisks: 1}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded+faulty) = %q, want FAILED", got)
|
||||
}
|
||||
// QNAP-style: raid_disks=32 but only 4 populated; degraded=28 but no faulty devices.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 28, faultyDisks: 0, raidDisks: 32, populatedDisks: 4}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(qnap sparse) = %q, want WARNING", got)
|
||||
}
|
||||
// A member disappearing from the same sparse array is indistinguishable
|
||||
// from another reserved slot, so it must not be reported as healthy.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 29, faultyDisks: 0, raidDisks: 32, populatedDisks: 3}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(qnap sparse missing member) = %q, want WARNING", got)
|
||||
}
|
||||
// A genuinely missing member (removed dev-* entry, not just an unpopulated
|
||||
// QNAP reserve slot) must still fail: raid_disks=4, only 3 populated, all
|
||||
// of them in_sync, so faultyDisks==0 but degraded==1.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 3}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(missing member) = %q, want FAILED", got)
|
||||
}
|
||||
// Degraded with no member-state info at all (e.g. sysfs read failed) must
|
||||
// still fail rather than being silently treated as a sparse QNAP array.
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", degraded: 1, faultyDisks: 0, raidDisks: 4, populatedDisks: 0}); got != "FAILED" {
|
||||
t.Fatalf("mdraidSmartStatus(degraded, no member info) = %q, want FAILED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "active", syncAction: "recover"}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(recover) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "check"}); got != "PASSED" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+check) = %q, want PASSED", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "check", mismatchCnt: 1}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+check+mismatch) = %q, want WARNING", got)
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", mismatchCnt: 1}); got != "WARNING" {
|
||||
t.Fatalf("mdraidSmartStatus(clean+mismatch) = %q, want WARNING", got)
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
health mdraidHealth
|
||||
want string
|
||||
}{
|
||||
{"clean", mdraidHealth{arrayState: "clean"}, "PASSED"},
|
||||
{"active", mdraidHealth{arrayState: "active"}, "PASSED"},
|
||||
{"mismatch", mdraidHealth{arrayState: "active", mismatchCnt: 1}, "WARNING"},
|
||||
{"degraded", mdraidHealth{arrayState: "active", degraded: 1}, "FAILED"},
|
||||
{"faulty member", mdraidHealth{arrayState: "active", faultyDisks: 1}, "FAILED"},
|
||||
{"inactive", mdraidHealth{arrayState: "inactive"}, "FAILED"},
|
||||
{"unknown", mdraidHealth{arrayState: "unknown"}, "UNKNOWN"},
|
||||
} {
|
||||
t.Run("repair/"+tc.name, func(t *testing.T) {
|
||||
tc.health.syncAction = "repair"
|
||||
if got := mdraidSmartStatus(tc.health); got != tc.want {
|
||||
t.Fatalf("mdraidSmartStatus(%+v) = %q, want %s", tc.health, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean"}); got != "PASSED" {
|
||||
t.Fatalf("mdraidSmartStatus(clean) = %q, want PASSED", got)
|
||||
}
|
||||
|
||||
195
agent/network_monitor.go
Normal file
195
agent/network_monitor.go
Normal file
@@ -0,0 +1,195 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
)
|
||||
|
||||
// MonitorManager manages network monitor configurations and task lifetimes.
|
||||
type MonitorManager struct {
|
||||
mu sync.RWMutex
|
||||
monitors map[string]*monitorTask // keyed by monitor ID
|
||||
probe monitorProbe
|
||||
certCheck certChecker
|
||||
resumeGuard monitorResumeGuard
|
||||
}
|
||||
|
||||
func newMonitorManager() *MonitorManager {
|
||||
return newMonitorManagerWithProbe(networkMonitorProbe(&http.Client{Timeout: monitor.MaxProbeTimeout}))
|
||||
}
|
||||
|
||||
func newMonitorManagerWithProbe(probe monitorProbe) *MonitorManager {
|
||||
return &MonitorManager{monitors: make(map[string]*monitorTask), probe: probe, certCheck: checkCert}
|
||||
}
|
||||
|
||||
// SyncMonitors replaces all monitor tasks with the given configs.
|
||||
func (pm *MonitorManager) SyncMonitors(configs []monitor.Config) {
|
||||
pm.mu.Lock()
|
||||
defer pm.mu.Unlock()
|
||||
|
||||
// Build set of new keys
|
||||
newKeys := make(map[string]monitor.Config, len(configs))
|
||||
for _, cfg := range configs {
|
||||
if cfg.ID == "" {
|
||||
continue
|
||||
}
|
||||
newKeys[cfg.ID] = cfg
|
||||
}
|
||||
|
||||
// Stop removed monitors
|
||||
for key, task := range pm.monitors {
|
||||
if _, exists := newKeys[key]; !exists {
|
||||
task.cancel()
|
||||
delete(pm.monitors, key)
|
||||
}
|
||||
}
|
||||
|
||||
// Start new monitors and restart tasks whose config changed.
|
||||
for key, cfg := range newKeys {
|
||||
task, exists := pm.monitors[key]
|
||||
if exists && task.config == cfg {
|
||||
continue
|
||||
}
|
||||
if exists {
|
||||
task.cancel()
|
||||
}
|
||||
task = newMonitorTaskFromExisting(cfg, task)
|
||||
task.resumeGuard = &pm.resumeGuard
|
||||
pm.resumeGuard.start()
|
||||
pm.monitors[key] = task
|
||||
pm.startMonitor(task)
|
||||
}
|
||||
if len(pm.monitors) == 0 {
|
||||
pm.resumeGuard.shutdown()
|
||||
}
|
||||
}
|
||||
|
||||
// HandleSyncRequest applies a full or incremental monitor sync request.
|
||||
func (pm *MonitorManager) HandleSyncRequest(req monitor.SyncRequest) (monitor.SyncResponse, error) {
|
||||
switch req.Action {
|
||||
case monitor.SyncActionReplace:
|
||||
pm.SyncMonitors(req.Configs)
|
||||
return monitor.SyncResponse{}, nil
|
||||
case monitor.SyncActionUpsert:
|
||||
result, err := pm.UpsertMonitor(req.Config, req.RunNow)
|
||||
if err != nil {
|
||||
return monitor.SyncResponse{}, err
|
||||
}
|
||||
if result == nil {
|
||||
return monitor.SyncResponse{}, nil
|
||||
}
|
||||
return monitor.SyncResponse{Result: *result}, nil
|
||||
case monitor.SyncActionDelete:
|
||||
if req.Config.ID == "" {
|
||||
return monitor.SyncResponse{}, errors.New("missing monitor ID for delete")
|
||||
}
|
||||
pm.DeleteMonitor(req.Config.ID)
|
||||
return monitor.SyncResponse{}, nil
|
||||
default:
|
||||
return monitor.SyncResponse{}, fmt.Errorf("unknown monitor sync action: %d", req.Action)
|
||||
}
|
||||
}
|
||||
|
||||
// UpsertMonitor creates or replaces a single monitor task.
|
||||
func (pm *MonitorManager) UpsertMonitor(config monitor.Config, runNow bool) (*monitor.Result, error) {
|
||||
if config.ID == "" {
|
||||
return nil, errors.New("missing monitor ID")
|
||||
}
|
||||
|
||||
pm.mu.Lock()
|
||||
task, exists := pm.monitors[config.ID]
|
||||
if exists && task.config == config {
|
||||
pm.mu.Unlock()
|
||||
if !runNow {
|
||||
return nil, nil
|
||||
}
|
||||
return pm.runNow(task), nil
|
||||
}
|
||||
if exists {
|
||||
task.cancel()
|
||||
}
|
||||
task = newMonitorTaskFromExisting(config, task)
|
||||
task.resumeGuard = &pm.resumeGuard
|
||||
pm.resumeGuard.start()
|
||||
pm.monitors[config.ID] = task
|
||||
pm.mu.Unlock()
|
||||
|
||||
if runNow {
|
||||
result := pm.runNow(task)
|
||||
pm.startMonitor(task)
|
||||
return result, nil
|
||||
}
|
||||
pm.startMonitor(task)
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// runNow runs a probe and any due certificate check concurrently, so the
|
||||
// response fits within the hub's single probe timeout budget.
|
||||
func (pm *MonitorManager) runNow(task *monitorTask) *monitor.Result {
|
||||
var wg sync.WaitGroup
|
||||
wg.Go(func() { task.refreshCert(pm.certCheck) })
|
||||
result := task.runProbe(pm.probe)
|
||||
wg.Wait()
|
||||
if result != nil {
|
||||
result.Cert = task.certInfo()
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// DeleteMonitor stops and removes a single monitor task.
|
||||
func (pm *MonitorManager) DeleteMonitor(id string) {
|
||||
if id == "" {
|
||||
return
|
||||
}
|
||||
pm.mu.Lock()
|
||||
defer pm.mu.Unlock()
|
||||
if task, exists := pm.monitors[id]; exists {
|
||||
task.cancel()
|
||||
delete(pm.monitors, id)
|
||||
}
|
||||
if len(pm.monitors) == 0 {
|
||||
pm.resumeGuard.shutdown()
|
||||
}
|
||||
}
|
||||
|
||||
// GetResults returns aggregated results for all monitors over the last supplied duration in ms.
|
||||
func (pm *MonitorManager) GetResults(durationMs uint16) map[string]monitor.Result {
|
||||
pm.mu.RLock()
|
||||
defer pm.mu.RUnlock()
|
||||
|
||||
results := make(map[string]monitor.Result, len(pm.monitors))
|
||||
now := time.Now()
|
||||
duration := time.Duration(durationMs) * time.Millisecond
|
||||
|
||||
for _, task := range pm.monitors {
|
||||
result, ok := task.history.result(duration, now)
|
||||
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// Only the default interval updates monitor records on the hub, so
|
||||
// realtime requests must not consume the unsent certificate.
|
||||
if durationMs == defaultDataCacheTimeMs {
|
||||
result.Cert = task.takeUnsentCert()
|
||||
}
|
||||
results[task.config.ID] = result
|
||||
}
|
||||
|
||||
return results
|
||||
}
|
||||
|
||||
// Stop stops all monitor tasks.
|
||||
func (pm *MonitorManager) Stop() {
|
||||
pm.mu.Lock()
|
||||
defer pm.mu.Unlock()
|
||||
for key, task := range pm.monitors {
|
||||
task.cancel()
|
||||
delete(pm.monitors, key)
|
||||
}
|
||||
pm.resumeGuard.shutdown()
|
||||
}
|
||||
74
agent/network_monitor_cert.go
Normal file
74
agent/network_monitor_cert.go
Normal file
@@ -0,0 +1,74 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
)
|
||||
|
||||
const (
|
||||
certCheckInterval = 24 * time.Hour
|
||||
certCheckRetryInterval = time.Hour
|
||||
)
|
||||
|
||||
// certChecker fetches the leaf certificate for an HTTPS target.
|
||||
type certChecker func(context.Context, string) (monitor.CertInfo, error)
|
||||
|
||||
// certCheckEnabled reports whether a monitor's certificate is checked, which is
|
||||
// the case for every HTTP monitor with an https target.
|
||||
func certCheckEnabled(config monitor.Config) bool {
|
||||
return config.Protocol == "http" && len(config.Target) > 8 && strings.EqualFold(config.Target[:8], "https://")
|
||||
}
|
||||
|
||||
// checkCert reads the leaf certificate presented by an HTTPS target. The chain is
|
||||
// not verified, so expired or self-signed certificates are still reported.
|
||||
func checkCert(ctx context.Context, target string) (monitor.CertInfo, error) {
|
||||
address, host, err := certAddress(target)
|
||||
if err != nil {
|
||||
return monitor.CertInfo{}, err
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, monitor.MaxProbeTimeout)
|
||||
defer cancel()
|
||||
dialer := tls.Dialer{Config: &tls.Config{ServerName: host, InsecureSkipVerify: true}}
|
||||
conn, err := dialer.DialContext(ctx, "tcp", address)
|
||||
if err != nil {
|
||||
return monitor.CertInfo{}, err
|
||||
}
|
||||
defer conn.Close()
|
||||
certs := conn.(*tls.Conn).ConnectionState().PeerCertificates
|
||||
if len(certs) == 0 {
|
||||
return monitor.CertInfo{}, errors.New("no peer certificates")
|
||||
}
|
||||
leaf := certs[0]
|
||||
return monitor.CertInfo{
|
||||
Expires: leaf.NotAfter.UnixMilli(),
|
||||
Issuer: leaf.Issuer.CommonName,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// certAddress returns the dial address and server name for an HTTPS URL.
|
||||
func certAddress(target string) (address, host string, err error) {
|
||||
u, err := url.Parse(target)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
if !strings.EqualFold(u.Scheme, "https") {
|
||||
return "", "", fmt.Errorf("certificate check requires an https target: %s", target)
|
||||
}
|
||||
host = u.Hostname()
|
||||
if host == "" {
|
||||
return "", "", fmt.Errorf("missing host in target: %s", target)
|
||||
}
|
||||
port := u.Port()
|
||||
if port == "" {
|
||||
port = "443"
|
||||
}
|
||||
return net.JoinHostPort(host, port), host, nil
|
||||
}
|
||||
184
agent/network_monitor_cert_test.go
Normal file
184
agent/network_monitor_cert_test.go
Normal file
@@ -0,0 +1,184 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"testing/synctest"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestCheckCertReadsUnverifiedLeaf(t *testing.T) {
|
||||
server := httptest.NewTLSServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {}))
|
||||
defer server.Close()
|
||||
|
||||
// httptest uses a self-signed certificate, which must still be reported.
|
||||
info, err := checkCert(context.Background(), server.URL)
|
||||
require.NoError(t, err)
|
||||
leaf := server.Certificate()
|
||||
assert.Equal(t, leaf.NotAfter.UnixMilli(), info.Expires)
|
||||
assert.Equal(t, leaf.Issuer.CommonName, info.Issuer)
|
||||
}
|
||||
|
||||
func TestCertAddress(t *testing.T) {
|
||||
tests := []struct {
|
||||
target, address, host string
|
||||
wantErr bool
|
||||
}{
|
||||
{target: "https://example.com", address: "example.com:443", host: "example.com"},
|
||||
{target: "https://example.com:8443/path?q=1", address: "example.com:8443", host: "example.com"},
|
||||
{target: "HTTPS://[::1]:9443", address: "[::1]:9443", host: "::1"},
|
||||
{target: "http://example.com", wantErr: true},
|
||||
{target: "https://", wantErr: true},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
address, host, err := certAddress(tt.target)
|
||||
if tt.wantErr {
|
||||
assert.Error(t, err, tt.target)
|
||||
continue
|
||||
}
|
||||
require.NoError(t, err, tt.target)
|
||||
assert.Equal(t, tt.address, address)
|
||||
assert.Equal(t, tt.host, host)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRefreshCertCadence(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
task := newMonitorTask(monitor.Config{ID: "test", Target: "https://example.test", Protocol: "http"})
|
||||
defer task.cancel()
|
||||
var calls int
|
||||
var fail error
|
||||
// Far enough out that the regular interval applies for the whole test.
|
||||
expires := time.Now().Add(365 * 24 * time.Hour).UnixMilli()
|
||||
check := func(context.Context, string) (monitor.CertInfo, error) {
|
||||
calls++
|
||||
if fail != nil {
|
||||
return monitor.CertInfo{}, fail
|
||||
}
|
||||
return monitor.CertInfo{Expires: expires + int64(calls)}, nil
|
||||
}
|
||||
|
||||
task.refreshCert(check)
|
||||
require.NotNil(t, task.certInfo())
|
||||
assert.Equal(t, expires+1, task.certInfo().Expires)
|
||||
|
||||
// Not due again until the check interval passes.
|
||||
time.Sleep(certCheckInterval - time.Second)
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 1, calls)
|
||||
time.Sleep(time.Second)
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 2, calls)
|
||||
|
||||
// Failures keep the last known certificate and retry sooner.
|
||||
fail = errors.New("connection refused")
|
||||
time.Sleep(certCheckInterval)
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 3, calls)
|
||||
assert.Equal(t, expires+2, task.certInfo().Expires)
|
||||
time.Sleep(certCheckRetryInterval)
|
||||
fail = nil
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 4, calls)
|
||||
assert.Equal(t, expires+4, task.certInfo().Expires)
|
||||
})
|
||||
}
|
||||
|
||||
func TestRefreshCertRetriesSoonerNearExpiry(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
expires time.Duration // relative to the check
|
||||
interval time.Duration
|
||||
}{
|
||||
{"expired", -time.Hour, certCheckRetryInterval},
|
||||
{"expires before next regular check", certCheckInterval - time.Minute, certCheckRetryInterval},
|
||||
{"expires after next regular check", certCheckInterval + time.Minute, certCheckInterval},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
task := newMonitorTask(monitor.Config{ID: "test", Target: "https://example.test", Protocol: "http"})
|
||||
defer task.cancel()
|
||||
var calls int
|
||||
check := func(context.Context, string) (monitor.CertInfo, error) {
|
||||
calls++
|
||||
return monitor.CertInfo{Expires: time.Now().Add(tc.expires).UnixMilli()}, nil
|
||||
}
|
||||
task.refreshCert(check)
|
||||
time.Sleep(tc.interval - time.Second)
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 1, calls)
|
||||
time.Sleep(time.Second)
|
||||
task.refreshCert(check)
|
||||
assert.Equal(t, 2, calls)
|
||||
})
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCertCheckEnabled(t *testing.T) {
|
||||
tests := []struct {
|
||||
protocol, target string
|
||||
want bool
|
||||
}{
|
||||
{"http", "https://example.com", true},
|
||||
{"http", "HTTPS://example.com:8443/path", true},
|
||||
{"http", "http://example.com", false},
|
||||
{"http", "https://", false},
|
||||
{"tcp", "https://example.com", false},
|
||||
{"icmp", "example.com", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
assert.Equal(t, tt.want, certCheckEnabled(monitor.Config{Protocol: tt.protocol, Target: tt.target}), tt.protocol+" "+tt.target)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRefreshCertSkipsNonHTTPS(t *testing.T) {
|
||||
task := newMonitorTask(monitor.Config{ID: "test", Target: "http://example.test", Protocol: "http"})
|
||||
defer task.cancel()
|
||||
task.refreshCert(func(context.Context, string) (monitor.CertInfo, error) {
|
||||
t.Fatal("certificate check must not run for non-https targets")
|
||||
return monitor.CertInfo{}, nil
|
||||
})
|
||||
assert.Nil(t, task.certInfo())
|
||||
}
|
||||
|
||||
func TestUpsertMonitorRunNowIncludesCert(t *testing.T) {
|
||||
server := httptest.NewTLSServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {}))
|
||||
defer server.Close()
|
||||
|
||||
pm := newMonitorManagerWithProbe(func(context.Context, monitor.Config) (int64, error) { return 100, nil })
|
||||
defer pm.Stop()
|
||||
config := monitor.Config{ID: "cert", Target: server.URL, Protocol: "http", Interval: 60}
|
||||
result, err := pm.UpsertMonitor(config, true)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
require.NotNil(t, result.Cert)
|
||||
assert.Equal(t, server.Certificate().NotAfter.UnixMilli(), result.Cert.Expires)
|
||||
|
||||
// Realtime results never carry the certificate, and the default interval
|
||||
// sends it only once per check.
|
||||
assert.Nil(t, pm.GetResults(1000)["cert"].Cert)
|
||||
results := pm.GetResults(defaultDataCacheTimeMs)
|
||||
require.NotNil(t, results["cert"].Cert)
|
||||
assert.Equal(t, result.Cert.Expires, results["cert"].Cert.Expires)
|
||||
assert.Nil(t, pm.GetResults(defaultDataCacheTimeMs)["cert"].Cert)
|
||||
|
||||
// Changing the interval keeps the known certificate without resending it.
|
||||
config.Interval = 30
|
||||
_, err = pm.UpsertMonitor(config, false)
|
||||
require.NoError(t, err)
|
||||
pm.mu.RLock()
|
||||
task := pm.monitors["cert"]
|
||||
pm.mu.RUnlock()
|
||||
assert.NotNil(t, task.certInfo())
|
||||
assert.Nil(t, pm.GetResults(defaultDataCacheTimeMs)["cert"].Cert)
|
||||
}
|
||||
274
agent/network_monitor_history.go
Normal file
274
agent/network_monitor_history.go
Normal file
@@ -0,0 +1,274 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"math"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
)
|
||||
|
||||
// Monitors run at user-defined intervals (e.g., every 10s).
|
||||
// To keep memory usage low and constant, data is stored in two layers:
|
||||
// 1. Raw samples: The most recent individual results (kept for monitorRawRetention).
|
||||
// 2. Minute buckets: A ring buffer of 61 buckets, each representing one
|
||||
// wall-clock minute. Samples collected within the same minute are aggregated
|
||||
// (sum, min, max, count) into a single bucket.
|
||||
//
|
||||
// Short-term requests (<= 61s) use raw samples.
|
||||
// Long-term requests (up to 1h) use the minute buckets to avoid storing thousands
|
||||
// of individual data points.
|
||||
|
||||
const (
|
||||
// monitorRawRetention is the duration to keep individual samples
|
||||
monitorRawRetention = 61 * time.Second
|
||||
// monitorMinuteBucketLen is the number of 1-minute buckets to keep (1 hour + 1 for partials)
|
||||
monitorMinuteBucketLen int32 = 61
|
||||
)
|
||||
|
||||
// monitorHistory owns retention and aggregation, independently of probe execution.
|
||||
type monitorHistory struct {
|
||||
mu sync.Mutex
|
||||
sampleCount int64
|
||||
samples []monitorSample
|
||||
buckets [monitorMinuteBucketLen]monitorBucket
|
||||
}
|
||||
|
||||
func newMonitorHistory() *monitorHistory {
|
||||
// Start small for typical intervals; append grows the buffer for faster probes.
|
||||
return &monitorHistory{samples: make([]monitorSample, 0, 4)}
|
||||
}
|
||||
|
||||
func (h *monitorHistory) clone() *monitorHistory {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
cloned := newMonitorHistory()
|
||||
cloned.samples = append(cloned.samples, h.samples...)
|
||||
cloned.buckets = h.buckets
|
||||
cloned.sampleCount = h.sampleCount
|
||||
return cloned
|
||||
}
|
||||
|
||||
func (h *monitorHistory) result(duration time.Duration, now time.Time) (monitor.Result, bool) {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
return h.resultLocked(duration, now)
|
||||
}
|
||||
|
||||
func (h *monitorHistory) record(sample monitorSample) monitor.Result {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
h.addSampleLocked(sample)
|
||||
result, _ := h.resultLocked(time.Minute, sample.timestamp)
|
||||
return result
|
||||
}
|
||||
|
||||
// monitorSample stores one monitor attempt and its collection time.
|
||||
type monitorSample struct {
|
||||
responseUs int64 // -1 means loss
|
||||
timestamp time.Time
|
||||
}
|
||||
|
||||
// monitorBucket stores one minute of aggregated monitor data.
|
||||
type monitorBucket struct {
|
||||
minute int32
|
||||
filled bool
|
||||
stats monitorAggregate
|
||||
}
|
||||
|
||||
// monitorAggregate accumulates successful response stats and total sample counts.
|
||||
type monitorAggregate struct {
|
||||
sumUs int64
|
||||
minUs int64
|
||||
maxUs int64
|
||||
totalCount int64
|
||||
successCount int64
|
||||
}
|
||||
|
||||
// newMonitorAggregate initializes an aggregate with an unset minimum value.
|
||||
func newMonitorAggregate() monitorAggregate {
|
||||
return monitorAggregate{minUs: math.MaxInt64}
|
||||
}
|
||||
|
||||
// addResponse folds a single monitor sample into the aggregate.
|
||||
func (agg *monitorAggregate) addResponse(responseUs int64) {
|
||||
agg.totalCount++
|
||||
if responseUs < 0 {
|
||||
return
|
||||
}
|
||||
agg.successCount++
|
||||
agg.sumUs += responseUs
|
||||
if responseUs < agg.minUs {
|
||||
agg.minUs = responseUs
|
||||
}
|
||||
if responseUs > agg.maxUs {
|
||||
agg.maxUs = responseUs
|
||||
}
|
||||
}
|
||||
|
||||
// addAggregate merges another aggregate into this one.
|
||||
func (agg *monitorAggregate) addAggregate(other monitorAggregate) {
|
||||
if other.totalCount == 0 {
|
||||
return
|
||||
}
|
||||
agg.totalCount += other.totalCount
|
||||
agg.successCount += other.successCount
|
||||
agg.sumUs += other.sumUs
|
||||
if other.successCount == 0 {
|
||||
return
|
||||
}
|
||||
if agg.minUs == math.MaxInt64 || other.minUs < agg.minUs {
|
||||
agg.minUs = other.minUs
|
||||
}
|
||||
if other.maxUs > agg.maxUs {
|
||||
agg.maxUs = other.maxUs
|
||||
}
|
||||
}
|
||||
|
||||
// hasData reports whether the aggregate contains any samples.
|
||||
func (agg monitorAggregate) hasData() bool {
|
||||
return agg.totalCount > 0
|
||||
}
|
||||
|
||||
// result converts the aggregate into the monitor result format.
|
||||
func (agg monitorAggregate) result() monitor.Result {
|
||||
avg := agg.avgResponse()
|
||||
result := monitor.Result{
|
||||
AvgResponse: avg,
|
||||
MinResponse: agg.minUs,
|
||||
MaxResponse: agg.maxUs,
|
||||
PacketLoss: agg.lossPercentage(),
|
||||
TotalCount: agg.totalCount,
|
||||
SuccessCount: agg.successCount,
|
||||
ResponseSum: agg.sumUs,
|
||||
}
|
||||
if agg.successCount == 0 {
|
||||
result.MinResponse, result.MaxResponse = 0, 0
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// avgResponse returns the rounded average of successful samples.
|
||||
func (agg monitorAggregate) avgResponse() int64 {
|
||||
if agg.successCount == 0 {
|
||||
return 0
|
||||
}
|
||||
return agg.sumUs / agg.successCount
|
||||
|
||||
}
|
||||
|
||||
// lossPercentage returns the rounded failure rate for the aggregate.
|
||||
func (agg monitorAggregate) lossPercentage() float64 {
|
||||
if agg.totalCount == 0 {
|
||||
return 0
|
||||
}
|
||||
return math.Round(float64(agg.totalCount-agg.successCount)/float64(agg.totalCount)*10000) / 100
|
||||
}
|
||||
|
||||
// resultLocked returns the aggregated monitor result for the requested duration along with a bool indicating whether any data was available.
|
||||
func (h *monitorHistory) resultLocked(duration time.Duration, now time.Time) (monitor.Result, bool) {
|
||||
agg := h.aggregateLocked(duration, now)
|
||||
if !agg.hasData() {
|
||||
// short realtime windows (e.g. the 1s window used for 1m/realtime charts) often fall
|
||||
// between monitor samples since monitors run at longer, user-defined intervals; fall back to
|
||||
// the most recent sample so realtime requests still report current status.
|
||||
agg = h.latestSampleAggregateLocked()
|
||||
}
|
||||
hourAgg := h.aggregateLocked(time.Hour, now)
|
||||
if !agg.hasData() {
|
||||
return monitor.Result{}, false
|
||||
}
|
||||
|
||||
result := agg.result()
|
||||
if len(h.samples) > 0 {
|
||||
result.LastProbeAt = h.samples[len(h.samples)-1].timestamp.UnixMilli()
|
||||
}
|
||||
|
||||
result.AvgResponse1h = hourAgg.avgResponse()
|
||||
result.MinResponse1h = hourAgg.minUs
|
||||
result.MaxResponse1h = hourAgg.maxUs
|
||||
result.PacketLoss1h = hourAgg.lossPercentage()
|
||||
result.SampleCount = h.sampleCount
|
||||
|
||||
if hourAgg.successCount == 0 {
|
||||
result.MinResponse1h, result.MaxResponse1h = 0, 0
|
||||
}
|
||||
return result, true
|
||||
}
|
||||
|
||||
// latestSampleAggregateLocked returns an aggregate containing only the most recent sample, if any.
|
||||
func (h *monitorHistory) latestSampleAggregateLocked() monitorAggregate {
|
||||
agg := newMonitorAggregate()
|
||||
if len(h.samples) == 0 {
|
||||
return agg
|
||||
}
|
||||
agg.addResponse(h.samples[len(h.samples)-1].responseUs)
|
||||
return agg
|
||||
}
|
||||
|
||||
// aggregateLocked collects monitor data for the requested time window.
|
||||
func (h *monitorHistory) aggregateLocked(duration time.Duration, now time.Time) monitorAggregate {
|
||||
cutoff := now.Add(-duration)
|
||||
// Keep short windows exact; longer windows read from minute buckets to avoid raw-sample retention.
|
||||
if duration <= monitorRawRetention {
|
||||
return aggregateSamplesSince(h.samples, cutoff)
|
||||
}
|
||||
return aggregateBucketsSince(h.buckets[:], cutoff, now)
|
||||
}
|
||||
|
||||
// aggregateSamplesSince aggregates raw samples newer than the cutoff.
|
||||
func aggregateSamplesSince(samples []monitorSample, cutoff time.Time) monitorAggregate {
|
||||
agg := newMonitorAggregate()
|
||||
for _, sample := range samples {
|
||||
if sample.timestamp.Before(cutoff) {
|
||||
continue
|
||||
}
|
||||
agg.addResponse(sample.responseUs)
|
||||
}
|
||||
return agg
|
||||
}
|
||||
|
||||
// aggregateBucketsSince aggregates minute buckets overlapping the requested window.
|
||||
func aggregateBucketsSince(buckets []monitorBucket, cutoff, now time.Time) monitorAggregate {
|
||||
agg := newMonitorAggregate()
|
||||
startMinute := int32(cutoff.Unix() / 60)
|
||||
endMinute := int32(now.Unix() / 60)
|
||||
for _, bucket := range buckets {
|
||||
if !bucket.filled || bucket.minute < startMinute || bucket.minute > endMinute {
|
||||
continue
|
||||
}
|
||||
agg.addAggregate(bucket.stats)
|
||||
}
|
||||
return agg
|
||||
}
|
||||
|
||||
// addSampleLocked stores a fresh sample in both raw and per-minute retention buffers.
|
||||
func (h *monitorHistory) addSampleLocked(sample monitorSample) {
|
||||
h.sampleCount++
|
||||
cutoff := sample.timestamp.Add(-monitorRawRetention)
|
||||
start := 0
|
||||
for i := range h.samples {
|
||||
if !h.samples[i].timestamp.Before(cutoff) {
|
||||
start = i
|
||||
break
|
||||
}
|
||||
if i == len(h.samples)-1 {
|
||||
start = len(h.samples)
|
||||
}
|
||||
}
|
||||
if start > 0 {
|
||||
size := copy(h.samples, h.samples[start:])
|
||||
h.samples = h.samples[:size]
|
||||
}
|
||||
h.samples = append(h.samples, sample)
|
||||
|
||||
minute := int32(sample.timestamp.Unix() / 60)
|
||||
// Each slot stores one wall-clock minute, so the ring stays fixed-size at ~1h per monitor.
|
||||
bucket := &h.buckets[minute%monitorMinuteBucketLen]
|
||||
if !bucket.filled || bucket.minute != minute {
|
||||
bucket.minute = minute
|
||||
bucket.filled = true
|
||||
bucket.stats = newMonitorAggregate()
|
||||
}
|
||||
bucket.stats.addResponse(sample.responseUs)
|
||||
}
|
||||
154
agent/network_monitor_history_test.go
Normal file
154
agent/network_monitor_history_test.go
Normal file
@@ -0,0 +1,154 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/fxamacker/cbor/v2"
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestMonitorHistoryWindowCounts(t *testing.T) {
|
||||
history := newMonitorHistory()
|
||||
now := time.Now()
|
||||
// This older success counts toward lifetime warm-up, but not this window.
|
||||
history.record(monitorSample{responseUs: 1000, timestamp: now.Add(-2 * time.Minute)})
|
||||
history.record(monitorSample{responseUs: 10, timestamp: now.Add(-30 * time.Second)})
|
||||
history.record(monitorSample{responseUs: 21, timestamp: now.Add(-20 * time.Second)})
|
||||
history.record(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
|
||||
result, ok := history.result(time.Minute, now)
|
||||
require.True(t, ok)
|
||||
assert.EqualValues(t, 4, result.SampleCount)
|
||||
assert.EqualValues(t, 3, result.TotalCount)
|
||||
assert.EqualValues(t, 2, result.SuccessCount)
|
||||
assert.EqualValues(t, 31, result.ResponseSum, "preserve the sum before average rounding")
|
||||
assert.EqualValues(t, 15, result.AvgResponse)
|
||||
assert.Equal(t, 33.33, result.PacketLoss)
|
||||
|
||||
encoded, err := cbor.Marshal(result)
|
||||
require.NoError(t, err)
|
||||
var decoded monitor.Result
|
||||
require.NoError(t, cbor.Unmarshal(encoded, &decoded))
|
||||
assert.Equal(t, result, decoded)
|
||||
stats := monitor.Stats{}.FromResult(decoded)
|
||||
assert.Equal(t, result.TotalCount, stats.TotalCount)
|
||||
assert.Equal(t, result.SuccessCount, stats.SuccessCount)
|
||||
assert.Equal(t, result.ResponseSum, stats.ResponseSum)
|
||||
|
||||
// Reads do not consume samples. A short window's latest-sample fallback
|
||||
// carries the count for that single failure, not the minute or lifetime count.
|
||||
repeated, _ := history.result(time.Minute, now)
|
||||
assert.Equal(t, result, repeated)
|
||||
fallback, ok := history.result(time.Second, now)
|
||||
require.True(t, ok)
|
||||
assert.EqualValues(t, 1, fallback.TotalCount)
|
||||
assert.Zero(t, fallback.SuccessCount)
|
||||
assert.Zero(t, fallback.ResponseSum)
|
||||
assert.Equal(t, 100.0, fallback.PacketLoss)
|
||||
assert.EqualValues(t, 4, fallback.SampleCount)
|
||||
}
|
||||
|
||||
func TestMonitorHistoryAggregateLockedUsesRawSamplesForShortWindows(t *testing.T) {
|
||||
now := time.Date(2026, time.April, 21, 12, 0, 0, 0, time.UTC)
|
||||
history := newMonitorHistory()
|
||||
|
||||
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-90 * time.Second)})
|
||||
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-30 * time.Second)})
|
||||
history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
|
||||
|
||||
agg := history.aggregateLocked(time.Minute, now)
|
||||
require.True(t, agg.hasData())
|
||||
assert.Equal(t, int64(2), agg.totalCount)
|
||||
assert.Equal(t, int64(1), agg.successCount)
|
||||
result := agg.result()
|
||||
assert.Equal(t, int64(20), result.AvgResponse)
|
||||
assert.Equal(t, int64(20), result.MinResponse)
|
||||
assert.Equal(t, int64(20), result.MaxResponse)
|
||||
assert.Equal(t, 50.0, result.PacketLoss)
|
||||
}
|
||||
|
||||
func TestMonitorHistoryAggregateLockedUsesMinuteBucketsForLongWindows(t *testing.T) {
|
||||
now := time.Date(2026, time.April, 21, 12, 0, 30, 0, time.UTC)
|
||||
history := newMonitorHistory()
|
||||
|
||||
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-11 * time.Minute)})
|
||||
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-9 * time.Minute)})
|
||||
history.addSampleLocked(monitorSample{responseUs: 40, timestamp: now.Add(-5 * time.Minute)})
|
||||
history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-90 * time.Second)})
|
||||
history.addSampleLocked(monitorSample{responseUs: 30, timestamp: now.Add(-30 * time.Second)})
|
||||
|
||||
agg := history.aggregateLocked(10*time.Minute, now)
|
||||
require.True(t, agg.hasData())
|
||||
assert.Equal(t, int64(4), agg.totalCount)
|
||||
assert.Equal(t, int64(3), agg.successCount)
|
||||
result := agg.result()
|
||||
assert.Equal(t, int64(30), result.AvgResponse)
|
||||
assert.Equal(t, int64(20), result.MinResponse)
|
||||
assert.Equal(t, int64(40), result.MaxResponse)
|
||||
assert.Equal(t, 25.0, result.PacketLoss)
|
||||
}
|
||||
|
||||
func TestMonitorHistoryAddSampleLockedTrimsRawSamplesButKeepsBucketHistory(t *testing.T) {
|
||||
now := time.Date(2026, time.April, 21, 12, 0, 0, 0, time.UTC)
|
||||
history := newMonitorHistory()
|
||||
|
||||
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-10 * time.Minute)})
|
||||
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now})
|
||||
|
||||
require.Len(t, history.samples, 1)
|
||||
assert.Equal(t, int64(20), history.samples[0].responseUs)
|
||||
|
||||
agg := history.aggregateLocked(10*time.Minute, now)
|
||||
require.True(t, agg.hasData())
|
||||
assert.Equal(t, int64(2), agg.totalCount)
|
||||
assert.Equal(t, int64(2), agg.successCount)
|
||||
result := agg.result()
|
||||
assert.Equal(t, int64(15), result.AvgResponse)
|
||||
assert.Equal(t, int64(10), result.MinResponse)
|
||||
assert.Equal(t, int64(20), result.MaxResponse)
|
||||
assert.Equal(t, 0.0, result.PacketLoss)
|
||||
}
|
||||
|
||||
func TestMonitorHistoryProbeTimestamp(t *testing.T) {
|
||||
history := newMonitorHistory()
|
||||
start := time.Date(2026, time.September, 14, 12, 0, 0, 0, time.UTC)
|
||||
_, ok := history.result(time.Minute, start)
|
||||
require.False(t, ok)
|
||||
first := history.record(monitorSample{responseUs: 20, timestamp: start})
|
||||
assert.Equal(t, start.UnixMilli(), first.LastProbeAt)
|
||||
for minute := 0; minute < 5; minute++ {
|
||||
now := start.Add(time.Duration(minute)*time.Minute + time.Second)
|
||||
// Realtime reads must not consume freshness for the persistence request.
|
||||
for _, window := range []time.Duration{time.Second, time.Minute} {
|
||||
result, ok := history.result(window, now)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, first.LastProbeAt, result.LastProbeAt)
|
||||
assert.Equal(t, int64(20), result.AvgResponse)
|
||||
}
|
||||
}
|
||||
next := start.Add(5 * time.Minute)
|
||||
failed := history.record(monitorSample{responseUs: -1, timestamp: next})
|
||||
assert.Equal(t, next.UnixMilli(), failed.LastProbeAt)
|
||||
assert.Equal(t, float64(100), failed.PacketLoss)
|
||||
repeated, ok := history.result(time.Minute, next.Add(2*time.Minute))
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, failed.LastProbeAt, repeated.LastProbeAt)
|
||||
assert.Equal(t, float64(100), repeated.PacketLoss)
|
||||
}
|
||||
|
||||
func TestMonitorHistorySampleCount(t *testing.T) {
|
||||
history := newMonitorHistory()
|
||||
now := time.Now()
|
||||
// Both failed and successful probes count, including older samples so
|
||||
// monitors with hourly intervals can finish warming up.
|
||||
history.record(monitorSample{responseUs: -1, timestamp: now.Add(-2 * time.Hour)})
|
||||
for i, response := range []int64{10, -1, 20} {
|
||||
result := history.record(monitorSample{responseUs: response, timestamp: now.Add(time.Duration(i) * time.Second)})
|
||||
assert.EqualValues(t, i+2, result.SampleCount)
|
||||
}
|
||||
result, ok := history.clone().result(time.Minute, now.Add(3*time.Second))
|
||||
require.True(t, ok)
|
||||
assert.EqualValues(t, 4, result.SampleCount)
|
||||
}
|
||||
312
agent/network_monitor_ping.go
Normal file
312
agent/network_monitor_ping.go
Normal file
@@ -0,0 +1,312 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"net"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"golang.org/x/net/icmp"
|
||||
"golang.org/x/net/ipv4"
|
||||
"golang.org/x/net/ipv6"
|
||||
|
||||
"log/slog"
|
||||
)
|
||||
|
||||
// Match the numeric RTT independently of the localized label used by Windows.
|
||||
var pingTimeRegex = regexp.MustCompile(`(?i)[=<]\s*([0-9]+(?:[.,][0-9]+)?)\s*ms\b`)
|
||||
|
||||
var icmpSequence atomic.Uint32
|
||||
|
||||
type icmpPacketConn interface {
|
||||
Close() error
|
||||
}
|
||||
|
||||
// icmpMethod tracks which ICMP approach to use. Once a method succeeds or
|
||||
// all native methods fail, the choice is cached so subsequent monitors skip
|
||||
// the trial-and-error overhead.
|
||||
type icmpMethod uint8
|
||||
|
||||
const (
|
||||
icmpUntried icmpMethod = iota // haven't tried yet
|
||||
icmpRaw // privileged raw socket
|
||||
icmpDatagram // unprivileged datagram socket
|
||||
icmpExecFallback // shell out to system ping command
|
||||
)
|
||||
|
||||
// icmpFamily holds the network parameters and cached detection result for one address family.
|
||||
type icmpFamily struct {
|
||||
rawNetwork string // e.g. "ip4:icmp" or "ip6:ipv6-icmp"
|
||||
dgramNetwork string // e.g. "udp4" or "udp6"
|
||||
listenAddr string // "0.0.0.0" or "::"
|
||||
echoType icmp.Type // outgoing echo request type
|
||||
replyType icmp.Type // expected echo reply type
|
||||
proto int // IANA protocol number for parsing replies
|
||||
isIPv6 bool
|
||||
mode icmpMethod // cached detection result (guarded by icmpModeMu)
|
||||
}
|
||||
|
||||
var (
|
||||
icmpV4 = icmpFamily{
|
||||
rawNetwork: "ip4:icmp",
|
||||
dgramNetwork: "udp4",
|
||||
listenAddr: "0.0.0.0",
|
||||
echoType: ipv4.ICMPTypeEcho,
|
||||
replyType: ipv4.ICMPTypeEchoReply,
|
||||
proto: 1,
|
||||
}
|
||||
icmpV6 = icmpFamily{
|
||||
rawNetwork: "ip6:ipv6-icmp",
|
||||
dgramNetwork: "udp6",
|
||||
listenAddr: "::",
|
||||
echoType: ipv6.ICMPTypeEchoRequest,
|
||||
replyType: ipv6.ICMPTypeEchoReply,
|
||||
proto: 58,
|
||||
isIPv6: true,
|
||||
}
|
||||
icmpModeMu sync.Mutex
|
||||
icmpListen = func(network, listenAddr string) (icmpPacketConn, error) {
|
||||
return icmp.ListenPacket(network, listenAddr)
|
||||
}
|
||||
)
|
||||
|
||||
// monitorICMP sends an ICMP echo request and measures round-trip response.
|
||||
// Supports both IPv4 and IPv6 targets. The ICMP method (raw socket,
|
||||
// unprivileged datagram, or exec fallback) is detected once per address
|
||||
// family and cached for subsequent monitors.
|
||||
// Returns response in microseconds, or -1 and an error on failure.
|
||||
func monitorICMP(ctx context.Context, target string) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
defer cancel()
|
||||
|
||||
family, ip, err := resolveICMPTarget(ctx, target)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
|
||||
icmpModeMu.Lock()
|
||||
if family.mode == icmpUntried {
|
||||
family.mode = detectICMPMode(family, icmpListen)
|
||||
}
|
||||
mode := family.mode
|
||||
icmpModeMu.Unlock()
|
||||
|
||||
switch mode {
|
||||
case icmpRaw:
|
||||
return monitorICMPNative(ctx, family.rawNetwork, family, &net.IPAddr{IP: ip})
|
||||
case icmpDatagram:
|
||||
return monitorICMPNative(ctx, family.dgramNetwork, family, &net.UDPAddr{IP: ip})
|
||||
case icmpExecFallback:
|
||||
return monitorICMPExec(ctx, ip.String(), family.isIPv6)
|
||||
default:
|
||||
return -1, errors.New("unsupported ICMP mode")
|
||||
}
|
||||
}
|
||||
|
||||
// resolveICMPTarget resolves a target hostname or IP to determine the address
|
||||
// family and concrete IP address. Prefers IPv4 for dual-stack hostnames.
|
||||
func resolveICMPTarget(ctx context.Context, target string) (*icmpFamily, net.IP, error) {
|
||||
if ip := net.ParseIP(target); ip != nil {
|
||||
if ip.To4() != nil {
|
||||
return &icmpV4, ip.To4(), nil
|
||||
}
|
||||
return &icmpV6, ip, nil
|
||||
}
|
||||
|
||||
ips, err := net.DefaultResolver.LookupIP(ctx, "ip", target)
|
||||
if err != nil || len(ips) == 0 {
|
||||
return nil, nil, err
|
||||
}
|
||||
for _, ip := range ips {
|
||||
if v4 := ip.To4(); v4 != nil {
|
||||
return &icmpV4, v4, nil
|
||||
}
|
||||
}
|
||||
return &icmpV6, ips[0], nil
|
||||
}
|
||||
|
||||
func detectICMPMode(family *icmpFamily, listen func(network, listenAddr string) (icmpPacketConn, error)) icmpMethod {
|
||||
label := "IPv4"
|
||||
if family.isIPv6 {
|
||||
label = "IPv6"
|
||||
}
|
||||
|
||||
conn, err := listen(family.rawNetwork, family.listenAddr)
|
||||
slog.Debug("ICMP raw socket test", "family", label, "err", err)
|
||||
if err == nil {
|
||||
conn.Close()
|
||||
return icmpRaw
|
||||
}
|
||||
|
||||
conn, err = listen(family.dgramNetwork, family.listenAddr)
|
||||
slog.Debug("ICMP datagram socket test", "family", label, "err", err)
|
||||
if err == nil {
|
||||
conn.Close()
|
||||
return icmpDatagram
|
||||
}
|
||||
|
||||
return icmpExecFallback
|
||||
}
|
||||
|
||||
// monitorICMPNative sends an ICMP echo request using Go's x/net/icmp package.
|
||||
func monitorICMPNative(ctx context.Context, network string, family *icmpFamily, dst net.Addr) (int64, error) {
|
||||
conn, err := icmp.ListenPacket(network, family.listenAddr)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
return monitorICMPPacket(ctx, conn, family, dst)
|
||||
}
|
||||
|
||||
func monitorICMPPacket(ctx context.Context, conn net.PacketConn, family *icmpFamily, dst net.Addr) (int64, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return -1, err
|
||||
}
|
||||
// Closing the socket interrupts both reads and writes on cancellation.
|
||||
stop := context.AfterFunc(ctx, func() { _ = conn.Close() })
|
||||
defer stop()
|
||||
|
||||
// Prepare correlation data before starting the round-trip timer. The token
|
||||
// also distinguishes delayed replies after the 16-bit sequence wraps.
|
||||
token := make([]byte, 16)
|
||||
if _, err := rand.Read(token); err != nil {
|
||||
return -1, err
|
||||
}
|
||||
echo := &icmp.Echo{
|
||||
ID: os.Getpid() & 0xffff,
|
||||
Seq: int(icmpSequence.Add(1) & 0xffff),
|
||||
Data: token,
|
||||
}
|
||||
// Linux ping sockets replace the Echo ID with their bound port. Darwin
|
||||
// datagram sockets and raw sockets preserve the supplied ID.
|
||||
if local, ok := conn.LocalAddr().(*net.UDPAddr); ok && runtime.GOOS == "linux" {
|
||||
echo.ID = local.Port
|
||||
}
|
||||
targetIP := icmpAddrIP(dst)
|
||||
msg := &icmp.Message{
|
||||
Type: family.echoType,
|
||||
Code: 0,
|
||||
Body: echo,
|
||||
}
|
||||
msgBytes, err := msg.Marshal(nil)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
|
||||
// Set deadline before sending
|
||||
if err := conn.SetDeadline(time.Now().Add(3 * time.Second)); err != nil {
|
||||
return -1, err
|
||||
}
|
||||
|
||||
buf := make([]byte, 1500)
|
||||
start := time.Now()
|
||||
if _, err := conn.WriteTo(msgBytes, dst); err != nil {
|
||||
return -1, err
|
||||
}
|
||||
|
||||
// Read reply
|
||||
for {
|
||||
n, peer, err := conn.ReadFrom(buf)
|
||||
received := time.Now()
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
if !targetIP.Equal(icmpAddrIP(peer)) {
|
||||
continue
|
||||
}
|
||||
|
||||
reply, err := icmp.ParseMessage(family.proto, buf[:n])
|
||||
if err != nil || reply.Type != family.replyType || reply.Code != 0 {
|
||||
continue
|
||||
}
|
||||
|
||||
body, ok := reply.Body.(*icmp.Echo)
|
||||
if ok && body.ID == echo.ID && body.Seq == echo.Seq && bytes.Equal(body.Data, echo.Data) {
|
||||
return received.Sub(start).Microseconds(), nil
|
||||
}
|
||||
// Keep waiting for our reply without extending the original deadline.
|
||||
}
|
||||
}
|
||||
|
||||
func icmpAddrIP(addr net.Addr) net.IP {
|
||||
switch addr := addr.(type) {
|
||||
case *net.IPAddr:
|
||||
return addr.IP
|
||||
case *net.UDPAddr:
|
||||
return addr.IP
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// pingCommand selects the executable and arguments for the supported agent platforms.
|
||||
// The context deadline enforces the timeout: -W has incompatible meanings across
|
||||
// Linux, BSD IPv4 ping, and macOS ping6.
|
||||
func pingCommand(goos, target string, isIPv6 bool) (string, []string, error) {
|
||||
family := "-4"
|
||||
if isIPv6 {
|
||||
family = "-6"
|
||||
}
|
||||
switch goos {
|
||||
case "windows":
|
||||
return "ping", []string{family, "-n", "1", "-w", "3000", target}, nil
|
||||
case "linux":
|
||||
return "ping", []string{family, "-n", "-c", "1", target}, nil
|
||||
case "darwin", "freebsd", "openbsd":
|
||||
command := "ping"
|
||||
if isIPv6 {
|
||||
command = "ping6"
|
||||
}
|
||||
return command, []string{"-n", "-c", "1", target}, nil
|
||||
default:
|
||||
return "", nil, fmt.Errorf("ping fallback is unsupported on %s", goos)
|
||||
}
|
||||
}
|
||||
|
||||
// monitorICMPExec falls back to the system ping command. Returns -1 and an error on failure.
|
||||
func monitorICMPExec(ctx context.Context, target string, isIPv6 bool) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
defer cancel()
|
||||
name, args, err := pingCommand(runtime.GOOS, target, isIPv6)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, name, args...)
|
||||
// Keep Unix output and decimal formatting stable. Windows ignores LC_ALL.
|
||||
cmd.Env = append(os.Environ(), "LC_ALL=C")
|
||||
output, err := cmd.Output()
|
||||
if ctx.Err() != nil {
|
||||
return -1, ctx.Err()
|
||||
}
|
||||
if err != nil {
|
||||
return -1, fmt.Errorf("%s failed: %w", name, err)
|
||||
}
|
||||
return parsePingResponse(output)
|
||||
}
|
||||
|
||||
// parsePingResponse returns the reported RTT, never subprocess execution time.
|
||||
// For a bounded value such as Windows' time<1ms, retain the reported upper bound.
|
||||
func parsePingResponse(output []byte) (int64, error) {
|
||||
matches := pingTimeRegex.FindSubmatch(output)
|
||||
if len(matches) < 2 {
|
||||
return -1, errors.New("ping output contains no round-trip time")
|
||||
}
|
||||
ms, err := strconv.ParseFloat(strings.ReplaceAll(string(matches[1]), ",", "."), 64)
|
||||
if err != nil || math.IsInf(ms, 0) || ms >= float64(math.MaxInt64)/1000 {
|
||||
return -1, errors.New("invalid round-trip time in ping output")
|
||||
}
|
||||
return int64(math.Round(ms * 1000)), nil
|
||||
}
|
||||
433
agent/network_monitor_ping_test.go
Normal file
433
agent/network_monitor_ping_test.go
Normal file
@@ -0,0 +1,433 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/net/icmp"
|
||||
)
|
||||
|
||||
type testICMPPacketConn struct{}
|
||||
|
||||
func (testICMPPacketConn) Close() error { return nil }
|
||||
|
||||
type blockingICMPConn struct {
|
||||
net.PacketConn
|
||||
reading chan struct{}
|
||||
}
|
||||
|
||||
func (c *blockingICMPConn) WriteTo(p []byte, addr net.Addr) (int, error) {
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func (c *blockingICMPConn) ReadFrom(p []byte) (int, net.Addr, error) {
|
||||
close(c.reading)
|
||||
return c.PacketConn.ReadFrom(p)
|
||||
}
|
||||
|
||||
func TestMonitorICMPPacketCancellation(t *testing.T) {
|
||||
conn, err := net.ListenPacket("udp4", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
defer conn.Close()
|
||||
blocking := &blockingICMPConn{PacketConn: conn, reading: make(chan struct{})}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := monitorICMPPacket(ctx, blocking, &icmpV4, conn.LocalAddr())
|
||||
done <- err
|
||||
}()
|
||||
select {
|
||||
case <-blocking.reading:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("probe did not begin reading")
|
||||
}
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
require.Error(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("cancellation did not interrupt the socket read")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorICMPExecCancellation(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("test uses a POSIX shell stub for ping")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "ping"), []byte("#!/bin/sh\nexec sleep 30\n"), 0o755))
|
||||
t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH"))
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond)
|
||||
defer cancel()
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := monitorICMPExec(ctx, "127.0.0.1", false)
|
||||
done <- err
|
||||
}()
|
||||
select {
|
||||
case err := <-done:
|
||||
require.ErrorIs(t, err, context.DeadlineExceeded)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("cancellation did not terminate ping")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPingCommand(t *testing.T) {
|
||||
for _, goos := range []string{"linux", "windows", "darwin", "freebsd", "openbsd"} {
|
||||
for _, ipv6 := range []bool{false, true} {
|
||||
t.Run(fmt.Sprintf("%s/ipv6=%t", goos, ipv6), func(t *testing.T) {
|
||||
target, family := "192.0.2.1", "-4"
|
||||
if ipv6 {
|
||||
target, family = "2001:db8::1", "-6"
|
||||
}
|
||||
name, args, err := pingCommand(goos, target, ipv6)
|
||||
require.NoError(t, err)
|
||||
wantName := "ping"
|
||||
wantArgs := []string{"-n", "-c", "1", target}
|
||||
switch goos {
|
||||
case "windows":
|
||||
wantArgs = []string{family, "-n", "1", "-w", "3000", target}
|
||||
case "linux":
|
||||
wantArgs = append([]string{family}, wantArgs...)
|
||||
default:
|
||||
if ipv6 {
|
||||
wantName = "ping6"
|
||||
}
|
||||
}
|
||||
assert.Equal(t, wantName, name)
|
||||
assert.Equal(t, wantArgs, args)
|
||||
})
|
||||
}
|
||||
}
|
||||
_, _, err := pingCommand("unsupported", "192.0.2.1", false)
|
||||
require.Error(t, err)
|
||||
}
|
||||
|
||||
func TestParsePingResponse(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
output string
|
||||
wantUs int64
|
||||
}{
|
||||
{"linux", "64 bytes from 192.0.2.1: icmp_seq=1 ttl=64 time=12.345 ms", 12345},
|
||||
{"bsd", "64 bytes from 192.0.2.1: icmp_seq=0 ttl=64 time=0.023 ms", 23},
|
||||
{"ipv6", "64 bytes from 2001:db8::1: icmp_seq=0 hlim=64 time=1.234 ms", 1234},
|
||||
{"windows", "Reply from 192.0.2.1: bytes=32 time=12ms TTL=128", 12000},
|
||||
{"windows submillisecond", "Reply from ::1: time<1ms", 1000},
|
||||
{"localized windows", "Antwort von 192.0.2.1: Bytes=32 Zeit=12ms TTL=128", 12000},
|
||||
{"decimal comma", "64 bytes from 192.0.2.1: time=1,234 ms", 1234},
|
||||
{"rounding", "time=0.1236 ms", 124},
|
||||
{"empty", "", -1},
|
||||
{"timeout", "Request timed out.", -1},
|
||||
{"unreachable", "Reply from 192.0.2.1: Destination host unreachable.", -1},
|
||||
{"malformed", "time=oops ms", -1},
|
||||
{"negative", "time=-1 ms", -1},
|
||||
{"overflow", "time=999999999999999999999 ms", -1},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
responseUs, err := parsePingResponse([]byte(tc.output))
|
||||
if tc.wantUs < 0 {
|
||||
require.Error(t, err)
|
||||
} else {
|
||||
require.NoError(t, err)
|
||||
}
|
||||
assert.Equal(t, tc.wantUs, responseUs)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorICMPExecOutput(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("test uses a POSIX shell stub for ping")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
output string
|
||||
exit int
|
||||
wantUs int64
|
||||
}{
|
||||
{"success", "time=1.234 ms", 0, 1234},
|
||||
{"missing RTT", "unrecognized output", 0, -1},
|
||||
{"failed command with RTT", "time=1.234 ms", 1, -1},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
// Also verify an inherited locale cannot override the C locale.
|
||||
script := fmt.Sprintf("#!/bin/sh\n[ \"$LC_ALL\" = C ] || exit 2\nprintf '%%s\\n' '%s'\nexit %d\n", tc.output, tc.exit)
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "ping"), []byte(script), 0o755))
|
||||
t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH"))
|
||||
t.Setenv("LC_ALL", "de_DE.UTF-8")
|
||||
responseUs, err := monitorICMPExec(t.Context(), "127.0.0.1", false)
|
||||
if tc.wantUs < 0 {
|
||||
require.Error(t, err)
|
||||
} else {
|
||||
require.NoError(t, err)
|
||||
}
|
||||
assert.Equal(t, tc.wantUs, responseUs)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
type icmpTestReply struct {
|
||||
data []byte
|
||||
peer net.Addr
|
||||
}
|
||||
|
||||
type scriptedICMPConn struct {
|
||||
net.PacketConn
|
||||
local net.Addr
|
||||
onWrite func([]byte, net.Addr)
|
||||
replies []icmpTestReply
|
||||
reads int
|
||||
deadlineSets int
|
||||
}
|
||||
|
||||
func (c *scriptedICMPConn) LocalAddr() net.Addr { return c.local }
|
||||
|
||||
func (c *scriptedICMPConn) SetDeadline(deadline time.Time) error {
|
||||
c.deadlineSets++
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *scriptedICMPConn) WriteTo(data []byte, dst net.Addr) (int, error) {
|
||||
c.onWrite(data, dst)
|
||||
return len(data), nil
|
||||
}
|
||||
|
||||
func (c *scriptedICMPConn) ReadFrom(buf []byte) (int, net.Addr, error) {
|
||||
c.reads++
|
||||
if len(c.replies) == 0 {
|
||||
return 0, nil, os.ErrDeadlineExceeded
|
||||
}
|
||||
reply := c.replies[0]
|
||||
c.replies = c.replies[1:]
|
||||
return copy(buf, reply.data), reply.peer, nil
|
||||
}
|
||||
|
||||
func TestMonitorICMPReplyCorrelation(t *testing.T) {
|
||||
for _, family := range []*icmpFamily{&icmpV4, &icmpV6} {
|
||||
for _, datagram := range []bool{false, true} {
|
||||
network := family.rawNetwork
|
||||
ip, other := net.ParseIP("192.0.2.1"), net.ParseIP("192.0.2.2")
|
||||
if family.isIPv6 {
|
||||
ip, other = net.ParseIP("2001:db8::1"), net.ParseIP("2001:db8::2")
|
||||
}
|
||||
var dst net.Addr = &net.IPAddr{IP: ip}
|
||||
var wrongPeer net.Addr = &net.IPAddr{IP: other}
|
||||
if datagram {
|
||||
network = family.dgramNetwork
|
||||
dst = &net.UDPAddr{IP: ip}
|
||||
wrongPeer = &net.UDPAddr{IP: other}
|
||||
}
|
||||
for _, mismatch := range []string{"source", "id", "sequence", "payload", "type", "code", "malformed"} {
|
||||
for _, eventuallyMatches := range []bool{false, true} {
|
||||
ending := "timeout"
|
||||
if eventuallyMatches {
|
||||
ending = "success"
|
||||
}
|
||||
t.Run(network+"/"+mismatch+"/"+ending, func(t *testing.T) {
|
||||
conn := &scriptedICMPConn{local: &net.IPAddr{IP: net.IPv4zero}}
|
||||
if datagram {
|
||||
conn.local = &net.UDPAddr{Port: 12345}
|
||||
if runtime.GOOS == "linux" {
|
||||
// Deliberately differ from the process ID.
|
||||
conn.local = &net.UDPAddr{Port: (os.Getpid() % 65534) + 1}
|
||||
}
|
||||
}
|
||||
conn.onWrite = func(data []byte, target net.Addr) {
|
||||
require.Equal(t, dst, target)
|
||||
request, err := icmp.ParseMessage(family.proto, data)
|
||||
require.NoError(t, err)
|
||||
echo := request.Body.(*icmp.Echo)
|
||||
expectedID := os.Getpid() & 0xffff
|
||||
if datagram && runtime.GOOS == "linux" {
|
||||
expectedID = conn.local.(*net.UDPAddr).Port
|
||||
}
|
||||
require.Equal(t, expectedID, echo.ID)
|
||||
reply := &icmp.Message{Type: family.replyType, Body: echo}
|
||||
valid, err := reply.Marshal(nil)
|
||||
require.NoError(t, err)
|
||||
peer := dst
|
||||
switch mismatch {
|
||||
case "source":
|
||||
peer = wrongPeer
|
||||
case "id":
|
||||
echo.ID ^= 1
|
||||
case "sequence":
|
||||
echo.Seq ^= 1
|
||||
case "payload":
|
||||
echo.Data[0] ^= 1
|
||||
case "type":
|
||||
reply.Type = family.echoType
|
||||
case "code":
|
||||
reply.Code = 1
|
||||
}
|
||||
invalid, err := reply.Marshal(nil)
|
||||
require.NoError(t, err)
|
||||
if mismatch == "malformed" {
|
||||
invalid = invalid[:2]
|
||||
}
|
||||
conn.replies = []icmpTestReply{{invalid, peer}}
|
||||
if eventuallyMatches {
|
||||
conn.replies = append(conn.replies, icmpTestReply{valid, dst})
|
||||
}
|
||||
}
|
||||
elapsed, err := monitorICMPPacket(context.Background(), conn, family, dst)
|
||||
if eventuallyMatches {
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, elapsed, int64(0))
|
||||
} else {
|
||||
require.ErrorIs(t, err, os.ErrDeadlineExceeded)
|
||||
assert.Equal(t, int64(-1), elapsed)
|
||||
}
|
||||
assert.Equal(t, 2, conn.reads)
|
||||
assert.Equal(t, 1, conn.deadlineSets)
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorICMPLoopback(t *testing.T) {
|
||||
for _, family := range []*icmpFamily{&icmpV4, &icmpV6} {
|
||||
for _, network := range []string{family.rawNetwork, family.dgramNetwork} {
|
||||
t.Run(network, func(t *testing.T) {
|
||||
conn, err := icmp.ListenPacket(network, family.listenAddr)
|
||||
if err != nil {
|
||||
t.Skipf("ICMP socket unavailable: %v", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
ip := net.ParseIP("127.0.0.1")
|
||||
if family.isIPv6 {
|
||||
ip = net.ParseIP("::1")
|
||||
}
|
||||
var dst net.Addr = &net.IPAddr{IP: ip}
|
||||
if network == family.dgramNetwork {
|
||||
dst = &net.UDPAddr{IP: ip}
|
||||
}
|
||||
elapsed, err := monitorICMPPacket(context.Background(), conn, family, dst)
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, elapsed, int64(0))
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDetectICMPMode(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
family *icmpFamily
|
||||
rawErr error
|
||||
udpErr error
|
||||
want icmpMethod
|
||||
wantNetworks []string
|
||||
}{
|
||||
{
|
||||
name: "IPv4 prefers raw socket when available",
|
||||
family: &icmpV4,
|
||||
want: icmpRaw,
|
||||
wantNetworks: []string{"ip4:icmp"},
|
||||
},
|
||||
{
|
||||
name: "IPv4 uses datagram when raw unavailable",
|
||||
family: &icmpV4,
|
||||
rawErr: errors.New("operation not permitted"),
|
||||
want: icmpDatagram,
|
||||
wantNetworks: []string{"ip4:icmp", "udp4"},
|
||||
},
|
||||
{
|
||||
name: "IPv4 falls back to exec when both unavailable",
|
||||
family: &icmpV4,
|
||||
rawErr: errors.New("operation not permitted"),
|
||||
udpErr: errors.New("protocol not supported"),
|
||||
want: icmpExecFallback,
|
||||
wantNetworks: []string{"ip4:icmp", "udp4"},
|
||||
},
|
||||
{
|
||||
name: "IPv6 prefers raw socket when available",
|
||||
family: &icmpV6,
|
||||
want: icmpRaw,
|
||||
wantNetworks: []string{"ip6:ipv6-icmp"},
|
||||
},
|
||||
{
|
||||
name: "IPv6 uses datagram when raw unavailable",
|
||||
family: &icmpV6,
|
||||
rawErr: errors.New("operation not permitted"),
|
||||
want: icmpDatagram,
|
||||
wantNetworks: []string{"ip6:ipv6-icmp", "udp6"},
|
||||
},
|
||||
{
|
||||
name: "IPv6 falls back to exec when both unavailable",
|
||||
family: &icmpV6,
|
||||
rawErr: errors.New("operation not permitted"),
|
||||
udpErr: errors.New("protocol not supported"),
|
||||
want: icmpExecFallback,
|
||||
wantNetworks: []string{"ip6:ipv6-icmp", "udp6"},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
calls := make([]string, 0, 2)
|
||||
listen := func(network, listenAddr string) (icmpPacketConn, error) {
|
||||
require.Equal(t, tt.family.listenAddr, listenAddr)
|
||||
calls = append(calls, network)
|
||||
switch network {
|
||||
case tt.family.rawNetwork:
|
||||
if tt.rawErr != nil {
|
||||
return nil, tt.rawErr
|
||||
}
|
||||
case tt.family.dgramNetwork:
|
||||
if tt.udpErr != nil {
|
||||
return nil, tt.udpErr
|
||||
}
|
||||
default:
|
||||
t.Fatalf("unexpected network %q", network)
|
||||
}
|
||||
return testICMPPacketConn{}, nil
|
||||
}
|
||||
|
||||
assert.Equal(t, tt.want, detectICMPMode(tt.family, listen))
|
||||
assert.Equal(t, tt.wantNetworks, calls)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolveICMPTarget(t *testing.T) {
|
||||
t.Run("IPv4 literal", func(t *testing.T) {
|
||||
family, ip, err := resolveICMPTarget(context.Background(), "127.0.0.1")
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, family)
|
||||
assert.False(t, family.isIPv6)
|
||||
assert.Equal(t, "127.0.0.1", ip.String())
|
||||
})
|
||||
|
||||
t.Run("IPv6 literal", func(t *testing.T) {
|
||||
family, ip, err := resolveICMPTarget(context.Background(), "::1")
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, family)
|
||||
assert.True(t, family.isIPv6)
|
||||
assert.Equal(t, "::1", ip.String())
|
||||
})
|
||||
|
||||
t.Run("IPv4-mapped IPv6 resolves as IPv4", func(t *testing.T) {
|
||||
family, ip, err := resolveICMPTarget(context.Background(), "::ffff:127.0.0.1")
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, family)
|
||||
assert.False(t, family.isIPv6)
|
||||
assert.Equal(t, "127.0.0.1", ip.String())
|
||||
})
|
||||
}
|
||||
133
agent/network_monitor_probe.go
Normal file
133
agent/network_monitor_probe.go
Normal file
@@ -0,0 +1,133 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
)
|
||||
|
||||
const networkMonitorUserAgent = "Beszel-Agent/" + beszel.Version + " (+https://beszel.dev)"
|
||||
|
||||
// monitorProbe performs one check. Errors are recorded as loss by the task runner.
|
||||
// Implementations must honor cancellation and bound their execution time.
|
||||
type monitorProbe func(context.Context, monitor.Config) (int64, error)
|
||||
|
||||
func networkMonitorProbe(client *http.Client) monitorProbe {
|
||||
return func(ctx context.Context, config monitor.Config) (int64, error) {
|
||||
switch config.Protocol {
|
||||
case "icmp":
|
||||
return monitorICMP(ctx, config.Target)
|
||||
case "tcp":
|
||||
return monitorTCP(ctx, config.Target, config.Port)
|
||||
case "http":
|
||||
return monitorHTTP(ctx, client, config.Target)
|
||||
case "dns":
|
||||
return monitorDNS(ctx, config.Target, config.Server)
|
||||
default:
|
||||
return -1, fmt.Errorf("unknown monitor protocol: %s", config.Protocol)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// monitorTCP measures connection establishment time, including address fallback
|
||||
// but excluding DNS resolution.
|
||||
// Returns -1 and an error on failure.
|
||||
func monitorTCP(ctx context.Context, target string, port uint16) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
defer cancel()
|
||||
|
||||
// Resolve DNS first, outside the timing window but within the probe deadline.
|
||||
ips, err := net.DefaultResolver.LookupHost(ctx, target)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
if len(ips) == 0 {
|
||||
return -1, errors.New("no addresses resolved for TCP monitor")
|
||||
}
|
||||
portString := fmt.Sprintf("%d", port)
|
||||
deadline, _ := ctx.Deadline()
|
||||
|
||||
// Share the remaining probe budget across addresses so an unresponsive
|
||||
// first address cannot consume all the time available for alternatives.
|
||||
start := time.Now()
|
||||
for i, ip := range ips {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return -1, err
|
||||
}
|
||||
dialer := net.Dialer{Timeout: time.Until(deadline) / time.Duration(len(ips)-i)}
|
||||
var conn net.Conn
|
||||
conn, err = dialer.DialContext(ctx, "tcp", net.JoinHostPort(ip, portString))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
responseUs := time.Since(start).Microseconds()
|
||||
conn.Close()
|
||||
return responseUs, nil
|
||||
}
|
||||
return -1, err
|
||||
}
|
||||
|
||||
// monitorDNS measures DNS resolution response time in microseconds. If server is
|
||||
// non-empty, the lookup is sent to that DNS server (host or host:port, default
|
||||
// port 53) instead of the system resolver. Returns -1 and an error on failure.
|
||||
func monitorDNS(ctx context.Context, target, server string) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
defer cancel()
|
||||
|
||||
resolver := net.DefaultResolver
|
||||
if server != "" {
|
||||
resolver = dnsResolverForServer(server)
|
||||
}
|
||||
|
||||
start := time.Now()
|
||||
ips, err := resolver.LookupHost(ctx, target)
|
||||
if err != nil || len(ips) == 0 {
|
||||
return -1, err
|
||||
}
|
||||
return time.Since(start).Microseconds(), nil
|
||||
}
|
||||
|
||||
// dnsResolverForServer builds a resolver that sends lookups to the given DNS
|
||||
// server address instead of the system resolver. server may be a bare host or
|
||||
// host:port; when no port is given, the standard DNS port 53 is used.
|
||||
func dnsResolverForServer(server string) *net.Resolver {
|
||||
address := server
|
||||
if _, _, err := net.SplitHostPort(server); err != nil {
|
||||
address = net.JoinHostPort(server, "53")
|
||||
}
|
||||
return &net.Resolver{
|
||||
PreferGo: true,
|
||||
Dial: func(ctx context.Context, network, _ string) (net.Conn, error) {
|
||||
var dialer net.Dialer
|
||||
return dialer.DialContext(ctx, network, address)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// monitorHTTP measures HTTP GET request response in microseconds. Returns -1 and an error on failure.
|
||||
func monitorHTTP(ctx context.Context, client *http.Client, url string) (int64, error) {
|
||||
if client == nil {
|
||||
client = http.DefaultClient
|
||||
}
|
||||
start := time.Now()
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
req.Header.Set("User-Agent", networkMonitorUserAgent)
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return -1, err
|
||||
}
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return -1, fmt.Errorf("HTTP error: %s", resp.Status)
|
||||
}
|
||||
return time.Since(start).Microseconds(), nil
|
||||
}
|
||||
88
agent/network_monitor_resume.go
Normal file
88
agent/network_monitor_resume.go
Normal file
@@ -0,0 +1,88 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
const (
|
||||
monitorResumeHeartbeat = 10 * time.Second
|
||||
// Allow scheduling jitter without mistaking an ordinary tick for resume.
|
||||
monitorResumeGap = 2 * monitorResumeHeartbeat
|
||||
monitorResumePause = 10 * time.Second
|
||||
)
|
||||
|
||||
// monitorResumeGuard detects likely suspend/resume using wall time. A long
|
||||
// process stall or forward clock adjustment can also trigger the bounded pause.
|
||||
// One heartbeat is shared by all configured monitors.
|
||||
type monitorResumeGuard struct {
|
||||
mu sync.Mutex
|
||||
stop chan struct{}
|
||||
lastTick time.Time
|
||||
pauseUntil time.Time
|
||||
generation uint32
|
||||
}
|
||||
|
||||
func (g *monitorResumeGuard) start() {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
if g.stop != nil {
|
||||
return
|
||||
}
|
||||
stop := make(chan struct{})
|
||||
g.stop = stop
|
||||
g.lastTick = time.Now().Round(0)
|
||||
g.pauseUntil = time.Time{}
|
||||
go func() {
|
||||
ticker := time.NewTicker(monitorResumeHeartbeat)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
g.mu.Lock()
|
||||
if g.stop == stop {
|
||||
g.observe(time.Now())
|
||||
}
|
||||
g.mu.Unlock()
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
func (g *monitorResumeGuard) shutdown() {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
if g.stop != nil {
|
||||
close(g.stop)
|
||||
g.stop = nil
|
||||
g.generation++
|
||||
}
|
||||
}
|
||||
|
||||
// observe requires mu. Strip the monotonic component because it can stop during
|
||||
// suspend. Read the current time rather than the ticker's queued timestamp.
|
||||
func (g *monitorResumeGuard) observe(now time.Time) {
|
||||
now = now.Round(0)
|
||||
if now.Sub(g.lastTick) > monitorResumeGap {
|
||||
g.pauseUntil = now.Add(monitorResumePause)
|
||||
g.generation++
|
||||
}
|
||||
g.lastTick = now
|
||||
}
|
||||
|
||||
// snapshot also observes time so a probe waking before the heartbeat detects
|
||||
// resume itself. A changed generation invalidates probes spanning suspend.
|
||||
func (g *monitorResumeGuard) snapshot() (generation uint32, allowed bool) {
|
||||
if g == nil {
|
||||
return 0, true
|
||||
}
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
if g.stop == nil {
|
||||
return g.generation, true
|
||||
}
|
||||
g.observe(time.Now())
|
||||
return g.generation, !g.lastTick.Before(g.pauseUntil)
|
||||
}
|
||||
121
agent/network_monitor_resume_test.go
Normal file
121
agent/network_monitor_resume_test.go
Normal file
@@ -0,0 +1,121 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"testing/synctest"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func simulateMonitorSleep(g *monitorResumeGuard) {
|
||||
g.mu.Lock()
|
||||
g.lastTick = time.Now().Add(-time.Hour).Round(0)
|
||||
g.mu.Unlock()
|
||||
}
|
||||
|
||||
func TestMonitorResumePause(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
var g monitorResumeGuard
|
||||
g.start()
|
||||
defer g.shutdown()
|
||||
generation, allowed := g.snapshot()
|
||||
require.True(t, allowed)
|
||||
// Heartbeats alone must keep the guard current between infrequent probes.
|
||||
time.Sleep(time.Minute)
|
||||
synctest.Wait()
|
||||
steadyGeneration, allowed := g.snapshot()
|
||||
require.True(t, allowed)
|
||||
require.Equal(t, generation, steadyGeneration)
|
||||
// The probe, rather than the heartbeat, must detect this gap.
|
||||
simulateMonitorSleep(&g)
|
||||
next, allowed := g.snapshot()
|
||||
assert.False(t, allowed)
|
||||
assert.NotEqual(t, generation, next)
|
||||
time.Sleep(9 * time.Second)
|
||||
_, allowed = g.snapshot()
|
||||
assert.False(t, allowed)
|
||||
time.Sleep(time.Second)
|
||||
end, allowed := g.snapshot()
|
||||
assert.True(t, allowed)
|
||||
assert.Equal(t, next, end)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorResumeGuardLifecycle(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
pm := newMonitorManagerWithProbe(func(context.Context, monitor.Config) (int64, error) { return 1, nil })
|
||||
defer pm.Stop()
|
||||
assert.Nil(t, pm.resumeGuard.stop)
|
||||
pm.SyncMonitors([]monitor.Config{{ID: "a", Interval: 3600}, {ID: "b", Interval: 3600}})
|
||||
stop := pm.resumeGuard.stop
|
||||
require.NotNil(t, stop)
|
||||
pm.DeleteMonitor("a")
|
||||
assert.Equal(t, stop, pm.resumeGuard.stop)
|
||||
pm.DeleteMonitor("b")
|
||||
assert.Nil(t, pm.resumeGuard.stop)
|
||||
select {
|
||||
case <-stop:
|
||||
default:
|
||||
t.Fatal("heartbeat was not stopped")
|
||||
}
|
||||
time.Sleep(time.Hour)
|
||||
_, err := pm.UpsertMonitor(monitor.Config{ID: "c", Interval: 3600}, false)
|
||||
require.NoError(t, err)
|
||||
_, allowed := pm.resumeGuard.snapshot()
|
||||
assert.True(t, allowed, "idle time must not trigger a resume pause")
|
||||
pm.SyncMonitors(nil)
|
||||
assert.Nil(t, pm.resumeGuard.stop)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorResumeDiscardsInflightProbe(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
var g monitorResumeGuard
|
||||
g.start()
|
||||
defer g.shutdown()
|
||||
task := newMonitorTask(monitor.Config{ID: "test"})
|
||||
defer task.cancel()
|
||||
task.resumeGuard = &g
|
||||
result := task.runProbe(func(context.Context, monitor.Config) (int64, error) {
|
||||
simulateMonitorSleep(&g)
|
||||
return 0, errors.New("network not ready")
|
||||
})
|
||||
assert.Nil(t, result)
|
||||
assert.Empty(t, task.history.samples)
|
||||
// Explicit requests may still run during the pause and record real failures.
|
||||
result = task.runProbe(func(context.Context, monitor.Config) (int64, error) {
|
||||
return 0, errors.New("unreachable")
|
||||
})
|
||||
require.NotNil(t, result)
|
||||
assert.Equal(t, 100.0, result.PacketLoss)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorResumeSkipsScheduledProbes(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
var calls atomic.Int32
|
||||
pm := newMonitorManagerWithProbe(func(context.Context, monitor.Config) (int64, error) {
|
||||
calls.Add(1)
|
||||
return 1, nil
|
||||
})
|
||||
defer pm.Stop()
|
||||
pm.SyncMonitors([]monitor.Config{{ID: "test", Interval: 1}})
|
||||
simulateMonitorSleep(&pm.resumeGuard)
|
||||
pm.resumeGuard.snapshot()
|
||||
time.Sleep(9 * time.Second)
|
||||
synctest.Wait()
|
||||
assert.Zero(t, calls.Load())
|
||||
assert.Empty(t, pm.GetResults(1000))
|
||||
time.Sleep(2 * time.Second)
|
||||
synctest.Wait()
|
||||
assert.Positive(t, calls.Load())
|
||||
})
|
||||
}
|
||||
63
agent/network_monitor_schedule.go
Normal file
63
agent/network_monitor_schedule.go
Normal file
@@ -0,0 +1,63 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"math/rand"
|
||||
"time"
|
||||
)
|
||||
|
||||
func (pm *MonitorManager) startMonitor(task *monitorTask) {
|
||||
interval := time.Duration(task.config.Interval) * time.Second
|
||||
if interval < time.Second {
|
||||
interval = 30 * time.Second
|
||||
}
|
||||
delay := getStagger(interval.Milliseconds())
|
||||
slog.Debug("starting monitor task", "target", task.config.Target, "delay", delay, "interval", interval)
|
||||
// Certificate checks piggyback on probe ticks, so they run at most once per
|
||||
// probe interval after they become due.
|
||||
go runMonitorSchedule(task.ctx, interval, delay, func() {
|
||||
if _, allowed := task.resumeGuard.snapshot(); allowed {
|
||||
task.runProbe(pm.probe)
|
||||
task.refreshCert(pm.certCheck)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// runMonitorSchedule owns only timing. Checks run serially, and slow checks
|
||||
// naturally drop missed ticks rather than building an execution backlog.
|
||||
func runMonitorSchedule(ctx context.Context, interval, delay time.Duration, run func()) {
|
||||
timer := time.NewTimer(delay)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-timer.C:
|
||||
}
|
||||
if ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
run()
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
if ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
run()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// getStagger returns an initial delay between half an interval and one interval.
|
||||
func getStagger(intervalMilli int64) time.Duration {
|
||||
delay := rand.Intn(int(intervalMilli))
|
||||
if delay < int(intervalMilli)/2 {
|
||||
delay += int(intervalMilli) / 2
|
||||
}
|
||||
return time.Duration(delay) * time.Millisecond
|
||||
}
|
||||
167
agent/network_monitor_schedule_test.go
Normal file
167
agent/network_monitor_schedule_test.go
Normal file
@@ -0,0 +1,167 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"testing/synctest"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestMonitorScheduleTiming(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(t.Context())
|
||||
defer cancel()
|
||||
var calls atomic.Int32
|
||||
go runMonitorSchedule(ctx, 10*time.Second, 5*time.Second, func() { calls.Add(1) })
|
||||
synctest.Wait()
|
||||
time.Sleep(4 * time.Second)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 0, int(calls.Load()))
|
||||
time.Sleep(time.Second)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 1, int(calls.Load()))
|
||||
time.Sleep(10 * time.Second)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 2, int(calls.Load()))
|
||||
cancel()
|
||||
synctest.Wait()
|
||||
time.Sleep(time.Minute)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 2, int(calls.Load()))
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorScheduleSlowProbe(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(t.Context())
|
||||
defer cancel()
|
||||
var calls atomic.Int32
|
||||
release := make(chan struct{})
|
||||
go runMonitorSchedule(ctx, time.Second, 0, func() {
|
||||
calls.Add(1)
|
||||
select {
|
||||
case <-release:
|
||||
case <-ctx.Done():
|
||||
}
|
||||
})
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 1, int(calls.Load()))
|
||||
time.Sleep(time.Minute)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 1, int(calls.Load()), "a slow probe must not spawn overlapping checks")
|
||||
close(release)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 1, int(calls.Load()), "missed intervals must not accumulate a backlog")
|
||||
time.Sleep(time.Second)
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 2, int(calls.Load()))
|
||||
cancel()
|
||||
synctest.Wait()
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorScheduledAndImmediateRequestsShareProbe(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
var calls atomic.Int32
|
||||
release := make(chan struct{})
|
||||
cfg := monitor.Config{ID: "test", Interval: 10}
|
||||
pm := newMonitorManagerWithProbe(func(ctx context.Context, config monitor.Config) (int64, error) {
|
||||
assert.Equal(t, cfg, config)
|
||||
calls.Add(1)
|
||||
<-release
|
||||
return 42, nil
|
||||
})
|
||||
defer pm.Stop()
|
||||
task := newMonitorTask(cfg)
|
||||
pm.monitors[cfg.ID] = task
|
||||
go runMonitorSchedule(task.ctx, 10*time.Second, 0, func() { task.runProbe(pm.probe) })
|
||||
synctest.Wait()
|
||||
results := make(chan *monitor.Result, 2)
|
||||
for range 2 {
|
||||
go func() {
|
||||
result, _ := pm.UpsertMonitor(cfg, true)
|
||||
results <- result
|
||||
}()
|
||||
}
|
||||
synctest.Wait()
|
||||
assert.Equal(t, 1, int(calls.Load()))
|
||||
assert.Empty(t, pm.GetResults(1000), "reading history must not wait for network I/O")
|
||||
close(release)
|
||||
synctest.Wait()
|
||||
first, second := <-results, <-results
|
||||
require.NotNil(t, first)
|
||||
require.NotNil(t, second)
|
||||
assert.Equal(t, int64(42), first.AvgResponse)
|
||||
assert.Equal(t, first, second)
|
||||
assert.NotSame(t, first, second, "callers must not share mutable result pointers")
|
||||
assert.Len(t, task.history.samples, 1)
|
||||
// A later explicit request must still perform a fresh probe.
|
||||
_, err := pm.UpsertMonitor(cfg, true)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, 2, int(calls.Load()))
|
||||
assert.Len(t, task.history.samples, 2)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorReplacementCancelsSharedProbe(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
cfg := monitor.Config{ID: "test", Interval: 10}
|
||||
pm := newMonitorManagerWithProbe(func(ctx context.Context, config monitor.Config) (int64, error) {
|
||||
if config.Interval == 10 {
|
||||
<-ctx.Done()
|
||||
return 0, ctx.Err()
|
||||
}
|
||||
return 30, nil
|
||||
})
|
||||
defer pm.Stop()
|
||||
task := newMonitorTask(cfg)
|
||||
task.history.record(monitorSample{responseUs: 10, timestamp: time.Now()})
|
||||
pm.monitors[cfg.ID] = task
|
||||
results := make(chan *monitor.Result, 2)
|
||||
for range 2 {
|
||||
go func() {
|
||||
result, _ := pm.UpsertMonitor(cfg, true)
|
||||
results <- result
|
||||
}()
|
||||
}
|
||||
synctest.Wait()
|
||||
updated := cfg
|
||||
updated.Interval = 20
|
||||
result, err := pm.UpsertMonitor(updated, true)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
assert.Equal(t, int64(20), result.AvgResponse)
|
||||
assert.Zero(t, result.PacketLoss)
|
||||
synctest.Wait()
|
||||
assert.Nil(t, <-results)
|
||||
assert.Nil(t, <-results)
|
||||
assert.Len(t, task.history.samples, 1)
|
||||
assert.Len(t, pm.monitors[cfg.ID].history.samples, 2)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorInjectedProbeTimeoutRecordsLoss(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
pm := newMonitorManagerWithProbe(func(ctx context.Context, _ monitor.Config) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
defer cancel()
|
||||
<-ctx.Done()
|
||||
return 0, ctx.Err()
|
||||
})
|
||||
defer pm.Stop()
|
||||
start := time.Now()
|
||||
result, err := pm.UpsertMonitor(monitor.Config{ID: "test", Interval: 3600}, true)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
assert.Equal(t, 3*time.Second, time.Since(start))
|
||||
assert.Equal(t, 100.0, result.PacketLoss)
|
||||
assert.NoError(t, pm.monitors["test"].ctx.Err())
|
||||
})
|
||||
}
|
||||
191
agent/network_monitor_task.go
Normal file
191
agent/network_monitor_task.go
Normal file
@@ -0,0 +1,191 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
)
|
||||
|
||||
const monitorFailureLogInterval = 5 * time.Minute
|
||||
|
||||
// monitorTask coordinates a probe and its history for one immutable configuration.
|
||||
type monitorTask struct {
|
||||
config monitor.Config
|
||||
ctx context.Context
|
||||
cancel context.CancelFunc
|
||||
history *monitorHistory
|
||||
resumeGuard *monitorResumeGuard
|
||||
runMu sync.Mutex
|
||||
inflight *monitorRun
|
||||
lastFailureLog int64 // Unix nanoseconds
|
||||
|
||||
certMu sync.Mutex
|
||||
cert *monitor.CertInfo
|
||||
certUnsent bool // cert has not been included in a stats result yet
|
||||
certChecking bool
|
||||
nextCertCheck time.Time
|
||||
}
|
||||
|
||||
type monitorRun struct {
|
||||
done chan struct{}
|
||||
result *monitor.Result // published by closing done; never mutated afterwards
|
||||
}
|
||||
|
||||
func newMonitorTask(config monitor.Config) *monitorTask {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
task := &monitorTask{config: config, ctx: ctx, history: newMonitorHistory()}
|
||||
// Serialize cancellation with publication, so canceled probes cannot enter
|
||||
// history copied into a replacement task.
|
||||
task.cancel = func() {
|
||||
task.runMu.Lock()
|
||||
cancel()
|
||||
task.runMu.Unlock()
|
||||
}
|
||||
return task
|
||||
}
|
||||
|
||||
func newMonitorTaskFromExisting(config monitor.Config, existing *monitorTask) *monitorTask {
|
||||
task := newMonitorTask(config)
|
||||
if existing != nil {
|
||||
task.history = existing.history.clone()
|
||||
// Keep the last known certificate, but check again soon for the new config.
|
||||
// The hub already stores it, so it is not marked unsent.
|
||||
if config.Target == existing.config.Target {
|
||||
task.cert = existing.certInfo()
|
||||
}
|
||||
}
|
||||
return task
|
||||
}
|
||||
|
||||
// runProbe shares an in-flight check between scheduled and immediate requests.
|
||||
// Every completed check contributes exactly one sample, regardless of how many
|
||||
// callers were waiting for it. No task or history lock is held during network I/O.
|
||||
func (task *monitorTask) runProbe(probe monitorProbe) *monitor.Result {
|
||||
task.runMu.Lock()
|
||||
if task.ctx.Err() != nil {
|
||||
task.runMu.Unlock()
|
||||
return nil
|
||||
}
|
||||
if run := task.inflight; run != nil {
|
||||
task.runMu.Unlock()
|
||||
select {
|
||||
case <-task.ctx.Done():
|
||||
return nil
|
||||
case <-run.done:
|
||||
if task.ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
return copyMonitorResult(run.result)
|
||||
}
|
||||
}
|
||||
run := &monitorRun{done: make(chan struct{})}
|
||||
task.inflight = run
|
||||
task.runMu.Unlock()
|
||||
|
||||
generation, _ := task.resumeGuard.snapshot()
|
||||
responseUs, err := probe(task.ctx, task.config)
|
||||
var logFailure bool
|
||||
task.runMu.Lock()
|
||||
currentGeneration, _ := task.resumeGuard.snapshot()
|
||||
if task.ctx.Err() == nil && generation == currentGeneration {
|
||||
now := time.Now()
|
||||
if err != nil {
|
||||
responseUs = -1
|
||||
logAt := now.UnixNano()
|
||||
if task.lastFailureLog == 0 || logAt < task.lastFailureLog || logAt-task.lastFailureLog >= int64(monitorFailureLogInterval) {
|
||||
logFailure = true
|
||||
task.lastFailureLog = logAt
|
||||
}
|
||||
} else {
|
||||
task.lastFailureLog = 0
|
||||
}
|
||||
result := task.history.record(monitorSample{responseUs: responseUs, timestamp: now})
|
||||
run.result = &result
|
||||
}
|
||||
|
||||
task.inflight = nil
|
||||
close(run.done)
|
||||
task.runMu.Unlock()
|
||||
if logFailure {
|
||||
slog.Warn("monitor failed", "err", err, "target", task.config.Target, "protocol", task.config.Protocol)
|
||||
}
|
||||
if task.ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
return copyMonitorResult(run.result)
|
||||
}
|
||||
|
||||
// refreshCert checks the certificate of an HTTPS target when due. A failed
|
||||
// check keeps the last known certificate and retries sooner, as does a
|
||||
// certificate that expires before the next regular check, so renewals show up
|
||||
// quickly. Concurrent callers skip rather than wait, and no lock is held during
|
||||
// network I/O.
|
||||
func (task *monitorTask) refreshCert(check certChecker) {
|
||||
if check == nil || !certCheckEnabled(task.config) {
|
||||
return
|
||||
}
|
||||
task.certMu.Lock()
|
||||
if task.certChecking || time.Now().Before(task.nextCertCheck) {
|
||||
task.certMu.Unlock()
|
||||
return
|
||||
}
|
||||
task.certChecking = true
|
||||
task.certMu.Unlock()
|
||||
|
||||
info, err := check(task.ctx, task.config.Target)
|
||||
|
||||
task.certMu.Lock()
|
||||
defer task.certMu.Unlock()
|
||||
task.certChecking = false
|
||||
if task.ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
task.nextCertCheck = time.Now().Add(certCheckRetryInterval)
|
||||
slog.Warn("certificate check failed", "err", err, "target", task.config.Target)
|
||||
return
|
||||
}
|
||||
task.cert = &info
|
||||
task.certUnsent = true
|
||||
now := time.Now()
|
||||
interval := certCheckInterval
|
||||
if time.UnixMilli(info.Expires).Before(now.Add(certCheckInterval)) {
|
||||
interval = certCheckRetryInterval
|
||||
}
|
||||
task.nextCertCheck = now.Add(interval)
|
||||
}
|
||||
|
||||
// certInfo returns a copy of the latest certificate info, or nil if unknown.
|
||||
func (task *monitorTask) certInfo() *monitor.CertInfo {
|
||||
task.certMu.Lock()
|
||||
defer task.certMu.Unlock()
|
||||
if task.cert == nil {
|
||||
return nil
|
||||
}
|
||||
cert := *task.cert
|
||||
return &cert
|
||||
}
|
||||
|
||||
// takeUnsentCert returns the latest certificate info once after each successful
|
||||
// check, so unchanged info is not resent with every stats result.
|
||||
func (task *monitorTask) takeUnsentCert() *monitor.CertInfo {
|
||||
task.certMu.Lock()
|
||||
defer task.certMu.Unlock()
|
||||
if !task.certUnsent {
|
||||
return nil
|
||||
}
|
||||
task.certUnsent = false
|
||||
cert := *task.cert
|
||||
return &cert
|
||||
}
|
||||
|
||||
func copyMonitorResult(result *monitor.Result) *monitor.Result {
|
||||
if result == nil {
|
||||
return nil
|
||||
}
|
||||
copy := *result
|
||||
return ©
|
||||
}
|
||||
79
agent/network_monitor_task_test.go
Normal file
79
agent/network_monitor_task_test.go
Normal file
@@ -0,0 +1,79 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"testing"
|
||||
"testing/synctest"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestMonitorFailureLogCooldown(t *testing.T) {
|
||||
var logs bytes.Buffer
|
||||
previous := slog.Default()
|
||||
slog.SetDefault(slog.New(slog.NewTextHandler(&logs, nil)))
|
||||
t.Cleanup(func() { slog.SetDefault(previous) })
|
||||
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
task := newMonitorTask(monitor.Config{ID: "test", Target: "example.test", Protocol: "tcp"})
|
||||
defer task.cancel()
|
||||
failure := errors.New("connection refused")
|
||||
probe := func(context.Context, monitor.Config) (int64, error) { return 42, failure }
|
||||
var samples int64
|
||||
check := func(wantLog bool) {
|
||||
t.Helper()
|
||||
logs.Reset()
|
||||
result := task.runProbe(probe)
|
||||
require.NotNil(t, result)
|
||||
samples++
|
||||
assert.Equal(t, samples, result.SampleCount, "suppressed warnings must still record samples")
|
||||
if !wantLog {
|
||||
assert.Empty(t, logs.String())
|
||||
} else {
|
||||
assert.Contains(t, logs.String(), `msg="monitor failed"`)
|
||||
assert.Equal(t, 1, bytes.Count(logs.Bytes(), []byte("\n")))
|
||||
}
|
||||
}
|
||||
|
||||
check(true)
|
||||
check(false)
|
||||
time.Sleep(5*time.Minute - time.Nanosecond)
|
||||
check(false)
|
||||
time.Sleep(time.Nanosecond)
|
||||
check(true)
|
||||
check(false)
|
||||
time.Sleep(5 * time.Minute)
|
||||
check(true)
|
||||
check(false)
|
||||
|
||||
// Recovery clears the cooldown.
|
||||
failure = nil
|
||||
check(false)
|
||||
failure = errors.New("connection refused again")
|
||||
check(true)
|
||||
|
||||
// Another monitor has its own cooldown.
|
||||
other := newMonitorTask(task.config)
|
||||
defer other.cancel()
|
||||
logs.Reset()
|
||||
require.NotNil(t, other.runProbe(probe))
|
||||
assert.Contains(t, logs.String(), `msg="monitor failed"`)
|
||||
|
||||
// A canceled probe must not publish a failure or emit a warning.
|
||||
logs.Reset()
|
||||
result := other.runProbe(func(context.Context, monitor.Config) (int64, error) {
|
||||
other.cancel()
|
||||
return -1, context.Canceled
|
||||
})
|
||||
assert.Nil(t, result)
|
||||
assert.Empty(t, logs.String())
|
||||
})
|
||||
}
|
||||
590
agent/network_monitor_test.go
Normal file
590
agent/network_monitor_test.go
Normal file
@@ -0,0 +1,590 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/internal/entities/monitor"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/net/dns/dnsmessage"
|
||||
)
|
||||
|
||||
func TestMonitorManagerGetResultsIncludesHourResponseRange(t *testing.T) {
|
||||
now := time.Now().UTC()
|
||||
task := newMonitorTask(monitor.Config{ID: "monitor-1"})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-30 * time.Minute)})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-9 * time.Minute)})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: 40, timestamp: now.Add(-5 * time.Minute)})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: 30, timestamp: now.Add(-50 * time.Second)})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-30 * time.Second)})
|
||||
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{"icmp:example.com": task}
|
||||
|
||||
results := pm.GetResults(uint16(time.Minute / time.Millisecond))
|
||||
result, ok := results["monitor-1"]
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, int64(30), result.AvgResponse)
|
||||
assert.Equal(t, int64(25), result.AvgResponse1h)
|
||||
assert.Equal(t, int64(30), result.MinResponse)
|
||||
assert.Equal(t, int64(10), result.MinResponse1h)
|
||||
assert.Equal(t, int64(30), result.MaxResponse)
|
||||
assert.Equal(t, int64(40), result.MaxResponse1h)
|
||||
assert.Equal(t, 50.0, result.PacketLoss)
|
||||
assert.Equal(t, 20.0, result.PacketLoss1h)
|
||||
}
|
||||
|
||||
func TestMonitorManagerGetResultsIncludesLossOnlyHourData(t *testing.T) {
|
||||
now := time.Now().UTC()
|
||||
task := newMonitorTask(monitor.Config{ID: "monitor-1"})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-30 * time.Second)})
|
||||
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
|
||||
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{"icmp:example.com": task}
|
||||
|
||||
results := pm.GetResults(uint16(time.Minute / time.Millisecond))
|
||||
result, ok := results["monitor-1"]
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, int64(0), result.AvgResponse)
|
||||
assert.Equal(t, int64(0), result.AvgResponse1h)
|
||||
assert.Equal(t, int64(0), result.MinResponse)
|
||||
assert.Equal(t, int64(0), result.MinResponse1h)
|
||||
assert.Equal(t, int64(0), result.MaxResponse)
|
||||
assert.Equal(t, int64(0), result.MaxResponse1h)
|
||||
assert.Equal(t, 100.0, result.PacketLoss)
|
||||
assert.Equal(t, 100.0, result.PacketLoss1h)
|
||||
}
|
||||
|
||||
func TestMonitorConfigResultKeyUsesSyncedID(t *testing.T) {
|
||||
cfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
|
||||
assert.Equal(t, "monitor-1", cfg.ID)
|
||||
}
|
||||
|
||||
func TestMonitorManagerSyncMonitorsSkipsConfigsWithoutStableID(t *testing.T) {
|
||||
validCfg := monitor.Config{ID: "monitor-1", Target: "ignored", Protocol: "noop", Interval: 10}
|
||||
invalidCfg := monitor.Config{Target: "ignored", Protocol: "noop", Interval: 10}
|
||||
|
||||
pm := newMonitorManager()
|
||||
pm.SyncMonitors([]monitor.Config{validCfg, invalidCfg})
|
||||
defer pm.Stop()
|
||||
|
||||
_, validExists := pm.monitors[validCfg.ID]
|
||||
_, invalidExists := pm.monitors[invalidCfg.ID]
|
||||
assert.True(t, validExists)
|
||||
assert.False(t, invalidExists)
|
||||
}
|
||||
|
||||
func TestMonitorManagerSyncMonitorsStopsRemovedTasksButKeepsExisting(t *testing.T) {
|
||||
keepCfg := monitor.Config{ID: "monitor-1", Target: "ignored", Protocol: "noop", Interval: 10}
|
||||
removeCfg := monitor.Config{ID: "monitor-2", Target: "ignored", Protocol: "noop", Interval: 10}
|
||||
|
||||
keptTask := newMonitorTask(keepCfg)
|
||||
removedTask := newMonitorTask(removeCfg)
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{
|
||||
keepCfg.ID: keptTask,
|
||||
removeCfg.ID: removedTask,
|
||||
}
|
||||
|
||||
pm.SyncMonitors([]monitor.Config{keepCfg})
|
||||
|
||||
assert.Same(t, keptTask, pm.monitors[keepCfg.ID])
|
||||
_, exists := pm.monitors[removeCfg.ID]
|
||||
assert.False(t, exists)
|
||||
|
||||
select {
|
||||
case <-removedTask.ctx.Done():
|
||||
default:
|
||||
t.Fatal("expected removed monitor task to be cancelled")
|
||||
}
|
||||
|
||||
select {
|
||||
case <-keptTask.ctx.Done():
|
||||
t.Fatal("expected existing monitor task to remain active")
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorManagerSyncMonitorsRestartsChangedConfig(t *testing.T) {
|
||||
originalCfg := monitor.Config{ID: "monitor-1", Target: "ignored-a", Protocol: "noop", Interval: 10}
|
||||
updatedCfg := monitor.Config{ID: "monitor-1", Target: "ignored-b", Protocol: "noop", Interval: 10}
|
||||
originalTask := newMonitorTask(originalCfg)
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{
|
||||
originalCfg.ID: originalTask,
|
||||
}
|
||||
|
||||
pm.SyncMonitors([]monitor.Config{updatedCfg})
|
||||
defer pm.Stop()
|
||||
|
||||
restartedTask := pm.monitors[updatedCfg.ID]
|
||||
assert.NotSame(t, originalTask, restartedTask)
|
||||
assert.Equal(t, updatedCfg, restartedTask.config)
|
||||
|
||||
select {
|
||||
case <-originalTask.ctx.Done():
|
||||
default:
|
||||
t.Fatal("expected changed monitor task to be cancelled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorManagerApplySyncUpsertRunsImmediatelyAndReturnsResult(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
pm := &MonitorManager{
|
||||
monitors: make(map[string]*monitorTask),
|
||||
probe: networkMonitorProbe(server.Client()),
|
||||
}
|
||||
|
||||
resp, err := pm.HandleSyncRequest(monitor.SyncRequest{
|
||||
Action: monitor.SyncActionUpsert,
|
||||
Config: monitor.Config{ID: "monitor-1", Target: server.URL, Protocol: "http", Interval: 10},
|
||||
RunNow: true,
|
||||
})
|
||||
defer pm.Stop()
|
||||
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, resp.Result.AvgResponse, int64(0))
|
||||
assert.Equal(t, 0.0, resp.Result.PacketLoss)
|
||||
assert.Equal(t, 0.0, resp.Result.PacketLoss1h)
|
||||
|
||||
task := pm.monitors["monitor-1"]
|
||||
require.NotNil(t, task)
|
||||
task.history.mu.Lock()
|
||||
defer task.history.mu.Unlock()
|
||||
require.Len(t, task.history.samples, 1)
|
||||
}
|
||||
|
||||
func TestMonitorManagerUpsertMonitorKeepsHistoryWhenOnlyIntervalChanges(t *testing.T) {
|
||||
originalCfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
|
||||
updatedCfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 30}
|
||||
now := time.Now().UTC()
|
||||
|
||||
existingTask := newMonitorTask(originalCfg)
|
||||
existingTask.history.addSampleLocked(monitorSample{responseUs: 12, timestamp: now.Add(-50 * time.Minute)})
|
||||
existingTask.history.addSampleLocked(monitorSample{responseUs: 24, timestamp: now.Add(-30 * time.Second)})
|
||||
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{originalCfg.ID: existingTask}
|
||||
|
||||
result, err := pm.UpsertMonitor(updatedCfg, false)
|
||||
defer pm.Stop()
|
||||
|
||||
require.NoError(t, err)
|
||||
assert.Nil(t, result)
|
||||
|
||||
updatedTask := pm.monitors[updatedCfg.ID]
|
||||
require.NotNil(t, updatedTask)
|
||||
assert.NotSame(t, existingTask, updatedTask)
|
||||
assert.Equal(t, updatedCfg, updatedTask.config)
|
||||
|
||||
updatedTask.history.mu.Lock()
|
||||
defer updatedTask.history.mu.Unlock()
|
||||
require.Len(t, updatedTask.history.samples, 1)
|
||||
assert.Equal(t, int64(24), updatedTask.history.samples[0].responseUs)
|
||||
|
||||
agg := updatedTask.history.aggregateLocked(time.Hour, now)
|
||||
require.True(t, agg.hasData())
|
||||
assert.Equal(t, int64(2), agg.totalCount)
|
||||
assert.Equal(t, int64(2), agg.successCount)
|
||||
assert.Equal(t, int64(18), agg.avgResponse())
|
||||
|
||||
select {
|
||||
case <-existingTask.ctx.Done():
|
||||
default:
|
||||
t.Fatal("expected original monitor task to be cancelled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorManagerApplySyncDeleteRemovesTask(t *testing.T) {
|
||||
config := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
|
||||
task := newMonitorTask(config)
|
||||
pm := newMonitorManager()
|
||||
pm.monitors = map[string]*monitorTask{config.ID: task}
|
||||
|
||||
_, err := pm.HandleSyncRequest(monitor.SyncRequest{
|
||||
Action: monitor.SyncActionDelete,
|
||||
Config: monitor.Config{ID: config.ID},
|
||||
})
|
||||
|
||||
require.NoError(t, err)
|
||||
_, exists := pm.monitors[config.ID]
|
||||
assert.False(t, exists)
|
||||
|
||||
select {
|
||||
case <-task.ctx.Done():
|
||||
default:
|
||||
t.Fatal("expected deleted monitor task to be cancelled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorManagerGetRandomDelay(t *testing.T) {
|
||||
for i := 1000; i < 360_000; i += 1000 {
|
||||
delay := getStagger(int64(i))
|
||||
assert.GreaterOrEqual(t, delay, time.Duration(i/2)*time.Millisecond)
|
||||
assert.LessOrEqual(t, delay, time.Duration(i)*time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorHTTP(t *testing.T) {
|
||||
t.Run("success", func(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
assert.Equal(t, "Beszel-Agent/"+beszel.Version+" (+https://beszel.dev)", r.Header.Get("User-Agent"))
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
responseUs, err := monitorHTTP(context.Background(), server.Client(), server.URL)
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, responseUs, int64(0))
|
||||
})
|
||||
|
||||
t.Run("server error", func(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "boom", http.StatusInternalServerError)
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
responseUs, err := monitorHTTP(context.Background(), server.Client(), server.URL)
|
||||
assert.Equal(t, int64(-1), responseUs)
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorTCP(t *testing.T) {
|
||||
t.Run("success", func(t *testing.T) {
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
defer listener.Close()
|
||||
|
||||
accepted := make(chan struct{})
|
||||
go func() {
|
||||
defer close(accepted)
|
||||
conn, err := listener.Accept()
|
||||
if err == nil {
|
||||
_ = conn.Close()
|
||||
}
|
||||
}()
|
||||
|
||||
port := uint16(listener.Addr().(*net.TCPAddr).Port)
|
||||
responseUs, err := monitorTCP(context.Background(), "127.0.0.1", port)
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, responseUs, int64(0))
|
||||
<-accepted
|
||||
})
|
||||
|
||||
t.Run("connection failure", func(t *testing.T) {
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
|
||||
port := uint16(listener.Addr().(*net.TCPAddr).Port)
|
||||
require.NoError(t, listener.Close())
|
||||
|
||||
responseUs, err := monitorTCP(context.Background(), "127.0.0.1", port)
|
||||
assert.Equal(t, int64(-1), responseUs)
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorTCPAddressFallback(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
ips []string
|
||||
loss bool
|
||||
}{
|
||||
{"first address fails", []string{"127.0.0.2", "127.0.0.1"}, false},
|
||||
{"first address succeeds", []string{"127.0.0.1", "127.0.0.2"}, false},
|
||||
{"all addresses fail", []string{"127.0.0.2", "127.0.0.3"}, true},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
listener, err := net.Listen("tcp4", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
defer listener.Close()
|
||||
original := net.DefaultResolver
|
||||
net.DefaultResolver = tcpMonitorTestResolver(tc.ips)
|
||||
defer func() { net.DefaultResolver = original }()
|
||||
|
||||
// Verify the resolver preserves the intended order, so success cannot
|
||||
// accidentally bypass the failed first address in the regression case.
|
||||
ips, err := net.DefaultResolver.LookupHost(t.Context(), "tcp-monitor.invalid.")
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, tc.ips, ips)
|
||||
responseUs, err := monitorTCP(t.Context(), "tcp-monitor.invalid.", uint16(listener.Addr().(*net.TCPAddr).Port))
|
||||
if tc.loss {
|
||||
require.Error(t, err)
|
||||
assert.Equal(t, int64(-1), responseUs)
|
||||
} else {
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, responseUs, int64(0))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// tcpMonitorTestResolver supplies multiple A records without external DNS.
|
||||
func tcpMonitorTestResolver(ips []string) *net.Resolver {
|
||||
return &net.Resolver{PreferGo: true, Dial: func(ctx context.Context, network, address string) (net.Conn, error) {
|
||||
client, server := net.Pipe()
|
||||
go func() {
|
||||
defer server.Close()
|
||||
// net.Resolver uses TCP framing when its connection is not a PacketConn.
|
||||
var size uint16
|
||||
if err := binary.Read(server, binary.BigEndian, &size); err != nil {
|
||||
return
|
||||
}
|
||||
packet := make([]byte, size)
|
||||
if _, err := io.ReadFull(server, packet); err != nil {
|
||||
return
|
||||
}
|
||||
var msg dnsmessage.Message
|
||||
if err := msg.Unpack(packet); err != nil {
|
||||
return
|
||||
}
|
||||
msg.Header.Response = true
|
||||
msg.Header.RecursionAvailable = true
|
||||
for _, question := range msg.Questions {
|
||||
if question.Type != dnsmessage.TypeA {
|
||||
continue
|
||||
}
|
||||
for _, ip := range ips {
|
||||
msg.Answers = append(msg.Answers, dnsmessage.Resource{
|
||||
Header: dnsmessage.ResourceHeader{Name: question.Name, Type: dnsmessage.TypeA, Class: dnsmessage.ClassINET},
|
||||
Body: &dnsmessage.AResource{A: [4]byte(net.ParseIP(ip).To4())},
|
||||
})
|
||||
}
|
||||
}
|
||||
packet, err := msg.Pack()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
response := binary.BigEndian.AppendUint16(nil, uint16(len(packet)))
|
||||
_, _ = server.Write(append(response, packet...))
|
||||
}()
|
||||
return client, nil
|
||||
}}
|
||||
}
|
||||
|
||||
// udpDNSTestServer starts a UDP server on loopback that answers A queries with the
|
||||
// given IPs, and returns its listen address (host:port).
|
||||
func udpDNSTestServer(t *testing.T, ips []string) string {
|
||||
t.Helper()
|
||||
conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.ParseIP("127.0.0.1")})
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { conn.Close() })
|
||||
|
||||
go func() {
|
||||
buf := make([]byte, 512)
|
||||
for {
|
||||
n, addr, err := conn.ReadFromUDP(buf)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
var msg dnsmessage.Message
|
||||
if err := msg.Unpack(buf[:n]); err != nil {
|
||||
continue
|
||||
}
|
||||
msg.Header.Response = true
|
||||
msg.Header.RecursionAvailable = true
|
||||
for _, question := range msg.Questions {
|
||||
if question.Type != dnsmessage.TypeA {
|
||||
continue
|
||||
}
|
||||
for _, ip := range ips {
|
||||
msg.Answers = append(msg.Answers, dnsmessage.Resource{
|
||||
Header: dnsmessage.ResourceHeader{Name: question.Name, Type: dnsmessage.TypeA, Class: dnsmessage.ClassINET},
|
||||
Body: &dnsmessage.AResource{A: [4]byte(net.ParseIP(ip).To4())},
|
||||
})
|
||||
}
|
||||
}
|
||||
packet, err := msg.Pack()
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
_, _ = conn.WriteToUDP(packet, addr)
|
||||
}
|
||||
}()
|
||||
|
||||
return conn.LocalAddr().String()
|
||||
}
|
||||
|
||||
func TestMonitorDNS(t *testing.T) {
|
||||
t.Run("success", func(t *testing.T) {
|
||||
responseUs, err := monitorDNS(context.Background(), "localhost", "")
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, responseUs, int64(0))
|
||||
})
|
||||
|
||||
t.Run("lookup failure", func(t *testing.T) {
|
||||
responseUs, err := monitorDNS(context.Background(), "", "")
|
||||
assert.Equal(t, int64(-1), responseUs)
|
||||
require.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("custom server", func(t *testing.T) {
|
||||
serverAddr := udpDNSTestServer(t, []string{"192.0.2.10"})
|
||||
responseUs, err := monitorDNS(context.Background(), "example.test.", serverAddr)
|
||||
require.NoError(t, err)
|
||||
assert.GreaterOrEqual(t, responseUs, int64(0))
|
||||
})
|
||||
|
||||
t.Run("custom server without port defaults to 53", func(t *testing.T) {
|
||||
resolver := dnsResolverForServer("127.0.0.1")
|
||||
conn, err := resolver.Dial(context.Background(), "udp", "")
|
||||
require.NoError(t, err)
|
||||
defer conn.Close()
|
||||
assert.Equal(t, "127.0.0.1:53", conn.RemoteAddr().String())
|
||||
})
|
||||
|
||||
t.Run("custom server unreachable", func(t *testing.T) {
|
||||
responseUs, err := monitorDNS(context.Background(), "example.test.", "127.0.0.1:1")
|
||||
assert.Equal(t, int64(-1), responseUs)
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
func TestMonitorManagerCancelsActiveProbe(t *testing.T) {
|
||||
for _, action := range []string{"stop", "delete", "upsert", "sync replace", "sync remove"} {
|
||||
t.Run(action, func(t *testing.T) {
|
||||
started := make(chan struct{})
|
||||
canceled := make(chan struct{})
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
close(started)
|
||||
select {
|
||||
case <-r.Context().Done():
|
||||
close(canceled)
|
||||
case <-release:
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
defer close(release)
|
||||
pm := newMonitorManager()
|
||||
defer pm.Stop()
|
||||
cfg := monitor.Config{ID: "test", Protocol: "http", Target: server.URL, Interval: 3600}
|
||||
task := newMonitorTask(cfg)
|
||||
// Seed history to ensure a canceled RunNow does not return an old result.
|
||||
task.history.addSampleLocked(monitorSample{responseUs: 123, timestamp: time.Now()})
|
||||
pm.monitors[cfg.ID] = task
|
||||
done := make(chan *monitor.Result, 1)
|
||||
go func() {
|
||||
result, _ := pm.UpsertMonitor(cfg, true)
|
||||
done <- result
|
||||
}()
|
||||
select {
|
||||
case <-started:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("probe did not start")
|
||||
}
|
||||
updated := cfg
|
||||
updated.Interval--
|
||||
switch action {
|
||||
case "stop":
|
||||
pm.Stop()
|
||||
case "delete":
|
||||
pm.DeleteMonitor(cfg.ID)
|
||||
case "upsert":
|
||||
_, err := pm.UpsertMonitor(updated, false)
|
||||
require.NoError(t, err)
|
||||
case "sync replace":
|
||||
pm.SyncMonitors([]monitor.Config{updated})
|
||||
case "sync remove":
|
||||
pm.SyncMonitors(nil)
|
||||
}
|
||||
select {
|
||||
case <-canceled:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("active HTTP request was not canceled")
|
||||
}
|
||||
select {
|
||||
case result := <-done:
|
||||
assert.Nil(t, result)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("RunNow did not return after cancellation")
|
||||
}
|
||||
task.history.mu.Lock()
|
||||
assert.Len(t, task.history.samples, 1, "cancellation must not record packet loss")
|
||||
task.history.mu.Unlock()
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorResolutionCancellation(t *testing.T) {
|
||||
for _, protocol := range []string{"tcp", "dns", "icmp"} {
|
||||
t.Run(protocol, func(t *testing.T) {
|
||||
started := make(chan struct{}, 1)
|
||||
original := net.DefaultResolver
|
||||
net.DefaultResolver = &net.Resolver{PreferGo: true, Dial: func(ctx context.Context, network, address string) (net.Conn, error) {
|
||||
select {
|
||||
case started <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
<-ctx.Done()
|
||||
return nil, ctx.Err()
|
||||
}}
|
||||
defer func() { net.DefaultResolver = original }()
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
var err error
|
||||
switch protocol {
|
||||
case "tcp":
|
||||
_, err = monitorTCP(ctx, "monitor-cancellation.invalid.", 80)
|
||||
case "dns":
|
||||
_, err = monitorDNS(ctx, "monitor-cancellation.invalid.", "")
|
||||
case "icmp":
|
||||
_, err = monitorICMP(ctx, "monitor-cancellation.invalid.")
|
||||
}
|
||||
done <- err
|
||||
}()
|
||||
select {
|
||||
case <-started:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("lookup did not start")
|
||||
}
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
require.Error(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("lookup did not cancel")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorProbeTimeoutRecordsLoss(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
select {
|
||||
case <-r.Context().Done():
|
||||
case <-release:
|
||||
}
|
||||
}))
|
||||
defer server.Close()
|
||||
defer close(release)
|
||||
pm := newMonitorManager()
|
||||
pm.probe = networkMonitorProbe(&http.Client{Timeout: 20 * time.Millisecond})
|
||||
task := newMonitorTask(monitor.Config{ID: "timeout", Protocol: "http", Target: server.URL})
|
||||
defer task.cancel()
|
||||
|
||||
result := task.runProbe(pm.probe)
|
||||
require.NotNil(t, result)
|
||||
assert.Equal(t, 100.0, result.PacketLoss)
|
||||
assert.Equal(t, 100.0, result.PacketLoss1h)
|
||||
require.Len(t, task.history.samples, 1)
|
||||
assert.Equal(t, int64(-1), task.history.samples[0].responseUs)
|
||||
assert.NoError(t, task.ctx.Err(), "a probe timeout must not cancel the task")
|
||||
}
|
||||
310
agent/package_updates.go
Normal file
310
agent/package_updates.go
Normal file
@@ -0,0 +1,310 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultPackageUpdatesInterval = time.Hour
|
||||
packageUpdatesTimeout = 5 * time.Minute
|
||||
// pacmanSyncInterval limits how often checkupdates downloads fresh sync
|
||||
// databases. Checks in between reuse the last synced copy.
|
||||
pacmanSyncInterval = 12 * time.Hour
|
||||
)
|
||||
|
||||
// packageUpdatesCheck returns [total] or [total, security] pending package updates.
|
||||
type packageUpdatesCheck func(ctx context.Context) ([]uint16, error)
|
||||
|
||||
// packageUpdatesManager periodically checks the host package manager for pending
|
||||
// updates in the background and caches the result, so checks never delay metrics.
|
||||
type packageUpdatesManager struct {
|
||||
sync.Mutex
|
||||
check packageUpdatesCheck
|
||||
interval time.Duration
|
||||
counts []uint16
|
||||
checkedAt time.Time
|
||||
running bool
|
||||
}
|
||||
|
||||
// newPackageUpdatesManager returns nil if disabled or no supported package manager
|
||||
// is found. Agents running in a container are skipped because the container's
|
||||
// package database is not the host's. dataDir holds pacman's private sync databases.
|
||||
func newPackageUpdatesManager(dataDir string) *packageUpdatesManager {
|
||||
if runtime.GOOS != "linux" || runningInContainer() {
|
||||
return nil
|
||||
}
|
||||
interval := defaultPackageUpdatesInterval
|
||||
if env, exists := utils.GetEnv("PACKAGE_UPDATES_INTERVAL"); exists {
|
||||
duration, err := time.ParseDuration(env)
|
||||
switch {
|
||||
case err == nil && duration == 0:
|
||||
return nil
|
||||
case err == nil && duration > 0:
|
||||
interval = duration
|
||||
default:
|
||||
slog.Warn("Invalid PACKAGE_UPDATES_INTERVAL", "value", env)
|
||||
}
|
||||
}
|
||||
name, check := detectPackageManager(dataDir)
|
||||
if check == nil {
|
||||
return nil
|
||||
}
|
||||
slog.Debug("Package updates", "manager", name, "interval", interval)
|
||||
return &packageUpdatesManager{check: check, interval: interval}
|
||||
}
|
||||
|
||||
// get returns the last cached counts and starts a background check if they are stale.
|
||||
func (pm *packageUpdatesManager) get(now time.Time) []uint16 {
|
||||
pm.Lock()
|
||||
defer pm.Unlock()
|
||||
if !pm.running && (pm.checkedAt.IsZero() || now.Sub(pm.checkedAt) >= pm.interval) {
|
||||
pm.running = true
|
||||
go pm.refresh()
|
||||
}
|
||||
return pm.counts
|
||||
}
|
||||
|
||||
func (pm *packageUpdatesManager) refresh() {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), packageUpdatesTimeout)
|
||||
defer cancel()
|
||||
counts, err := pm.check(ctx)
|
||||
if err != nil {
|
||||
slog.Debug("Package updates check failed", "err", err)
|
||||
counts = nil
|
||||
}
|
||||
pm.Lock()
|
||||
pm.counts = counts
|
||||
pm.checkedAt = time.Now()
|
||||
pm.running = false
|
||||
pm.Unlock()
|
||||
}
|
||||
|
||||
func runningInContainer() bool {
|
||||
for _, path := range []string{"/.dockerenv", "/run/.containerenv"} {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func detectPackageManager(dataDir string) (string, packageUpdatesCheck) {
|
||||
switch {
|
||||
case commandExists("apt-get"):
|
||||
return "apt", checkApt
|
||||
case commandExists("dnf"):
|
||||
return "dnf", checkDnf
|
||||
case commandExists("zypper"):
|
||||
return "zypper", checkZypper
|
||||
case commandExists("checkupdates"):
|
||||
return "pacman", newPacmanCheck(dataDir)
|
||||
case commandExists("apk"):
|
||||
return "apk", checkApk
|
||||
}
|
||||
return "", nil
|
||||
}
|
||||
|
||||
func commandExists(name string) bool {
|
||||
_, err := exec.LookPath(name)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// runPackageCommand runs a read-only package manager command and returns stdout.
|
||||
// okCodes lists non-zero exit codes that still mean success.
|
||||
func runPackageCommand(ctx context.Context, okCodes []int, name string, args ...string) (string, error) {
|
||||
return runPackageCommandEnv(ctx, nil, okCodes, name, args...)
|
||||
}
|
||||
|
||||
// runPackageCommandEnv is runPackageCommand with extra environment variables.
|
||||
func runPackageCommandEnv(ctx context.Context, env []string, okCodes []int, name string, args ...string) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, name, args...)
|
||||
cmd.Env = append(os.Environ(), "LC_ALL=C")
|
||||
cmd.Env = append(cmd.Env, env...)
|
||||
// checkupdates is a shell script, so a timeout kills only the script and its
|
||||
// children can keep stdout open. WaitDelay stops Output from waiting on them.
|
||||
cmd.WaitDelay = 10 * time.Second
|
||||
out, err := cmd.Output()
|
||||
if exitErr, ok := errors.AsType[*exec.ExitError](err); ok && slices.Contains(okCodes, exitErr.ExitCode()) {
|
||||
return string(out), nil
|
||||
}
|
||||
return string(out), err
|
||||
}
|
||||
|
||||
// checkApt simulates a full upgrade against the current package lists.
|
||||
// It never refreshes the lists; apt-daily or the user does that.
|
||||
func checkApt(ctx context.Context) ([]uint16, error) {
|
||||
out, err := runPackageCommand(ctx, nil, "apt-get", "-s", "dist-upgrade")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
total, security := parseAptSimulate(out)
|
||||
return []uint16{total, security}, nil
|
||||
}
|
||||
|
||||
// checkDnf uses the system metadata cache only (-C), so it never downloads metadata.
|
||||
func checkDnf(ctx context.Context) ([]uint16, error) {
|
||||
out, err := runPackageCommand(ctx, []int{100}, "dnf", "-q", "-C", "check-update")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
total := parseDnfCheckUpdate(out)
|
||||
out, err = runPackageCommand(ctx, []int{100}, "dnf", "-q", "-C", "check-update", "--security")
|
||||
if err != nil {
|
||||
return []uint16{total}, nil
|
||||
}
|
||||
return []uint16{total, parseDnfCheckUpdate(out)}, nil
|
||||
}
|
||||
|
||||
func checkZypper(ctx context.Context) ([]uint16, error) {
|
||||
out, err := runPackageCommand(ctx, nil, "zypper", "--no-refresh", "-q", "list-updates")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
total := parseZypperTable(out)
|
||||
out, err = runPackageCommand(ctx, nil, "zypper", "--no-refresh", "-q", "list-patches", "--category", "security")
|
||||
if err != nil {
|
||||
return []uint16{total}, nil
|
||||
}
|
||||
return []uint16{total, parseZypperTable(out)}, nil
|
||||
}
|
||||
|
||||
// newPacmanCheck uses checkupdates (pacman-contrib), which syncs a private copy of
|
||||
// the databases and never touches pacman's own. The copy lives in dataDir because
|
||||
// the systemd unit's ProtectSystem=strict makes the default /tmp location read-only.
|
||||
// It syncs every pacmanSyncInterval and uses the existing copy (-n) in between.
|
||||
// Local upgrades show up right away since checkupdates links the live local DB.
|
||||
// Exit code 2 means no updates.
|
||||
func newPacmanCheck(dataDir string) packageUpdatesCheck {
|
||||
var env []string
|
||||
var syncDir string
|
||||
if dataDir != "" {
|
||||
dbPath := filepath.Join(dataDir, "checkup-db")
|
||||
env = []string{"CHECKUPDATES_DB=" + dbPath}
|
||||
syncDir = filepath.Join(dbPath, "sync")
|
||||
}
|
||||
// checks never overlap (packageUpdatesManager.running), so no lock is needed
|
||||
var lastSync time.Time
|
||||
return func(ctx context.Context) ([]uint16, error) {
|
||||
// -n with a missing database reports no updates rather than failing,
|
||||
// so always sync first and whenever the private copy is missing
|
||||
sync := lastSync.IsZero() || time.Since(lastSync) >= pacmanSyncInterval
|
||||
if !sync && syncDir != "" {
|
||||
if _, err := os.Stat(syncDir); err != nil {
|
||||
sync = true
|
||||
}
|
||||
}
|
||||
var args []string
|
||||
if !sync {
|
||||
args = append(args, "-n")
|
||||
}
|
||||
out, err := runPackageCommandEnv(ctx, env, []int{2}, "checkupdates", args...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if sync {
|
||||
lastSync = time.Now()
|
||||
}
|
||||
return []uint16{parsePacmanCheckUpdates(out)}, nil
|
||||
}
|
||||
}
|
||||
|
||||
func checkApk(ctx context.Context) ([]uint16, error) {
|
||||
out, err := runPackageCommand(ctx, nil, "apk", "--no-network", "-u", "list")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return []uint16{parseApkUpgradable(out)}, nil
|
||||
}
|
||||
|
||||
// parseAptSimulate counts upgrades in `apt-get -s` output. Upgrade lines look like
|
||||
// "Inst libc6 [2.35-0ubuntu3.4] (2.35-0ubuntu3.15 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])".
|
||||
// New dependencies have no "[old version]" and are not counted.
|
||||
func parseAptSimulate(out string) (total, security uint16) {
|
||||
scanner := bufio.NewScanner(strings.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) < 4 || fields[0] != "Inst" || !strings.HasPrefix(fields[2], "[") {
|
||||
continue
|
||||
}
|
||||
total++
|
||||
start := strings.IndexByte(line, '(')
|
||||
end := strings.IndexByte(line, ')')
|
||||
if start >= 0 && end > start && strings.Contains(line[start:end], "-security") {
|
||||
security++
|
||||
}
|
||||
}
|
||||
return total, security
|
||||
}
|
||||
|
||||
// parseDnfCheckUpdate counts "name.arch version repo" lines, stopping at the
|
||||
// obsoletes section so obsoleted packages are not counted twice.
|
||||
func parseDnfCheckUpdate(out string) (count uint16) {
|
||||
scanner := bufio.NewScanner(strings.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if strings.HasPrefix(line, "Obsoleting") {
|
||||
break
|
||||
}
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) == 3 && strings.Contains(fields[0], ".") {
|
||||
count++
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// parseZypperTable counts the data rows of a zypper table (the lines after the
|
||||
// "---+---" separator).
|
||||
func parseZypperTable(out string) (count uint16) {
|
||||
inTable := false
|
||||
scanner := bufio.NewScanner(strings.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
switch {
|
||||
case !inTable:
|
||||
inTable = strings.HasPrefix(line, "--") && strings.Contains(line, "-+-")
|
||||
case strings.Contains(line, "|"):
|
||||
count++
|
||||
default:
|
||||
return count
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// parsePacmanCheckUpdates counts "name old -> new" lines.
|
||||
func parsePacmanCheckUpdates(out string) (count uint16) {
|
||||
scanner := bufio.NewScanner(strings.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
if strings.Contains(scanner.Text(), " -> ") {
|
||||
count++
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// parseApkUpgradable counts lines of `apk -u list`, which look like
|
||||
// "musl-1.2.5-r3 aarch64 {musl} (MIT) [upgradable from: musl-1.2.5-r0]".
|
||||
func parseApkUpgradable(out string) (count uint16) {
|
||||
scanner := bufio.NewScanner(strings.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
if strings.Contains(scanner.Text(), "[upgradable from:") {
|
||||
count++
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
196
agent/package_updates_test.go
Normal file
196
agent/package_updates_test.go
Normal file
@@ -0,0 +1,196 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func readPackageUpdatesTestData(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
data, err := os.ReadFile(filepath.Join("test-data", "package_updates", name))
|
||||
require.NoError(t, err)
|
||||
return string(data)
|
||||
}
|
||||
|
||||
// Test data files are real command outputs captured in containers.
|
||||
|
||||
func TestParseAptSimulate(t *testing.T) {
|
||||
tests := []struct {
|
||||
file string
|
||||
total, security uint16
|
||||
}{
|
||||
{"apt_debian12.txt", 44, 5},
|
||||
{"apt_ubuntu2204.txt", 58, 45},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.file, func(t *testing.T) {
|
||||
total, security := parseAptSimulate(readPackageUpdatesTestData(t, tt.file))
|
||||
assert.Equal(t, tt.total, total)
|
||||
assert.Equal(t, tt.security, security)
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("new dependencies and trailing brackets", func(t *testing.T) {
|
||||
out := `Inst linux-image-6.8.0-50-generic (6.8.0-50.51 Ubuntu:24.04/noble-updates, Ubuntu:24.04/noble-security [amd64])
|
||||
Inst linux-image-generic [6.8.0-49.49] (6.8.0-50.50 Ubuntu:24.04/noble-updates, Ubuntu:24.04/noble-security [amd64])
|
||||
Inst gcc-12-base [12.3.0-1ubuntu1~22.04] (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates [arm64]) [libstdc++6:arm64 libgcc-s1:arm64 ]
|
||||
Conf linux-image-generic (6.8.0-50.50 Ubuntu:24.04/noble-updates, Ubuntu:24.04/noble-security [amd64])
|
||||
Remv oldpkg [1.0]`
|
||||
total, security := parseAptSimulate(out)
|
||||
assert.Equal(t, uint16(2), total)
|
||||
assert.Equal(t, uint16(1), security)
|
||||
})
|
||||
|
||||
t.Run("no updates", func(t *testing.T) {
|
||||
total, security := parseAptSimulate("Reading package lists...\n0 upgraded, 0 newly installed, 0 to remove and 0 not upgraded.\n")
|
||||
assert.Zero(t, total)
|
||||
assert.Zero(t, security)
|
||||
})
|
||||
}
|
||||
|
||||
func TestParseDnfCheckUpdate(t *testing.T) {
|
||||
tests := []struct {
|
||||
file string
|
||||
count uint16
|
||||
}{
|
||||
{"dnf4_rocky9_check_update.txt", 110},
|
||||
{"dnf4_rocky9_check_update_security.txt", 53},
|
||||
{"dnf5_fedora42_check_update.txt", 20},
|
||||
{"dnf5_fedora42_check_update_security.txt", 5},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.file, func(t *testing.T) {
|
||||
assert.Equal(t, tt.count, parseDnfCheckUpdate(readPackageUpdatesTestData(t, tt.file)))
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("obsoletes section and notices", func(t *testing.T) {
|
||||
out := `
|
||||
kernel.x86_64 5.14.0-503.el9 baseos
|
||||
Security: kernel-core-5.14.0-427.el9.x86_64 is an installed security update
|
||||
Obsoleting Packages
|
||||
grub2-tools.x86_64 1:2.06-80.el9 baseos
|
||||
grub2-tools.x86_64 1:2.06-77.el9 @baseos
|
||||
`
|
||||
assert.Equal(t, uint16(1), parseDnfCheckUpdate(out))
|
||||
})
|
||||
}
|
||||
|
||||
func TestParseZypperTable(t *testing.T) {
|
||||
tests := []struct {
|
||||
file string
|
||||
count uint16
|
||||
}{
|
||||
{"zypper_leap155_list_updates.txt", 22},
|
||||
{"zypper_leap155_list_patches_security.txt", 4},
|
||||
{"zypper_leap156_list_updates_none.txt", 0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.file, func(t *testing.T) {
|
||||
assert.Equal(t, tt.count, parseZypperTable(readPackageUpdatesTestData(t, tt.file)))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestParsePacmanCheckUpdates(t *testing.T) {
|
||||
assert.Equal(t, uint16(4), parsePacmanCheckUpdates(readPackageUpdatesTestData(t, "pacman_checkupdates.txt")))
|
||||
assert.Zero(t, parsePacmanCheckUpdates(""))
|
||||
}
|
||||
|
||||
func TestParseApkUpgradable(t *testing.T) {
|
||||
assert.Equal(t, uint16(10), parseApkUpgradable(readPackageUpdatesTestData(t, "apk_alpine320_list_upgradable.txt")))
|
||||
assert.Zero(t, parseApkUpgradable(""))
|
||||
}
|
||||
|
||||
func TestPackageUpdatesManagerCaching(t *testing.T) {
|
||||
calls := make(chan struct{}, 10)
|
||||
result := []uint16{3, 1}
|
||||
var resultErr error
|
||||
pm := &packageUpdatesManager{
|
||||
interval: time.Hour,
|
||||
check: func(context.Context) ([]uint16, error) {
|
||||
calls <- struct{}{}
|
||||
return result, resultErr
|
||||
},
|
||||
}
|
||||
waitIdle := func() {
|
||||
require.Eventually(t, func() bool {
|
||||
pm.Lock()
|
||||
defer pm.Unlock()
|
||||
return !pm.running
|
||||
}, time.Second, time.Millisecond)
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
// first call starts a background check and returns nothing yet
|
||||
assert.Nil(t, pm.get(now))
|
||||
waitIdle()
|
||||
assert.Len(t, calls, 1)
|
||||
|
||||
// cached result within interval, no new check
|
||||
assert.Equal(t, []uint16{3, 1}, pm.get(now.Add(time.Minute)))
|
||||
assert.Len(t, calls, 1)
|
||||
|
||||
// stale after interval: returns cached value and refreshes in background
|
||||
result, resultErr = nil, errors.New("boom")
|
||||
assert.Equal(t, []uint16{3, 1}, pm.get(now.Add(2*time.Hour)))
|
||||
waitIdle()
|
||||
assert.Len(t, calls, 2)
|
||||
|
||||
// failed check clears the counts
|
||||
assert.Nil(t, pm.get(time.Now()))
|
||||
}
|
||||
|
||||
func TestPacmanCheckSync(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("requires a shell script on PATH")
|
||||
}
|
||||
binDir := t.TempDir()
|
||||
dataDir := t.TempDir()
|
||||
logFile := filepath.Join(binDir, "calls.log")
|
||||
// fake checkupdates logs its args and db path, and creates the sync dir when syncing
|
||||
script := `#!/bin/sh
|
||||
echo "args=[$*] db=$CHECKUPDATES_DB" >> ` + logFile + `
|
||||
[ "$1" = "-n" ] || mkdir -p "$CHECKUPDATES_DB/sync"
|
||||
echo "linux 6.1-1 -> 6.2-1"
|
||||
`
|
||||
require.NoError(t, os.WriteFile(filepath.Join(binDir, "checkupdates"), []byte(script), 0o755))
|
||||
t.Setenv("PATH", binDir+string(os.PathListSeparator)+os.Getenv("PATH"))
|
||||
|
||||
check := newPacmanCheck(dataDir)
|
||||
dbPath := filepath.Join(dataDir, "checkup-db")
|
||||
readCalls := func() []string {
|
||||
data, err := os.ReadFile(logFile)
|
||||
require.NoError(t, err)
|
||||
return strings.Split(strings.TrimSpace(string(data)), "\n")
|
||||
}
|
||||
|
||||
// first check syncs
|
||||
counts, err := check(context.Background())
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, []uint16{1}, counts)
|
||||
// later checks reuse the synced copy
|
||||
_, err = check(context.Background())
|
||||
require.NoError(t, err)
|
||||
// a missing private copy forces a sync
|
||||
require.NoError(t, os.RemoveAll(dbPath))
|
||||
_, err = check(context.Background())
|
||||
require.NoError(t, err)
|
||||
|
||||
assert.Equal(t, []string{
|
||||
"args=[] db=" + dbPath,
|
||||
"args=[-n] db=" + dbPath,
|
||||
"args=[] db=" + dbPath,
|
||||
}, readCalls())
|
||||
}
|
||||
@@ -21,6 +21,9 @@ func newAgentResponse(data any, requestID *uint32) common.AgentResponse {
|
||||
response.String = &v
|
||||
case map[string]smart.SmartData:
|
||||
response.SmartData = v
|
||||
case smart.SmartDataResponse:
|
||||
response.SmartData = v.Data
|
||||
response.SmartComplete = v.Complete
|
||||
case systemd.ServiceDetails:
|
||||
response.ServiceInfo = v
|
||||
default:
|
||||
|
||||
125
agent/sensors.go
125
agent/sensors.go
@@ -5,7 +5,9 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -32,6 +34,8 @@ type SensorConfig struct {
|
||||
isBlacklist bool
|
||||
hasWildcards bool
|
||||
skipCollection bool
|
||||
skipGPU bool
|
||||
sensorShadow string
|
||||
firstRun bool
|
||||
}
|
||||
|
||||
@@ -41,13 +45,14 @@ func (a *Agent) newSensorConfig() *SensorConfig {
|
||||
sensorsEnvVal, sensorsSet := utils.GetEnv("SENSORS")
|
||||
skipCollection := sensorsSet && sensorsEnvVal == ""
|
||||
sensorsTimeout, _ := utils.GetEnv("SENSORS_TIMEOUT")
|
||||
skipGPU, _ := utils.GetEnv("SKIP_GPU")
|
||||
|
||||
return a.newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout, skipCollection)
|
||||
return a.newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout, skipCollection, skipGPU == "true")
|
||||
}
|
||||
|
||||
// newSensorConfigWithEnv creates a SensorConfig with the provided environment variables
|
||||
// sensorsSet indicates if the SENSORS environment variable was explicitly set (even to empty string)
|
||||
func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout string, skipCollection bool) *SensorConfig {
|
||||
func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal, sensorsTimeout string, skipCollection, skipGPU bool) *SensorConfig {
|
||||
timeout := 2 * time.Second
|
||||
if sensorsTimeout != "" {
|
||||
if d, err := time.ParseDuration(sensorsTimeout); err == nil {
|
||||
@@ -62,6 +67,7 @@ func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal,
|
||||
primarySensor: primarySensor,
|
||||
timeout: timeout,
|
||||
skipCollection: skipCollection,
|
||||
skipGPU: skipGPU,
|
||||
firstRun: true,
|
||||
sensors: make(map[string]struct{}),
|
||||
}
|
||||
@@ -73,6 +79,19 @@ func (a *Agent) newSensorConfigWithEnv(primarySensor, sysSensors, sensorsEnvVal,
|
||||
common.EnvKey, common.EnvMap{common.HostSysEnvKey: sysSensors},
|
||||
)
|
||||
}
|
||||
if skipGPU && runtime.GOOS == "linux" {
|
||||
// gopsutil reads every temp*_input before results can be filtered, so
|
||||
// point it at a shadow tree built from the effective sysfs root instead.
|
||||
if shadow, err := buildNonGpuSysShadow(effectiveSysRoot(config.context)); err == nil {
|
||||
slog.Info("SKIP_GPU enabled, using non-GPU sensor sysfs shadow", "path", shadow)
|
||||
config.sensorShadow = shadow
|
||||
config.context = context.WithValue(config.context,
|
||||
common.EnvKey, common.EnvMap{common.HostSysEnvKey: shadow},
|
||||
)
|
||||
} else {
|
||||
slog.Warn("SKIP_GPU sensor shadow unavailable, falling back to post-read filtering", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// handle blacklist
|
||||
if strings.HasPrefix(sensorsEnvVal, "-") {
|
||||
@@ -149,6 +168,9 @@ func (a *Agent) updateTemperatures(systemStats *system.Stats) {
|
||||
if !isValidSensor(sensorName, a.sensorConfig) {
|
||||
continue
|
||||
}
|
||||
if a.sensorConfig.skipGPU && isGpuSensorKey(sensorName) {
|
||||
continue
|
||||
}
|
||||
// set dashboard temperature
|
||||
switch a.sensorConfig.primarySensor {
|
||||
case "":
|
||||
@@ -245,3 +267,102 @@ func scaleTemperature(temp float64) float64 {
|
||||
}
|
||||
return scaled100
|
||||
}
|
||||
|
||||
// effectiveSysRoot mirrors gopsutil's HostSys lookup, which lives in its
|
||||
// internal package: context override, then HOST_SYS env, then /sys.
|
||||
func effectiveSysRoot(ctx context.Context) string {
|
||||
if envMap, ok := ctx.Value(common.EnvKey).(common.EnvMap); ok {
|
||||
if v := envMap[common.HostSysEnvKey]; v != "" {
|
||||
return v
|
||||
}
|
||||
}
|
||||
if v := os.Getenv("HOST_SYS"); v != "" {
|
||||
return v
|
||||
}
|
||||
return "/sys"
|
||||
}
|
||||
|
||||
func (config *SensorConfig) cleanupSensorShadow() {
|
||||
if config.sensorShadow == "" {
|
||||
return
|
||||
}
|
||||
if err := os.RemoveAll(config.sensorShadow); err != nil {
|
||||
slog.Warn("Error removing sensor sysfs shadow", "path", config.sensorShadow, "err", err)
|
||||
return
|
||||
}
|
||||
config.sensorShadow = ""
|
||||
}
|
||||
|
||||
func (a *Agent) cleanupSensorShadow() {
|
||||
if a.sensorConfig != nil {
|
||||
a.sensorConfig.cleanupSensorShadow()
|
||||
}
|
||||
}
|
||||
|
||||
func isGpuThermalZone(zoneType string) bool {
|
||||
zoneType = strings.ToLower(strings.TrimSpace(zoneType))
|
||||
return isGpuChipName(zoneType) || strings.Contains(zoneType, "gpu")
|
||||
}
|
||||
|
||||
// buildNonGpuSysShadow links non-GPU sensor directories into a temp dir. Only
|
||||
// static chip names and thermal-zone types are read; no sensor values are touched.
|
||||
func buildNonGpuSysShadow(sysRoot string) (string, error) {
|
||||
shadow, err := os.MkdirTemp("", "beszel-sensors-*")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
shadowHwmon := filepath.Join(shadow, "class", "hwmon")
|
||||
if err := os.MkdirAll(shadowHwmon, 0o755); err != nil {
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
entries, err := os.ReadDir(filepath.Join(sysRoot, "class", "hwmon"))
|
||||
if err != nil && !os.IsNotExist(err) {
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
for _, entry := range entries {
|
||||
chipDir := filepath.Join(sysRoot, "class", "hwmon", entry.Name())
|
||||
// Some hwmon devices expose name under device/ (gopsutil's CentOS fallback).
|
||||
name, ok := utils.ReadStringFileOK(filepath.Join(chipDir, "name"))
|
||||
if !ok {
|
||||
name, ok = utils.ReadStringFileOK(filepath.Join(chipDir, "device", "name"))
|
||||
}
|
||||
if !ok || isGpuChipName(name) {
|
||||
continue
|
||||
}
|
||||
if err := os.Symlink(chipDir, filepath.Join(shadowHwmon, entry.Name())); err != nil {
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
}
|
||||
|
||||
thermalEntries, err := os.ReadDir(filepath.Join(sysRoot, "class", "thermal"))
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return shadow, nil
|
||||
}
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
shadowThermal := filepath.Join(shadow, "class", "thermal")
|
||||
if err := os.MkdirAll(shadowThermal, 0o755); err != nil {
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
for _, entry := range thermalEntries {
|
||||
if !strings.HasPrefix(entry.Name(), "thermal_zone") {
|
||||
continue
|
||||
}
|
||||
zoneDir := filepath.Join(sysRoot, "class", "thermal", entry.Name())
|
||||
zoneType, ok := utils.ReadStringFileOK(filepath.Join(zoneDir, "type"))
|
||||
if !ok || isGpuThermalZone(zoneType) {
|
||||
continue
|
||||
}
|
||||
if err := os.Symlink(zoneDir, filepath.Join(shadowThermal, entry.Name())); err != nil {
|
||||
os.RemoveAll(shadow)
|
||||
return "", err
|
||||
}
|
||||
}
|
||||
return shadow, nil
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//go:build !windows
|
||||
//go:build !windows && !freebsd
|
||||
|
||||
package agent
|
||||
|
||||
|
||||
14
agent/sensors_freebsd.go
Normal file
14
agent/sensors_freebsd.go
Normal file
@@ -0,0 +1,14 @@
|
||||
//go:build freebsd
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/sensors"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
var getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
return getFreeBSDSensorTemps(ctx, unix.SysctlUint32)
|
||||
}
|
||||
81
agent/sensors_freebsd_common.go
Normal file
81
agent/sensors_freebsd_common.go
Normal file
@@ -0,0 +1,81 @@
|
||||
//go:build freebsd || testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"github.com/shirou/gopsutil/v4/sensors"
|
||||
)
|
||||
|
||||
const (
|
||||
freebsdZeroCelsiusDeciKelvin = 2731
|
||||
freebsdAcpiThermalZoneCount = 16
|
||||
)
|
||||
|
||||
type freebsdSysctlUintReader func(name string) (uint32, error)
|
||||
|
||||
func getFreeBSDSensorTemps(ctx context.Context, readSysctl freebsdSysctlUintReader) ([]sensors.TemperatureStat, error) {
|
||||
cpuCount, err := readSysctl("hw.ncpu")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
temps := make([]sensors.TemperatureStat, 0, int(cpuCount)+freebsdAcpiThermalZoneCount)
|
||||
for cpu := range cpuCount {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return temps, ctx.Err()
|
||||
default:
|
||||
}
|
||||
|
||||
sysctlName := fmt.Sprintf("dev.cpu.%d.temperature", cpu)
|
||||
value, err := readSysctl(sysctlName)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
temp, ok := freebsdDeciKelvinToCelsius(value)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
temps = append(temps, sensors.TemperatureStat{
|
||||
SensorKey: fmt.Sprintf("cpu.%d", cpu),
|
||||
Temperature: temp,
|
||||
})
|
||||
}
|
||||
|
||||
for zone := 0; zone < freebsdAcpiThermalZoneCount; zone++ {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return temps, ctx.Err()
|
||||
default:
|
||||
}
|
||||
|
||||
sysctlName := fmt.Sprintf("hw.acpi.thermal.tz%d.temperature", zone)
|
||||
value, err := readSysctl(sysctlName)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
temp, ok := freebsdDeciKelvinToCelsius(value)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
temps = append(temps, sensors.TemperatureStat{
|
||||
SensorKey: fmt.Sprintf("acpi.thermal.tz%d", zone),
|
||||
Temperature: temp,
|
||||
})
|
||||
}
|
||||
|
||||
return temps, nil
|
||||
}
|
||||
|
||||
func freebsdDeciKelvinToCelsius(value uint32) (float64, bool) {
|
||||
if value <= freebsdZeroCelsiusDeciKelvin {
|
||||
return 0, false
|
||||
}
|
||||
temp := float64(int64(value)-freebsdZeroCelsiusDeciKelvin) / 10
|
||||
if temp <= 0 || temp >= 200 {
|
||||
return 0, false
|
||||
}
|
||||
return temp, true
|
||||
}
|
||||
167
agent/sensors_freebsd_common_test.go
Normal file
167
agent/sensors_freebsd_common_test.go
Normal file
@@ -0,0 +1,167 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
var errFakeFreeBSDSysctlNotFound = errors.New("sysctl not found")
|
||||
|
||||
type fakeFreeBSDSysctls struct {
|
||||
values map[string]uint32
|
||||
errs map[string]error
|
||||
}
|
||||
|
||||
func (f fakeFreeBSDSysctls) read(name string) (uint32, error) {
|
||||
if err, ok := f.errs[name]; ok {
|
||||
return 0, err
|
||||
}
|
||||
if value, ok := f.values[name]; ok {
|
||||
return value, nil
|
||||
}
|
||||
return 0, errFakeFreeBSDSysctlNotFound
|
||||
}
|
||||
|
||||
func TestFreeBSDDeciKelvinToCelsius(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
value uint32
|
||||
expected float64
|
||||
ok bool
|
||||
}{
|
||||
{
|
||||
name: "45 Celsius",
|
||||
value: 3181,
|
||||
expected: 45,
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "fractional Celsius",
|
||||
value: 3186,
|
||||
expected: 45.5,
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "zero deci-Kelvin",
|
||||
value: 0,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "zero Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "below zero Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin - 1,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "invalid signed integer",
|
||||
value: 1<<32 - 1,
|
||||
ok: false,
|
||||
},
|
||||
{
|
||||
name: "unreasonably high Celsius",
|
||||
value: freebsdZeroCelsiusDeciKelvin + 2000,
|
||||
ok: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result, ok := freebsdDeciKelvinToCelsius(tt.value)
|
||||
assert.Equal(t, tt.ok, ok)
|
||||
assert.InDelta(t, tt.expected, result, 0.001)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTemps(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{
|
||||
"hw.ncpu": 4,
|
||||
"dev.cpu.0.temperature": 3231,
|
||||
"dev.cpu.1.temperature": 3242,
|
||||
"dev.cpu.3.temperature": freebsdZeroCelsiusDeciKelvin,
|
||||
"hw.acpi.thermal.tz0.temperature": 3101,
|
||||
"hw.acpi.thermal.tz2.temperature": 3116,
|
||||
"hw.acpi.thermal.tz3.temperature": freebsdZeroCelsiusDeciKelvin,
|
||||
"unrelated.sensor.value": 9999,
|
||||
"dev.cpu.99.temperature": 9999,
|
||||
"dev.amdtemp.0.core0.foo": 9999,
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
require.Len(t, temps, 4)
|
||||
assert.Equal(t, "cpu.0", temps[0].SensorKey)
|
||||
assert.InDelta(t, 50.0, temps[0].Temperature, 0.001)
|
||||
assert.Equal(t, "cpu.1", temps[1].SensorKey)
|
||||
assert.InDelta(t, 51.1, temps[1].Temperature, 0.001)
|
||||
assert.Equal(t, "acpi.thermal.tz0", temps[2].SensorKey)
|
||||
assert.InDelta(t, 37.0, temps[2].Temperature, 0.001)
|
||||
assert.Equal(t, "acpi.thermal.tz2", temps[3].SensorKey)
|
||||
assert.InDelta(t, 38.5, temps[3].Temperature, 0.001)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsCpuCountError(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
errs: map[string]error{
|
||||
"hw.ncpu": errors.New("permission denied"),
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
assert.Nil(t, temps)
|
||||
assert.EqualError(t, err, "permission denied")
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsNoTemperatureSysctls(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{"hw.ncpu": 2},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, temps)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsAcpiOnly(t *testing.T) {
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{
|
||||
"hw.ncpu": 0,
|
||||
"hw.acpi.thermal.tz0.temperature": 3081,
|
||||
},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(context.Background(), reader.read)
|
||||
|
||||
require.NoError(t, err)
|
||||
require.Len(t, temps, 1)
|
||||
assert.Equal(t, "acpi.thermal.tz0", temps[0].SensorKey)
|
||||
assert.InDelta(t, 35.0, temps[0].Temperature, 0.001)
|
||||
}
|
||||
|
||||
func TestGetFreeBSDSensorTempsContextCancelled(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
reader := fakeFreeBSDSysctls{
|
||||
values: map[string]uint32{"hw.ncpu": 2},
|
||||
}
|
||||
|
||||
temps, err := getFreeBSDSensorTemps(ctx, reader.read)
|
||||
|
||||
assert.Empty(t, temps)
|
||||
assert.ErrorIs(t, err, context.Canceled)
|
||||
}
|
||||
@@ -5,6 +5,8 @@ package agent
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -328,7 +330,7 @@ func TestNewSensorConfigWithEnv(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := agent.newSensorConfigWithEnv(tt.primarySensor, tt.sysSensors, tt.sensors, tt.sensorsTimeout, tt.skipCollection)
|
||||
result := agent.newSensorConfigWithEnv(tt.primarySensor, tt.sysSensors, tt.sensors, tt.sensorsTimeout, tt.skipCollection, false)
|
||||
|
||||
// Check primary sensor
|
||||
assert.Equal(t, tt.expectedConfig.primarySensor, result.primarySensor)
|
||||
@@ -602,8 +604,9 @@ func TestUpdateTemperaturesSkipsOnTimeout(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
originalGetSensorTemps := getSensorTemps
|
||||
t.Cleanup(func() {
|
||||
getSensorTemps = sensors.TemperaturesWithContext
|
||||
getSensorTemps = originalGetSensorTemps
|
||||
})
|
||||
getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
@@ -619,3 +622,143 @@ func TestUpdateTemperaturesSkipsOnTimeout(t *testing.T) {
|
||||
assert.Equal(t, 0.0, agent.systemInfo.DashboardTemp)
|
||||
assert.Equal(t, map[string]float64{}, stats.Temperatures)
|
||||
}
|
||||
|
||||
func TestIsGpuSensorKey(t *testing.T) {
|
||||
for _, key := range []string{"xe", "XE_temp1", "amdgpu_edge", "NVIDIA"} {
|
||||
assert.True(t, isGpuSensorKey(key), key)
|
||||
}
|
||||
for _, key := range []string{"coretemp_core_0", "acpitz", "xen_temp", "myxe", ""} {
|
||||
assert.False(t, isGpuSensorKey(key), key)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSkipGpuSensorShadow(t *testing.T) {
|
||||
sysRoot := t.TempDir()
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "name"), "coretemp\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "temp1_input"), "55000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "name"), "xe\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "temp1_input"), "48000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone0", "type"), "cpu-thermal\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone0", "temp"), "55000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone1", "type"), "gpu\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone1", "temp"), "48000\n")
|
||||
|
||||
shadow, err := buildNonGpuSysShadow(sysRoot)
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { os.RemoveAll(shadow) })
|
||||
|
||||
assert.FileExists(t, filepath.Join(shadow, "class", "hwmon", "hwmon0", "temp1_input"))
|
||||
assert.NoFileExists(t, filepath.Join(shadow, "class", "hwmon", "hwmon1"))
|
||||
assert.FileExists(t, filepath.Join(shadow, "class", "thermal", "thermal_zone0", "temp"))
|
||||
assert.NoFileExists(t, filepath.Join(shadow, "class", "thermal", "thermal_zone1"))
|
||||
}
|
||||
|
||||
func TestSkipGpuSensorShadowDeviceName(t *testing.T) {
|
||||
sysRoot := t.TempDir()
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "device", "name"), "coretemp\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "device", "temp1_input"), "55000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "device", "name"), "xe\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "device", "temp1_input"), "48000\n")
|
||||
|
||||
shadow, err := buildNonGpuSysShadow(sysRoot)
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { os.RemoveAll(shadow) })
|
||||
|
||||
assert.FileExists(t, filepath.Join(shadow, "class", "hwmon", "hwmon0", "device", "temp1_input"))
|
||||
assert.NoFileExists(t, filepath.Join(shadow, "class", "hwmon", "hwmon1"))
|
||||
}
|
||||
|
||||
func TestSkipGpuSensorShadowKeepsThermalZonesWithoutNonGpuHwmon(t *testing.T) {
|
||||
sysRoot := t.TempDir()
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "name"), "xe\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "temp1_input"), "48000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone0", "type"), "cpu-thermal\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone0", "temp"), "55000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone1", "type"), "gpu\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "thermal", "thermal_zone1", "temp"), "48000\n")
|
||||
|
||||
shadow, err := buildNonGpuSysShadow(sysRoot)
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { os.RemoveAll(shadow) })
|
||||
|
||||
hwmonTemps, err := filepath.Glob(filepath.Join(shadow, "class", "hwmon", "hwmon*", "temp*_input"))
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, hwmonTemps)
|
||||
assert.FileExists(t, filepath.Join(shadow, "class", "thermal", "thermal_zone0", "temp"))
|
||||
assert.NoFileExists(t, filepath.Join(shadow, "class", "thermal", "thermal_zone1"))
|
||||
}
|
||||
|
||||
func TestNewSensorConfigSkipGpuWiresShadow(t *testing.T) {
|
||||
t.Setenv("SKIP_GPU", "true")
|
||||
|
||||
agent := &Agent{}
|
||||
config := agent.newSensorConfig()
|
||||
|
||||
assert.True(t, config.skipGPU)
|
||||
envMap, ok := config.context.Value(common.EnvKey).(common.EnvMap)
|
||||
require.True(t, ok, "SKIP_GPU should point the sensor context at a sysfs shadow")
|
||||
shadow, ok := envMap[common.HostSysEnvKey]
|
||||
require.True(t, ok)
|
||||
assert.DirExists(t, filepath.Join(shadow, "class", "hwmon"))
|
||||
assert.Equal(t, shadow, config.sensorShadow)
|
||||
config.cleanupSensorShadow()
|
||||
assert.NoDirExists(t, shadow)
|
||||
assert.Empty(t, config.sensorShadow)
|
||||
}
|
||||
|
||||
func TestSkipGpuShadowUsesSysSensorsRoot(t *testing.T) {
|
||||
sysRoot := t.TempDir()
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "name"), "coretemp\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0", "temp1_input"), "55000\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "name"), "xe\n")
|
||||
writeFile(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon1", "temp1_input"), "48000\n")
|
||||
|
||||
agent := &Agent{}
|
||||
config := agent.newSensorConfigWithEnv("", sysRoot, "", "", false, true)
|
||||
t.Cleanup(config.cleanupSensorShadow)
|
||||
|
||||
envMap, ok := config.context.Value(common.EnvKey).(common.EnvMap)
|
||||
require.True(t, ok, "SKIP_GPU should point the sensor context at a sysfs shadow")
|
||||
shadow, ok := envMap[common.HostSysEnvKey]
|
||||
require.True(t, ok)
|
||||
require.NotEqual(t, sysRoot, shadow, "shadow must not be the SYS_SENSORS tree itself")
|
||||
|
||||
target, err := os.Readlink(filepath.Join(shadow, "class", "hwmon", "hwmon0"))
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, filepath.Join(sysRoot, "class", "hwmon", "hwmon0"), target)
|
||||
assert.NoFileExists(t, filepath.Join(shadow, "class", "hwmon", "hwmon1"))
|
||||
}
|
||||
|
||||
func TestUpdateTemperaturesSkipGpu(t *testing.T) {
|
||||
originalGetSensorTemps := getSensorTemps
|
||||
t.Cleanup(func() {
|
||||
getSensorTemps = originalGetSensorTemps
|
||||
})
|
||||
getSensorTemps = func(ctx context.Context) ([]sensors.TemperatureStat, error) {
|
||||
return []sensors.TemperatureStat{
|
||||
{SensorKey: "coretemp_core_0", Temperature: 55},
|
||||
{SensorKey: "XE", Temperature: 48},
|
||||
}, nil
|
||||
}
|
||||
|
||||
newAgent := func(skipGPU bool) *Agent {
|
||||
agent := &Agent{
|
||||
systemInfo: system.Info{},
|
||||
sensorConfig: &SensorConfig{
|
||||
context: context.Background(),
|
||||
timeout: 2 * time.Second,
|
||||
sensors: map[string]struct{}{},
|
||||
skipGPU: skipGPU,
|
||||
},
|
||||
}
|
||||
return agent
|
||||
}
|
||||
|
||||
stats := &system.Stats{}
|
||||
newAgent(true).updateTemperatures(stats)
|
||||
assert.Equal(t, map[string]float64{"coretemp_core_0": 55}, stats.Temperatures)
|
||||
|
||||
stats = &system.Stats{}
|
||||
newAgent(false).updateTemperatures(stats)
|
||||
assert.Len(t, stats.Temperatures, 2)
|
||||
}
|
||||
|
||||
@@ -214,9 +214,12 @@ func (lhm *lhmProcess) getTemps(ctx context.Context) (temps []sensors.Temperatur
|
||||
return temps, nil
|
||||
}
|
||||
|
||||
// getSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
|
||||
// getSensorTemps is a variable so tests can replace the platform sensor collector.
|
||||
var getSensorTemps = getWindowsSensorTemps
|
||||
|
||||
// getWindowsSensorTemps attempts to pull sensor temperatures from the embedded LHM process.
|
||||
// NB: LibreHardwareMonitorLib requires admin privileges to access all available sensors.
|
||||
func getSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
|
||||
func getWindowsSensorTemps(ctx context.Context) (temps []sensors.TemperatureStat, err error) {
|
||||
defer func() {
|
||||
if err != nil {
|
||||
slog.Debug("Error reading sensors", "err", err)
|
||||
|
||||
@@ -29,9 +29,6 @@ type ServerOptions struct {
|
||||
Keys []gossh.PublicKey // SSH public keys for authentication
|
||||
}
|
||||
|
||||
// hubVersions caches hub versions by session ID to avoid repeated parsing.
|
||||
var hubVersions map[string]semver.Version
|
||||
|
||||
// StartServer starts the SSH server with the provided options.
|
||||
// It configures the server with secure defaults, sets up authentication,
|
||||
// and begins listening for connections. Returns an error if the server
|
||||
@@ -99,24 +96,15 @@ func (a *Agent) StartServer(opts ServerOptions) error {
|
||||
return a.server.Serve(ln)
|
||||
}
|
||||
|
||||
// getHubVersion retrieves and caches the hub version for a given session.
|
||||
// It extracts the version from the SSH client version string and caches
|
||||
// it to avoid repeated parsing. Returns a zero version if parsing fails.
|
||||
func (a *Agent) getHubVersion(sessionId string, sessionCtx ssh.Context) semver.Version {
|
||||
if hubVersions == nil {
|
||||
hubVersions = make(map[string]semver.Version, 1)
|
||||
}
|
||||
hubVersion, ok := hubVersions[sessionId]
|
||||
if ok {
|
||||
return hubVersion
|
||||
}
|
||||
// Extract hub version from SSH client version
|
||||
// getHubVersion extracts the hub version from the SSH client version string
|
||||
// for a given session. Returns a zero version if parsing fails.
|
||||
func (a *Agent) getHubVersion(sessionCtx ssh.Context) semver.Version {
|
||||
clientVersion := sessionCtx.Value(ssh.ContextKeyClientVersion)
|
||||
if versionStr, ok := clientVersion.(string); ok {
|
||||
hubVersion, _ = extractHubVersion(versionStr)
|
||||
hubVersion, _ := extractHubVersion(versionStr)
|
||||
return hubVersion
|
||||
}
|
||||
hubVersions[sessionId] = hubVersion
|
||||
return hubVersion
|
||||
return semver.Version{}
|
||||
}
|
||||
|
||||
// handleSession handles an incoming SSH session by gathering system statistics
|
||||
@@ -127,9 +115,8 @@ func (a *Agent) handleSession(s ssh.Session) {
|
||||
a.connectionManager.eventChan <- SSHConnect
|
||||
|
||||
sessionCtx := s.Context()
|
||||
sessionID := sessionCtx.SessionID()
|
||||
|
||||
hubVersion := a.getHubVersion(sessionID, sessionCtx)
|
||||
hubVersion := a.getHubVersion(sessionCtx)
|
||||
|
||||
// Legacy one-shot behavior for older hubs
|
||||
if hubVersion.LT(beszel.MinVersionAgentResponse) {
|
||||
@@ -278,6 +265,5 @@ func (a *Agent) StopServer() error {
|
||||
slog.Info("Stopping SSH server")
|
||||
_ = a.server.Close()
|
||||
a.server = nil
|
||||
a.connectionManager.eventChan <- SSHDisconnect
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -198,6 +198,28 @@ func TestStartServerDisableSSH(t *testing.T) {
|
||||
assert.Contains(t, err.Error(), "SSH disabled")
|
||||
}
|
||||
|
||||
func TestStopServerDoesNotBlockWhenEventQueueFull(t *testing.T) {
|
||||
agent := createTestAgent(t)
|
||||
agent.server = &ssh.Server{}
|
||||
agent.connectionManager.eventChan = make(chan ConnectionEvent, 1)
|
||||
agent.connectionManager.eventChan <- WebSocketConnect
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- agent.StopServer()
|
||||
}()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
require.NoError(t, err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("StopServer blocked on the connection event queue")
|
||||
}
|
||||
|
||||
assert.Nil(t, agent.server)
|
||||
assert.Equal(t, WebSocketConnect, <-agent.connectionManager.eventChan)
|
||||
}
|
||||
|
||||
/////////////////////////////////////////////////////////////////
|
||||
//////////////////// ParseKeys Tests ////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////
|
||||
@@ -404,27 +426,23 @@ func TestGetHubVersion(t *testing.T) {
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
}
|
||||
|
||||
// Test first call - should extract and cache version
|
||||
version := agent.getHubVersion("test-session-123", mockCtx)
|
||||
// Test first call - should extract version
|
||||
version := agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.12.0", version.String())
|
||||
|
||||
// Test second call - should return cached version
|
||||
mockCtx.clientVersion = "SSH-2.0-beszel_0.11.0" // Change version but should still return cached
|
||||
version = agent.getHubVersion("test-session-123", mockCtx)
|
||||
assert.Equal(t, "0.12.0", version.String()) // Should still be cached version
|
||||
|
||||
// Test different session - should extract new version
|
||||
version = agent.getHubVersion("different-session", mockCtx)
|
||||
// Test that version reflects the current client version (no stale caching)
|
||||
mockCtx.clientVersion = "SSH-2.0-beszel_0.11.0"
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.11.0", version.String())
|
||||
|
||||
// Test with invalid version string (non-beszel client)
|
||||
mockCtx.clientVersion = "SSH-2.0-OpenSSH_8.0"
|
||||
version = agent.getHubVersion("invalid-session", mockCtx)
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.Equal(t, "0.0.0", version.String()) // Should be empty version for non-beszel clients
|
||||
|
||||
// Test with no client version
|
||||
mockCtx.clientVersion = ""
|
||||
version = agent.getHubVersion("no-version-session", mockCtx)
|
||||
version = agent.getHubVersion(mockCtx)
|
||||
assert.True(t, version.EQ(semver.Version{})) // Should be empty version
|
||||
}
|
||||
|
||||
@@ -501,9 +519,6 @@ func TestWriteToSessionEncoding(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
// Reset the global hubVersions map to ensure clean state for each test
|
||||
hubVersions = nil
|
||||
|
||||
agent, err := NewAgent("")
|
||||
require.NoError(t, err)
|
||||
|
||||
@@ -585,39 +600,28 @@ func createTestCombinedData() *system.CombinedData {
|
||||
}
|
||||
}
|
||||
|
||||
func TestHubVersionCaching(t *testing.T) {
|
||||
// Reset the global hubVersions map to ensure clean state
|
||||
hubVersions = nil
|
||||
|
||||
// TestGetHubVersionConcurrent guards against a regression of the
|
||||
// "concurrent map writes" panic previously caused by a shared, unsynchronized
|
||||
// hubVersions cache (see https://github.com/henrygd/beszel/issues/2128).
|
||||
// getHubVersion no longer shares mutable state between sessions, so calling
|
||||
// it concurrently from many goroutines must be safe under `go test -race`.
|
||||
func TestGetHubVersionConcurrent(t *testing.T) {
|
||||
agent, err := NewAgent("")
|
||||
require.NoError(t, err)
|
||||
|
||||
ctx1 := &mockSSHContext{
|
||||
sessionID: "session1",
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
const goroutines = 50
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(goroutines)
|
||||
for i := 0; i < goroutines; i++ {
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
ctx := &mockSSHContext{
|
||||
sessionID: fmt.Sprintf("session-%d", i),
|
||||
clientVersion: "SSH-2.0-beszel_0.12.0",
|
||||
}
|
||||
version := agent.getHubVersion(ctx)
|
||||
assert.Equal(t, "0.12.0", version.String())
|
||||
}(i)
|
||||
}
|
||||
ctx2 := &mockSSHContext{
|
||||
sessionID: "session2",
|
||||
clientVersion: "SSH-2.0-beszel_0.11.0",
|
||||
}
|
||||
|
||||
// First calls should cache the versions
|
||||
v1 := agent.getHubVersion("session1", ctx1)
|
||||
v2 := agent.getHubVersion("session2", ctx2)
|
||||
|
||||
assert.Equal(t, "0.12.0", v1.String())
|
||||
assert.Equal(t, "0.11.0", v2.String())
|
||||
|
||||
// Verify caching by changing context but keeping same session ID
|
||||
ctx1.clientVersion = "SSH-2.0-beszel_0.10.0"
|
||||
v1Cached := agent.getHubVersion("session1", ctx1)
|
||||
assert.Equal(t, "0.12.0", v1Cached.String()) // Should still be cached version
|
||||
|
||||
// New session should get new version
|
||||
ctx3 := &mockSSHContext{
|
||||
sessionID: "session3",
|
||||
clientVersion: "SSH-2.0-beszel_0.13.0",
|
||||
}
|
||||
v3 := agent.getHubVersion("session3", ctx3)
|
||||
assert.Equal(t, "0.13.0", v3.String())
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
@@ -55,6 +55,11 @@ type DeviceInfo struct {
|
||||
typeVerified bool
|
||||
// parserType holds the parser type (nvme, sat, scsi) that last succeeded.
|
||||
parserType string
|
||||
// explicitType reports whether Type came from an explicit ":type" hint in
|
||||
// SMART_DEVICES. Such a type is a deliberate user override and must always be
|
||||
// passed to smartctl via -d, even for scsi/ata where a scan-detected type is
|
||||
// otherwise left off (see smartctlArgs and issue #1345).
|
||||
explicitType bool
|
||||
}
|
||||
|
||||
// deviceKey is a composite key for a device, used to identify a device uniquely.
|
||||
@@ -65,8 +70,9 @@ type deviceKey struct {
|
||||
|
||||
var errNoValidSmartData = fmt.Errorf("no valid SMART data found") // Error for missing data
|
||||
|
||||
// Refresh updates SMART data for all known devices
|
||||
func (sm *SmartManager) Refresh(forceScan bool) error {
|
||||
// Refresh updates SMART data for all known devices and reports whether every
|
||||
// discovered device was collected successfully.
|
||||
func (sm *SmartManager) Refresh(forceScan bool) (bool, error) {
|
||||
sm.refreshMutex.Lock()
|
||||
defer sm.refreshMutex.Unlock()
|
||||
|
||||
@@ -87,7 +93,7 @@ func (sm *SmartManager) Refresh(forceScan bool) error {
|
||||
}
|
||||
}
|
||||
|
||||
return sm.resolveRefreshError(scanErr, collectErr)
|
||||
return scanErr == nil && collectErr == nil, sm.resolveRefreshError(scanErr, collectErr)
|
||||
}
|
||||
|
||||
// devicesSnapshot returns a copy of the current device slice to avoid iterating
|
||||
@@ -251,8 +257,9 @@ func (sm *SmartManager) parseConfiguredDevices(config string) ([]*DeviceInfo, er
|
||||
}
|
||||
|
||||
devices = append(devices, &DeviceInfo{
|
||||
Name: name,
|
||||
Type: devType,
|
||||
Name: name,
|
||||
Type: devType,
|
||||
explicitType: devType != "",
|
||||
})
|
||||
}
|
||||
|
||||
@@ -368,9 +375,15 @@ func (sm *SmartManager) parseSmartOutput(deviceInfo *DeviceInfo, output []byte)
|
||||
Type string
|
||||
Parse func([]byte) (bool, int)
|
||||
}{
|
||||
{Type: "nvme", Parse: sm.parseSmartForNvme},
|
||||
{Type: "sat", Parse: sm.parseSmartForSata},
|
||||
{Type: "scsi", Parse: sm.parseSmartForScsi},
|
||||
{Type: "nvme", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForNvme(output, deviceInfo.Type)
|
||||
}},
|
||||
{Type: "sat", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForSata(output, deviceInfo.Type)
|
||||
}},
|
||||
{Type: "scsi", Parse: func(output []byte) (bool, int) {
|
||||
return sm.parseSmartForScsi(output, deviceInfo.Type)
|
||||
}},
|
||||
}
|
||||
|
||||
deviceType := normalizeParserType(deviceInfo.parserType)
|
||||
@@ -479,10 +492,11 @@ func (sm *SmartManager) CollectSmart(deviceInfo *DeviceInfo) error {
|
||||
return errNoValidSmartData
|
||||
}
|
||||
|
||||
// slog.Info("collecting SMART data", "device", deviceInfo.Name, "type", deviceInfo.Type, "has_existing_data", sm.hasDataForDevice(deviceInfo.Name))
|
||||
// slog.Info("collecting SMART data", "device", deviceInfo.Name, "type", deviceInfo.Type, "has_existing_data", sm.hasDataForDevice(deviceInfo))
|
||||
|
||||
// Check if we have any existing data for this device
|
||||
hasExistingData := sm.hasDataForDevice(deviceInfo.Name)
|
||||
// Check if we have existing data for this exact device identity. Multiple
|
||||
// bridge slots can share a path, so a name-only match is not sufficient.
|
||||
hasExistingData := sm.hasDataForDevice(deviceInfo)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
@@ -558,7 +572,9 @@ func (sm *SmartManager) smartctlArgs(deviceInfo *DeviceInfo, includeStandby bool
|
||||
deviceType = strings.ToLower(deviceInfo.Type)
|
||||
parserType = strings.ToLower(deviceInfo.parserType)
|
||||
// types sometimes misidentified in scan; see github.com/henrygd/beszel/issues/1345
|
||||
if deviceType != "" && deviceType != "scsi" && deviceType != "ata" {
|
||||
// An explicit SMART_DEVICES ":type" hint is a deliberate override, so always
|
||||
// pass it through; otherwise scsi/ata are left off so smartctl can auto-detect.
|
||||
if deviceType != "" && (deviceInfo.explicitType || (deviceType != "scsi" && deviceType != "ata")) {
|
||||
args = append(args, "-d", deviceInfo.Type)
|
||||
}
|
||||
}
|
||||
@@ -583,14 +599,18 @@ func (sm *SmartManager) smartctlArgs(deviceInfo *DeviceInfo, includeStandby bool
|
||||
return args
|
||||
}
|
||||
|
||||
// hasDataForDevice checks if we have cached SMART data for a specific device
|
||||
func (sm *SmartManager) hasDataForDevice(deviceName string) bool {
|
||||
// hasDataForDevice checks if we have cached SMART data for a specific device identity.
|
||||
func (sm *SmartManager) hasDataForDevice(deviceInfo *DeviceInfo) bool {
|
||||
if deviceInfo == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
sm.Lock()
|
||||
defer sm.Unlock()
|
||||
|
||||
// Check if any cached data has this device name
|
||||
deviceKey := makeDeviceKey(deviceInfo.Name, deviceInfo.Type)
|
||||
for _, data := range sm.SmartDataMap {
|
||||
if data != nil && data.DiskName == deviceName {
|
||||
if data != nil && makeDeviceKey(data.DiskName, data.DiskType) == deviceKey {
|
||||
return true
|
||||
}
|
||||
}
|
||||
@@ -663,6 +683,9 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
target.Type = prev.Type
|
||||
target.typeVerified = true
|
||||
target.parserType = prev.parserType
|
||||
if prev.explicitType {
|
||||
target.explicitType = true
|
||||
}
|
||||
}
|
||||
|
||||
// applyConfiguredMetadata updates a matched device with any configured
|
||||
@@ -676,6 +699,9 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
existingDev.typeVerified = false
|
||||
existingDev.parserType = normalizeParserType(newType)
|
||||
}
|
||||
if configuredDev.explicitType {
|
||||
existingDev.explicitType = true
|
||||
}
|
||||
if configuredDev.InfoName != "" {
|
||||
existingDev.InfoName = configuredDev.InfoName
|
||||
}
|
||||
@@ -732,7 +758,14 @@ func mergeDeviceLists(existing, scanned, configured []*DeviceInfo) []*DeviceInfo
|
||||
continue
|
||||
}
|
||||
if existingDev := deviceIndexByName[configuredDevice.Name]; existingDev != nil {
|
||||
oldKey := makeDeviceKey(existingDev.Name, existingDev.Type)
|
||||
if prev := existingIndex[key]; prev != nil {
|
||||
preserveVerifiedType(existingDev, prev)
|
||||
}
|
||||
applyConfiguredMetadata(existingDev, configuredDevice)
|
||||
delete(deviceIndex, oldKey)
|
||||
deviceIndex[makeDeviceKey(existingDev.Name, existingDev.Type)] = existingDev
|
||||
delete(deviceIndexByName, configuredDevice.Name)
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -836,9 +869,11 @@ func (sm *SmartManager) isVirtualDeviceFromStrings(fields ...string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// parseSmartForSata parses the output of smartctl --all -j for SATA/ATA devices and updates the SmartDataMap
|
||||
// parseSmartForSata parses the output of smartctl --all -j for SATA/ATA devices and updates the SmartDataMap.
|
||||
// deviceType is the exact type used to identify and query the device; when set,
|
||||
// it takes precedence over the generic type reported by smartctl.
|
||||
// Returns hasValidData and exitStatus
|
||||
func (sm *SmartManager) parseSmartForSata(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForSata(output []byte, deviceType string) (bool, int) {
|
||||
var data smart.SmartInfoForSata
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -877,6 +912,9 @@ func (sm *SmartManager) parseSmartForSata(output []byte) (bool, int) {
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
// get values from ata_device_statistics if necessary
|
||||
var ataDeviceStats smart.AtaDeviceStatistics
|
||||
@@ -950,7 +988,7 @@ func findAtaDeviceStatisticsValue(data *smart.SmartInfoForSata, ataDeviceStats *
|
||||
return nil
|
||||
}
|
||||
|
||||
func (sm *SmartManager) parseSmartForScsi(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForScsi(output []byte, deviceType string) (bool, int) {
|
||||
var data smart.SmartInfoForScsi
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -985,6 +1023,9 @@ func (sm *SmartManager) parseSmartForScsi(output []byte) (bool, int) {
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
attributes := make([]*smart.SmartAttribute, 0, 10)
|
||||
attributes = append(attributes, &smart.SmartAttribute{Name: "PowerOnHours", RawValue: data.PowerOnTime.Hours})
|
||||
@@ -1082,9 +1123,11 @@ func (sm *SmartManager) lookupDarwinNvmeCapacity(serial string) uint64 {
|
||||
return sm.darwinNvmeCapacity[serial]
|
||||
}
|
||||
|
||||
// parseSmartForNvme parses the output of smartctl --all -j /dev/nvmeX and updates the SmartDataMap
|
||||
// parseSmartForNvme parses the output of smartctl --all -j /dev/nvmeX and updates the SmartDataMap.
|
||||
// deviceType is the exact type used to identify and query the device; when set,
|
||||
// it takes precedence over the generic type reported by smartctl.
|
||||
// Returns hasValidData and exitStatus
|
||||
func (sm *SmartManager) parseSmartForNvme(output []byte) (bool, int) {
|
||||
func (sm *SmartManager) parseSmartForNvme(output []byte, deviceType string) (bool, int) {
|
||||
data := &smart.SmartInfoForNvme{}
|
||||
|
||||
if err := json.Unmarshal(output, &data); err != nil {
|
||||
@@ -1128,6 +1171,9 @@ func (sm *SmartManager) parseSmartForNvme(output []byte) (bool, int) {
|
||||
smartData.SmartStatus = getSmartStatus(smartData.Temperature, data.SmartStatus.Passed)
|
||||
smartData.DiskName = data.Device.Name
|
||||
smartData.DiskType = data.Device.Type
|
||||
if deviceType != "" {
|
||||
smartData.DiskType = deviceType
|
||||
}
|
||||
|
||||
// nvme attributes does not follow the same format as ata attributes,
|
||||
// so we manually map each field to SmartAttributes
|
||||
|
||||
@@ -4,6 +4,7 @@ package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
@@ -24,7 +25,7 @@ func TestParseSmartForScsi(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForScsi(data)
|
||||
hasData, exitStatus := sm.parseSmartForScsi(data, "")
|
||||
if !hasData {
|
||||
t.Fatalf("expected SCSI data to parse successfully")
|
||||
}
|
||||
@@ -69,7 +70,7 @@ func TestParseSmartForSata(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForSata(data)
|
||||
hasData, exitStatus := sm.parseSmartForSata(data, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 64, exitStatus)
|
||||
|
||||
@@ -88,6 +89,31 @@ func TestParseSmartForSata(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForSataPreservesFailedAndUnknownStatus(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
temperature int
|
||||
want string
|
||||
}{
|
||||
{name: "failed", temperature: 30, want: "FAILED"},
|
||||
{name: "unknown", want: "UNKNOWN"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
jsonPayload := []byte(fmt.Sprintf(`{
|
||||
"device": {"name": "/dev/sda", "type": "sat"},
|
||||
"serial_number": "PRESERVE%s",
|
||||
"temperature": {"current": %d},
|
||||
"ata_smart_attributes": {"table": [{"id": 197, "raw": {"value": 1, "string": "1"}}]}
|
||||
}`, test.name, test.temperature))
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, test.want, sm.SmartDataMap["PRESERVE"+test.name].SmartStatus)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
|
||||
jsonPayload := []byte(`{
|
||||
"smartctl": {"exit_status": 0},
|
||||
@@ -112,7 +138,7 @@ func TestParseSmartForSataDeviceStatisticsTemperature(t *testing.T) {
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -147,7 +173,7 @@ func TestParseSmartForSataAtaDeviceStatistics(t *testing.T) {
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -184,7 +210,7 @@ func TestParseSmartForSataNegativeDeviceStatistics(t *testing.T) {
|
||||
}`)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -223,7 +249,7 @@ func TestParseSmartForSataParentheticalRawValue(t *testing.T) {
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload)
|
||||
hasData, exitStatus := sm.parseSmartForSata(jsonPayload, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -245,7 +271,7 @@ func TestParseSmartForNvme(t *testing.T) {
|
||||
SmartDataMap: make(map[string]*smart.SmartData),
|
||||
}
|
||||
|
||||
hasData, exitStatus := sm.parseSmartForNvme(data)
|
||||
hasData, exitStatus := sm.parseSmartForNvme(data, "")
|
||||
require.True(t, hasData)
|
||||
assert.Equal(t, 0, exitStatus)
|
||||
|
||||
@@ -268,13 +294,15 @@ func TestParseSmartForNvme(t *testing.T) {
|
||||
func TestHasDataForDevice(t *testing.T) {
|
||||
sm := &SmartManager{
|
||||
SmartDataMap: map[string]*smart.SmartData{
|
||||
"serial-1": {DiskName: "/dev/sda"},
|
||||
"serial-1": {DiskName: "/dev/sda", DiskType: "jms56x,0"},
|
||||
"serial-2": nil,
|
||||
},
|
||||
}
|
||||
|
||||
assert.True(t, sm.hasDataForDevice("/dev/sda"))
|
||||
assert.False(t, sm.hasDataForDevice("/dev/sdb"))
|
||||
assert.True(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sda", Type: "jms56x,0"}))
|
||||
assert.False(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sda", Type: "jms56x,1"}))
|
||||
assert.False(t, sm.hasDataForDevice(&DeviceInfo{Name: "/dev/sdb", Type: "jms56x,0"}))
|
||||
assert.False(t, sm.hasDataForDevice(nil))
|
||||
}
|
||||
|
||||
func TestDevicesSnapshotReturnsCopy(t *testing.T) {
|
||||
@@ -392,6 +420,81 @@ func TestSmartctlArgs(t *testing.T) {
|
||||
)
|
||||
}
|
||||
|
||||
// TestSmartctlArgsExplicitType verifies that an explicit SMART_DEVICES type hint
|
||||
// is always passed to smartctl via -d, while a scan-detected scsi/ata type is
|
||||
// still left off so smartctl can auto-detect it (see issue #1345).
|
||||
func TestSmartctlArgsExplicitType(t *testing.T) {
|
||||
sm := &SmartManager{}
|
||||
|
||||
// Scan-detected scsi: -d is intentionally omitted.
|
||||
scanScsi := &DeviceInfo{Name: "/dev/sda", Type: "scsi"}
|
||||
assert.Equal(t,
|
||||
[]string{"-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(scanScsi, false),
|
||||
)
|
||||
|
||||
// Explicit scsi from SMART_DEVICES: -d scsi must be passed.
|
||||
explicitScsi := &DeviceInfo{Name: "/dev/sda", Type: "scsi", explicitType: true}
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "scsi", "-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(explicitScsi, false),
|
||||
)
|
||||
|
||||
// Explicit ata from SMART_DEVICES: -d ata must be passed (devstat still added).
|
||||
explicitAta := &DeviceInfo{Name: "/dev/sdb", Type: "ata", explicitType: true}
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "ata", "-a", "--json=c", "-l", "devstat", "/dev/sdb"},
|
||||
sm.smartctlArgs(explicitAta, false),
|
||||
)
|
||||
}
|
||||
|
||||
// TestSmartDevicesExplicitTypeFlowsToSmartctlArgs is a regression test for
|
||||
// issue #2072: an explicit SMART_DEVICES type (e.g. /dev/sda:scsi) must win over
|
||||
// a wrong scan-detected type (sat) and be handed to smartctl as -d scsi.
|
||||
func TestSmartDevicesExplicitTypeFlowsToSmartctlArgs(t *testing.T) {
|
||||
sm := &SmartManager{}
|
||||
|
||||
configured, err := sm.parseConfiguredDevices("/dev/sda:scsi")
|
||||
require.NoError(t, err)
|
||||
require.Len(t, configured, 1)
|
||||
assert.True(t, configured[0].explicitType)
|
||||
|
||||
// smartctl --scan misreports this USB drive as sat, which fails on it.
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat", Protocol: "ATA"},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 1)
|
||||
|
||||
device := merged[0]
|
||||
assert.Equal(t, "scsi", device.Type, "configured type should win over scan-detected sat")
|
||||
assert.True(t, device.explicitType, "explicit hint must survive the merge")
|
||||
|
||||
assert.Equal(t,
|
||||
[]string{"-d", "scsi", "-a", "--json=c", "/dev/sda"},
|
||||
sm.smartctlArgs(device, false),
|
||||
"explicit scsi type must be passed to smartctl, not dropped",
|
||||
)
|
||||
}
|
||||
|
||||
// TestMergeDeviceListsPreservesExplicitTypeAcrossRescan ensures a verified,
|
||||
// explicitly-typed device keeps its explicit flag when a later scan re-reports
|
||||
// it with a different auto-detected type.
|
||||
func TestMergeDeviceListsPreservesExplicitTypeAcrossRescan(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "scsi", parserType: "scsi", typeVerified: true, explicitType: true},
|
||||
}
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat"},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(existing, scanned, nil)
|
||||
require.Len(t, merged, 1)
|
||||
assert.Equal(t, "scsi", merged[0].Type)
|
||||
assert.True(t, merged[0].explicitType, "explicit type flag should survive a rescan")
|
||||
}
|
||||
|
||||
func TestResolveRefreshError(t *testing.T) {
|
||||
scanErr := errors.New("scan failed")
|
||||
collectErr := errors.New("collect failed")
|
||||
@@ -534,6 +637,74 @@ func TestMergeDeviceListsPrefersConfigured(t *testing.T) {
|
||||
assert.Equal(t, "sat", byName["/dev/sdb"].Type)
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsExpandsConfiguredDevicesWithSamePath(t *testing.T) {
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "sat", InfoName: "scan-info", Protocol: "ATA"},
|
||||
}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 2)
|
||||
|
||||
byKey := make(map[deviceKey]*DeviceInfo, len(merged))
|
||||
for _, device := range merged {
|
||||
byKey[makeDeviceKey(device.Name, device.Type)] = device
|
||||
}
|
||||
|
||||
first := byKey[makeDeviceKey("/dev/sdb", "jms56x,0")]
|
||||
require.NotNil(t, first)
|
||||
assert.Equal(t, "scan-info", first.InfoName)
|
||||
assert.Equal(t, "ATA", first.Protocol)
|
||||
assert.True(t, first.explicitType)
|
||||
|
||||
second := byKey[makeDeviceKey("/dev/sdb", "jms56x,1")]
|
||||
require.NotNil(t, second)
|
||||
assert.True(t, second.explicitType)
|
||||
assert.NotContains(t, byKey, makeDeviceKey("/dev/sdb", "sat"))
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsPreservesSamePathVerificationAcrossRescan(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", parserType: "sat", typeVerified: true, explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", parserType: "sat", typeVerified: true, explicitType: true},
|
||||
}
|
||||
scanned := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "sat", Protocol: "ATA"},
|
||||
}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,1", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(existing, scanned, configured)
|
||||
require.Len(t, merged, 2)
|
||||
byKey := make(map[deviceKey]*DeviceInfo, len(merged))
|
||||
for _, device := range merged {
|
||||
byKey[makeDeviceKey(device.Name, device.Type)] = device
|
||||
assert.True(t, device.typeVerified, device.Type)
|
||||
assert.Equal(t, "sat", device.parserType, device.Type)
|
||||
assert.True(t, device.explicitType, device.Type)
|
||||
}
|
||||
assert.Contains(t, byKey, makeDeviceKey("/dev/sdb", "jms56x,0"))
|
||||
assert.Contains(t, byKey, makeDeviceKey("/dev/sdb", "jms56x,1"))
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsDeduplicatesConfiguredIdentityAfterRekey(t *testing.T) {
|
||||
scanned := []*DeviceInfo{{Name: "/dev/sdb", Type: "sat"}}
|
||||
configured := []*DeviceInfo{
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
{Name: "/dev/sdb", Type: "jms56x,0", explicitType: true},
|
||||
}
|
||||
|
||||
merged := mergeDeviceLists(nil, scanned, configured)
|
||||
require.Len(t, merged, 1)
|
||||
assert.Equal(t, "/dev/sdb", merged[0].Name)
|
||||
assert.Equal(t, "jms56x,0", merged[0].Type)
|
||||
}
|
||||
|
||||
func TestMergeDeviceListsPreservesVerification(t *testing.T) {
|
||||
existing := []*DeviceInfo{
|
||||
{Name: "/dev/sda", Type: "sat+megaraid", parserType: "sat", typeVerified: true},
|
||||
@@ -678,6 +849,20 @@ func TestParseSmartOutputKeepsCustomType(t *testing.T) {
|
||||
assert.Equal(t, "sat+megaraid", device.Type)
|
||||
assert.Equal(t, "sat", device.parserType)
|
||||
assert.True(t, device.typeVerified)
|
||||
assert.Equal(t, "sat+megaraid", sm.SmartDataMap["9C40918040082"].DiskType)
|
||||
}
|
||||
|
||||
func TestParseSmartOutputDoesNotNormalizeDeviceIdentity(t *testing.T) {
|
||||
fixturePath := filepath.Join("test-data", "smart", "sda.json")
|
||||
data, err := os.ReadFile(fixturePath)
|
||||
require.NoError(t, err)
|
||||
|
||||
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
|
||||
device := &DeviceInfo{Name: "/dev/sda", Type: "ata", explicitType: true}
|
||||
|
||||
require.True(t, sm.parseSmartOutput(device, data))
|
||||
assert.Equal(t, "sat", device.parserType)
|
||||
assert.Equal(t, "ata", sm.SmartDataMap["9C40918040082"].DiskType)
|
||||
}
|
||||
|
||||
func TestParseSmartOutputResetsVerificationOnFailure(t *testing.T) {
|
||||
@@ -1225,7 +1410,7 @@ func TestParseSmartForNvmeAppleSSD(t *testing.T) {
|
||||
darwinNvmeProvider: fakeProvider,
|
||||
}
|
||||
|
||||
hasData, _ := sm.parseSmartForNvme(data)
|
||||
hasData, _ := sm.parseSmartForNvme(data, "")
|
||||
require.True(t, hasData)
|
||||
|
||||
deviceData, ok := sm.SmartDataMap["0ba0147940253c15"]
|
||||
@@ -1237,7 +1422,7 @@ func TestParseSmartForNvmeAppleSSD(t *testing.T) {
|
||||
assert.Equal(t, 1, providerCalls, "system_profiler should be called once")
|
||||
|
||||
// Second parse: provider should NOT be called again (cache hit)
|
||||
_, _ = sm.parseSmartForNvme(data)
|
||||
_, _ = sm.parseSmartForNvme(data, "")
|
||||
assert.Equal(t, 1, providerCalls, "system_profiler should not be called again after caching")
|
||||
}
|
||||
|
||||
|
||||
512
agent/storage_pool.go
Normal file
512
agent/storage_pool.go
Normal file
@@ -0,0 +1,512 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log/slog"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
zfsentity "github.com/henrygd/beszel/internal/entities/zfs"
|
||||
)
|
||||
|
||||
// zfsDatasetUsage holds usage values for a ZFS dataset mountpoint.
|
||||
type zfsDatasetUsage struct {
|
||||
used uint64
|
||||
avail uint64
|
||||
}
|
||||
|
||||
// datasetUsageRefreshInterval controls how often `zfs list` is re-run for the
|
||||
// mountpoint usage map. Dataset inventory changes rarely.
|
||||
const datasetUsageRefreshInterval = 5 * time.Minute
|
||||
|
||||
// poolStatsRefreshInterval controls how often `zpool list` is re-run for pool
|
||||
// capacity. Health and I/O are read from procfs on Linux, so the utility only
|
||||
// needs to refresh slow-moving space accounting.
|
||||
const poolStatsRefreshInterval = time.Minute
|
||||
|
||||
// btrfsFilesystems is the btrfs source; overridable in tests.
|
||||
var btrfsFilesystems = btrfs.Filesystems
|
||||
|
||||
type poolKernelSample struct {
|
||||
nread uint64
|
||||
nwrite uint64
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// StoragePoolManager combines independent backend inventories. Metrics and
|
||||
// dataset usage require the agent lock; GetDetail is safe for concurrent calls.
|
||||
type StoragePoolManager struct {
|
||||
backends []*poolBackend
|
||||
detailInterval time.Duration
|
||||
}
|
||||
|
||||
// poolBackend owns one backend's collectors and caches. Collector functions
|
||||
// are immutable after construction and may run concurrently for metrics/details.
|
||||
type poolBackend struct {
|
||||
name string
|
||||
poolStatsFn func() ([]zfs.PoolStat, error) // capacity/health source
|
||||
datasetsFn func() ([]zfs.Dataset, error) // dataset inventory source
|
||||
kernelStatsFn func() ([]zfs.PoolKernelStat, error) // procfs pool state/I/O source
|
||||
poolStatusesFn func() ([]zfs.PoolStatus, error) // scrub/vdev detail source
|
||||
|
||||
// Utility-backed caches below are refreshed in the background after the
|
||||
// first collection, so cacheMu guards them against those goroutines.
|
||||
cacheMu sync.Mutex
|
||||
poolData []zfs.PoolStat // cached pool inventory (TTL below)
|
||||
lastPoolStats time.Time
|
||||
poolRefreshing bool
|
||||
|
||||
datasetUsage map[string]zfsDatasetUsage // mountpoint -> usage
|
||||
lastUsageRefresh time.Time
|
||||
usageRefreshing bool
|
||||
|
||||
kernelSamples map[string]poolKernelSample
|
||||
|
||||
// Detail data (pools, vdevs, scrub, datasets) is cached and refreshed on
|
||||
// an interval. Accessed from handler goroutines, so it is mutex-protected.
|
||||
detailMu sync.Mutex
|
||||
detail *zfsentity.ZfsData
|
||||
lastDetailRefresh time.Time
|
||||
detailFailed bool
|
||||
}
|
||||
|
||||
func newStoragePoolManager() *StoragePoolManager {
|
||||
return &StoragePoolManager{
|
||||
backends: []*poolBackend{newZfsBackend(), newBtrfsBackend()},
|
||||
detailInterval: time.Hour,
|
||||
}
|
||||
}
|
||||
|
||||
func newZfsBackend() *poolBackend {
|
||||
return &poolBackend{
|
||||
name: "zfs",
|
||||
poolStatsFn: optionalPoolSource(zfs.PoolStats),
|
||||
datasetsFn: optionalPoolSource(zfs.Datasets),
|
||||
kernelStatsFn: optionalPoolSource(zfs.PoolKernelStats),
|
||||
poolStatusesFn: optionalPoolSource(zfs.PoolStatuses),
|
||||
}
|
||||
}
|
||||
|
||||
func newBtrfsBackend() *poolBackend {
|
||||
return &poolBackend{
|
||||
name: "btrfs",
|
||||
poolStatsFn: btrfsSource(btrfsPoolStats),
|
||||
kernelStatsFn: btrfsSource(btrfsKernelStats),
|
||||
poolStatusesFn: btrfsSource(btrfsPoolStatuses),
|
||||
}
|
||||
}
|
||||
|
||||
// datasets is optional: only backends that expose datasets provide a collector.
|
||||
func (b *poolBackend) datasets() ([]zfs.Dataset, error) {
|
||||
if b.datasetsFn == nil {
|
||||
return nil, nil
|
||||
}
|
||||
return b.datasetsFn()
|
||||
}
|
||||
|
||||
// A missing utility/interface is a successfully observed absent backend.
|
||||
func optionalPoolSource[T any](source func() ([]T, error)) func() ([]T, error) {
|
||||
return func() ([]T, error) {
|
||||
items, err := source()
|
||||
if errors.Is(err, zfs.ErrNoZfs) || errors.Is(err, exec.ErrNotFound) || errors.Is(err, errors.ErrUnsupported) {
|
||||
return nil, nil
|
||||
}
|
||||
return items, err
|
||||
}
|
||||
}
|
||||
|
||||
func btrfsSource[T any](convert func(btrfs.Filesystem) T) func() ([]T, error) {
|
||||
return func() ([]T, error) {
|
||||
filesystems, err := optionalPoolSource(btrfsFilesystems)()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
items := make([]T, 0, len(filesystems))
|
||||
for _, fs := range filesystems {
|
||||
items = append(items, convert(fs))
|
||||
}
|
||||
return items, nil
|
||||
}
|
||||
}
|
||||
|
||||
// Update refreshes systemStats.ZfsPools with the latest pool data. I/O
|
||||
// throughput and health come from inexpensive kernel kstats on Linux. Pool
|
||||
// capacity and dataset usage come from separately cached utility calls. The
|
||||
// pool map is empty when both backends are absent.
|
||||
func (m *StoragePoolManager) Update(systemStats *system.Stats) {
|
||||
// Rebuild the combined map so successful pool removals clear old samples.
|
||||
systemStats.ZfsPools = nil
|
||||
for _, backend := range m.backends {
|
||||
backend.updateBackendStats(systemStats)
|
||||
}
|
||||
}
|
||||
|
||||
func (b *poolBackend) updateBackendStats(systemStats *system.Stats) {
|
||||
pools := b.poolStats()
|
||||
if len(pools) == 0 {
|
||||
b.kernelSamples = nil
|
||||
return
|
||||
}
|
||||
|
||||
kernelStats, ioRates := b.kernelStats()
|
||||
|
||||
if systemStats.ZfsPools == nil {
|
||||
systemStats.ZfsPools = make(map[string]*system.ZfsPool, len(pools))
|
||||
}
|
||||
for i := range pools {
|
||||
pool := &pools[i]
|
||||
// Full precision, matching the dataset values below; the frontend
|
||||
// formats any magnitude.
|
||||
stats := &system.ZfsPool{
|
||||
DisplayName: pool.DisplayName,
|
||||
Raw: pool.Raw,
|
||||
Total: float64(pool.Size) / (1024 * 1024 * 1024),
|
||||
Used: float64(pool.Alloc) / (1024 * 1024 * 1024),
|
||||
Health: pool.Health,
|
||||
}
|
||||
if kernel, exists := kernelStats[pool.Name]; exists && kernel.Health != "" {
|
||||
stats.Health = kernel.Health
|
||||
}
|
||||
if io, exists := ioRates[pool.Name]; exists {
|
||||
stats.ReadBytes = io.NRead
|
||||
stats.WriteBytes = io.NWrite
|
||||
}
|
||||
slog.Debug("Storage pool sample", "backend", b.name, "pool", pool.Name, "health", stats.Health, "used_gb", stats.Used, "read_bps", stats.ReadBytes, "write_bps", stats.WriteBytes)
|
||||
systemStats.ZfsPools[pool.Name] = stats
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// poolStats returns the cached pool inventory, calling its collector at most
|
||||
// every poolStatsRefreshInterval. Only the first collection blocks; later
|
||||
// refreshes run in the background because utilities like `zpool list` can hang
|
||||
// for seconds on busy hosts, which would otherwise delay the hub's stats
|
||||
// response. On failure the previous inventory is retained and the refresh is
|
||||
// retried on the next cadence.
|
||||
func (b *poolBackend) poolStats() []zfs.PoolStat {
|
||||
b.cacheMu.Lock()
|
||||
defer b.cacheMu.Unlock()
|
||||
if b.poolRefreshing || (!b.lastPoolStats.IsZero() && time.Since(b.lastPoolStats) < poolStatsRefreshInterval) {
|
||||
return b.poolData
|
||||
}
|
||||
if b.lastPoolStats.IsZero() {
|
||||
b.storePoolStats(b.poolStatsFn())
|
||||
return b.poolData
|
||||
}
|
||||
b.poolRefreshing = true
|
||||
go func() {
|
||||
pools, err := b.poolStatsFn()
|
||||
b.cacheMu.Lock()
|
||||
defer b.cacheMu.Unlock()
|
||||
b.poolRefreshing = false
|
||||
b.storePoolStats(pools, err)
|
||||
}()
|
||||
return b.poolData
|
||||
}
|
||||
|
||||
// storePoolStats records a pool inventory result. Callers must hold cacheMu.
|
||||
func (b *poolBackend) storePoolStats(pools []zfs.PoolStat, err error) {
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool stats unavailable", "backend", b.name, "err", err)
|
||||
} else {
|
||||
b.poolData = pools
|
||||
}
|
||||
b.lastPoolStats = time.Now()
|
||||
}
|
||||
|
||||
// kernelStats reads cumulative pool counters and converts them to per-second
|
||||
// rates. Counter decreases indicate a pool export/import and reset the
|
||||
// baseline instead of producing an underflow spike.
|
||||
func (b *poolBackend) kernelStats() (map[string]zfs.PoolKernelStat, map[string]zfs.PoolIoStats) {
|
||||
if b.kernelStatsFn == nil {
|
||||
return nil, nil
|
||||
}
|
||||
stats, err := b.kernelStatsFn()
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool kernel stats unavailable", "backend", b.name, "err", err)
|
||||
return nil, nil
|
||||
}
|
||||
now := time.Now()
|
||||
byName := make(map[string]zfs.PoolKernelStat, len(stats))
|
||||
rates := make(map[string]zfs.PoolIoStats, len(stats))
|
||||
nextSamples := make(map[string]poolKernelSample, len(stats))
|
||||
for _, stat := range stats {
|
||||
byName[stat.Name] = stat
|
||||
if previous, ok := b.kernelSamples[stat.Name]; ok && now.After(previous.at) &&
|
||||
stat.NRead >= previous.nread && stat.NWrite >= previous.nwrite {
|
||||
seconds := now.Sub(previous.at).Seconds()
|
||||
rates[stat.Name] = zfs.PoolIoStats{
|
||||
NRead: uint64(float64(stat.NRead-previous.nread) / seconds),
|
||||
NWrite: uint64(float64(stat.NWrite-previous.nwrite) / seconds),
|
||||
}
|
||||
}
|
||||
nextSamples[stat.Name] = poolKernelSample{nread: stat.NRead, nwrite: stat.NWrite, at: now}
|
||||
}
|
||||
b.kernelSamples = nextSamples
|
||||
return byName, rates
|
||||
}
|
||||
|
||||
// refreshDatasetUsage re-runs `zfs list` when the refresh window has elapsed
|
||||
// and returns the mountpoint-keyed usage map. Like poolStats, only the first
|
||||
// collection blocks and later refreshes run in the background.
|
||||
func (b *poolBackend) refreshDatasetUsage() map[string]zfsDatasetUsage {
|
||||
b.cacheMu.Lock()
|
||||
defer b.cacheMu.Unlock()
|
||||
if b.usageRefreshing || (!b.lastUsageRefresh.IsZero() && time.Since(b.lastUsageRefresh) < datasetUsageRefreshInterval) {
|
||||
return b.datasetUsage
|
||||
}
|
||||
if b.lastUsageRefresh.IsZero() {
|
||||
b.storeDatasetUsage(b.datasets())
|
||||
return b.datasetUsage
|
||||
}
|
||||
b.usageRefreshing = true
|
||||
go func() {
|
||||
datasets, err := b.datasets()
|
||||
b.cacheMu.Lock()
|
||||
defer b.cacheMu.Unlock()
|
||||
b.usageRefreshing = false
|
||||
b.storeDatasetUsage(datasets, err)
|
||||
}()
|
||||
return b.datasetUsage
|
||||
}
|
||||
|
||||
// storeDatasetUsage rebuilds the usage map from a dataset listing. The map is
|
||||
// replaced rather than mutated so returned references stay safe to read.
|
||||
// Callers must hold cacheMu.
|
||||
func (b *poolBackend) storeDatasetUsage(datasets []zfs.Dataset, err error) {
|
||||
if err != nil {
|
||||
slog.Debug("Storage pool dataset usage unavailable", "backend", b.name, "err", err)
|
||||
} else {
|
||||
usage := make(map[string]zfsDatasetUsage, len(datasets))
|
||||
for _, ds := range datasets {
|
||||
if ds.Mountpoint != "" && ds.Mountpoint != "-" {
|
||||
usage[ds.Mountpoint] = zfsDatasetUsage{used: ds.Used, avail: ds.Avail}
|
||||
}
|
||||
}
|
||||
b.datasetUsage = usage
|
||||
}
|
||||
b.lastUsageRefresh = time.Now()
|
||||
}
|
||||
|
||||
// DatasetUsage returns ZFS dataset usage keyed by mountpoint, refreshed at
|
||||
// most every datasetUsageRefreshInterval. On failure the previous map is
|
||||
// retained and a debug log is emitted.
|
||||
func (m *StoragePoolManager) DatasetUsage() map[string]zfsDatasetUsage {
|
||||
for _, backend := range m.backends {
|
||||
if backend.name == "zfs" {
|
||||
return backend.refreshDatasetUsage()
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// GetDetail combines backend snapshots, identifying successful inventories so
|
||||
// the hub can accept partial updates without deleting failed backend records.
|
||||
func (m *StoragePoolManager) GetDetail(force bool) *zfsentity.ZfsData {
|
||||
data := &zfsentity.ZfsData{Complete: true}
|
||||
for _, backend := range m.backends {
|
||||
snapshot := backend.getBackendDetail(force, m.detailInterval)
|
||||
data.Pools = append(data.Pools, snapshot.Pools...)
|
||||
if snapshot.Complete {
|
||||
data.CompleteBackends = append(data.CompleteBackends, backend.name)
|
||||
} else {
|
||||
data.Complete = false
|
||||
}
|
||||
}
|
||||
return data
|
||||
}
|
||||
|
||||
func (b *poolBackend) getBackendDetail(force bool, interval time.Duration) *zfsentity.ZfsData {
|
||||
b.detailMu.Lock()
|
||||
defer b.detailMu.Unlock()
|
||||
|
||||
if force || b.detailFailed || b.detail == nil || time.Since(b.lastDetailRefresh) >= interval {
|
||||
if data, err := b.collectDetail(b.detail); err != nil {
|
||||
b.detailFailed = true
|
||||
slog.Debug("Storage pool detail collection failed", "backend", b.name, "err", err)
|
||||
if b.detail == nil {
|
||||
return &zfsentity.ZfsData{}
|
||||
}
|
||||
return &zfsentity.ZfsData{Pools: b.detail.Pools}
|
||||
} else {
|
||||
b.detailFailed = false
|
||||
b.detail = data
|
||||
b.lastDetailRefresh = time.Now()
|
||||
}
|
||||
}
|
||||
if b.detail == nil {
|
||||
return &zfsentity.ZfsData{}
|
||||
}
|
||||
return b.detail
|
||||
}
|
||||
|
||||
// collectDetail builds a ZfsData payload from the current system state.
|
||||
func (b *poolBackend) collectDetail(previous *zfsentity.ZfsData) (*zfsentity.ZfsData, error) {
|
||||
pools, err := b.poolStatsFn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(pools) == 0 {
|
||||
return &zfsentity.ZfsData{Pools: []*zfsentity.PoolDetail{}, Complete: true}, nil
|
||||
}
|
||||
|
||||
statuses, statusErr := b.poolStatusesFn()
|
||||
if statusErr != nil {
|
||||
slog.Debug("Storage pool status unavailable", "backend", b.name, "err", statusErr)
|
||||
}
|
||||
datasets, datasetsErr := b.datasets()
|
||||
if datasetsErr != nil {
|
||||
slog.Debug("Storage pool datasets unavailable", "backend", b.name, "err", datasetsErr)
|
||||
}
|
||||
|
||||
statusByPool := make(map[string]zfs.PoolStatus, len(statuses))
|
||||
for _, st := range statuses {
|
||||
statusByPool[st.Name] = st
|
||||
}
|
||||
|
||||
previousByPool := make(map[string]*zfsentity.PoolDetail)
|
||||
if previous != nil {
|
||||
for _, pool := range previous.Pools {
|
||||
if pool != nil {
|
||||
previousByPool[pool.Name] = pool
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
data := &zfsentity.ZfsData{Pools: make([]*zfsentity.PoolDetail, 0, len(pools)), Complete: true}
|
||||
for i := range pools {
|
||||
p := &pools[i]
|
||||
detail := &zfsentity.PoolDetail{
|
||||
DisplayName: p.DisplayName,
|
||||
Raw: p.Raw,
|
||||
Name: p.Name,
|
||||
Health: p.Health,
|
||||
Size: p.Size,
|
||||
Alloc: p.Alloc,
|
||||
Free: p.Free,
|
||||
}
|
||||
if st, ok := statusByPool[p.Name]; statusErr == nil && ok {
|
||||
if st.Scrub.State != "" && st.Scrub.State != "NONE" {
|
||||
detail.Scrub = &zfsentity.Scrub{
|
||||
State: st.Scrub.State,
|
||||
Progress: st.Scrub.Progress,
|
||||
Errors: st.Scrub.Errors,
|
||||
}
|
||||
}
|
||||
for _, v := range st.Vdevs {
|
||||
detail.Vdevs = append(detail.Vdevs, &zfsentity.Vdev{
|
||||
Name: v.Name,
|
||||
State: v.State,
|
||||
ReadErrs: v.ReadErrs,
|
||||
WriteErrs: v.WriteErrs,
|
||||
ChecksumErrs: v.ChecksumErrs,
|
||||
})
|
||||
}
|
||||
} else {
|
||||
if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Scrub = cached.Scrub
|
||||
detail.Vdevs = cached.Vdevs
|
||||
}
|
||||
}
|
||||
if datasetsErr == nil {
|
||||
foundDataset := false
|
||||
for _, ds := range datasets {
|
||||
if poolOfDataset(ds.Name) == p.Name {
|
||||
foundDataset = true
|
||||
detail.Datasets = append(detail.Datasets, &zfsentity.Dataset{
|
||||
Name: ds.Name,
|
||||
Used: ds.Used,
|
||||
Avail: ds.Avail,
|
||||
Mountpoint: ds.Mountpoint,
|
||||
})
|
||||
}
|
||||
}
|
||||
if !foundDataset {
|
||||
if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Datasets = cached.Datasets
|
||||
}
|
||||
}
|
||||
} else if cached := previousByPool[p.Name]; cached != nil {
|
||||
detail.Datasets = cached.Datasets
|
||||
}
|
||||
data.Pools = append(data.Pools, detail)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
// poolOfDataset returns the pool name for a dataset name (everything before
|
||||
// the first '/'). Datasets without a separator belong to a pool of the same
|
||||
// name.
|
||||
func poolOfDataset(name string) string {
|
||||
if idx := strings.IndexByte(name, '/'); idx >= 0 {
|
||||
return name[:idx]
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// ZfsMountpoints returns the set of mountpoints backed by ZFS datasets.
|
||||
func (m *StoragePoolManager) ZfsMountpoints() map[string]bool {
|
||||
usage := m.DatasetUsage()
|
||||
mountpoints := make(map[string]bool, len(usage))
|
||||
for mountpoint := range usage {
|
||||
mountpoints[mountpoint] = true
|
||||
}
|
||||
return mountpoints
|
||||
}
|
||||
|
||||
func btrfsPoolStats(fs btrfs.Filesystem) zfs.PoolStat {
|
||||
return zfs.PoolStat{MountID: fs.MountID, IODevice: fs.IODevice, Raw: fs.Raw, DisplayName: fs.Name, Name: "b:" + fs.UUID, Size: fs.Size, Alloc: fs.Alloc, Free: fs.Size - min(fs.Alloc, fs.Size), Health: fs.Health}
|
||||
}
|
||||
|
||||
func btrfsKernelStats(fs btrfs.Filesystem) zfs.PoolKernelStat {
|
||||
return zfs.PoolKernelStat{Name: "b:" + fs.UUID, Health: fs.Health, NRead: fs.NRead, NWrite: fs.NWrite}
|
||||
}
|
||||
|
||||
func btrfsPoolStatuses(fs btrfs.Filesystem) zfs.PoolStatus {
|
||||
status := zfs.PoolStatus{Name: "b:" + fs.UUID, State: fs.Health, Scrub: zfs.ScrubStatus{State: "NONE"}}
|
||||
for _, dev := range fs.Devices {
|
||||
status.Vdevs = append(status.Vdevs, zfs.VdevStatus{
|
||||
Name: dev.Name, State: dev.State,
|
||||
ReadErrs: dev.ReadErrs, WriteErrs: dev.WriteErrs, ChecksumErrs: dev.CorruptionErrs,
|
||||
})
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
// markDuplicateCharts leaves pool telemetry and detail intact, but tells the
|
||||
// hub which charts already have a filesystem equivalent. Only exact kernel
|
||||
// filesystem and I/O-device matches qualify; labels are never used.
|
||||
func (m *StoragePoolManager) markDuplicateCharts(stats *system.Stats, filesystems map[string]*system.FsStats, mountID func(string) string) {
|
||||
identities := make(map[string]string, len(filesystems))
|
||||
for device, fs := range filesystems {
|
||||
if fs.DiskTotal > 0 {
|
||||
identities[device] = mountID(fs.Mountpoint)
|
||||
}
|
||||
}
|
||||
for _, backend := range m.backends {
|
||||
backend.cacheMu.Lock()
|
||||
pools := backend.poolData
|
||||
backend.cacheMu.Unlock()
|
||||
for _, pool := range pools {
|
||||
sample := stats.ZfsPools[pool.Name]
|
||||
if sample == nil || pool.MountID == "" {
|
||||
continue
|
||||
}
|
||||
for device, identity := range identities {
|
||||
if identity != pool.MountID {
|
||||
continue
|
||||
}
|
||||
// Raw physical usage is not equivalent to a filesystem usage chart.
|
||||
sample.HideUsage = !pool.Raw
|
||||
if pool.IODevice != "" && pool.IODevice == device {
|
||||
sample.HideIO = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
548
agent/storage_pool_test.go
Normal file
548
agent/storage_pool_test.go
Normal file
@@ -0,0 +1,548 @@
|
||||
//go:build testing
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestOptionalPoolSource(t *testing.T) {
|
||||
failure := errors.New("timeout")
|
||||
for _, err := range []error{nil, zfs.ErrNoZfs, fmt.Errorf("zpool: %w", exec.ErrNotFound), errors.ErrUnsupported, failure} {
|
||||
_, got := optionalPoolSource(func() ([]zfs.PoolStat, error) { return nil, err })()
|
||||
if err == failure {
|
||||
assert.ErrorIs(t, got, failure)
|
||||
} else {
|
||||
assert.NoError(t, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type poolTestBackend struct {
|
||||
name string
|
||||
err error
|
||||
alloc uint64
|
||||
read uint64
|
||||
empty bool
|
||||
}
|
||||
|
||||
func (state *poolTestBackend) backend() *poolBackend {
|
||||
name := "zfs"
|
||||
if strings.HasPrefix(state.name, "b:") {
|
||||
name = "btrfs"
|
||||
}
|
||||
return &poolBackend{
|
||||
name: name,
|
||||
poolStatsFn: func() ([]zfs.PoolStat, error) {
|
||||
if state.empty {
|
||||
return nil, state.err
|
||||
}
|
||||
return []zfs.PoolStat{{Name: state.name, Size: 100, Alloc: state.alloc}}, state.err
|
||||
},
|
||||
kernelStatsFn: func() ([]zfs.PoolKernelStat, error) {
|
||||
return []zfs.PoolKernelStat{{Name: state.name, NRead: state.read}}, state.err
|
||||
},
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) { return nil, nil },
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return nil, nil },
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndependentPoolBackendCaches(t *testing.T) {
|
||||
for _, failed := range []int{0, 1} {
|
||||
t.Run([]string{"zfs", "btrfs"}[failed], func(t *testing.T) {
|
||||
states := []*poolTestBackend{{name: "tank", alloc: 10}, {name: "b:uuid", alloc: 10}}
|
||||
managers := []*poolBackend{states[0].backend(), states[1].backend()}
|
||||
zm := &StoragePoolManager{backends: managers, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 2)
|
||||
require.True(t, zm.GetDetail(true).Complete)
|
||||
baseline := poolKernelSample{at: time.Now().Add(-time.Second)}
|
||||
for i, m := range managers {
|
||||
m.lastPoolStats = time.Time{}
|
||||
m.kernelSamples[states[i].name] = baseline
|
||||
states[i].alloc = 20
|
||||
states[i].read = 100
|
||||
}
|
||||
states[failed].err = errors.New("collection failed")
|
||||
zm.Update(&stats)
|
||||
healthy := 1 - failed
|
||||
assert.Equal(t, uint64(10), managers[failed].poolData[0].Alloc)
|
||||
assert.Equal(t, uint64(20), managers[healthy].poolData[0].Alloc)
|
||||
assert.Equal(t, baseline, managers[failed].kernelSamples[states[failed].name])
|
||||
assert.Zero(t, stats.ZfsPools[states[failed].name].ReadBytes)
|
||||
assert.Positive(t, stats.ZfsPools[states[healthy].name].ReadBytes)
|
||||
partial := zm.GetDetail(true)
|
||||
assert.False(t, partial.Complete)
|
||||
assert.False(t, partial.CanRefreshPool(states[failed].name))
|
||||
assert.True(t, partial.CanRefreshPool(states[healthy].name))
|
||||
assert.Equal(t, uint64(10), partial.Pools[failed].Alloc)
|
||||
assert.Equal(t, uint64(20), partial.Pools[healthy].Alloc)
|
||||
assert.False(t, zm.GetDetail(false).Complete, "a failed forced refresh must not become complete from cache")
|
||||
|
||||
// Successful empty inventory removes only the healthy backend's pool.
|
||||
states[healthy].empty = true
|
||||
managers[healthy].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 1)
|
||||
assert.Contains(t, stats.ZfsPools, states[failed].name)
|
||||
partial = zm.GetDetail(true)
|
||||
require.Len(t, partial.Pools, 1)
|
||||
assert.True(t, partial.CanRefreshPool(states[healthy].name))
|
||||
|
||||
// Recovery uses the retained I/O baseline, then normal removal works.
|
||||
states[failed].err = nil
|
||||
managers[failed].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
assert.Positive(t, stats.ZfsPools[states[failed].name].ReadBytes)
|
||||
assert.True(t, zm.GetDetail(true).Complete)
|
||||
states[failed].empty = true
|
||||
managers[failed].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
assert.Empty(t, stats.ZfsPools)
|
||||
assert.Empty(t, zm.GetDetail(true).Pools)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndependentBackendsWithoutCache(t *testing.T) {
|
||||
z := &poolTestBackend{name: "tank", err: errors.New("ZFS failure")}
|
||||
b := &poolTestBackend{name: "b:uuid", alloc: 20}
|
||||
zm := &StoragePoolManager{backends: []*poolBackend{z.backend(), b.backend()}, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 1)
|
||||
assert.Contains(t, stats.ZfsPools, "b:uuid")
|
||||
detail := zm.GetDetail(true)
|
||||
require.Len(t, detail.Pools, 1)
|
||||
assert.False(t, detail.Complete)
|
||||
assert.Equal(t, []string{"btrfs"}, detail.CompleteBackends)
|
||||
}
|
||||
|
||||
func TestConcurrentBackendDetailsAndMetrics(t *testing.T) {
|
||||
zm := &StoragePoolManager{backends: []*poolBackend{(&poolTestBackend{name: "tank"}).backend(), (&poolTestBackend{name: "b:uuid"}).backend()}, detailInterval: time.Hour}
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < 3; i++ {
|
||||
wg.Add(1)
|
||||
go func(metrics bool) {
|
||||
defer wg.Done()
|
||||
for j := 0; j < 10; j++ {
|
||||
if metrics {
|
||||
zm.Update(&system.Stats{})
|
||||
} else {
|
||||
zm.GetDetail(true)
|
||||
}
|
||||
}
|
||||
}(i == 0)
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
func TestStoragePoolBackendOrder(t *testing.T) {
|
||||
z := (&poolTestBackend{name: "tank"}).backend()
|
||||
b := (&poolTestBackend{name: "b:uuid"}).backend()
|
||||
b.datasetsFn = nil // Btrfs does not expose datasets.
|
||||
z.datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank/data", Mountpoint: "/tank", Used: 10}}, nil
|
||||
}
|
||||
m := &StoragePoolManager{backends: []*poolBackend{b, z}, detailInterval: time.Hour}
|
||||
var stats system.Stats
|
||||
m.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 2)
|
||||
detail := m.GetDetail(true)
|
||||
require.True(t, detail.Complete)
|
||||
assert.Equal(t, []string{"btrfs", "zfs"}, detail.CompleteBackends)
|
||||
assert.Empty(t, detail.Pools[0].Datasets)
|
||||
assert.Len(t, detail.Pools[1].Datasets, 1)
|
||||
assert.Equal(t, uint64(10), m.DatasetUsage()["/tank"].used)
|
||||
|
||||
b.poolData[0].MountID = "uuid"
|
||||
b.poolData[0].IODevice = "sda"
|
||||
calls := 0
|
||||
m.markDuplicateCharts(&stats, map[string]*system.FsStats{
|
||||
"sda": {Mountpoint: "/", DiskTotal: 100},
|
||||
}, func(string) string { calls++; return "uuid" })
|
||||
assert.Equal(t, 1, calls, "resolve each filesystem once across all backends")
|
||||
assert.True(t, stats.ZfsPools["b:uuid"].HideUsage)
|
||||
assert.True(t, stats.ZfsPools["b:uuid"].HideIO)
|
||||
}
|
||||
|
||||
func TestUpdatePopulatesZfsPools(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Size: 23999000000000, Alloc: 12000000000000, Free: 11999000000000, Health: "DEGRADED"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank/apps", Used: 5000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
|
||||
{Name: "tank/backup", Used: 6000000000000, Avail: 11999000000000, Mountpoint: "/tank/backup"},
|
||||
// Small zvol (Proxmox VM EFI disk): must not round to zero.
|
||||
{Name: "rpool/vm-100-disk-2", Used: 4194304, Avail: 0, Mountpoint: "-"},
|
||||
}, nil
|
||||
}
|
||||
var kernelCalls int
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
kernelCalls++
|
||||
return []zfs.PoolKernelStat{{
|
||||
Name: "tank", Health: "ONLINE",
|
||||
NRead: uint64(kernelCalls-1) * 1250, NWrite: uint64(kernelCalls-1) * 5120,
|
||||
}}, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
// The first kernel sample establishes the cumulative-counter baseline.
|
||||
zm.Update(&stats)
|
||||
zm.backends[0].kernelSamples["tank"] = poolKernelSample{at: time.Now().Add(-time.Second)}
|
||||
zm.Update(&stats)
|
||||
require.NotNil(t, stats.ZfsPools)
|
||||
require.Contains(t, stats.ZfsPools, "tank")
|
||||
assert.InDelta(t, 22350.8105, stats.ZfsPools["tank"].Total, 0.0001) // Size in GiB
|
||||
assert.InDelta(t, 11175.8709, stats.ZfsPools["tank"].Used, 0.0001) // Alloc in GiB
|
||||
assert.Equal(t, "ONLINE", stats.ZfsPools["tank"].Health)
|
||||
assert.InDelta(t, 1250, stats.ZfsPools["tank"].ReadBytes, 5)
|
||||
assert.InDelta(t, 5120, stats.ZfsPools["tank"].WriteBytes, 5)
|
||||
|
||||
}
|
||||
|
||||
// TestUpdateKernelStatsMissing verifies pools without a kernel sample report zero
|
||||
// I/O instead of erroring.
|
||||
func TestUpdateKernelStatsMissing(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Size: 1, Alloc: 1, Health: "ONLINE"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.NotNil(t, stats.ZfsPools)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
|
||||
}
|
||||
|
||||
func TestUpdateKernelCounterReset(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank", Health: "ONLINE"}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
zm.backends[0].kernelSamples = map[string]poolKernelSample{
|
||||
"tank": {nread: 100, nwrite: 200, at: time.Now().Add(-time.Second)},
|
||||
}
|
||||
zm.backends[0].kernelStatsFn = func() ([]zfs.PoolKernelStat, error) {
|
||||
return []zfs.PoolKernelStat{{Name: "tank", Health: "ONLINE", NRead: 10, NWrite: 20}}, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].ReadBytes)
|
||||
assert.Equal(t, uint64(0), stats.ZfsPools["tank"].WriteBytes)
|
||||
}
|
||||
|
||||
func TestUpdateNoZfs(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
calls++
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
zm.Update(&stats)
|
||||
assert.Nil(t, stats.ZfsPools)
|
||||
assert.Equal(t, 1, calls, "failed pool discovery should be cached until the next refresh interval")
|
||||
}
|
||||
|
||||
func TestUpdateEmptyPools(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
calls++
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
zm.Update(&stats)
|
||||
assert.Nil(t, stats.ZfsPools)
|
||||
assert.Equal(t, 1, calls, "an empty pool inventory should be cached until the next refresh interval")
|
||||
}
|
||||
|
||||
func TestDatasetUsage(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
calls := 0
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
calls++
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Used: 12000000000000, Avail: 11999000000000, Mountpoint: "/tank"},
|
||||
{Name: "tank/apps", Used: 1000000000000, Avail: 11999000000000, Mountpoint: "/tank/apps"},
|
||||
{Name: "rpool", Used: 900000000000, Avail: 300000000000, Mountpoint: "-"}, // zvol/unmounted: excluded
|
||||
}, nil
|
||||
}
|
||||
|
||||
usage := zm.DatasetUsage()
|
||||
require.Len(t, usage, 2)
|
||||
assert.Equal(t, zfsDatasetUsage{used: 12000000000000, avail: 11999000000000}, usage["/tank"])
|
||||
assert.Equal(t, zfsDatasetUsage{used: 1000000000000, avail: 11999000000000}, usage["/tank/apps"])
|
||||
assert.Equal(t, 1, calls)
|
||||
|
||||
// Second call within the refresh window must not re-run the collector.
|
||||
zm.DatasetUsage()
|
||||
assert.Equal(t, 1, calls)
|
||||
}
|
||||
|
||||
func TestDatasetUsageRefreshOnErrorKeepsPrevious(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank", Used: 1, Avail: 1, Mountpoint: "/tank"}}, nil
|
||||
}
|
||||
assert.Len(t, zm.DatasetUsage(), 1)
|
||||
|
||||
// Force refresh window expiry, then a failing collector.
|
||||
zm.backends[0].lastUsageRefresh = time.Now().Add(-10 * time.Minute)
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return nil, zfs.ErrNoZfs
|
||||
}
|
||||
usage := zm.DatasetUsage()
|
||||
assert.Len(t, usage, 1, "previous usage should be retained on error")
|
||||
}
|
||||
|
||||
func TestDatasetUsageClearsAbsentBackend(t *testing.T) {
|
||||
b := newZfsBackend()
|
||||
b.datasetUsage = map[string]zfsDatasetUsage{"/tank": {used: 1, avail: 1}}
|
||||
b.datasetsFn = optionalPoolSource(func() ([]zfs.Dataset, error) {
|
||||
return nil, zfs.ErrNoZfs
|
||||
})
|
||||
|
||||
datasets, err := b.datasets()
|
||||
require.NoError(t, err, "an absent backend must not produce an error to log")
|
||||
assert.Empty(t, datasets)
|
||||
b.refreshDatasetUsage()
|
||||
assert.Empty(t, b.datasetUsage)
|
||||
assert.False(t, b.lastUsageRefresh.IsZero())
|
||||
}
|
||||
|
||||
func TestGetDetailForceRefresh(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
poolCalls := 0
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
poolCalls++
|
||||
return []zfs.PoolStat{{Name: "tank", Alloc: uint64(poolCalls)}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
|
||||
first := zm.GetDetail(false)
|
||||
assert.True(t, first.Complete)
|
||||
require.Len(t, first.Pools, 1)
|
||||
assert.Equal(t, uint64(1), first.Pools[0].Alloc)
|
||||
|
||||
cached := zm.GetDetail(false)
|
||||
require.Len(t, cached.Pools, 1)
|
||||
assert.Equal(t, uint64(1), cached.Pools[0].Alloc)
|
||||
assert.Equal(t, 1, poolCalls)
|
||||
|
||||
refreshed := zm.GetDetail(true)
|
||||
assert.True(t, refreshed.Complete)
|
||||
require.Len(t, refreshed.Pools, 1)
|
||||
assert.Equal(t, uint64(2), refreshed.Pools[0].Alloc)
|
||||
assert.Equal(t, 2, poolCalls)
|
||||
}
|
||||
|
||||
func TestGetDetailSuccessfulEmptyInventoryClearsCache(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank"}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, nil }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, nil }
|
||||
|
||||
require.Len(t, zm.GetDetail(false).Pools, 1)
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, nil }
|
||||
empty := zm.GetDetail(true)
|
||||
assert.True(t, empty.Complete)
|
||||
assert.Empty(t, empty.Pools)
|
||||
}
|
||||
|
||||
func TestGetDetailFailureReturnsIncompleteCachedInventory(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{{Name: "tank"}}, nil
|
||||
}
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) {
|
||||
return []zfs.PoolStatus{{Name: "tank", Vdevs: []zfs.VdevStatus{{Name: "mirror-0"}}}}, nil
|
||||
}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{{Name: "tank/data"}}, nil
|
||||
}
|
||||
first := zm.GetDetail(false)
|
||||
require.True(t, first.Complete)
|
||||
require.Len(t, first.Pools[0].Vdevs, 1)
|
||||
require.Len(t, first.Pools[0].Datasets, 1)
|
||||
|
||||
zm.backends[0].poolStatusesFn = func() ([]zfs.PoolStatus, error) { return nil, zfs.ErrNoZfs }
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) { return nil, zfs.ErrNoZfs }
|
||||
partial := zm.GetDetail(true)
|
||||
require.True(t, partial.Complete)
|
||||
require.Len(t, partial.Pools[0].Vdevs, 1)
|
||||
require.Len(t, partial.Pools[0].Datasets, 1)
|
||||
|
||||
zm.backends[0].poolStatsFn = func() ([]zfs.PoolStat, error) { return nil, zfs.ErrNoZfs }
|
||||
lastSuccessfulRefresh := zm.backends[0].lastDetailRefresh
|
||||
failed := zm.GetDetail(true)
|
||||
assert.False(t, failed.Complete)
|
||||
require.Len(t, failed.Pools, 1)
|
||||
assert.Equal(t, "tank", failed.Pools[0].Name)
|
||||
assert.Equal(t, lastSuccessfulRefresh, zm.backends[0].lastDetailRefresh)
|
||||
}
|
||||
|
||||
func TestZfsMountpoints(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
|
||||
zm.backends[0].datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
return []zfs.Dataset{
|
||||
{Name: "tank", Mountpoint: "/tank"},
|
||||
{Name: "rpool/ROOT/pve-1", Mountpoint: "/"},
|
||||
}, nil
|
||||
}
|
||||
mountpoints := zm.ZfsMountpoints()
|
||||
assert.Len(t, mountpoints, 2)
|
||||
assert.True(t, mountpoints["/tank"])
|
||||
assert.True(t, mountpoints["/"])
|
||||
}
|
||||
|
||||
func TestBtrfsRawCapacityPropagates(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolStatsFn: func() ([]zfs.PoolStat, error) {
|
||||
return []zfs.PoolStat{btrfsPoolStats(btrfs.Filesystem{UUID: "raw", Name: "raw", Size: 200, Alloc: 100, Raw: true})}, nil
|
||||
},
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) { return nil, nil },
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return nil, nil }}}}
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.True(t, stats.ZfsPools["b:raw"].Raw)
|
||||
detail := zm.GetDetail(true)
|
||||
require.True(t, detail.Complete)
|
||||
require.Len(t, detail.Pools, 1)
|
||||
assert.True(t, detail.Pools[0].Raw)
|
||||
}
|
||||
|
||||
func TestMarkDuplicatePoolCharts(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name, poolID, device string
|
||||
raw bool
|
||||
diskTotal float64
|
||||
wantUsage, wantIO bool
|
||||
}{
|
||||
{"single device root", "fs1", "dm-0", false, 100, true, true},
|
||||
{"multi device", "fs1", "", false, 100, true, false},
|
||||
{"different IO device", "fs1", "nvme0n1", false, 100, true, false},
|
||||
{"different filesystem", "fs2", "dm-0", false, 100, false, false},
|
||||
{"unknown identity", "", "dm-0", false, 100, false, false},
|
||||
{"raw usage", "fs1", "dm-0", true, 100, false, true},
|
||||
{"failed disk collection", "fs1", "dm-0", false, 0, false, false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolData: []zfs.PoolStat{{Name: "arbitrary label", MountID: tc.poolID, IODevice: tc.device, Raw: tc.raw}}}}}
|
||||
stats := &system.Stats{ZfsPools: map[string]*system.ZfsPool{"arbitrary label": {}}}
|
||||
fs := map[string]*system.FsStats{"dm-0": {Root: true, Mountpoint: "/", DiskTotal: tc.diskTotal}}
|
||||
zm.markDuplicateCharts(stats, fs, func(string) string { return "fs1" })
|
||||
assert.Equal(t, tc.wantUsage, stats.ZfsPools["arbitrary label"].HideUsage)
|
||||
assert.Equal(t, tc.wantIO, stats.ZfsPools["arbitrary label"].HideIO)
|
||||
// Bind mounts and custom extra-filesystem names have the same identity.
|
||||
fs["dm-0"].Root = false
|
||||
fs["dm-0"].Mountpoint = "/extra-filesystems/storage"
|
||||
fs["dm-0"].Name = "custom name"
|
||||
stats.ZfsPools["arbitrary label"] = &system.ZfsPool{}
|
||||
zm.markDuplicateCharts(stats, fs, func(string) string { return "fs1" })
|
||||
assert.Equal(t, tc.wantUsage, stats.ZfsPools["arbitrary label"].HideUsage)
|
||||
assert.Equal(t, tc.wantIO, stats.ZfsPools["arbitrary label"].HideIO)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBtrfsPoolIdentities(t *testing.T) {
|
||||
old := btrfsFilesystems
|
||||
t.Cleanup(func() { btrfsFilesystems = old })
|
||||
label := "tank"
|
||||
btrfsFilesystems = func() ([]btrfs.Filesystem, error) {
|
||||
return []btrfs.Filesystem{
|
||||
{UUID: "11111111-1111-4111-8111-111111111111", Name: label, Size: 100, Health: "ONLINE", NRead: 100, Devices: []btrfs.Device{{Name: "first"}}},
|
||||
{UUID: "22222222-2222-4222-8222-222222222222", Name: "tank", Size: 200, Health: "DEGRADED", NRead: 200, Devices: []btrfs.Device{{Name: "second"}}},
|
||||
}, nil
|
||||
}
|
||||
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs", poolStatsFn: func() ([]zfs.PoolStat, error) { return []zfs.PoolStat{{Name: "tank", Size: 300}}, nil },
|
||||
kernelStatsFn: func() ([]zfs.PoolKernelStat, error) { return []zfs.PoolKernelStat{{Name: "tank", NRead: 300}}, nil },
|
||||
poolStatusesFn: func() ([]zfs.PoolStatus, error) {
|
||||
return []zfs.PoolStatus{{Name: "tank", Vdevs: []zfs.VdevStatus{{Name: "zfs-device"}}}}, nil
|
||||
},
|
||||
|
||||
datasetsFn: func() ([]zfs.Dataset, error) { return []zfs.Dataset{{Name: "tank/data"}}, nil }}, newBtrfsBackend()}}
|
||||
first := "b:11111111-1111-4111-8111-111111111111"
|
||||
second := "b:22222222-2222-4222-8222-222222222222"
|
||||
var stats system.Stats
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 3)
|
||||
assert.Contains(t, stats.ZfsPools, "tank")
|
||||
assert.Equal(t, "ONLINE", stats.ZfsPools[first].Health)
|
||||
assert.Equal(t, "DEGRADED", stats.ZfsPools[second].Health)
|
||||
assert.Equal(t, uint64(100), zm.backends[1].kernelSamples[first].nread)
|
||||
assert.Equal(t, uint64(200), zm.backends[1].kernelSamples[second].nread)
|
||||
detail := zm.GetDetail(true)
|
||||
require.Len(t, detail.Pools, 3)
|
||||
assert.Equal(t, "zfs-device", detail.Pools[0].Vdevs[0].Name)
|
||||
assert.Len(t, detail.Pools[0].Datasets, 1)
|
||||
assert.Equal(t, "first", detail.Pools[1].Vdevs[0].Name)
|
||||
assert.Empty(t, detail.Pools[1].Datasets)
|
||||
assert.Equal(t, "second", detail.Pools[2].Vdevs[0].Name)
|
||||
label = "renamed"
|
||||
zm.backends[0].lastPoolStats = time.Time{}
|
||||
zm.backends[1].lastPoolStats = time.Time{}
|
||||
zm.Update(&stats)
|
||||
require.Len(t, stats.ZfsPools, 3)
|
||||
assert.Equal(t, "renamed", stats.ZfsPools[first].DisplayName)
|
||||
assert.Equal(t, first, zm.GetDetail(true).Pools[1].Name)
|
||||
assert.Equal(t, "renamed", zm.GetDetail(true).Pools[1].DisplayName)
|
||||
}
|
||||
|
||||
func TestStaleUtilityCachesRefreshInBackground(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
b := &poolBackend{name: "zfs"}
|
||||
b.poolStatsFn = func() ([]zfs.PoolStat, error) {
|
||||
<-release
|
||||
return []zfs.PoolStat{{Name: "new"}}, nil
|
||||
}
|
||||
b.datasetsFn = func() ([]zfs.Dataset, error) {
|
||||
<-release
|
||||
return []zfs.Dataset{{Name: "new", Mountpoint: "/new"}}, nil
|
||||
}
|
||||
b.poolData = []zfs.PoolStat{{Name: "old"}}
|
||||
b.lastPoolStats = time.Now().Add(-2 * poolStatsRefreshInterval)
|
||||
b.datasetUsage = map[string]zfsDatasetUsage{"/old": {}}
|
||||
b.lastUsageRefresh = time.Now().Add(-2 * datasetUsageRefreshInterval)
|
||||
|
||||
// A hung utility must not block collection; cached data is served meanwhile.
|
||||
for range 2 {
|
||||
assert.Equal(t, "old", b.poolStats()[0].Name)
|
||||
assert.Contains(t, b.refreshDatasetUsage(), "/old")
|
||||
}
|
||||
|
||||
close(release)
|
||||
require.Eventually(t, func() bool {
|
||||
return b.poolStats()[0].Name == "new" && b.refreshDatasetUsage()["/new"] == zfsDatasetUsage{}
|
||||
}, time.Second, time.Millisecond)
|
||||
}
|
||||
157
agent/system.go
157
agent/system.go
@@ -4,6 +4,7 @@ import (
|
||||
"bufio"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"os"
|
||||
"runtime"
|
||||
@@ -11,7 +12,9 @@ import (
|
||||
|
||||
"github.com/henrygd/beszel"
|
||||
"github.com/henrygd/beszel/agent/battery"
|
||||
"github.com/henrygd/beszel/agent/btrfs"
|
||||
"github.com/henrygd/beszel/agent/utils"
|
||||
"github.com/henrygd/beszel/agent/wifi"
|
||||
"github.com/henrygd/beszel/agent/zfs"
|
||||
"github.com/henrygd/beszel/internal/entities/container"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
@@ -31,7 +34,11 @@ func (a *Agent) refreshSystemDetails() {
|
||||
|
||||
if a.dockerManager != nil {
|
||||
a.systemDetails.Podman = a.dockerManager.IsPodman()
|
||||
hostInfo, _ = a.dockerManager.GetHostInfo()
|
||||
// Docker's host info describes the machine its daemon runs on. On macOS and
|
||||
// Windows that is a Linux VM, so its CPU and memory totals are not this host's.
|
||||
if runtime.GOOS != "darwin" && runtime.GOOS != "windows" {
|
||||
hostInfo, _ = a.dockerManager.GetHostInfo()
|
||||
}
|
||||
}
|
||||
|
||||
a.systemDetails.Hostname, _ = os.Hostname()
|
||||
@@ -78,6 +85,12 @@ func (a *Agent) refreshSystemDetails() {
|
||||
if info, err := cpu.Info(); err == nil && len(info) > 0 {
|
||||
a.systemDetails.CpuModel = info[0].ModelName
|
||||
}
|
||||
// gopsutil doesn't parse the "cpu model" field from /proc/cpuinfo, which
|
||||
// is the only source of the CPU model name on MIPS. Fall back to reading
|
||||
// it directly when ModelName is empty.
|
||||
if a.systemDetails.CpuModel == "" {
|
||||
a.systemDetails.CpuModel = getCpuModelFromCpuinfo()
|
||||
}
|
||||
// cores / threads
|
||||
cores, _ := cpu.Counts(false)
|
||||
threads := hostInfo.NCPU
|
||||
@@ -132,9 +145,14 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
var systemStats system.Stats
|
||||
|
||||
// battery
|
||||
if batteryPercent, batteryState, err := battery.GetBatteryStats(); err == nil {
|
||||
systemStats.Battery[0] = batteryPercent
|
||||
systemStats.Battery[1] = batteryState
|
||||
if batteries, err := battery.GetBatteryStats(); err == nil {
|
||||
systemStats.Batteries = make(map[string]uint8, len(batteries))
|
||||
for _, device := range batteries {
|
||||
systemStats.Batteries[device.Name] = device.Percent
|
||||
}
|
||||
if primary, ok := battery.Primary(batteries); ok {
|
||||
systemStats.Battery = [2]uint8{primary.Percent, primary.State}
|
||||
}
|
||||
}
|
||||
|
||||
// cpu metrics
|
||||
@@ -159,9 +177,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
|
||||
// load average
|
||||
if avgstat, err := load.Avg(); err == nil {
|
||||
systemStats.LoadAvg[0] = avgstat.Load1
|
||||
systemStats.LoadAvg[1] = avgstat.Load5
|
||||
systemStats.LoadAvg[2] = avgstat.Load15
|
||||
systemStats.LoadAvg[0] = utils.TwoDecimals(avgstat.Load1)
|
||||
systemStats.LoadAvg[1] = utils.TwoDecimals(avgstat.Load5)
|
||||
systemStats.LoadAvg[2] = utils.TwoDecimals(avgstat.Load15)
|
||||
slog.Debug("Load average", "5m", avgstat.Load5, "15m", avgstat.Load15)
|
||||
} else {
|
||||
slog.Error("Error getting load average", "err", err)
|
||||
@@ -169,21 +187,11 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
|
||||
// memory
|
||||
if v, err := mem.VirtualMemory(); err == nil {
|
||||
used, cacheBuff, swapUsed := calculateHostMemoryUsage(v, a.memCalc == "htop")
|
||||
// swap
|
||||
systemStats.Swap = utils.BytesToGigabytes(v.SwapTotal)
|
||||
systemStats.SwapUsed = utils.BytesToGigabytes(v.SwapTotal - v.SwapFree - v.SwapCached)
|
||||
// cache + buffers value for default mem calculation
|
||||
// note: gopsutil automatically adds SReclaimable to v.Cached
|
||||
cacheBuff := v.Cached + v.Buffers - v.Shared
|
||||
if cacheBuff <= 0 {
|
||||
cacheBuff = max(v.Total-v.Free-v.Used, 0)
|
||||
}
|
||||
// htop memory calculation overrides (likely outdated as of mid 2025)
|
||||
if a.memCalc == "htop" {
|
||||
// cacheBuff = v.Cached + v.Buffers - v.Shared
|
||||
v.Used = v.Total - (v.Free + cacheBuff)
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
}
|
||||
systemStats.SwapUsed = utils.BytesToGigabytes(swapUsed)
|
||||
v.Used = used
|
||||
// if a.memCalc == "legacy" {
|
||||
// v.Used = v.Total - v.Free - v.Buffers - v.Cached
|
||||
// cacheBuff = v.Total - v.Free - v.Used
|
||||
@@ -193,10 +201,14 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
if a.zfs {
|
||||
if arcSize, _ := zfs.ARCSize(); arcSize > 0 && arcSize < v.Used {
|
||||
v.Used = v.Used - arcSize
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
systemStats.MemZfsArc = utils.BytesToGigabytes(arcSize)
|
||||
}
|
||||
}
|
||||
if v.Total > 0 {
|
||||
v.UsedPercent = float64(v.Used) / float64(v.Total) * 100.0
|
||||
} else {
|
||||
v.UsedPercent = 0
|
||||
}
|
||||
systemStats.Mem = utils.BytesToGigabytes(v.Total)
|
||||
systemStats.MemBuffCache = utils.BytesToGigabytes(cacheBuff)
|
||||
systemStats.MemUsed = utils.BytesToGigabytes(v.Used)
|
||||
@@ -209,6 +221,10 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
// disk i/o (cache-aware per interval)
|
||||
a.updateDiskIo(cacheTimeMs, &systemStats)
|
||||
|
||||
// storage pool stats
|
||||
a.storagePoolManager.Update(&systemStats)
|
||||
a.storagePoolManager.markDuplicateCharts(&systemStats, a.fsStats, btrfs.MountID)
|
||||
|
||||
// network stats (per cache interval)
|
||||
a.updateNetworkStats(cacheTimeMs, &systemStats)
|
||||
|
||||
@@ -216,6 +232,9 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
// TODO: maybe refactor to methods on systemStats
|
||||
a.updateTemperatures(&systemStats)
|
||||
|
||||
// fan speeds (Linux-only; sysfs hwmon)
|
||||
a.updateFans(&systemStats)
|
||||
|
||||
// GPU data
|
||||
if a.gpuManager != nil {
|
||||
// reset high gpu percent
|
||||
@@ -249,6 +268,14 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
}
|
||||
}
|
||||
|
||||
// Wi-Fi collection spawns a process on macOS and dumps the BSS cache on
|
||||
// Linux, so only refresh on the default interval. Real-time requests reuse
|
||||
// the last snapshot.
|
||||
if cacheTimeMs == defaultDataCacheTimeMs {
|
||||
a.systemInfo.WiFi = wifi.Collect()
|
||||
}
|
||||
systemStats.WiFi = wifi.Signals(a.systemInfo.WiFi)
|
||||
|
||||
// update system info
|
||||
a.systemInfo.ConnectionType = a.connectionManager.ConnectionType
|
||||
a.systemInfo.Cpu = systemStats.Cpu
|
||||
@@ -256,13 +283,99 @@ func (a *Agent) getSystemStats(cacheTimeMs uint16) system.Stats {
|
||||
a.systemInfo.MemPct = systemStats.MemPct
|
||||
a.systemInfo.DiskPct = systemStats.DiskPct
|
||||
a.systemInfo.Battery = systemStats.Battery
|
||||
a.systemInfo.Uptime, _ = host.Uptime()
|
||||
a.systemInfo.Uptime, _ = getUptime()
|
||||
a.systemInfo.BandwidthBytes = systemStats.Bandwidth[0] + systemStats.Bandwidth[1]
|
||||
a.systemInfo.Threads = a.systemDetails.Threads
|
||||
|
||||
return systemStats
|
||||
}
|
||||
|
||||
// cpuModelFallbackKeys are the field names to look for in /proc/cpuinfo when
|
||||
// gopsutil fails to return a ModelName. The "cpu model" key is used on MIPS
|
||||
// (e.g. "MIPS 1004Kc V2.15"), while "system type" provides SoC information
|
||||
// on various embedded architectures.
|
||||
var cpuModelFallbackKeys = []string{"cpu model", "system type"}
|
||||
|
||||
// getCpuModelFromCpuinfo reads /proc/cpuinfo and returns a CPU model string.
|
||||
// This is a fallback for architectures where gopsutil's cpu.Info() does not
|
||||
// populate ModelName, most notably MIPS.
|
||||
func getCpuModelFromCpuinfo() string {
|
||||
file, err := os.Open("/proc/cpuinfo")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer file.Close()
|
||||
return parseCpuModel(file)
|
||||
}
|
||||
|
||||
// parseCpuModel scans r (expected to be /proc/cpuinfo content) and returns
|
||||
// a combined CPU model string. It collects values from all matching keys
|
||||
// and joins them with " / " when multiple are found.
|
||||
func parseCpuModel(r io.Reader) string {
|
||||
lines := readLines(r)
|
||||
var parts []string
|
||||
for _, key := range cpuModelFallbackKeys {
|
||||
for _, line := range lines {
|
||||
after, found := strings.CutPrefix(line, key)
|
||||
if !found {
|
||||
continue
|
||||
}
|
||||
after = strings.TrimSpace(after)
|
||||
if len(after) < 2 || after[0] != ':' {
|
||||
continue
|
||||
}
|
||||
if value := strings.TrimSpace(after[1:]); value != "" {
|
||||
parts = append(parts, value)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, " / ")
|
||||
}
|
||||
|
||||
// readLines reads all lines from r into a slice.
|
||||
func readLines(r io.Reader) []string {
|
||||
scanner := bufio.NewScanner(r)
|
||||
var lines []string
|
||||
for scanner.Scan() {
|
||||
lines = append(lines, scanner.Text())
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
// calculateHostMemoryUsage derives counters defensively because /proc/meminfo may
|
||||
// change while gopsutil reads it. Invalid unsigned subtractions saturate at zero.
|
||||
func calculateHostMemoryUsage(v *mem.VirtualMemoryStat, htop bool) (used, cacheBuff, swapUsed uint64) {
|
||||
used = v.Used
|
||||
if used > v.Total {
|
||||
used = saturatingSub(v.Total, v.Available)
|
||||
}
|
||||
|
||||
// gopsutil automatically adds SReclaimable to Cached.
|
||||
cacheBuff = min(v.Cached, v.Total)
|
||||
cacheBuff += min(v.Buffers, v.Total-cacheBuff)
|
||||
cacheBuff = saturatingSub(cacheBuff, min(v.Shared, v.Total))
|
||||
if v.Cached == 0 && v.Buffers == 0 {
|
||||
cacheBuff = saturatingSub(v.Total, v.Free, used)
|
||||
}
|
||||
if htop {
|
||||
used = saturatingSub(v.Total, v.Free, cacheBuff)
|
||||
}
|
||||
// Cached swap pages still occupy swap slots and are included in `free`'s used value.
|
||||
return used, cacheBuff, saturatingSub(v.SwapTotal, v.SwapFree)
|
||||
}
|
||||
|
||||
// saturatingSub subtracts each value, returning zero on underflow.
|
||||
func saturatingSub(value uint64, subtrahends ...uint64) uint64 {
|
||||
for _, subtrahend := range subtrahends {
|
||||
if subtrahend > value {
|
||||
return 0
|
||||
}
|
||||
value -= subtrahend
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
// getOsPrettyName attempts to get the pretty OS name from /etc/os-release on Linux systems
|
||||
func getOsPrettyName() (string, error) {
|
||||
file, err := os.Open("/etc/os-release")
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/henrygd/beszel/internal/common"
|
||||
"github.com/henrygd/beszel/internal/entities/system"
|
||||
"github.com/shirou/gopsutil/v4/mem"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
@@ -33,6 +35,59 @@ func TestGatherStatsDoesNotAttachDetailsToCachedRequests(t *testing.T) {
|
||||
assert.Nil(t, secondResponse.Details)
|
||||
}
|
||||
|
||||
func TestCalculateHostMemoryUsage(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
memory mem.VirtualMemoryStat
|
||||
htop bool
|
||||
used, cacheBuff, swapUsed uint64
|
||||
}{
|
||||
{
|
||||
name: "normal",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 40, Used: 60, Free: 20, Cached: 25, Buffers: 10, Shared: 5, SwapTotal: 20, SwapFree: 8, SwapCached: 2},
|
||||
used: 60,
|
||||
cacheBuff: 30,
|
||||
swapUsed: 12,
|
||||
},
|
||||
{
|
||||
name: "inconsistent counters saturate",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 110, Used: ^uint64(0) - 9, Free: 90, Cached: 5, Buffers: 10, Shared: 20, SwapTotal: 10, SwapFree: 9, SwapCached: 2},
|
||||
used: 0,
|
||||
cacheBuff: 0,
|
||||
swapUsed: 1,
|
||||
},
|
||||
{
|
||||
name: "htop subtraction saturates",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Available: 20, Used: 80, Free: 90, Cached: 20, Buffers: 5, SwapTotal: 30, SwapFree: 10, SwapCached: 5},
|
||||
htop: true,
|
||||
used: 0,
|
||||
cacheBuff: 25,
|
||||
swapUsed: 20,
|
||||
},
|
||||
{
|
||||
name: "zero cache from shared cancellation does not fall back",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Used: 60, Free: 10, Cached: 20, Buffers: 10, Shared: 30},
|
||||
used: 60,
|
||||
cacheBuff: 0,
|
||||
},
|
||||
{
|
||||
name: "absent cache counters use fallback",
|
||||
memory: mem.VirtualMemoryStat{Total: 100, Used: 60, Free: 10},
|
||||
used: 60,
|
||||
cacheBuff: 30,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
used, cacheBuff, swapUsed := calculateHostMemoryUsage(&tt.memory, tt.htop)
|
||||
assert.Equal(t, tt.used, used)
|
||||
assert.Equal(t, tt.cacheBuff, cacheBuff)
|
||||
assert.Equal(t, tt.swapUsed, swapUsed)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUpdateSystemDetailsMarksDetailsDirty(t *testing.T) {
|
||||
agent := &Agent{}
|
||||
|
||||
@@ -59,3 +114,81 @@ func TestUpdateSystemDetailsMarksDetailsDirty(t *testing.T) {
|
||||
assert.False(t, agent.detailsDirty)
|
||||
assert.Nil(t, original.Details)
|
||||
}
|
||||
|
||||
func TestParseCpuModel(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
expected string
|
||||
}{
|
||||
{
|
||||
name: "MIPS with both cpu model and system type",
|
||||
input: `system type : MediaTek MT7621 ver:1 eco:3
|
||||
machine : ASUS RT-AX53U
|
||||
processor : 0
|
||||
cpu model : MIPS 1004Kc V2.15
|
||||
BogoMIPS : 586.13
|
||||
wait instruction : yes`,
|
||||
expected: "MIPS 1004Kc V2.15 / MediaTek MT7621 ver:1 eco:3",
|
||||
},
|
||||
{
|
||||
name: "MIPS with different SoC",
|
||||
input: `system type : Atheros AR7161 rev 2
|
||||
machine : NETGEAR WNDR3700
|
||||
processor : 0
|
||||
cpu model : MIPS 24Kc V7.4
|
||||
BogoMIPS : 452.19`,
|
||||
expected: "MIPS 24Kc V7.4 / Atheros AR7161 rev 2",
|
||||
},
|
||||
{
|
||||
name: "only system type when cpu model missing",
|
||||
input: `system type : Broadcom BCM47xx
|
||||
processor : 0
|
||||
BogoMIPS : 296.11`,
|
||||
expected: "Broadcom BCM47xx",
|
||||
},
|
||||
{
|
||||
name: "only cpu model when system type missing",
|
||||
input: `processor : 0
|
||||
cpu model : MIPS 34Kc V2.15
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "MIPS 34Kc V2.15",
|
||||
},
|
||||
{
|
||||
name: "x86 cpuinfo returns empty",
|
||||
input: `processor : 0
|
||||
vendor_id : GenuineIntel
|
||||
cpu family : 6
|
||||
model : 142
|
||||
model name : Intel(R) Core(TM) i5-8250U CPU @ 1.60GHz
|
||||
stepping : 10`,
|
||||
expected: "",
|
||||
},
|
||||
{
|
||||
name: "empty input",
|
||||
input: "",
|
||||
expected: "",
|
||||
},
|
||||
{
|
||||
name: "cpu model with extra whitespace",
|
||||
input: `processor : 0
|
||||
cpu model : MIPS 34Kc V2.15
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "MIPS 34Kc V2.15",
|
||||
},
|
||||
{
|
||||
name: "cpu model without value",
|
||||
input: `processor : 0
|
||||
cpu model :
|
||||
BogoMIPS : 300.00`,
|
||||
expected: "",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := parseCpuModel(strings.NewReader(tt.input))
|
||||
assert.Equal(t, tt.expected, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
apk-tools-2.14.4-r1 aarch64 {apk-tools} (GPL-2.0-only) [upgradable from: apk-tools-2.14.4-r0]
|
||||
busybox-1.36.1-r31 aarch64 {busybox} (GPL-2.0-only) [upgradable from: busybox-1.36.1-r28]
|
||||
busybox-binsh-1.36.1-r31 aarch64 {busybox} (GPL-2.0-only) [upgradable from: busybox-binsh-1.36.1-r28]
|
||||
ca-certificates-bundle-20260413-r0 aarch64 {ca-certificates} (MPL-2.0 AND MIT) [upgradable from: ca-certificates-bundle-20240226-r0]
|
||||
libcrypto3-3.3.7-r0 aarch64 {openssl} (Apache-2.0) [upgradable from: libcrypto3-3.3.0-r2]
|
||||
libssl3-3.3.7-r0 aarch64 {openssl} (Apache-2.0) [upgradable from: libssl3-3.3.0-r2]
|
||||
musl-1.2.5-r3 aarch64 {musl} (MIT) [upgradable from: musl-1.2.5-r0]
|
||||
musl-utils-1.2.5-r3 aarch64 {musl} (MIT AND BSD-2-Clause AND GPL-2.0-or-later) [upgradable from: musl-utils-1.2.5-r0]
|
||||
ssl_client-1.36.1-r31 aarch64 {busybox} (GPL-2.0-only) [upgradable from: ssl_client-1.36.1-r28]
|
||||
zlib-1.3.2-r0 aarch64 {zlib} (Zlib) [upgradable from: zlib-1.3.1-r1]
|
||||
101
agent/test-data/package_updates/apt_debian12.txt
Normal file
101
agent/test-data/package_updates/apt_debian12.txt
Normal file
@@ -0,0 +1,101 @@
|
||||
Reading package lists...
|
||||
Building dependency tree...
|
||||
Reading state information...
|
||||
Calculating upgrade...
|
||||
The following packages will be upgraded:
|
||||
base-files bash bsdutils debian-archive-keyring dpkg e2fsprogs gcc-12-base
|
||||
gpgv init-system-helpers libblkid1 libc-bin libc6 libcap2 libcom-err2
|
||||
libext2fs2 libgcc-s1 libgcrypt20 libgnutls30 liblzma5 libmount1
|
||||
libpam-modules libpam-modules-bin libpam-runtime libpam0g libpcre2-8-0
|
||||
libseccomp2 libsmartcols1 libss2 libstdc++6 libsystemd0 libtasn1-6 libudev1
|
||||
libuuid1 login logsave mount passwd perl-base sed tar tzdata usr-is-merged
|
||||
util-linux util-linux-extra
|
||||
44 upgraded, 0 newly installed, 0 to remove and 0 not upgraded.
|
||||
Inst base-files [12.4+deb12u4] (12.4+deb12u15 Debian:12.15/oldstable [arm64])
|
||||
Conf base-files (12.4+deb12u15 Debian:12.15/oldstable [arm64])
|
||||
Inst bash [5.2.15-2+b2] (5.2.15-2+b13 Debian:12.15/oldstable [arm64])
|
||||
Conf bash (5.2.15-2+b13 Debian:12.15/oldstable [arm64])
|
||||
Inst bsdutils [1:2.38.1-5+b1] (1:2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf bsdutils (1:2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst tar [1.34+dfsg-1.2] (1.34+dfsg-1.2+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Conf tar (1.34+dfsg-1.2+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Inst dpkg [1.21.22] (1.21.23 Debian:12.15/oldstable [arm64])
|
||||
Conf dpkg (1.21.23 Debian:12.15/oldstable [arm64])
|
||||
Inst login [1:4.13+dfsg1-1+b1] (1:4.13+dfsg1-1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf login (1:4.13+dfsg1-1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst perl-base [5.36.0-7+deb12u1] (5.36.0-7+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf perl-base (5.36.0-7+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst sed [4.9-1] (4.9-1+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Conf sed (4.9-1+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Inst gcc-12-base [12.2.0-14] (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64]) [libstdc++6:arm64 libgcc-s1:arm64 ]
|
||||
Conf gcc-12-base (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64]) [libstdc++6:arm64 libgcc-s1:arm64 ]
|
||||
Inst libgcc-s1 [12.2.0-14] (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64]) [libstdc++6:arm64 ]
|
||||
Conf libgcc-s1 (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64]) [libstdc++6:arm64 ]
|
||||
Inst libstdc++6 [12.2.0-14] (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Conf libstdc++6 (12.2.0-14+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Inst libc6 [2.36-9+deb12u3] (2.36-9+deb12u14 Debian:12.15/oldstable [arm64])
|
||||
Conf libc6 (2.36-9+deb12u14 Debian:12.15/oldstable [arm64])
|
||||
Inst libsmartcols1 [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf libsmartcols1 (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst util-linux-extra [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf util-linux-extra (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst util-linux [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf util-linux (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst usr-is-merged [35] (37~deb12u1 Debian:12.15/oldstable [all])
|
||||
Conf usr-is-merged (37~deb12u1 Debian:12.15/oldstable [all])
|
||||
Inst init-system-helpers [1.65.2] (1.65.2+deb12u1 Debian:12.15/oldstable [all])
|
||||
Conf init-system-helpers (1.65.2+deb12u1 Debian:12.15/oldstable [all])
|
||||
Inst libc-bin [2.36-9+deb12u3] (2.36-9+deb12u14 Debian:12.15/oldstable [arm64])
|
||||
Conf libc-bin (2.36-9+deb12u14 Debian:12.15/oldstable [arm64])
|
||||
Inst libpam0g [1.5.2-6+deb12u1] (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf libpam0g (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst libpam-modules-bin [1.5.2-6+deb12u1] (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64]) [libpam-modules:arm64 on libpam-modules-bin:arm64] [libpam-modules:arm64 ]
|
||||
Conf libpam-modules-bin (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64]) [libpam-modules:arm64 ]
|
||||
Inst libpam-modules [1.5.2-6+deb12u1] (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf libpam-modules (1.5.2-6+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst logsave [1.47.0-2] (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Inst libext2fs2 [1.47.0-2] (1.47.0-2+b2 Debian:12.15/oldstable [arm64]) [e2fsprogs:arm64 on libext2fs2:arm64] [e2fsprogs:arm64 ]
|
||||
Conf libext2fs2 (1.47.0-2+b2 Debian:12.15/oldstable [arm64]) [e2fsprogs:arm64 ]
|
||||
Inst e2fsprogs [1.47.0-2] (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Inst mount [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst libpam-runtime [1.5.2-6+deb12u1] (1.5.2-6+deb12u2 Debian:12.15/oldstable [all])
|
||||
Conf libpam-runtime (1.5.2-6+deb12u2 Debian:12.15/oldstable [all])
|
||||
Inst passwd [1:4.13+dfsg1-1+b1] (1:4.13+dfsg1-1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf passwd (1:4.13+dfsg1-1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst debian-archive-keyring [2023.3+deb12u1] (2023.3+deb12u2 Debian:12.15/oldstable [all])
|
||||
Conf debian-archive-keyring (2023.3+deb12u2 Debian:12.15/oldstable [all])
|
||||
Inst libgcrypt20 [1.10.1-3] (1.10.1-3+deb12u1 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Conf libgcrypt20 (1.10.1-3+deb12u1 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Inst gpgv [2.2.40-1.1] (2.2.40-1.1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf gpgv (2.2.40-1.1+deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst libblkid1 [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf libblkid1 (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst libcap2 [1:2.66-4] (1:2.66-4+deb12u3+b1 Debian:12.15/oldstable [arm64])
|
||||
Conf libcap2 (1:2.66-4+deb12u3+b1 Debian:12.15/oldstable [arm64])
|
||||
Inst libtasn1-6 [4.19.0-2] (4.19.0-2+deb12u1 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Conf libtasn1-6 (4.19.0-2+deb12u1 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Inst libgnutls30 [3.7.9-2+deb12u1] (3.7.9-2+deb12u7 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Conf libgnutls30 (3.7.9-2+deb12u7 Debian:12.15/oldstable, Debian-Security:12/oldstable-security [arm64])
|
||||
Inst liblzma5 [5.4.1-0.2] (5.4.1-1+deb12u2 Debian-Security:12/oldstable-security [arm64])
|
||||
Conf liblzma5 (5.4.1-1+deb12u2 Debian-Security:12/oldstable-security [arm64])
|
||||
Inst libmount1 [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf libmount1 (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst libpcre2-8-0 [10.42-1] (10.42-1+deb12u1 Debian-Security:12/oldstable-security [arm64])
|
||||
Conf libpcre2-8-0 (10.42-1+deb12u1 Debian-Security:12/oldstable-security [arm64])
|
||||
Inst libseccomp2 [2.5.4-1+b3] (2.5.4-1+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Conf libseccomp2 (2.5.4-1+deb12u1 Debian:12.15/oldstable [arm64])
|
||||
Inst libsystemd0 [252.19-1~deb12u1] (252.39-1~deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf libsystemd0 (252.39-1~deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst libudev1 [252.19-1~deb12u1] (252.39-1~deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Conf libudev1 (252.39-1~deb12u2 Debian:12.15/oldstable [arm64])
|
||||
Inst libuuid1 [2.38.1-5+b1] (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf libuuid1 (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Inst tzdata [2023c-5+deb12u1] (2026b-0+deb12u1 Debian:12.15/oldstable [all])
|
||||
Inst libcom-err2 [1.47.0-2] (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Inst libss2 [1.47.0-2] (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Conf logsave (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Conf e2fsprogs (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Conf mount (2.38.1-5+deb12u3 Debian:12.15/oldstable [arm64])
|
||||
Conf tzdata (2026b-0+deb12u1 Debian:12.15/oldstable [all])
|
||||
Conf libcom-err2 (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
Conf libss2 (1.47.0-2+b2 Debian:12.15/oldstable [arm64])
|
||||
130
agent/test-data/package_updates/apt_ubuntu2204.txt
Normal file
130
agent/test-data/package_updates/apt_ubuntu2204.txt
Normal file
@@ -0,0 +1,130 @@
|
||||
Reading package lists...
|
||||
Building dependency tree...
|
||||
Reading state information...
|
||||
Calculating upgrade...
|
||||
The following packages will be upgraded:
|
||||
apt base-files bash bsdutils coreutils diffutils dpkg e2fsprogs gcc-12-base
|
||||
gpgv gzip libapt-pkg6.0 libattr1 libblkid1 libbz2-1.0 libc-bin libc6 libcap2
|
||||
libcom-err2 libext2fs2 libgcc-s1 libgcrypt20 libgnutls30 libgssapi-krb5-2
|
||||
libk5crypto3 libkrb5-3 libkrb5support0 liblzma5 libmount1 libncurses6
|
||||
libncursesw6 libp11-kit0 libpam-modules libpam-modules-bin libpam-runtime
|
||||
libpam0g libprocps8 libseccomp2 libsmartcols1 libss2 libssl3 libstdc++6
|
||||
libsystemd0 libtasn1-6 libtinfo6 libudev1 libuuid1 login logsave mount
|
||||
ncurses-base ncurses-bin passwd perl-base procps sed tar util-linux
|
||||
58 upgraded, 0 newly installed, 0 to remove and 0 not upgraded.
|
||||
Inst gcc-12-base [12.3.0-1ubuntu1~22.04] (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libstdc++6:arm64 libgcc-s1:arm64 ]
|
||||
Conf gcc-12-base (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libstdc++6:arm64 libgcc-s1:arm64 ]
|
||||
Inst libgcc-s1 [12.3.0-1ubuntu1~22.04] (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libstdc++6:arm64 ]
|
||||
Conf libgcc-s1 (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libstdc++6:arm64 ]
|
||||
Inst libstdc++6 [12.3.0-1ubuntu1~22.04] (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libstdc++6 (12.3.0-1ubuntu1~22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libc6 [2.35-0ubuntu3.4] (2.35-0ubuntu3.15 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libc6 (2.35-0ubuntu3.15 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst base-files [12ubuntu4.4] (12ubuntu4.7 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf base-files (12ubuntu4.7 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst bash [5.1-6ubuntu1] (5.1-6ubuntu1.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf bash (5.1-6ubuntu1.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst bsdutils [1:2.37.2-4ubuntu3] (1:2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf bsdutils (1:2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst coreutils [8.32-4.1ubuntu1] (8.32-4.1ubuntu1.4 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf coreutils (8.32-4.1ubuntu1.4 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst diffutils [1:3.8-0ubuntu2] (1:3.8-0ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf diffutils (1:3.8-0ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libbz2-1.0 [1.0.8-5build1] (1.0.8-5ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libbz2-1.0 (1.0.8-5ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libgcrypt20 [1.9.4-3ubuntu3] (1.9.4-3ubuntu3.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libgcrypt20 (1.9.4-3ubuntu3.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst liblzma5 [5.2.5-2ubuntu1] (5.2.5-2ubuntu1.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf liblzma5 (5.2.5-2ubuntu1.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libsystemd0 [249.11-0ubuntu3.10] (249.11-0ubuntu3.22 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libsystemd0 (249.11-0ubuntu3.22 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libudev1 [249.11-0ubuntu3.10] (249.11-0ubuntu3.22 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libudev1 (249.11-0ubuntu3.22 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libapt-pkg6.0 [2.4.10] (2.4.14 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf libapt-pkg6.0 (2.4.14 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst tar [1.34+dfsg-1ubuntu0.1.22.04.1] (1.34+dfsg-1ubuntu0.1.22.04.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf tar (1.34+dfsg-1ubuntu0.1.22.04.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst dpkg [1.21.1ubuntu2.2] (1.21.1ubuntu2.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf dpkg (1.21.1ubuntu2.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst gzip [1.10-4ubuntu4.1] (1.10-4ubuntu4.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf gzip (1.10-4ubuntu4.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst login [1:4.8.1-2ubuntu2.1] (1:4.8.1-2ubuntu2.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf login (1:4.8.1-2ubuntu2.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst ncurses-bin [6.3-2ubuntu0.1] (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf ncurses-bin (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst perl-base [5.34.0-3ubuntu1.2] (5.34.0-3ubuntu1.9 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf perl-base (5.34.0-3ubuntu1.9 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst sed [4.8-1ubuntu2] (4.8-1ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf sed (4.8-1ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst util-linux [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf util-linux (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libc-bin [2.35-0ubuntu3.4] (2.35-0ubuntu3.15 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libc-bin (2.35-0ubuntu3.15 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst ncurses-base [6.3-2ubuntu0.1] (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [all])
|
||||
Conf ncurses-base (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [all])
|
||||
Inst gpgv [2.2.27-3ubuntu2.1] (2.2.27-3ubuntu2.5 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf gpgv (2.2.27-3ubuntu2.5 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libp11-kit0 [0.24.0-6build1] (0.24.0-6ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libp11-kit0 (0.24.0-6ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libtasn1-6 [4.18.0-4build1] (4.18.0-4ubuntu0.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libtasn1-6 (4.18.0-4ubuntu0.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libgnutls30 [3.7.3-4ubuntu1.2] (3.7.3-4ubuntu1.9 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libgnutls30 (3.7.3-4ubuntu1.9 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libseccomp2 [2.5.3-2ubuntu2] (2.5.3-2ubuntu3~22.04.1 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf libseccomp2 (2.5.3-2ubuntu3~22.04.1 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst apt [2.4.10] (2.4.14 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf apt (2.4.14 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst libpam0g [1.4.0-11ubuntu2.3] (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libpam0g (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libpam-modules-bin [1.4.0-11ubuntu2.3] (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libpam-modules:arm64 on libpam-modules-bin:arm64] [libpam-modules:arm64 ]
|
||||
Conf libpam-modules-bin (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) [libpam-modules:arm64 ]
|
||||
Inst libpam-modules [1.4.0-11ubuntu2.3] (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libpam-modules (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst logsave [1.46.5-2ubuntu1.1] (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst libext2fs2 [1.46.5-2ubuntu1.1] (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64]) [e2fsprogs:arm64 on libext2fs2:arm64] [e2fsprogs:arm64 ]
|
||||
Conf libext2fs2 (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64]) [e2fsprogs:arm64 ]
|
||||
Inst e2fsprogs [1.46.5-2ubuntu1.1] (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst mount [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libattr1 [1:2.5.1-1build1] (1:2.5.1-1ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libattr1 (1:2.5.1-1ubuntu0.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libblkid1 [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libblkid1 (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libcap2 [1:2.44-1ubuntu0.22.04.1] (1:2.44-1ubuntu0.22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libcap2 (1:2.44-1ubuntu0.22.04.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libcom-err2 [1.46.5-2ubuntu1.1] (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf libcom-err2 (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst libk5crypto3 [1.19.2-2ubuntu0.2] (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf libk5crypto3 (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst libkrb5support0 [1.19.2-2ubuntu0.2] (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64]) [libkrb5-3:arm64 ]
|
||||
Conf libkrb5support0 (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64]) [libkrb5-3:arm64 ]
|
||||
Inst libkrb5-3 [1.19.2-2ubuntu0.2] (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64]) [libgssapi-krb5-2:arm64 ]
|
||||
Conf libkrb5-3 (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64]) [libgssapi-krb5-2:arm64 ]
|
||||
Inst libgssapi-krb5-2 [1.19.2-2ubuntu0.2] (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf libgssapi-krb5-2 (1.19.2-2ubuntu0.10 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst libssl3 [3.0.2-0ubuntu1.10] (3.0.2-0ubuntu1.29 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libssl3 (3.0.2-0ubuntu1.29 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libmount1 [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libmount1 (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libpam-runtime [1.4.0-11ubuntu2.3] (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [all])
|
||||
Conf libpam-runtime (1.4.0-11ubuntu2.8 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [all])
|
||||
Inst libsmartcols1 [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libsmartcols1 (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libncurses6 [6.3-2ubuntu0.1] (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) []
|
||||
Inst libncursesw6 [6.3-2ubuntu0.1] (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64]) []
|
||||
Inst libtinfo6 [6.3-2ubuntu0.1] (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libtinfo6 (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libuuid1 [2.37.2-4ubuntu3] (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libuuid1 (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst passwd [1:4.8.1-2ubuntu2.1] (1:4.8.1-2ubuntu2.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf passwd (1:4.8.1-2ubuntu2.2 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libprocps8 [2:3.3.17-6ubuntu2] (2:3.3.17-6ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Inst libss2 [1.46.5-2ubuntu1.1] (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Inst procps [2:3.3.17-6ubuntu2] (2:3.3.17-6ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf logsave (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf e2fsprogs (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf mount (2.37.2-4ubuntu3.6 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libncurses6 (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libncursesw6 (6.3-2ubuntu0.3 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libprocps8 (2:3.3.17-6ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
Conf libss2 (1.46.5-2ubuntu1.2 Ubuntu:22.04/jammy-updates [arm64])
|
||||
Conf procps (2:3.3.17-6ubuntu2.1 Ubuntu:22.04/jammy-updates, Ubuntu:22.04/jammy-security [arm64])
|
||||
111
agent/test-data/package_updates/dnf4_rocky9_check_update.txt
Normal file
111
agent/test-data/package_updates/dnf4_rocky9_check_update.txt
Normal file
@@ -0,0 +1,111 @@
|
||||
|
||||
alternatives.aarch64 1.24-2.el9 baseos
|
||||
audit-libs.aarch64 3.1.5-8.el9 baseos
|
||||
basesystem.noarch 11-13.el9.0.1 baseos
|
||||
bash.aarch64 5.1.8-9.el9 baseos
|
||||
binutils.aarch64 2.35.2-72.el9 baseos
|
||||
binutils-gold.aarch64 2.35.2-72.el9 baseos
|
||||
bzip2-libs.aarch64 1.0.8-11.el9 baseos
|
||||
ca-certificates.noarch 2025.2.80_v9.0.305-91.el9 baseos
|
||||
coreutils-single.aarch64 8.32-41.el9_8.1 baseos
|
||||
cracklib.aarch64 2.9.6-28.el9 baseos
|
||||
cracklib-dicts.aarch64 2.9.6-28.el9 baseos
|
||||
crypto-policies.noarch 20260224-1.gitea0f072.el9 baseos
|
||||
crypto-policies-scripts.noarch 20260224-1.gitea0f072.el9 baseos
|
||||
curl-minimal.aarch64 7.76.1-40.el9_8.5 baseos
|
||||
cyrus-sasl-lib.aarch64 2.1.27-22.el9_7 baseos
|
||||
dnf.noarch 4.14.0-34.el9_8.rocky.0.1 baseos
|
||||
dnf-data.noarch 4.14.0-34.el9_8.rocky.0.1 baseos
|
||||
elfutils-debuginfod-client.aarch64 0.194-1.el9.rocky.0.1 baseos
|
||||
elfutils-default-yama-scope.noarch 0.194-1.el9.rocky.0.1 baseos
|
||||
elfutils-libelf.aarch64 0.194-1.el9.rocky.0.1 baseos
|
||||
elfutils-libs.aarch64 0.194-1.el9.rocky.0.1 baseos
|
||||
expat.aarch64 2.5.0-6.el9_8.3 baseos
|
||||
file-libs.aarch64 5.39-17.el9 baseos
|
||||
filesystem.aarch64 3.16-5.el9 baseos
|
||||
findutils.aarch64 1:4.8.0-7.el9 baseos
|
||||
gdbm-libs.aarch64 1:1.23-1.el9 baseos
|
||||
glib2.aarch64 2.68.4-19.el9_8.10 baseos
|
||||
glibc.aarch64 2.34-275.el9_8 baseos
|
||||
glibc-common.aarch64 2.34-275.el9_8 baseos
|
||||
glibc-minimal-langpack.aarch64 2.34-275.el9_8 baseos
|
||||
gnupg2.aarch64 2.3.3-5.el9_7 baseos
|
||||
gnutls.aarch64 3.8.10-8.el9_8 baseos
|
||||
gzip.aarch64 1.12-2.el9_8 baseos
|
||||
ima-evm-utils.aarch64 1.6.2-2.el9.rocky.0.2 baseos
|
||||
krb5-libs.aarch64 1.21.1-10.el9_8 baseos
|
||||
less.aarch64 590-6.el9 baseos
|
||||
libacl.aarch64 2.4.0-1.el9_8 baseos
|
||||
libarchive.aarch64 3.5.3-11.el9_8 baseos
|
||||
libatomic.aarch64 11.5.0-14.el9 baseos
|
||||
libattr.aarch64 2.6.0-1.el9_8 baseos
|
||||
libblkid.aarch64 2.37.4-25.el9 baseos
|
||||
libcap.aarch64 2.48-10.el9_7.1 baseos
|
||||
libcom_err.aarch64 1.46.5-8.el9 baseos
|
||||
libcurl-minimal.aarch64 7.76.1-40.el9_8.5 baseos
|
||||
libdb.aarch64 5.3.28-57.el9_6 baseos
|
||||
libdnf.aarch64 0.69.0-18.el9.rocky.0.1 baseos
|
||||
libeconf.aarch64 0.4.1-7.el9_8 baseos
|
||||
libevent.aarch64 2.1.13-1.el9_8 baseos
|
||||
libfdisk.aarch64 2.37.4-25.el9 baseos
|
||||
libgcc.aarch64 11.5.0-14.el9 baseos
|
||||
libgcrypt.aarch64 1.10.0-13.el9_8 baseos
|
||||
libgomp.aarch64 11.5.0-14.el9 baseos
|
||||
libksba.aarch64 1.5.1-7.el9 baseos
|
||||
libmount.aarch64 2.37.4-25.el9 baseos
|
||||
libnghttp2.aarch64 1.43.0-6.el9_8.2 baseos
|
||||
librepo.aarch64 1.19.0-1.el9 baseos
|
||||
libselinux.aarch64 3.6-3.el9 baseos
|
||||
libsemanage.aarch64 3.6-5.el9_6 baseos
|
||||
libsepol.aarch64 3.6-3.el9 baseos
|
||||
libsmartcols.aarch64 2.37.4-25.el9 baseos
|
||||
libsolv.aarch64 0.7.24-6.el9_8 baseos
|
||||
libstdc++.aarch64 11.5.0-14.el9 baseos
|
||||
libtasn1.aarch64 4.16.0-10.el9_8 baseos
|
||||
libusbx.aarch64 1.0.30-1.el9_8 baseos
|
||||
libuser.aarch64 0.63-17.el9 baseos
|
||||
libuuid.aarch64 2.37.4-25.el9 baseos
|
||||
libxml2.aarch64 2.9.13-14.el9_8.4 baseos
|
||||
libzstd.aarch64 1.5.5-1.el9 baseos
|
||||
mpfr.aarch64 4.1.0-10.el9 baseos
|
||||
ncurses-base.noarch 6.2-12.20210508.el9 baseos
|
||||
ncurses-libs.aarch64 6.2-12.20210508.el9 baseos
|
||||
nettle.aarch64 3.10.1-1.el9 baseos
|
||||
openldap.aarch64 2.6.8-4.el9.0.1 baseos
|
||||
openssl.aarch64 1:3.5.8-1.el9_8 baseos
|
||||
openssl-libs.aarch64 1:3.5.8-1.el9_8 baseos
|
||||
p11-kit.aarch64 0.26.4-1.el9_8 baseos
|
||||
p11-kit-trust.aarch64 0.26.4-1.el9_8 baseos
|
||||
pam.aarch64 1.5.1-28.el9_8.1 baseos
|
||||
pcre.aarch64 8.44-4.el9 baseos
|
||||
pcre2.aarch64 10.40-6.el9 baseos
|
||||
pcre2-syntax.noarch 10.40-6.el9 baseos
|
||||
python3.aarch64 3.9.25-7.el9_8.3 baseos
|
||||
python3-dnf.noarch 4.14.0-34.el9_8.rocky.0.1 baseos
|
||||
python3-hawkey.aarch64 0.69.0-18.el9.rocky.0.1 baseos
|
||||
python3-libdnf.aarch64 0.69.0-18.el9.rocky.0.1 baseos
|
||||
python3-libs.aarch64 3.9.25-7.el9_8.3 baseos
|
||||
python3-pip-wheel.noarch 21.3.1-2.el9_8.rocky.0.1 baseos
|
||||
python3-rpm.aarch64 4.16.1.3-40.el9 baseos
|
||||
python3-setuptools-wheel.noarch 53.0.0-15.el9 baseos
|
||||
rocky-gpg-keys.noarch 9.8-1.2.el9 baseos
|
||||
rocky-release.noarch 9.8-1.2.el9 baseos
|
||||
rocky-repos.noarch 9.8-1.2.el9 baseos
|
||||
rootfiles.noarch 8.1-35.el9 baseos
|
||||
rpm.aarch64 4.16.1.3-40.el9 baseos
|
||||
rpm-build-libs.aarch64 4.16.1.3-40.el9 baseos
|
||||
rpm-libs.aarch64 4.16.1.3-40.el9 baseos
|
||||
rpm-sign-libs.aarch64 4.16.1.3-40.el9 baseos
|
||||
sed.aarch64 4.8-10.el9_8 baseos
|
||||
setup.noarch 2.13.7-10.el9 baseos
|
||||
shadow-utils.aarch64 2:4.9-16.el9 baseos
|
||||
sqlite-libs.aarch64 3.34.1-11.el9_8 baseos
|
||||
systemd-libs.aarch64 252-67.el9_8.6.rocky.0.1 baseos
|
||||
tar.aarch64 2:1.34-13.el9_8 baseos
|
||||
tpm2-tss.aarch64 3.2.3-1.el9 baseos
|
||||
tzdata.noarch 2026c-1.el9_8 baseos
|
||||
usermode.aarch64 1.114-7.el9 baseos
|
||||
util-linux.aarch64 2.37.4-25.el9 baseos
|
||||
util-linux-core.aarch64 2.37.4-25.el9 baseos
|
||||
vim-minimal.aarch64 2:8.2.2637-26.el9_8.21 baseos
|
||||
yum.noarch 4.14.0-34.el9_8.rocky.0.1 baseos
|
||||
@@ -0,0 +1,54 @@
|
||||
|
||||
binutils.aarch64 2.35.2-72.el9 baseos
|
||||
binutils-gold.aarch64 2.35.2-72.el9 baseos
|
||||
bzip2-libs.aarch64 1.0.8-11.el9 baseos
|
||||
coreutils-single.aarch64 8.32-41.el9_8.1 baseos
|
||||
curl-minimal.aarch64 7.76.1-40.el9_8.5 baseos
|
||||
expat.aarch64 2.5.0-6.el9_8.3 baseos
|
||||
file-libs.aarch64 5.39-17.el9 baseos
|
||||
glib2.aarch64 2.68.4-19.el9_8.10 baseos
|
||||
glibc.aarch64 2.34-275.el9_8 baseos
|
||||
glibc-common.aarch64 2.34-275.el9_8 baseos
|
||||
glibc-minimal-langpack.aarch64 2.34-275.el9_8 baseos
|
||||
gnupg2.aarch64 2.3.3-5.el9_7 baseos
|
||||
gnutls.aarch64 3.8.10-8.el9_8 baseos
|
||||
gzip.aarch64 1.12-2.el9_8 baseos
|
||||
krb5-libs.aarch64 1.21.1-10.el9_8 baseos
|
||||
less.aarch64 590-6.el9 baseos
|
||||
libacl.aarch64 2.4.0-1.el9_8 baseos
|
||||
libarchive.aarch64 3.5.3-11.el9_8 baseos
|
||||
libatomic.aarch64 11.5.0-14.el9 baseos
|
||||
libattr.aarch64 2.6.0-1.el9_8 baseos
|
||||
libblkid.aarch64 2.37.4-25.el9 baseos
|
||||
libcap.aarch64 2.48-10.el9_7.1 baseos
|
||||
libcurl-minimal.aarch64 7.76.1-40.el9_8.5 baseos
|
||||
libevent.aarch64 2.1.13-1.el9_8 baseos
|
||||
libfdisk.aarch64 2.37.4-25.el9 baseos
|
||||
libgcc.aarch64 11.5.0-14.el9 baseos
|
||||
libgcrypt.aarch64 1.10.0-13.el9_8 baseos
|
||||
libgomp.aarch64 11.5.0-14.el9 baseos
|
||||
libmount.aarch64 2.37.4-25.el9 baseos
|
||||
libnghttp2.aarch64 1.43.0-6.el9_8.2 baseos
|
||||
libsmartcols.aarch64 2.37.4-25.el9 baseos
|
||||
libsolv.aarch64 0.7.24-6.el9_8 baseos
|
||||
libstdc++.aarch64 11.5.0-14.el9 baseos
|
||||
libtasn1.aarch64 4.16.0-10.el9_8 baseos
|
||||
libuuid.aarch64 2.37.4-25.el9 baseos
|
||||
libxml2.aarch64 2.9.13-14.el9_8.4 baseos
|
||||
ncurses-base.noarch 6.2-12.20210508.el9 baseos
|
||||
ncurses-libs.aarch64 6.2-12.20210508.el9 baseos
|
||||
openssl.aarch64 1:3.5.8-1.el9_8 baseos
|
||||
openssl-libs.aarch64 1:3.5.8-1.el9_8 baseos
|
||||
p11-kit.aarch64 0.26.4-1.el9_8 baseos
|
||||
p11-kit-trust.aarch64 0.26.4-1.el9_8 baseos
|
||||
pam.aarch64 1.5.1-28.el9_8.1 baseos
|
||||
python3.aarch64 3.9.25-7.el9_8.3 baseos
|
||||
python3-libs.aarch64 3.9.25-7.el9_8.3 baseos
|
||||
python3-setuptools-wheel.noarch 53.0.0-15.el9 baseos
|
||||
shadow-utils.aarch64 2:4.9-16.el9 baseos
|
||||
sqlite-libs.aarch64 3.34.1-11.el9_8 baseos
|
||||
systemd-libs.aarch64 252-67.el9_8.6.rocky.0.1 baseos
|
||||
tar.aarch64 2:1.34-13.el9_8 baseos
|
||||
util-linux.aarch64 2.37.4-25.el9 baseos
|
||||
util-linux-core.aarch64 2.37.4-25.el9 baseos
|
||||
vim-minimal.aarch64 2:8.2.2637-26.el9_8.21 baseos
|
||||
@@ -0,0 +1,20 @@
|
||||
dnf5.aarch64 5.2.18.0-3.fc42 updates
|
||||
dnf5-plugins.aarch64 5.2.18.0-3.fc42 updates
|
||||
elfutils-default-yama-scope.noarch 0.195-1.fc42 updates
|
||||
elfutils-libelf.aarch64 0.195-1.fc42 updates
|
||||
elfutils-libs.aarch64 0.195-1.fc42 updates
|
||||
fedora-release-common.noarch 42-31 updates
|
||||
fedora-release-container.noarch 42-31 updates
|
||||
fedora-release-identity-container.noarch 42-31 updates
|
||||
glibc.aarch64 2.41-18.fc42 updates
|
||||
glibc-common.aarch64 2.41-18.fc42 updates
|
||||
glibc-minimal-langpack.aarch64 2.41-18.fc42 updates
|
||||
krb5-libs.aarch64 1.21.3-7.fc42 updates
|
||||
libdnf5.aarch64 5.2.18.0-3.fc42 updates
|
||||
libdnf5-cli.aarch64 5.2.18.0-3.fc42 updates
|
||||
libsolv.aarch64 0.7.37-2.fc42 updates
|
||||
openssl-libs.aarch64 1:3.2.6-4.fc42 updates
|
||||
rpm-sequoia.aarch64 1.10.2-2.fc42 updates
|
||||
tzdata.noarch 2026b-1.fc42 updates
|
||||
vim-data.noarch 2:9.2.390-1.fc42 updates
|
||||
vim-minimal.aarch64 2:9.2.390-1.fc42 updates
|
||||
@@ -0,0 +1,5 @@
|
||||
krb5-libs.aarch64 1.21.3-7.fc42 updates
|
||||
openssl-libs.aarch64 1:3.2.6-4.fc42 updates
|
||||
rpm-sequoia.aarch64 1.10.2-2.fc42 updates
|
||||
vim-data.noarch 2:9.2.390-1.fc42 updates
|
||||
vim-minimal.aarch64 2:9.2.390-1.fc42 updates
|
||||
4
agent/test-data/package_updates/pacman_checkupdates.txt
Normal file
4
agent/test-data/package_updates/pacman_checkupdates.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
libpcap 1.10.7-1 -> 1.11.0-1
|
||||
libsecret 0.21.7-1 -> 0.21.8.2-1
|
||||
libtirpc 1.3.7-1 -> 1.3.8-1
|
||||
tzdata 2026c-1 -> 2026d-1
|
||||
@@ -0,0 +1,15 @@
|
||||
Warning: Repository 'Update repository of openSUSE Backports' metadata expired since 2025-03-02 19:18:12 UTC.
|
||||
Warning: Repository 'Main Update Repository' metadata expired since 2025-08-30 08:17:31 UTC.
|
||||
Warning: Repository 'Update Repository (Non-Oss)' metadata expired since 2025-04-10 11:03:28 UTC.
|
||||
|
||||
|
||||
|
||||
Repository | Name | Category | Severity | Interactive | Status | Summary
|
||||
-------------------------------------------------------------+-----------------------------+----------+-----------+-------------+--------+--------------------------------
|
||||
Update repository with updates from SUSE Linux Enterprise 15 | openSUSE-SLE-15.5-2024-3765 | security | moderate | --- | needed | Security update for openssl-1_1
|
||||
Update repository with updates from SUSE Linux Enterprise 15 | openSUSE-SLE-15.5-2024-3926 | security | moderate | --- | needed | Security update for curl
|
||||
Update repository with updates from SUSE Linux Enterprise 15 | openSUSE-SLE-15.5-2024-4078 | security | important | --- | needed | Security update for glib2
|
||||
Update repository with updates from SUSE Linux Enterprise 15 | openSUSE-SLE-15.5-2024-4359 | security | moderate | --- | needed | Security update for curl
|
||||
|
||||
4 patches needed (4 security patches)
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
Warning: Repository 'Update repository of openSUSE Backports' metadata expired since 2025-03-02 19:18:12 UTC.
|
||||
Warning: Repository 'Main Update Repository' metadata expired since 2025-08-30 08:17:31 UTC.
|
||||
Warning: Repository 'Update Repository (Non-Oss)' metadata expired since 2025-04-10 11:03:28 UTC.
|
||||
|
||||
|
||||
S | Repository | Name | Current Version | Available Version | Arch
|
||||
---+--------------------------------------------------------------+--------------------+------------------------------------------+------------------------------------------+--------
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | aaa_base | 84.87+git20180409.04c9dae-150300.10.20.1 | 84.87+git20180409.04c9dae-150300.10.23.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | bash | 4.4-150400.25.22 | 4.4-150400.27.3.2 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | bash-sh | 4.4-150400.25.22 | 4.4-150400.27.3.2 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | crypto-policies | 20210917.c9d86d1-150400.3.6.1 | 20210917.c9d86d1-150400.3.8.1 | noarch
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | curl | 8.0.1-150400.5.50.1 | 8.0.1-150400.5.59.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | glibc | 2.31-150300.86.3 | 2.31-150300.89.2 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libcom_err2 | 1.46.4-150400.3.6.2 | 1.46.4-150400.3.9.2 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libcurl4 | 8.0.1-150400.5.50.1 | 8.0.1-150400.5.59.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libgcc_s1 | 13.3.0+git8781-150000.1.12.1 | 14.2.0+git10526-150000.1.6.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libglib-2_0-0 | 2.70.5-150400.3.14.1 | 2.70.5-150400.3.17.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libopenssl1_1 | 1.1.1l-150500.17.34.1 | 1.1.1l-150500.17.37.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libopenssl1_1-hmac | 1.1.1l-150500.17.34.1 | 1.1.1l-150500.17.37.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libreadline7 | 7.0-150400.25.22 | 7.0-150400.27.3.2 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libsolv-tools | 0.7.30-150400.3.27.2 | 0.7.31-150500.6.5.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libsolv-tools-base | 0.7.30-150400.3.27.2 | 0.7.31-150500.6.5.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libstdc++6 | 13.3.0+git8781-150000.1.12.1 | 14.2.0+git10526-150000.1.6.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libudev1 | 249.17-150400.8.43.1 | 249.17-150400.8.46.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | libzypp | 17.35.8-150500.6.13.1 | 17.35.16-150500.6.31.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | login_defs | 4.8.1-150400.10.21.1 | 4.8.1-150400.10.24.1 | noarch
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | openssl-1_1 | 1.1.1l-150500.17.34.1 | 1.1.1l-150500.17.37.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | shadow | 4.8.1-150400.10.21.1 | 4.8.1-150400.10.24.1 | aarch64
|
||||
v | Update repository with updates from SUSE Linux Enterprise 15 | zypper | 1.14.76-150500.6.6.15 | 1.14.78-150500.6.14.1 | aarch64
|
||||
@@ -0,0 +1,3 @@
|
||||
Warning: Repository 'Update repository of openSUSE Backports' metadata expired since 2026-07-10 11:19:15 UTC.
|
||||
|
||||
|
||||
9
agent/test-data/zfs/zfs_list.txt
Normal file
9
agent/test-data/zfs/zfs_list.txt
Normal file
@@ -0,0 +1,9 @@
|
||||
tank 12000000000000 11999000000000 /tank
|
||||
tank/apps 1000000000000 11999000000000 /tank/apps
|
||||
tank/backup 2000000000000 11999000000000 /tank/backup
|
||||
tank/media 1000000000000 11999000000000 /tank/my media
|
||||
rpool 900000000000 300000000000 -
|
||||
rpool/ROOT 1000000000 300000000000 -
|
||||
rpool/ROOT/pve-1 890000000000 300000000000 /
|
||||
rpool/data 9000000000 300000000000 -
|
||||
rpool/data/subvol-100-disk-0 400000000000 300000000000 /subvol-100-disk-0
|
||||
2
agent/test-data/zfs/zpool_list.txt
Normal file
2
agent/test-data/zfs/zpool_list.txt
Normal file
@@ -0,0 +1,2 @@
|
||||
tank 23999000000000 12000000000000 11999000000000 ONLINE
|
||||
rpool 1200000000000 900000000000 300000000000 DEGRADED
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user