Compare commits

..

32 Commits

Author SHA1 Message Date
hank
218c8b65e5 New translations en.po (Vietnamese)
[ci skip]
2026-09-16 06:11:50 -04:00
hank
6914f01b92 New translations en.po (German)
[ci skip]
2026-09-09 21:46:10 -04:00
hank
b1b13c3102 New translations en.po (Arabic)
[ci skip]
2026-09-09 21:46:09 -04:00
hank
4f22bf0246 New translations en.po (Uzbek)
[ci skip]
2026-09-09 21:46:08 -04:00
hank
c8e4ff6936 New translations en.po (Chinese Traditional, Hong Kong)
[ci skip]
2026-09-09 21:46:06 -04:00
hank
009a47be93 New translations en.po (Croatian)
[ci skip]
2026-09-09 21:46:05 -04:00
hank
100512bcfe New translations en.po (Persian)
[ci skip]
2026-09-09 21:46:04 -04:00
hank
d3364d5658 New translations en.po (Indonesian)
[ci skip]
2026-09-09 21:46:02 -04:00
hank
c9e4aeefa5 New translations en.po (Vietnamese)
[ci skip]
2026-09-09 21:46:01 -04:00
hank
6cdc829acd New translations en.po (Chinese Traditional)
[ci skip]
2026-09-09 21:46:00 -04:00
hank
791a9ffa05 New translations en.po (Chinese Simplified)
[ci skip]
2026-09-09 21:45:58 -04:00
hank
d8a5a91413 New translations en.po (Ukrainian)
[ci skip]
2026-09-09 21:45:57 -04:00
hank
2fde2c204e New translations en.po (Turkish)
[ci skip]
2026-09-09 21:45:55 -04:00
hank
fad6e33604 New translations en.po (Swedish)
[ci skip]
2026-09-09 21:45:54 -04:00
hank
444ea9d8f7 New translations en.po (Serbian (Cyrillic))
[ci skip]
2026-09-09 21:45:53 -04:00
hank
4848335c31 New translations en.po (Slovenian)
[ci skip]
2026-09-09 21:45:51 -04:00
hank
bbaedb6a06 New translations en.po (Russian)
[ci skip]
2026-09-09 21:45:50 -04:00
hank
3a3cc6811a New translations en.po (Portuguese)
[ci skip]
2026-09-09 21:45:48 -04:00
hank
3a982dbbb7 New translations en.po (Polish)
[ci skip]
2026-09-09 21:45:47 -04:00
hank
7874378682 New translations en.po (Norwegian)
[ci skip]
2026-09-09 21:45:46 -04:00
hank
8bd484ea61 New translations en.po (Dutch)
[ci skip]
2026-09-09 21:45:44 -04:00
hank
643db14caa New translations en.po (Korean)
[ci skip]
2026-09-09 21:45:43 -04:00
hank
db12911578 New translations en.po (Japanese)
[ci skip]
2026-09-09 21:45:42 -04:00
hank
d7dcfbf49e New translations en.po (Italian)
[ci skip]
2026-09-09 21:45:40 -04:00
hank
f17a7320d4 New translations en.po (Hungarian)
[ci skip]
2026-09-09 21:45:39 -04:00
hank
815a207509 New translations en.po (Hebrew)
[ci skip]
2026-09-09 21:45:37 -04:00
hank
31dacaff38 New translations en.po (Greek)
[ci skip]
2026-09-09 21:45:36 -04:00
hank
fd4e03de08 New translations en.po (Danish)
[ci skip]
2026-09-09 21:45:35 -04:00
hank
6453869371 New translations en.po (Czech)
[ci skip]
2026-09-09 21:45:33 -04:00
hank
e0de0b3260 New translations en.po (Bulgarian)
[ci skip]
2026-09-09 21:45:32 -04:00
hank
73b9b86cb7 New translations en.po (Spanish)
[ci skip]
2026-09-09 21:45:30 -04:00
hank
8db92a72ef New translations en.po (French)
[ci skip]
2026-09-09 21:45:29 -04:00
180 changed files with 925 additions and 18458 deletions

View File

@@ -1,6 +1,6 @@
# Node.js dependencies
node_modules/
**/node_modules/
node_modules
internalsite/node_modules
# Go build artifacts and binaries
build

View File

@@ -29,7 +29,6 @@ jobs:
# henrygd/beszel-agent:alpine
- image: henrygd/beszel-agent
dockerfile: ./internal/dockerfile_agent_alpine
flavor: latest=false
registry: docker.io
username_secret: DOCKERHUB_USERNAME
password_secret: DOCKERHUB_TOKEN
@@ -56,7 +55,6 @@ jobs:
# henrygd/beszel-agent-nvidia:slim
- image: henrygd/beszel-agent-nvidia
dockerfile: ./internal/dockerfile_agent_nvidia_slim
flavor: latest=false
platforms: linux/amd64,linux/arm64
registry: docker.io
username_secret: DOCKERHUB_USERNAME
@@ -125,7 +123,6 @@ jobs:
# ghcr.io/henrygd/beszel-agent-nvidia:slim
- image: ghcr.io/${{ github.repository }}/beszel-agent-nvidia
dockerfile: ./internal/dockerfile_agent_nvidia_slim
flavor: latest=false
platforms: linux/amd64,linux/arm64
registry: ghcr.io
username: ${{ github.actor }}
@@ -153,7 +150,6 @@ jobs:
# ghcr.io/henrygd/beszel-agent:alpine
- image: ghcr.io/${{ github.repository }}/beszel-agent
dockerfile: ./internal/dockerfile_agent_alpine
flavor: latest=false
registry: ghcr.io
username: ${{ github.actor }}
password_secret: GITHUB_TOKEN
@@ -163,7 +159,7 @@ jobs:
type=semver,pattern={{major}}.{{minor}}-alpine
type=semver,pattern={{major}}-alpine
# henrygd/beszel-agent
# henrygd/beszel-agent (keep at bottom so it gets built after :alpine and gets the latest tag)
- image: henrygd/beszel-agent
dockerfile: ./internal/dockerfile_agent
registry: docker.io
@@ -204,8 +200,6 @@ jobs:
uses: docker/metadata-action@v6
with:
images: ${{ matrix.image }}
# Variant images must not overwrite the standard image's latest tag.
flavor: ${{ matrix.flavor || 'latest=auto' }}
tags: ${{ matrix.tags }}
# https://github.com/docker/login-action

View File

@@ -48,7 +48,6 @@ type Agent struct {
keys []gossh.PublicKey // SSH public keys
smartManager *SmartManager // Manages SMART data
systemdManager *systemdManager // Manages systemd services
monitorManager *MonitorManager // Manages network monitors
storagePoolManager *StoragePoolManager // Manages storage pool and dataset data
}
@@ -123,9 +122,6 @@ func NewAgent(dataDir ...string) (agent *Agent, err error) {
// initialize handler registry
agent.handlerRegistry = NewHandlerRegistry()
// initialize monitor manager
agent.monitorManager = newMonitorManager()
agent.storagePoolManager = newStoragePoolManager()
// Retain ZFS_INTERVAL for the shared storage pool detail refresh interval.
@@ -196,11 +192,6 @@ func (a *Agent) gatherStats(options common.DataRequestOptions) *system.CombinedD
}
}
if a.monitorManager != nil {
data.Monitors = a.monitorManager.GetResults(cacheTimeMs)
slog.Debug("Monitors", "data", data.Monitors)
}
// skip updating systemd services if cache time is not the default 60sec interval
if a.systemdManager != nil && cacheTimeMs == defaultDataCacheTimeMs {
totalCount := uint16(a.systemdManager.getServiceStatsCount())

View File

@@ -30,11 +30,6 @@ const (
wsDeadline = 120 * time.Second
)
// errNoHubURL is returned when HUB_URL is unset. This is not a failure
// condition: an agent configured with only a public key runs in SSH-only mode,
// where the hub dials the agent and no outbound WebSocket client is expected.
var errNoHubURL = errors.New("HUB_URL environment variable not set")
type caCertFileError struct {
err error
}
@@ -68,7 +63,7 @@ type WebSocketClient struct {
func newWebSocketClient(agent *Agent) (client *WebSocketClient, err error) {
hubURLStr, exists := utils.GetEnv("HUB_URL")
if !exists {
return nil, errNoHubURL
return nil, errors.New("HUB_URL environment variable not set")
}
client = &WebSocketClient{}

View File

@@ -32,28 +32,6 @@ import (
"golang.org/x/crypto/ssh"
)
// TestNewWebSocketClientNoHubURL verifies that an unset HUB_URL returns the
// errNoHubURL sentinel rather than an opaque error. Callers rely on this to
// distinguish SSH-only mode -- a supported configuration in which the hub dials
// the agent -- from an actual misconfiguration.
func TestNewWebSocketClientNoHubURL(t *testing.T) {
agent := createTestAgent(t)
// t.Setenv registers restoration of the original value; unset afterwards so
// GetEnv's LookupEnv reports the variable as absent rather than empty.
t.Setenv("BESZEL_AGENT_HUB_URL", "")
os.Unsetenv("BESZEL_AGENT_HUB_URL")
t.Setenv("HUB_URL", "")
os.Unsetenv("HUB_URL")
t.Setenv("BESZEL_AGENT_TOKEN", "test-token")
client, err := newWebSocketClient(agent)
require.Error(t, err)
assert.Nil(t, client)
assert.ErrorIs(t, err, errNoHubURL)
}
// TestNewWebSocketClient tests WebSocket client creation
func TestNewWebSocketClient(t *testing.T) {
agent := createTestAgent(t)

View File

@@ -91,15 +91,7 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
if errors.As(err, &caCertErr) {
return err
}
disableSSH, _ := utils.GetEnv("DISABLE_SSH")
if errors.Is(err, errNoHubURL) && disableSSH != "true" {
// SSH-only mode: the hub dials the agent, so there is nothing to warn
// about. With SSH also disabled there is no connection method at all,
// so that case still warns.
slog.Debug("WebSocket client not configured", "err", err)
} else {
slog.Warn("Error creating WebSocket client", "err", err)
}
slog.Warn("Error creating WebSocket client", "err", err)
}
c.wsClient = wsClient
@@ -153,7 +145,6 @@ func (c *ConnectionManager) Start(serverOptions ServerOptions) error {
// }
func (c *ConnectionManager) stop() error {
_ = c.agent.StopServer()
c.agent.monitorManager.Stop()
c.closeWebSocket()
return health.CleanUp()
}

View File

@@ -65,15 +65,10 @@ type dockerManager struct {
dockerVersionChecked bool // Whether a version probe has completed successfully
isWindows bool // Whether the Docker Engine API is running on Windows
buf *bytes.Buffer // Buffer to store and read response bodies
apiStats *container.ApiStats // Reusable API stats object
excludeContainers []string // Patterns to exclude containers by name
usingPodman bool // Whether the Docker Engine API is running on Podman
registryClient *http.Client // Client for registry requests; nil uses a client with a 10-second timeout
imageUpdatesDisabled bool // Whether image update checks are disabled by configuration
imageUpdatesMutex sync.RWMutex // Protects imageUpdates, its entries, and imageUpdatesRunning
imageUpdates map[string]*imageUpdateStatus // Shared update status keyed by normalized image reference
imageUpdatesRunning bool // Whether a background image-update batch is in progress
// Cache-time-aware tracking for CPU stats (similar to cpu.go)
// Maps cache time intervals to container-specific CPU usage tracking
lastCpuContainer map[uint16]map[string]uint64 // cacheTimeMs -> containerId -> last cpu container usage
@@ -166,9 +161,6 @@ func (dm *dockerManager) getDockerStats(cacheTimeMs uint16) ([]*container.Stats,
clear(dm.validIds)
}
// Only schedule auxiliary work here; metrics never wait for image discovery.
dm.refreshImageUpdates(dm.apiContainerList, time.Now())
var failedContainers []*container.ApiInfo
for _, ctr := range dm.apiContainerList {
@@ -514,17 +506,6 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
}
}
// Read and decode the response before locking shared stats to avoid blocking
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return fmt.Errorf("container stats request failed: %s", resp.Status)
}
res := &container.ApiStats{}
if err := json.NewDecoder(resp.Body).Decode(res); err != nil {
return err
}
updateAvailable := dm.cachedImageUpdate(ctr.Image)
dm.containerStatsMutex.Lock()
defer dm.containerStatsMutex.Unlock()
@@ -539,9 +520,6 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
stats.Status = statusText
stats.Health = health
stats.Image = ctr.Image
stats.UpdateAvailable = updateAvailable
if len(ctr.Ports) > 0 {
stats.Ports = convertContainerPortsToString(ctr)
}
@@ -554,6 +532,12 @@ func (dm *dockerManager) updateContainerStats(ctr *container.ApiInfo, cacheTimeM
stats.NetworkSent = 0
stats.NetworkRecv = 0
res := dm.apiStats
res.Networks = nil
if err := dm.decode(resp, res); err != nil {
return err
}
// Initialize CPU tracking for this cache time interval
dm.initializeCpuTracking(cacheTimeMs)
@@ -689,8 +673,6 @@ func newDockerManager(agent *Agent) *dockerManager {
userAgent: "Docker-Client/",
}
dockerImageCheck, _ := utils.GetEnv("DOCKER_IMAGE_CHECK")
// Read container exclusion patterns from environment variable
var excludeContainers []string
if excludeStr, set := utils.GetEnv("EXCLUDE_CONTAINERS"); set && excludeStr != "" {
@@ -710,11 +692,11 @@ func newDockerManager(agent *Agent) *dockerManager {
Timeout: timeout,
Transport: userAgentTransport,
},
containerStatsMap: make(map[string]*container.Stats),
sem: make(chan struct{}, 5),
apiContainerList: []*container.ApiInfo{},
excludeContainers: excludeContainers,
imageUpdatesDisabled: dockerImageCheck == "false",
containerStatsMap: make(map[string]*container.Stats),
sem: make(chan struct{}, 5),
apiContainerList: []*container.ApiInfo{},
apiStats: &container.ApiStats{},
excludeContainers: excludeContainers,
// Initialize cache-time-aware tracking structures
lastCpuContainer: make(map[uint16]map[string]uint64),

View File

@@ -1,108 +0,0 @@
package agent
import (
"log/slog"
"sync"
"time"
"github.com/distribution/reference"
"github.com/henrygd/beszel/internal/entities/container"
)
const imageUpdateInterval = time.Hour
type imageUpdateStatus struct {
available bool
checkedAt time.Time
}
func normalizedImageReference(image string) string {
named, err := reference.ParseNormalizedNamed(image)
if err != nil {
return ""
}
// Digest-pinned references cannot move to a new version.
if _, pinned := named.(reference.Digested); pinned {
return ""
}
return reference.TagNameOnly(named).String()
}
// refreshImageUpdates starts at most one background batch. Neither its network
// work nor its completion is part of the container metrics wait group.
func (dm *dockerManager) refreshImageUpdates(containers []*container.ApiInfo, now time.Time) {
if dm.imageUpdatesDisabled {
return
}
dm.imageUpdatesMutex.Lock()
defer dm.imageUpdatesMutex.Unlock()
if dm.imageUpdatesRunning {
return
}
if dm.imageUpdates == nil {
dm.imageUpdates = make(map[string]*imageUpdateStatus)
}
active := make(map[string]struct{}, len(containers))
pending := make(map[string]*imageUpdateStatus)
for _, ctr := range containers {
if len(ctr.Names) > 0 && dm.shouldExcludeContainer(ctr.Names[0][1:]) {
continue
}
key := normalizedImageReference(ctr.Image)
if key == "" {
continue
}
active[key] = struct{}{}
entry := dm.imageUpdates[key]
if entry == nil {
entry = &imageUpdateStatus{}
dm.imageUpdates[key] = entry
}
if entry.checkedAt.IsZero() || now.Sub(entry.checkedAt) >= imageUpdateInterval {
pending[key] = entry
}
}
for key := range dm.imageUpdates {
if _, ok := active[key]; !ok {
delete(dm.imageUpdates, key)
}
}
if len(pending) == 0 {
return
}
dm.imageUpdatesRunning = true
go func() {
// Limit auxiliary requests even on hosts running many different images.
sem := make(chan struct{}, 2)
var wg sync.WaitGroup
for key, entry := range pending {
sem <- struct{}{}
wg.Add(1)
go func() {
defer wg.Done()
defer func() { <-sem }()
available, err := dm.checkImageUpdate(key)
if err != nil {
available = false
slog.Debug("Image update check failed", "image", key, "err", err)
}
dm.imageUpdatesMutex.Lock()
entry.available = available
entry.checkedAt = time.Now()
dm.imageUpdatesMutex.Unlock()
}()
}
wg.Wait()
dm.imageUpdatesMutex.Lock()
dm.imageUpdatesRunning = false
dm.imageUpdatesMutex.Unlock()
}()
}
func (dm *dockerManager) cachedImageUpdate(image string) bool {
key := normalizedImageReference(image)
dm.imageUpdatesMutex.RLock()
defer dm.imageUpdatesMutex.RUnlock()
entry := dm.imageUpdates[key]
return entry != nil && entry.available
}

View File

@@ -1,248 +0,0 @@
//go:build testing
package agent
import (
"encoding/json"
"fmt"
"github.com/fxamacker/cbor/v2"
"io"
"net/http"
"net/http/httptest"
"strings"
"sync/atomic"
"testing"
"time"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/stretchr/testify/require"
)
func waitForImageUpdates(t *testing.T, dm *dockerManager) {
t.Helper()
require.Eventually(t, func() bool {
dm.imageUpdatesMutex.RLock()
defer dm.imageUpdatesMutex.RUnlock()
return !dm.imageUpdatesRunning
}, time.Second*3, time.Millisecond)
}
func TestDisableDockerImageUpdateCheck(t *testing.T) {
t.Setenv("BESZEL_AGENT_DOCKER_IMAGE_CHECK", "false")
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path == "/version" {
fmt.Fprint(w, `{"Version":"25.0.0"}`)
return
}
http.NotFound(w, r)
}))
defer server.Close()
t.Setenv("BESZEL_AGENT_DOCKER_HOST", server.URL)
dm := newDockerManager(nil)
require.True(t, dm.imageUpdatesDisabled)
dm.registryClient = &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
t.Fatal("disabled image update check made a registry request")
return nil, nil
})}
dm.refreshImageUpdates([]*container.ApiInfo{{Image: "nginx", Names: []string{"/nginx"}}}, time.Now())
require.False(t, dm.imageUpdatesRunning)
require.Nil(t, dm.imageUpdates)
}
func TestImageUpdateCacheAndStats(t *testing.T) {
local := "sha256:" + strings.Repeat("a", 64)
remote := "sha256:" + strings.Repeat("b", 64)
var inspections, lookups atomic.Int32
var fail atomic.Bool
var upToDate atomic.Bool
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
switch {
case strings.HasPrefix(r.URL.Path, "/images/"):
inspections.Add(1)
fmt.Fprintf(w, `{"RepoDigests":["docker.io/library/nginx@%s"]}`, local)
case r.URL.Path == "/containers/json":
fmt.Fprint(w, `[{"Id":"aaaaaaaaaaaa","Names":["/one"],"Image":"nginx","Status":"Up 2 hours"},{"Id":"bbbbbbbbbbbb","Names":["/two"],"Image":"docker.io/library/nginx:latest","Status":"Up 2 hours"}]`)
case strings.Contains(r.URL.Path, "/stats"):
fmt.Fprint(w, `{"memory_stats":{"usage":1048576},"cpu_stats":{},"networks":{}}`)
default:
http.NotFound(w, r)
}
}))
defer server.Close()
dm := newDockerManagerForVersionTest(server)
dm.dockerVersionChecked = true
dm.registryClient = &http.Client{Timeout: time.Second, Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
if fail.Load() {
return nil, fmt.Errorf("registry unavailable")
}
response := &http.Response{StatusCode: 200, Header: make(http.Header), Body: io.NopCloser(strings.NewReader(`{"token":"test"}`))}
if r.Method == http.MethodHead {
lookups.Add(1)
digest := remote
if upToDate.Load() {
digest = local
}
response.Header.Set("Docker-Content-Digest", digest)
}
return response, nil
})}
stats, err := dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
require.Len(t, stats, 2)
waitForImageUpdates(t, dm)
require.EqualValues(t, 1, lookups.Load())
require.EqualValues(t, 1, inspections.Load())
stats, err = dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
for _, stat := range stats {
require.True(t, stat.UpdateAvailable)
if stat.Id == "aaaaaaaaaaaa" {
require.Equal(t, "nginx", stat.Image)
} else {
require.Equal(t, "docker.io/library/nginx:latest", stat.Image)
}
}
require.EqualValues(t, 1, lookups.Load())
expire := func() {
dm.imageUpdatesMutex.Lock()
dm.imageUpdates["docker.io/library/nginx:latest"].checkedAt = time.Now().Add(-imageUpdateInterval)
dm.imageUpdatesMutex.Unlock()
}
upToDate.Store(true)
expire()
_, err = dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
waitForImageUpdates(t, dm)
require.EqualValues(t, 2, lookups.Load())
require.False(t, dm.cachedImageUpdate("nginx:latest"))
// An expired positive result is cleared on failure, and the failure itself
// is cached so realtime stats do not retry a broken registry every second.
dm.imageUpdatesMutex.Lock()
dm.imageUpdates["docker.io/library/nginx:latest"].available = true
dm.imageUpdatesMutex.Unlock()
fail.Store(true)
expire()
_, err = dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
waitForImageUpdates(t, dm)
failedInspections := inspections.Load()
stats, err = dm.getDockerStats(defaultCacheTimeMs)
require.NoError(t, err)
require.Len(t, stats, 2)
require.Equal(t, failedInspections, inspections.Load())
for _, stat := range stats {
require.False(t, stat.UpdateAvailable)
require.Equal(t, 1.0, stat.Mem)
}
}
func TestImageDiscoveryDoesNotBlockStats(t *testing.T) {
started := make(chan struct{}, 1)
release := make(chan struct{})
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if strings.HasPrefix(r.URL.Path, "/images/") {
fmt.Fprintf(w, `{"RepoDigests":["example.com/app@sha256:%s"]}`, strings.Repeat("a", 64))
} else {
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
}
}))
defer server.Close()
dm := newDockerManagerForVersionTest(server)
defer func() { close(release); waitForImageUpdates(t, dm) }()
dm.registryClient = &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) {
started <- struct{}{}
<-release
return nil, fmt.Errorf("timeout")
})}
ctr := &container.ApiInfo{IdShort: "aaaaaaaaaaaa", Image: "example.com/app", Names: []string{"/one"}}
dm.refreshImageUpdates([]*container.ApiInfo{ctr}, time.Now())
select {
case <-started:
case <-time.After(3 * time.Second):
t.Fatal("check did not start")
}
done := make(chan error, 1)
go func() { done <- dm.updateContainerStats(ctr, defaultCacheTimeMs) }()
select {
case err := <-done:
require.NoError(t, err)
case <-time.After(time.Second):
t.Fatal("registry blocked stats")
}
dm.imageUpdatesMutex.RLock()
require.True(t, dm.imageUpdatesRunning)
dm.imageUpdatesMutex.RUnlock()
}
func TestNormalizeImageUpdateReferences(t *testing.T) {
require.Equal(t, normalizedImageReference("nginx"), normalizedImageReference("docker.io/library/nginx:latest"))
require.Empty(t, normalizedImageReference("bad reference"))
require.Empty(t, normalizedImageReference("nginx@sha256:"+strings.Repeat("a", 64)))
}
// A stats request can return headers promptly and then stall while reading its
// body. The stats-map mutex must remain available during that read.
func TestStatsResponseBodyDoesNotHoldStatsLock(t *testing.T) {
started := make(chan struct{})
release := make(chan struct{})
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusOK)
w.(http.Flusher).Flush()
close(started)
<-release
fmt.Fprint(w, `{"memory_stats":{"usage":1048576}}`)
}))
defer server.Close()
dm := newDockerManagerForVersionTest(server)
done := make(chan error, 1)
go func() {
done <- dm.updateContainerStats(&container.ApiInfo{IdShort: "aaaaaaaaaaaa", Names: []string{"/one"}, Image: "nginx"}, defaultCacheTimeMs)
}()
<-started
locked := make(chan struct{})
go func() { dm.containerStatsMutex.Lock(); dm.containerStatsMutex.Unlock(); close(locked) }()
select {
case <-locked:
case <-time.After(time.Second):
close(release)
<-done
t.Fatal("Docker response body held the stats mutex")
}
close(release)
require.NoError(t, <-done)
}
func TestImageUpdateStatsEncoding(t *testing.T) {
original := container.Stats{Image: "nginx:latest", UpdateAvailable: true}
encoded, err := cbor.Marshal(original)
require.NoError(t, err)
var fields map[int]any
require.NoError(t, cbor.Unmarshal(encoded, &fields))
require.Equal(t, true, fields[11])
require.Equal(t, "nginx:latest", fields[8])
var decoded container.Stats
require.NoError(t, cbor.Unmarshal(encoded, &decoded))
require.True(t, decoded.UpdateAvailable)
require.Equal(t, original.Image, decoded.Image)
encoded, err = json.Marshal(original)
require.NoError(t, err)
require.Contains(t, string(encoded), `"u":true`)
}
func TestImageUpdateCacheExpiryBoundaryAndPruning(t *testing.T) {
now := time.Now()
key := normalizedImageReference("nginx")
dm := &dockerManager{imageUpdates: map[string]*imageUpdateStatus{
key: {available: true, checkedAt: now},
"unused.example/image:latest": {checkedAt: now},
}}
dm.refreshImageUpdates([]*container.ApiInfo{{Image: "nginx"}}, now.Add(imageUpdateInterval-time.Nanosecond))
require.False(t, dm.imageUpdatesRunning)
require.Len(t, dm.imageUpdates, 1)
require.True(t, dm.cachedImageUpdate("nginx:latest"))
dm.refreshImageUpdates(nil, now)
require.Empty(t, dm.imageUpdates)
}

View File

@@ -1,222 +0,0 @@
package agent
import (
_ "crypto/sha256"
"encoding/json"
"fmt"
"net/http"
"net/url"
"strings"
"time"
"github.com/distribution/reference"
"github.com/opencontainers/go-digest"
)
const imageRegistryTimeout = 10 * time.Second
const imageManifestAccept = "application/vnd.docker.distribution.manifest.list.v2+json, " +
"application/vnd.docker.distribution.manifest.v2+json, " +
"application/vnd.oci.image.manifest.v1+json, " +
"application/vnd.oci.image.index.v1+json"
// checkImageUpdate compares the digest recorded by Docker for image with the
// digest currently advertised by its registry. A digest-pinned reference is
// immutable and therefore never has an update available.
func (dm *dockerManager) checkImageUpdate(image string) (bool, error) {
named, err := reference.ParseNormalizedNamed(image)
if err != nil {
return false, fmt.Errorf("parse image reference %q: %w", image, err)
}
if _, pinned := named.(reference.Digested); pinned {
return false, nil
}
named = reference.TagNameOnly(named)
registry := reference.Domain(named)
repository := reference.Path(named)
tag := named.(reference.Tagged).Tag()
localDigest, err := dm.inspectImageDigest(image, registry, repository)
if err != nil {
return false, err
}
remoteDigest, err := dm.registryImageDigest(registry, repository, tag)
if err != nil {
return false, err
}
return remoteDigest != localDigest, nil
}
// inspectImageDigest reads Docker's image metadata without using dm.decode.
// The checker runs in the image-discovery goroutine, so it must not hold any
// of the container statistics locks while waiting on the Docker API.
func (dm *dockerManager) inspectImageDigest(image, registry, repository string) (string, error) {
if dm.client == nil {
return "", fmt.Errorf("inspect image %q: Docker client is unavailable", image)
}
endpoint := "http://localhost/images/" + url.PathEscape(image) + "/json"
resp, err := dm.client.Get(endpoint)
if err != nil {
return "", fmt.Errorf("inspect image %q: %w", image, err)
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return "", fmt.Errorf("inspect image %q failed: %s", image, responseStatus(resp))
}
var inspect struct {
RepoDigests []string `json:"RepoDigests"`
}
if err := json.NewDecoder(resp.Body).Decode(&inspect); err != nil {
return "", fmt.Errorf("decode image inspect %q: %w", image, err)
}
if len(inspect.RepoDigests) == 0 {
return "", fmt.Errorf("inspect image %q returned no repository digests", image)
}
localDigest, ok := matchingRepositoryDigest(inspect.RepoDigests, registry, repository)
if !ok {
return "", fmt.Errorf("inspect image %q returned no valid digest for %s/%s", image, registry, repository)
}
return localDigest, nil
}
// matchingRepositoryDigest returns a valid digest belonging to the requested
// repository. Docker can return multiple RepoDigests for one local image; an
// unrelated first entry must never be used for the comparison.
func matchingRepositoryDigest(repoDigests []string, registry, repository string) (string, bool) {
for _, repoDigest := range repoDigests {
repoDigest = strings.TrimSpace(repoDigest)
at := strings.LastIndexByte(repoDigest, '@')
if at <= 0 || at == len(repoDigest)-1 || strings.Contains(repoDigest[:at], "@") {
continue
}
repoRef, err := reference.ParseNormalizedNamed(repoDigest[:at])
if err != nil || reference.Path(repoRef) != repository || !sameRegistry(reference.Domain(repoRef), registry) {
continue
}
if _, hasTag := repoRef.(reference.Tagged); hasTag {
continue
}
d, err := digest.Parse(repoDigest[at+1:])
if err != nil {
continue
}
return d.String(), true
}
return "", false
}
func sameRegistry(left, right string) bool {
left = canonicalRegistry(left)
right = canonicalRegistry(right)
return left == right ||
(left == "ghcr.io" && right == "lscr.io") ||
(left == "lscr.io" && right == "ghcr.io")
}
func canonicalRegistry(registry string) string {
if registry == "index.docker.io" {
return "docker.io"
}
return registry
}
func (dm *dockerManager) registryImageDigest(registry, repository, tag string) (string, error) {
client := dm.registryClient
if client == nil {
client = &http.Client{Timeout: imageRegistryTimeout}
}
token, err := dm.registryToken(client, registry, repository)
if err != nil {
return "", err
}
host := registry
if registry == "docker.io" {
host = "registry-1.docker.io"
}
manifestURL := "https://" + host + "/v2/" + repository + "/manifests/" + url.PathEscape(tag)
req, err := http.NewRequest(http.MethodHead, manifestURL, nil)
if err != nil {
return "", fmt.Errorf("create manifest request: %w", err)
}
req.Header.Set("Accept", imageManifestAccept)
if token != "" {
req.Header.Set("Authorization", "Bearer "+token)
}
resp, err := client.Do(req)
if err != nil {
return "", fmt.Errorf("fetch manifest %s:%s: %w", registry, repository, err)
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return "", fmt.Errorf("manifest request for %s:%s failed: %s", repository, tag, responseStatus(resp))
}
remote := strings.TrimSpace(resp.Header.Get("Docker-Content-Digest"))
d, err := digest.Parse(remote)
if err != nil {
return "", fmt.Errorf("manifest request for %s:%s returned invalid digest: %w", repository, tag, err)
}
return d.String(), nil
}
func (dm *dockerManager) registryToken(client *http.Client, registry, repository string) (string, error) {
var authURL string
switch registry {
case "docker.io":
authURL = "https://auth.docker.io/token?service=registry.docker.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
case "ghcr.io", "lscr.io":
// lscr.io is the LinuxServer alias for its GHCR-backed images.
authURL = "https://ghcr.io/token?service=ghcr.io&scope=" + url.QueryEscape("repository:"+repository+":pull")
default:
// Anonymous registries remain supported, as they were before the
// authenticated Docker Hub and GHCR paths were added.
return "", nil
}
req, err := http.NewRequest(http.MethodGet, authURL, nil)
if err != nil {
return "", fmt.Errorf("create registry auth request: %w", err)
}
resp, err := client.Do(req)
if err != nil {
return "", fmt.Errorf("fetch registry auth token for %s: %w", repository, err)
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return "", fmt.Errorf("registry auth request for %s failed: %s", repository, responseStatus(resp))
}
var tokenResponse struct {
Token string `json:"token"`
AccessToken string `json:"access_token"`
}
if err := json.NewDecoder(resp.Body).Decode(&tokenResponse); err != nil {
return "", fmt.Errorf("decode registry auth response for %s: %w", repository, err)
}
token := strings.TrimSpace(tokenResponse.Token)
if token == "" {
token = strings.TrimSpace(tokenResponse.AccessToken)
}
if token == "" {
return "", fmt.Errorf("registry auth response for %s contained no token", repository)
}
return token, nil
}
func responseStatus(resp *http.Response) string {
if resp.Status != "" {
return resp.Status
}
return http.StatusText(resp.StatusCode)
}

View File

@@ -1,204 +0,0 @@
//go:build testing
package agent
import (
"fmt"
"io"
"net/http"
"net/http/httptest"
"strings"
"sync/atomic"
"testing"
"github.com/stretchr/testify/require"
)
type registryTransportFunc func(*http.Request) (*http.Response, error)
func (fn registryTransportFunc) RoundTrip(req *http.Request) (*http.Response, error) {
return fn(req)
}
func registryResponse(status int, body string) *http.Response {
return &http.Response{
StatusCode: status,
Status: fmt.Sprintf("%d %s", status, http.StatusText(status)),
Header: make(http.Header),
Body: io.NopCloser(strings.NewReader(body)),
}
}
func registryDigest(fill byte) string {
return "sha256:" + strings.Repeat(string(fill), 64)
}
func newRegistryChecker(t *testing.T, inspectBody string, transport http.RoundTripper) *dockerManager {
t.Helper()
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if strings.HasPrefix(r.URL.Path, "/images/") {
w.Header().Set("Content-Type", "application/json")
_, _ = io.WriteString(w, inspectBody)
return
}
http.NotFound(w, r)
}))
t.Cleanup(server.Close)
return &dockerManager{
client: newDockerManagerForVersionTest(server).client,
registryClient: &http.Client{Transport: transport},
}
}
func TestCheckImageUpdateUsesInspectAndManifestDigests(t *testing.T) {
local := registryDigest('a')
remote := registryDigest('b')
var authCalls, manifestCalls atomic.Int32
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
switch {
case req.Method == http.MethodGet && req.URL.Host == "auth.docker.io":
authCalls.Add(1)
require.Equal(t, "/token", req.URL.Path)
return registryResponse(http.StatusOK, `{"token":"test-token"}`), nil
case req.Method == http.MethodHead && req.URL.Host == "registry-1.docker.io":
manifestCalls.Add(1)
require.Equal(t, "/v2/library/alpine/manifests/latest", req.URL.Path)
require.Equal(t, "Bearer test-token", req.Header.Get("Authorization"))
resp := registryResponse(http.StatusOK, "")
resp.Header.Set("Docker-Content-Digest", remote)
return resp, nil
default:
return registryResponse(http.StatusNotFound, ""), nil
}
}))
available, err := dm.checkImageUpdate("alpine")
require.NoError(t, err)
require.True(t, available)
require.EqualValues(t, 1, authCalls.Load())
require.EqualValues(t, 1, manifestCalls.Load())
}
func TestCheckImageUpdateReportsUnknownInspectState(t *testing.T) {
for _, test := range []struct {
name string
body string
}{
{name: "missing field", body: `{}`},
{name: "empty field", body: `{"RepoDigests":[]}`},
{name: "malformed reference", body: `{"RepoDigests":["not-a-repo-digest"]}`},
{name: "wrong repository", body: `{"RepoDigests":["docker.io/library/busybox@` + registryDigest('a') + `"]}`},
{name: "malformed digest", body: `{"RepoDigests":["docker.io/library/alpine@sha256:not-a-digest"]}`},
} {
t.Run(test.name, func(t *testing.T) {
var registryCalls atomic.Int32
dm := newRegistryChecker(t, test.body, registryTransportFunc(func(req *http.Request) (*http.Response, error) {
registryCalls.Add(1)
return registryResponse(http.StatusOK, `{"token":"unexpected"}`), nil
}))
available, err := dm.checkImageUpdate("alpine")
require.Error(t, err)
require.False(t, available)
require.EqualValues(t, 0, registryCalls.Load(), "invalid local state must not query a registry")
})
}
}
func TestCheckImageUpdateChecksInspectAuthAndManifestStatuses(t *testing.T) {
local := registryDigest('a')
validInspect := fmt.Sprintf(`{"RepoDigests":["docker.io/library/alpine@%s"]}`, local)
tests := []struct {
name string
inspectCode int
authCode int
manifestCode int
remote string
want string
}{
{name: "inspect status", inspectCode: http.StatusNotFound, want: "inspect image"},
{name: "auth status", inspectCode: http.StatusOK, authCode: http.StatusUnauthorized, want: "registry auth"},
{name: "manifest status", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusNotFound, remote: local, want: "manifest request"},
{name: "missing digest", inspectCode: http.StatusOK, authCode: http.StatusOK, manifestCode: http.StatusOK, want: "invalid digest"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if test.inspectCode != http.StatusOK && strings.HasPrefix(r.URL.Path, "/images/") {
w.WriteHeader(test.inspectCode)
return
}
_, _ = io.WriteString(w, validInspect)
}))
t.Cleanup(server.Close)
calls := 0
dm := &dockerManager{client: newDockerManagerForVersionTest(server).client, registryClient: &http.Client{Transport: registryTransportFunc(func(req *http.Request) (*http.Response, error) {
calls++
if req.Method == http.MethodGet {
return registryResponse(test.authCode, `{"token":"test"}`), nil
}
response := registryResponse(test.manifestCode, "")
response.Header.Set("Docker-Content-Digest", test.remote)
return response, nil
})}}
_, err := dm.checkImageUpdate("alpine")
require.Error(t, err)
require.Contains(t, err.Error(), test.want)
if test.inspectCode != http.StatusOK {
require.Zero(t, calls)
}
})
}
}
func TestCheckImageUpdateSupportsAnonymousAndLSCRRegistries(t *testing.T) {
t.Run("anonymous registry", func(t *testing.T) {
local := registryDigest('a')
var calls atomic.Int32
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["example.com/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
calls.Add(1)
require.Equal(t, http.MethodHead, req.Method)
require.Equal(t, "example.com", req.URL.Host)
resp := registryResponse(http.StatusOK, "")
resp.Header.Set("Docker-Content-Digest", local)
return resp, nil
}))
available, err := dm.checkImageUpdate("example.com/app")
require.NoError(t, err)
require.False(t, available)
require.EqualValues(t, 1, calls.Load())
})
t.Run("lscr ghcr alias", func(t *testing.T) {
local := registryDigest('a')
var authCalls, manifestCalls atomic.Int32
dm := newRegistryChecker(t, fmt.Sprintf(`{"RepoDigests":["ghcr.io/linuxserver/app@%s"]}`, local), registryTransportFunc(func(req *http.Request) (*http.Response, error) {
if req.Method == http.MethodGet {
authCalls.Add(1)
return registryResponse(http.StatusOK, `{"token":"test"}`), nil
}
manifestCalls.Add(1)
require.Equal(t, "lscr.io", req.URL.Host)
resp := registryResponse(http.StatusOK, "")
resp.Header.Set("Docker-Content-Digest", local)
return resp, nil
}))
available, err := dm.checkImageUpdate("lscr.io/linuxserver/app")
require.NoError(t, err)
require.False(t, available)
require.EqualValues(t, 1, authCalls.Load())
require.EqualValues(t, 1, manifestCalls.Load())
})
}
func TestCheckImageUpdateSkipsPinnedDigest(t *testing.T) {
image := "docker.io/library/alpine@" + registryDigest('a')
dm := &dockerManager{}
available, err := dm.checkImageUpdate(image)
require.NoError(t, err)
require.False(t, available)
}

View File

@@ -1184,6 +1184,7 @@ func TestUpdateContainerStatsPodmanCpuCalculation(t *testing.T) {
}
})},
containerStatsMap: make(map[string]*container.Stats),
apiStats: &container.ApiStats{},
usingPodman: true,
lastCpuContainer: map[uint16]map[string]uint64{
defaultCacheTimeMs: {"0123456789ab": prevCpuUsage},
@@ -1675,6 +1676,7 @@ func TestUpdateContainerStatsUsesPodmanInspectHealthFallback(t *testing.T) {
}
})},
containerStatsMap: make(map[string]*container.Stats),
apiStats: &container.ApiStats{},
usingPodman: true,
lastCpuContainer: make(map[uint16]map[string]uint64),
lastCpuSystem: make(map[uint16]map[string]uint64),

View File

@@ -1119,6 +1119,7 @@ func TestCalculateGPUAverage(t *testing.T) {
}
func TestGPUCapabilitiesAndLegacyPriority(t *testing.T) {
// Save original PATH
hasAmdSysfs := (&GPUManager{}).hasAmdSysfs()
tests := []struct {
@@ -1212,7 +1213,7 @@ echo "[]"`
{
name: "no gpu tools available",
setupCommands: func(_ string) error {
// The subtest already restricts PATH to its empty temporary directory.
t.Setenv("PATH", "")
return nil
},
wantErr: true,

View File

@@ -7,7 +7,6 @@ import (
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/smart"
"log/slog"
@@ -52,7 +51,6 @@ func NewHandlerRegistry() *HandlerRegistry {
registry.Register(common.GetContainerInfo, &GetContainerInfoHandler{})
registry.Register(common.GetSmartData, &GetSmartDataHandler{})
registry.Register(common.GetSystemdInfo, &GetSystemdInfoHandler{})
registry.Register(common.SyncNetworkMonitors, &SyncNetworkMonitorsHandler{})
registry.Register(common.GetZfsData, &GetZfsDataHandler{})
return registry
@@ -225,21 +223,3 @@ func (h *GetSystemdInfoHandler) Handle(hctx *HandlerContext) error {
return hctx.SendResponse(details, hctx.RequestID)
}
////////////////////////////////////////////////////////////////////////////
////////////////////////////////////////////////////////////////////////////
// SyncNetworkMonitorsHandler handles monitor configuration sync from hub
type SyncNetworkMonitorsHandler struct{}
func (h *SyncNetworkMonitorsHandler) Handle(hctx *HandlerContext) error {
var req monitor.SyncRequest
if err := cbor.Unmarshal(hctx.Request.Data, &req); err != nil {
return err
}
resp, err := hctx.Agent.monitorManager.HandleSyncRequest(req)
if err != nil {
return err
}
return hctx.SendResponse(resp, hctx.RequestID)
}

View File

@@ -201,9 +201,12 @@ func mdraidSmartStatus(health mdraidHealth) string {
if health.mismatchCnt > 0 {
return "WARNING"
}
// "check" and "repair" are requested consistency scans, not evidence of
// array failure. With no health issues above, keep scrubbing green while
// reporting the sync action and progress attributes.
// "check" scans for consistency problems without repairing mismatches.
// With no mismatches, keep it green while reporting progress attributes.
switch syncAction {
case "repair":
return "WARNING"
}
switch state {
case "clean", "active", "active-idle", "write-pending", "read-auto", "readonly":
return "PASSED"

View File

@@ -174,25 +174,8 @@ func TestMdraidSmartStatus(t *testing.T) {
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", mismatchCnt: 1}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(clean+mismatch) = %q, want WARNING", got)
}
for _, tc := range []struct {
name string
health mdraidHealth
want string
}{
{"clean", mdraidHealth{arrayState: "clean"}, "PASSED"},
{"active", mdraidHealth{arrayState: "active"}, "PASSED"},
{"mismatch", mdraidHealth{arrayState: "active", mismatchCnt: 1}, "WARNING"},
{"degraded", mdraidHealth{arrayState: "active", degraded: 1}, "FAILED"},
{"faulty member", mdraidHealth{arrayState: "active", faultyDisks: 1}, "FAILED"},
{"inactive", mdraidHealth{arrayState: "inactive"}, "FAILED"},
{"unknown", mdraidHealth{arrayState: "unknown"}, "UNKNOWN"},
} {
t.Run("repair/"+tc.name, func(t *testing.T) {
tc.health.syncAction = "repair"
if got := mdraidSmartStatus(tc.health); got != tc.want {
t.Fatalf("mdraidSmartStatus(%+v) = %q, want %s", tc.health, got, tc.want)
}
})
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean", syncAction: "repair"}); got != "WARNING" {
t.Fatalf("mdraidSmartStatus(repair) = %q, want WARNING", got)
}
if got := mdraidSmartStatus(mdraidHealth{arrayState: "clean"}); got != "PASSED" {
t.Fatalf("mdraidSmartStatus(clean) = %q, want PASSED", got)

View File

@@ -1,176 +0,0 @@
package agent
import (
"errors"
"fmt"
"net/http"
"sync"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
)
// MonitorManager manages network monitor configurations and task lifetimes.
type MonitorManager struct {
mu sync.RWMutex
monitors map[string]*monitorTask // keyed by monitor ID
probe monitorProbe
resumeGuard monitorResumeGuard
}
func newMonitorManager() *MonitorManager {
return newMonitorManagerWithProbe(networkMonitorProbe(&http.Client{Timeout: monitor.MaxProbeTimeout}))
}
func newMonitorManagerWithProbe(probe monitorProbe) *MonitorManager {
return &MonitorManager{monitors: make(map[string]*monitorTask), probe: probe}
}
// SyncMonitors replaces all monitor tasks with the given configs.
func (pm *MonitorManager) SyncMonitors(configs []monitor.Config) {
pm.mu.Lock()
defer pm.mu.Unlock()
// Build set of new keys
newKeys := make(map[string]monitor.Config, len(configs))
for _, cfg := range configs {
if cfg.ID == "" {
continue
}
newKeys[cfg.ID] = cfg
}
// Stop removed monitors
for key, task := range pm.monitors {
if _, exists := newKeys[key]; !exists {
task.cancel()
delete(pm.monitors, key)
}
}
// Start new monitors and restart tasks whose config changed.
for key, cfg := range newKeys {
task, exists := pm.monitors[key]
if exists && task.config == cfg {
continue
}
if exists {
task.cancel()
}
task = newMonitorTaskFromExisting(cfg, task)
task.resumeGuard = &pm.resumeGuard
pm.resumeGuard.start()
pm.monitors[key] = task
pm.startMonitor(task)
}
if len(pm.monitors) == 0 {
pm.resumeGuard.shutdown()
}
}
// HandleSyncRequest applies a full or incremental monitor sync request.
func (pm *MonitorManager) HandleSyncRequest(req monitor.SyncRequest) (monitor.SyncResponse, error) {
switch req.Action {
case monitor.SyncActionReplace:
pm.SyncMonitors(req.Configs)
return monitor.SyncResponse{}, nil
case monitor.SyncActionUpsert:
result, err := pm.UpsertMonitor(req.Config, req.RunNow)
if err != nil {
return monitor.SyncResponse{}, err
}
if result == nil {
return monitor.SyncResponse{}, nil
}
return monitor.SyncResponse{Result: *result}, nil
case monitor.SyncActionDelete:
if req.Config.ID == "" {
return monitor.SyncResponse{}, errors.New("missing monitor ID for delete")
}
pm.DeleteMonitor(req.Config.ID)
return monitor.SyncResponse{}, nil
default:
return monitor.SyncResponse{}, fmt.Errorf("unknown monitor sync action: %d", req.Action)
}
}
// UpsertMonitor creates or replaces a single monitor task.
func (pm *MonitorManager) UpsertMonitor(config monitor.Config, runNow bool) (*monitor.Result, error) {
if config.ID == "" {
return nil, errors.New("missing monitor ID")
}
pm.mu.Lock()
task, exists := pm.monitors[config.ID]
if exists && task.config == config {
pm.mu.Unlock()
if !runNow {
return nil, nil
}
return task.runProbe(pm.probe), nil
}
if exists {
task.cancel()
}
task = newMonitorTaskFromExisting(config, task)
task.resumeGuard = &pm.resumeGuard
pm.resumeGuard.start()
pm.monitors[config.ID] = task
pm.mu.Unlock()
if runNow {
result := task.runProbe(pm.probe)
pm.startMonitor(task)
return result, nil
}
pm.startMonitor(task)
return nil, nil
}
// DeleteMonitor stops and removes a single monitor task.
func (pm *MonitorManager) DeleteMonitor(id string) {
if id == "" {
return
}
pm.mu.Lock()
defer pm.mu.Unlock()
if task, exists := pm.monitors[id]; exists {
task.cancel()
delete(pm.monitors, id)
}
if len(pm.monitors) == 0 {
pm.resumeGuard.shutdown()
}
}
// GetResults returns aggregated results for all monitors over the last supplied duration in ms.
func (pm *MonitorManager) GetResults(durationMs uint16) map[string]monitor.Result {
pm.mu.RLock()
defer pm.mu.RUnlock()
results := make(map[string]monitor.Result, len(pm.monitors))
now := time.Now()
duration := time.Duration(durationMs) * time.Millisecond
for _, task := range pm.monitors {
result, ok := task.history.result(duration, now)
if !ok {
continue
}
results[task.config.ID] = result
}
return results
}
// Stop stops all monitor tasks.
func (pm *MonitorManager) Stop() {
pm.mu.Lock()
defer pm.mu.Unlock()
for key, task := range pm.monitors {
task.cancel()
delete(pm.monitors, key)
}
pm.resumeGuard.shutdown()
}

View File

@@ -1,274 +0,0 @@
package agent
import (
"math"
"sync"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
)
// Monitors run at user-defined intervals (e.g., every 10s).
// To keep memory usage low and constant, data is stored in two layers:
// 1. Raw samples: The most recent individual results (kept for monitorRawRetention).
// 2. Minute buckets: A ring buffer of 61 buckets, each representing one
// wall-clock minute. Samples collected within the same minute are aggregated
// (sum, min, max, count) into a single bucket.
//
// Short-term requests (<= 61s) use raw samples.
// Long-term requests (up to 1h) use the minute buckets to avoid storing thousands
// of individual data points.
const (
// monitorRawRetention is the duration to keep individual samples
monitorRawRetention = 61 * time.Second
// monitorMinuteBucketLen is the number of 1-minute buckets to keep (1 hour + 1 for partials)
monitorMinuteBucketLen int32 = 61
)
// monitorHistory owns retention and aggregation, independently of probe execution.
type monitorHistory struct {
mu sync.Mutex
sampleCount int64
samples []monitorSample
buckets [monitorMinuteBucketLen]monitorBucket
}
func newMonitorHistory() *monitorHistory {
// Start small for typical intervals; append grows the buffer for faster probes.
return &monitorHistory{samples: make([]monitorSample, 0, 4)}
}
func (h *monitorHistory) clone() *monitorHistory {
h.mu.Lock()
defer h.mu.Unlock()
cloned := newMonitorHistory()
cloned.samples = append(cloned.samples, h.samples...)
cloned.buckets = h.buckets
cloned.sampleCount = h.sampleCount
return cloned
}
func (h *monitorHistory) result(duration time.Duration, now time.Time) (monitor.Result, bool) {
h.mu.Lock()
defer h.mu.Unlock()
return h.resultLocked(duration, now)
}
func (h *monitorHistory) record(sample monitorSample) monitor.Result {
h.mu.Lock()
defer h.mu.Unlock()
h.addSampleLocked(sample)
result, _ := h.resultLocked(time.Minute, sample.timestamp)
return result
}
// monitorSample stores one monitor attempt and its collection time.
type monitorSample struct {
responseUs int64 // -1 means loss
timestamp time.Time
}
// monitorBucket stores one minute of aggregated monitor data.
type monitorBucket struct {
minute int32
filled bool
stats monitorAggregate
}
// monitorAggregate accumulates successful response stats and total sample counts.
type monitorAggregate struct {
sumUs int64
minUs int64
maxUs int64
totalCount int64
successCount int64
}
// newMonitorAggregate initializes an aggregate with an unset minimum value.
func newMonitorAggregate() monitorAggregate {
return monitorAggregate{minUs: math.MaxInt64}
}
// addResponse folds a single monitor sample into the aggregate.
func (agg *monitorAggregate) addResponse(responseUs int64) {
agg.totalCount++
if responseUs < 0 {
return
}
agg.successCount++
agg.sumUs += responseUs
if responseUs < agg.minUs {
agg.minUs = responseUs
}
if responseUs > agg.maxUs {
agg.maxUs = responseUs
}
}
// addAggregate merges another aggregate into this one.
func (agg *monitorAggregate) addAggregate(other monitorAggregate) {
if other.totalCount == 0 {
return
}
agg.totalCount += other.totalCount
agg.successCount += other.successCount
agg.sumUs += other.sumUs
if other.successCount == 0 {
return
}
if agg.minUs == math.MaxInt64 || other.minUs < agg.minUs {
agg.minUs = other.minUs
}
if other.maxUs > agg.maxUs {
agg.maxUs = other.maxUs
}
}
// hasData reports whether the aggregate contains any samples.
func (agg monitorAggregate) hasData() bool {
return agg.totalCount > 0
}
// result converts the aggregate into the monitor result format.
func (agg monitorAggregate) result() monitor.Result {
avg := agg.avgResponse()
result := monitor.Result{
AvgResponse: avg,
MinResponse: agg.minUs,
MaxResponse: agg.maxUs,
PacketLoss: agg.lossPercentage(),
TotalCount: agg.totalCount,
SuccessCount: agg.successCount,
ResponseSum: agg.sumUs,
}
if agg.successCount == 0 {
result.MinResponse, result.MaxResponse = 0, 0
}
return result
}
// avgResponse returns the rounded average of successful samples.
func (agg monitorAggregate) avgResponse() int64 {
if agg.successCount == 0 {
return 0
}
return agg.sumUs / agg.successCount
}
// lossPercentage returns the rounded failure rate for the aggregate.
func (agg monitorAggregate) lossPercentage() float64 {
if agg.totalCount == 0 {
return 0
}
return math.Round(float64(agg.totalCount-agg.successCount)/float64(agg.totalCount)*10000) / 100
}
// resultLocked returns the aggregated monitor result for the requested duration along with a bool indicating whether any data was available.
func (h *monitorHistory) resultLocked(duration time.Duration, now time.Time) (monitor.Result, bool) {
agg := h.aggregateLocked(duration, now)
if !agg.hasData() {
// short realtime windows (e.g. the 1s window used for 1m/realtime charts) often fall
// between monitor samples since monitors run at longer, user-defined intervals; fall back to
// the most recent sample so realtime requests still report current status.
agg = h.latestSampleAggregateLocked()
}
hourAgg := h.aggregateLocked(time.Hour, now)
if !agg.hasData() {
return monitor.Result{}, false
}
result := agg.result()
if len(h.samples) > 0 {
result.LastProbeAt = h.samples[len(h.samples)-1].timestamp.UnixMilli()
}
result.AvgResponse1h = hourAgg.avgResponse()
result.MinResponse1h = hourAgg.minUs
result.MaxResponse1h = hourAgg.maxUs
result.PacketLoss1h = hourAgg.lossPercentage()
result.SampleCount = h.sampleCount
if hourAgg.successCount == 0 {
result.MinResponse1h, result.MaxResponse1h = 0, 0
}
return result, true
}
// latestSampleAggregateLocked returns an aggregate containing only the most recent sample, if any.
func (h *monitorHistory) latestSampleAggregateLocked() monitorAggregate {
agg := newMonitorAggregate()
if len(h.samples) == 0 {
return agg
}
agg.addResponse(h.samples[len(h.samples)-1].responseUs)
return agg
}
// aggregateLocked collects monitor data for the requested time window.
func (h *monitorHistory) aggregateLocked(duration time.Duration, now time.Time) monitorAggregate {
cutoff := now.Add(-duration)
// Keep short windows exact; longer windows read from minute buckets to avoid raw-sample retention.
if duration <= monitorRawRetention {
return aggregateSamplesSince(h.samples, cutoff)
}
return aggregateBucketsSince(h.buckets[:], cutoff, now)
}
// aggregateSamplesSince aggregates raw samples newer than the cutoff.
func aggregateSamplesSince(samples []monitorSample, cutoff time.Time) monitorAggregate {
agg := newMonitorAggregate()
for _, sample := range samples {
if sample.timestamp.Before(cutoff) {
continue
}
agg.addResponse(sample.responseUs)
}
return agg
}
// aggregateBucketsSince aggregates minute buckets overlapping the requested window.
func aggregateBucketsSince(buckets []monitorBucket, cutoff, now time.Time) monitorAggregate {
agg := newMonitorAggregate()
startMinute := int32(cutoff.Unix() / 60)
endMinute := int32(now.Unix() / 60)
for _, bucket := range buckets {
if !bucket.filled || bucket.minute < startMinute || bucket.minute > endMinute {
continue
}
agg.addAggregate(bucket.stats)
}
return agg
}
// addSampleLocked stores a fresh sample in both raw and per-minute retention buffers.
func (h *monitorHistory) addSampleLocked(sample monitorSample) {
h.sampleCount++
cutoff := sample.timestamp.Add(-monitorRawRetention)
start := 0
for i := range h.samples {
if !h.samples[i].timestamp.Before(cutoff) {
start = i
break
}
if i == len(h.samples)-1 {
start = len(h.samples)
}
}
if start > 0 {
size := copy(h.samples, h.samples[start:])
h.samples = h.samples[:size]
}
h.samples = append(h.samples, sample)
minute := int32(sample.timestamp.Unix() / 60)
// Each slot stores one wall-clock minute, so the ring stays fixed-size at ~1h per monitor.
bucket := &h.buckets[minute%monitorMinuteBucketLen]
if !bucket.filled || bucket.minute != minute {
bucket.minute = minute
bucket.filled = true
bucket.stats = newMonitorAggregate()
}
bucket.stats.addResponse(sample.responseUs)
}

View File

@@ -1,154 +0,0 @@
package agent
import (
"testing"
"time"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestMonitorHistoryWindowCounts(t *testing.T) {
history := newMonitorHistory()
now := time.Now()
// This older success counts toward lifetime warm-up, but not this window.
history.record(monitorSample{responseUs: 1000, timestamp: now.Add(-2 * time.Minute)})
history.record(monitorSample{responseUs: 10, timestamp: now.Add(-30 * time.Second)})
history.record(monitorSample{responseUs: 21, timestamp: now.Add(-20 * time.Second)})
history.record(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
result, ok := history.result(time.Minute, now)
require.True(t, ok)
assert.EqualValues(t, 4, result.SampleCount)
assert.EqualValues(t, 3, result.TotalCount)
assert.EqualValues(t, 2, result.SuccessCount)
assert.EqualValues(t, 31, result.ResponseSum, "preserve the sum before average rounding")
assert.EqualValues(t, 15, result.AvgResponse)
assert.Equal(t, 33.33, result.PacketLoss)
encoded, err := cbor.Marshal(result)
require.NoError(t, err)
var decoded monitor.Result
require.NoError(t, cbor.Unmarshal(encoded, &decoded))
assert.Equal(t, result, decoded)
stats := monitor.Stats{}.FromResult(decoded)
assert.Equal(t, result.TotalCount, stats.TotalCount)
assert.Equal(t, result.SuccessCount, stats.SuccessCount)
assert.Equal(t, result.ResponseSum, stats.ResponseSum)
// Reads do not consume samples. A short window's latest-sample fallback
// carries the count for that single failure, not the minute or lifetime count.
repeated, _ := history.result(time.Minute, now)
assert.Equal(t, result, repeated)
fallback, ok := history.result(time.Second, now)
require.True(t, ok)
assert.EqualValues(t, 1, fallback.TotalCount)
assert.Zero(t, fallback.SuccessCount)
assert.Zero(t, fallback.ResponseSum)
assert.Equal(t, 100.0, fallback.PacketLoss)
assert.EqualValues(t, 4, fallback.SampleCount)
}
func TestMonitorHistoryAggregateLockedUsesRawSamplesForShortWindows(t *testing.T) {
now := time.Date(2026, time.April, 21, 12, 0, 0, 0, time.UTC)
history := newMonitorHistory()
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-90 * time.Second)})
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-30 * time.Second)})
history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
agg := history.aggregateLocked(time.Minute, now)
require.True(t, agg.hasData())
assert.Equal(t, int64(2), agg.totalCount)
assert.Equal(t, int64(1), agg.successCount)
result := agg.result()
assert.Equal(t, int64(20), result.AvgResponse)
assert.Equal(t, int64(20), result.MinResponse)
assert.Equal(t, int64(20), result.MaxResponse)
assert.Equal(t, 50.0, result.PacketLoss)
}
func TestMonitorHistoryAggregateLockedUsesMinuteBucketsForLongWindows(t *testing.T) {
now := time.Date(2026, time.April, 21, 12, 0, 30, 0, time.UTC)
history := newMonitorHistory()
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-11 * time.Minute)})
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-9 * time.Minute)})
history.addSampleLocked(monitorSample{responseUs: 40, timestamp: now.Add(-5 * time.Minute)})
history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-90 * time.Second)})
history.addSampleLocked(monitorSample{responseUs: 30, timestamp: now.Add(-30 * time.Second)})
agg := history.aggregateLocked(10*time.Minute, now)
require.True(t, agg.hasData())
assert.Equal(t, int64(4), agg.totalCount)
assert.Equal(t, int64(3), agg.successCount)
result := agg.result()
assert.Equal(t, int64(30), result.AvgResponse)
assert.Equal(t, int64(20), result.MinResponse)
assert.Equal(t, int64(40), result.MaxResponse)
assert.Equal(t, 25.0, result.PacketLoss)
}
func TestMonitorHistoryAddSampleLockedTrimsRawSamplesButKeepsBucketHistory(t *testing.T) {
now := time.Date(2026, time.April, 21, 12, 0, 0, 0, time.UTC)
history := newMonitorHistory()
history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-10 * time.Minute)})
history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now})
require.Len(t, history.samples, 1)
assert.Equal(t, int64(20), history.samples[0].responseUs)
agg := history.aggregateLocked(10*time.Minute, now)
require.True(t, agg.hasData())
assert.Equal(t, int64(2), agg.totalCount)
assert.Equal(t, int64(2), agg.successCount)
result := agg.result()
assert.Equal(t, int64(15), result.AvgResponse)
assert.Equal(t, int64(10), result.MinResponse)
assert.Equal(t, int64(20), result.MaxResponse)
assert.Equal(t, 0.0, result.PacketLoss)
}
func TestMonitorHistoryProbeTimestamp(t *testing.T) {
history := newMonitorHistory()
start := time.Date(2026, time.September, 14, 12, 0, 0, 0, time.UTC)
_, ok := history.result(time.Minute, start)
require.False(t, ok)
first := history.record(monitorSample{responseUs: 20, timestamp: start})
assert.Equal(t, start.UnixMilli(), first.LastProbeAt)
for minute := 0; minute < 5; minute++ {
now := start.Add(time.Duration(minute)*time.Minute + time.Second)
// Realtime reads must not consume freshness for the persistence request.
for _, window := range []time.Duration{time.Second, time.Minute} {
result, ok := history.result(window, now)
require.True(t, ok)
assert.Equal(t, first.LastProbeAt, result.LastProbeAt)
assert.Equal(t, int64(20), result.AvgResponse)
}
}
next := start.Add(5 * time.Minute)
failed := history.record(monitorSample{responseUs: -1, timestamp: next})
assert.Equal(t, next.UnixMilli(), failed.LastProbeAt)
assert.Equal(t, float64(100), failed.PacketLoss)
repeated, ok := history.result(time.Minute, next.Add(2*time.Minute))
require.True(t, ok)
assert.Equal(t, failed.LastProbeAt, repeated.LastProbeAt)
assert.Equal(t, float64(100), repeated.PacketLoss)
}
func TestMonitorHistorySampleCount(t *testing.T) {
history := newMonitorHistory()
now := time.Now()
// Both failed and successful probes count, including older samples so
// monitors with hourly intervals can finish warming up.
history.record(monitorSample{responseUs: -1, timestamp: now.Add(-2 * time.Hour)})
for i, response := range []int64{10, -1, 20} {
result := history.record(monitorSample{responseUs: response, timestamp: now.Add(time.Duration(i) * time.Second)})
assert.EqualValues(t, i+2, result.SampleCount)
}
result, ok := history.clone().result(time.Minute, now.Add(3*time.Second))
require.True(t, ok)
assert.EqualValues(t, 4, result.SampleCount)
}

View File

@@ -1,312 +0,0 @@
package agent
import (
"bytes"
"context"
"crypto/rand"
"errors"
"fmt"
"math"
"net"
"os"
"os/exec"
"regexp"
"runtime"
"strconv"
"strings"
"sync"
"sync/atomic"
"time"
"golang.org/x/net/icmp"
"golang.org/x/net/ipv4"
"golang.org/x/net/ipv6"
"log/slog"
)
// Match the numeric RTT independently of the localized label used by Windows.
var pingTimeRegex = regexp.MustCompile(`(?i)[=<]\s*([0-9]+(?:[.,][0-9]+)?)\s*ms\b`)
var icmpSequence atomic.Uint32
type icmpPacketConn interface {
Close() error
}
// icmpMethod tracks which ICMP approach to use. Once a method succeeds or
// all native methods fail, the choice is cached so subsequent monitors skip
// the trial-and-error overhead.
type icmpMethod uint8
const (
icmpUntried icmpMethod = iota // haven't tried yet
icmpRaw // privileged raw socket
icmpDatagram // unprivileged datagram socket
icmpExecFallback // shell out to system ping command
)
// icmpFamily holds the network parameters and cached detection result for one address family.
type icmpFamily struct {
rawNetwork string // e.g. "ip4:icmp" or "ip6:ipv6-icmp"
dgramNetwork string // e.g. "udp4" or "udp6"
listenAddr string // "0.0.0.0" or "::"
echoType icmp.Type // outgoing echo request type
replyType icmp.Type // expected echo reply type
proto int // IANA protocol number for parsing replies
isIPv6 bool
mode icmpMethod // cached detection result (guarded by icmpModeMu)
}
var (
icmpV4 = icmpFamily{
rawNetwork: "ip4:icmp",
dgramNetwork: "udp4",
listenAddr: "0.0.0.0",
echoType: ipv4.ICMPTypeEcho,
replyType: ipv4.ICMPTypeEchoReply,
proto: 1,
}
icmpV6 = icmpFamily{
rawNetwork: "ip6:ipv6-icmp",
dgramNetwork: "udp6",
listenAddr: "::",
echoType: ipv6.ICMPTypeEchoRequest,
replyType: ipv6.ICMPTypeEchoReply,
proto: 58,
isIPv6: true,
}
icmpModeMu sync.Mutex
icmpListen = func(network, listenAddr string) (icmpPacketConn, error) {
return icmp.ListenPacket(network, listenAddr)
}
)
// monitorICMP sends an ICMP echo request and measures round-trip response.
// Supports both IPv4 and IPv6 targets. The ICMP method (raw socket,
// unprivileged datagram, or exec fallback) is detected once per address
// family and cached for subsequent monitors.
// Returns response in microseconds, or -1 and an error on failure.
func monitorICMP(ctx context.Context, target string) (int64, error) {
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
defer cancel()
family, ip, err := resolveICMPTarget(ctx, target)
if err != nil {
return -1, err
}
icmpModeMu.Lock()
if family.mode == icmpUntried {
family.mode = detectICMPMode(family, icmpListen)
}
mode := family.mode
icmpModeMu.Unlock()
switch mode {
case icmpRaw:
return monitorICMPNative(ctx, family.rawNetwork, family, &net.IPAddr{IP: ip})
case icmpDatagram:
return monitorICMPNative(ctx, family.dgramNetwork, family, &net.UDPAddr{IP: ip})
case icmpExecFallback:
return monitorICMPExec(ctx, ip.String(), family.isIPv6)
default:
return -1, errors.New("unsupported ICMP mode")
}
}
// resolveICMPTarget resolves a target hostname or IP to determine the address
// family and concrete IP address. Prefers IPv4 for dual-stack hostnames.
func resolveICMPTarget(ctx context.Context, target string) (*icmpFamily, net.IP, error) {
if ip := net.ParseIP(target); ip != nil {
if ip.To4() != nil {
return &icmpV4, ip.To4(), nil
}
return &icmpV6, ip, nil
}
ips, err := net.DefaultResolver.LookupIP(ctx, "ip", target)
if err != nil || len(ips) == 0 {
return nil, nil, err
}
for _, ip := range ips {
if v4 := ip.To4(); v4 != nil {
return &icmpV4, v4, nil
}
}
return &icmpV6, ips[0], nil
}
func detectICMPMode(family *icmpFamily, listen func(network, listenAddr string) (icmpPacketConn, error)) icmpMethod {
label := "IPv4"
if family.isIPv6 {
label = "IPv6"
}
conn, err := listen(family.rawNetwork, family.listenAddr)
slog.Debug("ICMP raw socket test", "family", label, "err", err)
if err == nil {
conn.Close()
return icmpRaw
}
conn, err = listen(family.dgramNetwork, family.listenAddr)
slog.Debug("ICMP datagram socket test", "family", label, "err", err)
if err == nil {
conn.Close()
return icmpDatagram
}
return icmpExecFallback
}
// monitorICMPNative sends an ICMP echo request using Go's x/net/icmp package.
func monitorICMPNative(ctx context.Context, network string, family *icmpFamily, dst net.Addr) (int64, error) {
conn, err := icmp.ListenPacket(network, family.listenAddr)
if err != nil {
return -1, err
}
defer conn.Close()
return monitorICMPPacket(ctx, conn, family, dst)
}
func monitorICMPPacket(ctx context.Context, conn net.PacketConn, family *icmpFamily, dst net.Addr) (int64, error) {
if err := ctx.Err(); err != nil {
return -1, err
}
// Closing the socket interrupts both reads and writes on cancellation.
stop := context.AfterFunc(ctx, func() { _ = conn.Close() })
defer stop()
// Prepare correlation data before starting the round-trip timer. The token
// also distinguishes delayed replies after the 16-bit sequence wraps.
token := make([]byte, 16)
if _, err := rand.Read(token); err != nil {
return -1, err
}
echo := &icmp.Echo{
ID: os.Getpid() & 0xffff,
Seq: int(icmpSequence.Add(1) & 0xffff),
Data: token,
}
// Linux ping sockets replace the Echo ID with their bound port. Darwin
// datagram sockets and raw sockets preserve the supplied ID.
if local, ok := conn.LocalAddr().(*net.UDPAddr); ok && runtime.GOOS == "linux" {
echo.ID = local.Port
}
targetIP := icmpAddrIP(dst)
msg := &icmp.Message{
Type: family.echoType,
Code: 0,
Body: echo,
}
msgBytes, err := msg.Marshal(nil)
if err != nil {
return -1, err
}
// Set deadline before sending
if err := conn.SetDeadline(time.Now().Add(3 * time.Second)); err != nil {
return -1, err
}
buf := make([]byte, 1500)
start := time.Now()
if _, err := conn.WriteTo(msgBytes, dst); err != nil {
return -1, err
}
// Read reply
for {
n, peer, err := conn.ReadFrom(buf)
received := time.Now()
if err != nil {
return -1, err
}
if !targetIP.Equal(icmpAddrIP(peer)) {
continue
}
reply, err := icmp.ParseMessage(family.proto, buf[:n])
if err != nil || reply.Type != family.replyType || reply.Code != 0 {
continue
}
body, ok := reply.Body.(*icmp.Echo)
if ok && body.ID == echo.ID && body.Seq == echo.Seq && bytes.Equal(body.Data, echo.Data) {
return received.Sub(start).Microseconds(), nil
}
// Keep waiting for our reply without extending the original deadline.
}
}
func icmpAddrIP(addr net.Addr) net.IP {
switch addr := addr.(type) {
case *net.IPAddr:
return addr.IP
case *net.UDPAddr:
return addr.IP
default:
return nil
}
}
// pingCommand selects the executable and arguments for the supported agent platforms.
// The context deadline enforces the timeout: -W has incompatible meanings across
// Linux, BSD IPv4 ping, and macOS ping6.
func pingCommand(goos, target string, isIPv6 bool) (string, []string, error) {
family := "-4"
if isIPv6 {
family = "-6"
}
switch goos {
case "windows":
return "ping", []string{family, "-n", "1", "-w", "3000", target}, nil
case "linux":
return "ping", []string{family, "-n", "-c", "1", target}, nil
case "darwin", "freebsd", "openbsd":
command := "ping"
if isIPv6 {
command = "ping6"
}
return command, []string{"-n", "-c", "1", target}, nil
default:
return "", nil, fmt.Errorf("ping fallback is unsupported on %s", goos)
}
}
// monitorICMPExec falls back to the system ping command. Returns -1 and an error on failure.
func monitorICMPExec(ctx context.Context, target string, isIPv6 bool) (int64, error) {
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
defer cancel()
name, args, err := pingCommand(runtime.GOOS, target, isIPv6)
if err != nil {
return -1, err
}
cmd := exec.CommandContext(ctx, name, args...)
// Keep Unix output and decimal formatting stable. Windows ignores LC_ALL.
cmd.Env = append(os.Environ(), "LC_ALL=C")
output, err := cmd.Output()
if ctx.Err() != nil {
return -1, ctx.Err()
}
if err != nil {
return -1, fmt.Errorf("%s failed: %w", name, err)
}
return parsePingResponse(output)
}
// parsePingResponse returns the reported RTT, never subprocess execution time.
// For a bounded value such as Windows' time<1ms, retain the reported upper bound.
func parsePingResponse(output []byte) (int64, error) {
matches := pingTimeRegex.FindSubmatch(output)
if len(matches) < 2 {
return -1, errors.New("ping output contains no round-trip time")
}
ms, err := strconv.ParseFloat(strings.ReplaceAll(string(matches[1]), ",", "."), 64)
if err != nil || math.IsInf(ms, 0) || ms >= float64(math.MaxInt64)/1000 {
return -1, errors.New("invalid round-trip time in ping output")
}
return int64(math.Round(ms * 1000)), nil
}

View File

@@ -1,433 +0,0 @@
//go:build testing
package agent
import (
"context"
"errors"
"fmt"
"net"
"os"
"path/filepath"
"runtime"
"testing"
"time"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"golang.org/x/net/icmp"
)
type testICMPPacketConn struct{}
func (testICMPPacketConn) Close() error { return nil }
type blockingICMPConn struct {
net.PacketConn
reading chan struct{}
}
func (c *blockingICMPConn) WriteTo(p []byte, addr net.Addr) (int, error) {
return len(p), nil
}
func (c *blockingICMPConn) ReadFrom(p []byte) (int, net.Addr, error) {
close(c.reading)
return c.PacketConn.ReadFrom(p)
}
func TestMonitorICMPPacketCancellation(t *testing.T) {
conn, err := net.ListenPacket("udp4", "127.0.0.1:0")
require.NoError(t, err)
defer conn.Close()
blocking := &blockingICMPConn{PacketConn: conn, reading: make(chan struct{})}
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
done := make(chan error, 1)
go func() {
_, err := monitorICMPPacket(ctx, blocking, &icmpV4, conn.LocalAddr())
done <- err
}()
select {
case <-blocking.reading:
case <-time.After(time.Second):
t.Fatal("probe did not begin reading")
}
cancel()
select {
case err := <-done:
require.Error(t, err)
case <-time.After(time.Second):
t.Fatal("cancellation did not interrupt the socket read")
}
}
func TestMonitorICMPExecCancellation(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("test uses a POSIX shell stub for ping")
}
dir := t.TempDir()
require.NoError(t, os.WriteFile(filepath.Join(dir, "ping"), []byte("#!/bin/sh\nexec sleep 30\n"), 0o755))
t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH"))
ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond)
defer cancel()
done := make(chan error, 1)
go func() {
_, err := monitorICMPExec(ctx, "127.0.0.1", false)
done <- err
}()
select {
case err := <-done:
require.ErrorIs(t, err, context.DeadlineExceeded)
case <-time.After(time.Second):
t.Fatal("cancellation did not terminate ping")
}
}
func TestPingCommand(t *testing.T) {
for _, goos := range []string{"linux", "windows", "darwin", "freebsd", "openbsd"} {
for _, ipv6 := range []bool{false, true} {
t.Run(fmt.Sprintf("%s/ipv6=%t", goos, ipv6), func(t *testing.T) {
target, family := "192.0.2.1", "-4"
if ipv6 {
target, family = "2001:db8::1", "-6"
}
name, args, err := pingCommand(goos, target, ipv6)
require.NoError(t, err)
wantName := "ping"
wantArgs := []string{"-n", "-c", "1", target}
switch goos {
case "windows":
wantArgs = []string{family, "-n", "1", "-w", "3000", target}
case "linux":
wantArgs = append([]string{family}, wantArgs...)
default:
if ipv6 {
wantName = "ping6"
}
}
assert.Equal(t, wantName, name)
assert.Equal(t, wantArgs, args)
})
}
}
_, _, err := pingCommand("unsupported", "192.0.2.1", false)
require.Error(t, err)
}
func TestParsePingResponse(t *testing.T) {
for _, tc := range []struct {
name string
output string
wantUs int64
}{
{"linux", "64 bytes from 192.0.2.1: icmp_seq=1 ttl=64 time=12.345 ms", 12345},
{"bsd", "64 bytes from 192.0.2.1: icmp_seq=0 ttl=64 time=0.023 ms", 23},
{"ipv6", "64 bytes from 2001:db8::1: icmp_seq=0 hlim=64 time=1.234 ms", 1234},
{"windows", "Reply from 192.0.2.1: bytes=32 time=12ms TTL=128", 12000},
{"windows submillisecond", "Reply from ::1: time<1ms", 1000},
{"localized windows", "Antwort von 192.0.2.1: Bytes=32 Zeit=12ms TTL=128", 12000},
{"decimal comma", "64 bytes from 192.0.2.1: time=1,234 ms", 1234},
{"rounding", "time=0.1236 ms", 124},
{"empty", "", -1},
{"timeout", "Request timed out.", -1},
{"unreachable", "Reply from 192.0.2.1: Destination host unreachable.", -1},
{"malformed", "time=oops ms", -1},
{"negative", "time=-1 ms", -1},
{"overflow", "time=999999999999999999999 ms", -1},
} {
t.Run(tc.name, func(t *testing.T) {
responseUs, err := parsePingResponse([]byte(tc.output))
if tc.wantUs < 0 {
require.Error(t, err)
} else {
require.NoError(t, err)
}
assert.Equal(t, tc.wantUs, responseUs)
})
}
}
func TestMonitorICMPExecOutput(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("test uses a POSIX shell stub for ping")
}
for _, tc := range []struct {
name string
output string
exit int
wantUs int64
}{
{"success", "time=1.234 ms", 0, 1234},
{"missing RTT", "unrecognized output", 0, -1},
{"failed command with RTT", "time=1.234 ms", 1, -1},
} {
t.Run(tc.name, func(t *testing.T) {
dir := t.TempDir()
// Also verify an inherited locale cannot override the C locale.
script := fmt.Sprintf("#!/bin/sh\n[ \"$LC_ALL\" = C ] || exit 2\nprintf '%%s\\n' '%s'\nexit %d\n", tc.output, tc.exit)
require.NoError(t, os.WriteFile(filepath.Join(dir, "ping"), []byte(script), 0o755))
t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH"))
t.Setenv("LC_ALL", "de_DE.UTF-8")
responseUs, err := monitorICMPExec(t.Context(), "127.0.0.1", false)
if tc.wantUs < 0 {
require.Error(t, err)
} else {
require.NoError(t, err)
}
assert.Equal(t, tc.wantUs, responseUs)
})
}
}
type icmpTestReply struct {
data []byte
peer net.Addr
}
type scriptedICMPConn struct {
net.PacketConn
local net.Addr
onWrite func([]byte, net.Addr)
replies []icmpTestReply
reads int
deadlineSets int
}
func (c *scriptedICMPConn) LocalAddr() net.Addr { return c.local }
func (c *scriptedICMPConn) SetDeadline(deadline time.Time) error {
c.deadlineSets++
return nil
}
func (c *scriptedICMPConn) WriteTo(data []byte, dst net.Addr) (int, error) {
c.onWrite(data, dst)
return len(data), nil
}
func (c *scriptedICMPConn) ReadFrom(buf []byte) (int, net.Addr, error) {
c.reads++
if len(c.replies) == 0 {
return 0, nil, os.ErrDeadlineExceeded
}
reply := c.replies[0]
c.replies = c.replies[1:]
return copy(buf, reply.data), reply.peer, nil
}
func TestMonitorICMPReplyCorrelation(t *testing.T) {
for _, family := range []*icmpFamily{&icmpV4, &icmpV6} {
for _, datagram := range []bool{false, true} {
network := family.rawNetwork
ip, other := net.ParseIP("192.0.2.1"), net.ParseIP("192.0.2.2")
if family.isIPv6 {
ip, other = net.ParseIP("2001:db8::1"), net.ParseIP("2001:db8::2")
}
var dst net.Addr = &net.IPAddr{IP: ip}
var wrongPeer net.Addr = &net.IPAddr{IP: other}
if datagram {
network = family.dgramNetwork
dst = &net.UDPAddr{IP: ip}
wrongPeer = &net.UDPAddr{IP: other}
}
for _, mismatch := range []string{"source", "id", "sequence", "payload", "type", "code", "malformed"} {
for _, eventuallyMatches := range []bool{false, true} {
ending := "timeout"
if eventuallyMatches {
ending = "success"
}
t.Run(network+"/"+mismatch+"/"+ending, func(t *testing.T) {
conn := &scriptedICMPConn{local: &net.IPAddr{IP: net.IPv4zero}}
if datagram {
conn.local = &net.UDPAddr{Port: 12345}
if runtime.GOOS == "linux" {
// Deliberately differ from the process ID.
conn.local = &net.UDPAddr{Port: (os.Getpid() % 65534) + 1}
}
}
conn.onWrite = func(data []byte, target net.Addr) {
require.Equal(t, dst, target)
request, err := icmp.ParseMessage(family.proto, data)
require.NoError(t, err)
echo := request.Body.(*icmp.Echo)
expectedID := os.Getpid() & 0xffff
if datagram && runtime.GOOS == "linux" {
expectedID = conn.local.(*net.UDPAddr).Port
}
require.Equal(t, expectedID, echo.ID)
reply := &icmp.Message{Type: family.replyType, Body: echo}
valid, err := reply.Marshal(nil)
require.NoError(t, err)
peer := dst
switch mismatch {
case "source":
peer = wrongPeer
case "id":
echo.ID ^= 1
case "sequence":
echo.Seq ^= 1
case "payload":
echo.Data[0] ^= 1
case "type":
reply.Type = family.echoType
case "code":
reply.Code = 1
}
invalid, err := reply.Marshal(nil)
require.NoError(t, err)
if mismatch == "malformed" {
invalid = invalid[:2]
}
conn.replies = []icmpTestReply{{invalid, peer}}
if eventuallyMatches {
conn.replies = append(conn.replies, icmpTestReply{valid, dst})
}
}
elapsed, err := monitorICMPPacket(context.Background(), conn, family, dst)
if eventuallyMatches {
require.NoError(t, err)
assert.GreaterOrEqual(t, elapsed, int64(0))
} else {
require.ErrorIs(t, err, os.ErrDeadlineExceeded)
assert.Equal(t, int64(-1), elapsed)
}
assert.Equal(t, 2, conn.reads)
assert.Equal(t, 1, conn.deadlineSets)
})
}
}
}
}
}
func TestMonitorICMPLoopback(t *testing.T) {
for _, family := range []*icmpFamily{&icmpV4, &icmpV6} {
for _, network := range []string{family.rawNetwork, family.dgramNetwork} {
t.Run(network, func(t *testing.T) {
conn, err := icmp.ListenPacket(network, family.listenAddr)
if err != nil {
t.Skipf("ICMP socket unavailable: %v", err)
}
defer conn.Close()
ip := net.ParseIP("127.0.0.1")
if family.isIPv6 {
ip = net.ParseIP("::1")
}
var dst net.Addr = &net.IPAddr{IP: ip}
if network == family.dgramNetwork {
dst = &net.UDPAddr{IP: ip}
}
elapsed, err := monitorICMPPacket(context.Background(), conn, family, dst)
require.NoError(t, err)
assert.GreaterOrEqual(t, elapsed, int64(0))
})
}
}
}
func TestDetectICMPMode(t *testing.T) {
tests := []struct {
name string
family *icmpFamily
rawErr error
udpErr error
want icmpMethod
wantNetworks []string
}{
{
name: "IPv4 prefers raw socket when available",
family: &icmpV4,
want: icmpRaw,
wantNetworks: []string{"ip4:icmp"},
},
{
name: "IPv4 uses datagram when raw unavailable",
family: &icmpV4,
rawErr: errors.New("operation not permitted"),
want: icmpDatagram,
wantNetworks: []string{"ip4:icmp", "udp4"},
},
{
name: "IPv4 falls back to exec when both unavailable",
family: &icmpV4,
rawErr: errors.New("operation not permitted"),
udpErr: errors.New("protocol not supported"),
want: icmpExecFallback,
wantNetworks: []string{"ip4:icmp", "udp4"},
},
{
name: "IPv6 prefers raw socket when available",
family: &icmpV6,
want: icmpRaw,
wantNetworks: []string{"ip6:ipv6-icmp"},
},
{
name: "IPv6 uses datagram when raw unavailable",
family: &icmpV6,
rawErr: errors.New("operation not permitted"),
want: icmpDatagram,
wantNetworks: []string{"ip6:ipv6-icmp", "udp6"},
},
{
name: "IPv6 falls back to exec when both unavailable",
family: &icmpV6,
rawErr: errors.New("operation not permitted"),
udpErr: errors.New("protocol not supported"),
want: icmpExecFallback,
wantNetworks: []string{"ip6:ipv6-icmp", "udp6"},
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
calls := make([]string, 0, 2)
listen := func(network, listenAddr string) (icmpPacketConn, error) {
require.Equal(t, tt.family.listenAddr, listenAddr)
calls = append(calls, network)
switch network {
case tt.family.rawNetwork:
if tt.rawErr != nil {
return nil, tt.rawErr
}
case tt.family.dgramNetwork:
if tt.udpErr != nil {
return nil, tt.udpErr
}
default:
t.Fatalf("unexpected network %q", network)
}
return testICMPPacketConn{}, nil
}
assert.Equal(t, tt.want, detectICMPMode(tt.family, listen))
assert.Equal(t, tt.wantNetworks, calls)
})
}
}
func TestResolveICMPTarget(t *testing.T) {
t.Run("IPv4 literal", func(t *testing.T) {
family, ip, err := resolveICMPTarget(context.Background(), "127.0.0.1")
require.NoError(t, err)
require.NotNil(t, family)
assert.False(t, family.isIPv6)
assert.Equal(t, "127.0.0.1", ip.String())
})
t.Run("IPv6 literal", func(t *testing.T) {
family, ip, err := resolveICMPTarget(context.Background(), "::1")
require.NoError(t, err)
require.NotNil(t, family)
assert.True(t, family.isIPv6)
assert.Equal(t, "::1", ip.String())
})
t.Run("IPv4-mapped IPv6 resolves as IPv4", func(t *testing.T) {
family, ip, err := resolveICMPTarget(context.Background(), "::ffff:127.0.0.1")
require.NoError(t, err)
require.NotNil(t, family)
assert.False(t, family.isIPv6)
assert.Equal(t, "127.0.0.1", ip.String())
})
}

View File

@@ -1,109 +0,0 @@
package agent
import (
"context"
"errors"
"fmt"
"net"
"net/http"
"time"
"github.com/henrygd/beszel"
"github.com/henrygd/beszel/internal/entities/monitor"
)
const networkMonitorUserAgent = "Beszel-Agent/" + beszel.Version + " (+https://beszel.dev)"
// monitorProbe performs one check. Errors are recorded as loss by the task runner.
// Implementations must honor cancellation and bound their execution time.
type monitorProbe func(context.Context, monitor.Config) (int64, error)
func networkMonitorProbe(client *http.Client) monitorProbe {
return func(ctx context.Context, config monitor.Config) (int64, error) {
switch config.Protocol {
case "icmp":
return monitorICMP(ctx, config.Target)
case "tcp":
return monitorTCP(ctx, config.Target, config.Port)
case "http":
return monitorHTTP(ctx, client, config.Target)
case "dns":
return monitorDNS(ctx, config.Target)
default:
return -1, fmt.Errorf("unknown monitor protocol: %s", config.Protocol)
}
}
}
// monitorTCP measures connection establishment time, including address fallback
// but excluding DNS resolution.
// Returns -1 and an error on failure.
func monitorTCP(ctx context.Context, target string, port uint16) (int64, error) {
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
defer cancel()
// Resolve DNS first, outside the timing window but within the probe deadline.
ips, err := net.DefaultResolver.LookupHost(ctx, target)
if err != nil {
return -1, err
}
if len(ips) == 0 {
return -1, errors.New("no addresses resolved for TCP monitor")
}
portString := fmt.Sprintf("%d", port)
deadline, _ := ctx.Deadline()
// Share the remaining probe budget across addresses so an unresponsive
// first address cannot consume all the time available for alternatives.
start := time.Now()
for i, ip := range ips {
if err := ctx.Err(); err != nil {
return -1, err
}
dialer := net.Dialer{Timeout: time.Until(deadline) / time.Duration(len(ips)-i)}
var conn net.Conn
conn, err = dialer.DialContext(ctx, "tcp", net.JoinHostPort(ip, portString))
if err != nil {
continue
}
responseUs := time.Since(start).Microseconds()
conn.Close()
return responseUs, nil
}
return -1, err
}
// monitorDNS measures DNS resolution response time in microseconds. Returns -1 and an error on failure.
func monitorDNS(ctx context.Context, target string) (int64, error) {
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
defer cancel()
start := time.Now()
ips, err := net.DefaultResolver.LookupHost(ctx, target)
if err != nil || len(ips) == 0 {
return -1, err
}
return time.Since(start).Microseconds(), nil
}
// monitorHTTP measures HTTP GET request response in microseconds. Returns -1 and an error on failure.
func monitorHTTP(ctx context.Context, client *http.Client, url string) (int64, error) {
if client == nil {
client = http.DefaultClient
}
start := time.Now()
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return -1, err
}
req.Header.Set("User-Agent", networkMonitorUserAgent)
resp, err := client.Do(req)
if err != nil {
return -1, err
}
resp.Body.Close()
if resp.StatusCode >= 400 {
return -1, fmt.Errorf("HTTP error: %s", resp.Status)
}
return time.Since(start).Microseconds(), nil
}

View File

@@ -1,88 +0,0 @@
package agent
import (
"sync"
"time"
)
const (
monitorResumeHeartbeat = 10 * time.Second
// Allow scheduling jitter without mistaking an ordinary tick for resume.
monitorResumeGap = 2 * monitorResumeHeartbeat
monitorResumePause = 10 * time.Second
)
// monitorResumeGuard detects likely suspend/resume using wall time. A long
// process stall or forward clock adjustment can also trigger the bounded pause.
// One heartbeat is shared by all configured monitors.
type monitorResumeGuard struct {
mu sync.Mutex
stop chan struct{}
lastTick time.Time
pauseUntil time.Time
generation uint32
}
func (g *monitorResumeGuard) start() {
g.mu.Lock()
defer g.mu.Unlock()
if g.stop != nil {
return
}
stop := make(chan struct{})
g.stop = stop
g.lastTick = time.Now().Round(0)
g.pauseUntil = time.Time{}
go func() {
ticker := time.NewTicker(monitorResumeHeartbeat)
defer ticker.Stop()
for {
select {
case <-stop:
return
case <-ticker.C:
g.mu.Lock()
if g.stop == stop {
g.observe(time.Now())
}
g.mu.Unlock()
}
}
}()
}
func (g *monitorResumeGuard) shutdown() {
g.mu.Lock()
defer g.mu.Unlock()
if g.stop != nil {
close(g.stop)
g.stop = nil
g.generation++
}
}
// observe requires mu. Strip the monotonic component because it can stop during
// suspend. Read the current time rather than the ticker's queued timestamp.
func (g *monitorResumeGuard) observe(now time.Time) {
now = now.Round(0)
if now.Sub(g.lastTick) > monitorResumeGap {
g.pauseUntil = now.Add(monitorResumePause)
g.generation++
}
g.lastTick = now
}
// snapshot also observes time so a probe waking before the heartbeat detects
// resume itself. A changed generation invalidates probes spanning suspend.
func (g *monitorResumeGuard) snapshot() (generation uint32, allowed bool) {
if g == nil {
return 0, true
}
g.mu.Lock()
defer g.mu.Unlock()
if g.stop == nil {
return g.generation, true
}
g.observe(time.Now())
return g.generation, !g.lastTick.Before(g.pauseUntil)
}

View File

@@ -1,121 +0,0 @@
//go:build testing
package agent
import (
"context"
"errors"
"sync/atomic"
"testing"
"testing/synctest"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func simulateMonitorSleep(g *monitorResumeGuard) {
g.mu.Lock()
g.lastTick = time.Now().Add(-time.Hour).Round(0)
g.mu.Unlock()
}
func TestMonitorResumePause(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
var g monitorResumeGuard
g.start()
defer g.shutdown()
generation, allowed := g.snapshot()
require.True(t, allowed)
// Heartbeats alone must keep the guard current between infrequent probes.
time.Sleep(time.Minute)
synctest.Wait()
steadyGeneration, allowed := g.snapshot()
require.True(t, allowed)
require.Equal(t, generation, steadyGeneration)
// The probe, rather than the heartbeat, must detect this gap.
simulateMonitorSleep(&g)
next, allowed := g.snapshot()
assert.False(t, allowed)
assert.NotEqual(t, generation, next)
time.Sleep(9 * time.Second)
_, allowed = g.snapshot()
assert.False(t, allowed)
time.Sleep(time.Second)
end, allowed := g.snapshot()
assert.True(t, allowed)
assert.Equal(t, next, end)
})
}
func TestMonitorResumeGuardLifecycle(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
pm := newMonitorManagerWithProbe(func(context.Context, monitor.Config) (int64, error) { return 1, nil })
defer pm.Stop()
assert.Nil(t, pm.resumeGuard.stop)
pm.SyncMonitors([]monitor.Config{{ID: "a", Interval: 3600}, {ID: "b", Interval: 3600}})
stop := pm.resumeGuard.stop
require.NotNil(t, stop)
pm.DeleteMonitor("a")
assert.Equal(t, stop, pm.resumeGuard.stop)
pm.DeleteMonitor("b")
assert.Nil(t, pm.resumeGuard.stop)
select {
case <-stop:
default:
t.Fatal("heartbeat was not stopped")
}
time.Sleep(time.Hour)
_, err := pm.UpsertMonitor(monitor.Config{ID: "c", Interval: 3600}, false)
require.NoError(t, err)
_, allowed := pm.resumeGuard.snapshot()
assert.True(t, allowed, "idle time must not trigger a resume pause")
pm.SyncMonitors(nil)
assert.Nil(t, pm.resumeGuard.stop)
})
}
func TestMonitorResumeDiscardsInflightProbe(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
var g monitorResumeGuard
g.start()
defer g.shutdown()
task := newMonitorTask(monitor.Config{ID: "test"})
defer task.cancel()
task.resumeGuard = &g
result := task.runProbe(func(context.Context, monitor.Config) (int64, error) {
simulateMonitorSleep(&g)
return 0, errors.New("network not ready")
})
assert.Nil(t, result)
assert.Empty(t, task.history.samples)
// Explicit requests may still run during the pause and record real failures.
result = task.runProbe(func(context.Context, monitor.Config) (int64, error) {
return 0, errors.New("unreachable")
})
require.NotNil(t, result)
assert.Equal(t, 100.0, result.PacketLoss)
})
}
func TestMonitorResumeSkipsScheduledProbes(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
var calls atomic.Int32
pm := newMonitorManagerWithProbe(func(context.Context, monitor.Config) (int64, error) {
calls.Add(1)
return 1, nil
})
defer pm.Stop()
pm.SyncMonitors([]monitor.Config{{ID: "test", Interval: 1}})
simulateMonitorSleep(&pm.resumeGuard)
pm.resumeGuard.snapshot()
time.Sleep(9 * time.Second)
synctest.Wait()
assert.Zero(t, calls.Load())
assert.Empty(t, pm.GetResults(1000))
time.Sleep(2 * time.Second)
synctest.Wait()
assert.Positive(t, calls.Load())
})
}

View File

@@ -1,60 +0,0 @@
package agent
import (
"context"
"log/slog"
"math/rand"
"time"
)
func (pm *MonitorManager) startMonitor(task *monitorTask) {
interval := time.Duration(task.config.Interval) * time.Second
if interval < time.Second {
interval = 30 * time.Second
}
delay := getStagger(interval.Milliseconds())
slog.Debug("starting monitor task", "target", task.config.Target, "delay", delay, "interval", interval)
go runMonitorSchedule(task.ctx, interval, delay, func() {
if _, allowed := task.resumeGuard.snapshot(); allowed {
task.runProbe(pm.probe)
}
})
}
// runMonitorSchedule owns only timing. Checks run serially, and slow checks
// naturally drop missed ticks rather than building an execution backlog.
func runMonitorSchedule(ctx context.Context, interval, delay time.Duration, run func()) {
timer := time.NewTimer(delay)
defer timer.Stop()
select {
case <-ctx.Done():
return
case <-timer.C:
}
if ctx.Err() != nil {
return
}
run()
ticker := time.NewTicker(interval)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
if ctx.Err() != nil {
return
}
run()
}
}
}
// getStagger returns an initial delay between half an interval and one interval.
func getStagger(intervalMilli int64) time.Duration {
delay := rand.Intn(int(intervalMilli))
if delay < int(intervalMilli)/2 {
delay += int(intervalMilli) / 2
}
return time.Duration(delay) * time.Millisecond
}

View File

@@ -1,167 +0,0 @@
//go:build testing
package agent
import (
"context"
"sync/atomic"
"testing"
"testing/synctest"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestMonitorScheduleTiming(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
ctx, cancel := context.WithCancel(t.Context())
defer cancel()
var calls atomic.Int32
go runMonitorSchedule(ctx, 10*time.Second, 5*time.Second, func() { calls.Add(1) })
synctest.Wait()
time.Sleep(4 * time.Second)
synctest.Wait()
assert.Equal(t, 0, int(calls.Load()))
time.Sleep(time.Second)
synctest.Wait()
assert.Equal(t, 1, int(calls.Load()))
time.Sleep(10 * time.Second)
synctest.Wait()
assert.Equal(t, 2, int(calls.Load()))
cancel()
synctest.Wait()
time.Sleep(time.Minute)
synctest.Wait()
assert.Equal(t, 2, int(calls.Load()))
})
}
func TestMonitorScheduleSlowProbe(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
ctx, cancel := context.WithCancel(t.Context())
defer cancel()
var calls atomic.Int32
release := make(chan struct{})
go runMonitorSchedule(ctx, time.Second, 0, func() {
calls.Add(1)
select {
case <-release:
case <-ctx.Done():
}
})
synctest.Wait()
assert.Equal(t, 1, int(calls.Load()))
time.Sleep(time.Minute)
synctest.Wait()
assert.Equal(t, 1, int(calls.Load()), "a slow probe must not spawn overlapping checks")
close(release)
synctest.Wait()
assert.Equal(t, 1, int(calls.Load()), "missed intervals must not accumulate a backlog")
time.Sleep(time.Second)
synctest.Wait()
assert.Equal(t, 2, int(calls.Load()))
cancel()
synctest.Wait()
})
}
func TestMonitorScheduledAndImmediateRequestsShareProbe(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
var calls atomic.Int32
release := make(chan struct{})
cfg := monitor.Config{ID: "test", Interval: 10}
pm := newMonitorManagerWithProbe(func(ctx context.Context, config monitor.Config) (int64, error) {
assert.Equal(t, cfg, config)
calls.Add(1)
<-release
return 42, nil
})
defer pm.Stop()
task := newMonitorTask(cfg)
pm.monitors[cfg.ID] = task
go runMonitorSchedule(task.ctx, 10*time.Second, 0, func() { task.runProbe(pm.probe) })
synctest.Wait()
results := make(chan *monitor.Result, 2)
for range 2 {
go func() {
result, _ := pm.UpsertMonitor(cfg, true)
results <- result
}()
}
synctest.Wait()
assert.Equal(t, 1, int(calls.Load()))
assert.Empty(t, pm.GetResults(1000), "reading history must not wait for network I/O")
close(release)
synctest.Wait()
first, second := <-results, <-results
require.NotNil(t, first)
require.NotNil(t, second)
assert.Equal(t, int64(42), first.AvgResponse)
assert.Equal(t, first, second)
assert.NotSame(t, first, second, "callers must not share mutable result pointers")
assert.Len(t, task.history.samples, 1)
// A later explicit request must still perform a fresh probe.
_, err := pm.UpsertMonitor(cfg, true)
require.NoError(t, err)
assert.Equal(t, 2, int(calls.Load()))
assert.Len(t, task.history.samples, 2)
})
}
func TestMonitorReplacementCancelsSharedProbe(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
cfg := monitor.Config{ID: "test", Interval: 10}
pm := newMonitorManagerWithProbe(func(ctx context.Context, config monitor.Config) (int64, error) {
if config.Interval == 10 {
<-ctx.Done()
return 0, ctx.Err()
}
return 30, nil
})
defer pm.Stop()
task := newMonitorTask(cfg)
task.history.record(monitorSample{responseUs: 10, timestamp: time.Now()})
pm.monitors[cfg.ID] = task
results := make(chan *monitor.Result, 2)
for range 2 {
go func() {
result, _ := pm.UpsertMonitor(cfg, true)
results <- result
}()
}
synctest.Wait()
updated := cfg
updated.Interval = 20
result, err := pm.UpsertMonitor(updated, true)
require.NoError(t, err)
require.NotNil(t, result)
assert.Equal(t, int64(20), result.AvgResponse)
assert.Zero(t, result.PacketLoss)
synctest.Wait()
assert.Nil(t, <-results)
assert.Nil(t, <-results)
assert.Len(t, task.history.samples, 1)
assert.Len(t, pm.monitors[cfg.ID].history.samples, 2)
})
}
func TestMonitorInjectedProbeTimeoutRecordsLoss(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
pm := newMonitorManagerWithProbe(func(ctx context.Context, _ monitor.Config) (int64, error) {
ctx, cancel := context.WithTimeout(ctx, 3*time.Second)
defer cancel()
<-ctx.Done()
return 0, ctx.Err()
})
defer pm.Stop()
start := time.Now()
result, err := pm.UpsertMonitor(monitor.Config{ID: "test", Interval: 3600}, true)
require.NoError(t, err)
require.NotNil(t, result)
assert.Equal(t, 3*time.Second, time.Since(start))
assert.Equal(t, 100.0, result.PacketLoss)
assert.NoError(t, pm.monitors["test"].ctx.Err())
})
}

View File

@@ -1,116 +0,0 @@
package agent
import (
"context"
"log/slog"
"sync"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
)
const monitorFailureLogInterval = 5 * time.Minute
// monitorTask coordinates a probe and its history for one immutable configuration.
type monitorTask struct {
config monitor.Config
ctx context.Context
cancel context.CancelFunc
history *monitorHistory
resumeGuard *monitorResumeGuard
runMu sync.Mutex
inflight *monitorRun
lastFailureLog int64 // Unix nanoseconds
}
type monitorRun struct {
done chan struct{}
result *monitor.Result // published by closing done; never mutated afterwards
}
func newMonitorTask(config monitor.Config) *monitorTask {
ctx, cancel := context.WithCancel(context.Background())
task := &monitorTask{config: config, ctx: ctx, history: newMonitorHistory()}
// Serialize cancellation with publication, so canceled probes cannot enter
// history copied into a replacement task.
task.cancel = func() {
task.runMu.Lock()
cancel()
task.runMu.Unlock()
}
return task
}
func newMonitorTaskFromExisting(config monitor.Config, existing *monitorTask) *monitorTask {
task := newMonitorTask(config)
if existing != nil {
task.history = existing.history.clone()
}
return task
}
// runProbe shares an in-flight check between scheduled and immediate requests.
// Every completed check contributes exactly one sample, regardless of how many
// callers were waiting for it. No task or history lock is held during network I/O.
func (task *monitorTask) runProbe(probe monitorProbe) *monitor.Result {
task.runMu.Lock()
if task.ctx.Err() != nil {
task.runMu.Unlock()
return nil
}
if run := task.inflight; run != nil {
task.runMu.Unlock()
select {
case <-task.ctx.Done():
return nil
case <-run.done:
if task.ctx.Err() != nil {
return nil
}
return copyMonitorResult(run.result)
}
}
run := &monitorRun{done: make(chan struct{})}
task.inflight = run
task.runMu.Unlock()
generation, _ := task.resumeGuard.snapshot()
responseUs, err := probe(task.ctx, task.config)
var logFailure bool
task.runMu.Lock()
currentGeneration, _ := task.resumeGuard.snapshot()
if task.ctx.Err() == nil && generation == currentGeneration {
now := time.Now()
if err != nil {
responseUs = -1
logAt := now.UnixNano()
if task.lastFailureLog == 0 || logAt < task.lastFailureLog || logAt-task.lastFailureLog >= int64(monitorFailureLogInterval) {
logFailure = true
task.lastFailureLog = logAt
}
} else {
task.lastFailureLog = 0
}
result := task.history.record(monitorSample{responseUs: responseUs, timestamp: now})
run.result = &result
}
task.inflight = nil
close(run.done)
task.runMu.Unlock()
if logFailure {
slog.Warn("monitor failed", "err", err, "target", task.config.Target, "protocol", task.config.Protocol)
}
if task.ctx.Err() != nil {
return nil
}
return copyMonitorResult(run.result)
}
func copyMonitorResult(result *monitor.Result) *monitor.Result {
if result == nil {
return nil
}
copy := *result
return &copy
}

View File

@@ -1,79 +0,0 @@
//go:build testing
package agent
import (
"bytes"
"context"
"errors"
"log/slog"
"testing"
"testing/synctest"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestMonitorFailureLogCooldown(t *testing.T) {
var logs bytes.Buffer
previous := slog.Default()
slog.SetDefault(slog.New(slog.NewTextHandler(&logs, nil)))
t.Cleanup(func() { slog.SetDefault(previous) })
synctest.Test(t, func(t *testing.T) {
task := newMonitorTask(monitor.Config{ID: "test", Target: "example.test", Protocol: "tcp"})
defer task.cancel()
failure := errors.New("connection refused")
probe := func(context.Context, monitor.Config) (int64, error) { return 42, failure }
var samples int64
check := func(wantLog bool) {
t.Helper()
logs.Reset()
result := task.runProbe(probe)
require.NotNil(t, result)
samples++
assert.Equal(t, samples, result.SampleCount, "suppressed warnings must still record samples")
if !wantLog {
assert.Empty(t, logs.String())
} else {
assert.Contains(t, logs.String(), `msg="monitor failed"`)
assert.Equal(t, 1, bytes.Count(logs.Bytes(), []byte("\n")))
}
}
check(true)
check(false)
time.Sleep(5*time.Minute - time.Nanosecond)
check(false)
time.Sleep(time.Nanosecond)
check(true)
check(false)
time.Sleep(5 * time.Minute)
check(true)
check(false)
// Recovery clears the cooldown.
failure = nil
check(false)
failure = errors.New("connection refused again")
check(true)
// Another monitor has its own cooldown.
other := newMonitorTask(task.config)
defer other.cancel()
logs.Reset()
require.NotNil(t, other.runProbe(probe))
assert.Contains(t, logs.String(), `msg="monitor failed"`)
// A canceled probe must not publish a failure or emit a warning.
logs.Reset()
result := other.runProbe(func(context.Context, monitor.Config) (int64, error) {
other.cancel()
return -1, context.Canceled
})
assert.Nil(t, result)
assert.Empty(t, logs.String())
})
}

View File

@@ -1,526 +0,0 @@
package agent
import (
"context"
"encoding/binary"
"io"
"net"
"net/http"
"net/http/httptest"
"testing"
"time"
"github.com/henrygd/beszel"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"golang.org/x/net/dns/dnsmessage"
)
func TestMonitorManagerGetResultsIncludesHourResponseRange(t *testing.T) {
now := time.Now().UTC()
task := newMonitorTask(monitor.Config{ID: "monitor-1"})
task.history.addSampleLocked(monitorSample{responseUs: 10, timestamp: now.Add(-30 * time.Minute)})
task.history.addSampleLocked(monitorSample{responseUs: 20, timestamp: now.Add(-9 * time.Minute)})
task.history.addSampleLocked(monitorSample{responseUs: 40, timestamp: now.Add(-5 * time.Minute)})
task.history.addSampleLocked(monitorSample{responseUs: 30, timestamp: now.Add(-50 * time.Second)})
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-30 * time.Second)})
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{"icmp:example.com": task}
results := pm.GetResults(uint16(time.Minute / time.Millisecond))
result, ok := results["monitor-1"]
require.True(t, ok)
assert.Equal(t, int64(30), result.AvgResponse)
assert.Equal(t, int64(25), result.AvgResponse1h)
assert.Equal(t, int64(30), result.MinResponse)
assert.Equal(t, int64(10), result.MinResponse1h)
assert.Equal(t, int64(30), result.MaxResponse)
assert.Equal(t, int64(40), result.MaxResponse1h)
assert.Equal(t, 50.0, result.PacketLoss)
assert.Equal(t, 20.0, result.PacketLoss1h)
}
func TestMonitorManagerGetResultsIncludesLossOnlyHourData(t *testing.T) {
now := time.Now().UTC()
task := newMonitorTask(monitor.Config{ID: "monitor-1"})
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-30 * time.Second)})
task.history.addSampleLocked(monitorSample{responseUs: -1, timestamp: now.Add(-10 * time.Second)})
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{"icmp:example.com": task}
results := pm.GetResults(uint16(time.Minute / time.Millisecond))
result, ok := results["monitor-1"]
require.True(t, ok)
assert.Equal(t, int64(0), result.AvgResponse)
assert.Equal(t, int64(0), result.AvgResponse1h)
assert.Equal(t, int64(0), result.MinResponse)
assert.Equal(t, int64(0), result.MinResponse1h)
assert.Equal(t, int64(0), result.MaxResponse)
assert.Equal(t, int64(0), result.MaxResponse1h)
assert.Equal(t, 100.0, result.PacketLoss)
assert.Equal(t, 100.0, result.PacketLoss1h)
}
func TestMonitorConfigResultKeyUsesSyncedID(t *testing.T) {
cfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
assert.Equal(t, "monitor-1", cfg.ID)
}
func TestMonitorManagerSyncMonitorsSkipsConfigsWithoutStableID(t *testing.T) {
validCfg := monitor.Config{ID: "monitor-1", Target: "ignored", Protocol: "noop", Interval: 10}
invalidCfg := monitor.Config{Target: "ignored", Protocol: "noop", Interval: 10}
pm := newMonitorManager()
pm.SyncMonitors([]monitor.Config{validCfg, invalidCfg})
defer pm.Stop()
_, validExists := pm.monitors[validCfg.ID]
_, invalidExists := pm.monitors[invalidCfg.ID]
assert.True(t, validExists)
assert.False(t, invalidExists)
}
func TestMonitorManagerSyncMonitorsStopsRemovedTasksButKeepsExisting(t *testing.T) {
keepCfg := monitor.Config{ID: "monitor-1", Target: "ignored", Protocol: "noop", Interval: 10}
removeCfg := monitor.Config{ID: "monitor-2", Target: "ignored", Protocol: "noop", Interval: 10}
keptTask := newMonitorTask(keepCfg)
removedTask := newMonitorTask(removeCfg)
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{
keepCfg.ID: keptTask,
removeCfg.ID: removedTask,
}
pm.SyncMonitors([]monitor.Config{keepCfg})
assert.Same(t, keptTask, pm.monitors[keepCfg.ID])
_, exists := pm.monitors[removeCfg.ID]
assert.False(t, exists)
select {
case <-removedTask.ctx.Done():
default:
t.Fatal("expected removed monitor task to be cancelled")
}
select {
case <-keptTask.ctx.Done():
t.Fatal("expected existing monitor task to remain active")
default:
}
}
func TestMonitorManagerSyncMonitorsRestartsChangedConfig(t *testing.T) {
originalCfg := monitor.Config{ID: "monitor-1", Target: "ignored-a", Protocol: "noop", Interval: 10}
updatedCfg := monitor.Config{ID: "monitor-1", Target: "ignored-b", Protocol: "noop", Interval: 10}
originalTask := newMonitorTask(originalCfg)
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{
originalCfg.ID: originalTask,
}
pm.SyncMonitors([]monitor.Config{updatedCfg})
defer pm.Stop()
restartedTask := pm.monitors[updatedCfg.ID]
assert.NotSame(t, originalTask, restartedTask)
assert.Equal(t, updatedCfg, restartedTask.config)
select {
case <-originalTask.ctx.Done():
default:
t.Fatal("expected changed monitor task to be cancelled")
}
}
func TestMonitorManagerApplySyncUpsertRunsImmediatelyAndReturnsResult(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusNoContent)
}))
defer server.Close()
pm := &MonitorManager{
monitors: make(map[string]*monitorTask),
probe: networkMonitorProbe(server.Client()),
}
resp, err := pm.HandleSyncRequest(monitor.SyncRequest{
Action: monitor.SyncActionUpsert,
Config: monitor.Config{ID: "monitor-1", Target: server.URL, Protocol: "http", Interval: 10},
RunNow: true,
})
defer pm.Stop()
require.NoError(t, err)
assert.GreaterOrEqual(t, resp.Result.AvgResponse, int64(0))
assert.Equal(t, 0.0, resp.Result.PacketLoss)
assert.Equal(t, 0.0, resp.Result.PacketLoss1h)
task := pm.monitors["monitor-1"]
require.NotNil(t, task)
task.history.mu.Lock()
defer task.history.mu.Unlock()
require.Len(t, task.history.samples, 1)
}
func TestMonitorManagerUpsertMonitorKeepsHistoryWhenOnlyIntervalChanges(t *testing.T) {
originalCfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
updatedCfg := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 30}
now := time.Now().UTC()
existingTask := newMonitorTask(originalCfg)
existingTask.history.addSampleLocked(monitorSample{responseUs: 12, timestamp: now.Add(-50 * time.Minute)})
existingTask.history.addSampleLocked(monitorSample{responseUs: 24, timestamp: now.Add(-30 * time.Second)})
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{originalCfg.ID: existingTask}
result, err := pm.UpsertMonitor(updatedCfg, false)
defer pm.Stop()
require.NoError(t, err)
assert.Nil(t, result)
updatedTask := pm.monitors[updatedCfg.ID]
require.NotNil(t, updatedTask)
assert.NotSame(t, existingTask, updatedTask)
assert.Equal(t, updatedCfg, updatedTask.config)
updatedTask.history.mu.Lock()
defer updatedTask.history.mu.Unlock()
require.Len(t, updatedTask.history.samples, 1)
assert.Equal(t, int64(24), updatedTask.history.samples[0].responseUs)
agg := updatedTask.history.aggregateLocked(time.Hour, now)
require.True(t, agg.hasData())
assert.Equal(t, int64(2), agg.totalCount)
assert.Equal(t, int64(2), agg.successCount)
assert.Equal(t, int64(18), agg.avgResponse())
select {
case <-existingTask.ctx.Done():
default:
t.Fatal("expected original monitor task to be cancelled")
}
}
func TestMonitorManagerApplySyncDeleteRemovesTask(t *testing.T) {
config := monitor.Config{ID: "monitor-1", Target: "1.1.1.1", Protocol: "icmp", Interval: 10}
task := newMonitorTask(config)
pm := newMonitorManager()
pm.monitors = map[string]*monitorTask{config.ID: task}
_, err := pm.HandleSyncRequest(monitor.SyncRequest{
Action: monitor.SyncActionDelete,
Config: monitor.Config{ID: config.ID},
})
require.NoError(t, err)
_, exists := pm.monitors[config.ID]
assert.False(t, exists)
select {
case <-task.ctx.Done():
default:
t.Fatal("expected deleted monitor task to be cancelled")
}
}
func TestMonitorManagerGetRandomDelay(t *testing.T) {
for i := 1000; i < 360_000; i += 1000 {
delay := getStagger(int64(i))
assert.GreaterOrEqual(t, delay, time.Duration(i/2)*time.Millisecond)
assert.LessOrEqual(t, delay, time.Duration(i)*time.Millisecond)
}
}
func TestMonitorHTTP(t *testing.T) {
t.Run("success", func(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
assert.Equal(t, "Beszel-Agent/"+beszel.Version+" (+https://beszel.dev)", r.Header.Get("User-Agent"))
w.WriteHeader(http.StatusNoContent)
}))
defer server.Close()
responseUs, err := monitorHTTP(context.Background(), server.Client(), server.URL)
require.NoError(t, err)
assert.GreaterOrEqual(t, responseUs, int64(0))
})
t.Run("server error", func(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
http.Error(w, "boom", http.StatusInternalServerError)
}))
defer server.Close()
responseUs, err := monitorHTTP(context.Background(), server.Client(), server.URL)
assert.Equal(t, int64(-1), responseUs)
require.Error(t, err)
})
}
func TestMonitorTCP(t *testing.T) {
t.Run("success", func(t *testing.T) {
listener, err := net.Listen("tcp", "127.0.0.1:0")
require.NoError(t, err)
defer listener.Close()
accepted := make(chan struct{})
go func() {
defer close(accepted)
conn, err := listener.Accept()
if err == nil {
_ = conn.Close()
}
}()
port := uint16(listener.Addr().(*net.TCPAddr).Port)
responseUs, err := monitorTCP(context.Background(), "127.0.0.1", port)
require.NoError(t, err)
assert.GreaterOrEqual(t, responseUs, int64(0))
<-accepted
})
t.Run("connection failure", func(t *testing.T) {
listener, err := net.Listen("tcp", "127.0.0.1:0")
require.NoError(t, err)
port := uint16(listener.Addr().(*net.TCPAddr).Port)
require.NoError(t, listener.Close())
responseUs, err := monitorTCP(context.Background(), "127.0.0.1", port)
assert.Equal(t, int64(-1), responseUs)
require.Error(t, err)
})
}
func TestMonitorTCPAddressFallback(t *testing.T) {
for _, tc := range []struct {
name string
ips []string
loss bool
}{
{"first address fails", []string{"127.0.0.2", "127.0.0.1"}, false},
{"first address succeeds", []string{"127.0.0.1", "127.0.0.2"}, false},
{"all addresses fail", []string{"127.0.0.2", "127.0.0.3"}, true},
} {
t.Run(tc.name, func(t *testing.T) {
listener, err := net.Listen("tcp4", "127.0.0.1:0")
require.NoError(t, err)
defer listener.Close()
original := net.DefaultResolver
net.DefaultResolver = tcpMonitorTestResolver(tc.ips)
defer func() { net.DefaultResolver = original }()
// Verify the resolver preserves the intended order, so success cannot
// accidentally bypass the failed first address in the regression case.
ips, err := net.DefaultResolver.LookupHost(t.Context(), "tcp-monitor.invalid.")
require.NoError(t, err)
require.Equal(t, tc.ips, ips)
responseUs, err := monitorTCP(t.Context(), "tcp-monitor.invalid.", uint16(listener.Addr().(*net.TCPAddr).Port))
if tc.loss {
require.Error(t, err)
assert.Equal(t, int64(-1), responseUs)
} else {
require.NoError(t, err)
assert.GreaterOrEqual(t, responseUs, int64(0))
}
})
}
}
// tcpMonitorTestResolver supplies multiple A records without external DNS.
func tcpMonitorTestResolver(ips []string) *net.Resolver {
return &net.Resolver{PreferGo: true, Dial: func(ctx context.Context, network, address string) (net.Conn, error) {
client, server := net.Pipe()
go func() {
defer server.Close()
// net.Resolver uses TCP framing when its connection is not a PacketConn.
var size uint16
if err := binary.Read(server, binary.BigEndian, &size); err != nil {
return
}
packet := make([]byte, size)
if _, err := io.ReadFull(server, packet); err != nil {
return
}
var msg dnsmessage.Message
if err := msg.Unpack(packet); err != nil {
return
}
msg.Header.Response = true
msg.Header.RecursionAvailable = true
for _, question := range msg.Questions {
if question.Type != dnsmessage.TypeA {
continue
}
for _, ip := range ips {
msg.Answers = append(msg.Answers, dnsmessage.Resource{
Header: dnsmessage.ResourceHeader{Name: question.Name, Type: dnsmessage.TypeA, Class: dnsmessage.ClassINET},
Body: &dnsmessage.AResource{A: [4]byte(net.ParseIP(ip).To4())},
})
}
}
packet, err := msg.Pack()
if err != nil {
return
}
response := binary.BigEndian.AppendUint16(nil, uint16(len(packet)))
_, _ = server.Write(append(response, packet...))
}()
return client, nil
}}
}
func TestMonitorDNS(t *testing.T) {
t.Run("success", func(t *testing.T) {
responseUs, err := monitorDNS(context.Background(), "localhost")
require.NoError(t, err)
assert.GreaterOrEqual(t, responseUs, int64(0))
})
t.Run("lookup failure", func(t *testing.T) {
responseUs, err := monitorDNS(context.Background(), "")
assert.Equal(t, int64(-1), responseUs)
require.Error(t, err)
})
}
func TestMonitorManagerCancelsActiveProbe(t *testing.T) {
for _, action := range []string{"stop", "delete", "upsert", "sync replace", "sync remove"} {
t.Run(action, func(t *testing.T) {
started := make(chan struct{})
canceled := make(chan struct{})
release := make(chan struct{})
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
close(started)
select {
case <-r.Context().Done():
close(canceled)
case <-release:
}
}))
defer server.Close()
defer close(release)
pm := newMonitorManager()
defer pm.Stop()
cfg := monitor.Config{ID: "test", Protocol: "http", Target: server.URL, Interval: 3600}
task := newMonitorTask(cfg)
// Seed history to ensure a canceled RunNow does not return an old result.
task.history.addSampleLocked(monitorSample{responseUs: 123, timestamp: time.Now()})
pm.monitors[cfg.ID] = task
done := make(chan *monitor.Result, 1)
go func() {
result, _ := pm.UpsertMonitor(cfg, true)
done <- result
}()
select {
case <-started:
case <-time.After(time.Second):
t.Fatal("probe did not start")
}
updated := cfg
updated.Interval--
switch action {
case "stop":
pm.Stop()
case "delete":
pm.DeleteMonitor(cfg.ID)
case "upsert":
_, err := pm.UpsertMonitor(updated, false)
require.NoError(t, err)
case "sync replace":
pm.SyncMonitors([]monitor.Config{updated})
case "sync remove":
pm.SyncMonitors(nil)
}
select {
case <-canceled:
case <-time.After(time.Second):
t.Fatal("active HTTP request was not canceled")
}
select {
case result := <-done:
assert.Nil(t, result)
case <-time.After(time.Second):
t.Fatal("RunNow did not return after cancellation")
}
task.history.mu.Lock()
assert.Len(t, task.history.samples, 1, "cancellation must not record packet loss")
task.history.mu.Unlock()
})
}
}
func TestMonitorResolutionCancellation(t *testing.T) {
for _, protocol := range []string{"tcp", "dns", "icmp"} {
t.Run(protocol, func(t *testing.T) {
started := make(chan struct{}, 1)
original := net.DefaultResolver
net.DefaultResolver = &net.Resolver{PreferGo: true, Dial: func(ctx context.Context, network, address string) (net.Conn, error) {
select {
case started <- struct{}{}:
default:
}
<-ctx.Done()
return nil, ctx.Err()
}}
defer func() { net.DefaultResolver = original }()
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
done := make(chan error, 1)
go func() {
var err error
switch protocol {
case "tcp":
_, err = monitorTCP(ctx, "monitor-cancellation.invalid.", 80)
case "dns":
_, err = monitorDNS(ctx, "monitor-cancellation.invalid.")
case "icmp":
_, err = monitorICMP(ctx, "monitor-cancellation.invalid.")
}
done <- err
}()
select {
case <-started:
case <-time.After(time.Second):
t.Fatal("lookup did not start")
}
cancel()
select {
case err := <-done:
require.Error(t, err)
case <-time.After(time.Second):
t.Fatal("lookup did not cancel")
}
})
}
}
func TestMonitorProbeTimeoutRecordsLoss(t *testing.T) {
release := make(chan struct{})
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
select {
case <-r.Context().Done():
case <-release:
}
}))
defer server.Close()
defer close(release)
pm := newMonitorManager()
pm.probe = networkMonitorProbe(&http.Client{Timeout: 20 * time.Millisecond})
task := newMonitorTask(monitor.Config{ID: "timeout", Protocol: "http", Target: server.URL})
defer task.cancel()
result := task.runProbe(pm.probe)
require.NotNil(t, result)
assert.Equal(t, 100.0, result.PacketLoss)
assert.Equal(t, 100.0, result.PacketLoss1h)
require.Len(t, task.history.samples, 1)
assert.Equal(t, int64(-1), task.history.samples[0].responseUs)
assert.NoError(t, task.ctx.Err(), "a probe timeout must not cancel the task")
}

View File

@@ -931,6 +931,9 @@ func (sm *SmartManager) parseSmartForSata(output []byte, deviceType string) (boo
if parsed, ok := smart.ParseSmartRawValueString(attr.Raw.String); ok {
rawValue = parsed
}
if smartData.SmartStatus == "PASSED" && rawValue > 0 && (attr.ID == 5 || attr.ID == 197 || attr.ID == 198) {
smartData.SmartStatus = "WARNING"
}
smartAttr := &smart.SmartAttribute{
ID: attr.ID,
Name: attr.Name,

View File

@@ -7,6 +7,7 @@ import (
"fmt"
"os"
"path/filepath"
"strconv"
"testing"
"github.com/henrygd/beszel/internal/entities/smart"
@@ -89,6 +90,27 @@ func TestParseSmartForSata(t *testing.T) {
}
}
func TestParseSmartForSataWarnsForCriticalAttributes(t *testing.T) {
for _, attrID := range []int{5, 197, 198} {
t.Run("attribute "+strconv.Itoa(attrID), func(t *testing.T) {
jsonPayload := []byte(fmt.Sprintf(`{
"smartctl": {"exit_status": 0},
"device": {"name": "/dev/sda", "type": "sat"},
"model_name": "Example",
"serial_number": "WARNING%d",
"smart_status": {"passed": true},
"temperature": {"current": 30},
"ata_smart_attributes": {"table": [{"id": %d, "raw": {"value": 1, "string": "1"}}]}
}`, attrID, attrID))
sm := &SmartManager{SmartDataMap: make(map[string]*smart.SmartData)}
hasData, _ := sm.parseSmartForSata(jsonPayload, "")
require.True(t, hasData)
assert.Equal(t, "WARNING", sm.SmartDataMap[fmt.Sprintf("WARNING%d", attrID)].SmartStatus)
})
}
}
func TestParseSmartForSataPreservesFailedAndUnknownStatus(t *testing.T) {
for _, test := range []struct {
name string

View File

@@ -80,7 +80,7 @@ func newZfsBackend() *poolBackend {
return &poolBackend{
name: "zfs",
poolStatsFn: optionalPoolSource(zfs.PoolStats),
datasetsFn: optionalPoolSource(zfs.Datasets),
datasetsFn: zfs.Datasets,
kernelStatsFn: optionalPoolSource(zfs.PoolKernelStats),
poolStatusesFn: optionalPoolSource(zfs.PoolStatuses),
}

View File

@@ -323,21 +323,6 @@ func TestDatasetUsageRefreshOnErrorKeepsPrevious(t *testing.T) {
assert.Len(t, usage, 1, "previous usage should be retained on error")
}
func TestDatasetUsageClearsAbsentBackend(t *testing.T) {
b := newZfsBackend()
b.datasetUsage = map[string]zfsDatasetUsage{"/tank": {used: 1, avail: 1}}
b.datasetsFn = optionalPoolSource(func() ([]zfs.Dataset, error) {
return nil, zfs.ErrNoZfs
})
datasets, err := b.datasets()
require.NoError(t, err, "an absent backend must not produce an error to log")
assert.Empty(t, datasets)
b.refreshDatasetUsage()
assert.Empty(t, b.datasetUsage)
assert.False(t, b.lastUsageRefresh.IsZero())
}
func TestGetDetailForceRefresh(t *testing.T) {
zm := &StoragePoolManager{detailInterval: time.Hour, backends: []*poolBackend{{name: "zfs"}}}
poolCalls := 0

View File

@@ -70,9 +70,6 @@ type Dataset struct {
// PoolStats returns capacity and health for all pools on the system using
// `zpool list`. Frequent health and I/O sampling uses PoolKernelStats instead.
func PoolStats() ([]PoolStat, error) {
if err := checkZfsDevice(); err != nil {
return nil, err
}
out, err := commandOutput("zpool", "list", "-Hp", "-o", "name,size,alloc,free,health")
if err != nil {
var exitErr *exec.ExitError
@@ -87,9 +84,6 @@ func PoolStats() ([]PoolStat, error) {
// Datasets returns all datasets on the system with usage and mountpoint
// information using `zfs list` (recursive by default).
func Datasets() ([]Dataset, error) {
if err := checkZfsDevice(); err != nil {
return nil, err
}
out, err := commandOutput("zfs", "list", "-Hp", "-o", "name,used,avail,mountpoint")
if err != nil {
return nil, fmt.Errorf("zfs list: %w", err)

View File

@@ -13,10 +13,7 @@ import (
"strings"
)
var (
procZfsPath = "/proc/spl/kstat/zfs"
devZfsPath = "/dev/zfs"
)
var procZfsPath = "/proc/spl/kstat/zfs"
func ARCSize() (uint64, error) {
file, err := os.Open(filepath.Join(procZfsPath, "arcstats"))
@@ -43,19 +40,6 @@ func ARCSize() (uint64, error) {
return 0, fmt.Errorf("size field not found in arcstats")
}
// checkZfsDevice lets containers without /dev/zfs fail fast instead of
// waiting for ZFS utility commands to time out.
func checkZfsDevice() error {
_, err := os.Stat(devZfsPath)
if err != nil {
if errors.Is(err, os.ErrNotExist) {
return ErrNoZfs
}
return err
}
return nil
}
// PoolKernelStats reads pool state and cumulative I/O counters directly from
// procfs. These kstats are the same interfaces used by node_exporter's Linux
// ZFS collector and avoid keeping a `zpool iostat` subprocess alive.

View File

@@ -88,66 +88,3 @@ func TestReadObjsetIORequiresAllCounters(t *testing.T) {
_, _, err := readObjsetIO(path)
require.Error(t, err)
}
func TestCollectorsSkipCommandsWhenDevZfsMissing(t *testing.T) {
root := t.TempDir()
oldDevZfsPath := devZfsPath
devZfsPath = filepath.Join(root, "missing")
t.Cleanup(func() { devZfsPath = oldDevZfsPath })
oldCommandOutput := commandOutput
commandOutput = func(name string, args ...string) ([]byte, error) {
t.Fatalf("unexpected %s call with %v", name, args)
return nil, nil
}
t.Cleanup(func() { commandOutput = oldCommandOutput })
_, err := PoolStats()
assert.ErrorIs(t, err, ErrNoZfs)
_, err = Datasets()
assert.ErrorIs(t, err, ErrNoZfs)
}
func TestDatasetsDelegatesWhenDevZfsPresent(t *testing.T) {
oldDevZfsPath := devZfsPath
devZfsPath = filepath.Join(t.TempDir(), "zfs")
require.NoError(t, os.WriteFile(devZfsPath, nil, 0o644))
t.Cleanup(func() { devZfsPath = oldDevZfsPath })
oldCommandOutput := commandOutput
commandOutput = func(name string, args ...string) ([]byte, error) {
assert.Equal(t, "zfs", name)
assert.Equal(t, []string{"list", "-Hp", "-o", "name,used,avail,mountpoint"}, args)
return []byte("tank\t50\t50\t/tank\n"), nil
}
t.Cleanup(func() { commandOutput = oldCommandOutput })
datasets, err := Datasets()
require.NoError(t, err)
assert.Equal(t, []Dataset{{Name: "tank", Used: 50, Avail: 50, Mountpoint: "/tank"}}, datasets)
}
func TestPoolStatsDelegatesToZpoolWhenDevZfsPresent(t *testing.T) {
root := t.TempDir()
devFile := filepath.Join(root, "zfs")
require.NoError(t, os.WriteFile(devFile, []byte(""), 0o644))
oldDevZfsPath := devZfsPath
devZfsPath = devFile
t.Cleanup(func() { devZfsPath = oldDevZfsPath })
oldCommandOutput := commandOutput
called := false
commandOutput = func(name string, args ...string) ([]byte, error) {
called = true
assert.Equal(t, "zpool", name)
assert.Equal(t, []string{"list", "-Hp", "-o", "name,size,alloc,free,health"}, args)
return []byte("tank\t100\t50\t50\tONLINE\n"), nil
}
t.Cleanup(func() { commandOutput = oldCommandOutput })
pools, err := PoolStats()
require.NoError(t, err)
assert.True(t, called)
assert.Equal(t, []PoolStat{{Name: "tank", Size: 100, Alloc: 50, Free: 50, Health: "ONLINE"}}, pools)
}

View File

@@ -1,9 +0,0 @@
//go:build !linux
package zfs
// The /dev/zfs probe is Linux-specific. Other platforms detect availability
// through the ZFS utilities themselves.
func checkZfsDevice() error {
return nil
}

View File

@@ -1,33 +0,0 @@
//go:build testing && !linux
package zfs
import (
"testing"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestCollectorsUseUtilitiesOnNonLinux(t *testing.T) {
oldCommandOutput := commandOutput
commandOutput = func(name string, args ...string) ([]byte, error) {
switch name {
case "zpool":
return []byte("tank\t100\t50\t50\tONLINE\n"), nil
case "zfs":
return []byte("tank\t50\t50\t/tank\n"), nil
default:
t.Fatalf("unexpected command %s", name)
return nil, nil
}
}
t.Cleanup(func() { commandOutput = oldCommandOutput })
pools, err := PoolStats()
require.NoError(t, err)
assert.Equal(t, []PoolStat{{Name: "tank", Size: 100, Alloc: 50, Free: 50, Health: "ONLINE"}}, pools)
datasets, err := Datasets()
require.NoError(t, err)
assert.Equal(t, []Dataset{{Name: "tank", Used: 50, Avail: 50, Mountpoint: "/tank"}}, datasets)
}

View File

@@ -6,7 +6,7 @@ import "github.com/blang/semver"
const (
// Version is the current version of the application.
Version = "0.20.0"
Version = "0.19.0"
// AppName is the name of the application.
AppName = "beszel"
)
@@ -19,6 +19,3 @@ var MinVersionAgentResponse = semver.MustParse("0.13.0")
// MinVersionZfsData is the minimum agent version that supports ZFS detail requests.
var MinVersionZfsData = semver.MustParse("0.18.9")
// MinVersionNetworkMonitors is the minimum agent version that supports network monitor sync.
var MinVersionNetworkMonitors = semver.MustParse("0.20.0")

40
go.mod
View File

@@ -5,25 +5,23 @@ go 1.27.1
require (
github.com/blang/semver v3.5.1+incompatible
github.com/coreos/go-systemd/v22 v22.7.0
github.com/distribution/reference v0.6.0
github.com/ebitengine/purego v0.11.0
github.com/fxamacker/cbor/v2 v2.9.4
github.com/fxamacker/cbor/v2 v2.9.3
github.com/gliderlabs/ssh v0.3.8
github.com/lxzan/gws v1.10.2
github.com/nicholas-fedor/shoutrrr v0.21.0
github.com/opencontainers/go-digest v1.0.0
github.com/google/uuid v1.6.0
github.com/lxzan/gws v1.10.1
github.com/nicholas-fedor/shoutrrr v0.20.0
github.com/pocketbase/dbx v1.12.0
github.com/pocketbase/pocketbase v0.40.4
github.com/pocketbase/pocketbase v0.40.2
github.com/shirou/gopsutil/v4 v4.26.8
github.com/spf13/cast v1.10.0
github.com/spf13/cobra v1.10.2
github.com/spf13/pflag v1.0.10
github.com/stretchr/testify v1.12.1
golang.org/x/crypto v0.57.0
golang.org/x/exp v0.0.0-20260908205506-85c1c2202aba
golang.org/x/net v0.59.0
golang.org/x/oauth2 v0.37.0
golang.org/x/sys v0.48.0
golang.org/x/crypto v0.56.0
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa
golang.org/x/net v0.58.0
golang.org/x/sys v0.47.0
gopkg.in/yaml.v3 v3.0.1
howett.net/plist v1.0.1
)
@@ -33,7 +31,7 @@ require (
github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 // indirect
github.com/disintegration/imaging v1.6.2 // indirect
github.com/domodwyer/mailyak/v3 v3.6.2 // indirect
github.com/dustin/go-humanize v1.1.0 // indirect
github.com/dustin/go-humanize v1.0.1 // indirect
github.com/eclipse/paho.golang v0.23.0 // indirect
github.com/fatih/color v1.19.0 // indirect
github.com/fsnotify/fsnotify v1.10.1 // indirect
@@ -43,7 +41,6 @@ require (
github.com/go-sql-driver/mysql v1.9.1 // indirect
github.com/godbus/dbus/v5 v5.2.2 // indirect
github.com/golang-jwt/jwt/v5 v5.3.1 // indirect
github.com/google/uuid v1.6.0 // indirect
github.com/gorilla/websocket v1.5.3 // indirect
github.com/inconshreveable/mousetrap v1.1.0 // indirect
github.com/klauspost/compress v1.20.0 // indirect
@@ -52,23 +49,18 @@ require (
github.com/mattn/go-isatty v0.0.24 // indirect
github.com/ncruces/go-strftime v1.0.0 // indirect
github.com/pocketbase/ozzo-validation/v4 v4.3.0 // indirect
github.com/power-devops/perfstat v0.0.0-20260916203055-22a1a467d9f0 // indirect
github.com/power-devops/perfstat v0.0.0-20260805114148-88456608a4f6 // indirect
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
github.com/tklauser/go-sysconf v0.4.0 // indirect
github.com/tklauser/numcpus v0.12.0 // indirect
github.com/x448/float16 v0.8.4 // indirect
github.com/yusufpapurcu/wmi v1.2.4 // indirect
go.yaml.in/yaml/v3 v3.0.5 // indirect
golang.org/x/image v0.46.0 // indirect
golang.org/x/mod v0.41.0 // indirect
golang.org/x/sync v0.23.0 // indirect
golang.org/x/term v0.46.0 // indirect
golang.org/x/text v0.42.0 // indirect
golang.org/x/tools v0.50.0 // indirect
mellium.im/reader v0.1.0 // indirect
mellium.im/sasl v0.3.2 // indirect
mellium.im/xmlstream v0.15.4 // indirect
mellium.im/xmpp v0.23.0 // indirect
golang.org/x/image v0.45.0 // indirect
golang.org/x/oauth2 v0.36.0 // indirect
golang.org/x/sync v0.22.0 // indirect
golang.org/x/term v0.45.0 // indirect
golang.org/x/text v0.41.0 // indirect
modernc.org/libc v1.74.4 // indirect
modernc.org/mathutil v1.7.1 // indirect
modernc.org/memory v1.12.1 // indirect

84
go.sum
View File

@@ -15,12 +15,10 @@ github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6N
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/disintegration/imaging v1.6.2 h1:w1LecBlG2Lnp8B3jk5zSuNqd7b4DXhcjwek1ei82L+c=
github.com/disintegration/imaging v1.6.2/go.mod h1:44/5580QXChDfwIclfc/PCwrr44amcmDAg8hxG0Ewe4=
github.com/distribution/reference v0.6.0 h1:0IXCQ5g4/QMHHkarYzh5l+u8T3t73zM5QvfrDyIgxBk=
github.com/distribution/reference v0.6.0/go.mod h1:BbU0aIcezP1/5jX/8MP0YiH4SdvB5Y4f/wlDRiLyi3E=
github.com/domodwyer/mailyak/v3 v3.6.2 h1:x3tGMsyFhTCaxp6ycgR0FE/bu5QiNp+hetUuCOBXMn8=
github.com/domodwyer/mailyak/v3 v3.6.2/go.mod h1:lOm/u9CyCVWHeaAmHIdF4RiKVxKUT/H5XX10lIKAL6c=
github.com/dustin/go-humanize v1.1.0 h1:dbKTrvD0klcbBV/h4AWJdMuZogJACoMlvWIWZ5b2xWg=
github.com/dustin/go-humanize v1.1.0/go.mod h1:hc1CvRkJMsgxqjmjMQF3QNRAZBwY8AXBAzKYoSX9sFI=
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
github.com/ebitengine/purego v0.11.0 h1:jhp/D+Nyv7UUW8HAcmcjt2N2rYrYi9m3SL21k0Ua/NI=
github.com/ebitengine/purego v0.11.0/go.mod h1:DCHPP08djqhNSoTfImcnHYQRZmd0qhakvrozqaEYhGQ=
github.com/eclipse/paho.golang v0.23.0 h1:KHgl2wz6EJo7cMBmkuhpt7C576vP+kpPv7jjvSyR6Mk=
@@ -31,8 +29,8 @@ github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHk
github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0=
github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx59Ho=
github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo=
github.com/fxamacker/cbor/v2 v2.9.4 h1:xwjVlxEMR3S605oUlgBjKLTTeGFciYPGYCtF/35LKGo=
github.com/fxamacker/cbor/v2 v2.9.4/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ=
github.com/fxamacker/cbor/v2 v2.9.3 h1:oQBnFATpNdY8gJHTndDDv5Xl4QqNaz51G5LLEPhng3Q=
github.com/fxamacker/cbor/v2 v2.9.3/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ=
github.com/gabriel-vasile/mimetype v1.4.15 h1:05iP/CYtZ/w455R/KZM6rZ5ieAdh99UPtd+d3YzLmaI=
github.com/gabriel-vasile/mimetype v1.4.15/go.mod h1:azpTcoLcDZRNgFou5j+APrqQx9HqVPWa6ijYQIIVswQ=
github.com/ganigeorgiev/fexpr v0.6.0 h1:Fza3O/QMBKEudUvxV862qe6GjxM60GJjjKytdp+VQus=
@@ -77,31 +75,29 @@ github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
github.com/lufia/plan9stats v0.0.0-20260802145828-341c2f0c90b5 h1:eveIIGn4BGM3qknO74omf6HYr30/exH+eVUTuAgwjZ0=
github.com/lufia/plan9stats v0.0.0-20260802145828-341c2f0c90b5/go.mod h1:autxFIvghDt3jPTLoqZ9OZ7s9qTGNAWmYCjVFWPX/zg=
github.com/lxzan/gws v1.10.2 h1:htReTvcY89iMk1ScVtUbk6J96kIZWaafj6r/lasK/NA=
github.com/lxzan/gws v1.10.2/go.mod h1:gXHSCPmTGryWJ4icuqy8Yho32E4YIMHH0fkDRYJRbdc=
github.com/lxzan/gws v1.10.1 h1:1xG+tDOV0lgDeVPf0wNT74u3cn0K3LpcavRrTPTrMwQ=
github.com/lxzan/gws v1.10.1/go.mod h1:gXHSCPmTGryWJ4icuqy8Yho32E4YIMHH0fkDRYJRbdc=
github.com/mattn/go-colorable v0.1.15 h1:+u9SLTRGnXv73cEsnsmoZBom+dMU88B2M0aDcWy0/jY=
github.com/mattn/go-colorable v0.1.15/go.mod h1:6LmQG8QLFO4G5z1gPvYEzlUgJ2wF+stgPZH1UqBm1s8=
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
github.com/nicholas-fedor/shoutrrr v0.21.0 h1:as/mEwdaZMijCVu0FkTUEXashhvC3Y7C5g9dsXMcmQc=
github.com/nicholas-fedor/shoutrrr v0.21.0/go.mod h1:dgg4kJv9K0tLXBH/1TXiSibNbM2hcd4SK6xb0sglyU4=
github.com/onsi/ginkgo/v2 v2.32.2 h1:2o6vyFvR6snrJWgRVztC+OwuqqPEMI1UzYl2s2iU7Cg=
github.com/onsi/ginkgo/v2 v2.32.2/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
github.com/nicholas-fedor/shoutrrr v0.20.0 h1:hMAxIYlfAeZ1FcTDgU0kUOvVXUsOirWo8IWlnzGLkac=
github.com/nicholas-fedor/shoutrrr v0.20.0/go.mod h1:hgde37yNWCXh8+N6WemyDRMNYLOFTf326GsBx8Z7CFA=
github.com/onsi/ginkgo/v2 v2.32.1 h1:6tlvcDm/3sE8lGJbZ4+d4mO3RLy24/tQWOFzVSQNIfw=
github.com/onsi/ginkgo/v2 v2.32.1/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
github.com/onsi/gomega v1.43.0 h1:VlG/1FxqNxhSO+lq/OHBNaaqwiBK/mO8JbVkX9Y+FeU=
github.com/onsi/gomega v1.43.0/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg=
github.com/opencontainers/go-digest v1.0.0 h1:apOUWs51W5PlhuyGyz9FCeeBIOUDA/6nW8Oi/yOhh5U=
github.com/opencontainers/go-digest v1.0.0/go.mod h1:0JzlMkj0TRzQZfJkVvzbP0HBR3IKzErnv2BNG4W4MAM=
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/pocketbase/dbx v1.12.0 h1:/oLErM+A0b4xI0PWTGPqSDVjzix48PqI/bng2l0PzoA=
github.com/pocketbase/dbx v1.12.0/go.mod h1:xXRCIAKTHMgUCyCKZm55pUOdvFziJjQfXaWKhu2vhMs=
github.com/pocketbase/ozzo-validation/v4 v4.3.0 h1:uKBDVma7bZqgR2a6AwE+k9hkuDFfiZMpBHQdZ1z3iQs=
github.com/pocketbase/ozzo-validation/v4 v4.3.0/go.mod h1:6XNjSTw/Jb2F8LOkKO3oyzIWExbrGiYoS4uVxVwz90g=
github.com/pocketbase/pocketbase v0.40.4 h1:0SvSUreR3NhUMCs9LchE59oEG53efZ3cKiMGyAGBN9U=
github.com/pocketbase/pocketbase v0.40.4/go.mod h1:2mU+80FLiY1fb13WZRg8Xx/lKg4nTjgKVJMigyRH6k0=
github.com/power-devops/perfstat v0.0.0-20260916203055-22a1a467d9f0 h1:XA01Vk/wv9YikCi1V51yRzIHPMT5of9+cMpZoZvDn/M=
github.com/power-devops/perfstat v0.0.0-20260916203055-22a1a467d9f0/go.mod h1:OmDBASR4679mdNQnz2pUhc2G8CO2JrUAVFDRBDP/hJE=
github.com/pocketbase/pocketbase v0.40.2 h1:7gTqvt3bmilkphyZZ1QNhX19g3BXHqT7ynDyU81RVT4=
github.com/pocketbase/pocketbase v0.40.2/go.mod h1:jc3YuyToy+ZXM4CeO7uSCN/htgR8yv+tjSE3eJZ8eh8=
github.com/power-devops/perfstat v0.0.0-20260805114148-88456608a4f6 h1:jL3a8soXdzuTCcRnKhOmtcsVOObdDTFf4O2B403HPRU=
github.com/power-devops/perfstat v0.0.0-20260805114148-88456608a4f6/go.mod h1:OmDBASR4679mdNQnz2pUhc2G8CO2JrUAVFDRBDP/hJE=
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
github.com/rogpeppe/go-internal v1.9.0 h1:73kH8U+JUqXU8lRuOHeVHaa/SZPifC7BkcraZVejAe8=
@@ -136,37 +132,37 @@ go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
golang.org/x/crypto v0.57.0 h1:3ZVCjf8Ggz7zneR/EHRVx68Ctf+2pmIMP2UFhh9cC6M=
golang.org/x/crypto v0.57.0/go.mod h1:Fdz0i5U6CoizGwLda9DttjSk6qlZo25zYNtR+ycvuZA=
golang.org/x/exp v0.0.0-20260908205506-85c1c2202aba h1:Ck8QetSgk912qxWLMCKxd0in+aiyBQyDSMae6e/xmpU=
golang.org/x/exp v0.0.0-20260908205506-85c1c2202aba/go.mod h1:50RgIsmK7OwqzTTeqcSXQW8SswW0o8fRcDxmqGluJ8E=
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa h1:QSyA8ishJCyT21kER9KwNt0b7BM3iRK4x9QXhjN5Fdk=
golang.org/x/exp v0.0.0-20260824195058-e88cd73687aa/go.mod h1:zeBbvyFKDaLwa7CH/zI8KXt7gTl14SF7sO08Pl5jBCM=
golang.org/x/image v0.0.0-20191009234506-e7c1f5e7dbb8/go.mod h1:FeLwcggjj3mMvU+oOTbSwawSJRM1uh48EjtB4UJZlP0=
golang.org/x/image v0.46.0 h1:b1+oYj0Jbp6K5MDT4i4/eZpYlk3V8SJhhDKh6LBHAyQ=
golang.org/x/image v0.46.0/go.mod h1:3B3W05VGVQyuXucLINLjXKrqISASfi4Xj+iCVkLMwew=
golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c=
golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o=
golang.org/x/image v0.45.0 h1:FMb1nTbH5H9vF55SriQHgFw5GnNL9Jg6L25BwXKzhB0=
golang.org/x/image v0.45.0/go.mod h1:n62x/7RqlwXDvGsSU4u6IUTUf6KghUZ9Bt7cG/T9Fx4=
golang.org/x/mod v0.40.0 h1:hUv+3cXcdRHz08UmSiOob7sadHig73uo5bkXxQ/tvUs=
golang.org/x/mod v0.40.0/go.mod h1:0/weTWkPWGBikyTWAX3dkjVztMmBA5hM0DH6BElSupE=
golang.org/x/net v0.0.0-20190603091049-60506f45cf65/go.mod h1:HSz+uSET+XFnRR8LxR5pz3Of3rY3CfYBVs4xY44aLks=
golang.org/x/net v0.59.0 h1:5zfYln+w5XCxwrnMMJPufRgNoXEaGxl0wo5GqPXyues=
golang.org/x/net v0.59.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg=
golang.org/x/oauth2 v0.37.0 h1:JUlcxA8oAtauLfiH8FX2/FkAWHAdi0QtGCGc+hofE98=
golang.org/x/oauth2 v0.37.0/go.mod h1:IxwZNxUULJmpBFf9K/9NTMSIfZZuvuTy1gGxhigP/58=
golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk=
golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0=
golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To=
golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU=
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
golang.org/x/sys v0.0.0-20190916202348-b4ddaad3f8a3/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20201204225414-ed752295db88/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
golang.org/x/term v0.46.0 h1:3+OXuTbaKDgwk8jTi3aSLHRlmWqHEUDUtxnbFigO4YE=
golang.org/x/term v0.46.0/go.mod h1:+K02xbkittuwc0Am4abfA3Fc+XRGXkvBXNO88NCXPoc=
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
golang.org/x/text v0.3.2/go.mod h1:bEr9sfX3Q8Zfm5fL9x+3itogRgK3+ptLWKqgva+5dAk=
golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI=
golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E=
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
golang.org/x/tools v0.50.0 h1:c2ifzfcuY7L90lZ2aKd8S4K2NpASF08SZx9ZuJkHmSU=
golang.org/x/tools v0.50.0/go.mod h1:7ulVMw3831Mwi5EZD6RomGyffr4VFjuNYXf2BbCEAV0=
golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI=
golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo=
google.golang.org/appengine v1.6.5/go.mod h1:8WjMMxjGQR8xUklV/ARdw2HLXBOI7O7uCIDZVag1xfc=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
@@ -176,14 +172,6 @@ gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
howett.net/plist v1.0.1 h1:37GdZ8tP09Q35o9ych3ehygcsL+HqKSwzctveSlarvM=
howett.net/plist v1.0.1/go.mod h1:lqaXoTrLY4hg8tnEzNru53gicrbv7rrk+2xJA/7hw9g=
mellium.im/reader v0.1.0 h1:UUEMev16gdvaxxZC7fC08j7IzuDKh310nB6BlwnxTww=
mellium.im/reader v0.1.0/go.mod h1:F+X5HXpkIfJ9EE1zHQG9lM/hO946iYAmU7xjg5dsQHI=
mellium.im/sasl v0.3.2 h1:PT6Xp7ccn9XaXAnJ03FcEjmAn7kK1x7aoXV6F+Vmrl0=
mellium.im/sasl v0.3.2/go.mod h1:NKXDi1zkr+BlMHLQjY3ofYuU4KSPFxknb8mfEu6SveY=
mellium.im/xmlstream v0.15.4 h1:gLKxcWl4rLMUpKgtzrTBvr4OexPeO/edYus+uK3F6ZI=
mellium.im/xmlstream v0.15.4/go.mod h1:yXaCW2++fmVO4L9piKVkyLDqnCmictVYF7FDQW8prb4=
mellium.im/xmpp v0.23.0 h1:rvKvOvMdIURCLaAWEJN8J0QpO3AJYCDKjxLsqtTPSjY=
mellium.im/xmpp v0.23.0/go.mod h1:GHDKlKKQe0LNmD9YqExyxnFEEBiz84KGqnfiA2VNzb8=
modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI=
modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=

View File

@@ -20,11 +20,10 @@ type hubLike interface {
}
type AlertManager struct {
hub hubLike
stopOnce sync.Once
pendingAlerts sync.Map
alertsCache *AlertsCache
networkMonitors *networkMonitorCache
hub hubLike
stopOnce sync.Once
pendingAlerts sync.Map
alertsCache *AlertsCache
}
type AlertMessageData struct {
@@ -108,9 +107,8 @@ var supportsTitle = map[string]struct{}{
// NewAlertManager creates a new AlertManager instance.
func NewAlertManager(app hubLike) *AlertManager {
am := &AlertManager{
hub: app,
alertsCache: NewAlertsCache(app),
networkMonitors: newNetworkMonitorCache(app),
hub: app,
alertsCache: NewAlertsCache(app),
}
am.bindEvents()
return am
@@ -118,7 +116,6 @@ func NewAlertManager(app hubLike) *AlertManager {
// Bind events to the alerts collection lifecycle
func (am *AlertManager) bindEvents() {
am.bindNetworkMonitorAlertEvents()
am.hub.OnRecordAfterUpdateSuccess("alerts").BindFunc(updateHistoryOnAlertUpdate)
am.hub.OnRecordAfterDeleteSuccess("alerts").BindFunc(resolveHistoryOnAlertDelete)
am.hub.OnRecordAfterUpdateSuccess("smart_devices").BindFunc(am.handleSmartDeviceAlert)

View File

@@ -29,13 +29,6 @@ func UpsertUserAlerts(e *core.RequestEvent) error {
return e.BadRequestError("Bad data", err)
}
if reqData.Name == alertNameNetworkMonitorLoss {
if reqData.Value < 0 || reqData.Value >= 100 {
return e.BadRequestError("Monitor loss threshold must be at least 0 and below 100", nil)
}
reqData.Min = 0
}
alertsCollection, err := e.App.FindCachedCollectionByNameOrId("alerts")
if err != nil {
return err

View File

@@ -11,7 +11,6 @@ import (
"strings"
"sync/atomic"
"testing"
"testing/synctest"
beszelTests "github.com/henrygd/beszel/internal/tests"
pbTests "github.com/pocketbase/pocketbase/tests"
@@ -534,20 +533,6 @@ func TestSendTestNotification(t *testing.T) {
for _, url := range []string{localURL, "smtp://user:pass@127.0.0.1/?fromAddress=sender@example.com&toAddresses=recipient@example.com", "mqtt://127.0.0.1/topic"} {
scenarios = append(scenarios, beszelTests.ApiScenario{
BeforeTestFunc: func(tb testing.TB, _ *pbTests.TestApp, e *core.ServeEvent) {
if !strings.HasPrefix(url, "mqtt://") {
return
}
// Keep the real MQTT rejection path, but advance its library's
// fixed timeout using virtual time instead of waiting 10 seconds.
e.Router.BindFunc(func(re *core.RequestEvent) error {
var err error
synctest.Test(tb.(*testing.T), func(t *testing.T) {
err = re.Next()
})
return err
})
},
Name: "readonly cannot send to " + url,
Method: http.MethodPost,
URL: "/api/beszel/test-notification",

View File

@@ -1,7 +1,6 @@
package alerts
import (
"sync"
"time"
"github.com/pocketbase/dbx"
@@ -19,9 +18,6 @@ type CachedAlertData struct {
Triggered bool
Min uint8
PendingSince time.Time
// Immutable after publication; decoded only when the alert record changes.
MonitorStates map[string]string
MonitorStatesValid bool
// Created types.DateTime
}
@@ -34,18 +30,11 @@ func (a *CachedAlertData) PopulateFromRecord(record *core.Record) {
a.Triggered = record.GetBool("triggered")
a.Min = uint8(record.GetInt("min"))
a.PendingSince = record.GetDateTime("pending_since").Time()
if a.Name == alertNameNetworkMonitorLoss {
var state networkMonitorAlertState
a.MonitorStatesValid = record.UnmarshalJSONField("state", &state) == nil
a.MonitorStates = state.Monitors
}
// a.Created = record.GetDateTime("created")
}
// AlertsCache provides an in-memory cache for system alerts.
type AlertsCache struct {
// Serialize lazy loads with updates so a late load cannot replace newer state.
loadMu sync.Mutex
app core.App
store *store.Store[string, *store.Store[string, CachedAlertData]]
populated bool
@@ -80,8 +69,6 @@ func (c *AlertsCache) bindEvents() *AlertsCache {
// PopulateFromDB clears current entries and loads all alerts from the database into the cache.
func (c *AlertsCache) PopulateFromDB(force bool) error {
c.loadMu.Lock()
defer c.loadMu.Unlock()
if !force && c.populated {
return nil
}
@@ -91,7 +78,7 @@ func (c *AlertsCache) PopulateFromDB(force bool) error {
}
c.store.RemoveAll()
for _, record := range records {
c.update(record)
c.Update(record)
}
c.populated = true
return nil
@@ -99,12 +86,6 @@ func (c *AlertsCache) PopulateFromDB(force bool) error {
// Update adds or updates an alert record in the cache.
func (c *AlertsCache) Update(record *core.Record) {
c.loadMu.Lock()
defer c.loadMu.Unlock()
c.update(record)
}
func (c *AlertsCache) update(record *core.Record) {
systemID := record.GetString("system")
if systemID == "" {
return
@@ -121,8 +102,6 @@ func (c *AlertsCache) update(record *core.Record) {
// Delete removes an alert record from the cache.
func (c *AlertsCache) Delete(record *core.Record) {
c.loadMu.Lock()
defer c.loadMu.Unlock()
systemID := record.GetString("system")
if systemID == "" {
return
@@ -136,23 +115,18 @@ func (c *AlertsCache) Delete(record *core.Record) {
func (c *AlertsCache) GetSystemAlerts(systemID string) []CachedAlertData {
systemStore, ok := c.store.GetOk(systemID)
if !ok {
c.loadMu.Lock()
defer c.loadMu.Unlock()
systemStore, ok = c.store.GetOk(systemID)
if !ok {
// Populate cache for this system
records, err := c.app.FindAllRecords("alerts", dbx.NewExp("system={:system}", dbx.Params{"system": systemID}))
if err != nil {
return nil
}
systemStore = store.New(map[string]CachedAlertData{})
for _, record := range records {
var ca CachedAlertData
ca.PopulateFromRecord(record)
systemStore.Set(record.Id, ca)
}
c.store.Set(systemID, systemStore)
// Populate cache for this system
records, err := c.app.FindAllRecords("alerts", dbx.NewExp("system={:system}", dbx.Params{"system": systemID}))
if err != nil {
return nil
}
systemStore = store.New(map[string]CachedAlertData{})
for _, record := range records {
var ca CachedAlertData
ca.PopulateFromRecord(record)
systemStore.Set(record.Id, ca)
}
c.store.Set(systemID, systemStore)
}
all := systemStore.GetAll()
alerts := make([]CachedAlertData, 0, len(all))

View File

@@ -9,12 +9,6 @@ import (
// On triggered alert record delete, set matching alert history record to resolved
func resolveHistoryOnAlertDelete(e *core.RecordEvent) error {
if e.Record.GetString("name") == alertNameNetworkMonitorLoss {
if err := resolveNetworkMonitorHistory(e.App, e.Record.Id); err != nil {
return err
}
return e.Next()
}
if !e.Record.GetBool("triggered") {
return e.Next()
}
@@ -24,10 +18,6 @@ func resolveHistoryOnAlertDelete(e *core.RecordEvent) error {
// On alert record update, update alert history record
func updateHistoryOnAlertUpdate(e *core.RecordEvent) error {
// Network monitor incidents have separate history entries per monitor.
if e.Record.GetString("name") == alertNameNetworkMonitorLoss {
return e.Next()
}
original := e.Record.Original()
new := e.Record

View File

@@ -1,269 +0,0 @@
package alerts
import (
"database/sql"
"errors"
"fmt"
"math"
"net"
"strconv"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
const alertNameNetworkMonitorLoss = "NetworkMonitorLoss"
// networkMonitorAlertState is this alert type's persisted runtime state.
// Monitor IDs map to their open history entries independently of history retention.
type networkMonitorAlertState struct {
Monitors map[string]string `json:"monitors"`
}
func (am *AlertManager) bindNetworkMonitorAlertEvents() {
// Hidden fields are still writable through the record API unless protected.
protectState := func(e *core.RecordRequestEvent) error {
e.Record.Set("state", e.Record.Original().Get("state"))
oldName, newName := e.Record.Original().GetString("name"), e.Record.GetString("name")
if oldName != "" && (oldName == alertNameNetworkMonitorLoss || newName == alertNameNetworkMonitorLoss) &&
(oldName != newName || e.Record.GetString("system") != e.Record.Original().GetString("system")) {
return e.BadRequestError("Delete and recreate the alert to change its type or system", nil)
}
if e.Record.GetString("name") == alertNameNetworkMonitorLoss {
if !e.HasSuperuserAuth() && (e.Auth == nil || !userHasSystem(e.App, e.Auth.Id, e.Record.GetString("system"))) {
return e.ForbiddenError("You do not have access to this system", nil)
}
e.Record.Set("triggered", e.Record.Original().GetBool("triggered"))
value := e.Record.GetFloat("value")
if math.IsNaN(value) || math.IsInf(value, 0) || value < 0 || value >= 100 {
return e.BadRequestError("Monitor loss threshold must be at least 0 and below 100", nil)
}
e.Record.Set("min", 0)
}
return e.Next()
}
am.hub.OnRecordCreateRequest("alerts").BindFunc(protectState)
am.hub.OnRecordUpdateRequest("alerts").BindFunc(protectState)
cleanup := func(e *core.RecordEvent) error {
if err := e.Next(); err != nil {
return err
}
return am.evaluateNetworkMonitorAlerts(e.App, e.Record.GetString("system"), nil)
}
am.hub.OnRecordAfterDeleteSuccess("network_monitors").BindFunc(cleanup)
am.hub.OnRecordAfterUpdateSuccess("network_monitors").BindFunc(func(e *core.RecordEvent) error {
if e.Record.GetBool("enabled") || !e.Record.Original().GetBool("enabled") {
return e.Next()
}
return cleanup(e)
})
}
// HandleNetworkMonitorAlerts runs after the full monitoring transaction commits,
// using its exact payload (dashboard requests can replace the cached payload).
// Omitted results and disconnected systems never imply recovery.
func (am *AlertManager) HandleNetworkMonitorAlerts(systemRecord *core.Record, results map[string]monitor.Result) error {
if systemRecord.GetString("status") != "up" {
return nil
}
alerts := am.alertsCache.GetAlertsByName(systemRecord.Id, alertNameNetworkMonitorLoss)
if len(alerts) == 0 {
return nil
}
monitors, err := am.networkMonitors.get(systemRecord.Id)
if err != nil {
return err
}
if !networkMonitorTransitionPending(alerts, monitors, results, time.Now()) {
return nil
}
// The cache only predicts a transition. Reload and recheck under the DB
// transaction before persisting, including current system/monitor status.
return am.evaluateNetworkMonitorAlerts(am.hub, systemRecord.Id, results)
}
// networkMonitorTransitionPending does no IO and never mutates cached maps.
func networkMonitorTransitionPending(alerts []CachedAlertData, monitors map[string]int, results map[string]monitor.Result, now time.Time) bool {
for _, alert := range alerts {
if !alert.MonitorStatesValid || alert.Triggered != (len(alert.MonitorStates) > 0) {
return true
}
for id := range alert.MonitorStates {
if _, enabled := monitors[id]; !enabled {
return true
}
}
for id, result := range results {
interval, enabled := monitors[id]
if !enabled || !monitorResultReady(result, interval, now) {
continue
}
_, active := alert.MonitorStates[id]
if (result.PacketLoss1h > alert.Value) != active {
return true
}
}
}
return false
}
func (am *AlertManager) evaluateNetworkMonitorAlerts(app core.App, systemID string, results map[string]monitor.Result) error {
var messages []AlertMessageData
err := app.RunInTransaction(func(tx core.App) error {
// Read configuration inside the transaction so concurrent threshold changes,
// disabling, and evaluations cannot overwrite each other's incident state.
alerts, err := tx.FindAllRecords("alerts", dbx.HashExp{"system": systemID, "name": alertNameNetworkMonitorLoss})
if err != nil || len(alerts) == 0 {
return err
}
system, err := tx.FindRecordById("systems", systemID)
if errors.Is(err, sql.ErrNoRows) {
// System deletion cascades to its alerts.
return nil
}
if err != nil {
return err
}
monitors, err := tx.FindAllRecords("network_monitors", dbx.HashExp{"system": systemID, "enabled": true})
if err != nil {
return err
}
enabled := make(map[string]*core.Record, len(monitors))
for _, m := range monitors {
enabled[m.Id] = m
}
now := time.Now()
for _, alert := range alerts {
var state networkMonitorAlertState
if err := alert.UnmarshalJSONField("state", &state); err != nil {
return err
}
states := state.Monitors
if states == nil {
states = map[string]string{}
}
changed := false
// Removing or disabling a monitor closes its incident silently.
for id, historyID := range states {
if _, ok := enabled[id]; !ok {
if err := resolveMonitorIncident(tx, historyID, now); err != nil {
return err
}
delete(states, id)
changed = true
}
}
if system.GetString("status") == "up" {
for _, m := range monitors {
result, ok := results[m.Id]
if !ok || !monitorResultReady(result, m.GetInt("interval"), now) {
continue
}
historyID, active := states[m.Id]
triggered := result.PacketLoss1h > alert.GetFloat("value")
if triggered == active {
continue
}
label := m.GetString("target")
if m.GetString("protocol") == "tcp" {
label = net.JoinHostPort(label, strconv.Itoa(m.GetInt("port")))
}
if triggered {
collection, err := tx.FindCachedCollectionByNameOrId("alerts_history")
if err != nil {
return err
}
history := core.NewRecord(collection)
history.Load(map[string]any{
"alert_id": alert.Id, "user": alert.GetString("user"), "system": systemID,
"name": alertNameNetworkMonitorLoss, "monitor_name": label, "value": result.PacketLoss1h,
})
if err := tx.Save(history); err != nil {
return err
}
states[m.Id] = history.Id
} else {
if err := resolveMonitorIncident(tx, historyID, now); err != nil {
return err
}
delete(states, m.Id)
}
changed = true
state, comparison := "loss", "exceeds"
if !triggered {
state, comparison = "recovered", "is at or below"
}
messages = append(messages, AlertMessageData{
UserID: alert.GetString("user"), SystemID: systemID,
Title: fmt.Sprintf("Network monitor %s on %s: %s", state, system.GetString("name"), label),
Message: fmt.Sprintf("%s on %s: loss over the past hour is %.2f%%, which %s the %.2f%% threshold.", label, system.GetString("name"), result.PacketLoss1h, comparison, alert.GetFloat("value")),
Link: am.hub.MakeLink("system", systemID), LinkText: "View " + system.GetString("name"),
})
}
}
if changed || alert.GetBool("triggered") != (len(states) > 0) {
alert.Set("state", networkMonitorAlertState{Monitors: states})
alert.Set("triggered", len(states) > 0)
if err := tx.Save(alert); err != nil {
return err
}
}
}
return nil
})
if err != nil {
return err
}
// Match other alert types: persist transitions before delivery, and respect
// the user's existing notification destinations and quiet hours.
for _, message := range messages {
if err := am.SendAlert(message); err != nil {
app.Logger().Error("Failed to send network monitor alert", "err", err)
}
}
return nil
}
func monitorResultReady(result monitor.Result, interval int, now time.Time) bool {
// Three completed attempts provide a short warm-up, including after an agent
// restart.
if result.SampleCount < 3 || result.LastProbeAt <= 0 || math.IsNaN(result.PacketLoss1h) || math.IsInf(result.PacketLoss1h, 0) || result.PacketLoss1h < 0 || result.PacketLoss1h > 100 {
return false
}
// Never interpret an empty one-hour window as zero loss.
maxAge := min(time.Hour, max(3*time.Duration(interval)*time.Second, 3*time.Minute))
age := now.Sub(time.UnixMilli(result.LastProbeAt))
return age >= -time.Minute && age <= maxAge
}
func resolveMonitorIncident(app core.App, id string, now time.Time) error {
record, err := app.FindRecordById("alerts_history", id)
if errors.Is(err, sql.ErrNoRows) {
// History can be purged independently.
return nil
}
if err != nil {
return err
}
if !record.GetDateTime("resolved").IsZero() {
return nil
}
record.Set("resolved", now.UTC())
return app.Save(record)
}
func resolveNetworkMonitorHistory(app core.App, alertID string) error {
records, err := app.FindAllRecords("alerts_history", dbx.HashExp{"alert_id": alertID, "resolved": ""})
if err != nil {
return err
}
for _, record := range records {
record.Set("resolved", time.Now().UTC())
if err := app.Save(record); err != nil {
return err
}
}
return nil
}

View File

@@ -1,513 +0,0 @@
//go:build testing
package alerts_test
import (
"sync"
"sync/atomic"
"testing"
"time"
"github.com/henrygd/beszel/internal/alerts"
"github.com/henrygd/beszel/internal/entities/monitor"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
pbTests "github.com/pocketbase/pocketbase/tests"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func networkAlertSetup(t *testing.T) (*beszelTests.TestHub, *core.Record, *core.Record, []*core.Record) {
t.Helper()
hub, system, alert := systemdTestSetup(t, false)
t.Cleanup(hub.Cleanup)
alert.Set("name", "NetworkMonitorLoss")
alert.Set("value", 5)
require.NoError(t, hub.Save(alert))
var monitors []*core.Record
for _, name := range []string{"gateway", "website"} {
record, err := beszelTests.CreateRecord(hub, "network_monitors", map[string]any{
"system": system.Id, "target": name + ".example.com", "protocol": "icmp", "interval": 60, "enabled": true,
})
require.NoError(t, err)
monitors = append(monitors, record)
}
// Avoid starting a system update worker in tests.
_, err := hub.DB().Update("systems", dbx.Params{"status": "up"}, dbx.HashExp{"id": system.Id}).Execute()
require.NoError(t, err)
system.Set("status", "up")
return hub, system, alert, monitors
}
func monitorResult(loss float64) monitor.Result {
return monitor.Result{LastProbeAt: time.Now().UnixMilli(), SampleCount: 60, PacketLoss1h: loss}
}
func TestNetworkMonitorAlertIndependentIncidents(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
count := hub.TestMailer.TotalSend()
results := map[string]monitor.Result{monitors[0].Id: monitorResult(10), monitors[1].Id: monitorResult(0)}
check := func(active bool, open, sent int) {
t.Helper()
record, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.Equal(t, active, record.GetBool("triggered"))
total, err := hub.CountRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id, "resolved": ""})
require.NoError(t, err)
assert.EqualValues(t, open, total)
assert.Equal(t, count+sent, hub.TestMailer.TotalSend())
}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
check(true, 1, 1)
message := hub.TestMailer.Messages()[count]
assert.Contains(t, message.Text, "gateway.example.com")
assert.Contains(t, message.Text, "10.00%")
assert.Contains(t, message.Text, "5.00%")
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
check(true, 1, 1)
// Persisted monitor state prevents duplicate notifications after a hub restart.
am = alerts.NewTestAlertManagerWithoutWorker(hub)
results[monitors[1].Id] = monitorResult(20)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
check(true, 2, 2)
results[monitors[0].Id] = monitorResult(5) // Equality is a recovery.
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
check(true, 1, 3)
results[monitors[1].Id] = monitorResult(0)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
check(false, 0, 4)
histories, err := hub.FindAllRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id})
require.NoError(t, err)
require.Len(t, histories, 2)
for _, history := range histories {
assert.NotEmpty(t, history.GetString("monitor_name"))
}
}
func TestNetworkMonitorAlertTargetLabel(t *testing.T) {
for _, tc := range []struct {
protocol, target, label string
port int
}{
{"icmp", "gateway.example.com", "gateway.example.com", 0},
{"http", "https://example.com/health", "https://example.com/health", 0},
{"tcp", "example.com", "example.com:8443", 8443},
{"tcp", "2001:db8::1", "[2001:db8::1]:443", 443},
} {
t.Run(tc.label, func(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
m := monitors[0]
m.Set("protocol", tc.protocol)
m.Set("target", tc.target)
m.Set("port", tc.port)
require.NoError(t, hub.Save(m))
am := alerts.NewTestAlertManagerWithoutWorker(hub)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, map[string]monitor.Result{m.Id: monitorResult(10)}))
histories, err := hub.FindAllRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id})
require.NoError(t, err)
require.Len(t, histories, 1)
assert.Equal(t, tc.label, histories[0].GetString("monitor_name"))
assert.Contains(t, hub.TestMailer.Messages()[hub.TestMailer.TotalSend()-1].Text, tc.label)
})
}
}
func TestNetworkMonitorAlertIgnoresUnknownResults(t *testing.T) {
for _, scenario := range []string{"missing", "stale", "warmup", "no probes", "down", "paused", "future", "expired hourly window"} {
t.Run(scenario, func(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
apply := func(loss float64) {
result := monitorResult(loss)
results := map[string]monitor.Result{monitors[0].Id: result}
switch scenario {
case "missing":
results = nil
case "stale":
result.LastProbeAt = time.Now().Add(-10 * time.Minute).UnixMilli()
results[monitors[0].Id] = result
case "expired hourly window":
monitors[0].Set("interval", 3600)
require.NoError(t, hub.Save(monitors[0]))
result.LastProbeAt = time.Now().Add(-2 * time.Hour).UnixMilli()
results[monitors[0].Id] = result
case "future":
result.LastProbeAt = time.Now().Add(time.Hour).UnixMilli()
results[monitors[0].Id] = result
case "warmup":
result.SampleCount = 2
results[monitors[0].Id] = result
case "no probes":
result.SampleCount = 0
results[monitors[0].Id] = result
case "down":
_, err := hub.DB().Update("systems", dbx.Params{"status": scenario}, dbx.HashExp{"id": system.Id}).Execute()
require.NoError(t, err)
case "paused":
record, err := hub.FindRecordById("systems", system.Id)
require.NoError(t, err)
record.Set("status", "paused")
require.NoError(t, hub.Save(record))
}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
}
count := hub.TestMailer.TotalSend()
apply(100)
assert.Equal(t, count, hub.TestMailer.TotalSend())
_, err := hub.DB().Update("systems", dbx.Params{"status": "up"}, dbx.HashExp{"id": system.Id}).Execute()
require.NoError(t, err)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, map[string]monitor.Result{monitors[0].Id: monitorResult(10)}))
apply(0)
assert.Equal(t, count+1, hub.TestMailer.TotalSend(), "unknown data must not recover an incident")
record, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, record.GetBool("triggered"))
})
}
}
func TestNetworkMonitorAlertCleanup(t *testing.T) {
for _, scenario := range []string{"disable monitor", "delete monitor", "disable alert", "purge history", "delete system"} {
t.Run(scenario, func(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(10), monitors[1].Id: monitorResult(20)}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
count := hub.TestMailer.TotalSend()
switch scenario {
case "disable monitor":
monitors[0].Set("enabled", false)
require.NoError(t, hub.Save(monitors[0]))
case "delete monitor":
require.NoError(t, hub.Delete(monitors[0]))
case "disable alert":
require.NoError(t, hub.Delete(alert))
case "delete system":
require.NoError(t, hub.Delete(system))
case "purge history":
history, err := hub.FindAllRecords("alerts_history")
require.NoError(t, err)
for _, record := range history {
require.NoError(t, hub.Delete(record))
}
}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count, hub.TestMailer.TotalSend())
open, err := hub.CountRecords("alerts_history", dbx.HashExp{"alert_id": alert.Id, "resolved": ""})
require.NoError(t, err)
if scenario == "disable monitor" || scenario == "delete monitor" {
assert.EqualValues(t, 1, open)
record, err := hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.True(t, record.GetBool("triggered"))
require.NoError(t, hub.Delete(monitors[1]))
record, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, record.GetBool("triggered"))
} else {
assert.Zero(t, open)
}
})
}
}
func TestNetworkMonitorAlertPerUserThresholds(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
user, err := beszelTests.CreateUser(hub, "monitor2@example.com", "password")
require.NoError(t, err)
other, err := beszelTests.CreateRecord(hub, "alerts", map[string]any{"name": "NetworkMonitorLoss", "system": system.Id, "user": user.Id, "value": 20})
require.NoError(t, err)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(10)}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
other, err = hub.FindRecordById("alerts", other.Id)
require.NoError(t, err)
assert.False(t, other.GetBool("triggered"))
// Editing the threshold re-evaluates on the next batch, without losing state.
alert, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
alert.Set("value", 15)
require.NoError(t, hub.Save(alert))
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
alert, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
assert.False(t, alert.GetBool("triggered"))
}
func TestNetworkMonitorAlertAPI(t *testing.T) {
for _, tc := range []struct {
name string
value float64
direct, denied, patch bool
status int
}{
{name: "zero threshold", value: 0, status: 200},
{name: "fractional threshold", value: 5.5, status: 200},
{name: "negative threshold", value: -1, status: 400},
{name: "unreachable threshold", value: 100, status: 400},
{name: "bulk inaccessible system", value: 5, denied: true, status: 200},
{name: "direct inaccessible system", value: 5, direct: true, denied: true, status: 403},
{name: "direct invalid threshold", value: -1, direct: true, status: 400},
{name: "direct private state", value: 5, direct: true, status: 200},
{name: "patch preserves state", value: 10, direct: true, patch: true, status: 200},
} {
t.Run(tc.name, func(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
owner := user.Id
if tc.denied {
other, err := beszelTests.CreateUser(hub, "other@example.com", "password")
require.NoError(t, err)
owner = other.Id
}
systems, err := beszelTests.CreateSystems(hub, 1, owner, "paused")
require.NoError(t, err)
token, err := user.NewAuthToken()
require.NoError(t, err)
body := map[string]any{"name": "NetworkMonitorLoss", "value": tc.value, "min": 60, "systems": []string{systems[0].Id}, "overwrite": true}
url, method := "/api/beszel/user-alerts", "POST"
if tc.direct {
url = "/api/collections/alerts/records"
body["system"], body["user"] = systems[0].Id, user.Id
body["state"], body["triggered"] = map[string]any{"monitors": map[string]string{"fake": "fake"}}, true
}
if tc.patch {
alert, err := beszelTests.CreateRecord(hub, "alerts", map[string]any{"name": "NetworkMonitorLoss", "system": systems[0].Id, "user": user.Id, "value": 5, "triggered": true, "state": map[string]any{"monitors": map[string]string{"real": "history"}}})
require.NoError(t, err)
url += "/" + alert.Id
method = "PATCH"
body["triggered"] = false
}
content := `"success":true`
if tc.direct {
content = `"name":"NetworkMonitorLoss"`
}
if tc.status == 400 {
content = `"status":400`
}
if tc.status == 403 {
content = `"status":403`
}
scenario := beszelTests.ApiScenario{
Name: tc.name, Method: method, URL: url, Body: jsonReader(body),
Headers: map[string]string{"Authorization": token}, ExpectedStatus: tc.status, ExpectedContent: []string{content},
TestAppFactory: func(testing.TB) *pbTests.TestApp { return hub.TestApp },
}
scenario.Test(t)
records, err := hub.FindAllRecords("alerts")
require.NoError(t, err)
if tc.status != 200 || tc.denied {
assert.Empty(t, records)
return
}
require.Len(t, records, 1)
assert.Equal(t, tc.value, records[0].GetFloat("value"))
assert.Zero(t, records[0].GetInt("min"))
state := struct {
Monitors map[string]string `json:"monitors"`
}{}
require.NoError(t, records[0].UnmarshalJSONField("state", &state))
states := state.Monitors
if tc.patch {
assert.Equal(t, map[string]string{"real": "history"}, states)
assert.True(t, records[0].GetBool("triggered"))
} else {
assert.Empty(t, states)
assert.False(t, records[0].GetBool("triggered"))
}
})
}
}
type monitorCountingHub struct {
*beszelTests.TestHub
transactions atomic.Int64
beforeTransaction func()
}
func (h *monitorCountingHub) RunInTransaction(fn func(core.App) error) error {
h.transactions.Add(1)
if h.beforeTransaction != nil {
h.beforeTransaction()
}
return h.App.RunInTransaction(fn)
}
// Count actual SQL on both DB connections, including queries through record APIs.
func monitorSQLCounter(t *testing.T, app core.App) *atomic.Int64 {
t.Helper()
count := &atomic.Int64{}
for _, builder := range []dbx.Builder{app.ConcurrentDB(), app.NonconcurrentDB()} {
db := builder.(*dbx.DB)
old := db.LogFunc
db.LogFunc = func(string, ...any) { count.Add(1) }
t.Cleanup(func() { db.LogFunc = old })
}
return count
}
func TestNetworkMonitorAlertSteadyStateNoDatabaseWork(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
counted := &monitorCountingHub{TestHub: hub}
am := alerts.NewTestAlertManagerWithoutWorker(counted)
sql := monitorSQLCounter(t, hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(0)}
evaluate := func() { t.Helper(); require.NoError(t, am.HandleNetworkMonitorAlerts(system, results)) }
noWork := func() {
t.Helper()
sql.Store(0)
counted.transactions.Store(0)
for range 100 {
evaluate()
}
assert.Zero(t, sql.Load(), "steady state must not issue SQL")
assert.Zero(t, counted.transactions.Load(), "steady state must not open transactions")
}
// One-time lazy loads are permitted, including on hub restart.
evaluate()
assert.Positive(t, sql.Load())
noWork()
// Realtime metric saves invoke record hooks but must not invalidate config.
fresh, err := hub.FindRecordById("network_monitors", monitors[0].Id)
require.NoError(t, err)
monitors[0] = fresh
monitors[0].Set("loss1h", 0)
monitors[0].Set("res", 100)
require.NoError(t, hub.Save(monitors[0]))
noWork()
results[monitors[0].Id] = monitorResult(10)
evaluate()
assert.Positive(t, sql.Load(), "transitions must still be persisted")
assert.EqualValues(t, 1, counted.transactions.Load())
noWork()
// Missing and stale observations must not enter the transaction either.
results = nil
noWork()
results = map[string]monitor.Result{monitors[0].Id: {SampleCount: 60, LastProbeAt: time.Now().Add(-10 * time.Minute).UnixMilli()}}
noWork()
results[monitors[0].Id] = monitorResult(0)
evaluate()
noWork()
require.NoError(t, hub.Delete(alert))
noWork()
}
func TestNetworkMonitorAlertConfigCacheInvalidation(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(0)}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
count := hub.TestMailer.TotalSend()
// Widening the interval makes this observation fresh. A stale interval cache
// would miss the failure indefinitely, even though results keep arriving.
result := monitorResult(10)
result.LastProbeAt = time.Now().Add(-4 * time.Minute).UnixMilli()
results[monitors[0].Id] = result
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count, hub.TestMailer.TotalSend())
monitors[0].Set("interval", 120)
require.NoError(t, hub.Save(monitors[0]))
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count+1, hub.TestMailer.TotalSend())
// Disable, then re-enable the same ID: its new failure must be detected.
monitors[0].Set("enabled", false)
require.NoError(t, hub.Save(monitors[0]))
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
monitors[0].Set("enabled", true)
require.NoError(t, hub.Save(monitors[0]))
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count+2, hub.TestMailer.TotalSend())
// A new monitor must also become eligible without restarting the hub.
created, err := beszelTests.CreateRecord(hub, "network_monitors", map[string]any{
"system": system.Id, "name": "new", "target": "new.example.com", "protocol": "icmp", "interval": 60, "enabled": true,
})
require.NoError(t, err)
results[created.Id] = monitorResult(10)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count+3, hub.TestMailer.TotalSend())
// Threshold changes refresh cached config and preserve the active incidents.
alert, err = hub.FindRecordById("alerts", alert.Id)
require.NoError(t, err)
alert.Set("value", 15)
require.NoError(t, hub.Save(alert))
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count+5, hub.TestMailer.TotalSend())
}
func TestNetworkMonitorAlertRevalidatesCandidate(t *testing.T) {
for _, change := range []string{"threshold", "disable alert", "disable monitor", "down"} {
t.Run(change, func(t *testing.T) {
hub, system, alert, monitors := networkAlertSetup(t)
counted := &monitorCountingHub{TestHub: hub}
am := alerts.NewTestAlertManagerWithoutWorker(counted)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(0)}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
count := hub.TestMailer.TotalSend()
// Change the DB after the cache predicts a transition, before its transaction.
counted.beforeTransaction = func() {
counted.beforeTransaction = nil
switch change {
case "threshold":
alert.Set("value", 20)
require.NoError(t, hub.Save(alert))
case "disable alert":
require.NoError(t, hub.Delete(alert))
case "disable monitor":
monitors[0].Set("enabled", false)
require.NoError(t, hub.Save(monitors[0]))
case "down":
_, err := hub.DB().Update("systems", dbx.Params{"status": "down"}, dbx.HashExp{"id": system.Id}).Execute()
require.NoError(t, err)
}
}
results[monitors[0].Id] = monitorResult(10)
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.EqualValues(t, 1, counted.transactions.Load())
assert.Equal(t, count, hub.TestMailer.TotalSend())
histories, err := hub.CountRecords("alerts_history")
require.NoError(t, err)
assert.Zero(t, histories)
})
}
}
func TestNetworkMonitorAlertConcurrentEvaluations(t *testing.T) {
hub, system, _, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(10)}
count := hub.TestMailer.TotalSend()
var wg sync.WaitGroup
errs := make(chan error, 8)
for range 8 {
wg.Go(func() { errs <- am.HandleNetworkMonitorAlerts(system, results) })
}
wg.Wait()
close(errs)
for err := range errs {
require.NoError(t, err)
}
assert.Equal(t, count+1, hub.TestMailer.TotalSend())
}
func TestNetworkMonitorAlertCacheAfterRollback(t *testing.T) {
hub, system, _, monitors := networkAlertSetup(t)
am := alerts.NewTestAlertManagerWithoutWorker(hub)
results := map[string]monitor.Result{monitors[0].Id: monitorResult(0)}
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
_, err := hub.DB().NewQuery(`CREATE TRIGGER fail_alert BEFORE UPDATE ON alerts BEGIN SELECT RAISE(ABORT, 'test rollback'); END`).Execute()
require.NoError(t, err)
count := hub.TestMailer.TotalSend()
results[monitors[0].Id] = monitorResult(10)
require.Error(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count, hub.TestMailer.TotalSend())
histories, err := hub.CountRecords("alerts_history")
require.NoError(t, err)
assert.Zero(t, histories)
_, err = hub.DB().NewQuery("DROP TRIGGER fail_alert").Execute()
require.NoError(t, err)
// A failed transition must not be published to the cache and mask the retry.
require.NoError(t, am.HandleNetworkMonitorAlerts(system, results))
assert.Equal(t, count+1, hub.TestMailer.TotalSend())
}

View File

@@ -47,7 +47,7 @@ func (am *AlertManager) HandleSystemAlerts(systemRecord *core.Record, data *syst
return nil
}
alerts := am.alertsCache.GetAlertsExcludingNames(systemRecord.Id, "Status", alertNameSystemdFailed, containerAlertName, alertNameNetworkMonitorLoss)
alerts := am.alertsCache.GetAlertsExcludingNames(systemRecord.Id, "Status", alertNameSystemdFailed, containerAlertName)
if len(alerts) == 0 {
return nil
}

View File

@@ -11,9 +11,8 @@ import (
func NewTestAlertManagerWithoutWorker(app hubLike) *AlertManager {
return &AlertManager{
hub: app,
alertsCache: NewAlertsCache(app),
networkMonitors: newNetworkMonitorCache(app),
hub: app,
alertsCache: NewAlertsCache(app),
}
}

View File

@@ -1,79 +0,0 @@
package alerts
import (
"sync"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
)
// networkMonitorCache keeps just the enabled monitor IDs and probe intervals
// needed for the alert fast path. Names and targets are read only on transitions.
// Returned maps are immutable; configuration changes invalidate the whole entry.
type networkMonitorCache struct {
app core.App
mu sync.RWMutex
systems map[string]map[string]int
}
func newNetworkMonitorCache(app core.App) *networkMonitorCache {
c := &networkMonitorCache{app: app, systems: make(map[string]map[string]int)}
invalidate := func(e *core.RecordEvent) error {
c.invalidate(e.Record.GetString("system"))
return e.Next()
}
app.OnRecordAfterCreateSuccess("network_monitors").BindFunc(invalidate)
app.OnRecordAfterDeleteSuccess("network_monitors").BindFunc(invalidate)
app.OnRecordAfterUpdateSuccess("network_monitors").BindFunc(func(e *core.RecordEvent) error {
old := e.Record.Original()
// Realtime metric saves also invoke this hook. They must not evict config.
if old.GetString("system") != e.Record.GetString("system") ||
old.GetBool("enabled") != e.Record.GetBool("enabled") ||
old.GetInt("interval") != e.Record.GetInt("interval") {
c.invalidate(old.GetString("system"))
c.invalidate(e.Record.GetString("system"))
}
return e.Next()
})
app.OnRecordAfterDeleteSuccess("systems").BindFunc(func(e *core.RecordEvent) error {
c.invalidate(e.Record.Id)
return e.Next()
})
return c
}
func (c *networkMonitorCache) invalidate(systemID string) {
c.mu.Lock()
delete(c.systems, systemID)
c.mu.Unlock()
}
func (c *networkMonitorCache) get(systemID string) (map[string]int, error) {
c.mu.RLock()
monitors, ok := c.systems[systemID]
c.mu.RUnlock()
if ok {
return monitors, nil
}
c.mu.Lock()
defer c.mu.Unlock()
if monitors, ok := c.systems[systemID]; ok {
return monitors, nil
}
// Keep the lock through the load so a concurrent config change cannot be
// invalidated first and then overwritten by the older query result.
var rows []struct {
ID string `db:"id"`
Interval int `db:"interval"`
}
if err := c.app.DB().Select("id", "interval").From("network_monitors").
Where(dbx.HashExp{"system": systemID, "enabled": true}).All(&rows); err != nil {
return nil, err
}
monitors = make(map[string]int, len(rows))
for _, row := range rows {
monitors[row.ID] = row.Interval
}
c.systems[systemID] = monitors
return monitors, nil
}

View File

@@ -11,7 +11,6 @@ import (
"strings"
"sync/atomic"
"testing"
"testing/synctest"
"github.com/nicholas-fedor/shoutrrr/pkg/types"
"golang.org/x/net/dns/dnsmessage"
@@ -179,45 +178,39 @@ func TestPublicNotificationTCP(t *testing.T) {
} {
t.Run(rawURL, func(t *testing.T) {
t.Parallel()
// MQTT waits for a fixed library timeout even after a dial failure.
// Virtual time preserves the full send/cleanup path without that delay.
t.Run("internal destination", func(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
err := sendPublicNotification(strings.ReplaceAll(rawURL, "HOST", "127.0.0.1"), "test")
if !errors.Is(err, errInternalDestination) {
t.Fatalf("expected blocked destination, got %v", err)
}
})
err := sendPublicNotification(strings.ReplaceAll(rawURL, "HOST", "127.0.0.1"), "test")
if !errors.Is(err, errInternalDestination) {
t.Fatalf("expected blocked destination, got %v", err)
}
})
t.Run("public destination uses injected dialer", func(t *testing.T) {
synctest.Test(t, func(t *testing.T) {
var calls atomic.Int32
stopped := errors.New("test dial stopped")
service, err := newPublicNotificationService(strings.ReplaceAll(rawURL, "HOST", "8.8.8.8"), types.SenderOptions{
DialContext: func(ctx context.Context, network, address string) (net.Conn, error) {
calls.Add(1)
if network != "tcp" || !strings.HasPrefix(address, "8.8.8.8:") {
t.Errorf("unexpected dial: %s %s", network, address)
}
if err := checkNotificationAddress(address); err != nil {
t.Error(err)
}
return nil, stopped
},
})
if err != nil {
t.Fatal(err)
}
if closer, ok := service.(io.Closer); ok {
defer closer.Close()
}
if err := service.Send("test", &types.Params{}); err == nil {
t.Fatal("expected dial failure")
}
if calls.Load() == 0 {
t.Fatal("custom dialer was not used")
}
var calls atomic.Int32
stopped := errors.New("test dial stopped")
service, err := newPublicNotificationService(strings.ReplaceAll(rawURL, "HOST", "8.8.8.8"), types.SenderOptions{
DialContext: func(ctx context.Context, network, address string) (net.Conn, error) {
calls.Add(1)
if network != "tcp" || !strings.HasPrefix(address, "8.8.8.8:") {
t.Errorf("unexpected dial: %s %s", network, address)
}
if err := checkNotificationAddress(address); err != nil {
t.Error(err)
}
return nil, stopped
},
})
if err != nil {
t.Fatal(err)
}
if closer, ok := service.(io.Closer); ok {
defer closer.Close()
}
if err := service.Send("test", &types.Params{}); err == nil {
t.Fatal("expected dial failure")
}
if calls.Load() == 0 {
t.Fatal("custom dialer was not used")
}
})
})
}

View File

@@ -1,11 +1,9 @@
package main
import (
"errors"
"fmt"
"log"
"os"
"runtime"
"strings"
"github.com/henrygd/beszel"
@@ -16,12 +14,6 @@ import (
"golang.org/x/crypto/ssh"
)
type noKeyProvidedError struct{}
func (noKeyProvidedError) Error() string {
return "no key provided: must set -key flag, KEY env var, or KEY_FILE env var. Use 'beszel-agent help' for usage"
}
// cli options
type cmdOptions struct {
key string // key is the public key(s) for SSH authentication.
@@ -132,7 +124,7 @@ func (opts *cmdOptions) loadPublicKeys() ([]ssh.PublicKey, error) {
// Try key file
keyFile, ok := utils.GetEnv("KEY_FILE")
if !ok {
return nil, noKeyProvidedError{}
return nil, fmt.Errorf("no key provided: must set -key flag, KEY env var, or KEY_FILE env var. Use 'beszel-agent help' for usage")
}
pubKey, err := os.ReadFile(keyFile)
@@ -146,14 +138,6 @@ func (opts *cmdOptions) getAddress() string {
return agent.GetAddress(opts.listen)
}
func isBenignStartupError(err error, goos string) bool {
if goos != "windows" {
return false
}
var noKeyErr noKeyProvidedError
return errors.As(err, &noKeyErr)
}
// handleFingerprint handles the "fingerprint" command with subcommands "view" and "reset".
func handleFingerprint() {
subCmd := ""
@@ -198,12 +182,6 @@ func main() {
var err error
serverConfig.Keys, err = opts.loadPublicKeys()
if err != nil {
if isBenignStartupError(err, runtime.GOOS) {
// WinGet launches the executable without configuration during validation.
// Exit successfully in that case while retaining the error on other platforms.
log.Print("Failed to load public keys:", err)
return
}
log.Fatal("Failed to load public keys:", err)
}

View File

@@ -2,7 +2,6 @@ package main
import (
"crypto/ed25519"
"errors"
"os"
"path/filepath"
"testing"
@@ -188,26 +187,6 @@ func TestLoadPublicKeys(t *testing.T) {
}
}
func TestIsBenignStartupError(t *testing.T) {
tests := []struct {
name string
err error
goos string
want bool
}{
{name: "missing key on windows", err: noKeyProvidedError{}, goos: "windows", want: true},
{name: "wrapped missing key on windows", err: errors.Join(errors.New("startup failed"), noKeyProvidedError{}), goos: "windows", want: true},
{name: "missing key on linux", err: noKeyProvidedError{}, goos: "linux", want: false},
{name: "different error on windows", err: errors.New("invalid key"), goos: "windows", want: false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
assert.Equal(t, tt.want, isBenignStartupError(tt.err, tt.goos))
})
}
}
func TestGetNetwork(t *testing.T) {
tests := []struct {
name string

View File

@@ -24,8 +24,6 @@ const (
GetSystemdInfo
// Request ZFS detail data from agent
GetZfsData
// Sync network monitor configuration to agent
SyncNetworkMonitors
// Add new actions here...
)

View File

@@ -12,7 +12,7 @@ RUN apk add --no-cache ca-certificates && update-ca-certificates
# Build
ARG TARGETOS TARGETARCH
RUN CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN rm -rf /tmp/*

View File

@@ -10,14 +10,14 @@ COPY . ./
# Build
ARG TARGETOS TARGETARCH
RUN CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN rm -rf /tmp/*
# --------------------------
# Final image: default scratch-based agent
# --------------------------
FROM alpine:3.24
FROM alpine:3.23
COPY --from=builder /agent /agent
# AMD GPU name lookup (used by agent on Linux when /usr/share/libdrm/amdgpu.ids is read)
@@ -28,4 +28,4 @@ RUN apk add --no-cache smartmontools zfs
# Ensure data persistence across container recreations
VOLUME ["/var/lib/beszel-agent"]
ENTRYPOINT ["/agent"]
ENTRYPOINT ["/agent"]

View File

@@ -10,13 +10,13 @@ COPY . ./
# Build
ARG TARGETOS TARGETARCH
RUN CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /agent ./internal/cmd/agent
# --------------------------
# Final image
# Note: must cap_add: [CAP_PERFMON] and mount /dev/dri/ as volume
# --------------------------
FROM alpine:3.24
FROM alpine:3.23
COPY --from=builder /agent /agent

View File

@@ -10,7 +10,7 @@ COPY . ./
# Build
ARG TARGETOS TARGETARCH
RUN CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -tags glibc -ldflags "-w -s" -o /agent ./internal/cmd/agent
RUN CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -tags glibc -ldflags "-w -s" -o /agent ./internal/cmd/agent
# --------------------------
# Smartmontools builder stage

View File

@@ -17,7 +17,7 @@ RUN set -eux; \
if [ "$TARGETARCH" = "arm" ] && [ -n "$TARGETVARIANT" ]; then \
export GOARM="${TARGETVARIANT#v}"; \
fi; \
CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH \
CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH \
go build -tags glibc -ldflags "-w -s" -o /agent ./internal/cmd/agent
# --------------------------
@@ -70,9 +70,7 @@ RUN set -eux; \
# --------------------------
FROM --platform=$TARGETPLATFORM debian:bookworm-slim AS zfsutils-builder
# zfsutils-linux is distributed in Debian's contrib component.
RUN sed -i 's/Components: main/Components: main contrib/' /etc/apt/sources.list.d/debian.sources \
&& apt-get update && apt-get install -y --no-install-recommends \
RUN apt-get update && apt-get install -y --no-install-recommends \
zfsutils-linux \
&& rm -rf /var/lib/apt/lists/*

View File

@@ -17,7 +17,7 @@ RUN update-ca-certificates
# Build
ARG TARGETOS TARGETARCH
RUN CGO_ENABLED=0 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /beszel ./internal/cmd/hub
RUN CGO_ENABLED=0 GOGC=75 GOOS=$TARGETOS GOARCH=$TARGETARCH go build -ldflags "-w -s" -o /beszel ./internal/cmd/hub
# ? -------------------------
FROM scratch
@@ -31,4 +31,4 @@ VOLUME ["/beszel_data"]
EXPOSE 8090
ENTRYPOINT [ "/beszel" ]
CMD ["serve", "--http=0.0.0.0:8090"]
CMD ["serve", "--http=0.0.0.0:8090"]

View File

@@ -186,12 +186,11 @@ type Stats struct {
NetworkRecv float64 `json:"nr,omitzero" cbor:"4,keyasint,omitzero"` // deprecated 0.18.3 (MB) - keep field for old agents/records
Bandwidth [2]uint64 `json:"b,omitzero" cbor:"9,keyasint,omitzero"` // [sent bytes, recv bytes]
Health DockerHealth `json:"-" cbor:"5,keyasint"`
Status string `json:"-" cbor:"6,keyasint"`
Id string `json:"-" cbor:"7,keyasint"`
Image string `json:"-" cbor:"8,keyasint"`
Ports string `json:"-" cbor:"10,keyasint"`
UpdateAvailable bool `json:"u,omitzero" cbor:"11,keyasint,omitzero"`
Health DockerHealth `json:"-" cbor:"5,keyasint"`
Status string `json:"-" cbor:"6,keyasint"`
Id string `json:"-" cbor:"7,keyasint"`
Image string `json:"-" cbor:"8,keyasint"`
Ports string `json:"-" cbor:"10,keyasint"`
// PrevCpu [2]uint64 `json:"-"`
CpuSystem uint64 `json:"-"`
CpuContainer uint64 `json:"-"`

View File

@@ -1,102 +0,0 @@
package monitor
import "time"
// MaxProbeTimeout is the longest agent probe timeout (currently HTTP).
// Hub requests that run a probe must allow this time in addition to transport overhead.
const MaxProbeTimeout = 10 * time.Second
type SyncAction uint8
const (
// SyncActionReplace indicates a full sync where the provided configs should replace all existing monitors for the system.
SyncActionReplace SyncAction = iota
// SyncActionUpsert indicates an incremental sync where the provided config should be added or updated.
SyncActionUpsert
// SyncActionDelete indicates an incremental sync where the provided config should be removed.
SyncActionDelete
)
// Config defines a network monitor task sent from hub to agent.
type Config struct {
// ID is the stable network_monitors record ID generated by the hub.
ID string `cbor:"0,keyasint"`
Target string `cbor:"1,keyasint"`
Protocol string `cbor:"2,keyasint"` // "icmp", "tcp", "http", or "dns"
Port uint16 `cbor:"3,keyasint,omitempty"`
Interval uint16 `cbor:"4,keyasint"` // seconds
}
// SyncRequest defines an incremental or full monitor sync request sent to the agent.
type SyncRequest struct {
Action SyncAction `cbor:"0,keyasint"`
Config Config `cbor:"1,keyasint,omitempty"`
Configs []Config `cbor:"2,keyasint,omitempty"`
RunNow bool `cbor:"3,keyasint,omitempty"`
}
// SyncResponse returns the immediate result for an upsert when requested.
type SyncResponse struct {
Result Result `cbor:"0,keyasint,omitempty"`
}
// Result holds aggregated monitor results for a single target.
//
// 0: avg response in microseconds
//
// 1: 1h average response in microseconds
//
// 2: min response in microseconds
//
// 3: 1h min response in microseconds
//
// 4: max response in microseconds
//
// 5: 1h max response in microseconds
//
// 6: packet loss percentage (0-100)
//
// 7: 1h packet loss percentage (0-100)
type Result struct {
AvgResponse int64 `cbor:"0,keyasint,omitempty"`
AvgResponse1h int64 `cbor:"1,keyasint,omitempty"`
MinResponse int64 `cbor:"2,keyasint,omitempty"`
MinResponse1h int64 `cbor:"3,keyasint,omitempty"`
MaxResponse int64 `cbor:"4,keyasint,omitempty"`
MaxResponse1h int64 `cbor:"5,keyasint,omitempty"`
PacketLoss float64 `cbor:"6,keyasint,omitempty"`
PacketLoss1h float64 `cbor:"7,keyasint,omitempty"`
// LastProbeAt is the latest completed probe's Unix timestamp in milliseconds.
LastProbeAt int64 `cbor:"8,keyasint"`
// SampleCount includes all completed probes since this monitor started.
// Used for alert warm-up even when the interval is longer than 20 minutes.
SampleCount int64 `cbor:"9,keyasint,omitempty"`
// Counts and sum cover the current response window (or latest-sample
// fallback), not the hourly window or lifetime SampleCount.
TotalCount int64 `cbor:"10,keyasint"`
SuccessCount int64 `cbor:"11,keyasint"`
ResponseSum int64 `cbor:"12,keyasint"`
}
// Stats holds response times in microseconds and packet loss percentage (0-100).
type Stats struct {
ResAvg float64 `json:"res_avg" db:"-"` // Derived for display; not stored.
ResMin float64 `json:"res_min" db:"res_min"`
ResMax float64 `json:"res_max" db:"res_max"`
Loss float64 `json:"loss" db:"-"` // Derived for display; not stored.
TotalCount int64 `json:"-" db:"total_count"`
SuccessCount int64 `json:"-" db:"success_count"`
ResponseSum int64 `json:"-" db:"res_sum"`
}
func (s Stats) FromResult(result Result) Stats {
return Stats{
ResAvg: float64(result.AvgResponse),
ResMin: float64(result.MinResponse),
ResMax: float64(result.MaxResponse),
Loss: result.PacketLoss,
TotalCount: result.TotalCount,
SuccessCount: result.SuccessCount,
ResponseSum: result.ResponseSum,
}
}

View File

@@ -7,7 +7,6 @@ import (
"time"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/systemd"
)
@@ -211,6 +210,5 @@ type CombinedData struct {
Details *Details `cbor:"4,keyasint,omitempty"`
// SystemdServicesUpdated distinguishes a fresh empty snapshot from a response
// that omitted systemd data (for example, a short-cache dashboard request).
SystemdServicesUpdated bool `json:"systemdUpdated,omitempty" cbor:"5,keyasint,omitempty"`
Monitors map[string]monitor.Result `cbor:"6,keyasint"`
SystemdServicesUpdated bool `json:"systemdUpdated,omitempty" cbor:"5,keyasint,omitempty"`
}

View File

@@ -978,18 +978,11 @@ func TestAgentWebSocketIntegration(t *testing.T) {
}
}
// A connected WebSocket does not mean the hub has finished verifying
// the agent and updating the system. Wait for the database state rather
// than assuming that work completes within a fixed sleep under load.
var status string
require.EventuallyWithT(t, func(c *assert.CollectT) {
updatedSystemRecord, err := testApp.FindRecordById("systems", systemRecord.Id)
if !assert.NoError(c, err) {
return
}
status = updatedSystemRecord.GetString("status")
assert.Equal(c, tc.expectSystemStatus, status, "System status should match expected value")
}, 5*time.Second, 20*time.Millisecond)
// Verify system status
updatedSystemRecord, err := testApp.FindRecordById("systems", systemRecord.Id)
require.NoError(t, err)
status := updatedSystemRecord.GetString("status")
assert.Equal(t, tc.expectSystemStatus, status, "System status should match expected value")
t.Logf("%s - System status: %s, Fingerprint: %s", tc.description, status, finalFingerprint)
})
@@ -1149,43 +1142,42 @@ func TestMultipleSystemsWithSameUniversalToken(t *testing.T) {
// Verify system creation/reuse behavior
if tc.expectConnection {
expectedSystemsAfter := systemsBeforeCount
// Count systems after connection
systemsAfter, err := testApp.FindRecordsByFilter("systems", "users ~ {:userId}", "", -1, 0, map[string]any{"userId": userRecord.Id})
require.NoError(t, err)
systemsAfterCount := len(systemsAfter)
if tc.expectNewSystem {
expectedSystemsAfter++
// Should have created a new system
systemCount++
assert.Equal(t, systemsBeforeCount+1, systemsAfterCount, "Should have created a new system")
assert.Equal(t, systemCount, systemsAfterCount, "Total system count should match expected")
} else {
// Should have reused existing system
assert.Equal(t, systemsBeforeCount, systemsAfterCount, "Should not have created a new system")
assert.Equal(t, systemCount, systemsAfterCount, "Total system count should remain the same")
}
// WebSocket connection precedes the hub's asynchronous system
// setup. Re-read all database state until setup is complete.
var systemId, status string
require.EventuallyWithT(t, func(c *assert.CollectT) {
systemsAfter, err := testApp.FindRecordsByFilter("systems", "users ~ {:userId}", "", -1, 0, map[string]any{"userId": userRecord.Id})
if !assert.NoError(c, err) {
return
}
assert.Len(c, systemsAfter, expectedSystemsAfter, "System creation/reuse should match expected behavior")
assert.Len(c, systemsAfter, systemCount, "Total system count should match expected")
time.Sleep(20 * time.Millisecond)
fingerprints, err := testApp.FindRecordsByFilter("fingerprints", "token = {:token} && fingerprint = {:fingerprint}", "", -1, 0, map[string]any{
"token": universalToken,
"fingerprint": tc.agentFingerprint,
})
if !assert.NoError(c, err) || !assert.Len(c, fingerprints, 1, "Should have exactly one fingerprint record for this token+fingerprint combination") {
return
}
// Verify that a fingerprint record exists for this fingerprint
fingerprints, err := testApp.FindRecordsByFilter("fingerprints", "token = {:token} && fingerprint = {:fingerprint}", "", -1, 0, map[string]any{
"token": universalToken,
"fingerprint": tc.agentFingerprint,
})
require.NoError(t, err)
require.Len(t, fingerprints, 1, "Should have exactly one fingerprint record for this token+fingerprint combination")
fingerprint := fingerprints[0]
assert.Equal(c, universalToken, fingerprint.GetString("token"), "Fingerprint should have the universal token")
assert.Equal(c, tc.agentFingerprint, fingerprint.GetString("fingerprint"), "Fingerprint should match agent's fingerprint")
fingerprint := fingerprints[0]
assert.Equal(t, universalToken, fingerprint.GetString("token"), "Fingerprint should have the universal token")
assert.Equal(t, tc.agentFingerprint, fingerprint.GetString("fingerprint"), "Fingerprint should match agent's fingerprint")
systemId = fingerprint.GetString("system")
system, err := testApp.FindRecordById("systems", systemId)
if !assert.NoError(c, err) {
return
}
status = system.GetString("status")
assert.Equal(c, tc.expectSystemStatus, status, "System status should match expected value")
}, 5*time.Second, 20*time.Millisecond)
// Verify system status
systemId := fingerprint.GetString("system")
system, err := testApp.FindRecordById("systems", systemId)
require.NoError(t, err)
status := system.GetString("status")
assert.Equal(t, tc.expectSystemStatus, status, "System status should match expected value")
t.Logf("%s - System ID: %s, Status: %s, New System: %v", tc.description, systemId, status, tc.expectNewSystem)
}

View File

@@ -2,17 +2,13 @@ package hub
import (
"context"
"fmt"
"log/slog"
"net"
"net/http"
"net/netip"
"regexp"
"strings"
"time"
"uuid"
"github.com/blang/semver"
"github.com/google/uuid"
"github.com/henrygd/beszel"
"github.com/henrygd/beszel/internal/alerts"
"github.com/henrygd/beszel/internal/ghupdate"
@@ -82,81 +78,12 @@ func (h *Hub) registerMiddlewares(se *core.ServeEvent) {
}
// authenticate with trusted header
if trustedHeader, _ := utils.GetEnv("TRUSTED_AUTH_HEADER"); trustedHeader != "" {
// only honor the header from these peers, if set
trustedProxies, restricted := parseTrustedProxies()
se.Router.BindFunc(func(e *core.RequestEvent) error {
if restricted && !isTrustedProxy(trustedProxies, e.Request.RemoteAddr) {
return e.Next()
}
return authorizeRequestWithEmail(e, e.Request.Header.Get(trustedHeader))
})
}
}
// parseTrustedProxies reads TRUSTED_PROXY_IPS (comma-separated IPs or CIDRs).
// restricted is false when the variable is unset or empty, meaning the trusted
// header is accepted from any peer. Invalid entries are skipped with a warning,
// so a typo narrows the allowlist rather than widening it.
func parseTrustedProxies() (prefixes []netip.Prefix, restricted bool) {
value, _ := utils.GetEnv("TRUSTED_PROXY_IPS")
if value == "" {
return nil, false
}
for entry := range strings.SplitSeq(value, ",") {
entry = strings.TrimSpace(entry)
if entry == "" {
continue
}
if prefix, err := parseProxyPrefix(entry); err == nil {
prefixes = append(prefixes, prefix)
} else {
slog.Warn("Ignoring invalid TRUSTED_PROXY_IPS entry", "entry", entry)
}
}
return prefixes, true
}
// parseProxyPrefix parses an IP or CIDR into a masked prefix. IPv4-mapped IPv6
// entries are converted to IPv4 so they match IPv4 peers.
func parseProxyPrefix(entry string) (netip.Prefix, error) {
prefix, err := netip.ParsePrefix(entry)
if err != nil {
addr, err := netip.ParseAddr(entry)
if err != nil {
return netip.Prefix{}, err
}
addr = addr.Unmap()
return netip.PrefixFrom(addr, addr.BitLen()), nil
}
if prefix.Addr().Is4In6() {
if prefix.Bits() < 96 {
return netip.Prefix{}, fmt.Errorf("%s covers more than the IPv4-mapped range", entry)
}
prefix = netip.PrefixFrom(prefix.Addr().Unmap(), prefix.Bits()-96)
}
return prefix.Masked(), nil
}
// isTrustedProxy reports whether the peer address of a request (host:port) is
// within one of the prefixes.
func isTrustedProxy(prefixes []netip.Prefix, remoteAddr string) bool {
host, _, err := net.SplitHostPort(remoteAddr)
if err != nil {
host = remoteAddr
}
addr, err := netip.ParseAddr(host)
if err != nil {
return false
}
addr = addr.Unmap().WithZone("")
for _, prefix := range prefixes {
if prefix.Contains(addr) {
return true
}
}
return false
}
// registerApiRoutes registers custom API routes
func (h *Hub) registerApiRoutes(se *core.ServeEvent) error {
// auth protected routes

View File

@@ -6,16 +6,12 @@ import (
"fmt"
"io"
"net/http"
"net/http/httptest"
"sort"
"testing"
"time"
beszelTests "github.com/henrygd/beszel/internal/tests"
"github.com/henrygd/beszel/internal/migrations"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/apis"
"github.com/pocketbase/pocketbase/core"
pbTests "github.com/pocketbase/pocketbase/tests"
"github.com/stretchr/testify/require"
@@ -30,59 +26,6 @@ func jsonReader(v any) io.Reader {
return bytes.NewReader(data)
}
type gatedReader struct {
data []byte
started chan struct{}
release chan struct{}
offset int
}
func (r *gatedReader) Read(p []byte) (int, error) {
if r.offset == 0 {
close(r.started)
<-r.release
}
if r.offset >= len(r.data) {
return 0, io.EOF
}
n := copy(p, r.data[r.offset:])
r.offset += n
return n, nil
}
func firstUserTestMux(t *testing.T) (*beszelTests.TestHub, http.Handler) {
t.Helper()
hub, err := beszelTests.NewTestHub(t.TempDir())
require.NoError(t, err)
_ = hub.StartHub()
router, err := apis.NewRouter(hub.TestApp)
require.NoError(t, err)
serveEvent := &core.ServeEvent{App: hub.TestApp, Router: router}
var handler http.Handler
err = hub.TestApp.OnServe().Trigger(serveEvent, func(e *core.ServeEvent) error {
var buildErr error
handler, buildErr = e.Router.BuildMux()
return buildErr
})
require.NoError(t, err)
require.NotNil(t, handler)
return hub, handler
}
func postFirstUser(handler http.Handler, email string) *httptest.ResponseRecorder {
body, _ := json.Marshal(map[string]string{
"email": email,
"password": "password123",
})
req := httptest.NewRequest(http.MethodPost, "/api/beszel/create-user", bytes.NewReader(body))
req.Header.Set("Content-Type", "application/json")
recorder := httptest.NewRecorder()
handler.ServeHTTP(recorder, req)
return recorder
}
func TestApiRoutesAuthentication(t *testing.T) {
hub, user := beszelTests.GetHubWithUser(t)
defer hub.Cleanup()
@@ -846,87 +789,6 @@ func TestFirstUserCreation(t *testing.T) {
})
}
func TestFirstUserBootstrapAtomicity(t *testing.T) {
t.Run("concurrent complete requests produce exactly one winner", func(t *testing.T) {
hub, handler := firstUserTestMux(t)
defer hub.Cleanup()
start := make(chan struct{})
statuses := make(chan int, 2)
for _, email := range []string{"first@example.com", "second@example.com"} {
go func(email string) {
<-start
statuses <- postFirstUser(handler, email).Code
}(email)
}
close(start)
got := []int{<-statuses, <-statuses}
sort.Ints(got)
require.Equal(t, []int{http.StatusOK, http.StatusForbidden}, got)
users, err := hub.FindAllRecords("users")
require.NoError(t, err)
require.Len(t, users, 1)
superusers, err := hub.FindAllRecords(core.CollectionNameSuperusers)
require.NoError(t, err)
require.Len(t, superusers, 1)
require.NotEqual(t, migrations.TempAdminEmail, superusers[0].Email())
})
t.Run("partial body cannot retain stale bootstrap authorization", func(t *testing.T) {
hub, handler := firstUserTestMux(t)
defer hub.Cleanup()
body, err := json.Marshal(map[string]string{
"email": "parked@example.com",
"password": "password123",
})
require.NoError(t, err)
gated := &gatedReader{
data: body,
started: make(chan struct{}),
release: make(chan struct{}),
}
parkedRequest := httptest.NewRequest(http.MethodPost, "/api/beszel/create-user", gated)
parkedRequest.Header.Set("Content-Type", "application/json")
parkedRecorder := httptest.NewRecorder()
parkedDone := make(chan struct{})
go func() {
handler.ServeHTTP(parkedRecorder, parkedRequest)
close(parkedDone)
}()
select {
case <-gated.started:
case <-time.After(2 * time.Second):
t.Fatal("parked request did not begin reading its body")
}
operatorRecorder := postFirstUser(handler, "operator@example.com")
require.Equal(t, http.StatusOK, operatorRecorder.Code)
lateRecorder := postFirstUser(handler, "late@example.com")
require.Equal(t, http.StatusForbidden, lateRecorder.Code)
close(gated.release)
select {
case <-parkedDone:
case <-time.After(2 * time.Second):
t.Fatal("parked request did not finish")
}
require.Equal(t, http.StatusForbidden, parkedRecorder.Code)
users, err := hub.FindAllRecords("users")
require.NoError(t, err)
require.Len(t, users, 1)
require.Equal(t, "operator@example.com", users[0].Email())
superusers, err := hub.FindAllRecords(core.CollectionNameSuperusers)
require.NoError(t, err)
require.Len(t, superusers, 1)
require.Equal(t, "operator@example.com", superusers[0].Email())
})
}
func TestCreateUserEndpointAvailability(t *testing.T) {
t.Run("CreateUserEndpoint available when no users exist", func(t *testing.T) {
hub, _ := beszelTests.NewTestHub(t.TempDir())
@@ -1107,79 +969,6 @@ func TestTrustedHeaderMiddleware(t *testing.T) {
}
}
func TestTrustedHeaderProxyAllowlist(t *testing.T) {
var hubs []*beszelTests.TestHub
defer func() {
for _, hub := range hubs {
hub.Cleanup()
}
}()
testAppFactory := func(t testing.TB) *pbTests.TestApp {
hub, _ := beszelTests.NewTestHub(t.TempDir())
hubs = append(hubs, hub)
hub.StartHub()
return hub.TestApp
}
// httptest requests arrive from 192.0.2.1:1234
testCases := []struct {
name string
proxies string
expectedStatus int
expectedContent []string
}{
{
name: "peer inside an allowed range",
proxies: "10.0.0.0/8, 192.0.2.0/24",
expectedStatus: 200,
expectedContent: []string{"\"key\":", "\"v\":"},
},
{
name: "peer is the listed address",
proxies: "192.0.2.1",
expectedStatus: 200,
expectedContent: []string{"\"key\":", "\"v\":"},
},
{
name: "peer outside the allowlist",
proxies: "10.0.0.0/8",
expectedStatus: 401,
expectedContent: []string{"requires valid"},
},
{
name: "allowlist with no valid entry",
proxies: "proxy.internal",
expectedStatus: 401,
expectedContent: []string{"requires valid"},
},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
t.Setenv("TRUSTED_AUTH_HEADER", "X-Beszel-Trusted")
t.Setenv("TRUSTED_PROXY_IPS", tc.proxies)
scenario := beszelTests.ApiScenario{
Name: "GET /getkey - with trusted header",
Method: http.MethodGet,
URL: "/api/beszel/getkey",
Headers: map[string]string{
"X-Beszel-Trusted": "user@test.com",
},
ExpectedStatus: tc.expectedStatus,
ExpectedContent: tc.expectedContent,
TestAppFactory: testAppFactory,
BeforeTestFunc: func(t testing.TB, app *pbTests.TestApp, e *core.ServeEvent) {
beszelTests.CreateUser(app, "user@test.com", "password123")
},
}
scenario.Test(t)
})
}
}
func TestUpdateEndpoint(t *testing.T) {
t.Setenv("CHECK_UPDATES", "true")

View File

@@ -78,7 +78,7 @@ func setCollectionAuthSettings(app core.App) error {
return err
}
if err := applyCollectionRules(app, []string{"containers", "container_stats", "system_stats", "systemd_services", "network_monitor_stats"}, collectionRules{
if err := applyCollectionRules(app, []string{"containers", "container_stats", "system_stats", "systemd_services"}, collectionRules{
list: &systemScopedReadRule,
}); err != nil {
return err
@@ -108,16 +108,6 @@ func setCollectionAuthSettings(app core.App) error {
return err
}
if err := applyCollectionRules(app, []string{"network_monitors"}, collectionRules{
list: &systemScopedReadRule,
view: &systemScopedReadRule,
create: &systemScopedWriteRule,
update: &systemScopedWriteRule,
delete: &systemScopedWriteRule,
}); err != nil {
return err
}
if err := applyCollectionRules(app, []string{"system_details"}, collectionRules{
list: &systemScopedReadRule,
view: &systemScopedReadRule,

View File

@@ -6,8 +6,8 @@ import (
"log"
"os"
"path/filepath"
"uuid"
"github.com/google/uuid"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/pocketbase/dbx"

View File

@@ -106,12 +106,9 @@ func (h *Hub) StartHub() error {
// TODO: move to users package
// handle default values for user / user_settings creation
h.App.OnRecordAuthWithOAuth2Request("users").BindFunc(h.um.InitializeOAuthUserRole)
h.App.OnRecordCreate("users").BindFunc(h.um.InitializeUserRole)
h.App.OnRecordCreate("user_settings").BindFunc(h.um.InitializeUserSettings)
bindNetworkMonitorsEvents(h)
pb, ok := h.App.(*pocketbase.PocketBase)
if !ok {
return errors.New("not a pocketbase app")
@@ -125,8 +122,6 @@ func (h *Hub) initialize(app core.App) error {
settings := app.Settings()
// batch requests (for alerts)
settings.Batch.Enabled = true
settings.Batch.MaxRequests = 100
settings.Batch.MaxBodySize = 1 << 20 // 1 MiB
// set URL if APP_URL env is set
if appURL, isSet := utils.GetEnv("APP_URL"); isSet {
h.appURL = appURL

View File

@@ -1,158 +0,0 @@
package hub
import (
"strconv"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/hub/systems"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/types"
)
// generateMonitorID creates a stable hash ID for a monitor based on its configuration and the system it belongs to.
func generateMonitorID(systemId string, config monitor.Config) string {
args := []string{systemId, config.Target, config.Protocol}
// only use port for TCP monitors, since for other protocols it's not relevant as standalone value
if config.Protocol == "tcp" {
args = append(args, strconv.FormatUint(uint64(config.Port), 10))
}
return systems.MakeStableHashId(args...)
}
// bindNetworkMonitorsEvents keeps monitor records and agent monitor state in sync.
func bindNetworkMonitorsEvents(hub *Hub) {
// on create, make sure the id is set to a stable hash
hub.OnRecordCreate("network_monitors").BindFunc(func(e *core.RecordEvent) error {
systemID := e.Record.GetString("system")
config := monitorConfigFromRecord(e.Record)
id := generateMonitorID(systemID, *config)
e.Record.Set("id", id)
return e.Next()
})
// sync monitor to agent on creation and persist the first result immediately when available
hub.OnRecordAfterCreateSuccess("network_monitors").BindFunc(func(e *core.RecordEvent) error {
err := e.Next()
if err != nil {
return err
}
if !e.Record.GetBool("enabled") {
return nil
}
// If connected, run the monitor immediately. Paused systems may be absent
// from the manager; their monitors will sync when they reconnect.
system, err := hub.sm.GetSystem(e.Record.GetString("system"))
if err == nil && system.Status == "up" {
go hub.upsertNetworkMonitor(e.Record, true)
}
return nil
})
// On API update requests, if the monitor config changed in a way that requires a new ID, create a new
// record with the new ID and delete the old one. Otherwise, just update the existing monitor on the agent.
hub.OnRecordUpdateRequest("network_monitors").BindFunc(func(e *core.RecordRequestEvent) error {
systemID := e.Record.GetString("system")
// only tcp uses port - set other protocols port to zero
if e.Record.GetString("protocol") != "tcp" {
e.Record.Set("port", 0)
}
ID := generateMonitorID(systemID, *monitorConfigFromRecord(e.Record))
if ID != e.Record.Id {
newRecord := copyMonitorToNewRecord(e.Record, ID)
if err := e.App.Save(newRecord); err != nil {
return err
}
if err := e.App.Delete(e.Record); err != nil {
return err
}
return nil
}
err := e.Next()
if err != nil {
return err
}
if e.Record.GetBool("enabled") {
// if the monitor is enabled, sync the updated config to the agent now
runNow := !e.Record.Original().GetBool("enabled")
err = hub.upsertNetworkMonitor(e.Record, runNow)
} else {
// if the monitor is paused, remove it from the agent
err = hub.deleteNetworkMonitor(e.Record)
}
if err != nil {
hub.Logger().Warn("failed to sync updated monitor", "system", systemID, "monitor", e.Record.Id, "err", err)
}
return nil
})
// sync monitor to agent on delete
hub.OnRecordAfterDeleteSuccess("network_monitors").BindFunc(func(e *core.RecordEvent) error {
if err := hub.deleteNetworkMonitor(e.Record); err != nil {
hub.Logger().Warn("failed to delete monitor on agent", "system", e.Record.GetString("system"), "monitor", e.Record.Id, "err", err)
}
return e.Next()
})
}
// monitorConfigFromRecord builds a monitor config from a network_monitors record.
func monitorConfigFromRecord(record *core.Record) *monitor.Config {
return &monitor.Config{
ID: record.Id,
Target: record.GetString("target"),
Protocol: record.GetString("protocol"),
Port: uint16(record.GetInt("port")),
Interval: uint16(record.GetInt("interval")),
}
}
// setMonitorResultFields stores the latest monitor result values on the record.
func setMonitorResultFields(record *core.Record, result monitor.Result) {
nowString := time.Now().UTC().Format(types.DefaultDateLayout)
record.Set("res", result.AvgResponse)
record.Set("resAvg1h", result.AvgResponse1h)
record.Set("resMin1h", result.MinResponse1h)
record.Set("resMax1h", result.MaxResponse1h)
record.Set("loss1h", result.PacketLoss1h)
record.Set("updated", nowString)
}
// copyMonitorToNewRecord creates a new record with the same field values as the old one.
// This is used when the monitor config changes in a way that requires a new ID, so we need
// to create a new record with the new ID and delete the old one.
func copyMonitorToNewRecord(oldRecord *core.Record, newID string) *core.Record {
collection := oldRecord.Collection()
newRecord := core.NewRecord(collection)
newRecord.Id = newID
fields := []string{"system", "target", "protocol", "port", "interval", "enabled"}
for _, field := range fields {
newRecord.Set(field, oldRecord.Get(field))
}
return newRecord
}
// upsertNetworkMonitor creates or updates the record's monitor on the target system. If runNow
// is true, it will also trigger an immediate monitor run and update the record with the result.
func (h *Hub) upsertNetworkMonitor(record *core.Record, runNow bool) error {
systemID := record.GetString("system")
system, err := h.sm.GetSystem(systemID)
if err != nil {
return err
}
result, err := system.UpsertNetworkMonitor(*monitorConfigFromRecord(record), runNow)
if err != nil || result == nil {
return err
}
setMonitorResultFields(record, *result)
return h.App.SaveNoValidate(record)
}
// deleteNetworkMonitor removes the record's monitor from the target system.
func (h *Hub) deleteNetworkMonitor(record *core.Record) error {
systemID := record.GetString("system")
system, err := h.sm.GetSystem(systemID)
if err != nil {
return err
}
return system.DeleteNetworkMonitor(record.Id)
}

View File

@@ -1,225 +0,0 @@
package hub
import (
"bytes"
"encoding/json"
"net/http"
"net/http/httptest"
"testing"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/pocketbase/pocketbase/apis"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestCreateNetworkMonitorsOnPausedSystem(t *testing.T) {
for _, batch := range []bool{false, true} {
name := "single"
if batch {
name = "batch"
}
t.Run(name, func(t *testing.T) {
hub, testApp, err := createTestHub(t)
require.NoError(t, err)
defer cleanupTestHub(hub, testApp)
bindNetworkMonitorsEvents(hub)
user, err := createTestUser(hub)
require.NoError(t, err)
system, err := createTestRecord(hub, "systems", map[string]any{
"name": "Paused", "host": "localhost", "port": "45876",
"status": "paused", "users": []string{user.Id},
})
require.NoError(t, err)
// Paused systems are not loaded into the manager at startup.
_, err = hub.sm.GetSystem(system.Id)
require.Error(t, err)
payload := func(target string) map[string]any {
return map[string]any{
"system": system.Id, "target": target, "protocol": "icmp",
"interval": 60, "enabled": true,
}
}
url := "/api/collections/network_monitors/records"
var body any = payload("1.1.1.1")
count := 1
if batch {
body = map[string]any{"requests": []map[string]any{
{"method": "POST", "url": url, "body": payload("1.1.1.1")},
{"method": "POST", "url": url, "body": payload("8.8.8.8")},
}}
url = "/api/batch"
count = 2
}
data, err := json.Marshal(body)
require.NoError(t, err)
token, err := user.NewAuthToken()
require.NoError(t, err)
router, err := apis.NewRouter(hub)
require.NoError(t, err)
handler, err := router.BuildMux()
require.NoError(t, err)
request := httptest.NewRequest(http.MethodPost, url, bytes.NewReader(data))
request.Header.Set("Content-Type", "application/json")
request.Header.Set("Authorization", token)
response := httptest.NewRecorder()
handler.ServeHTTP(response, request)
assert.Equal(t, http.StatusOK, response.Code, response.Body.String())
records, err := hub.FindAllRecords("network_monitors")
require.NoError(t, err)
require.Len(t, records, count)
for _, record := range records {
assert.Equal(t, system.Id, record.GetString("system"))
assert.True(t, record.GetBool("enabled"))
}
})
}
}
func TestGenerateMonitorID(t *testing.T) {
tests := []struct {
name string
systemID string
config monitor.Config
expected string
}{
{
name: "HTTP monitor on example.com",
systemID: "sys123",
config: monitor.Config{
Protocol: "http",
Target: "example.com",
Port: 0,
Interval: 60,
},
expected: "a20a5827",
},
{
name: "HTTP monitor on example.com with different port",
systemID: "sys123",
config: monitor.Config{
Protocol: "http",
Target: "example.com",
Port: 8080,
Interval: 60,
},
expected: "a20a5827",
},
{
name: "HTTP monitor on example.com with different system ID",
systemID: "sys1234",
config: monitor.Config{
Protocol: "http",
Target: "example.com",
Port: 80,
Interval: 60,
},
expected: "ab602ae7",
},
{
name: "Same monitor, different interval",
systemID: "sys1234",
config: monitor.Config{
Protocol: "http",
Target: "example.com",
Port: 80,
Interval: 120,
},
expected: "ab602ae7",
},
{
name: "ICMP monitor on 1.1.1.1",
systemID: "sys456",
config: monitor.Config{
Protocol: "icmp",
Target: "1.1.1.1",
Port: 0,
Interval: 10,
},
expected: "6d13a4a4",
}, {
name: "ICMP monitor on 1.1.1.1 with different system ID",
systemID: "sys4567",
config: monitor.Config{
Protocol: "icmp",
Target: "1.1.1.1",
Port: 0,
Interval: 10,
},
expected: "ddd6c81",
},
{
name: "TCP monitor on example.com with port 443",
systemID: "sys789",
config: monitor.Config{
Protocol: "tcp",
Target: "example.com",
Port: 443,
Interval: 30,
},
expected: "677b991",
},
{
name: "TCP monitor on example.com with port 8443",
systemID: "sys789",
config: monitor.Config{
Protocol: "tcp",
Target: "example.com",
Port: 8443,
Interval: 30,
},
expected: "84167969",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := generateMonitorID(tt.systemID, tt.config)
assert.Equal(t, tt.expected, got, "generateMonitorID() = %v, want %v", got, tt.expected)
})
}
}
func TestCopyMonitorToNewRecordDropsResultFields(t *testing.T) {
hub, testApp, err := createTestHub(t)
require.NoError(t, err)
defer cleanupTestHub(hub, testApp)
collection, err := hub.FindCachedCollectionByNameOrId("network_monitors")
require.NoError(t, err)
assert.Nil(t, collection.Fields.GetByName("name"))
oldRecord := core.NewRecord(collection)
oldRecord.Load(map[string]any{
"system": "sys123",
"target": "https://example.com",
"protocol": "http",
"port": 443,
"interval": 60,
"enabled": true,
"res": 1200,
"resAvg1h": 1300,
"resMin1h": 900,
"resMax1h": 1600,
"loss1h": 5,
"updated": "2026-04-29 12:00:00.000Z",
})
newRecord := copyMonitorToNewRecord(oldRecord, "next12345")
assert.Equal(t, "next12345", newRecord.Id)
assert.Equal(t, "https://example.com", newRecord.GetString("target"))
assert.Equal(t, "http", newRecord.GetString("protocol"))
assert.Equal(t, 443, newRecord.GetInt("port"))
assert.True(t, newRecord.GetBool("enabled"))
assert.Zero(t, newRecord.GetFloat("res"))
assert.Zero(t, newRecord.GetFloat("resAvg1h"))
assert.Zero(t, newRecord.GetFloat("resMin1h"))
assert.Zero(t, newRecord.GetFloat("resMax1h"))
assert.Zero(t, newRecord.GetFloat("loss1h"))
assert.Equal(t, "", newRecord.GetString("updated"))
}

View File

@@ -1,46 +0,0 @@
//go:build testing
package systems
import (
"testing"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/pocketbase/dbx"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestCreateContainerRecordsPersistsImageUpdateAvailability(t *testing.T) {
_, app := newTestSystemWithHub(t)
const (
systemID = "system123"
containerID = "abcdef123456"
image = "nginx:latest"
)
data := &container.Stats{
Id: containerID,
Name: "web",
Image: image,
UpdateAvailable: true,
}
require.NoError(t, createContainerRecords(app, []*container.Stats{data}, systemID))
var record struct {
Image string `db:"image"`
UpdateAvailable bool `db:"updatable"`
}
require.NoError(t, app.DB().Select("image", "updatable").From("containers").
Where(dbx.HashExp{"id": containerID}).One(&record))
assert.Equal(t, image, record.Image)
assert.True(t, record.UpdateAvailable)
data.UpdateAvailable = false
require.NoError(t, createContainerRecords(app, []*container.Stats{data}, systemID))
require.NoError(t, app.DB().Select("image", "updatable").From("containers").
Where(dbx.HashExp{"id": containerID}).One(&record))
assert.Equal(t, image, record.Image)
assert.False(t, record.UpdateAvailable)
}

View File

@@ -1,159 +0,0 @@
//go:build testing
package systems
import (
"context"
"crypto/ed25519"
"crypto/rand"
"net"
"sync/atomic"
"testing"
"time"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/monitor"
esystem "github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/hub/expirymap"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/require"
"golang.org/x/crypto/ssh"
)
func TestSSHNetworkMonitorReconnectSync(t *testing.T) {
sys, app := newTestSystemWithHub(t)
sys.manager.zfsFetchMap = expirymap.New[zfsFetchState](time.Hour)
t.Cleanup(sys.manager.zfsFetchMap.StopCleaner)
sys.ctx = context.Background()
sys.Status = up
_, key, err := ed25519.GenerateKey(rand.Reader)
require.NoError(t, err)
signer, err := ssh.NewSignerFromKey(key)
require.NoError(t, err)
config := &ssh.ServerConfig{NoClientAuth: true, ServerVersion: "SSH-2.0-beszel_0.20.0"}
config.AddHostKey(signer)
listener, err := net.Listen("tcp", "127.0.0.1:0")
require.NoError(t, err)
t.Cleanup(func() { _ = listener.Close() })
sys.Host, sys.Port, err = net.SplitHostPort(listener.Addr().String())
require.NoError(t, err)
sys.manager.sshConfig = &ssh.ClientConfig{User: "test", HostKeyCallback: ssh.InsecureIgnoreHostKey(), Timeout: time.Second}
t.Cleanup(sys.closeSSHConnection)
requests := make(chan monitor.SyncRequest, 10)
var failSync atomic.Bool
go func() {
for {
conn, err := listener.Accept()
if err != nil {
return
}
go func() {
server, channels, reqs, err := ssh.NewServerConn(conn, config)
if err != nil {
_ = conn.Close()
return
}
defer server.Close()
go ssh.DiscardRequests(reqs)
for channel := range channels {
ch, reqs, err := channel.Accept()
if err != nil {
return
}
go func() {
defer ch.Close()
for req := range reqs {
if req.Type != "shell" {
_ = req.Reply(false, nil)
continue
}
_ = req.Reply(true, nil)
var request common.HubRequest[cbor.RawMessage]
if cbor.NewDecoder(ch).Decode(&request) != nil {
return
}
response := common.AgentResponse{}
switch request.Action {
case common.GetData:
response.SystemData = &esystem.CombinedData{}
case common.SyncNetworkMonitors:
var syncReq monitor.SyncRequest
if cbor.Unmarshal(request.Data, &syncReq) != nil {
return
}
requests <- syncReq
if failSync.Load() {
response.Error = "test sync failure"
} else {
response.Data, _ = cbor.Marshal(monitor.SyncResponse{})
}
}
_ = cbor.NewEncoder(ch).Encode(response)
_, _ = ch.SendRequest("exit-status", false, ssh.Marshal(struct{ Status uint32 }{0}))
return
}
}()
}
}()
}
}()
collection, err := app.FindCachedCollectionByNameOrId("network_monitors")
require.NoError(t, err)
probe := core.NewRecord(collection)
probe.Load(map[string]any{"system": sys.Id, "target": "localhost", "protocol": "tcp", "port": 80, "interval": 60, "enabled": true})
require.NoError(t, app.SaveNoValidate(probe))
fetch := func() {
t.Helper()
_, err := sys.fetchDataFromAgent(common.DataRequestOptions{})
require.NoError(t, err, "monitor sync failure must not fail stats fetching")
}
receive := func() monitor.SyncRequest {
t.Helper()
select {
case req := <-requests:
require.Equal(t, monitor.SyncActionReplace, req.Action)
return req
case <-time.After(time.Second):
t.Fatal("missing full monitor sync")
return monitor.SyncRequest{}
}
}
fetch()
require.Equal(t, probe.Id, receive().Configs[0].ID)
require.False(t, sys.monitorsNeedSync.Load())
fetch()
require.Empty(t, requests, "steady-state fetch must not resync")
// Simulate loss of the agent process/connection and its in-memory monitors.
require.NoError(t, sys.client.Load().Close())
fetch()
require.Equal(t, probe.Id, receive().Configs[0].ID)
require.False(t, sys.monitorsNeedSync.Load())
// Failed replacements are retried on the next successful stats fetch.
require.NoError(t, sys.client.Load().Close())
failSync.Store(true)
fetch()
receive()
require.True(t, sys.monitorsNeedSync.Load())
failSync.Store(false)
fetch()
receive()
require.False(t, sys.monitorsNeedSync.Load())
probe.Set("enabled", false)
require.NoError(t, app.SaveNoValidate(probe))
require.NoError(t, sys.client.Load().Close())
fetch()
require.Empty(t, receive().Configs, "empty replacement must clear stale monitors")
}
func TestPendingNetworkMonitorSyncQueryFailure(t *testing.T) {
sys, app := newTestSystemWithHub(t)
_, err := app.DB().NewQuery("DROP TABLE network_monitors").Execute()
require.NoError(t, err)
sys.monitorsNeedSync.Store(true)
sys.syncPendingNetworkMonitors()
require.True(t, sys.monitorsNeedSync.Load())
}

View File

@@ -1,224 +0,0 @@
//go:build testing
package systems
import (
"fmt"
"testing"
"time"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/subscriptions"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestNetworkMonitorProbePruning(t *testing.T) {
for _, tc := range []struct {
name string
monitors map[string]monitor.Result
fail bool
want map[string]int64
}{
{"nil report", nil, false, map[string]int64{"monitor1": 1000, "monitor2": 1000}},
{"empty report", map[string]monitor.Result{}, false, map[string]int64{}},
{"removed monitor", map[string]monitor.Result{"monitor1": {LastProbeAt: 1000}}, false, map[string]int64{"monitor1": 1000}},
{"rolled back report", map[string]monitor.Result{}, true, map[string]int64{"monitor1": 1000, "monitor2": 1000}},
} {
t.Run(tc.name, func(t *testing.T) {
sys, app := newTestSystemWithHub(t)
sys.lastSavedMonitorProbe = map[string]int64{"monitor1": 1000, "monitor2": 1000}
// Preserve the distinction between nil and empty across the agent transport.
encoded, err := cbor.Marshal(system.CombinedData{Monitors: tc.monitors})
require.NoError(t, err)
var data system.CombinedData
require.NoError(t, cbor.Unmarshal(encoded, &data))
if tc.fail {
_, err = app.DB().NewQuery(`CREATE TRIGGER fail_system_update BEFORE UPDATE ON systems BEGIN SELECT RAISE(ABORT, 'test rollback'); END`).Execute()
require.NoError(t, err)
}
_, err = sys.createRecords(&data)
if tc.fail {
require.Error(t, err)
} else {
require.NoError(t, err)
}
assert.Equal(t, tc.want, sys.lastSavedMonitorProbe)
})
}
}
func TestNetworkMonitorStatsFreshness(t *testing.T) {
for _, realtime := range []bool{false, true} {
name := "sql"
if realtime {
name = "realtime"
}
t.Run(name, func(t *testing.T) {
sys, app := newTestSystemWithHub(t)
if realtime {
client := subscriptions.NewDefaultClient()
client.Subscribe("network_monitors/*")
app.SubscriptionsBroker().Register(client)
t.Cleanup(func() { app.SubscriptionsBroker().Unregister(client.Id()) })
}
col, err := app.FindCachedCollectionByNameOrId("network_monitors")
require.NoError(t, err)
for _, id := range []string{"monitor1", "monitor2"} {
record := core.NewRecord(col)
record.Id = id
record.Set("system", sys.Id)
require.NoError(t, app.SaveNoValidate(record))
}
data := &system.CombinedData{Monitors: map[string]monitor.Result{
"monitor1": {LastProbeAt: 1000, AvgResponse: 20, TotalCount: 6, SuccessCount: 6, ResponseSum: 123},
"monitor2": {LastProbeAt: 1000, PacketLoss: 100, TotalCount: 1},
}}
count := func(want int64) {
t.Helper()
got, err := app.CountRecords("network_monitor_stats")
require.NoError(t, err)
assert.Equal(t, want, got)
}
save := func() {
t.Helper()
_, err := sys.createRecords(data)
require.NoError(t, err)
}
save()
count(2)
stored, err := app.FindAllRecords("network_monitor_stats")
require.NoError(t, err)
for _, record := range stored {
result := data.Monitors[record.GetString("monitor")]
assert.EqualValues(t, result.TotalCount, record.GetInt("total_count"))
assert.EqualValues(t, result.SuccessCount, record.GetInt("success_count"))
assert.EqualValues(t, result.ResponseSum, record.GetInt("res_sum"))
}
// A resume can overlap the scheduled update with the same probe.
errs := make(chan error, 4)
for range 4 {
go func() {
_, err := sys.createRecords(data)
errs <- err
}()
}
for range 4 {
require.NoError(t, <-errs)
}
count(2)
// A rolling hourly value can change without a new probe.
result := data.Monitors["monitor1"]
result.AvgResponse1h = 42
data.Monitors["monitor1"] = result
save()
count(2)
record, err := app.FindRecordById("network_monitors", "monitor1")
require.NoError(t, err)
assert.Equal(t, 42, record.GetInt("resAvg1h"))
// Identical response values and failed probes still count as new measurements.
for id, result := range data.Monitors {
result.LastProbeAt = 301000
data.Monitors[id] = result
}
save()
count(4)
// A failed individual insert must remain retryable, even if others commit.
_, err = app.DB().NewQuery(`CREATE TRIGGER fail_monitor_insert BEFORE INSERT ON network_monitor_stats WHEN NEW.monitor = 'monitor1' BEGIN SELECT RAISE(ABORT, 'test insert failure'); END`).Execute()
require.NoError(t, err)
for id, result := range data.Monitors {
result.LastProbeAt = 601000
data.Monitors[id] = result
}
save()
count(5)
assert.Equal(t, int64(301000), sys.lastSavedMonitorProbe["monitor1"])
assert.Equal(t, int64(601000), sys.lastSavedMonitorProbe["monitor2"])
_, err = app.DB().NewQuery("DROP TRIGGER fail_monitor_insert").Execute()
require.NoError(t, err)
save()
count(6)
// Failure after inserting stats rolls back the whole transaction and its markers.
_, err = app.DB().NewQuery(`CREATE TRIGGER fail_system_update BEFORE UPDATE ON systems BEGIN SELECT RAISE(ABORT, 'test rollback'); END`).Execute()
require.NoError(t, err)
result = data.Monitors["monitor1"]
result.LastProbeAt = 901000
data.Monitors["monitor1"] = result
_, err = sys.createRecords(data)
require.Error(t, err)
count(6)
assert.Equal(t, int64(601000), sys.lastSavedMonitorProbe["monitor1"])
_, err = app.DB().NewQuery("DROP TRIGGER fail_system_update").Execute()
require.NoError(t, err)
save()
count(7)
// Clock rollback is a new probe identity, not a reason to stall writes.
result.LastProbeAt = 500
data.Monitors["monitor1"] = result
save()
count(8)
// Recreated systems intentionally accept the first result without restoring state.
sys = &System{Id: sys.Id, manager: sys.manager}
save()
count(10)
})
}
}
// Observes the committed DB through the hub, not the transaction's app.
type monitorAlertHub struct {
stubHub
handle func(*core.Record, map[string]monitor.Result) error
}
func (h monitorAlertHub) HandleNetworkMonitorAlerts(record *core.Record, results map[string]monitor.Result) error {
return h.handle(record, results)
}
func TestNetworkMonitorAlertsAfterCommit(t *testing.T) {
for _, realtime := range []bool{false, true} {
t.Run(fmt.Sprint(realtime), func(t *testing.T) {
sys, app := newTestSystemWithHub(t)
if realtime {
client := subscriptions.NewDefaultClient()
client.Subscribe("network_monitors/*")
app.SubscriptionsBroker().Register(client)
}
collection, err := app.FindCachedCollectionByNameOrId("network_monitors")
require.NoError(t, err)
record := core.NewRecord(collection)
record.Set("system", sys.Id)
require.NoError(t, app.SaveNoValidate(record))
called := 0
result := monitor.Result{LastProbeAt: time.Now().UnixMilli(), SampleCount: 3, PacketLoss1h: 10}
sys.manager.hub = monitorAlertHub{stubHub: stubHub{app}, handle: func(systemRecord *core.Record, results map[string]monitor.Result) error {
called++
assert.Equal(t, sys.Id, systemRecord.Id)
assert.Equal(t, result, results[record.Id])
saved, err := app.FindRecordById("network_monitors", record.Id)
require.NoError(t, err)
assert.Equal(t, 10.0, saved.GetFloat("loss1h"))
return nil
}}
data := &system.CombinedData{Monitors: map[string]monitor.Result{record.Id: result}}
_, err = sys.createRecords(data)
require.NoError(t, err)
assert.Equal(t, 1, called)
// A transaction that fails after writing monitor stats must not notify.
_, err = app.DB().NewQuery(`CREATE TRIGGER fail_system BEFORE UPDATE ON systems BEGIN SELECT RAISE(ABORT, 'test rollback'); END`).Execute()
require.NoError(t, err)
_, err = sys.createRecords(data)
require.Error(t, err)
assert.Equal(t, 1, called)
})
}
}

View File

@@ -1,184 +0,0 @@
//go:build testing
package systems
import (
"net/http"
"net/http/httptest"
"strings"
"sync/atomic"
"testing"
"time"
"github.com/blang/semver"
"github.com/fxamacker/cbor/v2"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/monitor"
esystem "github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/hub/ws"
"github.com/lxzan/gws"
"github.com/pocketbase/pocketbase/core"
"github.com/stretchr/testify/require"
)
type monitorSyncClient struct {
gws.BuiltinEventHandler
requests chan common.HubRequest[monitor.SyncRequest]
failSync atomic.Bool
}
func (c *monitorSyncClient) OnMessage(conn *gws.Conn, message *gws.Message) {
defer message.Close()
var req common.HubRequest[cbor.RawMessage]
if err := cbor.Unmarshal(message.Bytes(), &req); err != nil {
return
}
resp := common.AgentResponse{Id: req.Id}
if req.Action == common.GetData {
resp.SystemData = &esystem.CombinedData{}
} else {
var data monitor.SyncRequest
if err := cbor.Unmarshal(req.Data, &data); err != nil {
return
}
c.requests <- common.HubRequest[monitor.SyncRequest]{Id: req.Id, Action: req.Action, Data: data}
if c.failSync.Load() {
resp.Error = "test sync failure"
} else {
resp.Data, _ = cbor.Marshal(monitor.SyncResponse{})
}
}
response, _ := cbor.Marshal(resp)
_ = conn.WriteMessage(gws.OpcodeBinary, response)
}
// Avoid the production delayed disconnect notification; these tests explicitly
// remove each connection from the manager before reconnecting.
type monitorSyncServer struct{ ws.Handler }
func (*monitorSyncServer) OnClose(*gws.Conn, error) {}
func TestNetworkMonitorSyncSkipsOlderAgents(t *testing.T) {
for _, version := range []string{"0.0.0", "0.18.0", "0.19.0"} {
t.Run(version, func(t *testing.T) {
// No transport: attempting to send any request would fail.
sys := &System{agentVersion: semver.MustParse(version)}
require.NoError(t, sys.SyncNetworkMonitors(nil))
result, err := sys.UpsertNetworkMonitor(monitor.Config{ID: "test"}, true)
require.NoError(t, err)
require.Nil(t, result)
require.NoError(t, sys.DeleteNetworkMonitor("test"))
})
}
}
func TestNetworkMonitorReconnectSync(t *testing.T) {
for _, change := range []string{"delete", "disable", "retry"} {
t.Run(change, func(t *testing.T) {
sys, app := newTestSystemWithHub(t)
record, err := app.FindRecordById("systems", sys.Id)
require.NoError(t, err)
// Suppress unrelated system-stat requests while exercising reconnects.
record.Set("status", paused)
require.NoError(t, app.SaveNoValidate(record))
collection, err := app.FindCachedCollectionByNameOrId("network_monitors")
require.NoError(t, err)
probe := core.NewRecord(collection)
probe.Load(map[string]any{
"system": sys.Id, "target": "localhost", "protocol": "tcp",
"port": 80, "interval": 60, "enabled": true,
})
require.NoError(t, app.SaveNoValidate(probe))
sm := NewSystemManager(stubHub{app})
t.Cleanup(func() {
sm.cancel()
_ = sm.RemoveSystem(sys.Id)
sm.smartFetchMap.StopCleaner()
sm.zfsFetchMap.StopCleaner()
})
version := semver.MustParse("0.20.0")
connections := make(chan *ws.WsConn, 1)
upgrader := gws.NewUpgrader(&monitorSyncServer{}, nil)
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
conn, err := upgrader.Upgrade(w, r)
if err != nil {
t.Error(err)
return
}
wsConn := ws.NewWsConnection(conn, version)
conn.Session().Store("wsConn", wsConn)
connections <- wsConn
conn.ReadLoop()
}))
t.Cleanup(server.Close)
client := &monitorSyncClient{requests: make(chan common.HubRequest[monitor.SyncRequest], 2)}
connect := func() monitor.SyncRequest {
t.Helper()
conn, _, err := gws.NewClient(client, &gws.ClientOption{Addr: "ws" + strings.TrimPrefix(server.URL, "http")})
require.NoError(t, err)
t.Cleanup(func() { _ = conn.NetConn().Close() })
go conn.ReadLoop()
select {
case wsConn := <-connections:
require.NoError(t, sm.AddWebSocketSystem(sys.Id, version, wsConn))
case <-time.After(3 * time.Second):
t.Fatal("websocket connection was not established")
}
select {
case req := <-client.requests:
require.Equal(t, common.SyncNetworkMonitors, req.Action)
require.Equal(t, monitor.SyncActionReplace, req.Data.Action)
return req.Data
case <-time.After(3 * time.Second):
t.Fatal("reconnected agent did not receive a monitor replacement")
return monitor.SyncRequest{}
}
}
client.failSync.Store(change == "retry")
initial := connect()
require.Len(t, initial.Configs, 1)
require.Equal(t, probe.Id, initial.Configs[0].ID)
if change == "retry" {
system, err := sm.GetSystem(sys.Id)
require.NoError(t, err)
require.Eventually(t, system.monitorsNeedSync.Load, time.Second, time.Millisecond)
// A second failed sync must not fail the stats fetch or clear pending state.
_, err = system.fetchDataFromAgent(common.DataRequestOptions{})
require.NoError(t, err)
require.True(t, system.monitorsNeedSync.Load())
require.Len(t, client.requests, 1)
<-client.requests
client.failSync.Store(false)
_, err = system.fetchDataFromAgent(common.DataRequestOptions{})
require.NoError(t, err)
require.False(t, system.monitorsNeedSync.Load())
require.Len(t, client.requests, 1)
retry := <-client.requests
require.Equal(t, monitor.SyncActionReplace, retry.Data.Action)
require.Equal(t, initial.Configs, retry.Data.Configs)
_, err = system.fetchDataFromAgent(common.DataRequestOptions{})
require.NoError(t, err)
require.Empty(t, client.requests, "successful sync must not repeat on every fetch")
return
}
require.NoError(t, sm.RemoveSystem(sys.Id))
if change == "delete" {
require.NoError(t, app.Delete(probe))
} else {
probe.Set("enabled", false)
require.NoError(t, app.SaveNoValidate(probe))
}
require.Empty(t, connect().Configs, "reconnect must clear the agent's previous probe")
})
}
}
func TestGetMonitorConfigsForSystemQueryError(t *testing.T) {
sys, app := newTestSystemWithHub(t)
_, err := app.DB().NewQuery("DROP TABLE network_monitors").Execute()
require.NoError(t, err)
_, err = sys.manager.GetMonitorConfigsForSystem(sys.Id)
require.Error(t, err, "a failed query must not be treated as an empty monitor set")
}

View File

@@ -1,80 +0,0 @@
package systems
import (
"context"
"fmt"
"time"
"github.com/henrygd/beszel"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/monitor"
)
// syncPendingNetworkMonitors runs on WebSocket connect and after successful stats
// fetches. Failed syncs retry on the next update without taking the system down.
func (sys *System) syncPendingNetworkMonitors() {
if !sys.monitorsNeedSync.Swap(false) {
return
}
if err := sys.syncAllNetworkMonitors(); err != nil {
sys.monitorsNeedSync.Store(true)
sys.manager.hub.Logger().Warn("failed to sync monitors to agent", "system", sys.Id, "err", err)
}
}
func (sys *System) syncAllNetworkMonitors() error {
configs, err := sys.manager.GetMonitorConfigsForSystem(sys.Id)
if err != nil {
return fmt.Errorf("failed to load monitors: %w", err)
}
// An empty set must also replace probes retained across a disconnect.
return sys.SyncNetworkMonitors(configs)
}
// SyncNetworkMonitors sends monitor configurations to the agent.
func (sys *System) SyncNetworkMonitors(configs []monitor.Config) error {
_, err := sys.syncNetworkMonitors(monitor.SyncRequest{Action: monitor.SyncActionReplace, Configs: configs})
return err
}
// UpsertNetworkMonitor sends a single monitor configuration change to the agent.
func (sys *System) UpsertNetworkMonitor(config monitor.Config, runNow bool) (*monitor.Result, error) {
resp, err := sys.syncNetworkMonitors(monitor.SyncRequest{
Action: monitor.SyncActionUpsert,
Config: config,
RunNow: runNow,
})
if err != nil {
return nil, err
}
if resp.Result == (monitor.Result{}) {
return nil, nil
}
result := resp.Result
return &result, nil
}
// DeleteNetworkMonitor removes a single monitor task from the agent.
func (sys *System) DeleteNetworkMonitor(id string) error {
_, err := sys.syncNetworkMonitors(monitor.SyncRequest{
Action: monitor.SyncActionDelete,
Config: monitor.Config{ID: id},
})
return err
}
func (sys *System) syncNetworkMonitors(req monitor.SyncRequest) (monitor.SyncResponse, error) {
if sys.agentVersion.LT(beszel.MinVersionNetworkMonitors) {
return monitor.SyncResponse{}, nil
}
timeout := 5 * time.Second
if req.Action == monitor.SyncActionUpsert && req.RunNow {
// Allow the probe to finish, including a timeout result, while preserving
// the normal request budget for transport and response handling.
timeout += monitor.MaxProbeTimeout
}
ctx, cancel := context.WithTimeout(context.Background(), timeout)
defer cancel()
var result monitor.SyncResponse
return result, sys.request(ctx, common.SyncNetworkMonitors, req, &result)
}

View File

@@ -9,7 +9,6 @@ import (
"math/rand"
"net"
"strings"
"sync"
"sync/atomic"
"time"
@@ -19,7 +18,6 @@ import (
"github.com/henrygd/beszel/internal/hub/ws"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/smart"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/entities/systemd"
@@ -32,8 +30,6 @@ import (
"github.com/lxzan/gws"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/security"
"github.com/pocketbase/pocketbase/tools/types"
"golang.org/x/crypto/ssh"
)
@@ -56,13 +52,6 @@ type System struct {
smartInterval time.Duration // Interval for periodic SMART data updates
zfsFetching atomic.Bool // True if ZFS pools are currently being fetched
zfsInterval time.Duration // Interval for periodic ZFS detail data updates
// A fresh connection needs a full monitor configuration sync.
monitorsNeedSync atomic.Bool
// Serialize persistence from scheduled updates and resumes through commit.
recordsMu sync.Mutex
// Protected by recordsMu; realtime reads don't consume probes.
lastSavedMonitorProbe map[string]int64
}
func (sm *SystemManager) NewSystem(systemId string) *System {
@@ -222,15 +211,11 @@ func (sys *System) handlePaused() {
// createRecords updates the system record and adds system_stats and container_stats records
func (sys *System) createRecords(data *system.CombinedData) (*core.Record, error) {
sys.recordsMu.Lock()
defer sys.recordsMu.Unlock()
systemRecord, err := sys.getRecord(sys.manager.hub)
if err != nil {
return nil, err
}
hub := sys.manager.hub
savedMonitorProbes := make(map[string]int64)
err = hub.RunInTransaction(func(txApp core.App) error {
// add system_stats record
systemStatsCollection, err := txApp.FindCachedCollectionByNameOrId("system_stats")
@@ -281,12 +266,6 @@ func (sys *System) createRecords(data *system.CombinedData) (*core.Record, error
}
}
if data.Monitors != nil {
if err := sys.updateNetworkMonitorsRecords(txApp, data.Monitors, savedMonitorProbes); err != nil {
return err
}
}
if err := sys.syncZfsPoolHealth(txApp, data.Stats.ZfsPools); err != nil {
return err
}
@@ -308,29 +287,6 @@ func (sys *System) createRecords(data *system.CombinedData) (*core.Record, error
return nil
})
// Publish only successful inserts after the entire transaction commits.
if err == nil && len(savedMonitorProbes) > 0 {
if sys.lastSavedMonitorProbe == nil {
sys.lastSavedMonitorProbe = savedMonitorProbes
} else {
for id, timestamp := range savedMonitorProbes {
sys.lastSavedMonitorProbe[id] = timestamp
}
}
}
// A non-nil report includes cached results for all remaining monitors.
if err == nil && data.Monitors != nil {
for id := range sys.lastSavedMonitorProbe {
if _, exists := data.Monitors[id]; !exists {
delete(sys.lastSavedMonitorProbe, id)
}
}
}
if err == nil {
if alertErr := hub.HandleNetworkMonitorAlerts(systemRecord, data.Monitors); alertErr != nil {
hub.Logger().Error("Error handling network monitor alerts", "err", alertErr)
}
}
return systemRecord, err
}
@@ -381,7 +337,7 @@ func createSystemdStatsRecords(app core.App, data []*systemd.Service, systemId s
}
suffix := fmt.Sprintf("%d", i)
valueStrings = append(valueStrings, fmt.Sprintf("({:id%[1]s}, {:system}, {:name%[1]s}, {:state%[1]s}, {:sub%[1]s}, {:cpu%[1]s}, {:cpuPeak%[1]s}, {:memory%[1]s}, {:memPeak%[1]s}, {:updated})", suffix))
params["id"+suffix] = MakeStableHashId(systemId, service.Name)
params["id"+suffix] = makeStableHashId(systemId, service.Name)
params["name"+suffix] = service.Name
params["state"+suffix] = service.State
params["sub"+suffix] = service.Sub
@@ -407,106 +363,6 @@ func createSystemdStatsRecords(app core.App, data []*systemd.Service, systemId s
return err
}
func (sys *System) updateNetworkMonitorsRecords(app core.App, monitorResults map[string]monitor.Result, savedProbes map[string]int64) error {
if len(monitorResults) == 0 {
return nil
}
var err error
systemId := sys.Id
const monitorCollectionName = "network_monitors"
// If realtime updates are active, we save via PocketBase records to trigger realtime events.
// Otherwise we can do a more efficient direct update via SQL
realtimeActive := utils.RealtimeActiveForCollection(app, monitorCollectionName, func(filterQuery string) bool {
return !strings.Contains(filterQuery, "system") || strings.Contains(filterQuery, systemId)
})
now := time.Now().UTC()
nowMilli := now.UnixMilli()
nowString := now.Format(types.DefaultDateLayout)
var db dbx.Builder
var updateQuery *dbx.Query
if !realtimeActive {
db = app.DB()
monitorFields := []string{"res", "resMin1h", "resMax1h", "resAvg1h", "loss1h", "updated"}
setClauses := make([]string, len(monitorFields))
for i, f := range monitorFields {
setClauses[i] = fmt.Sprintf("%s={:%s}", f, f)
}
queryString := fmt.Sprintf("UPDATE %s SET %s WHERE id={:id}", monitorCollectionName, strings.Join(setClauses, ", "))
updateQuery = db.NewQuery(queryString)
}
// update network_monitors records
for id, result := range monitorResults {
monitorData := map[string]any{
"id": id,
"res": result.AvgResponse,
"resAvg1h": result.AvgResponse1h,
"resMin1h": result.MinResponse1h,
"resMax1h": result.MaxResponse1h,
"loss1h": result.PacketLoss1h,
"updated": nowString,
}
switch realtimeActive {
case true:
var record *core.Record
record, err = app.FindRecordById(monitorCollectionName, id)
if err == nil {
record.Load(monitorData)
err = app.SaveNoValidate(record)
}
default:
_, err = updateQuery.Bind(dbx.Params(monitorData)).Execute()
}
if err != nil {
app.Logger().Warn("Failed to update monitor", "system", systemId, "monitor", id, "err", err)
}
}
// handle stats collection — one record per monitor
const statsCollectionName = "network_monitor_stats"
var statsCollection *core.Collection
if realtimeActive {
statsCollection, _ = app.FindCachedCollectionByNameOrId(statsCollectionName)
}
for monitorId, result := range monitorResults {
// Compare identity, not ordering, so agent clock changes don't stall writes.
if result.LastProbeAt == sys.lastSavedMonitorProbe[monitorId] {
continue
}
statsRecordData := map[string]any{
"system": systemId,
"monitor": monitorId,
"type": "1m",
"created": nowMilli,
"res_min": result.MinResponse,
"res_max": result.MaxResponse,
"total_count": result.TotalCount,
"success_count": result.SuccessCount,
"res_sum": result.ResponseSum,
}
switch realtimeActive {
case true:
record := core.NewRecord(statsCollection)
record.Load(statsRecordData)
err = app.SaveNoValidate(record)
default:
statsRecordData["id"] = security.PseudorandomStringWithAlphabet(10, core.DefaultIdAlphabet)
_, err = db.Insert(statsCollectionName, dbx.Params(statsRecordData)).Execute()
}
if err != nil {
app.Logger().Error("Failed to update monitor stats", "system", systemId, "monitor", monitorId, "err", err)
} else {
savedProbes[monitorId] = result.LastProbeAt
}
}
return nil
}
// createContainerRecords creates container records
func createContainerRecords(app core.App, data []*container.Stats, systemId string) error {
if len(data) == 0 {
@@ -520,7 +376,7 @@ func createContainerRecords(app core.App, data []*container.Stats, systemId stri
valueStrings := make([]string, 0, len(data))
for i, container := range data {
suffix := fmt.Sprintf("%d", i)
valueStrings = append(valueStrings, fmt.Sprintf("({:id%[1]s}, {:system}, {:name%[1]s}, {:image%[1]s}, {:ports%[1]s}, {:status%[1]s}, {:health%[1]s}, {:cpu%[1]s}, {:memory%[1]s}, {:net%[1]s}, {:updateAvailable%[1]s}, {:updated})", suffix))
valueStrings = append(valueStrings, fmt.Sprintf("({:id%[1]s}, {:system}, {:name%[1]s}, {:image%[1]s}, {:ports%[1]s}, {:status%[1]s}, {:health%[1]s}, {:cpu%[1]s}, {:memory%[1]s}, {:net%[1]s}, {:updated})", suffix))
params["id"+suffix] = container.Id
params["name"+suffix] = container.Name
params["image"+suffix] = container.Image
@@ -534,10 +390,9 @@ func createContainerRecords(app core.App, data []*container.Stats, systemId stri
netBytes = uint64((container.NetworkSent + container.NetworkRecv) * 1024 * 1024)
}
params["net"+suffix] = netBytes
params["updateAvailable"+suffix] = container.UpdateAvailable
}
queryString := fmt.Sprintf(
"INSERT INTO containers (id, system, name, image, ports, status, health, cpu, memory, net, updatable, updated) VALUES %s ON CONFLICT(id) DO UPDATE SET system = excluded.system, name = excluded.name, image = excluded.image, ports = excluded.ports, status = excluded.status, health = excluded.health, cpu = excluded.cpu, memory = excluded.memory, net = excluded.net, updatable = excluded.updatable, updated = excluded.updated",
"INSERT INTO containers (id, system, name, image, ports, status, health, cpu, memory, net, updated) VALUES %s ON CONFLICT(id) DO UPDATE SET system = excluded.system, name = excluded.name, image = excluded.image, ports = excluded.ports, status = excluded.status, health = excluded.health, cpu = excluded.cpu, memory = excluded.memory, net = excluded.net, updated = excluded.updated",
strings.Join(valueStrings, ","),
)
_, err := app.DB().NewQuery(queryString).Bind(params).Execute()
@@ -633,10 +488,7 @@ func (sys *System) request(ctx context.Context, action common.WebSocketAction, r
err := sys.sshTransport.RequestWithRetry(ctx, action, req, dest, 1)
// Keep legacy SSH client/version fields in sync for other code paths.
if sys.sshTransport != nil {
client := sys.sshTransport.GetClient()
if previous := sys.client.Swap(client); client != nil && client != previous {
sys.monitorsNeedSync.Store(true)
}
sys.client.Store(sys.sshTransport.GetClient())
sys.agentVersion = sys.sshTransport.GetAgentVersion()
}
return err
@@ -694,7 +546,6 @@ func (sys *System) fetchDataFromAgent(options common.DataRequestOptions) (*syste
if sys.WsConn != nil && sys.WsConn.IsConnected() {
wsData, err := sys.fetchDataViaWebSocket(options)
if err == nil {
sys.syncPendingNetworkMonitors()
return wsData, nil
}
// close the WebSocket connection if error and try SSH
@@ -705,7 +556,6 @@ func (sys *System) fetchDataFromAgent(options common.DataRequestOptions) (*syste
if err != nil {
return nil, err
}
sys.syncPendingNetworkMonitors()
return sshData, nil
}
@@ -771,7 +621,7 @@ func (sys *System) FetchZfsDataFromAgent(force bool) (*zfs.ZfsData, error) {
return &result, err
}
func MakeStableHashId(strings ...string) string {
func makeStableHashId(strings ...string) string {
hash := fnv.New32a()
for _, str := range strings {
hash.Write([]byte(str))
@@ -940,7 +790,6 @@ func (s *System) createSSHClient() error {
return err
}
s.agentVersion, _ = extractAgentVersion(string(client.Conn.ServerVersion()))
s.monitorsNeedSync.Store(true)
s.manager.resetFailedSmartFetchState(s.Id)
s.manager.resetFailedZfsFetchState(s.Id)
return nil

View File

@@ -9,7 +9,6 @@ import (
"github.com/henrygd/beszel/internal/hub/ws"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/hub/expirymap"
@@ -18,7 +17,6 @@ import (
"github.com/henrygd/beszel"
"github.com/blang/semver"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/store"
"golang.org/x/crypto/ssh"
@@ -64,7 +62,6 @@ type hubLike interface {
core.App
GetSSHKey(dataDir string) (ssh.Signer, error)
HandleSystemAlerts(systemRecord *core.Record, data *system.CombinedData) error
HandleNetworkMonitorAlerts(systemRecord *core.Record, results map[string]monitor.Result) error
HandleStatusAlerts(status string, systemRecord *core.Record) error
HandleContainerAlerts(systemRecord *core.Record, data *system.CombinedData, fetchLogs func(containerID string) (string, error)) error
CancelPendingStatusAlerts(systemID string)
@@ -349,15 +346,10 @@ func (sm *SystemManager) AddWebSocketSystem(systemId string, agentVersion semver
system := sm.NewSystem(systemId)
system.WsConn = wsConn
system.agentVersion = agentVersion
system.monitorsNeedSync.Store(true)
if err := sm.AddRecord(systemRecord, system); err != nil {
return err
}
// Sync network monitors to the newly connected agent
go system.syncPendingNetworkMonitors()
return nil
}
@@ -370,16 +362,6 @@ func (sm *SystemManager) resetFailedSmartFetchState(systemID string) {
}
}
// GetMonitorConfigsForSystem returns all enabled monitor configs for a system.
func (sm *SystemManager) GetMonitorConfigsForSystem(systemID string) ([]monitor.Config, error) {
var configs []monitor.Config
err := sm.hub.DB().
NewQuery("SELECT id, target, protocol, port, interval FROM network_monitors WHERE system = {:system} AND enabled = true").
Bind(dbx.Params{"system": systemID}).
All(&configs)
return configs, err
}
// resetFailedZfsFetchState clears only failed ZFS cooldown entries so a fresh
// agent reconnect retries ZFS discovery immediately after configuration changes.
func (sm *SystemManager) resetFailedZfsFetchState(systemID string) {
@@ -415,12 +397,11 @@ func (sm *SystemManager) createSSHClientConfig() error {
// deactivateAlerts finds all triggered alerts for a system and sets them to inactive.
// This is called when a system is paused or goes offline to prevent continued alerts.
// Monitor incidents remain open: a missing observation does not establish recovery.
func deactivateAlerts(app core.App, systemID string) error {
// Note: Direct SQL updates don't trigger SSE, so we use the PocketBase API
// _, err := app.DB().NewQuery(fmt.Sprintf("UPDATE alerts SET triggered = false WHERE system = '%s'", systemID)).Execute()
alerts, err := app.FindRecordsByFilter("alerts", fmt.Sprintf("system = '%s' && triggered = 1 && name != 'NetworkMonitorLoss'", systemID), "", -1, 0)
alerts, err := app.FindRecordsByFilter("alerts", fmt.Sprintf("system = '%s' && triggered = 1", systemID), "", -1, 0)
if err != nil {
return err
}

View File

@@ -6,8 +6,6 @@ import (
"time"
"github.com/henrygd/beszel/internal/common"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/hub/utils"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/apis"
@@ -167,7 +165,7 @@ func (sm *SystemManager) fetchRealtimeDataAndNotify() {
if err != nil {
return
}
bytes, err := marshalRealtimeData(data)
bytes, err := json.Marshal(data)
if err == nil {
notify(sm.hub, system, fetch.subscription, bytes)
}
@@ -206,22 +204,6 @@ func (sm *SystemManager) finishRealtimeFetch(fetch realtimeFetch) {
}
}
// marshalRealtimeData marshals combined agent data for a realtime broadcast, converting
// the per-monitor results into the derived metric fields the frontend charts expect.
func marshalRealtimeData(data *system.CombinedData) ([]byte, error) {
if len(data.Monitors) == 0 {
return json.Marshal(data)
}
monitorStats := make(map[string]monitor.Stats, len(data.Monitors))
for id, result := range data.Monitors {
monitorStats[id] = monitor.Stats{}.FromResult(result)
}
return json.Marshal(struct {
*system.CombinedData
Monitors map[string]monitor.Stats `json:"Monitors"`
}{data, monitorStats})
}
// notify broadcasts realtime data to all clients subscribed to a specific subscription.
// Custom topics bypass collection rules, so check current access for every
// recipient, including clients whose authentication or membership was revoked.

View File

@@ -77,7 +77,7 @@ func (sys *System) saveSmartDevices(smartData map[string]smart.SmartData, comple
currentIDs := make(map[string]struct{}, len(smartData))
for deviceKey := range smartData {
currentIDs[MakeStableHashId(sys.Id, deviceKey)] = struct{}{}
currentIDs[makeStableHashId(sys.Id, deviceKey)] = struct{}{}
}
err = hub.RunInTransaction(func(txApp core.App) error {
@@ -115,7 +115,7 @@ func (sys *System) saveSmartDevices(smartData map[string]smart.SmartData, comple
}
func (sys *System) upsertSmartDeviceRecord(app core.App, collection *core.Collection, deviceKey string, device smart.SmartData) error {
recordID := MakeStableHashId(sys.Id, deviceKey)
recordID := makeStableHashId(sys.Id, deviceKey)
record, err := app.FindRecordById(collection, recordID)
if err != nil {

View File

@@ -7,7 +7,6 @@ import (
"testing"
"time"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/smart"
esystem "github.com/henrygd/beszel/internal/entities/system"
"github.com/henrygd/beszel/internal/hub/expirymap"
@@ -29,8 +28,7 @@ func (stubHub) GetSSHKey(dataDir string) (ssh.Signer, error) { return nil, nil }
func (stubHub) HandleSystemAlerts(systemRecord *core.Record, data *esystem.CombinedData) error {
return nil
}
func (stubHub) HandleNetworkMonitorAlerts(*core.Record, map[string]monitor.Result) error { return nil }
func (stubHub) HandleStatusAlerts(status string, systemRecord *core.Record) error { return nil }
func (stubHub) HandleStatusAlerts(status string, systemRecord *core.Record) error { return nil }
func (stubHub) HandleContainerAlerts(systemRecord *core.Record, data *esystem.CombinedData, fetchLogs func(containerID string) (string, error)) error {
return nil
}
@@ -214,7 +212,7 @@ func TestSaveSmartDevices_IncompleteDataDoesNotRemoveDevices(t *testing.T) {
}, false))
assert.Len(t, countSmartDeviceRecords(t, testApp, sys.Id), 2)
recordA, err := testApp.FindRecordById("smart_devices", MakeStableHashId(sys.Id, "AAA"))
recordA, err := testApp.FindRecordById("smart_devices", makeStableHashId(sys.Id, "AAA"))
require.NoError(t, err)
assert.EqualValues(t, 42, recordA.GetInt("temp"))
}

View File

@@ -14,9 +14,9 @@ func TestGetSystemdServiceId(t *testing.T) {
serviceName := "nginx.service"
// Call multiple times and ensure same result
id1 := MakeStableHashId(systemId, serviceName)
id2 := MakeStableHashId(systemId, serviceName)
id3 := MakeStableHashId(systemId, serviceName)
id1 := makeStableHashId(systemId, serviceName)
id2 := makeStableHashId(systemId, serviceName)
id3 := makeStableHashId(systemId, serviceName)
assert.Equal(t, id1, id2)
assert.Equal(t, id2, id3)
@@ -29,10 +29,10 @@ func TestGetSystemdServiceId(t *testing.T) {
serviceName1 := "nginx.service"
serviceName2 := "apache.service"
id1 := MakeStableHashId(systemId1, serviceName1)
id2 := MakeStableHashId(systemId2, serviceName1)
id3 := MakeStableHashId(systemId1, serviceName2)
id4 := MakeStableHashId(systemId2, serviceName2)
id1 := makeStableHashId(systemId1, serviceName1)
id2 := makeStableHashId(systemId2, serviceName1)
id3 := makeStableHashId(systemId1, serviceName2)
id4 := makeStableHashId(systemId2, serviceName2)
// All IDs should be different
assert.NotEqual(t, id1, id2)
@@ -56,14 +56,14 @@ func TestGetSystemdServiceId(t *testing.T) {
}
for _, tc := range testCases {
id := MakeStableHashId(tc.systemId, tc.serviceName)
id := makeStableHashId(tc.systemId, tc.serviceName)
// FNV-32 produces 8 hex characters
assert.Len(t, id, 8, "ID should be 8 characters for systemId='%s', serviceName='%s'", tc.systemId, tc.serviceName)
}
})
t.Run("hexadecimal output", func(t *testing.T) {
id := MakeStableHashId("test-system", "test-service")
id := makeStableHashId("test-system", "test-service")
assert.NotEmpty(t, id)
// Should only contain hexadecimal characters

View File

@@ -129,7 +129,7 @@ func (sys *System) saveZfsPools(zfsData *zfs.ZfsData) error {
}
func (sys *System) upsertZfsPoolRecord(app core.App, collection *core.Collection, pool *zfs.PoolDetail) error {
recordID := MakeStableHashId(sys.Id, pool.Name)
recordID := makeStableHashId(sys.Id, pool.Name)
record, err := app.FindRecordById(collection, recordID)
if err != nil {
@@ -171,7 +171,7 @@ func (sys *System) syncZfsPoolHealth(app core.App, pools map[string]*system.ZfsP
if pool == nil {
continue
}
recordID := MakeStableHashId(sys.Id, name)
recordID := makeStableHashId(sys.Id, name)
record, err := app.FindRecordById(collection, recordID)
if err != nil {
if !errors.Is(err, sql.ErrNoRows) {

View File

@@ -135,14 +135,14 @@ func TestSavePartialBackendInventory(t *testing.T) {
{Name: healthyKey, Alloc: 10}, {Name: failedKey, Alloc: 10},
}}
require.NoError(t, sys.saveZfsPools(initial))
failedID := MakeStableHashId(sys.Id, failedKey)
failedID := makeStableHashId(sys.Id, failedKey)
before, err := app.FindRecordById("zfs_pools", failedID)
require.NoError(t, err)
partial := &zfs.ZfsData{CompleteBackends: []string{healthy}, Pools: []*zfs.PoolDetail{
{Name: healthyKey, Alloc: 20}, {Name: failedKey, Alloc: 99},
}}
assert.ErrorIs(t, sys.saveZfsPools(partial), errIncompleteZfsData)
fresh, err := app.FindRecordById("zfs_pools", MakeStableHashId(sys.Id, healthyKey))
fresh, err := app.FindRecordById("zfs_pools", makeStableHashId(sys.Id, healthyKey))
require.NoError(t, err)
assert.EqualValues(t, 20, fresh.GetInt("alloc"))
cached, err := app.FindRecordById("zfs_pools", failedID)
@@ -169,7 +169,7 @@ func TestSyncZfsPoolHealthWritesOnlyTransitions(t *testing.T) {
require.NoError(t, sys.syncZfsPoolHealth(app, map[string]*system.ZfsPool{
"tank": {Total: 100, Used: 25, Health: "ONLINE"},
}))
record, err := app.FindRecordById(collection, MakeStableHashId(sys.Id, "tank"))
record, err := app.FindRecordById(collection, makeStableHashId(sys.Id, "tank"))
require.NoError(t, err)
firstUpdated := record.GetDateTime("updated")
assert.Equal(t, "ONLINE", record.GetString("health"))
@@ -193,7 +193,7 @@ func TestSyncZfsPoolHealthWritesOnlyTransitions(t *testing.T) {
func TestZfsRawCapacityPersistence(t *testing.T) {
sys, app := newTestSystemWithHub(t)
require.NoError(t, sys.saveZfsPools(&zfs.ZfsData{Complete: true, Pools: []*zfs.PoolDetail{{Name: "btrfs", Size: 200, Alloc: 10, Raw: true}}}))
record, err := app.FindRecordById("zfs_pools", MakeStableHashId(sys.Id, "btrfs"))
record, err := app.FindRecordById("zfs_pools", makeStableHashId(sys.Id, "btrfs"))
require.NoError(t, err)
require.True(t, record.GetBool("raw"))
require.NoError(t, sys.syncZfsPoolHealth(app, map[string]*system.ZfsPool{"btrfs": {Total: 1, Used: 0.25}}))
@@ -210,7 +210,7 @@ func TestBtrfsDisplayNameKeepsRecordIdentity(t *testing.T) {
key: {DisplayName: "tank", Health: "ONLINE"},
"tank": {Health: "ONLINE"},
}))
id := MakeStableHashId(sys.Id, key)
id := makeStableHashId(sys.Id, key)
record, err := app.FindRecordById("zfs_pools", id)
require.NoError(t, err)
assert.Equal(t, "tank", record.GetString("display_name"))
@@ -225,6 +225,6 @@ func TestBtrfsDisplayNameKeepsRecordIdentity(t *testing.T) {
record, err = app.FindRecordById("zfs_pools", id)
require.NoError(t, err)
assert.Equal(t, "detail name", record.GetString("display_name"))
_, err = app.FindRecordById("zfs_pools", MakeStableHashId(sys.Id, "tank"))
_, err = app.FindRecordById("zfs_pools", makeStableHashId(sys.Id, "tank"))
require.NoError(t, err)
}

View File

@@ -1,127 +0,0 @@
//go:build testing
package hub
import (
"net/netip"
"os"
"testing"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestParseTrustedProxies(t *testing.T) {
testCases := []struct {
name string
value string
prefixes []string
restricted bool
}{
{
name: "empty",
value: "",
restricted: false,
},
{
name: "blank",
value: " , ",
prefixes: nil,
restricted: true,
},
{
name: "single addresses become host prefixes",
value: "10.0.0.5, 2001:db8::1",
prefixes: []string{"10.0.0.5/32", "2001:db8::1/128"},
restricted: true,
},
{
name: "cidrs are masked",
value: "172.16.5.9/12,fd00::1/64",
prefixes: []string{"172.16.0.0/12", "fd00::/64"},
restricted: true,
},
{
name: "ipv4-mapped entries become ipv4",
value: "::ffff:10.0.0.5, ::ffff:10.0.0.0/104",
prefixes: []string{"10.0.0.5/32", "10.0.0.0/8"},
restricted: true,
},
{
name: "invalid entries are skipped, valid ones kept",
value: "proxy.internal, 10.0.0.0/8, 300.1.1.1, ::ffff:0.0.0.0/64",
prefixes: []string{"10.0.0.0/8"},
restricted: true,
},
{
name: "only invalid entries trust nobody",
value: "proxy.internal",
prefixes: nil,
restricted: true,
},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
t.Setenv("TRUSTED_PROXY_IPS", tc.value)
prefixes, restricted := parseTrustedProxies()
assert.Equal(t, tc.restricted, restricted)
var got []string
for _, p := range prefixes {
got = append(got, p.String())
}
assert.Equal(t, tc.prefixes, got)
})
}
t.Run("unset", func(t *testing.T) {
t.Setenv("TRUSTED_PROXY_IPS", "")
os.Unsetenv("TRUSTED_PROXY_IPS")
prefixes, restricted := parseTrustedProxies()
assert.False(t, restricted)
assert.Nil(t, prefixes)
})
t.Run("prefixed env var takes precedence", func(t *testing.T) {
t.Setenv("TRUSTED_PROXY_IPS", "10.0.0.0/8")
t.Setenv("BESZEL_HUB_TRUSTED_PROXY_IPS", "192.168.0.0/16")
prefixes, restricted := parseTrustedProxies()
assert.True(t, restricted)
require.Len(t, prefixes, 1)
assert.Equal(t, "192.168.0.0/16", prefixes[0].String())
})
}
func TestIsTrustedProxy(t *testing.T) {
prefixes := []netip.Prefix{
netip.MustParsePrefix("10.0.0.0/8"),
netip.MustParsePrefix("2001:db8::/32"),
netip.MustParsePrefix("fe80::/10"),
}
testCases := []struct {
name string
remoteAddr string
trusted bool
}{
{"ipv4 in prefix", "10.20.30.40:51234", true},
{"ipv4 outside prefix", "11.0.0.1:51234", false},
{"ipv6 in prefix", "[2001:db8:1::2]:443", true},
{"ipv6 outside prefix", "[2001:db9::1]:443", false},
{"ipv4-mapped ipv6 matches ipv4 prefix", "[::ffff:10.1.2.3]:80", true},
{"zone is ignored", "[fe80::1%eth0]:80", true},
{"no port", "10.1.2.3", true},
{"empty", "", false},
{"garbage", "not-an-address:80", false},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
assert.Equal(t, tc.trusted, isTrustedProxy(prefixes, tc.remoteAddr))
})
}
t.Run("empty allowlist trusts nobody", func(t *testing.T) {
assert.False(t, isTrustedProxy(nil, "10.0.0.1:1"))
})
}

View File

@@ -1,11 +1,7 @@
// Package utils provides utility functions for the hub.
package utils
import (
"os"
"github.com/pocketbase/pocketbase/core"
)
import "os"
// GetEnv retrieves an environment variable with a "BESZEL_HUB_" prefix, or falls back to the unprefixed key.
func GetEnv(key string) (value string, exists bool) {
@@ -14,26 +10,3 @@ func GetEnv(key string) (value string, exists bool) {
}
return os.LookupEnv(key)
}
// realtimeActiveForCollection checks if there are active WebSocket subscriptions for the given collection.
func RealtimeActiveForCollection(app core.App, collectionName string, validateFn func(filterQuery string) bool) bool {
broker := app.SubscriptionsBroker()
if broker.TotalClients() == 0 {
return false
}
for _, client := range broker.Clients() {
subs := client.Subscriptions(collectionName)
if len(subs) > 0 {
if validateFn == nil {
return true
}
for k := range subs {
filter := subs[k].Query["filter"]
if validateFn(filter) {
return true
}
}
}
}
return false
}

View File

@@ -83,8 +83,7 @@ func init() {
"ContainerHealth",
"SystemdFailed",
"CPUIOWait",
"CPUSteal",
"NetworkMonitorLoss"
"CPUSteal"
]
},
{
@@ -120,16 +119,6 @@ func init() {
"system": false,
"type": "bool"
},
{
"hidden": true,
"id": "json4000656575",
"maxSize": 0,
"name": "state",
"presentable": false,
"required": false,
"system": false,
"type": "json"
},
{
"hidden": true,
"id": "date1302749137",
@@ -163,7 +152,6 @@ func init() {
}
],
"indexes": [
"CREATE INDEX idx_alerts_system_name ON alerts (system, name)",
"CREATE UNIQUE INDEX ` + "`" + `idx_MnhEt21L5r` + "`" + ` ON ` + "`" + `alerts` + "`" + ` (\n ` + "`" + `user` + "`" + `,\n ` + "`" + `system` + "`" + `,\n ` + "`" + `name` + "`" + `\n)"
],
"system": false
@@ -246,20 +234,6 @@ func init() {
"system": false,
"type": "text"
},
{
"autogeneratePattern": "",
"hidden": false,
"id": "text3888135399",
"max": 0,
"min": 0,
"name": "monitor_name",
"pattern": "",
"presentable": false,
"primaryKey": false,
"required": false,
"system": false,
"type": "text"
},
{
"hidden": false,
"id": "number494360628",
@@ -1057,15 +1031,6 @@ func init() {
"required": true,
"system": false,
"type": "number"
},
{
"hidden": false,
"id": "bool2084032502",
"name": "updatable",
"presentable": false,
"required": false,
"system": false,
"type": "bool"
}
],
"indexes": [
@@ -1756,7 +1721,6 @@ func init() {
"fields": [
{
"autogeneratePattern": "[a-z0-9]{15}",
"help": "",
"hidden": false,
"id": "text3208210256",
"max": 15,
@@ -1772,7 +1736,6 @@ func init() {
{
"cascadeDelete": true,
"collectionId": "2hz5ncl8tizk5nx",
"help": "",
"hidden": false,
"id": "relation1204987316",
"maxSelect": 1,
@@ -1785,7 +1748,6 @@ func init() {
},
{
"autogeneratePattern": "",
"help": "",
"hidden": false,
"id": "text7739291048",
"max": 0,
@@ -1800,7 +1762,6 @@ func init() {
},
{
"autogeneratePattern": "",
"help": "",
"hidden": false,
"id": "text5528164482",
"max": 0,
@@ -1814,7 +1775,6 @@ func init() {
"type": "text"
},
{
"help": "",
"hidden": false,
"id": "number8862034195",
"max": null,
@@ -1827,7 +1787,6 @@ func init() {
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number4418907321",
"max": null,
@@ -1840,7 +1799,6 @@ func init() {
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number2904183765",
"max": null,
@@ -1853,7 +1811,6 @@ func init() {
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "json4466109723",
"maxSize": 0,
@@ -1864,7 +1821,6 @@ func init() {
"type": "json"
},
{
"help": "",
"hidden": false,
"id": "json9012873456",
"maxSize": 0,
@@ -1875,7 +1831,6 @@ func init() {
"type": "json"
},
{
"help": "",
"hidden": false,
"id": "json7182045639",
"maxSize": 0,
@@ -1886,7 +1841,6 @@ func init() {
"type": "json"
},
{
"help": "",
"hidden": false,
"id": "date9274163058",
"max": "",
@@ -1906,401 +1860,19 @@ func init() {
"presentable": false,
"system": false,
"type": "autodate"
},
{
"autogeneratePattern": "",
"help": "",
"hidden": false,
"id": "text3578368839",
"max": 0,
"min": 0,
"name": "display_name",
"pattern": "",
"presentable": false,
"primaryKey": false,
"required": false,
"system": false,
"type": "text"
},
{
"help": "",
"hidden": false,
"id": "bool447994709",
"name": "raw",
"presentable": false,
"required": false,
"system": false,
"type": "bool"
}
],
"id": "pbc_8441057391",
"indexes": [
"CREATE INDEX ` + "`" + `idx_zfsPoolsSystem` + "`" + ` ON ` + "`" + `zfs_pools` + "`" + ` (` + "`" + `system` + "`" + `)"
],
"listRule": "@request.auth.id != \"\" && system.users.id ?= @request.auth.id",
"listRule": null,
"name": "zfs_pools",
"system": false,
"type": "base",
"updateRule": null,
"viewRule": "@request.auth.id != \"\" && system.users.id ?= @request.auth.id"
},
{
"createRule": null,
"deleteRule": null,
"fields": [
{
"autogeneratePattern": "[a-z0-9]{10}",
"help": "",
"hidden": false,
"id": "text3208210256",
"max": 10,
"min": 6,
"name": "id",
"pattern": "^[a-z0-9]+$",
"presentable": false,
"primaryKey": true,
"required": true,
"system": true,
"type": "text"
},
{
"cascadeDelete": true,
"collectionId": "2hz5ncl8tizk5nx",
"help": "",
"hidden": false,
"id": "nm_system",
"maxSelect": 1,
"minSelect": 0,
"name": "system",
"presentable": false,
"required": true,
"system": false,
"type": "relation"
},
{
"autogeneratePattern": "",
"help": "",
"hidden": false,
"id": "nm_target",
"max": 500,
"min": 1,
"name": "target",
"pattern": "",
"presentable": false,
"primaryKey": false,
"required": true,
"system": false,
"type": "text"
},
{
"help": "",
"hidden": false,
"id": "nm_protocol",
"maxSelect": 1,
"name": "protocol",
"presentable": false,
"required": true,
"system": false,
"type": "select",
"values": [
"icmp",
"tcp",
"http",
"dns"
]
},
{
"help": "",
"hidden": false,
"id": "nm_port",
"max": 65535,
"min": 0,
"name": "port",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "nm_interval",
"max": 3600,
"min": 1,
"name": "interval",
"onlyInt": true,
"presentable": false,
"required": true,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number926446584",
"max": null,
"min": null,
"name": "res",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number1006954605",
"max": null,
"min": null,
"name": "resAvg1h",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number4267669802",
"max": null,
"min": null,
"name": "resMin1h",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number591433223",
"max": null,
"min": null,
"name": "resMax1h",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "number3726709001",
"max": null,
"min": null,
"name": "loss1h",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "nm_enabled",
"name": "enabled",
"presentable": false,
"required": false,
"system": false,
"type": "bool"
},
{
"hidden": false,
"id": "autodate2990389176",
"name": "created",
"onCreate": true,
"onUpdate": false,
"presentable": false,
"system": false,
"type": "autodate"
},
{
"help": "",
"hidden": false,
"id": "date3332085495",
"max": "",
"min": "",
"name": "updated",
"presentable": false,
"required": false,
"system": false,
"type": "date"
}
],
"id": "nm_monitors_001",
"indexes": [
"CREATE INDEX ` + "`" + `idx_nm_system_enabled` + "`" + ` ON ` + "`" + `network_monitors` + "`" + ` (` + "`" + `system` + "`" + `, ` + "`" + `enabled` + "`" + `)"
],
"listRule": null,
"name": "network_monitors",
"system": false,
"type": "base",
"updateRule": null,
"viewRule": null
},
{
"createRule": null,
"deleteRule": null,
"fields": [
{
"autogeneratePattern": "[a-z0-9]{10}",
"help": "",
"hidden": false,
"id": "text3208210256",
"max": 10,
"min": 10,
"name": "id",
"pattern": "^[a-z0-9]+$",
"presentable": false,
"primaryKey": true,
"required": true,
"system": true,
"type": "text"
},
{
"cascadeDelete": true,
"collectionId": "2hz5ncl8tizk5nx",
"help": "",
"hidden": false,
"id": "nms_system",
"maxSelect": 1,
"minSelect": 0,
"name": "system",
"presentable": false,
"required": true,
"system": false,
"type": "relation"
},
{
"cascadeDelete": true,
"collectionId": "nm_monitors_001",
"help": "",
"hidden": false,
"id": "nms_monitor",
"maxSelect": 1,
"minSelect": 0,
"name": "monitor",
"presentable": false,
"required": false,
"system": false,
"type": "relation"
},
{
"help": "Number of probe attempts",
"hidden": false,
"id": "nms_total_count",
"max": null,
"min": 0,
"name": "total_count",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "Number of successful probe attempts",
"hidden": false,
"id": "nms_success_count",
"max": null,
"min": 0,
"name": "success_count",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "Sum of successful response times in microseconds",
"hidden": false,
"id": "nms_res_sum",
"max": null,
"min": 0,
"name": "res_sum",
"onlyInt": true,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "Response time in microseconds",
"hidden": false,
"id": "nms_res_min",
"max": null,
"min": 0,
"name": "res_min",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "Response time in microseconds",
"hidden": false,
"id": "nms_res_max",
"max": null,
"min": 0,
"name": "res_max",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
},
{
"help": "",
"hidden": false,
"id": "nms_type",
"maxSelect": 1,
"name": "type",
"presentable": false,
"required": true,
"system": false,
"type": "select",
"values": [
"1m",
"10m",
"20m",
"120m",
"480m"
]
},
{
"help": "",
"hidden": false,
"id": "number2990389176",
"max": null,
"min": null,
"name": "created",
"onlyInt": false,
"presentable": false,
"required": false,
"system": false,
"type": "number"
}
],
"id": "nm_stats_001",
"indexes": [
"CREATE INDEX IF NOT EXISTS ` + "`" + `idx_nms_system_type_created` + "`" + ` ON ` + "`" + `network_monitor_stats` + "`" + ` (` + "`" + `system` + "`" + `, ` + "`" + `type` + "`" + `, ` + "`" + `created` + "`" + `)",
"CREATE INDEX IF NOT EXISTS ` + "`" + `idx_nms_monitor_type_created` + "`" + ` ON ` + "`" + `network_monitor_stats` + "`" + ` (` + "`" + `monitor` + "`" + `, ` + "`" + `type` + "`" + `, ` + "`" + `created` + "`" + `)",
"CREATE INDEX IF NOT EXISTS ` + "`" + `idx_nms_type_created` + "`" + ` ON ` + "`" + `network_monitor_stats` + "`" + ` (` + "`" + `type` + "`" + `, ` + "`" + `created` + "`" + `)"
],
"listRule": null,
"name": "network_monitor_stats",
"system": false,
"type": "base",
"updateRule": null,
"viewRule": null
}
}
]`
err := app.ImportCollectionsByMarshaledJSON([]byte(jsonData), false)

View File

@@ -0,0 +1,27 @@
package migrations
import (
"github.com/pocketbase/pocketbase/core"
m "github.com/pocketbase/pocketbase/migrations"
)
func init() {
m.Register(func(app core.App) error {
c, err := app.FindCollectionByNameOrId("zfs_pools")
if err != nil {
return err
}
c.Fields.Add(&core.TextField{Name: "display_name"})
c.Fields.Add(&core.BoolField{Name: "raw"})
return app.Save(c)
}, func(app core.App) error {
c, err := app.FindCollectionByNameOrId("zfs_pools")
if err != nil {
return err
}
c.Fields.RemoveByName("display_name")
c.Fields.RemoveByName("raw")
return app.Save(c)
})
}

View File

@@ -1,226 +0,0 @@
//go:build testing
package records_test
import (
"testing"
"time"
monitorEntity "github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/records"
"github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestAverageMonitorStats(t *testing.T) {
hub, err := tests.NewTestHub(t.TempDir())
require.NoError(t, err)
defer hub.Cleanup()
collection, err := hub.FindCachedCollectionByNameOrId("network_monitor_stats")
require.NoError(t, err)
assert.Nil(t, collection.Fields.GetByName("res_avg"))
assert.Nil(t, collection.Fields.GetByName("loss"))
rm := records.NewRecordManager(hub)
user, err := tests.CreateUser(hub, "monitor-avg@example.com", "testtesttest")
require.NoError(t, err)
sys, err := tests.CreateRecord(hub, "systems", map[string]any{
"name": "monitor-avg-system",
"host": "localhost",
"port": "45876",
"status": "up",
"users": []string{user.Id},
})
require.NoError(t, err)
monitor, err := tests.CreateRecord(hub, "network_monitors", map[string]any{
"system": sys.Id,
"name": "cloudflare",
"target": "1.1.1.1",
"protocol": "icmp",
"interval": 30,
"enabled": true,
})
require.NoError(t, err)
created := time.Now().UnixMilli()
// Unequal probe counts must weight both latency and loss.
recordA, err := tests.CreateRecord(hub, "network_monitor_stats", map[string]any{
"system": sys.Id,
"monitor": monitor.Id,
"type": "1m",
"created": created,
"res_min": 5,
"res_max": 20,
"total_count": 6, "success_count": 6, "res_sum": 60,
})
require.NoError(t, err)
recordB, err := tests.CreateRecord(hub, "network_monitor_stats", map[string]any{
"system": sys.Id,
"monitor": monitor.Id,
"type": "1m",
"created": created,
"res_min": 10,
"res_max": 60,
"total_count": 1, "success_count": 1, "res_sum": 22,
})
require.NoError(t, err)
result, count, err := rm.AverageMonitorStats(hub.DB(), monitor.Id, "1m", created-1)
require.NoError(t, err)
assert.Equal(t, 2, count)
assert.Equal(t, monitorEntity.Stats{ResAvg: 11.71, ResMin: 5, ResMax: 60, TotalCount: 7, SuccessCount: 7, ResponseSum: 82}, result)
for _, tc := range []struct {
name, monitor, recordType string
after int64
}{
{"other monitor", "missing", "1m", created - 1},
{"other type", monitor.Id, "10m", created - 1},
{"exclusive cutoff", monitor.Id, "1m", created},
} {
t.Run(tc.name, func(t *testing.T) {
stats, count, err := rm.AverageMonitorStats(hub.DB(), tc.monitor, tc.recordType, tc.after)
require.NoError(t, err)
assert.Zero(t, count)
assert.Equal(t, monitorEntity.Stats{}, stats)
})
}
// A failure-only bucket counts toward loss but must not lower latency.
recordB.Set("res_min", 0)
recordB.Set("res_max", 0)
recordB.Set("success_count", 0)
recordB.Set("res_sum", 0)
require.NoError(t, hub.Save(recordB))
result, count, err = rm.AverageMonitorStats(hub.DB(), monitor.Id, "1m", created-1)
require.NoError(t, err)
assert.Equal(t, 2, count)
assert.Equal(t, monitorEntity.Stats{ResAvg: 10, ResMin: 5, ResMax: 20, Loss: 14.29, TotalCount: 7, SuccessCount: 6, ResponseSum: 60}, result)
// Sparse monitor records must propagate through every rollup level.
rm.CreateLongerRecords()
for _, recordType := range []string{"10m", "20m", "120m", "480m"} {
rollups, err := hub.FindAllRecords("network_monitor_stats", dbx.HashExp{"monitor": monitor.Id, "type": recordType})
require.NoError(t, err)
require.Len(t, rollups, 1, recordType)
assert.Equal(t, 5.0, rollups[0].GetFloat("res_min"))
assert.Equal(t, 20.0, rollups[0].GetFloat("res_max"))
assert.Equal(t, 7, rollups[0].GetInt("total_count"))
assert.Equal(t, 6, rollups[0].GetInt("success_count"))
assert.Equal(t, 60, rollups[0].GetInt("res_sum"))
// A sibling with a different number of probes must retain its actual
// weight when the next tier combines their underlying counts.
_, err = tests.CreateRecord(hub, "network_monitor_stats", map[string]any{
"system": sys.Id, "monitor": monitor.Id, "type": recordType, "created": created,
"res_min": 100, "res_max": 100,
"total_count": 3, "success_count": 1, "res_sum": 100,
})
require.NoError(t, err)
merged, count, err := rm.AverageMonitorStats(hub.DB(), monitor.Id, recordType, created-1)
require.NoError(t, err)
assert.Equal(t, 2, count)
assert.Equal(t, monitorEntity.Stats{
ResAvg: 22.86, ResMin: 5, ResMax: 100, Loss: 30,
TotalCount: 10, SuccessCount: 7, ResponseSum: 160,
}, merged)
}
// All failures produce zero latency, while a genuine zero-microsecond
// success remains a valid minimum (it must not be filtered out as a sentinel).
recordA.Set("success_count", 0)
recordA.Set("res_sum", 0)
require.NoError(t, hub.Save(recordA))
result, _, err = rm.AverageMonitorStats(hub.DB(), monitor.Id, "1m", created-1)
require.NoError(t, err)
assert.Equal(t, monitorEntity.Stats{TotalCount: 7, Loss: 100}, result)
recordA.Set("success_count", 1)
recordA.Set("res_min", 0)
recordA.Set("res_max", 0)
require.NoError(t, hub.Save(recordA))
result, _, err = rm.AverageMonitorStats(hub.DB(), monitor.Id, "1m", created-1)
require.NoError(t, err)
assert.Equal(t, monitorEntity.Stats{TotalCount: 7, SuccessCount: 1, Loss: 85.71}, result)
}
func TestSparseMonitorRollups(t *testing.T) {
for _, tc := range []struct {
name string
interval int
samples int
disabled bool
}{
{"five minute interval", 300, 2, false},
{"ten minute interval", 600, 1, false},
{"fifteen minute interval", 900, 1, false},
{"empty window", 900, 0, false},
{"disabled with pending history", 300, 2, true},
{"disabled empty window", 900, 0, true},
} {
t.Run(tc.name, func(t *testing.T) {
hub, err := tests.NewTestHub(t.TempDir())
require.NoError(t, err)
defer hub.Cleanup()
user, err := tests.CreateUser(hub, "sparse-monitor@example.com", "testtesttest")
require.NoError(t, err)
sys, err := tests.CreateRecord(hub, "systems", map[string]any{
"name": "sparse-monitor-system", "host": "localhost", "port": "45876", "status": "up",
"users": []string{user.Id},
})
require.NoError(t, err)
monitor, err := tests.CreateRecord(hub, "network_monitors", map[string]any{
"system": sys.Id, "target": "1.1.1.1", "protocol": "icmp",
"interval": tc.interval, "enabled": true,
})
require.NoError(t, err)
now := time.Now()
for i := range tc.samples {
_, err := tests.CreateRecord(hub, "network_monitor_stats", map[string]any{
"system": sys.Id, "monitor": monitor.Id, "type": "1m",
"created": now.Add(-time.Minute - time.Duration(i*tc.interval)*time.Second).UnixMilli(),
"res_min": 8, "res_max": 20,
"total_count": 4, "success_count": 3, "res_sum": 36,
})
require.NoError(t, err)
}
// Other collections must still reject fewer than nine minute records.
for _, collection := range []string{"system_stats", "container_stats"} {
stats := `{"cpu":10}`
if collection == "container_stats" {
stats = `[{"name":"test","cpu":10}]`
}
for range 8 {
_, err := tests.CreateRecord(hub, collection, map[string]any{
"system": sys.Id, "type": "1m", "stats": stats,
})
require.NoError(t, err)
}
}
if tc.disabled {
monitor.Set("enabled", false)
require.NoError(t, hub.Save(monitor))
}
records.NewRecordManager(hub).CreateLongerRecords()
for _, recordType := range []string{"10m", "20m", "120m", "480m"} {
rollups, err := hub.FindAllRecords("network_monitor_stats", dbx.HashExp{"monitor": monitor.Id, "type": recordType})
require.NoError(t, err)
if tc.samples == 0 {
assert.Empty(t, rollups, recordType)
} else {
require.Len(t, rollups, 1, recordType)
assert.Equal(t, 8.0, rollups[0].GetFloat("res_min"))
assert.Equal(t, 20.0, rollups[0].GetFloat("res_max"))
assert.Equal(t, 4*tc.samples, rollups[0].GetInt("total_count"))
assert.Equal(t, 3*tc.samples, rollups[0].GetInt("success_count"))
assert.Equal(t, 36*tc.samples, rollups[0].GetInt("res_sum"))
}
for _, collection := range []string{"system_stats", "container_stats"} {
count, err := hub.CountRecords(collection, dbx.HashExp{"system": sys.Id, "type": recordType})
require.NoError(t, err)
assert.Zero(t, count, "%s %s", collection, recordType)
}
}
})
}
}

View File

@@ -3,16 +3,15 @@ package records
import (
"encoding/json"
"log/slog"
"math"
"time"
"github.com/henrygd/beszel/internal/entities/container"
"github.com/henrygd/beszel/internal/entities/monitor"
"github.com/henrygd/beszel/internal/entities/system"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/core"
"github.com/pocketbase/pocketbase/tools/types"
)
type RecordManager struct {
@@ -40,7 +39,7 @@ type StatsRecord struct {
// Create longer records by averaging shorter records
func (rm *RecordManager) CreateLongerRecords() {
now := time.Now().UTC()
// start := time.Now()
longerRecordData := []LongerRecordData{
{
shorterType: "1m",
@@ -70,28 +69,23 @@ func (rm *RecordManager) CreateLongerRecords() {
}
// wrap the operations in a transaction
// Pocketbase cron does not handle errors, log them here.
err := rm.app.RunInTransaction(func(txApp core.App) error {
rm.app.RunInTransaction(func(txApp core.App) error {
var err error
collections := [2]*core.Collection{}
collections[0], err = txApp.FindCachedCollectionByNameOrId("system_stats")
if err != nil {
slog.Error("Error finding cached collection using system stats:", "err", err)
return err
}
collections[1], err = txApp.FindCachedCollectionByNameOrId("container_stats")
if err != nil {
return err
}
monitorStatsColl, err := txApp.FindCachedCollectionByNameOrId("network_monitor_stats")
if err != nil {
slog.Error("Error finding cached collection using container stats:", "err", err)
return err
}
var systems RecordIds
db := txApp.DB()
if err := db.NewQuery("SELECT id FROM systems WHERE status='up'").All(&systems); err != nil {
return err
}
db.NewQuery("SELECT id FROM systems WHERE status='up'").All(&systems)
// loop through all active systems, time periods, and collections
for _, system := range systems {
@@ -100,52 +94,44 @@ func (rm *RecordManager) CreateLongerRecords() {
recordData := longerRecordData[i]
// log.Println("processing longer record type", recordData.longerType)
// add one minute padding for longer records because they are created slightly later than the job start time
longerRecordPeriod := now.Add(recordData.longerTimeDuration + time.Minute)
longerRecordPeriod := time.Now().UTC().Add(recordData.longerTimeDuration + time.Minute)
// shorter records are created independently of longer records, so we shouldn't need to add padding
shorterRecordPeriod := now.Add(recordData.longerTimeDuration)
shorterRecordPeriod := time.Now().UTC().Add(recordData.longerTimeDuration)
// loop through both collections
for _, collection := range collections {
// check creation time of last longer record if not 10m, since 10m is created every run
if recordData.longerType != "10m" {
count, err := txApp.CountRecords(collection.Id, dbx.NewExp(
"system = {:system} AND type = {:type} AND created > {:created}",
dbx.Params{
"type": recordData.longerType,
"system": system.Id,
"created": longerRecordPeriod.Format(types.DefaultDateLayout),
},
))
if err != nil {
return err
}
count, err := txApp.CountRecords(
collection.Id,
dbx.NewExp(
"system = {:system} AND type = {:type} AND created > {:created}",
dbx.Params{"type": recordData.longerType, "system": system.Id, "created": longerRecordPeriod},
),
)
// continue if longer record exists
if count > 0 {
if err != nil || count > 0 {
continue
}
}
// get shorter records from the past x minutes
var recordIds RecordIds
params := dbx.Params{
"type": recordData.shorterType,
"system": system.Id,
"created": shorterRecordPeriod.Format(types.DefaultDateLayout),
}
err := db.
err := txApp.DB().
Select("id").
From(collection.Name).
Where(dbx.NewExp(
AndWhere(dbx.NewExp(
"system={:system} AND type={:type} AND created > {:created}",
params,
dbx.Params{
"type": recordData.shorterType,
"system": system.Id,
"created": shorterRecordPeriod,
},
)).
OrderBy("created").
All(&recordIds)
if err != nil {
return err
}
// continue if not enough shorter records
if len(recordIds) < recordData.minShorterRecords {
if err != nil || len(recordIds) < recordData.minShorterRecords {
continue
}
// average the shorter records and create longer record
@@ -156,88 +142,20 @@ func (rm *RecordManager) CreateLongerRecords() {
case "system_stats":
longerRecord.Set("stats", rm.AverageSystemStats(db, recordIds))
case "container_stats":
longerRecord.Set("stats", rm.AverageContainerStats(db, recordIds))
}
if err := txApp.SaveNoValidate(longerRecord); err != nil {
txApp.Logger().Error("failed to save longer record", "err", err)
slog.Error("failed to save longer record", "err", err)
}
}
}
}
// network_monitor_stats is aggregated per monitor (not per system)
var monitors []struct {
Id string `db:"id"`
System string `db:"system"`
}
// Disabled monitors still have history that must advance through retention tiers.
if err := db.NewQuery("SELECT id, system FROM network_monitors").All(&monitors); err != nil {
return err
}
for _, monitorRec := range monitors {
for i := range longerRecordData {
recordData := longerRecordData[i]
longerRecordPeriod := now.Add(recordData.longerTimeDuration + time.Minute)
shorterRecordPeriod := now.Add(recordData.longerTimeDuration)
if recordData.longerType != "10m" {
count, err := txApp.CountRecords(monitorStatsColl.Id, dbx.NewExp(
"monitor={:monitor} AND type={:type} AND created>{:created}",
dbx.Params{
"monitor": monitorRec.Id,
"type": recordData.longerType,
"created": longerRecordPeriod.UnixMilli(),
},
))
if err != nil {
return err
}
if count > 0 {
continue
}
}
stats, count, err := rm.AverageMonitorStats(db, monitorRec.Id, recordData.shorterType, shorterRecordPeriod.UnixMilli())
if err != nil {
txApp.Logger().Error("failed to average monitor stats", "monitor", monitorRec.Id, "err", err)
continue
}
// Monitor intervals can exceed the aggregation window, so average
// any available records at every level and skip only empty windows.
if count == 0 {
continue
}
longerRecord := core.NewRecord(monitorStatsColl)
longerRecord.Set("system", monitorRec.System)
longerRecord.Set("monitor", monitorRec.Id)
longerRecord.Set("type", recordData.longerType)
longerRecord.Set("created", now.UnixMilli())
longerRecord.Set("res_min", stats.ResMin)
longerRecord.Set("res_max", stats.ResMax)
longerRecord.Set("total_count", stats.TotalCount)
longerRecord.Set("success_count", stats.SuccessCount)
longerRecord.Set("res_sum", stats.ResponseSum)
if err := txApp.SaveNoValidate(longerRecord); err != nil {
txApp.Logger().Error("failed to save monitor longer record", "err", err)
}
}
}
return nil
})
if err != nil {
rm.app.Logger().Error("failed to create longer records", "err", err)
}
}
func getCreatedTimeField(collectionName string, period time.Time) any {
// network_monitor_stats stores created as unix timestamp in ms, not as a date string
if collectionName == "network_monitor_stats" {
return period.UnixMilli()
}
return period.Format(types.DefaultDateLayout)
// log.Println("finished creating longer records", "time (ms)", time.Since(start).Milliseconds())
}
// Calculate the average stats of a list of system_stats records without reflect
@@ -678,36 +596,6 @@ func AverageContainerStatsSlice(records [][]container.Stats) []container.Stats {
return result
}
// AverageMonitorStats merges probe counts and response sums, preserving their
// weights through every retention tier. Failed probes do not contribute latency.
func (rm *RecordManager) AverageMonitorStats(db dbx.Builder, monitorID, recordType string, createdAfter int64) (monitor.Stats, int, error) {
var result struct {
monitor.Stats
Count int `db:"count"`
}
err := db.Select(
"COUNT(*) AS count",
"COALESCE(SUM(total_count), 0) AS total_count",
"COALESCE(SUM(success_count), 0) AS success_count",
"COALESCE(SUM(res_sum), 0) AS res_sum",
"COALESCE(MIN(CASE WHEN success_count > 0 THEN res_min END), 0) AS res_min",
"COALESCE(MAX(CASE WHEN success_count > 0 THEN res_max END), 0) AS res_max",
).From("network_monitor_stats").Where(dbx.NewExp(
"monitor={:monitor} AND type={:type} AND created>{:created}",
dbx.Params{"monitor": monitorID, "type": recordType, "created": createdAfter},
)).One(&result)
if err != nil {
return monitor.Stats{}, 0, err
}
if result.SuccessCount > 0 {
result.ResAvg = twoDecimals(float64(result.ResponseSum) / float64(result.SuccessCount))
}
if result.TotalCount > 0 {
result.Loss = twoDecimals(float64(result.TotalCount-result.SuccessCount) * 100 / float64(result.TotalCount))
}
return result.Stats, result.Count, nil
}
/* Round float to two decimals */
func twoDecimals(value float64) float64 {
return math.Round(value*100) / 100

View File

@@ -3,6 +3,7 @@ package records
import (
"fmt"
"log/slog"
"strings"
"time"
"github.com/pocketbase/dbx"
@@ -59,7 +60,7 @@ func deleteOldAlertsHistory(app core.App, countToKeep, countBeforeDeletion int)
// Deletes system_stats records older than what is displayed in the UI
func deleteOldSystemStats(app core.App) error {
// Collections to process
collections := [3]string{"system_stats", "container_stats", "network_monitor_stats"}
collections := [2]string{"system_stats", "container_stats"}
// Record types and their retention periods
type RecordDeletionData struct {
@@ -75,17 +76,24 @@ func deleteOldSystemStats(app core.App) error {
}
now := time.Now().UTC()
db := app.DB()
for _, collection := range collections {
query := db.Delete(collection, dbx.NewExp("type={:type} AND created<{:created}"))
for _, rd := range recordData {
if _, err := query.Bind(dbx.Params{
"type": rd.recordType,
"created": getCreatedTimeField(collection, now.Add(-rd.retention)),
}).Execute(); err != nil {
return fmt.Errorf("failed to delete from %s: %v", collection, err)
}
// Build the WHERE clause
var conditionParts []string
var params dbx.Params = make(map[string]any)
for i := range recordData {
rd := recordData[i]
// Create parameterized condition for this record type
dateParam := fmt.Sprintf("date%d", i)
conditionParts = append(conditionParts, fmt.Sprintf("(type = '%s' AND created < {:%s})", rd.recordType, dateParam))
params[dateParam] = now.Add(-rd.retention)
}
// Combine conditions with OR
conditionStr := strings.Join(conditionParts, " OR ")
// Construct and execute the full raw query
rawQuery := fmt.Sprintf("DELETE FROM %s WHERE %s", collection, conditionStr)
if _, err := app.DB().NewQuery(rawQuery).Bind(params).Execute(); err != nil {
return fmt.Errorf("failed to delete from %s: %v", collection, err)
}
}
return nil

View File

@@ -1,86 +0,0 @@
//go:build testing
package records_test
import (
"testing"
"time"
"github.com/henrygd/beszel/internal/records"
"github.com/henrygd/beszel/internal/tests"
"github.com/pocketbase/dbx"
"github.com/pocketbase/pocketbase/tools/types"
"github.com/stretchr/testify/require"
)
func TestLongerRecordsPreventDuplicates(t *testing.T) {
for _, collection := range []string{"system_stats", "container_stats", "network_monitor_stats"} {
for _, tier := range []struct {
shorter, longer string
count int
}{
{"10m", "20m", 2},
{"20m", "120m", 6},
{"120m", "480m", 4},
} {
t.Run(collection+"/"+tier.longer, func(t *testing.T) {
hub, err := tests.NewTestHub(t.TempDir())
require.NoError(t, err)
defer hub.Cleanup()
user, err := tests.CreateUser(hub, "rollup@example.com", "testtesttest")
require.NoError(t, err)
sys, err := tests.CreateRecord(hub, "systems", map[string]any{
"name": "rollup-system", "host": "localhost", "port": "45876",
"status": "up", "users": []string{user.Id},
})
require.NoError(t, err)
created := time.Now().UTC().Add(-time.Minute)
data := map[string]any{
"system": sys.Id, "type": tier.shorter,
"created": created.Format(types.DefaultDateLayout),
}
filter := dbx.HashExp{"system": sys.Id, "type": tier.longer}
switch collection {
case "system_stats":
data["stats"] = `{"cpu":10}`
case "container_stats":
data["stats"] = `[{"name":"test","cpu":10}]`
case "network_monitor_stats":
monitor, err := tests.CreateRecord(hub, "network_monitors", map[string]any{
"system": sys.Id, "target": "1.1.1.1", "protocol": "icmp",
"interval": 30, "enabled": true,
})
require.NoError(t, err)
data["monitor"] = monitor.Id
data["created"] = created.UnixMilli()
data["total_count"] = 1
data["success_count"] = 1
data["res_sum"] = 10
data["res_min"] = 10
data["res_max"] = 10
filter["monitor"] = monitor.Id
}
for range tier.count {
_, err := tests.CreateRecord(hub, collection, data)
require.NoError(t, err)
}
rm := records.NewRecordManager(hub)
rm.CreateLongerRecords()
first, err := hub.FindAllRecords(collection, filter)
require.NoError(t, err)
require.Len(t, first, 1)
// The shorter records remain eligible, but the existing longer
// record must prevent another rollup on a subsequent invocation.
rm.CreateLongerRecords()
second, err := hub.FindAllRecords(collection, filter)
require.NoError(t, err)
require.Len(t, second, 1)
require.Equal(t, first[0].Id, second[0].Id)
})
}
}
}

View File

@@ -1,7 +1,7 @@
{
"name": "beszel",
"private": true,
"version": "0.20.0",
"version": "0.19.0",
"type": "module",
"scripts": {
"dev": "vite --host",
@@ -74,4 +74,4 @@
"optionalDependencies": {
"@esbuild/linux-arm64": "^0.21.5"
}
}
}

View File

@@ -61,8 +61,6 @@ export const ActiveAlerts = () => {
<AlertDescription>
{info.triggeredDesc ? (
info.triggeredDesc()
) : alert.name === "NetworkMonitorLoss" ? (
<Trans>One or more monitors exceed {alert.value}% loss</Trans>
) : alert.name === "Status" ? (
<Trans>Connection is down</Trans>
) : info.invert ? (

View File

@@ -30,8 +30,7 @@ export const alertsHistoryColumns: ColumnDef<AlertsHistoryRecord>[] = [
accessorFn: (record) => {
const name = record.name
const info = alertInfo[name]
const label = info?.name().replace("cpu", "CPU") || name
return record.monitor_name ? `${label}: ${record.monitor_name}` : label
return info?.name().replace("cpu", "CPU") || name
},
header: ({ column }) => (
<Button variant="ghost" onClick={() => column.toggleSorting(column.getIsSorted() === "asc")}>

View File

@@ -239,13 +239,13 @@ export function AlertContent({
/** Alerts that fire on first observation have no duration to configure */
const noDuration = alertData.noDuration === true
/** Binary alerts have no threshold to configure */
const noThreshold = !!singleDescription || alertData.noThreshold === true
const noThreshold = !!singleDescription || noDuration
/** Whether enabling the alert reveals anything to configure */
const hasControls = !(noThreshold && noDuration)
const [checked, setChecked] = useState(global ? false : !!alert)
const [min, setMin] = useState(alert?.min || (noDuration ? 0 : 10))
const [value, setValue] = useState(alert?.value ?? (noThreshold ? 0 : (alertData.start ?? 80)))
const [value, setValue] = useState(alert?.value || (noThreshold ? 0 : (alertData.start ?? 80)))
const Icon = alertData.icon
@@ -319,7 +319,7 @@ export function AlertContent({
<div className="grid sm:grid-cols-2 mt-1.5 gap-5 px-4 pb-5 tabular-nums text-muted-foreground">
<Suspense fallback={<div className="h-10" />}>
{!noThreshold && (
<div className={cn(noDuration && "col-span-full")}>
<div>
<p id={`v${name}`} className="text-sm block h-6">
{alertData.invert ? (
<Trans>

View File

@@ -66,7 +66,7 @@ export default function AreaChartDefault({
}) {
const { yAxisWidth, updateYAxisWidth } = useYAxisWidth()
const { isIntersecting, ref } = useIntersectionObserver({ freeze: false })
const sourceData = customData ?? chartData.systemStats ?? []
const sourceData = customData ?? chartData.systemStats
const [displayData, setDisplayData] = useState(sourceData)
const [displayMaxToggled, setDisplayMaxToggled] = useState(maxToggled)
@@ -111,8 +111,6 @@ export default function AreaChartDefault({
})
}, [areasKey, displayMaxToggled])
const XAxis = xAxis(chartData.chartTime, displayData.at(-1)?.created)
return useMemo(() => {
if (displayData.length === 0) {
return null
@@ -148,7 +146,7 @@ export default function AreaChartDefault({
axisLine={false}
/>
)}
{XAxis}
{xAxis(chartData)}
<ChartTooltip
animationEasing="ease-out"
animationDuration={150}
@@ -169,5 +167,5 @@ export default function AreaChartDefault({
</AreaChart>
</ChartContainer>
)
}, [displayData, yAxisWidth, filter, Areas, XAxis])
}, [displayData, yAxisWidth, filter, Areas])
}

View File

@@ -9,21 +9,14 @@ import { memo } from "react"
export default memo(function ChartTimeSelect({
className,
agentVersion,
chartTimeStore = $chartTime,
allowRealtime = true,
}: {
className?: string
agentVersion: SemVer
chartTimeStore?: typeof $chartTime
allowRealtime?: boolean
}) {
const chartTime = useStore(chartTimeStore)
const chartTime = useStore($chartTime)
// remove chart times that are not supported by the system agent version
const availableChartTimes = Object.entries(chartTimeData).filter(([value, { minVersion }]) => {
if (value === "1m" && !allowRealtime) {
return false
}
const availableChartTimes = Object.entries(chartTimeData).filter(([_, { minVersion }]) => {
if (!minVersion) {
return true
}
@@ -31,7 +24,7 @@ export default memo(function ChartTimeSelect({
})
return (
<Select defaultValue="1h" value={chartTime} onValueChange={(value: ChartTimes) => chartTimeStore.set(value)}>
<Select defaultValue="1h" value={chartTime} onValueChange={(value: ChartTimes) => $chartTime.set(value)}>
<SelectTrigger className={cn(className, "relative ps-10 pe-5")}>
<HistoryIcon className="h-4 w-4 absolute start-4 top-1/2 -translate-y-1/2 opacity-85" />
<SelectValue />

View File

@@ -22,10 +22,6 @@ export type DataPoint<T = SystemStatsRecord> = {
order?: number
strokeOpacity?: number
activeDot?: boolean
dot?: boolean
/** Which Y axis this series plots against. Defaults to "left". */
yAxisId?: "left" | "right"
strokeDasharray?: string
}
export default function LineChartDefault({
@@ -34,12 +30,9 @@ export default function LineChartDefault({
max,
maxToggled,
tickFormatter,
tickFormatter2,
contentFormatter,
dataPoints,
domain,
domain2,
max2,
legend,
itemSorter,
showTotal = false,
@@ -48,24 +41,18 @@ export default function LineChartDefault({
filter,
truncate = false,
chartProps,
connectNulls,
}: {
chartData: ChartData
// biome-ignore lint/suspicious/noExplicitAny: accepts different data source types (systemStats or containerData)
customData?: any[]
max?: number
max2?: number
maxToggled?: boolean
tickFormatter: (value: number, index: number) => string
/** Tick formatter for the right ("right"-yAxisId) axis, when any dataPoint uses it. */
tickFormatter2?: (value: number, index: number) => string
// biome-ignore lint/suspicious/noExplicitAny: recharts tooltip item interop
contentFormatter: (item: any, key: string) => ReactNode
// biome-ignore lint/suspicious/noExplicitAny: accepts DataPoint with different generic types
dataPoints?: DataPoint<any>[]
domain?: AxisDomain
/** Domain for the right axis, when any dataPoint uses it. */
domain2?: AxisDomain
legend?: boolean
showTotal?: boolean
// biome-ignore lint/suspicious/noExplicitAny: recharts tooltip item interop
@@ -75,15 +62,10 @@ export default function LineChartDefault({
filter?: string
truncate?: boolean
chartProps?: Omit<React.ComponentProps<typeof LineChart>, "data" | "margin">
connectNulls?: boolean
}) {
const { yAxisWidth, updateYAxisWidth } = useYAxisWidth()
const hasRightAxis = !!dataPoints?.some((dp) => dp.yAxisId === "right")
// fixed width for the secondary axis rather than measured, since its labels (e.g. loss %) are short
// and predictable, and this avoids depending on a second async width-measurement pass to settle
const rightAxisWidth = 38
const { isIntersecting, ref } = useIntersectionObserver({ freeze: false })
const sourceData = customData ?? chartData.systemStats ?? []
const sourceData = customData ?? chartData.systemStats
const [displayData, setDisplayData] = useState(sourceData)
const [displayMaxToggled, setDisplayMaxToggled] = useState(maxToggled)
@@ -101,9 +83,7 @@ export default function LineChartDefault({
}, [displayData, displayMaxToggled, isIntersecting, maxToggled, sourceData])
// Use a stable key derived from data point identities and visual properties
const linesKey = dataPoints?.map((d) => `${d.label}:${d.strokeOpacity}${d.dot}${d.yAxisId}${d.strokeDasharray}`).join("\0")
const XAxis = xAxis(chartData.chartTime, displayData.at(-1)?.created)
const linesKey = dataPoints?.map((d) => `${d.label}:${d.strokeOpacity ?? ""}`).join("\0")
const Lines = useMemo(() => {
return dataPoints?.map((dataPoint, i) => {
@@ -114,20 +94,17 @@ export default function LineChartDefault({
return (
<Line
key={dataPoint.label}
yAxisId={dataPoint.yAxisId ?? "left"}
dataKey={dataPoint.dataKey}
name={dataPoint.label}
type="monotoneX"
dot={dataPoint.dot || false}
dot={false}
strokeWidth={1.5}
stroke={color}
strokeOpacity={dataPoint.strokeOpacity}
strokeDasharray={dataPoint.strokeDasharray}
isAnimationActive={false}
// stackId={dataPoint.stackId}
order={dataPoint.order || i}
activeDot={dataPoint.activeDot ?? true}
connectNulls={connectNulls}
/>
)
})
@@ -158,7 +135,6 @@ export default function LineChartDefault({
<CartesianGrid vertical={false} />
{!hideYAxis && (
<YAxis
yAxisId="left"
direction="ltr"
orientation={chartData.orientation}
className="tracking-tighter"
@@ -169,20 +145,7 @@ export default function LineChartDefault({
axisLine={false}
/>
)}
{!hideYAxis && hasRightAxis && (
<YAxis
yAxisId="right"
direction="ltr"
orientation={chartData.orientation === "left" ? "right" : "left"}
className="tracking-tighter"
width={rightAxisWidth}
domain={domain2 ?? [0, max2 ?? "auto"]}
tickFormatter={tickFormatter2 ?? tickFormatter}
tickLine={false}
axisLine={false}
/>
)}
{XAxis}
{xAxis(chartData)}
<ChartTooltip
animationEasing="ease-out"
animationDuration={150}
@@ -203,5 +166,5 @@ export default function LineChartDefault({
</LineChart>
</ChartContainer>
)
}, [displayData, yAxisWidth, hasRightAxis, filter, Lines, XAxis])
}, [displayData, yAxisWidth, filter, Lines])
}

Some files were not shown because too many files have changed in this diff Show More