fix(agent): skip first disk I/O sample of an interval when its window is too short

Right after agent start the hub requests stats immediately, and the first
sample of the interval was measured from the baseline taken at startup, only
a second or so earlier. A burst of startup I/O was then stored as the rate
for the whole minute, causing large spikes in disk I/O charts.

The first sample of an interval now must span at least half the interval.
Otherwise it only stores the snapshot, and the next sample is measured from it.
This commit is contained in:
henrygd
2026-09-30 13:02:29 -04:00
parent e925fafd7b
commit 266f45f104
2 changed files with 67 additions and 12 deletions

View File

@@ -679,11 +679,11 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
}
// Previous snapshot for this interval and device
prev, hasPrev := a.diskPrev[cacheTimeMs][name]
if !hasPrev {
prev, ok := a.diskPrev[cacheTimeMs][name]
firstSample := !ok
if firstSample {
// Seed from the latest counters of any interval, else seed from current
prev, hasPrev = a.diskBaseline[name]
if !hasPrev {
if prev, ok = a.diskBaseline[name]; !ok {
prev = prevDiskFromCounter(d, now)
}
}
@@ -697,6 +697,12 @@ func (a *Agent) updateDiskIo(cacheTimeMs uint16, systemStats *system.Stats) {
if msElapsed < 100 {
continue
}
// The first sample of an interval must span at least half the interval.
// Right after agent start the baseline is only a second or so old, and a
// burst of startup I/O would be recorded as the rate for the whole interval.
if firstSample && msElapsed < uint64(cacheTimeMs)/2 {
continue
}
diskIORead := (d.ReadBytes - prev.readBytes) * 1000 / msElapsed
diskIOWrite := (d.WriteBytes - prev.writeBytes) * 1000 / msElapsed