fix(v5): guard against psutil's no-baseline first cpu_times_percent sample

psutil.cpu_times_percent(interval=0.0) requires two anchor points to
compute a delta. Before the second anchor is laid down (~1 ms after the
first), it returns either all zeros or a partial sample like
(idle=1.0, every other field=0.0). The cpu plugin computes
total = 100 - idle, so an unsettled sample produces total=99-100% — a
visible spike that persists for one TTL window (1 s by default) after
startup, while v4 stays at the real value because CpuPercent primes
psutil in __init__.

Sampler-level fix: detect unsettled samples (sum of time-percent
fields < 50%), sleep 50 ms, and re-sample once before caching. The
guard runs inside the asyncio lock so concurrent get_aggregate calls
share the same retry.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
nicolargoandClaude Opus 4.7 committed 2026-05-13 16:00:57 +02:00
1 parent ff1784dbe8
commit 656d902585
4 files changed
+111 -5

No files matched your search

+39 -2
View File
@@ -73,12 +73,49 @@ class CpuSamplerV5:
# ----------------------------------------------------------- aggregate
@staticmethod
def _is_unsettled(sample: Any) -> bool:
"""Detect a psutil sample taken before its baseline is built.
``psutil.cpu_times_percent(interval=0.0)`` returns ``0.0`` for
every field on the very first call (no anchor). The next few
calls — until enough wall time has elapsed since the anchor —
return partial samples that don't sum to ~100% (typical
signature: ``idle≈1.0, every other field == 0.0``). Cache-ing
such a sample would propagate the "100% CPU spike" for a full
TTL window after startup.
Heuristic: a settled cpu_times_percent sample sums to roughly
100%. Below 50% means no real baseline yet.
"""
names = ("user", "system", "idle", "nice", "iowait", "irq", "softirq", "steal", "guest", "dpc")
total = 0.0
for n in names:
v = getattr(sample, n, 0.0)
if isinstance(v, (int, float)):
total += float(v)
return total < 50.0
async def _fetch_aggregate(self, percpu: bool = False) -> Any:
"""Pull a psutil sample; if it looks unsettled, sleep briefly and
retry once. Caller is responsible for holding ``self._lock``."""
result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu)
first = result[0] if (percpu and result) else result
if first is not None and self._is_unsettled(first):
# Give psutil's anchor enough wall time to register a real
# delta on the next call. 50 ms is comfortably above the
# ~1 ms granularity below which psutil reports all zeros and
# imperceptible to the user at startup.
await asyncio.sleep(0.05)
result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu)
return result
async def get_aggregate(self) -> Any:
"""System-wide ``cpu_times_percent`` (cached over ``ttl``)."""
async with self._lock:
if self._is_fresh(self._aggregate_ts) and self._aggregate is not None:
return self._aggregate
self._aggregate = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0)
self._aggregate = await self._fetch_aggregate(percpu=False)
self._aggregate_ts = time.monotonic()
return self._aggregate
@@ -89,7 +126,7 @@ class CpuSamplerV5:
async with self._lock:
if self._is_fresh(self._per_core_ts) and self._per_core:
return self._per_core
self._per_core = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=True)
self._per_core = await self._fetch_aggregate(percpu=True)
self._per_core_ts = time.monotonic()
return self._per_core
+5 -3
View File
@@ -185,9 +185,11 @@ class PluginModel(GlancesPluginBase[dict]):
}
async def _grab_stats(self) -> dict:
# Both psutil calls are coalesced through the shared sampler — when the
# `percpu` plugin is updated in the same scheduler tick, only one
# psutil sample fires per sub-call.
# Both psutil calls are coalesced through the shared sampler —
# when the `percpu` plugin is updated in the same scheduler
# tick, only one psutil sample fires per sub-call. The sampler
# also guards against psutil's "no baseline yet" first-call
# behaviour (cf. `_is_unsettled`).
agg = await sampler.get_aggregate()
cpu_stats = await sampler.get_stats()
+2
View File
@@ -111,6 +111,8 @@ class PluginModel(GlancesPluginBase[list]):
}
async def _grab_stats(self) -> list:
# The shared sampler guards against psutil's "no baseline yet"
# first-call behaviour (cf. `cpu_sampler_v5._is_unsettled`).
per_core = await sampler.get_per_core()
out: list[dict[str, Any]] = []
for cpu_number, cpu_times in enumerate(per_core):
+65
View File
@@ -159,3 +159,68 @@ def test_module_level_singleton_exists():
from glances.cpu_sampler_v5 import sampler
assert isinstance(sampler, CpuSamplerV5)
# ---------------------------------------------------------- unsettled-sample guard
def test_is_unsettled_detects_all_zero_sample():
"""psutil returns 0.0 everywhere on the first call (no baseline)."""
zeroed = _agg(idle=0.0)._replace(user=0.0, system=0.0, nice=0.0, iowait=0.0)
assert CpuSamplerV5._is_unsettled(zeroed) is True
def test_is_unsettled_detects_partial_first_call():
"""Real-world bug: first call after init returns e.g. idle=1.0 and
everything else 0.0 — sum ≪ 100 → unsettled."""
partial = CpuTimesPercent(
user=0.0,
system=0.0,
idle=1.0,
nice=0.0,
iowait=0.0,
irq=0.0,
softirq=0.0,
steal=0.0,
guest=0.0,
guest_nice=0.0,
)
assert CpuSamplerV5._is_unsettled(partial) is True
def test_is_unsettled_accepts_settled_sample():
"""A real sample sums to ~100% across the time-percent fields."""
settled = _agg(idle=72.5) # user=10+system=15+idle=72.5+... ≈ 100
assert CpuSamplerV5._is_unsettled(settled) is False
async def test_fetch_aggregate_retries_after_unsettled_sample():
"""If the first psutil call returns an unsettled sample, the sampler
sleeps briefly and re-samples once."""
sampler = CpuSamplerV5(ttl=10.0)
unsettled = CpuTimesPercent(
user=0.0,
system=0.0,
idle=1.0,
nice=0.0,
iowait=0.0,
irq=0.0,
softirq=0.0,
steal=0.0,
guest=0.0,
guest_nice=0.0,
)
settled = _agg(idle=72.0)
results = [unsettled, settled]
def stub(*args, **kwargs):
return results.pop(0)
with patch("glances.cpu_sampler_v5.psutil.cpu_times_percent", side_effect=stub):
# Bypass the asyncio.sleep so the test runs instantly.
with patch("glances.cpu_sampler_v5.asyncio.sleep") as fake_sleep:
fake_sleep.return_value = None
actual = await sampler.get_aggregate()
assert actual.idle == 72.0 # the settled second sample is what we cached
assert results == [] # both samples were consumed