mirror of
https://github.com/nicolargo/glances.git
synced 2026-10-04 16:05:19 -04:00
fix(v5): guard against psutil's no-baseline first cpu_times_percent sample
psutil.cpu_times_percent(interval=0.0) requires two anchor points to compute a delta. Before the second anchor is laid down (~1 ms after the first), it returns either all zeros or a partial sample like (idle=1.0, every other field=0.0). The cpu plugin computes total = 100 - idle, so an unsettled sample produces total=99-100% — a visible spike that persists for one TTL window (1 s by default) after startup, while v4 stays at the real value because CpuPercent primes psutil in __init__. Sampler-level fix: detect unsettled samples (sum of time-percent fields < 50%), sleep 50 ms, and re-sample once before caching. The guard runs inside the asyncio lock so concurrent get_aggregate calls share the same retry. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
1 parent
ff1784dbe8
commit
656d902585
4 files changed
+111
-5
No files matched your search
@@ -73,12 +73,49 @@ class CpuSamplerV5:
|
||||
|
||||
# ----------------------------------------------------------- aggregate
|
||||
|
||||
@staticmethod
|
||||
def _is_unsettled(sample: Any) -> bool:
|
||||
"""Detect a psutil sample taken before its baseline is built.
|
||||
|
||||
``psutil.cpu_times_percent(interval=0.0)`` returns ``0.0`` for
|
||||
every field on the very first call (no anchor). The next few
|
||||
calls — until enough wall time has elapsed since the anchor —
|
||||
return partial samples that don't sum to ~100% (typical
|
||||
signature: ``idle≈1.0, every other field == 0.0``). Cache-ing
|
||||
such a sample would propagate the "100% CPU spike" for a full
|
||||
TTL window after startup.
|
||||
|
||||
Heuristic: a settled cpu_times_percent sample sums to roughly
|
||||
100%. Below 50% means no real baseline yet.
|
||||
"""
|
||||
names = ("user", "system", "idle", "nice", "iowait", "irq", "softirq", "steal", "guest", "dpc")
|
||||
total = 0.0
|
||||
for n in names:
|
||||
v = getattr(sample, n, 0.0)
|
||||
if isinstance(v, (int, float)):
|
||||
total += float(v)
|
||||
return total < 50.0
|
||||
|
||||
async def _fetch_aggregate(self, percpu: bool = False) -> Any:
|
||||
"""Pull a psutil sample; if it looks unsettled, sleep briefly and
|
||||
retry once. Caller is responsible for holding ``self._lock``."""
|
||||
result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu)
|
||||
first = result[0] if (percpu and result) else result
|
||||
if first is not None and self._is_unsettled(first):
|
||||
# Give psutil's anchor enough wall time to register a real
|
||||
# delta on the next call. 50 ms is comfortably above the
|
||||
# ~1 ms granularity below which psutil reports all zeros and
|
||||
# imperceptible to the user at startup.
|
||||
await asyncio.sleep(0.05)
|
||||
result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu)
|
||||
return result
|
||||
|
||||
async def get_aggregate(self) -> Any:
|
||||
"""System-wide ``cpu_times_percent`` (cached over ``ttl``)."""
|
||||
async with self._lock:
|
||||
if self._is_fresh(self._aggregate_ts) and self._aggregate is not None:
|
||||
return self._aggregate
|
||||
self._aggregate = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0)
|
||||
self._aggregate = await self._fetch_aggregate(percpu=False)
|
||||
self._aggregate_ts = time.monotonic()
|
||||
return self._aggregate
|
||||
|
||||
@@ -89,7 +126,7 @@ class CpuSamplerV5:
|
||||
async with self._lock:
|
||||
if self._is_fresh(self._per_core_ts) and self._per_core:
|
||||
return self._per_core
|
||||
self._per_core = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=True)
|
||||
self._per_core = await self._fetch_aggregate(percpu=True)
|
||||
self._per_core_ts = time.monotonic()
|
||||
return self._per_core
|
||||
|
||||
|
||||
@@ -185,9 +185,11 @@ class PluginModel(GlancesPluginBase[dict]):
|
||||
}
|
||||
|
||||
async def _grab_stats(self) -> dict:
|
||||
# Both psutil calls are coalesced through the shared sampler — when the
|
||||
# `percpu` plugin is updated in the same scheduler tick, only one
|
||||
# psutil sample fires per sub-call.
|
||||
# Both psutil calls are coalesced through the shared sampler —
|
||||
# when the `percpu` plugin is updated in the same scheduler
|
||||
# tick, only one psutil sample fires per sub-call. The sampler
|
||||
# also guards against psutil's "no baseline yet" first-call
|
||||
# behaviour (cf. `_is_unsettled`).
|
||||
agg = await sampler.get_aggregate()
|
||||
cpu_stats = await sampler.get_stats()
|
||||
|
||||
|
||||
@@ -111,6 +111,8 @@ class PluginModel(GlancesPluginBase[list]):
|
||||
}
|
||||
|
||||
async def _grab_stats(self) -> list:
|
||||
# The shared sampler guards against psutil's "no baseline yet"
|
||||
# first-call behaviour (cf. `cpu_sampler_v5._is_unsettled`).
|
||||
per_core = await sampler.get_per_core()
|
||||
out: list[dict[str, Any]] = []
|
||||
for cpu_number, cpu_times in enumerate(per_core):
|
||||
|
||||
@@ -159,3 +159,68 @@ def test_module_level_singleton_exists():
|
||||
from glances.cpu_sampler_v5 import sampler
|
||||
|
||||
assert isinstance(sampler, CpuSamplerV5)
|
||||
|
||||
|
||||
# ---------------------------------------------------------- unsettled-sample guard
|
||||
|
||||
|
||||
def test_is_unsettled_detects_all_zero_sample():
|
||||
"""psutil returns 0.0 everywhere on the first call (no baseline)."""
|
||||
zeroed = _agg(idle=0.0)._replace(user=0.0, system=0.0, nice=0.0, iowait=0.0)
|
||||
assert CpuSamplerV5._is_unsettled(zeroed) is True
|
||||
|
||||
|
||||
def test_is_unsettled_detects_partial_first_call():
|
||||
"""Real-world bug: first call after init returns e.g. idle=1.0 and
|
||||
everything else 0.0 — sum ≪ 100 → unsettled."""
|
||||
partial = CpuTimesPercent(
|
||||
user=0.0,
|
||||
system=0.0,
|
||||
idle=1.0,
|
||||
nice=0.0,
|
||||
iowait=0.0,
|
||||
irq=0.0,
|
||||
softirq=0.0,
|
||||
steal=0.0,
|
||||
guest=0.0,
|
||||
guest_nice=0.0,
|
||||
)
|
||||
assert CpuSamplerV5._is_unsettled(partial) is True
|
||||
|
||||
|
||||
def test_is_unsettled_accepts_settled_sample():
|
||||
"""A real sample sums to ~100% across the time-percent fields."""
|
||||
settled = _agg(idle=72.5) # user=10+system=15+idle=72.5+... ≈ 100
|
||||
assert CpuSamplerV5._is_unsettled(settled) is False
|
||||
|
||||
|
||||
async def test_fetch_aggregate_retries_after_unsettled_sample():
|
||||
"""If the first psutil call returns an unsettled sample, the sampler
|
||||
sleeps briefly and re-samples once."""
|
||||
sampler = CpuSamplerV5(ttl=10.0)
|
||||
unsettled = CpuTimesPercent(
|
||||
user=0.0,
|
||||
system=0.0,
|
||||
idle=1.0,
|
||||
nice=0.0,
|
||||
iowait=0.0,
|
||||
irq=0.0,
|
||||
softirq=0.0,
|
||||
steal=0.0,
|
||||
guest=0.0,
|
||||
guest_nice=0.0,
|
||||
)
|
||||
settled = _agg(idle=72.0)
|
||||
results = [unsettled, settled]
|
||||
|
||||
def stub(*args, **kwargs):
|
||||
return results.pop(0)
|
||||
|
||||
with patch("glances.cpu_sampler_v5.psutil.cpu_times_percent", side_effect=stub):
|
||||
# Bypass the asyncio.sleep so the test runs instantly.
|
||||
with patch("glances.cpu_sampler_v5.asyncio.sleep") as fake_sleep:
|
||||
fake_sleep.return_value = None
|
||||
actual = await sampler.get_aggregate()
|
||||
|
||||
assert actual.idle == 72.0 # the settled second sample is what we cached
|
||||
assert results == [] # both samples were consumed
|
||||
Reference in new issue
Block a user