diff --git a/glances/cpu_sampler_v5.py b/glances/cpu_sampler_v5.py index d371d4bf..5293f198 100755 --- a/glances/cpu_sampler_v5.py +++ b/glances/cpu_sampler_v5.py @@ -73,12 +73,49 @@ class CpuSamplerV5: # ----------------------------------------------------------- aggregate + @staticmethod + def _is_unsettled(sample: Any) -> bool: + """Detect a psutil sample taken before its baseline is built. + + ``psutil.cpu_times_percent(interval=0.0)`` returns ``0.0`` for + every field on the very first call (no anchor). The next few + calls — until enough wall time has elapsed since the anchor — + return partial samples that don't sum to ~100% (typical + signature: ``idle≈1.0, every other field == 0.0``). Cache-ing + such a sample would propagate the "100% CPU spike" for a full + TTL window after startup. + + Heuristic: a settled cpu_times_percent sample sums to roughly + 100%. Below 50% means no real baseline yet. + """ + names = ("user", "system", "idle", "nice", "iowait", "irq", "softirq", "steal", "guest", "dpc") + total = 0.0 + for n in names: + v = getattr(sample, n, 0.0) + if isinstance(v, (int, float)): + total += float(v) + return total < 50.0 + + async def _fetch_aggregate(self, percpu: bool = False) -> Any: + """Pull a psutil sample; if it looks unsettled, sleep briefly and + retry once. Caller is responsible for holding ``self._lock``.""" + result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu) + first = result[0] if (percpu and result) else result + if first is not None and self._is_unsettled(first): + # Give psutil's anchor enough wall time to register a real + # delta on the next call. 50 ms is comfortably above the + # ~1 ms granularity below which psutil reports all zeros and + # imperceptible to the user at startup. + await asyncio.sleep(0.05) + result = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=percpu) + return result + async def get_aggregate(self) -> Any: """System-wide ``cpu_times_percent`` (cached over ``ttl``).""" async with self._lock: if self._is_fresh(self._aggregate_ts) and self._aggregate is not None: return self._aggregate - self._aggregate = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0) + self._aggregate = await self._fetch_aggregate(percpu=False) self._aggregate_ts = time.monotonic() return self._aggregate @@ -89,7 +126,7 @@ class CpuSamplerV5: async with self._lock: if self._is_fresh(self._per_core_ts) and self._per_core: return self._per_core - self._per_core = await asyncio.to_thread(psutil.cpu_times_percent, interval=0.0, percpu=True) + self._per_core = await self._fetch_aggregate(percpu=True) self._per_core_ts = time.monotonic() return self._per_core diff --git a/glances/plugins/cpu/model_v5.py b/glances/plugins/cpu/model_v5.py index 03544ae0..3dcb9be7 100755 --- a/glances/plugins/cpu/model_v5.py +++ b/glances/plugins/cpu/model_v5.py @@ -185,9 +185,11 @@ class PluginModel(GlancesPluginBase[dict]): } async def _grab_stats(self) -> dict: - # Both psutil calls are coalesced through the shared sampler — when the - # `percpu` plugin is updated in the same scheduler tick, only one - # psutil sample fires per sub-call. + # Both psutil calls are coalesced through the shared sampler — + # when the `percpu` plugin is updated in the same scheduler + # tick, only one psutil sample fires per sub-call. The sampler + # also guards against psutil's "no baseline yet" first-call + # behaviour (cf. `_is_unsettled`). agg = await sampler.get_aggregate() cpu_stats = await sampler.get_stats() diff --git a/glances/plugins/percpu/model_v5.py b/glances/plugins/percpu/model_v5.py index f449c0a4..51ab1e18 100755 --- a/glances/plugins/percpu/model_v5.py +++ b/glances/plugins/percpu/model_v5.py @@ -111,6 +111,8 @@ class PluginModel(GlancesPluginBase[list]): } async def _grab_stats(self) -> list: + # The shared sampler guards against psutil's "no baseline yet" + # first-call behaviour (cf. `cpu_sampler_v5._is_unsettled`). per_core = await sampler.get_per_core() out: list[dict[str, Any]] = [] for cpu_number, cpu_times in enumerate(per_core): diff --git a/tests/test_cpu_sampler_v5.py b/tests/test_cpu_sampler_v5.py index 582b4d1b..074a4359 100755 --- a/tests/test_cpu_sampler_v5.py +++ b/tests/test_cpu_sampler_v5.py @@ -159,3 +159,68 @@ def test_module_level_singleton_exists(): from glances.cpu_sampler_v5 import sampler assert isinstance(sampler, CpuSamplerV5) + + +# ---------------------------------------------------------- unsettled-sample guard + + +def test_is_unsettled_detects_all_zero_sample(): + """psutil returns 0.0 everywhere on the first call (no baseline).""" + zeroed = _agg(idle=0.0)._replace(user=0.0, system=0.0, nice=0.0, iowait=0.0) + assert CpuSamplerV5._is_unsettled(zeroed) is True + + +def test_is_unsettled_detects_partial_first_call(): + """Real-world bug: first call after init returns e.g. idle=1.0 and + everything else 0.0 — sum ≪ 100 → unsettled.""" + partial = CpuTimesPercent( + user=0.0, + system=0.0, + idle=1.0, + nice=0.0, + iowait=0.0, + irq=0.0, + softirq=0.0, + steal=0.0, + guest=0.0, + guest_nice=0.0, + ) + assert CpuSamplerV5._is_unsettled(partial) is True + + +def test_is_unsettled_accepts_settled_sample(): + """A real sample sums to ~100% across the time-percent fields.""" + settled = _agg(idle=72.5) # user=10+system=15+idle=72.5+... ≈ 100 + assert CpuSamplerV5._is_unsettled(settled) is False + + +async def test_fetch_aggregate_retries_after_unsettled_sample(): + """If the first psutil call returns an unsettled sample, the sampler + sleeps briefly and re-samples once.""" + sampler = CpuSamplerV5(ttl=10.0) + unsettled = CpuTimesPercent( + user=0.0, + system=0.0, + idle=1.0, + nice=0.0, + iowait=0.0, + irq=0.0, + softirq=0.0, + steal=0.0, + guest=0.0, + guest_nice=0.0, + ) + settled = _agg(idle=72.0) + results = [unsettled, settled] + + def stub(*args, **kwargs): + return results.pop(0) + + with patch("glances.cpu_sampler_v5.psutil.cpu_times_percent", side_effect=stub): + # Bypass the asyncio.sleep so the test runs instantly. + with patch("glances.cpu_sampler_v5.asyncio.sleep") as fake_sleep: + fake_sleep.return_value = None + actual = await sampler.get_aggregate() + + assert actual.idle == 72.0 # the settled second sample is what we cached + assert results == [] # both samples were consumed