From 80b836fa9efdf4cce41431bb32c9a58d8ddfd958 Mon Sep 17 00:00:00 2001 From: Turgut Kural <58116817+TurgutKural@users.noreply.github.com> Date: Fri, 21 Aug 2026 10:46:12 +0300 Subject: [PATCH] fix(compression): preflight display-seed must not overwrite real provider usage The preflight rough-estimate seed used 'last >= 0' semantics, so any provider that reports real prompt_tokens got its reading replaced by the schema/reasoning-inflated rough estimate whenever the estimate was larger. last_prompt_tokens feeds both the CLI context meter (cli.py) and the post-response compression gate (conversation_loop.py), so one seed made the bar jump to an inflated number AND could push the real-usage gate over threshold on estimator noise. Observed in a production CLI session on a 1M-token window with a reasoning-heavy history: the status bar showed ~492K real provider prompt tokens; the next turn's preflight estimated ~685K (rough estimates can inflate 1.4-2.5x on reasoning-heavy sessions, #81481) and seeded it into last_prompt_tokens; compression then fired at ~69% of the window while real usage was ~49%. Policy change: a real provider reading (>0) always wins. The seed now only fills the 0 state ('no reading yet'), keeping the status bar live for usage-less providers (#34282's motivation); -1 remains protected as the post-compression sentinel (#36718). Updates TestPreflightSentinelGuard to encode the new policy and adds a regression test with the measured 492K/685K numbers. --- agent/turn_context.py | 18 ++++++++++++++++-- tests/agent/test_context_compressor.py | 18 ++++++++++++++---- 2 files changed, 30 insertions(+), 6 deletions(-) diff --git a/agent/turn_context.py b/agent/turn_context.py index 8f333870cf..e664034db7 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -1067,8 +1067,22 @@ def build_turn_context( if not _preflight_deferred: _last = _compressor.last_prompt_tokens - # Do NOT overwrite the -1 sentinel (#36718). - if _last >= 0 and _preflight_tokens > _last: + # The seed exists so the status bar shows current occupancy when a + # provider reports no usage (#34282's motivation). But + # ``last_prompt_tokens`` is also the post-response compression + # gate's "real tokens" input (conversation_loop) and the CLI + # context meter's source (cli.py). Overwriting a REAL provider + # reading with the rough preflight estimate makes the bar jump to + # an inflated number and can push the real-usage gate over the + # threshold on estimator noise — compression then fires at far + # below the user-configured threshold (observed: 492K real vs + # ~685K rough on a 1M window; reasoning-heavy sessions inflate + # the rough estimate 1.4-2.5x, see #81481). + # + # Policy: a real provider reading (>0) always wins. Seed only + # from the 0 state ("no reading yet"); -1 stays protected as the + # post-compression sentinel (#36718). + if _last == 0 and _preflight_tokens > _last: _compressor.last_prompt_tokens = _preflight_tokens _compression_cooldown = getattr( diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index 2b822a10cd..ae87008e04 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -2463,9 +2463,9 @@ class TestPreflightSentinelGuard: """ def _seed(self, last_prompt_tokens, preflight_tokens): - # Mirror the exact guard in agent/conversation_loop.py run_conversation. + # Mirror the exact guard in agent/turn_context.py build_turn_context. _last = last_prompt_tokens - if _last >= 0 and preflight_tokens > _last: + if _last == 0 and preflight_tokens > _last: return preflight_tokens # would overwrite return last_prompt_tokens # preserved @@ -2475,11 +2475,21 @@ class TestPreflightSentinelGuard: result = self._seed(compressor.last_prompt_tokens, 250_000) assert result == -1 - def test_real_value_still_revises_upward(self, compressor): - compressor.last_prompt_tokens = 10_000 + def test_zero_state_still_seeded(self, compressor): + # 0 means "no reading yet" — the seed keeps the status bar live when + # providers report no usage. + compressor.last_prompt_tokens = 0 result = self._seed(compressor.last_prompt_tokens, 50_000) assert result == 50_000 + def test_real_provider_reading_wins_over_rough_estimate(self, compressor): + # Regression for the 492K-vs-685K display jump: a real provider + # reading must never be replaced by the schema/reasoning-inflated + # rough preflight estimate (#81481 class inflation). + compressor.last_prompt_tokens = 492_000 + result = self._seed(compressor.last_prompt_tokens, 685_344) + assert result == 492_000 + class TestTurnPairPreservation: