From 0f4587e336f4f7ca2528db6862b6f2e4c94f1075 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 12:41:13 -0700 Subject: [PATCH] refactor(compression): every compaction gate asks real usage first; rough estimates only decide whether to wait MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two parallel "real usage" mechanisms fought each other: the usage anchor (real + delta) and the compressor's rough/real projection (should_defer_preflight_to_real_usage with last_rough_tokens_when_real_prompt_fit / _pending_request_rough_tokens / note_request_rough_estimate baselines). The projection stored an anchored, real-scale figure as its "rough" baseline, so a rewind that invalidated the anchor produced phantom growth and a spurious compaction (#103391). Now there is one authority: - Post-tool gate (turn_preflight.compress_after_tool_results): anchored figure first (the raw last_prompt_tokens ignored the tool results just appended), then real, then rough. - Gateway hygiene (run_turn._hmwa_hygiene_plan): real session count, else the anchor persisted on the session row, else rough. - Preflight / pre-API gates: an anchored figure is never deferred. A whole-context rough estimate over threshold waits ONE request for the provider's real count instead of compressing on a guess (first request, rewind/edit-resend, reloaded history without a persisted anchor). - The wait is one request, never a disable: a provider that omits usage (note_usage_less_response, #2153 class), a real reading already over threshold, a rough figure past the whole window, and provider-proven overflow all compress immediately; the post-compaction latch (#36718 / #104192) is unchanged. - Projection baselines and their bookkeeping deleted (-101 LOC in context_compressor); the fixtures that scripted whole-history estimates now state the fact they relied on (provider omits usage). Fixes #103391 (closes #103397 by construction — the baseline it repaired no longer exists). --- agent/agent_init.py | 1 + agent/codex_runtime.py | 2 + agent/context_compressor.py | 53 +++---- agent/turn_context.py | 1 + agent/turn_context_compaction.py | 7 +- agent/turn_preflight.py | 33 ++--- agent/turn_preflight_gate.py | 7 +- agent/turn_request_assembly.py | 6 +- agent/turn_usage.py | 3 + gateway/run_turn.py | 15 +- tests/agent/test_context_compressor.py | 133 ++++-------------- ..._context_compressor_cross_session_guard.py | 1 - ...ext_compressor_session_end_clears_state.py | 2 - .../test_turn_context_overflow_warning.py | 8 +- tests/run_agent/test_413_compression.py | 53 +++---- .../test_compression_budget_rearm.py | 6 +- 16 files changed, 115 insertions(+), 216 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index 5a5f22c1d8..562dfcc328 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -2136,6 +2136,7 @@ _USAGE_STATE: Dict[str, Any] = { # snapshot; invalidated on compaction/session switch so stale anchors never suppress compression. "_usage_anchor": None, "_turn_base_usage_anchor": None, + "_request_pressure_anchored": False, # whether the last pressure figure came from the anchor # Cumulative token usage for the session "session_prompt_tokens": 0, "session_completion_tokens": 0, diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index c90e2d3852..f15732f708 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -98,6 +98,8 @@ def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any] if compressor is not None and getattr(compressor, "awaiting_real_usage_after_compression", False): # No usage cannot adjudicate the pending compaction; unlatch preflight deferral. compressor.update_from_response({}) + if compressor is not None and callable(getattr(compressor, "note_usage_less_response", None)): + compressor.note_usage_less_response() _queue_token_counts(agent, "Codex app-server api-call persistence failed (session=%s): %s", counts=lambda: billing(billing_mode="subscription_included")) return {} diff --git a/agent/context_compressor.py b/agent/context_compressor.py index e9242fa371..cea803e34f 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1818,10 +1818,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._reset_session_compaction_state() def _reset_real_usage_pairing(self) -> None: - """Forget the real-vs-rough token pairing used by should_defer_preflight_to_real_usage().""" + """Forget the real-usage state read by real_usage_pending().""" self.last_real_prompt_tokens = self.last_compression_rough_tokens = 0 - self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens = 0 - self.awaiting_real_usage_after_compression = False + self.awaiting_real_usage_after_compression = self._provider_omits_usage = False def _reset_session_compaction_state(self) -> None: """Shared per-session reset for /new, /reset and session end.""" @@ -2355,19 +2354,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): """Pair the real prompt count with its rough estimate and judge the armed compaction verdict.""" if self.last_prompt_tokens > 0: self.last_real_prompt_tokens = self.last_prompt_tokens + self._provider_omits_usage = False if self.last_prompt_tokens < self.threshold_tokens: - if self.awaiting_real_usage_after_compression and self.last_compression_rough_tokens > 0: - self.last_rough_tokens_when_real_prompt_fit = self.last_compression_rough_tokens - elif self._pending_request_rough_tokens > 0: - # Pair the real prompt count with the same request's rough estimate so the defer baseline syncs on - # EVERY fitting response, not only after compaction; otherwise a never-compressed session has no - # baseline and preflight fires on the raw rough estimate (overcounts CJK / replay blobs severalfold). - self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens # Any real reading below the trigger proves the prompt fits: clear the latch. The fallback streak survives. self._record_ineffective_compression_verdict(0) - else: - self.last_rough_tokens_when_real_prompt_fit = 0 - self._pending_request_rough_tokens = 0 # Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count, # not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it. # Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's @@ -2409,12 +2399,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return self.last_prompt_tokens = snapshot - def note_request_rough_estimate(self, rough_tokens: int) -> None: - """Record the rough estimate of the request about to be sent, for pairing with real usage.""" - try: - self._pending_request_rough_tokens = max(0, int(rough_tokens)) - except (TypeError, ValueError): - self._pending_request_rough_tokens = 0 + def note_usage_less_response(self) -> None: + """A completed response carried no usage: until a real reading arrives, this provider cannot + adjudicate context pressure, so rough estimates decide instead of waiting forever (#2153).""" + self._provider_omits_usage = True def note_native_compaction_checkpoint(self) -> None: """Wait for real usage before trusting a newly checkpointed request. @@ -2431,26 +2419,23 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self.last_compression_rough_tokens = 0 def should_defer_preflight_to_real_usage(self, rough_tokens: int) -> bool: - """Return True when a high rough preflight estimate is known-noisy. - Projects real usage as ``last_real + (rough_now - rough_at_last_real)`` and fires only when the - projection, not the raw estimate, crosses the threshold. Not a strict upper bound for - chars/4-underestimated scripts (Cyrillic, Thai, Arabic); bounded by two backstops: a real - reading at/over threshold clears the baseline, and the overflow handler compacts reactively. - Callers with a smaller (raw-messages) basis can only over-defer; the pre-API pressure check - re-runs with the aligned basis.""" + """True when a whole-context ROUGH estimate over threshold must wait ONE request for the + provider's real usage. Callers skip this for usage-anchored figures (real prompt count + + delta of what was appended since), which never defer. A rough figure defers right after a + local or native compaction (the last real reading is stale — the latch) and on any + transcript the anchor does not cover (first request, rewind/edit-resend, reloaded history): + the next response re-anchors it. It never defers once the provider has proven it omits + usage, or the estimate would be the only signal and compression could never fire (#2153); + the overflow handler compacts reactively in every case.""" if rough_tokens < self.threshold_tokens: return False - # After local or native compaction, last_real_prompt_tokens is STALE - # (above threshold); defer one turn until real usage arrives. if self.awaiting_real_usage_after_compression: return True - if self.last_real_prompt_tokens <= 0 or self.last_real_prompt_tokens >= self.threshold_tokens: + # A real reading already at/over threshold needs no second opinion, and a rough figure past + # the whole window describes a request certain to fail — sending it only buys an overflow error. + if self.last_real_prompt_tokens >= self.threshold_tokens or rough_tokens >= self.context_length: return False - baseline = self.last_rough_tokens_when_real_prompt_fit or self.last_compression_rough_tokens - if baseline <= 0: - return False - # No baseline ratchet here: advancing rough without a matching real reading would defer on stale data. - return self.last_real_prompt_tokens + max(0, rough_tokens - baseline) < self.threshold_tokens + return not self._provider_omits_usage def should_compress(self, prompt_tokens: int = None) -> bool: """True when compression should run now (anti-thrash included; see :meth:`should_compress_info` for the reason).""" diff --git a/agent/turn_context.py b/agent/turn_context.py index 7d85aaf226..87463a9983 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -39,6 +39,7 @@ def _preflight_request_tokens( """Token estimate for automatic preflight compression: a valid provider usage anchor, else the checkpoint-pruned native wire payload, else the generic estimator.""" anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + agent._request_pressure_anchored = anchored is not None if anchored is not None: return anchored tools = getattr(agent, "tools", None) or None diff --git a/agent/turn_context_compaction.py b/agent/turn_context_compaction.py index 171f602048..590946c6b9 100644 --- a/agent/turn_context_compaction.py +++ b/agent/turn_context_compaction.py @@ -258,7 +258,8 @@ def _preflight_compression( # snapshot may arm the interrupted-turn rollback. if isinstance(_snapshot_val, int) and not isinstance(_snapshot_val, bool): agent._turn_preflight_display_snapshot = _snapshot_val - _preflight_deferred = getattr( + # An anchored figure is real usage + delta: never deferred. + _preflight_deferred = not getattr(agent, "_request_pressure_anchored", False) and getattr( _compressor, "should_defer_preflight_to_real_usage", lambda _tokens: False )(_preflight_tokens) _codex_native_auto = _codex_native_auto_compaction(agent) @@ -278,8 +279,8 @@ def _preflight_compression( _compress_block_reason = None if _preflight_deferred: logger.info( - "Skipping preflight compression: rough estimate ~%s >= %s, " - "but last real provider prompt was %s after compression", + "Skipping preflight compression: rough estimate ~%s >= %s is not anchored on " + "real usage (last real provider prompt %s); deferring to the next response", f"{_preflight_tokens:,}", f"{_compressor.threshold_tokens:,}", f"{_compressor.last_real_prompt_tokens:,}", ) diff --git a/agent/turn_preflight.py b/agent/turn_preflight.py index 15d661e53a..557a02661a 100644 --- a/agent/turn_preflight.py +++ b/agent/turn_preflight.py @@ -263,32 +263,23 @@ def compress_after_tool_results( ) _compressor = agent.context_compressor - # Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens - # is the actual context size the provider reported plus the assistant turn — a tight lower bound for the - # next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves - # ample headroom; if tool results push past it, the next API call will report the real total and trigger - # compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage - # data), fall back to rough estimate to avoid missing compression. Without this, a session can grow - # unbounded after disconnects because should_compress(0) never fires. (#2153) - if _compressor.last_prompt_tokens > 0: + # Real usage decides: the anchor is the provider's last prompt count plus a rough delta for + # ONLY the tool results appended since (the raw last_prompt_tokens ignores them). Right after + # a compaction (-1 sentinel) there is no real count yet: never treat the schema-heavy rough + # figure as pressure. The whole-request rough estimate is the last resort (usage-less + # provider, post-disconnect, gateway restart), kept route-aware (#96995/#97602). + from agent.usage_anchor import anchored_context_tokens + + _anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + if _anchored is not None: + _real_tokens = _anchored + elif _compressor.last_prompt_tokens > 0: # Only prompt_tokens: thinking models inflate completion_tokens with - # reasoning that uses no context → premature compression. - # Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026) + # reasoning that uses no context → premature compression. (#12026) _real_tokens = _compressor.last_prompt_tokens elif _compressor.last_prompt_tokens == -1: - # Compression just ran, no API prompt count yet: don't treat a rough - # schema-heavy post-compression estimate as real context pressure. _real_tokens = 0 else: - # Include tool schemas (20-30K tokens the messages-only estimate misses) and - # stay route-aware: on a compacted native-Codex session the generic - # durable-history figure would false-trigger. - # Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate - # misses, which can skip compression past the configured threshold (#14695). Route-aware - # (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure - # overstates the wire and would false-trigger compression here exactly like the pre-API guard — this - # fallback runs precisely when no provider usage is available (post-disconnect / gateway restart), - # the unanchored case from #97602's repro. _real_tokens = _midturn_request_pressure_tokens( agent, messages, active_system_prompt or "", estimate_request_tokens_rough(messages, tools=agent.tools or None), diff --git a/agent/turn_preflight_gate.py b/agent/turn_preflight_gate.py index 9982cd1afd..b01224107c 100644 --- a/agent/turn_preflight_gate.py +++ b/agent/turn_preflight_gate.py @@ -90,8 +90,11 @@ def run_preflight_gate( return run_preflight_compression( agent, v, compressor=_compressor, request_pressure_tokens=request_pressure_tokens, provider_overflow_preflight=_provider_overflow_preflight, - defer_preflight=getattr( - _compressor, "should_defer_preflight_to_real_usage", lambda _t: False + # An anchored figure is real usage + delta: never deferred. Only a whole-context rough + # estimate waits for the provider's count. + defer_preflight=( + (lambda _t: False) if getattr(agent, "_request_pressure_anchored", False) + else getattr(_compressor, "should_defer_preflight_to_real_usage", lambda _t: False) ), moa_prepared_request=_moa_prepared_request, system_message=system_message, user_message=user_message, max_compression_attempts=max_compression_attempts, diff --git a/agent/turn_request_assembly.py b/agent/turn_request_assembly.py index f090d7e81d..c4d98ad211 100644 --- a/agent/turn_request_assembly.py +++ b/agent/turn_request_assembly.py @@ -245,6 +245,7 @@ def assemble_api_request( # Usage-anchored override: real prompt_tokens (incl. system + tool schemas) + # delta estimate replaces the whole-history heuristic when the anchor is fresh. _anchored_pressure = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + agent._request_pressure_anchored = _anchored_pressure is not None if _anchored_pressure is not None: request_pressure_tokens = _anchored_pressure else: @@ -254,11 +255,6 @@ def assemble_api_request( request_pressure_tokens = _pressure_with_real_floor( agent.context_compressor, request_pressure_tokens ) - # Stash the rough estimate so update_from_response() can pair it with the real - # count (should_defer_preflight_to_real_usage). getattr: test doubles lack it. - _note_rough = getattr(agent.context_compressor, "note_request_rough_estimate", None) - if callable(_note_rough): - _note_rough(request_pressure_tokens) return AssembledRequest( "fallthrough", api_messages, tools_for_api, _moa_prepared_request, pending_moa_prepared_request, approx_tokens, request_pressure_tokens, approx_tokens * 4, diff --git a/agent/turn_usage.py b/agent/turn_usage.py index 3c9e0e9494..a9eb006196 100644 --- a/agent/turn_usage.py +++ b/agent/turn_usage.py @@ -81,6 +81,9 @@ def record_response_usage( # pending verdict so later readings aren't charged to it and # preflight deferral isn't latched indefinitely. compressor.update_from_response({}) + _note_usage_less = getattr(compressor, "note_usage_less_response", None) + if callable(_note_usage_less): + _note_usage_less() logger.info( "API call #%d: model=%s provider=%s in=? out=? total=? latency=%.1fs usage=unavailable", agent.session_api_calls, agent.model, agent.provider or "unknown", api_duration, diff --git a/gateway/run_turn.py b/gateway/run_turn.py index 53684bf60e..ba8cf52793 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -617,10 +617,21 @@ class GatewayTurnMixin: _warn_token_threshold = int(_hyg_context_length * 0.95) _msg_count = len(history) - # Prefer the API-reported prompt tokens over the rough estimate (runs 30-50% high, which only - # fires hygiene early — safe). Do NOT compensate with a threshold multiplier. + # Real usage decides: the API-reported prompt count, else the anchor persisted on the session + # row (real count + delta of what was appended since, survives gateway restarts), else the + # rough estimate (runs 30-50% high, which only fires hygiene early — safe). Do NOT compensate + # with a threshold multiplier. + _anchored = None + if session_entry.last_prompt_tokens <= 0: + from agent.usage_anchor import persisted_anchor_tokens + _session_db = getattr(self, "_session_db", None) + _anchored = persisted_anchor_tokens( + getattr(_session_db, "_db", _session_db), session_entry.session_id, history, + ) if session_entry.last_prompt_tokens > 0: _approx_tokens, _token_source = session_entry.last_prompt_tokens, "actual" + elif _anchored is not None: + _approx_tokens, _token_source = _anchored, "anchored" else: _approx_tokens, _token_source = estimate_messages_tokens_rough(history), "estimated" diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index 470e7a9661..42ee1d7c91 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -253,7 +253,6 @@ class TestShouldCompress: class TestUpdateFromResponse: def test_updates_fields(self, compressor): compressor.awaiting_real_usage_after_compression = True - compressor.last_compression_rough_tokens = 90_000 compressor.update_from_response({ "prompt_tokens": 5000, "completion_tokens": 1000, @@ -262,97 +261,39 @@ class TestUpdateFromResponse: assert compressor.last_prompt_tokens == 5000 assert compressor.last_completion_tokens == 1000 assert compressor.last_real_prompt_tokens == 5000 - assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000 assert compressor.awaiting_real_usage_after_compression is False def test_missing_fields_default_zero(self, compressor): compressor.update_from_response({}) assert compressor.last_prompt_tokens == 0 - def test_pairs_noted_rough_estimate_with_fitting_real_usage(self, compressor): - """note_request_rough_estimate() + a fitting response must anchor the - defer baseline even when no compression ever ran — this is what gives - fresh sessions a (rough, real) pair before their first compaction.""" - compressor.note_request_rough_estimate(120_000) - compressor.update_from_response({"prompt_tokens": 60_000}) - - assert compressor.last_real_prompt_tokens == 60_000 - assert compressor.last_rough_tokens_when_real_prompt_fit == 120_000 - # Consumed: a later usage-bearing response without a fresh note keeps - # the previous pair instead of re-pairing against a stale estimate. - assert compressor._pending_request_rough_tokens == 0 - - def test_post_compression_pairing_wins_over_noted_estimate(self, compressor): - """Right after a compaction the post-compression rough count is the - authoritative baseline (#36718); a stale pre-compression note must not - displace it.""" - compressor.note_request_rough_estimate(120_000) - compressor.awaiting_real_usage_after_compression = True - compressor.last_compression_rough_tokens = 40_000 - compressor.update_from_response({"prompt_tokens": 30_000}) - - assert compressor.last_rough_tokens_when_real_prompt_fit == 40_000 - - def test_usage_less_response_preserves_pending_note(self, compressor): - """Transports that report usage separately send usage-less responses - first; the pending pair must survive until real usage arrives.""" - compressor.note_request_rough_estimate(120_000) - compressor.update_from_response({}) - - assert compressor._pending_request_rough_tokens == 120_000 - - def test_over_threshold_real_usage_clears_pending_note(self, compressor): - compressor.note_request_rough_estimate(120_000) - compressor.update_from_response({"prompt_tokens": 90_000}) - - assert compressor.last_rough_tokens_when_real_prompt_fit == 0 - assert compressor._pending_request_rough_tokens == 0 class TestPreflightDeferral: + """A whole-context rough estimate over threshold waits ONE request for real usage; callers + never consult this for usage-anchored figures.""" - def test_defers_while_projected_real_usage_fits(self, compressor): - """Large rough growth alone must not trigger compaction: with real - usage at 50K and 10K of rough growth since that reading, projected - real usage is 60K — far under the 85K threshold. The old fixed 5% - growth tolerance compacted here at ~59% of the real window (CJK / - replay-blob overcount churn).""" + def test_rough_estimate_defers_until_provider_prices_it(self, compressor): + """Real usage far under threshold, rough estimate far over (CJK / replay-blob overcount): + the provider's next reading decides, not the estimate.""" + compressor.context_length = 200_000 compressor.threshold_tokens = 85_000 compressor.last_real_prompt_tokens = 50_000 - compressor.last_rough_tokens_when_real_prompt_fit = 90_000 - assert compressor.should_defer_preflight_to_real_usage(100_000) is True + assert compressor.should_defer_preflight_to_real_usage(80_000) is False - def test_does_not_defer_when_projected_real_usage_crosses_threshold(self, compressor): - """Projection = last real + rough growth. 80K real + 6K growth = 86K - >= 85K threshold: compression must run.""" - compressor.threshold_tokens = 85_000 - compressor.last_real_prompt_tokens = 80_000 - compressor.last_rough_tokens_when_real_prompt_fit = 90_000 - - assert compressor.should_defer_preflight_to_real_usage(96_000) is False - - def test_does_not_defer_without_a_baseline(self, compressor): - """No synchronized (rough, real) pair yet — fall back to trusting the - rough estimate (conservative: compress).""" + def test_never_defers_when_real_usage_cannot_arrive_or_is_already_over(self, compressor): + """Deferral is one request, never a disable: a provider that omits usage, a real reading + already over threshold, or a rough figure past the whole window all compress now.""" + compressor.context_length = 100_000 compressor.threshold_tokens = 85_000 + compressor.last_real_prompt_tokens = 90_000 + assert compressor.should_defer_preflight_to_real_usage(95_000) is False compressor.last_real_prompt_tokens = 50_000 - compressor.last_rough_tokens_when_real_prompt_fit = 0 - compressor.last_compression_rough_tokens = 0 - - assert compressor.should_defer_preflight_to_real_usage(100_000) is False - - def test_defer_does_not_ratchet_baseline(self, compressor): - """The baseline is refreshed only by update_from_response() pairing. - Deferring must not advance it: without a fresh real reading, a - ratcheted baseline would shrink apparent growth and defer on stale - data.""" - compressor.threshold_tokens = 85_000 - compressor.last_real_prompt_tokens = 50_000 - compressor.last_rough_tokens_when_real_prompt_fit = 90_000 - - assert compressor.should_defer_preflight_to_real_usage(100_000) is True - assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000 - + assert compressor.should_defer_preflight_to_real_usage(150_000) is False + compressor.note_usage_less_response() + assert compressor.should_defer_preflight_to_real_usage(95_000) is False + compressor.update_from_response({"prompt_tokens": 50_000}) + assert compressor.should_defer_preflight_to_real_usage(95_000) is True def test_defers_immediately_after_compaction_with_stale_real_prompt(self, compressor): """#36718: right after a compaction, last_real_prompt_tokens still holds @@ -360,47 +301,24 @@ class TestPreflightDeferral: must force deferral so preflight doesn't fire a SECOND compaction before real post-compaction usage arrives.""" compressor.threshold_tokens = 85_000 - # Stale pre-compression value — would hit the `>= threshold => False` - # short-circuit and defeat deferral without the flag guard. compressor.last_real_prompt_tokens = 120_000 compressor.awaiting_real_usage_after_compression = True assert compressor.should_defer_preflight_to_real_usage(95_000) is True def test_native_checkpoint_defers_until_provider_usage_reanchors(self, compressor): - """A newly captured native checkpoint is opaque ciphertext, not text. - - Its serialized size can add more than a million rough tokens even when - the provider reports that the checkpoint-pruned request fits. Defer the - local compressor for one request so real usage can establish the new - rough/real calibration instead of immediately summarizing again. - """ + """A newly captured native checkpoint is opaque ciphertext whose serialized size can add + more than a million rough tokens; the local compressor waits one request for real usage.""" compressor.threshold_tokens = 85_000 compressor.last_real_prompt_tokens = 60_000 - compressor.last_rough_tokens_when_real_prompt_fit = 70_000 compressor.last_compression_rough_tokens = 40_000 compressor.note_native_compaction_checkpoint() - encrypted_checkpoint_rough = 1_300_000 assert compressor.awaiting_real_usage_after_compression is True assert compressor.last_compression_rough_tokens == 0 - assert ( - compressor.should_defer_preflight_to_real_usage( - encrypted_checkpoint_rough - ) - is True - ) - - compressor.note_request_rough_estimate(encrypted_checkpoint_rough) + assert compressor.should_defer_preflight_to_real_usage(1_300_000) is True compressor.update_from_response({"prompt_tokens": 65_000}) - assert compressor.awaiting_real_usage_after_compression is False - assert ( - compressor.last_rough_tokens_when_real_prompt_fit - == encrypted_checkpoint_rough - ) - - class TestCompress: @@ -2221,7 +2139,6 @@ class TestUpdateModelResetsCalibration: # Simulate a large-model session that proved a prompt fit. comp.last_prompt_tokens = 120_000 comp.last_real_prompt_tokens = 120_000 - comp.last_rough_tokens_when_real_prompt_fit = 130_000 comp.last_compression_rough_tokens = 130_000 comp.awaiting_real_usage_after_compression = True comp._ineffective_compression_count = 2 @@ -2230,7 +2147,6 @@ class TestUpdateModelResetsCalibration: assert comp.last_prompt_tokens == 0 assert comp.last_real_prompt_tokens == 0 - assert comp.last_rough_tokens_when_real_prompt_fit == 0 assert comp.last_compression_rough_tokens == 0 assert comp.awaiting_real_usage_after_compression is False assert comp._ineffective_compression_count == 0 @@ -2240,15 +2156,14 @@ class TestUpdateModelResetsCalibration: preflight on the new smaller model.""" comp = self._comp() comp.last_real_prompt_tokens = 50_000 - comp.last_rough_tokens_when_real_prompt_fit = 90_000 # Before switch, a modest rough growth would defer. comp.threshold_tokens = 85_000 assert comp.should_defer_preflight_to_real_usage(93_000) is True - # After switching to a 65K model, the stale state is gone, so a rough - # estimate over the new threshold is NOT deferred — preflight will run. + # After switching to a 65K model the stale reading is gone; a rough estimate past the + # NEW window is a request certain to fail and is never deferred — preflight will run. comp.update_model("small-model", context_length=65_536) - assert comp.should_defer_preflight_to_real_usage(comp.threshold_tokens + 5_000) is False + assert comp.should_defer_preflight_to_real_usage(comp.context_length + 5_000) is False diff --git a/tests/agent/test_context_compressor_cross_session_guard.py b/tests/agent/test_context_compressor_cross_session_guard.py index e92edb1618..4ede5323fb 100644 --- a/tests/agent/test_context_compressor_cross_session_guard.py +++ b/tests/agent/test_context_compressor_cross_session_guard.py @@ -62,7 +62,6 @@ def _make_compressor(): c._last_aux_model_failure_model = None c.last_real_prompt_tokens = 0 c.last_compression_rough_tokens = 0 - c.last_rough_tokens_when_real_prompt_fit = 0 c.awaiting_real_usage_after_compression = False return c diff --git a/tests/agent/test_context_compressor_session_end_clears_state.py b/tests/agent/test_context_compressor_session_end_clears_state.py index d7655c6b7f..b2c9a88733 100644 --- a/tests/agent/test_context_compressor_session_end_clears_state.py +++ b/tests/agent/test_context_compressor_session_end_clears_state.py @@ -67,7 +67,6 @@ def _make_compressor(): c._last_aux_model_failure_model = None c.last_real_prompt_tokens = 0 c.last_compression_rough_tokens = 0 - c.last_rough_tokens_when_real_prompt_fit = 0 c.awaiting_real_usage_after_compression = False c._previous_summary = None c._summary_has_user_turn = None @@ -91,7 +90,6 @@ def _simulate_cron_session_state(c): c._context_probe_persistable = True c.last_real_prompt_tokens = 50000 c.last_compression_rough_tokens = 60000 - c.last_rough_tokens_when_real_prompt_fit = 55000 c.awaiting_real_usage_after_compression = True diff --git a/tests/agent/test_turn_context_overflow_warning.py b/tests/agent/test_turn_context_overflow_warning.py index 72c4f0b107..b9034d4b80 100644 --- a/tests/agent/test_turn_context_overflow_warning.py +++ b/tests/agent/test_turn_context_overflow_warning.py @@ -45,7 +45,7 @@ class TestShouldCompressInfo: def test_cooldown_reports_reason(self): comp = _make_compressor() - comp.last_prompt_tokens = 73_000 + comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000 comp._summary_failure_cooldown_until = time.monotonic() + 60 should, reason = comp.should_compress_info(73_000) assert should is False @@ -58,7 +58,7 @@ class TestShouldCompressInfo: """should_compress() must still return a bare bool for existing callers in conversation_loop.py (and/or chains).""" comp = _make_compressor() - comp.last_prompt_tokens = 73_000 + comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000 comp._summary_failure_cooldown_until = time.monotonic() + 60 result = comp.should_compress(73_000) assert result is False @@ -118,7 +118,7 @@ def _run_build(agent): class TestTurnContextOverflowWarning: def test_warns_on_cooldown_block(self): comp = _make_compressor() - comp.last_prompt_tokens = 73_000 + comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000 comp._summary_failure_cooldown_until = time.monotonic() + 30 agent = _build_warn_agent(comp) _run_build(agent) @@ -139,7 +139,7 @@ class TestTurnContextOverflowWarning: cooldown timer moves. """ comp = _make_compressor() - comp.last_prompt_tokens = 73_000 + comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000 comp._summary_failure_cooldown_until = time.monotonic() + 30 agent = _build_warn_agent(comp) # Turn 1: over threshold + cooldown -> warn. diff --git a/tests/run_agent/test_413_compression.py b/tests/run_agent/test_413_compression.py index e11c2ddef0..7bcc97c05a 100644 --- a/tests/run_agent/test_413_compression.py +++ b/tests/run_agent/test_413_compression.py @@ -624,6 +624,7 @@ class TestPreflightCompression: not leak even though compaction itself still runs. """ agent.compression_enabled = True + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 130_000 agent.context_compressor.emit_automatic_compaction_status = False @@ -667,6 +668,7 @@ class TestPreflightCompression: def test_preflight_compresses_oversized_history(self, agent): """When loaded history exceeds the model's context threshold, compress before API call.""" agent.compression_enabled = True + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides # Set a small context so the history is "oversized", but large enough # that the compressed result (2 short messages) fits in a single pass. agent.context_compressor.context_length = 2000 @@ -719,6 +721,7 @@ class TestPreflightCompression: def test_preflight_suppresses_status_when_context_engine_opts_out(self, agent): """LCM-style engines can keep routine automatic preflight maintenance silent.""" agent.compression_enabled = True + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 100_000 agent.context_compressor.emit_automatic_compaction_status = False @@ -763,6 +766,7 @@ class TestPreflightCompression: def test_preflight_uses_context_engine_custom_status_message(self, agent): """Plugin engines can replace generic built-in-compressor wording.""" agent.compression_enabled = True + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 100_000 @@ -811,46 +815,27 @@ class TestPreflightCompression: assert not any("Preflight compression" in msg for msg in lifecycle_messages) - def test_preflight_compresses_when_projected_real_usage_crosses(self, agent): - """Projected real usage (last real + rough growth) crossing the - threshold still triggers preflight: 95K real + 12K rough growth = - 107K >= 100K. Growth alone no longer decides — real usage far below - the threshold defers instead (see TestPreflightDeferral).""" + def test_rough_over_threshold_waits_one_request_then_real_usage_compresses(self, agent): + """Real usage decides, the estimate only decides whether to wait for it: a whole-history + rough estimate over threshold with no anchor defers ONE request; the provider's real + prompt count then drives the next gate — over threshold compresses, under does not.""" agent.compression_enabled = True agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 100_000 - agent.context_compressor.last_prompt_tokens = 95_000 - agent.context_compressor.last_real_prompt_tokens = 95_000 - agent.context_compressor.last_rough_tokens_when_real_prompt_fit = 113_000 big_history = [] for i in range(20): big_history.append({"role": "user", "content": f"Message {i} padded"}) big_history.append({"role": "assistant", "content": f"Response {i} padded"}) - ok_resp = _mock_response( - content="Compressed after growth", - finish_reason="stop", - usage={"prompt_tokens": 50_000, "completion_tokens": 100, "total_tokens": 50_100}, - ) - agent.client.chat.completions.create.side_effect = [ok_resp] - - # First rough estimate must clear the threshold so preflight fires - # (rough growth since the last fitting request is large, so the - # deferral path is NOT taken). Every estimate after compaction is - # sub-threshold. Use a callable side_effect rather than a fixed list - # so we don't have to predict how many times the loop re-estimates — - # the post-response real-token estimate is an extra call that a - # 2-element list would exhaust (StopIteration). - _rough_calls = {"n": 0} - - def _rough_estimate(*_args, **_kwargs): - _rough_calls["n"] += 1 - return 125_000 if _rough_calls["n"] == 1 else 40_000 + agent.client.chat.completions.create.side_effect = [ + _mock_response(content="first", usage={"prompt_tokens": 105_000, "completion_tokens": 10, "total_tokens": 105_010}), + _mock_response(content="second", usage={"prompt_tokens": 50_000, "completion_tokens": 10, "total_tokens": 50_010}), + ] with ( - patch("agent.turn_context.estimate_request_tokens_rough", side_effect=_rough_estimate), - patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_rough_estimate), + patch("agent.turn_context.estimate_request_tokens_rough", return_value=125_000), + patch("agent.model_metadata.estimate_request_tokens_rough", return_value=125_000), patch.object(agent, "_compress_context") as mock_compress, patch.object(agent, "_persist_session"), patch.object(agent, "_save_trajectory"), @@ -860,10 +845,12 @@ class TestPreflightCompression: [{"role": "user", "content": f"{SUMMARY_PREFIX}\nPrevious conversation"}], "new system prompt", ) - result = agent.run_conversation("hello", conversation_history=big_history) + r1 = agent.run_conversation("hello", conversation_history=big_history) + assert mock_compress.call_count == 0, "estimate alone must not compress before real usage" + agent.run_conversation("again", conversation_history=r1["messages"]) - mock_compress.assert_called_once() - assert result["completed"] is True + assert mock_compress.call_count == 1 + assert mock_compress.call_args.kwargs["approx_tokens"] >= 105_000 def test_no_preflight_when_under_threshold(self, agent): """When history fits within context, no preflight compression needed.""" @@ -928,6 +915,7 @@ class TestPreflightCompression: ): """The proactive retry block must not consume provider-overflow recovery.""" agent.compression_enabled = True + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 130_000 @@ -1183,6 +1171,7 @@ class TestPreflightCompression: agent.context_compressor.context_length = 200_000 agent.context_compressor.threshold_tokens = 130_000 agent.context_compressor.last_prompt_tokens = 74_400 + agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides agent.context_compressor._ineffective_compression_count = 1 big_history = [] diff --git a/tests/run_agent/test_compression_budget_rearm.py b/tests/run_agent/test_compression_budget_rearm.py index d91eef72d1..84d6548cbe 100644 --- a/tests/run_agent/test_compression_budget_rearm.py +++ b/tests/run_agent/test_compression_budget_rearm.py @@ -155,7 +155,11 @@ def test_pre_api_compression_budget_rearms_only_after_pressure_clears( estimate_values = iter([200, 190, 200, 10]) _last_estimate = [10] - def _next_estimate(*_args, **_kwargs): + def _next_estimate(messages=None, *_args, **_kwargs): + # The scripted sequence prices the WHOLE history; a usage-anchored gate + # estimates only the few messages appended since the real reading. + if isinstance(messages, list) and len(messages) <= 4: + return 10 # The provider-recovery variant re-runs the pre-API preflight after # fallback activation (#84733), consuming an extra estimate reading. # Hold the final low-pressure value once the scripted sequence is