refactor(compression): every compaction gate asks real usage first; rough estimates only decide whether to wait

Two parallel "real usage" mechanisms fought each other: the usage anchor (real + delta) and the
compressor's rough/real projection (should_defer_preflight_to_real_usage with
last_rough_tokens_when_real_prompt_fit / _pending_request_rough_tokens / note_request_rough_estimate
baselines). The projection stored an anchored, real-scale figure as its "rough" baseline, so a
rewind that invalidated the anchor produced phantom growth and a spurious compaction (#103391).

Now there is one authority:

- Post-tool gate (turn_preflight.compress_after_tool_results): anchored figure first (the raw
  last_prompt_tokens ignored the tool results just appended), then real, then rough.
- Gateway hygiene (run_turn._hmwa_hygiene_plan): real session count, else the anchor persisted on
  the session row, else rough.
- Preflight / pre-API gates: an anchored figure is never deferred. A whole-context rough estimate
  over threshold waits ONE request for the provider's real count instead of compressing on a guess
  (first request, rewind/edit-resend, reloaded history without a persisted anchor).
- The wait is one request, never a disable: a provider that omits usage
  (note_usage_less_response, #2153 class), a real reading already over threshold, a rough figure
  past the whole window, and provider-proven overflow all compress immediately; the post-compaction
  latch (#36718 / #104192) is unchanged.
- Projection baselines and their bookkeeping deleted (-101 LOC in context_compressor); the fixtures
  that scripted whole-history estimates now state the fact they relied on (provider omits usage).

Fixes #103391 (closes #103397 by construction — the baseline it repaired no longer exists).
This commit is contained in:
Teknium
2026-09-06 12:41:13 -07:00
parent c0aaa238f6
commit 0f4587e336
16 changed files with 115 additions and 216 deletions
+1
View File
@@ -2136,6 +2136,7 @@ _USAGE_STATE: Dict[str, Any] = {
# snapshot; invalidated on compaction/session switch so stale anchors never suppress compression.
"_usage_anchor": None,
"_turn_base_usage_anchor": None,
"_request_pressure_anchored": False, # whether the last pressure figure came from the anchor
# Cumulative token usage for the session
"session_prompt_tokens": 0,
"session_completion_tokens": 0,
+2
View File
@@ -98,6 +98,8 @@ def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any]
if compressor is not None and getattr(compressor, "awaiting_real_usage_after_compression", False):
# No usage cannot adjudicate the pending compaction; unlatch preflight deferral.
compressor.update_from_response({})
if compressor is not None and callable(getattr(compressor, "note_usage_less_response", None)):
compressor.note_usage_less_response()
_queue_token_counts(agent, "Codex app-server api-call persistence failed (session=%s): %s",
counts=lambda: billing(billing_mode="subscription_included"))
return {}
+19 -34
View File
@@ -1818,10 +1818,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._reset_session_compaction_state()
def _reset_real_usage_pairing(self) -> None:
"""Forget the real-vs-rough token pairing used by should_defer_preflight_to_real_usage()."""
"""Forget the real-usage state read by real_usage_pending()."""
self.last_real_prompt_tokens = self.last_compression_rough_tokens = 0
self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens = 0
self.awaiting_real_usage_after_compression = False
self.awaiting_real_usage_after_compression = self._provider_omits_usage = False
def _reset_session_compaction_state(self) -> None:
"""Shared per-session reset for /new, /reset and session end."""
@@ -2355,19 +2354,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
"""Pair the real prompt count with its rough estimate and judge the armed compaction verdict."""
if self.last_prompt_tokens > 0:
self.last_real_prompt_tokens = self.last_prompt_tokens
self._provider_omits_usage = False
if self.last_prompt_tokens < self.threshold_tokens:
if self.awaiting_real_usage_after_compression and self.last_compression_rough_tokens > 0:
self.last_rough_tokens_when_real_prompt_fit = self.last_compression_rough_tokens
elif self._pending_request_rough_tokens > 0:
# Pair the real prompt count with the same request's rough estimate so the defer baseline syncs on
# EVERY fitting response, not only after compaction; otherwise a never-compressed session has no
# baseline and preflight fires on the raw rough estimate (overcounts CJK / replay blobs severalfold).
self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens
# Any real reading below the trigger proves the prompt fits: clear the latch. The fallback streak survives.
self._record_ineffective_compression_verdict(0)
else:
self.last_rough_tokens_when_real_prompt_fit = 0
self._pending_request_rough_tokens = 0
# Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count,
# not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it.
# Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's
@@ -2409,12 +2399,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
return
self.last_prompt_tokens = snapshot
def note_request_rough_estimate(self, rough_tokens: int) -> None:
"""Record the rough estimate of the request about to be sent, for pairing with real usage."""
try:
self._pending_request_rough_tokens = max(0, int(rough_tokens))
except (TypeError, ValueError):
self._pending_request_rough_tokens = 0
def note_usage_less_response(self) -> None:
"""A completed response carried no usage: until a real reading arrives, this provider cannot
adjudicate context pressure, so rough estimates decide instead of waiting forever (#2153)."""
self._provider_omits_usage = True
def note_native_compaction_checkpoint(self) -> None:
"""Wait for real usage before trusting a newly checkpointed request.
@@ -2431,26 +2419,23 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self.last_compression_rough_tokens = 0
def should_defer_preflight_to_real_usage(self, rough_tokens: int) -> bool:
"""Return True when a high rough preflight estimate is known-noisy.
Projects real usage as ``last_real + (rough_now - rough_at_last_real)`` and fires only when the
projection, not the raw estimate, crosses the threshold. Not a strict upper bound for
chars/4-underestimated scripts (Cyrillic, Thai, Arabic); bounded by two backstops: a real
reading at/over threshold clears the baseline, and the overflow handler compacts reactively.
Callers with a smaller (raw-messages) basis can only over-defer; the pre-API pressure check
re-runs with the aligned basis."""
"""True when a whole-context ROUGH estimate over threshold must wait ONE request for the
provider's real usage. Callers skip this for usage-anchored figures (real prompt count +
delta of what was appended since), which never defer. A rough figure defers right after a
local or native compaction (the last real reading is stale — the latch) and on any
transcript the anchor does not cover (first request, rewind/edit-resend, reloaded history):
the next response re-anchors it. It never defers once the provider has proven it omits
usage, or the estimate would be the only signal and compression could never fire (#2153);
the overflow handler compacts reactively in every case."""
if rough_tokens < self.threshold_tokens:
return False
# After local or native compaction, last_real_prompt_tokens is STALE
# (above threshold); defer one turn until real usage arrives.
if self.awaiting_real_usage_after_compression:
return True
if self.last_real_prompt_tokens <= 0 or self.last_real_prompt_tokens >= self.threshold_tokens:
# A real reading already at/over threshold needs no second opinion, and a rough figure past
# the whole window describes a request certain to fail — sending it only buys an overflow error.
if self.last_real_prompt_tokens >= self.threshold_tokens or rough_tokens >= self.context_length:
return False
baseline = self.last_rough_tokens_when_real_prompt_fit or self.last_compression_rough_tokens
if baseline <= 0:
return False
# No baseline ratchet here: advancing rough without a matching real reading would defer on stale data.
return self.last_real_prompt_tokens + max(0, rough_tokens - baseline) < self.threshold_tokens
return not self._provider_omits_usage
def should_compress(self, prompt_tokens: int = None) -> bool:
"""True when compression should run now (anti-thrash included; see :meth:`should_compress_info` for the reason)."""
+1
View File
@@ -39,6 +39,7 @@ def _preflight_request_tokens(
"""Token estimate for automatic preflight compression: a valid provider usage anchor,
else the checkpoint-pruned native wire payload, else the generic estimator."""
anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
agent._request_pressure_anchored = anchored is not None
if anchored is not None:
return anchored
tools = getattr(agent, "tools", None) or None
+4 -3
View File
@@ -258,7 +258,8 @@ def _preflight_compression(
# snapshot may arm the interrupted-turn rollback.
if isinstance(_snapshot_val, int) and not isinstance(_snapshot_val, bool):
agent._turn_preflight_display_snapshot = _snapshot_val
_preflight_deferred = getattr(
# An anchored figure is real usage + delta: never deferred.
_preflight_deferred = not getattr(agent, "_request_pressure_anchored", False) and getattr(
_compressor, "should_defer_preflight_to_real_usage", lambda _tokens: False
)(_preflight_tokens)
_codex_native_auto = _codex_native_auto_compaction(agent)
@@ -278,8 +279,8 @@ def _preflight_compression(
_compress_block_reason = None
if _preflight_deferred:
logger.info(
"Skipping preflight compression: rough estimate ~%s >= %s, "
"but last real provider prompt was %s after compression",
"Skipping preflight compression: rough estimate ~%s >= %s is not anchored on "
"real usage (last real provider prompt %s); deferring to the next response",
f"{_preflight_tokens:,}", f"{_compressor.threshold_tokens:,}",
f"{_compressor.last_real_prompt_tokens:,}",
)
+12 -21
View File
@@ -263,32 +263,23 @@ def compress_after_tool_results(
)
_compressor = agent.context_compressor
# Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens
# is the actual context size the provider reported plus the assistant turn — a tight lower bound for the
# next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves
# ample headroom; if tool results push past it, the next API call will report the real total and trigger
# compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage
# data), fall back to rough estimate to avoid missing compression. Without this, a session can grow
# unbounded after disconnects because should_compress(0) never fires. (#2153)
if _compressor.last_prompt_tokens > 0:
# Real usage decides: the anchor is the provider's last prompt count plus a rough delta for
# ONLY the tool results appended since (the raw last_prompt_tokens ignores them). Right after
# a compaction (-1 sentinel) there is no real count yet: never treat the schema-heavy rough
# figure as pressure. The whole-request rough estimate is the last resort (usage-less
# provider, post-disconnect, gateway restart), kept route-aware (#96995/#97602).
from agent.usage_anchor import anchored_context_tokens
_anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
if _anchored is not None:
_real_tokens = _anchored
elif _compressor.last_prompt_tokens > 0:
# Only prompt_tokens: thinking models inflate completion_tokens with
# reasoning that uses no context → premature compression.
# Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026)
# reasoning that uses no context → premature compression. (#12026)
_real_tokens = _compressor.last_prompt_tokens
elif _compressor.last_prompt_tokens == -1:
# Compression just ran, no API prompt count yet: don't treat a rough
# schema-heavy post-compression estimate as real context pressure.
_real_tokens = 0
else:
# Include tool schemas (20-30K tokens the messages-only estimate misses) and
# stay route-aware: on a compacted native-Codex session the generic
# durable-history figure would false-trigger.
# Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate
# misses, which can skip compression past the configured threshold (#14695). Route-aware
# (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure
# overstates the wire and would false-trigger compression here exactly like the pre-API guard — this
# fallback runs precisely when no provider usage is available (post-disconnect / gateway restart),
# the unanchored case from #97602's repro.
_real_tokens = _midturn_request_pressure_tokens(
agent, messages, active_system_prompt or "",
estimate_request_tokens_rough(messages, tools=agent.tools or None),
+5 -2
View File
@@ -90,8 +90,11 @@ def run_preflight_gate(
return run_preflight_compression(
agent, v, compressor=_compressor, request_pressure_tokens=request_pressure_tokens,
provider_overflow_preflight=_provider_overflow_preflight,
defer_preflight=getattr(
_compressor, "should_defer_preflight_to_real_usage", lambda _t: False
# An anchored figure is real usage + delta: never deferred. Only a whole-context rough
# estimate waits for the provider's count.
defer_preflight=(
(lambda _t: False) if getattr(agent, "_request_pressure_anchored", False)
else getattr(_compressor, "should_defer_preflight_to_real_usage", lambda _t: False)
),
moa_prepared_request=_moa_prepared_request, system_message=system_message,
user_message=user_message, max_compression_attempts=max_compression_attempts,
+1 -5
View File
@@ -245,6 +245,7 @@ def assemble_api_request(
# Usage-anchored override: real prompt_tokens (incl. system + tool schemas) +
# delta estimate replaces the whole-history heuristic when the anchor is fresh.
_anchored_pressure = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
agent._request_pressure_anchored = _anchored_pressure is not None
if _anchored_pressure is not None:
request_pressure_tokens = _anchored_pressure
else:
@@ -254,11 +255,6 @@ def assemble_api_request(
request_pressure_tokens = _pressure_with_real_floor(
agent.context_compressor, request_pressure_tokens
)
# Stash the rough estimate so update_from_response() can pair it with the real
# count (should_defer_preflight_to_real_usage). getattr: test doubles lack it.
_note_rough = getattr(agent.context_compressor, "note_request_rough_estimate", None)
if callable(_note_rough):
_note_rough(request_pressure_tokens)
return AssembledRequest(
"fallthrough", api_messages, tools_for_api, _moa_prepared_request,
pending_moa_prepared_request, approx_tokens, request_pressure_tokens, approx_tokens * 4,
+3
View File
@@ -81,6 +81,9 @@ def record_response_usage(
# pending verdict so later readings aren't charged to it and
# preflight deferral isn't latched indefinitely.
compressor.update_from_response({})
_note_usage_less = getattr(compressor, "note_usage_less_response", None)
if callable(_note_usage_less):
_note_usage_less()
logger.info(
"API call #%d: model=%s provider=%s in=? out=? total=? latency=%.1fs usage=unavailable",
agent.session_api_calls, agent.model, agent.provider or "unknown", api_duration,
+13 -2
View File
@@ -617,10 +617,21 @@ class GatewayTurnMixin:
_warn_token_threshold = int(_hyg_context_length * 0.95)
_msg_count = len(history)
# Prefer the API-reported prompt tokens over the rough estimate (runs 30-50% high, which only
# fires hygiene early — safe). Do NOT compensate with a threshold multiplier.
# Real usage decides: the API-reported prompt count, else the anchor persisted on the session
# row (real count + delta of what was appended since, survives gateway restarts), else the
# rough estimate (runs 30-50% high, which only fires hygiene early — safe). Do NOT compensate
# with a threshold multiplier.
_anchored = None
if session_entry.last_prompt_tokens <= 0:
from agent.usage_anchor import persisted_anchor_tokens
_session_db = getattr(self, "_session_db", None)
_anchored = persisted_anchor_tokens(
getattr(_session_db, "_db", _session_db), session_entry.session_id, history,
)
if session_entry.last_prompt_tokens > 0:
_approx_tokens, _token_source = session_entry.last_prompt_tokens, "actual"
elif _anchored is not None:
_approx_tokens, _token_source = _anchored, "anchored"
else:
_approx_tokens, _token_source = estimate_messages_tokens_rough(history), "estimated"
+24 -109
View File
@@ -253,7 +253,6 @@ class TestShouldCompress:
class TestUpdateFromResponse:
def test_updates_fields(self, compressor):
compressor.awaiting_real_usage_after_compression = True
compressor.last_compression_rough_tokens = 90_000
compressor.update_from_response({
"prompt_tokens": 5000,
"completion_tokens": 1000,
@@ -262,97 +261,39 @@ class TestUpdateFromResponse:
assert compressor.last_prompt_tokens == 5000
assert compressor.last_completion_tokens == 1000
assert compressor.last_real_prompt_tokens == 5000
assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000
assert compressor.awaiting_real_usage_after_compression is False
def test_missing_fields_default_zero(self, compressor):
compressor.update_from_response({})
assert compressor.last_prompt_tokens == 0
def test_pairs_noted_rough_estimate_with_fitting_real_usage(self, compressor):
"""note_request_rough_estimate() + a fitting response must anchor the
defer baseline even when no compression ever ran — this is what gives
fresh sessions a (rough, real) pair before their first compaction."""
compressor.note_request_rough_estimate(120_000)
compressor.update_from_response({"prompt_tokens": 60_000})
assert compressor.last_real_prompt_tokens == 60_000
assert compressor.last_rough_tokens_when_real_prompt_fit == 120_000
# Consumed: a later usage-bearing response without a fresh note keeps
# the previous pair instead of re-pairing against a stale estimate.
assert compressor._pending_request_rough_tokens == 0
def test_post_compression_pairing_wins_over_noted_estimate(self, compressor):
"""Right after a compaction the post-compression rough count is the
authoritative baseline (#36718); a stale pre-compression note must not
displace it."""
compressor.note_request_rough_estimate(120_000)
compressor.awaiting_real_usage_after_compression = True
compressor.last_compression_rough_tokens = 40_000
compressor.update_from_response({"prompt_tokens": 30_000})
assert compressor.last_rough_tokens_when_real_prompt_fit == 40_000
def test_usage_less_response_preserves_pending_note(self, compressor):
"""Transports that report usage separately send usage-less responses
first; the pending pair must survive until real usage arrives."""
compressor.note_request_rough_estimate(120_000)
compressor.update_from_response({})
assert compressor._pending_request_rough_tokens == 120_000
def test_over_threshold_real_usage_clears_pending_note(self, compressor):
compressor.note_request_rough_estimate(120_000)
compressor.update_from_response({"prompt_tokens": 90_000})
assert compressor.last_rough_tokens_when_real_prompt_fit == 0
assert compressor._pending_request_rough_tokens == 0
class TestPreflightDeferral:
"""A whole-context rough estimate over threshold waits ONE request for real usage; callers
never consult this for usage-anchored figures."""
def test_defers_while_projected_real_usage_fits(self, compressor):
"""Large rough growth alone must not trigger compaction: with real
usage at 50K and 10K of rough growth since that reading, projected
real usage is 60K — far under the 85K threshold. The old fixed 5%
growth tolerance compacted here at ~59% of the real window (CJK /
replay-blob overcount churn)."""
def test_rough_estimate_defers_until_provider_prices_it(self, compressor):
"""Real usage far under threshold, rough estimate far over (CJK / replay-blob overcount):
the provider's next reading decides, not the estimate."""
compressor.context_length = 200_000
compressor.threshold_tokens = 85_000
compressor.last_real_prompt_tokens = 50_000
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
assert compressor.should_defer_preflight_to_real_usage(100_000) is True
assert compressor.should_defer_preflight_to_real_usage(80_000) is False
def test_does_not_defer_when_projected_real_usage_crosses_threshold(self, compressor):
"""Projection = last real + rough growth. 80K real + 6K growth = 86K
>= 85K threshold: compression must run."""
compressor.threshold_tokens = 85_000
compressor.last_real_prompt_tokens = 80_000
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
assert compressor.should_defer_preflight_to_real_usage(96_000) is False
def test_does_not_defer_without_a_baseline(self, compressor):
"""No synchronized (rough, real) pair yet — fall back to trusting the
rough estimate (conservative: compress)."""
def test_never_defers_when_real_usage_cannot_arrive_or_is_already_over(self, compressor):
"""Deferral is one request, never a disable: a provider that omits usage, a real reading
already over threshold, or a rough figure past the whole window all compress now."""
compressor.context_length = 100_000
compressor.threshold_tokens = 85_000
compressor.last_real_prompt_tokens = 90_000
assert compressor.should_defer_preflight_to_real_usage(95_000) is False
compressor.last_real_prompt_tokens = 50_000
compressor.last_rough_tokens_when_real_prompt_fit = 0
compressor.last_compression_rough_tokens = 0
assert compressor.should_defer_preflight_to_real_usage(100_000) is False
def test_defer_does_not_ratchet_baseline(self, compressor):
"""The baseline is refreshed only by update_from_response() pairing.
Deferring must not advance it: without a fresh real reading, a
ratcheted baseline would shrink apparent growth and defer on stale
data."""
compressor.threshold_tokens = 85_000
compressor.last_real_prompt_tokens = 50_000
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
assert compressor.should_defer_preflight_to_real_usage(100_000) is True
assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000
assert compressor.should_defer_preflight_to_real_usage(150_000) is False
compressor.note_usage_less_response()
assert compressor.should_defer_preflight_to_real_usage(95_000) is False
compressor.update_from_response({"prompt_tokens": 50_000})
assert compressor.should_defer_preflight_to_real_usage(95_000) is True
def test_defers_immediately_after_compaction_with_stale_real_prompt(self, compressor):
"""#36718: right after a compaction, last_real_prompt_tokens still holds
@@ -360,47 +301,24 @@ class TestPreflightDeferral:
must force deferral so preflight doesn't fire a SECOND compaction before
real post-compaction usage arrives."""
compressor.threshold_tokens = 85_000
# Stale pre-compression value — would hit the `>= threshold => False`
# short-circuit and defeat deferral without the flag guard.
compressor.last_real_prompt_tokens = 120_000
compressor.awaiting_real_usage_after_compression = True
assert compressor.should_defer_preflight_to_real_usage(95_000) is True
def test_native_checkpoint_defers_until_provider_usage_reanchors(self, compressor):
"""A newly captured native checkpoint is opaque ciphertext, not text.
Its serialized size can add more than a million rough tokens even when
the provider reports that the checkpoint-pruned request fits. Defer the
local compressor for one request so real usage can establish the new
rough/real calibration instead of immediately summarizing again.
"""
"""A newly captured native checkpoint is opaque ciphertext whose serialized size can add
more than a million rough tokens; the local compressor waits one request for real usage."""
compressor.threshold_tokens = 85_000
compressor.last_real_prompt_tokens = 60_000
compressor.last_rough_tokens_when_real_prompt_fit = 70_000
compressor.last_compression_rough_tokens = 40_000
compressor.note_native_compaction_checkpoint()
encrypted_checkpoint_rough = 1_300_000
assert compressor.awaiting_real_usage_after_compression is True
assert compressor.last_compression_rough_tokens == 0
assert (
compressor.should_defer_preflight_to_real_usage(
encrypted_checkpoint_rough
)
is True
)
compressor.note_request_rough_estimate(encrypted_checkpoint_rough)
assert compressor.should_defer_preflight_to_real_usage(1_300_000) is True
compressor.update_from_response({"prompt_tokens": 65_000})
assert compressor.awaiting_real_usage_after_compression is False
assert (
compressor.last_rough_tokens_when_real_prompt_fit
== encrypted_checkpoint_rough
)
class TestCompress:
@@ -2221,7 +2139,6 @@ class TestUpdateModelResetsCalibration:
# Simulate a large-model session that proved a prompt fit.
comp.last_prompt_tokens = 120_000
comp.last_real_prompt_tokens = 120_000
comp.last_rough_tokens_when_real_prompt_fit = 130_000
comp.last_compression_rough_tokens = 130_000
comp.awaiting_real_usage_after_compression = True
comp._ineffective_compression_count = 2
@@ -2230,7 +2147,6 @@ class TestUpdateModelResetsCalibration:
assert comp.last_prompt_tokens == 0
assert comp.last_real_prompt_tokens == 0
assert comp.last_rough_tokens_when_real_prompt_fit == 0
assert comp.last_compression_rough_tokens == 0
assert comp.awaiting_real_usage_after_compression is False
assert comp._ineffective_compression_count == 0
@@ -2240,15 +2156,14 @@ class TestUpdateModelResetsCalibration:
preflight on the new smaller model."""
comp = self._comp()
comp.last_real_prompt_tokens = 50_000
comp.last_rough_tokens_when_real_prompt_fit = 90_000
# Before switch, a modest rough growth would defer.
comp.threshold_tokens = 85_000
assert comp.should_defer_preflight_to_real_usage(93_000) is True
# After switching to a 65K model, the stale state is gone, so a rough
# estimate over the new threshold is NOT deferred — preflight will run.
# After switching to a 65K model the stale reading is gone; a rough estimate past the
# NEW window is a request certain to fail and is never deferred — preflight will run.
comp.update_model("small-model", context_length=65_536)
assert comp.should_defer_preflight_to_real_usage(comp.threshold_tokens + 5_000) is False
assert comp.should_defer_preflight_to_real_usage(comp.context_length + 5_000) is False
@@ -62,7 +62,6 @@ def _make_compressor():
c._last_aux_model_failure_model = None
c.last_real_prompt_tokens = 0
c.last_compression_rough_tokens = 0
c.last_rough_tokens_when_real_prompt_fit = 0
c.awaiting_real_usage_after_compression = False
return c
@@ -67,7 +67,6 @@ def _make_compressor():
c._last_aux_model_failure_model = None
c.last_real_prompt_tokens = 0
c.last_compression_rough_tokens = 0
c.last_rough_tokens_when_real_prompt_fit = 0
c.awaiting_real_usage_after_compression = False
c._previous_summary = None
c._summary_has_user_turn = None
@@ -91,7 +90,6 @@ def _simulate_cron_session_state(c):
c._context_probe_persistable = True
c.last_real_prompt_tokens = 50000
c.last_compression_rough_tokens = 60000
c.last_rough_tokens_when_real_prompt_fit = 55000
c.awaiting_real_usage_after_compression = True
@@ -45,7 +45,7 @@ class TestShouldCompressInfo:
def test_cooldown_reports_reason(self):
comp = _make_compressor()
comp.last_prompt_tokens = 73_000
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
comp._summary_failure_cooldown_until = time.monotonic() + 60
should, reason = comp.should_compress_info(73_000)
assert should is False
@@ -58,7 +58,7 @@ class TestShouldCompressInfo:
"""should_compress() must still return a bare bool for existing
callers in conversation_loop.py (and/or chains)."""
comp = _make_compressor()
comp.last_prompt_tokens = 73_000
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
comp._summary_failure_cooldown_until = time.monotonic() + 60
result = comp.should_compress(73_000)
assert result is False
@@ -118,7 +118,7 @@ def _run_build(agent):
class TestTurnContextOverflowWarning:
def test_warns_on_cooldown_block(self):
comp = _make_compressor()
comp.last_prompt_tokens = 73_000
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
comp._summary_failure_cooldown_until = time.monotonic() + 30
agent = _build_warn_agent(comp)
_run_build(agent)
@@ -139,7 +139,7 @@ class TestTurnContextOverflowWarning:
cooldown timer moves.
"""
comp = _make_compressor()
comp.last_prompt_tokens = 73_000
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
comp._summary_failure_cooldown_until = time.monotonic() + 30
agent = _build_warn_agent(comp)
# Turn 1: over threshold + cooldown -> warn.
+21 -32
View File
@@ -624,6 +624,7 @@ class TestPreflightCompression:
not leak even though compaction itself still runs.
"""
agent.compression_enabled = True
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 130_000
agent.context_compressor.emit_automatic_compaction_status = False
@@ -667,6 +668,7 @@ class TestPreflightCompression:
def test_preflight_compresses_oversized_history(self, agent):
"""When loaded history exceeds the model's context threshold, compress before API call."""
agent.compression_enabled = True
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
# Set a small context so the history is "oversized", but large enough
# that the compressed result (2 short messages) fits in a single pass.
agent.context_compressor.context_length = 2000
@@ -719,6 +721,7 @@ class TestPreflightCompression:
def test_preflight_suppresses_status_when_context_engine_opts_out(self, agent):
"""LCM-style engines can keep routine automatic preflight maintenance silent."""
agent.compression_enabled = True
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 100_000
agent.context_compressor.emit_automatic_compaction_status = False
@@ -763,6 +766,7 @@ class TestPreflightCompression:
def test_preflight_uses_context_engine_custom_status_message(self, agent):
"""Plugin engines can replace generic built-in-compressor wording."""
agent.compression_enabled = True
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 100_000
@@ -811,46 +815,27 @@ class TestPreflightCompression:
assert not any("Preflight compression" in msg for msg in lifecycle_messages)
def test_preflight_compresses_when_projected_real_usage_crosses(self, agent):
"""Projected real usage (last real + rough growth) crossing the
threshold still triggers preflight: 95K real + 12K rough growth =
107K >= 100K. Growth alone no longer decides — real usage far below
the threshold defers instead (see TestPreflightDeferral)."""
def test_rough_over_threshold_waits_one_request_then_real_usage_compresses(self, agent):
"""Real usage decides, the estimate only decides whether to wait for it: a whole-history
rough estimate over threshold with no anchor defers ONE request; the provider's real
prompt count then drives the next gate — over threshold compresses, under does not."""
agent.compression_enabled = True
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 100_000
agent.context_compressor.last_prompt_tokens = 95_000
agent.context_compressor.last_real_prompt_tokens = 95_000
agent.context_compressor.last_rough_tokens_when_real_prompt_fit = 113_000
big_history = []
for i in range(20):
big_history.append({"role": "user", "content": f"Message {i} padded"})
big_history.append({"role": "assistant", "content": f"Response {i} padded"})
ok_resp = _mock_response(
content="Compressed after growth",
finish_reason="stop",
usage={"prompt_tokens": 50_000, "completion_tokens": 100, "total_tokens": 50_100},
)
agent.client.chat.completions.create.side_effect = [ok_resp]
# First rough estimate must clear the threshold so preflight fires
# (rough growth since the last fitting request is large, so the
# deferral path is NOT taken). Every estimate after compaction is
# sub-threshold. Use a callable side_effect rather than a fixed list
# so we don't have to predict how many times the loop re-estimates —
# the post-response real-token estimate is an extra call that a
# 2-element list would exhaust (StopIteration).
_rough_calls = {"n": 0}
def _rough_estimate(*_args, **_kwargs):
_rough_calls["n"] += 1
return 125_000 if _rough_calls["n"] == 1 else 40_000
agent.client.chat.completions.create.side_effect = [
_mock_response(content="first", usage={"prompt_tokens": 105_000, "completion_tokens": 10, "total_tokens": 105_010}),
_mock_response(content="second", usage={"prompt_tokens": 50_000, "completion_tokens": 10, "total_tokens": 50_010}),
]
with (
patch("agent.turn_context.estimate_request_tokens_rough", side_effect=_rough_estimate),
patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_rough_estimate),
patch("agent.turn_context.estimate_request_tokens_rough", return_value=125_000),
patch("agent.model_metadata.estimate_request_tokens_rough", return_value=125_000),
patch.object(agent, "_compress_context") as mock_compress,
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
@@ -860,10 +845,12 @@ class TestPreflightCompression:
[{"role": "user", "content": f"{SUMMARY_PREFIX}\nPrevious conversation"}],
"new system prompt",
)
result = agent.run_conversation("hello", conversation_history=big_history)
r1 = agent.run_conversation("hello", conversation_history=big_history)
assert mock_compress.call_count == 0, "estimate alone must not compress before real usage"
agent.run_conversation("again", conversation_history=r1["messages"])
mock_compress.assert_called_once()
assert result["completed"] is True
assert mock_compress.call_count == 1
assert mock_compress.call_args.kwargs["approx_tokens"] >= 105_000
def test_no_preflight_when_under_threshold(self, agent):
"""When history fits within context, no preflight compression needed."""
@@ -928,6 +915,7 @@ class TestPreflightCompression:
):
"""The proactive retry block must not consume provider-overflow recovery."""
agent.compression_enabled = True
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 130_000
@@ -1183,6 +1171,7 @@ class TestPreflightCompression:
agent.context_compressor.context_length = 200_000
agent.context_compressor.threshold_tokens = 130_000
agent.context_compressor.last_prompt_tokens = 74_400
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
agent.context_compressor._ineffective_compression_count = 1
big_history = []
@@ -155,7 +155,11 @@ def test_pre_api_compression_budget_rearms_only_after_pressure_clears(
estimate_values = iter([200, 190, 200, 10])
_last_estimate = [10]
def _next_estimate(*_args, **_kwargs):
def _next_estimate(messages=None, *_args, **_kwargs):
# The scripted sequence prices the WHOLE history; a usage-anchored gate
# estimates only the few messages appended since the real reading.
if isinstance(messages, list) and len(messages) <= 4:
return 10
# The provider-recovery variant re-runs the pre-API preflight after
# fallback activation (#84733), consuming an extra estimate reading.
# Hold the final low-pressure value once the scripted sequence is