refactor(compression): every compaction gate asks real usage first; rough estimates only decide whether to wait
Two parallel "real usage" mechanisms fought each other: the usage anchor (real + delta) and the compressor's rough/real projection (should_defer_preflight_to_real_usage with last_rough_tokens_when_real_prompt_fit / _pending_request_rough_tokens / note_request_rough_estimate baselines). The projection stored an anchored, real-scale figure as its "rough" baseline, so a rewind that invalidated the anchor produced phantom growth and a spurious compaction (#103391). Now there is one authority: - Post-tool gate (turn_preflight.compress_after_tool_results): anchored figure first (the raw last_prompt_tokens ignored the tool results just appended), then real, then rough. - Gateway hygiene (run_turn._hmwa_hygiene_plan): real session count, else the anchor persisted on the session row, else rough. - Preflight / pre-API gates: an anchored figure is never deferred. A whole-context rough estimate over threshold waits ONE request for the provider's real count instead of compressing on a guess (first request, rewind/edit-resend, reloaded history without a persisted anchor). - The wait is one request, never a disable: a provider that omits usage (note_usage_less_response, #2153 class), a real reading already over threshold, a rough figure past the whole window, and provider-proven overflow all compress immediately; the post-compaction latch (#36718 / #104192) is unchanged. - Projection baselines and their bookkeeping deleted (-101 LOC in context_compressor); the fixtures that scripted whole-history estimates now state the fact they relied on (provider omits usage). Fixes #103391 (closes #103397 by construction — the baseline it repaired no longer exists).
This commit is contained in:
@@ -2136,6 +2136,7 @@ _USAGE_STATE: Dict[str, Any] = {
|
||||
# snapshot; invalidated on compaction/session switch so stale anchors never suppress compression.
|
||||
"_usage_anchor": None,
|
||||
"_turn_base_usage_anchor": None,
|
||||
"_request_pressure_anchored": False, # whether the last pressure figure came from the anchor
|
||||
# Cumulative token usage for the session
|
||||
"session_prompt_tokens": 0,
|
||||
"session_completion_tokens": 0,
|
||||
|
||||
@@ -98,6 +98,8 @@ def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any]
|
||||
if compressor is not None and getattr(compressor, "awaiting_real_usage_after_compression", False):
|
||||
# No usage cannot adjudicate the pending compaction; unlatch preflight deferral.
|
||||
compressor.update_from_response({})
|
||||
if compressor is not None and callable(getattr(compressor, "note_usage_less_response", None)):
|
||||
compressor.note_usage_less_response()
|
||||
_queue_token_counts(agent, "Codex app-server api-call persistence failed (session=%s): %s",
|
||||
counts=lambda: billing(billing_mode="subscription_included"))
|
||||
return {}
|
||||
|
||||
+19
-34
@@ -1818,10 +1818,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._reset_session_compaction_state()
|
||||
|
||||
def _reset_real_usage_pairing(self) -> None:
|
||||
"""Forget the real-vs-rough token pairing used by should_defer_preflight_to_real_usage()."""
|
||||
"""Forget the real-usage state read by real_usage_pending()."""
|
||||
self.last_real_prompt_tokens = self.last_compression_rough_tokens = 0
|
||||
self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens = 0
|
||||
self.awaiting_real_usage_after_compression = False
|
||||
self.awaiting_real_usage_after_compression = self._provider_omits_usage = False
|
||||
|
||||
def _reset_session_compaction_state(self) -> None:
|
||||
"""Shared per-session reset for /new, /reset and session end."""
|
||||
@@ -2355,19 +2354,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
"""Pair the real prompt count with its rough estimate and judge the armed compaction verdict."""
|
||||
if self.last_prompt_tokens > 0:
|
||||
self.last_real_prompt_tokens = self.last_prompt_tokens
|
||||
self._provider_omits_usage = False
|
||||
if self.last_prompt_tokens < self.threshold_tokens:
|
||||
if self.awaiting_real_usage_after_compression and self.last_compression_rough_tokens > 0:
|
||||
self.last_rough_tokens_when_real_prompt_fit = self.last_compression_rough_tokens
|
||||
elif self._pending_request_rough_tokens > 0:
|
||||
# Pair the real prompt count with the same request's rough estimate so the defer baseline syncs on
|
||||
# EVERY fitting response, not only after compaction; otherwise a never-compressed session has no
|
||||
# baseline and preflight fires on the raw rough estimate (overcounts CJK / replay blobs severalfold).
|
||||
self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens
|
||||
# Any real reading below the trigger proves the prompt fits: clear the latch. The fallback streak survives.
|
||||
self._record_ineffective_compression_verdict(0)
|
||||
else:
|
||||
self.last_rough_tokens_when_real_prompt_fit = 0
|
||||
self._pending_request_rough_tokens = 0
|
||||
# Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count,
|
||||
# not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it.
|
||||
# Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's
|
||||
@@ -2409,12 +2399,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
return
|
||||
self.last_prompt_tokens = snapshot
|
||||
|
||||
def note_request_rough_estimate(self, rough_tokens: int) -> None:
|
||||
"""Record the rough estimate of the request about to be sent, for pairing with real usage."""
|
||||
try:
|
||||
self._pending_request_rough_tokens = max(0, int(rough_tokens))
|
||||
except (TypeError, ValueError):
|
||||
self._pending_request_rough_tokens = 0
|
||||
def note_usage_less_response(self) -> None:
|
||||
"""A completed response carried no usage: until a real reading arrives, this provider cannot
|
||||
adjudicate context pressure, so rough estimates decide instead of waiting forever (#2153)."""
|
||||
self._provider_omits_usage = True
|
||||
|
||||
def note_native_compaction_checkpoint(self) -> None:
|
||||
"""Wait for real usage before trusting a newly checkpointed request.
|
||||
@@ -2431,26 +2419,23 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self.last_compression_rough_tokens = 0
|
||||
|
||||
def should_defer_preflight_to_real_usage(self, rough_tokens: int) -> bool:
|
||||
"""Return True when a high rough preflight estimate is known-noisy.
|
||||
Projects real usage as ``last_real + (rough_now - rough_at_last_real)`` and fires only when the
|
||||
projection, not the raw estimate, crosses the threshold. Not a strict upper bound for
|
||||
chars/4-underestimated scripts (Cyrillic, Thai, Arabic); bounded by two backstops: a real
|
||||
reading at/over threshold clears the baseline, and the overflow handler compacts reactively.
|
||||
Callers with a smaller (raw-messages) basis can only over-defer; the pre-API pressure check
|
||||
re-runs with the aligned basis."""
|
||||
"""True when a whole-context ROUGH estimate over threshold must wait ONE request for the
|
||||
provider's real usage. Callers skip this for usage-anchored figures (real prompt count +
|
||||
delta of what was appended since), which never defer. A rough figure defers right after a
|
||||
local or native compaction (the last real reading is stale — the latch) and on any
|
||||
transcript the anchor does not cover (first request, rewind/edit-resend, reloaded history):
|
||||
the next response re-anchors it. It never defers once the provider has proven it omits
|
||||
usage, or the estimate would be the only signal and compression could never fire (#2153);
|
||||
the overflow handler compacts reactively in every case."""
|
||||
if rough_tokens < self.threshold_tokens:
|
||||
return False
|
||||
# After local or native compaction, last_real_prompt_tokens is STALE
|
||||
# (above threshold); defer one turn until real usage arrives.
|
||||
if self.awaiting_real_usage_after_compression:
|
||||
return True
|
||||
if self.last_real_prompt_tokens <= 0 or self.last_real_prompt_tokens >= self.threshold_tokens:
|
||||
# A real reading already at/over threshold needs no second opinion, and a rough figure past
|
||||
# the whole window describes a request certain to fail — sending it only buys an overflow error.
|
||||
if self.last_real_prompt_tokens >= self.threshold_tokens or rough_tokens >= self.context_length:
|
||||
return False
|
||||
baseline = self.last_rough_tokens_when_real_prompt_fit or self.last_compression_rough_tokens
|
||||
if baseline <= 0:
|
||||
return False
|
||||
# No baseline ratchet here: advancing rough without a matching real reading would defer on stale data.
|
||||
return self.last_real_prompt_tokens + max(0, rough_tokens - baseline) < self.threshold_tokens
|
||||
return not self._provider_omits_usage
|
||||
|
||||
def should_compress(self, prompt_tokens: int = None) -> bool:
|
||||
"""True when compression should run now (anti-thrash included; see :meth:`should_compress_info` for the reason)."""
|
||||
|
||||
@@ -39,6 +39,7 @@ def _preflight_request_tokens(
|
||||
"""Token estimate for automatic preflight compression: a valid provider usage anchor,
|
||||
else the checkpoint-pruned native wire payload, else the generic estimator."""
|
||||
anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
|
||||
agent._request_pressure_anchored = anchored is not None
|
||||
if anchored is not None:
|
||||
return anchored
|
||||
tools = getattr(agent, "tools", None) or None
|
||||
|
||||
@@ -258,7 +258,8 @@ def _preflight_compression(
|
||||
# snapshot may arm the interrupted-turn rollback.
|
||||
if isinstance(_snapshot_val, int) and not isinstance(_snapshot_val, bool):
|
||||
agent._turn_preflight_display_snapshot = _snapshot_val
|
||||
_preflight_deferred = getattr(
|
||||
# An anchored figure is real usage + delta: never deferred.
|
||||
_preflight_deferred = not getattr(agent, "_request_pressure_anchored", False) and getattr(
|
||||
_compressor, "should_defer_preflight_to_real_usage", lambda _tokens: False
|
||||
)(_preflight_tokens)
|
||||
_codex_native_auto = _codex_native_auto_compaction(agent)
|
||||
@@ -278,8 +279,8 @@ def _preflight_compression(
|
||||
_compress_block_reason = None
|
||||
if _preflight_deferred:
|
||||
logger.info(
|
||||
"Skipping preflight compression: rough estimate ~%s >= %s, "
|
||||
"but last real provider prompt was %s after compression",
|
||||
"Skipping preflight compression: rough estimate ~%s >= %s is not anchored on "
|
||||
"real usage (last real provider prompt %s); deferring to the next response",
|
||||
f"{_preflight_tokens:,}", f"{_compressor.threshold_tokens:,}",
|
||||
f"{_compressor.last_real_prompt_tokens:,}",
|
||||
)
|
||||
|
||||
+12
-21
@@ -263,32 +263,23 @@ def compress_after_tool_results(
|
||||
)
|
||||
|
||||
_compressor = agent.context_compressor
|
||||
# Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens
|
||||
# is the actual context size the provider reported plus the assistant turn — a tight lower bound for the
|
||||
# next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves
|
||||
# ample headroom; if tool results push past it, the next API call will report the real total and trigger
|
||||
# compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage
|
||||
# data), fall back to rough estimate to avoid missing compression. Without this, a session can grow
|
||||
# unbounded after disconnects because should_compress(0) never fires. (#2153)
|
||||
if _compressor.last_prompt_tokens > 0:
|
||||
# Real usage decides: the anchor is the provider's last prompt count plus a rough delta for
|
||||
# ONLY the tool results appended since (the raw last_prompt_tokens ignores them). Right after
|
||||
# a compaction (-1 sentinel) there is no real count yet: never treat the schema-heavy rough
|
||||
# figure as pressure. The whole-request rough estimate is the last resort (usage-less
|
||||
# provider, post-disconnect, gateway restart), kept route-aware (#96995/#97602).
|
||||
from agent.usage_anchor import anchored_context_tokens
|
||||
|
||||
_anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
|
||||
if _anchored is not None:
|
||||
_real_tokens = _anchored
|
||||
elif _compressor.last_prompt_tokens > 0:
|
||||
# Only prompt_tokens: thinking models inflate completion_tokens with
|
||||
# reasoning that uses no context → premature compression.
|
||||
# Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026)
|
||||
# reasoning that uses no context → premature compression. (#12026)
|
||||
_real_tokens = _compressor.last_prompt_tokens
|
||||
elif _compressor.last_prompt_tokens == -1:
|
||||
# Compression just ran, no API prompt count yet: don't treat a rough
|
||||
# schema-heavy post-compression estimate as real context pressure.
|
||||
_real_tokens = 0
|
||||
else:
|
||||
# Include tool schemas (20-30K tokens the messages-only estimate misses) and
|
||||
# stay route-aware: on a compacted native-Codex session the generic
|
||||
# durable-history figure would false-trigger.
|
||||
# Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate
|
||||
# misses, which can skip compression past the configured threshold (#14695). Route-aware
|
||||
# (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure
|
||||
# overstates the wire and would false-trigger compression here exactly like the pre-API guard — this
|
||||
# fallback runs precisely when no provider usage is available (post-disconnect / gateway restart),
|
||||
# the unanchored case from #97602's repro.
|
||||
_real_tokens = _midturn_request_pressure_tokens(
|
||||
agent, messages, active_system_prompt or "",
|
||||
estimate_request_tokens_rough(messages, tools=agent.tools or None),
|
||||
|
||||
@@ -90,8 +90,11 @@ def run_preflight_gate(
|
||||
return run_preflight_compression(
|
||||
agent, v, compressor=_compressor, request_pressure_tokens=request_pressure_tokens,
|
||||
provider_overflow_preflight=_provider_overflow_preflight,
|
||||
defer_preflight=getattr(
|
||||
_compressor, "should_defer_preflight_to_real_usage", lambda _t: False
|
||||
# An anchored figure is real usage + delta: never deferred. Only a whole-context rough
|
||||
# estimate waits for the provider's count.
|
||||
defer_preflight=(
|
||||
(lambda _t: False) if getattr(agent, "_request_pressure_anchored", False)
|
||||
else getattr(_compressor, "should_defer_preflight_to_real_usage", lambda _t: False)
|
||||
),
|
||||
moa_prepared_request=_moa_prepared_request, system_message=system_message,
|
||||
user_message=user_message, max_compression_attempts=max_compression_attempts,
|
||||
|
||||
@@ -245,6 +245,7 @@ def assemble_api_request(
|
||||
# Usage-anchored override: real prompt_tokens (incl. system + tool schemas) +
|
||||
# delta estimate replaces the whole-history heuristic when the anchor is fresh.
|
||||
_anchored_pressure = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None))
|
||||
agent._request_pressure_anchored = _anchored_pressure is not None
|
||||
if _anchored_pressure is not None:
|
||||
request_pressure_tokens = _anchored_pressure
|
||||
else:
|
||||
@@ -254,11 +255,6 @@ def assemble_api_request(
|
||||
request_pressure_tokens = _pressure_with_real_floor(
|
||||
agent.context_compressor, request_pressure_tokens
|
||||
)
|
||||
# Stash the rough estimate so update_from_response() can pair it with the real
|
||||
# count (should_defer_preflight_to_real_usage). getattr: test doubles lack it.
|
||||
_note_rough = getattr(agent.context_compressor, "note_request_rough_estimate", None)
|
||||
if callable(_note_rough):
|
||||
_note_rough(request_pressure_tokens)
|
||||
return AssembledRequest(
|
||||
"fallthrough", api_messages, tools_for_api, _moa_prepared_request,
|
||||
pending_moa_prepared_request, approx_tokens, request_pressure_tokens, approx_tokens * 4,
|
||||
|
||||
@@ -81,6 +81,9 @@ def record_response_usage(
|
||||
# pending verdict so later readings aren't charged to it and
|
||||
# preflight deferral isn't latched indefinitely.
|
||||
compressor.update_from_response({})
|
||||
_note_usage_less = getattr(compressor, "note_usage_less_response", None)
|
||||
if callable(_note_usage_less):
|
||||
_note_usage_less()
|
||||
logger.info(
|
||||
"API call #%d: model=%s provider=%s in=? out=? total=? latency=%.1fs usage=unavailable",
|
||||
agent.session_api_calls, agent.model, agent.provider or "unknown", api_duration,
|
||||
|
||||
+13
-2
@@ -617,10 +617,21 @@ class GatewayTurnMixin:
|
||||
_warn_token_threshold = int(_hyg_context_length * 0.95)
|
||||
_msg_count = len(history)
|
||||
|
||||
# Prefer the API-reported prompt tokens over the rough estimate (runs 30-50% high, which only
|
||||
# fires hygiene early — safe). Do NOT compensate with a threshold multiplier.
|
||||
# Real usage decides: the API-reported prompt count, else the anchor persisted on the session
|
||||
# row (real count + delta of what was appended since, survives gateway restarts), else the
|
||||
# rough estimate (runs 30-50% high, which only fires hygiene early — safe). Do NOT compensate
|
||||
# with a threshold multiplier.
|
||||
_anchored = None
|
||||
if session_entry.last_prompt_tokens <= 0:
|
||||
from agent.usage_anchor import persisted_anchor_tokens
|
||||
_session_db = getattr(self, "_session_db", None)
|
||||
_anchored = persisted_anchor_tokens(
|
||||
getattr(_session_db, "_db", _session_db), session_entry.session_id, history,
|
||||
)
|
||||
if session_entry.last_prompt_tokens > 0:
|
||||
_approx_tokens, _token_source = session_entry.last_prompt_tokens, "actual"
|
||||
elif _anchored is not None:
|
||||
_approx_tokens, _token_source = _anchored, "anchored"
|
||||
else:
|
||||
_approx_tokens, _token_source = estimate_messages_tokens_rough(history), "estimated"
|
||||
|
||||
|
||||
@@ -253,7 +253,6 @@ class TestShouldCompress:
|
||||
class TestUpdateFromResponse:
|
||||
def test_updates_fields(self, compressor):
|
||||
compressor.awaiting_real_usage_after_compression = True
|
||||
compressor.last_compression_rough_tokens = 90_000
|
||||
compressor.update_from_response({
|
||||
"prompt_tokens": 5000,
|
||||
"completion_tokens": 1000,
|
||||
@@ -262,97 +261,39 @@ class TestUpdateFromResponse:
|
||||
assert compressor.last_prompt_tokens == 5000
|
||||
assert compressor.last_completion_tokens == 1000
|
||||
assert compressor.last_real_prompt_tokens == 5000
|
||||
assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000
|
||||
assert compressor.awaiting_real_usage_after_compression is False
|
||||
|
||||
def test_missing_fields_default_zero(self, compressor):
|
||||
compressor.update_from_response({})
|
||||
assert compressor.last_prompt_tokens == 0
|
||||
|
||||
def test_pairs_noted_rough_estimate_with_fitting_real_usage(self, compressor):
|
||||
"""note_request_rough_estimate() + a fitting response must anchor the
|
||||
defer baseline even when no compression ever ran — this is what gives
|
||||
fresh sessions a (rough, real) pair before their first compaction."""
|
||||
compressor.note_request_rough_estimate(120_000)
|
||||
compressor.update_from_response({"prompt_tokens": 60_000})
|
||||
|
||||
assert compressor.last_real_prompt_tokens == 60_000
|
||||
assert compressor.last_rough_tokens_when_real_prompt_fit == 120_000
|
||||
# Consumed: a later usage-bearing response without a fresh note keeps
|
||||
# the previous pair instead of re-pairing against a stale estimate.
|
||||
assert compressor._pending_request_rough_tokens == 0
|
||||
|
||||
def test_post_compression_pairing_wins_over_noted_estimate(self, compressor):
|
||||
"""Right after a compaction the post-compression rough count is the
|
||||
authoritative baseline (#36718); a stale pre-compression note must not
|
||||
displace it."""
|
||||
compressor.note_request_rough_estimate(120_000)
|
||||
compressor.awaiting_real_usage_after_compression = True
|
||||
compressor.last_compression_rough_tokens = 40_000
|
||||
compressor.update_from_response({"prompt_tokens": 30_000})
|
||||
|
||||
assert compressor.last_rough_tokens_when_real_prompt_fit == 40_000
|
||||
|
||||
def test_usage_less_response_preserves_pending_note(self, compressor):
|
||||
"""Transports that report usage separately send usage-less responses
|
||||
first; the pending pair must survive until real usage arrives."""
|
||||
compressor.note_request_rough_estimate(120_000)
|
||||
compressor.update_from_response({})
|
||||
|
||||
assert compressor._pending_request_rough_tokens == 120_000
|
||||
|
||||
def test_over_threshold_real_usage_clears_pending_note(self, compressor):
|
||||
compressor.note_request_rough_estimate(120_000)
|
||||
compressor.update_from_response({"prompt_tokens": 90_000})
|
||||
|
||||
assert compressor.last_rough_tokens_when_real_prompt_fit == 0
|
||||
assert compressor._pending_request_rough_tokens == 0
|
||||
|
||||
class TestPreflightDeferral:
|
||||
"""A whole-context rough estimate over threshold waits ONE request for real usage; callers
|
||||
never consult this for usage-anchored figures."""
|
||||
|
||||
def test_defers_while_projected_real_usage_fits(self, compressor):
|
||||
"""Large rough growth alone must not trigger compaction: with real
|
||||
usage at 50K and 10K of rough growth since that reading, projected
|
||||
real usage is 60K — far under the 85K threshold. The old fixed 5%
|
||||
growth tolerance compacted here at ~59% of the real window (CJK /
|
||||
replay-blob overcount churn)."""
|
||||
def test_rough_estimate_defers_until_provider_prices_it(self, compressor):
|
||||
"""Real usage far under threshold, rough estimate far over (CJK / replay-blob overcount):
|
||||
the provider's next reading decides, not the estimate."""
|
||||
compressor.context_length = 200_000
|
||||
compressor.threshold_tokens = 85_000
|
||||
compressor.last_real_prompt_tokens = 50_000
|
||||
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
|
||||
|
||||
assert compressor.should_defer_preflight_to_real_usage(100_000) is True
|
||||
assert compressor.should_defer_preflight_to_real_usage(80_000) is False
|
||||
|
||||
def test_does_not_defer_when_projected_real_usage_crosses_threshold(self, compressor):
|
||||
"""Projection = last real + rough growth. 80K real + 6K growth = 86K
|
||||
>= 85K threshold: compression must run."""
|
||||
compressor.threshold_tokens = 85_000
|
||||
compressor.last_real_prompt_tokens = 80_000
|
||||
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
|
||||
|
||||
assert compressor.should_defer_preflight_to_real_usage(96_000) is False
|
||||
|
||||
def test_does_not_defer_without_a_baseline(self, compressor):
|
||||
"""No synchronized (rough, real) pair yet — fall back to trusting the
|
||||
rough estimate (conservative: compress)."""
|
||||
def test_never_defers_when_real_usage_cannot_arrive_or_is_already_over(self, compressor):
|
||||
"""Deferral is one request, never a disable: a provider that omits usage, a real reading
|
||||
already over threshold, or a rough figure past the whole window all compress now."""
|
||||
compressor.context_length = 100_000
|
||||
compressor.threshold_tokens = 85_000
|
||||
compressor.last_real_prompt_tokens = 90_000
|
||||
assert compressor.should_defer_preflight_to_real_usage(95_000) is False
|
||||
compressor.last_real_prompt_tokens = 50_000
|
||||
compressor.last_rough_tokens_when_real_prompt_fit = 0
|
||||
compressor.last_compression_rough_tokens = 0
|
||||
|
||||
assert compressor.should_defer_preflight_to_real_usage(100_000) is False
|
||||
|
||||
def test_defer_does_not_ratchet_baseline(self, compressor):
|
||||
"""The baseline is refreshed only by update_from_response() pairing.
|
||||
Deferring must not advance it: without a fresh real reading, a
|
||||
ratcheted baseline would shrink apparent growth and defer on stale
|
||||
data."""
|
||||
compressor.threshold_tokens = 85_000
|
||||
compressor.last_real_prompt_tokens = 50_000
|
||||
compressor.last_rough_tokens_when_real_prompt_fit = 90_000
|
||||
|
||||
assert compressor.should_defer_preflight_to_real_usage(100_000) is True
|
||||
assert compressor.last_rough_tokens_when_real_prompt_fit == 90_000
|
||||
|
||||
assert compressor.should_defer_preflight_to_real_usage(150_000) is False
|
||||
compressor.note_usage_less_response()
|
||||
assert compressor.should_defer_preflight_to_real_usage(95_000) is False
|
||||
compressor.update_from_response({"prompt_tokens": 50_000})
|
||||
assert compressor.should_defer_preflight_to_real_usage(95_000) is True
|
||||
|
||||
def test_defers_immediately_after_compaction_with_stale_real_prompt(self, compressor):
|
||||
"""#36718: right after a compaction, last_real_prompt_tokens still holds
|
||||
@@ -360,47 +301,24 @@ class TestPreflightDeferral:
|
||||
must force deferral so preflight doesn't fire a SECOND compaction before
|
||||
real post-compaction usage arrives."""
|
||||
compressor.threshold_tokens = 85_000
|
||||
# Stale pre-compression value — would hit the `>= threshold => False`
|
||||
# short-circuit and defeat deferral without the flag guard.
|
||||
compressor.last_real_prompt_tokens = 120_000
|
||||
compressor.awaiting_real_usage_after_compression = True
|
||||
assert compressor.should_defer_preflight_to_real_usage(95_000) is True
|
||||
|
||||
def test_native_checkpoint_defers_until_provider_usage_reanchors(self, compressor):
|
||||
"""A newly captured native checkpoint is opaque ciphertext, not text.
|
||||
|
||||
Its serialized size can add more than a million rough tokens even when
|
||||
the provider reports that the checkpoint-pruned request fits. Defer the
|
||||
local compressor for one request so real usage can establish the new
|
||||
rough/real calibration instead of immediately summarizing again.
|
||||
"""
|
||||
"""A newly captured native checkpoint is opaque ciphertext whose serialized size can add
|
||||
more than a million rough tokens; the local compressor waits one request for real usage."""
|
||||
compressor.threshold_tokens = 85_000
|
||||
compressor.last_real_prompt_tokens = 60_000
|
||||
compressor.last_rough_tokens_when_real_prompt_fit = 70_000
|
||||
compressor.last_compression_rough_tokens = 40_000
|
||||
|
||||
compressor.note_native_compaction_checkpoint()
|
||||
|
||||
encrypted_checkpoint_rough = 1_300_000
|
||||
assert compressor.awaiting_real_usage_after_compression is True
|
||||
assert compressor.last_compression_rough_tokens == 0
|
||||
assert (
|
||||
compressor.should_defer_preflight_to_real_usage(
|
||||
encrypted_checkpoint_rough
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
compressor.note_request_rough_estimate(encrypted_checkpoint_rough)
|
||||
assert compressor.should_defer_preflight_to_real_usage(1_300_000) is True
|
||||
compressor.update_from_response({"prompt_tokens": 65_000})
|
||||
|
||||
assert compressor.awaiting_real_usage_after_compression is False
|
||||
assert (
|
||||
compressor.last_rough_tokens_when_real_prompt_fit
|
||||
== encrypted_checkpoint_rough
|
||||
)
|
||||
|
||||
|
||||
|
||||
|
||||
class TestCompress:
|
||||
@@ -2221,7 +2139,6 @@ class TestUpdateModelResetsCalibration:
|
||||
# Simulate a large-model session that proved a prompt fit.
|
||||
comp.last_prompt_tokens = 120_000
|
||||
comp.last_real_prompt_tokens = 120_000
|
||||
comp.last_rough_tokens_when_real_prompt_fit = 130_000
|
||||
comp.last_compression_rough_tokens = 130_000
|
||||
comp.awaiting_real_usage_after_compression = True
|
||||
comp._ineffective_compression_count = 2
|
||||
@@ -2230,7 +2147,6 @@ class TestUpdateModelResetsCalibration:
|
||||
|
||||
assert comp.last_prompt_tokens == 0
|
||||
assert comp.last_real_prompt_tokens == 0
|
||||
assert comp.last_rough_tokens_when_real_prompt_fit == 0
|
||||
assert comp.last_compression_rough_tokens == 0
|
||||
assert comp.awaiting_real_usage_after_compression is False
|
||||
assert comp._ineffective_compression_count == 0
|
||||
@@ -2240,15 +2156,14 @@ class TestUpdateModelResetsCalibration:
|
||||
preflight on the new smaller model."""
|
||||
comp = self._comp()
|
||||
comp.last_real_prompt_tokens = 50_000
|
||||
comp.last_rough_tokens_when_real_prompt_fit = 90_000
|
||||
# Before switch, a modest rough growth would defer.
|
||||
comp.threshold_tokens = 85_000
|
||||
assert comp.should_defer_preflight_to_real_usage(93_000) is True
|
||||
|
||||
# After switching to a 65K model, the stale state is gone, so a rough
|
||||
# estimate over the new threshold is NOT deferred — preflight will run.
|
||||
# After switching to a 65K model the stale reading is gone; a rough estimate past the
|
||||
# NEW window is a request certain to fail and is never deferred — preflight will run.
|
||||
comp.update_model("small-model", context_length=65_536)
|
||||
assert comp.should_defer_preflight_to_real_usage(comp.threshold_tokens + 5_000) is False
|
||||
assert comp.should_defer_preflight_to_real_usage(comp.context_length + 5_000) is False
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -62,7 +62,6 @@ def _make_compressor():
|
||||
c._last_aux_model_failure_model = None
|
||||
c.last_real_prompt_tokens = 0
|
||||
c.last_compression_rough_tokens = 0
|
||||
c.last_rough_tokens_when_real_prompt_fit = 0
|
||||
c.awaiting_real_usage_after_compression = False
|
||||
return c
|
||||
|
||||
|
||||
@@ -67,7 +67,6 @@ def _make_compressor():
|
||||
c._last_aux_model_failure_model = None
|
||||
c.last_real_prompt_tokens = 0
|
||||
c.last_compression_rough_tokens = 0
|
||||
c.last_rough_tokens_when_real_prompt_fit = 0
|
||||
c.awaiting_real_usage_after_compression = False
|
||||
c._previous_summary = None
|
||||
c._summary_has_user_turn = None
|
||||
@@ -91,7 +90,6 @@ def _simulate_cron_session_state(c):
|
||||
c._context_probe_persistable = True
|
||||
c.last_real_prompt_tokens = 50000
|
||||
c.last_compression_rough_tokens = 60000
|
||||
c.last_rough_tokens_when_real_prompt_fit = 55000
|
||||
c.awaiting_real_usage_after_compression = True
|
||||
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ class TestShouldCompressInfo:
|
||||
|
||||
def test_cooldown_reports_reason(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 60
|
||||
should, reason = comp.should_compress_info(73_000)
|
||||
assert should is False
|
||||
@@ -58,7 +58,7 @@ class TestShouldCompressInfo:
|
||||
"""should_compress() must still return a bare bool for existing
|
||||
callers in conversation_loop.py (and/or chains)."""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 60
|
||||
result = comp.should_compress(73_000)
|
||||
assert result is False
|
||||
@@ -118,7 +118,7 @@ def _run_build(agent):
|
||||
class TestTurnContextOverflowWarning:
|
||||
def test_warns_on_cooldown_block(self):
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
_run_build(agent)
|
||||
@@ -139,7 +139,7 @@ class TestTurnContextOverflowWarning:
|
||||
cooldown timer moves.
|
||||
"""
|
||||
comp = _make_compressor()
|
||||
comp.last_prompt_tokens = 73_000
|
||||
comp.last_prompt_tokens = comp.last_real_prompt_tokens = 73_000
|
||||
comp._summary_failure_cooldown_until = time.monotonic() + 30
|
||||
agent = _build_warn_agent(comp)
|
||||
# Turn 1: over threshold + cooldown -> warn.
|
||||
|
||||
@@ -624,6 +624,7 @@ class TestPreflightCompression:
|
||||
not leak even though compaction itself still runs.
|
||||
"""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 130_000
|
||||
agent.context_compressor.emit_automatic_compaction_status = False
|
||||
@@ -667,6 +668,7 @@ class TestPreflightCompression:
|
||||
def test_preflight_compresses_oversized_history(self, agent):
|
||||
"""When loaded history exceeds the model's context threshold, compress before API call."""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
# Set a small context so the history is "oversized", but large enough
|
||||
# that the compressed result (2 short messages) fits in a single pass.
|
||||
agent.context_compressor.context_length = 2000
|
||||
@@ -719,6 +721,7 @@ class TestPreflightCompression:
|
||||
def test_preflight_suppresses_status_when_context_engine_opts_out(self, agent):
|
||||
"""LCM-style engines can keep routine automatic preflight maintenance silent."""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 100_000
|
||||
agent.context_compressor.emit_automatic_compaction_status = False
|
||||
@@ -763,6 +766,7 @@ class TestPreflightCompression:
|
||||
def test_preflight_uses_context_engine_custom_status_message(self, agent):
|
||||
"""Plugin engines can replace generic built-in-compressor wording."""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 100_000
|
||||
|
||||
@@ -811,46 +815,27 @@ class TestPreflightCompression:
|
||||
assert not any("Preflight compression" in msg for msg in lifecycle_messages)
|
||||
|
||||
|
||||
def test_preflight_compresses_when_projected_real_usage_crosses(self, agent):
|
||||
"""Projected real usage (last real + rough growth) crossing the
|
||||
threshold still triggers preflight: 95K real + 12K rough growth =
|
||||
107K >= 100K. Growth alone no longer decides — real usage far below
|
||||
the threshold defers instead (see TestPreflightDeferral)."""
|
||||
def test_rough_over_threshold_waits_one_request_then_real_usage_compresses(self, agent):
|
||||
"""Real usage decides, the estimate only decides whether to wait for it: a whole-history
|
||||
rough estimate over threshold with no anchor defers ONE request; the provider's real
|
||||
prompt count then drives the next gate — over threshold compresses, under does not."""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 100_000
|
||||
agent.context_compressor.last_prompt_tokens = 95_000
|
||||
agent.context_compressor.last_real_prompt_tokens = 95_000
|
||||
agent.context_compressor.last_rough_tokens_when_real_prompt_fit = 113_000
|
||||
|
||||
big_history = []
|
||||
for i in range(20):
|
||||
big_history.append({"role": "user", "content": f"Message {i} padded"})
|
||||
big_history.append({"role": "assistant", "content": f"Response {i} padded"})
|
||||
|
||||
ok_resp = _mock_response(
|
||||
content="Compressed after growth",
|
||||
finish_reason="stop",
|
||||
usage={"prompt_tokens": 50_000, "completion_tokens": 100, "total_tokens": 50_100},
|
||||
)
|
||||
agent.client.chat.completions.create.side_effect = [ok_resp]
|
||||
|
||||
# First rough estimate must clear the threshold so preflight fires
|
||||
# (rough growth since the last fitting request is large, so the
|
||||
# deferral path is NOT taken). Every estimate after compaction is
|
||||
# sub-threshold. Use a callable side_effect rather than a fixed list
|
||||
# so we don't have to predict how many times the loop re-estimates —
|
||||
# the post-response real-token estimate is an extra call that a
|
||||
# 2-element list would exhaust (StopIteration).
|
||||
_rough_calls = {"n": 0}
|
||||
|
||||
def _rough_estimate(*_args, **_kwargs):
|
||||
_rough_calls["n"] += 1
|
||||
return 125_000 if _rough_calls["n"] == 1 else 40_000
|
||||
agent.client.chat.completions.create.side_effect = [
|
||||
_mock_response(content="first", usage={"prompt_tokens": 105_000, "completion_tokens": 10, "total_tokens": 105_010}),
|
||||
_mock_response(content="second", usage={"prompt_tokens": 50_000, "completion_tokens": 10, "total_tokens": 50_010}),
|
||||
]
|
||||
|
||||
with (
|
||||
patch("agent.turn_context.estimate_request_tokens_rough", side_effect=_rough_estimate),
|
||||
patch("agent.model_metadata.estimate_request_tokens_rough", side_effect=_rough_estimate),
|
||||
patch("agent.turn_context.estimate_request_tokens_rough", return_value=125_000),
|
||||
patch("agent.model_metadata.estimate_request_tokens_rough", return_value=125_000),
|
||||
patch.object(agent, "_compress_context") as mock_compress,
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
@@ -860,10 +845,12 @@ class TestPreflightCompression:
|
||||
[{"role": "user", "content": f"{SUMMARY_PREFIX}\nPrevious conversation"}],
|
||||
"new system prompt",
|
||||
)
|
||||
result = agent.run_conversation("hello", conversation_history=big_history)
|
||||
r1 = agent.run_conversation("hello", conversation_history=big_history)
|
||||
assert mock_compress.call_count == 0, "estimate alone must not compress before real usage"
|
||||
agent.run_conversation("again", conversation_history=r1["messages"])
|
||||
|
||||
mock_compress.assert_called_once()
|
||||
assert result["completed"] is True
|
||||
assert mock_compress.call_count == 1
|
||||
assert mock_compress.call_args.kwargs["approx_tokens"] >= 105_000
|
||||
|
||||
def test_no_preflight_when_under_threshold(self, agent):
|
||||
"""When history fits within context, no preflight compression needed."""
|
||||
@@ -928,6 +915,7 @@ class TestPreflightCompression:
|
||||
):
|
||||
"""The proactive retry block must not consume provider-overflow recovery."""
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 130_000
|
||||
|
||||
@@ -1183,6 +1171,7 @@ class TestPreflightCompression:
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.threshold_tokens = 130_000
|
||||
agent.context_compressor.last_prompt_tokens = 74_400
|
||||
agent.context_compressor.note_usage_less_response() # provider omits usage: the estimate decides
|
||||
agent.context_compressor._ineffective_compression_count = 1
|
||||
|
||||
big_history = []
|
||||
|
||||
@@ -155,7 +155,11 @@ def test_pre_api_compression_budget_rearms_only_after_pressure_clears(
|
||||
estimate_values = iter([200, 190, 200, 10])
|
||||
_last_estimate = [10]
|
||||
|
||||
def _next_estimate(*_args, **_kwargs):
|
||||
def _next_estimate(messages=None, *_args, **_kwargs):
|
||||
# The scripted sequence prices the WHOLE history; a usage-anchored gate
|
||||
# estimates only the few messages appended since the real reading.
|
||||
if isinstance(messages, list) and len(messages) <= 4:
|
||||
return 10
|
||||
# The provider-recovery variant re-runs the pre-API preflight after
|
||||
# fallback activation (#84733), consuming an extra estimate reading.
|
||||
# Hold the final low-pressure value once the scripted sequence is
|
||||
|
||||
Reference in New Issue
Block a user