From c49e2a496ec11232ce47a83c5baf653476695cfa Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 30 Aug 2026 04:34:57 -0700 Subject: [PATCH] fix(compression): route-aware pruned estimate at remaining pressure sibling sites MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the mid-turn pre-API guard fix (#96995 / #97602): sweep the remaining call sites that derive automatic compression pressure from a generic estimate over the assembled durable history, which on a compacted native-Codex session overstates the wire payload by orders of magnitude. - agent/turn_context.py idle-triggered compaction: use _preflight_request_tokens (anchor -> native pruned -> generic) instead of the raw generic request estimate, so resuming a compacted codex session after an idle gap does not fire a compaction the next request never needed. - agent/turn_context.py uncompressed-session overflow-warn RE-ARM: match the warn site's route-aware figure so the dedup re-arms correctly on native sessions. - agent/conversation_loop.py post-response should_compress fallback (last_prompt_tokens==0, i.e. no provider usage after a disconnect or gateway restart — the unanchored case in #97602's repro): route through _midturn_request_pressure_tokens instead of the generic figure. Left alone deliberately: provider-proven overflow recovery paths (413 / context-length errors — the provider already proved the request does not fit, figures there only arm recovery and score progress), compression progress before/after pairs (relative deltas on the same scale), manual /compress display estimates (gateway/CLI/ACP feedback, not automatic triggers), MoA advisor budget trimming (not a codex-native wire payload), and context_compressor internals (measure local durable-history shrink). --- agent/conversation_loop.py | 14 ++++++++++++-- agent/turn_context.py | 23 +++++++++++++++++------ 2 files changed, 29 insertions(+), 8 deletions(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 2d01ae9a85..438adf6647 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -7660,8 +7660,18 @@ def run_conversation( # these add 20-30K tokens the messages-only # estimate misses, which can skip compression # past the configured threshold (#14695). - _real_tokens = estimate_request_tokens_rough( - messages, tools=agent.tools or None + # Route-aware (#96995/#97602 class): on a compacted + # native-Codex session the generic durable-history + # figure overstates the wire and would false-trigger + # compression here exactly like the pre-API guard — + # this fallback runs precisely when no provider usage + # is available (post-disconnect / gateway restart), + # the unanchored case from #97602's repro. + _real_tokens = _midturn_request_pressure_tokens( + agent, + messages, + active_system_prompt or "", + estimate_messages_tokens_rough(messages), ) if ( diff --git a/agent/turn_context.py b/agent/turn_context.py index 01dbf27371..8959386bd4 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -883,10 +883,15 @@ def build_turn_context( _idle_gap = time.time() - getattr(agent, "_last_activity_ts", time.time()) if _idle_gap >= _idle_after: _compressor = agent.context_compressor - _idle_tokens = estimate_request_tokens_rough( + # Route-aware pressure (#96995/#97602 class): on a compacted + # native-Codex session the generic durable-history figure + # overstates the wire by orders of magnitude and would fire an + # idle compaction the next request never needed. Reuse the + # preflight estimator (anchor → native pruned → generic). + _idle_tokens = _preflight_request_tokens( + agent, messages, - system_prompt=active_system_prompt or "", - tools=agent.tools or None, + active_system_prompt or "", ) # Post-compression target size: don't summarise a thread already # below what compaction would reduce it to. @@ -1297,10 +1302,16 @@ def build_turn_context( if callable(_clear_warn): _clear_warn() else: - _uncompressed_tokens = estimate_request_tokens_rough( + # Route-aware (#96995/#97602 class): the warn site in the + # conversation loop now measures the checkpoint-pruned wire + # payload on native-Codex sessions, so the re-arm must use + # the same figure — otherwise a compacted session that fits + # on the wire never clears the dedup and future genuine + # overflow warnings stay suppressed. + _uncompressed_tokens = _preflight_request_tokens( + agent, messages, - system_prompt=active_system_prompt or "", - tools=agent.tools or None, + active_system_prompt or "", ) if _uncompressed_tokens <= _ctx_len: _clear_warn = getattr(