diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 2d01ae9a85..438adf6647 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -7660,8 +7660,18 @@ def run_conversation( # these add 20-30K tokens the messages-only # estimate misses, which can skip compression # past the configured threshold (#14695). - _real_tokens = estimate_request_tokens_rough( - messages, tools=agent.tools or None + # Route-aware (#96995/#97602 class): on a compacted + # native-Codex session the generic durable-history + # figure overstates the wire and would false-trigger + # compression here exactly like the pre-API guard — + # this fallback runs precisely when no provider usage + # is available (post-disconnect / gateway restart), + # the unanchored case from #97602's repro. + _real_tokens = _midturn_request_pressure_tokens( + agent, + messages, + active_system_prompt or "", + estimate_messages_tokens_rough(messages), ) if ( diff --git a/agent/turn_context.py b/agent/turn_context.py index 01dbf27371..8959386bd4 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -883,10 +883,15 @@ def build_turn_context( _idle_gap = time.time() - getattr(agent, "_last_activity_ts", time.time()) if _idle_gap >= _idle_after: _compressor = agent.context_compressor - _idle_tokens = estimate_request_tokens_rough( + # Route-aware pressure (#96995/#97602 class): on a compacted + # native-Codex session the generic durable-history figure + # overstates the wire by orders of magnitude and would fire an + # idle compaction the next request never needed. Reuse the + # preflight estimator (anchor → native pruned → generic). + _idle_tokens = _preflight_request_tokens( + agent, messages, - system_prompt=active_system_prompt or "", - tools=agent.tools or None, + active_system_prompt or "", ) # Post-compression target size: don't summarise a thread already # below what compaction would reduce it to. @@ -1297,10 +1302,16 @@ def build_turn_context( if callable(_clear_warn): _clear_warn() else: - _uncompressed_tokens = estimate_request_tokens_rough( + # Route-aware (#96995/#97602 class): the warn site in the + # conversation loop now measures the checkpoint-pruned wire + # payload on native-Codex sessions, so the re-arm must use + # the same figure — otherwise a compacted session that fits + # on the wire never clears the dedup and future genuine + # overflow warnings stay suppressed. + _uncompressed_tokens = _preflight_request_tokens( + agent, messages, - system_prompt=active_system_prompt or "", - tools=agent.tools or None, + active_system_prompt or "", ) if _uncompressed_tokens <= _ctx_len: _clear_warn = getattr(