fix(compression): use the route-aware pruned estimate in the mid-turn pre-API guard

The #96155 fix (#96644) made the turn-prologue preflight estimate the
checkpoint-pruned native Responses payload, but the independent mid-turn
pre-API pressure guard in conversation_loop still estimated the full
assembled durable history. On a compacted native-Codex session the
generic figure overstates the wire by orders of magnitude (the issue's
deterministic probe: 1,037,241 generic vs 6,036 pruned, 171x), so the
guard false-tripped a 600-second local compression the main request
never needed — the live sequence shows the actual request then fit at
164k input tokens against a 765k threshold (#96995).

Extract the guard's pressure figure into _midturn_request_pressure_tokens
and mirror the turn-prologue: when native Responses compaction is proven
eligible, use estimate_native_responses_preflight_tokens (system prompt
and tools included, checkpoint-pruned); otherwise keep the generic
message+tools figure. Passing the assembled api_messages alongside
effective_system counts the system prompt exactly once — the estimator's
converter skips system-role rows and adds the prompt separately.
total_chars (verbose log proxy) and the non-codex paths are unchanged.

Fixes #96995
This commit is contained in:
liuhao1024
2026-08-28 16:55:18 +08:00
committed by Teknium
parent a2af8405d1
commit 222fda3d6b
2 changed files with 101 additions and 2 deletions
+53 -2
View File
@@ -126,6 +126,51 @@ RUN_BUDGET_WRAPUP_NOTICE = (
)
def _midturn_request_pressure_tokens(
agent: Any,
api_messages: List[Dict[str, Any]],
effective_system: str,
approx_tokens: int,
) -> int:
"""Token figure the mid-turn pre-API compression guard compares.
When the upcoming request is eligible for native Responses compaction the
transport will checkpoint-prune the payload before sending, so the generic
durable-history estimate overstates the wire by orders of magnitude on a
compacted session and fires a 600s local compression the main request
never needed (#96995). Mirror the turn-prologue preflight (#96644 /
#96155): use the pruned estimate when native eligibility is proven, the
generic message+tools figure otherwise.
The native estimator adds the system prompt and tool schemas itself and
its converter skips system-role rows, so passing the assembled
``api_messages`` (which carries the system row) alongside
``effective_system`` counts the system prompt exactly once.
"""
try:
from agent.codex_responses_adapter import (
estimate_native_responses_preflight_tokens,
)
native = estimate_native_responses_preflight_tokens(
agent,
api_messages,
system_prompt=effective_system or "",
tools=getattr(agent, "tools", None) or None,
)
if isinstance(native, int) and not isinstance(native, bool) and native >= 0:
return native
except Exception:
logger.debug(
"native Responses mid-turn estimate unavailable; "
"using generic transcript estimate",
exc_info=True,
)
return approx_tokens + (
_estimate_tools_tokens_rough(agent.tools) if agent.tools else 0
)
def _review_input_budget_exhausted(agent: Any) -> bool:
"""True when a detached review fork has replayed its aggregate input budget.
@@ -2608,8 +2653,14 @@ def run_conversation(
# separately (compression needs them: 50+ tools = 20-30K tokens).
# total_chars is a rough (~) proxy — verbose log + hook metric only.
approx_tokens = estimate_messages_tokens_rough(api_messages)
request_pressure_tokens = approx_tokens + (
_estimate_tools_tokens_rough(agent.tools) if agent.tools else 0
# Route-aware pressure: when the upcoming request is eligible for
# native Responses compaction the transport will checkpoint-prune
# the payload before sending — the generic durable-history figure
# overstates the wire by orders of magnitude on a compacted session
# and fires a 600s local compression the main request never needed
# (#96995, mirroring the turn-prologue preflight #96644/#96155).
request_pressure_tokens = _midturn_request_pressure_tokens(
agent, api_messages, effective_system or "", approx_tokens
)
# Usage-anchored override: when the last provider response's exact
# usage is still valid for the durable transcript, replace the
@@ -98,3 +98,51 @@ def test_preflight_wrapper_falls_back_to_generic_when_ineligible():
generic = estimate_request_tokens_rough(messages)
assert _preflight_request_tokens(agent, messages, "") == generic
# ── Mid-turn pre-API guard parity (#96995) ────────────────────────────────
# The mid-turn guard in conversation_loop must measure the same pruned wire
# payload the turn-prologue preflight does (#96644/#96155); before #96995 it
# used the generic durable-history estimate and false-tripped 600s local
# compression on compacted native-Codex sessions.
def test_midturn_pressure_uses_pruned_estimate_when_eligible():
from agent.conversation_loop import (
_midturn_request_pressure_tokens,
estimate_messages_tokens_rough,
)
agent = _codex_agent()
messages = [{"role": "system", "content": "be brief"}] + _history_with_checkpoint()
native = estimate_native_responses_preflight_tokens(
agent, messages, system_prompt="be brief"
)
generic = estimate_messages_tokens_rough(messages)
assert native is not None
assert generic > native * 2
# The assembled api_messages carry the system row; the helper must not
# double-count it (converter skips system rows, system_prompt adds it once).
assert _midturn_request_pressure_tokens(
agent, messages, "be brief", generic
) == native
def test_midturn_pressure_falls_back_to_generic_plus_tools_when_ineligible():
from agent.conversation_loop import (
_estimate_tools_tokens_rough,
_midturn_request_pressure_tokens,
estimate_messages_tokens_rough,
)
agent = _codex_agent(
api_mode="chat_completions",
tools=[{"type": "function", "function": {"name": "t", "parameters": {}}}],
)
messages = [{"role": "system", "content": "be brief"}] + _history_with_checkpoint()
approx = estimate_messages_tokens_rough(messages)
assert _midturn_request_pressure_tokens(
agent, messages, "be brief", approx
) == approx + _estimate_tools_tokens_rough(agent.tools)