fix(compression): use the route-aware pruned estimate in the mid-turn pre-API guard
The #96155 fix (#96644) made the turn-prologue preflight estimate the checkpoint-pruned native Responses payload, but the independent mid-turn pre-API pressure guard in conversation_loop still estimated the full assembled durable history. On a compacted native-Codex session the generic figure overstates the wire by orders of magnitude (the issue's deterministic probe: 1,037,241 generic vs 6,036 pruned, 171x), so the guard false-tripped a 600-second local compression the main request never needed — the live sequence shows the actual request then fit at 164k input tokens against a 765k threshold (#96995). Extract the guard's pressure figure into _midturn_request_pressure_tokens and mirror the turn-prologue: when native Responses compaction is proven eligible, use estimate_native_responses_preflight_tokens (system prompt and tools included, checkpoint-pruned); otherwise keep the generic message+tools figure. Passing the assembled api_messages alongside effective_system counts the system prompt exactly once — the estimator's converter skips system-role rows and adds the prompt separately. total_chars (verbose log proxy) and the non-codex paths are unchanged. Fixes #96995
This commit is contained in:
@@ -126,6 +126,51 @@ RUN_BUDGET_WRAPUP_NOTICE = (
|
||||
)
|
||||
|
||||
|
||||
def _midturn_request_pressure_tokens(
|
||||
agent: Any,
|
||||
api_messages: List[Dict[str, Any]],
|
||||
effective_system: str,
|
||||
approx_tokens: int,
|
||||
) -> int:
|
||||
"""Token figure the mid-turn pre-API compression guard compares.
|
||||
|
||||
When the upcoming request is eligible for native Responses compaction the
|
||||
transport will checkpoint-prune the payload before sending, so the generic
|
||||
durable-history estimate overstates the wire by orders of magnitude on a
|
||||
compacted session and fires a 600s local compression the main request
|
||||
never needed (#96995). Mirror the turn-prologue preflight (#96644 /
|
||||
#96155): use the pruned estimate when native eligibility is proven, the
|
||||
generic message+tools figure otherwise.
|
||||
|
||||
The native estimator adds the system prompt and tool schemas itself and
|
||||
its converter skips system-role rows, so passing the assembled
|
||||
``api_messages`` (which carries the system row) alongside
|
||||
``effective_system`` counts the system prompt exactly once.
|
||||
"""
|
||||
try:
|
||||
from agent.codex_responses_adapter import (
|
||||
estimate_native_responses_preflight_tokens,
|
||||
)
|
||||
|
||||
native = estimate_native_responses_preflight_tokens(
|
||||
agent,
|
||||
api_messages,
|
||||
system_prompt=effective_system or "",
|
||||
tools=getattr(agent, "tools", None) or None,
|
||||
)
|
||||
if isinstance(native, int) and not isinstance(native, bool) and native >= 0:
|
||||
return native
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"native Responses mid-turn estimate unavailable; "
|
||||
"using generic transcript estimate",
|
||||
exc_info=True,
|
||||
)
|
||||
return approx_tokens + (
|
||||
_estimate_tools_tokens_rough(agent.tools) if agent.tools else 0
|
||||
)
|
||||
|
||||
|
||||
def _review_input_budget_exhausted(agent: Any) -> bool:
|
||||
"""True when a detached review fork has replayed its aggregate input budget.
|
||||
|
||||
@@ -2608,8 +2653,14 @@ def run_conversation(
|
||||
# separately (compression needs them: 50+ tools = 20-30K tokens).
|
||||
# total_chars is a rough (~) proxy — verbose log + hook metric only.
|
||||
approx_tokens = estimate_messages_tokens_rough(api_messages)
|
||||
request_pressure_tokens = approx_tokens + (
|
||||
_estimate_tools_tokens_rough(agent.tools) if agent.tools else 0
|
||||
# Route-aware pressure: when the upcoming request is eligible for
|
||||
# native Responses compaction the transport will checkpoint-prune
|
||||
# the payload before sending — the generic durable-history figure
|
||||
# overstates the wire by orders of magnitude on a compacted session
|
||||
# and fires a 600s local compression the main request never needed
|
||||
# (#96995, mirroring the turn-prologue preflight #96644/#96155).
|
||||
request_pressure_tokens = _midturn_request_pressure_tokens(
|
||||
agent, api_messages, effective_system or "", approx_tokens
|
||||
)
|
||||
# Usage-anchored override: when the last provider response's exact
|
||||
# usage is still valid for the durable transcript, replace the
|
||||
|
||||
@@ -98,3 +98,51 @@ def test_preflight_wrapper_falls_back_to_generic_when_ineligible():
|
||||
generic = estimate_request_tokens_rough(messages)
|
||||
|
||||
assert _preflight_request_tokens(agent, messages, "") == generic
|
||||
|
||||
|
||||
# ── Mid-turn pre-API guard parity (#96995) ────────────────────────────────
|
||||
# The mid-turn guard in conversation_loop must measure the same pruned wire
|
||||
# payload the turn-prologue preflight does (#96644/#96155); before #96995 it
|
||||
# used the generic durable-history estimate and false-tripped 600s local
|
||||
# compression on compacted native-Codex sessions.
|
||||
|
||||
|
||||
def test_midturn_pressure_uses_pruned_estimate_when_eligible():
|
||||
from agent.conversation_loop import (
|
||||
_midturn_request_pressure_tokens,
|
||||
estimate_messages_tokens_rough,
|
||||
)
|
||||
|
||||
agent = _codex_agent()
|
||||
messages = [{"role": "system", "content": "be brief"}] + _history_with_checkpoint()
|
||||
native = estimate_native_responses_preflight_tokens(
|
||||
agent, messages, system_prompt="be brief"
|
||||
)
|
||||
generic = estimate_messages_tokens_rough(messages)
|
||||
|
||||
assert native is not None
|
||||
assert generic > native * 2
|
||||
# The assembled api_messages carry the system row; the helper must not
|
||||
# double-count it (converter skips system rows, system_prompt adds it once).
|
||||
assert _midturn_request_pressure_tokens(
|
||||
agent, messages, "be brief", generic
|
||||
) == native
|
||||
|
||||
|
||||
def test_midturn_pressure_falls_back_to_generic_plus_tools_when_ineligible():
|
||||
from agent.conversation_loop import (
|
||||
_estimate_tools_tokens_rough,
|
||||
_midturn_request_pressure_tokens,
|
||||
estimate_messages_tokens_rough,
|
||||
)
|
||||
|
||||
agent = _codex_agent(
|
||||
api_mode="chat_completions",
|
||||
tools=[{"type": "function", "function": {"name": "t", "parameters": {}}}],
|
||||
)
|
||||
messages = [{"role": "system", "content": "be brief"}] + _history_with_checkpoint()
|
||||
approx = estimate_messages_tokens_rough(messages)
|
||||
|
||||
assert _midturn_request_pressure_tokens(
|
||||
agent, messages, "be brief", approx
|
||||
) == approx + _estimate_tools_tokens_rough(agent.tools)
|
||||
|
||||
Reference in New Issue
Block a user