fix(compression): anchor codex app-server preflight on reported usage
The codex_app_server runtime bypasses the conversation loop, so the usage-anchored context accounting captured there never ran: agent._usage_anchor stayed None forever. Every preflight estimate therefore fell back to the rough mirror-transcript heuristic, which is deliberately never compacted on this runtime (_record_codex_app_server_compaction preserves the mirror), so it grows monotonically while the real thread may be tiny or freshly compacted. With compression.codex_app_server_auto=hermes that estimate alone tripped the threshold and fired thread/compact/start on nearly every turn of a long-lived session (#100381). Mirror the main loop's post-response capture: _record_codex_app_server_usage now snapshots the reported thread usage into agent._usage_anchor (base prompt + completion exactly as the provider counted them, with estimation confined to messages appended since). A usage-less turn keeps the previous anchor, and the post-compaction invalidation site is unchanged.
This commit is contained in:
+12
-3
@@ -81,9 +81,12 @@ def _queue_token_counts(agent, fail_msg: str, *fail_extra: Any, counts: Callable
|
||||
logger.debug(fail_msg, agent.session_id, *fail_extra, exc)
|
||||
|
||||
|
||||
def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any]:
|
||||
"""Translate Codex app-server token usage into Hermes accounting. Prompt bucket = uncached + cached
|
||||
input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call."""
|
||||
input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call.
|
||||
``messages`` (the transcript mirror) lets real usage anchor the next preflight: this runtime bypasses
|
||||
the main loop's capture, and the mirror is never compacted natively, so without an anchor the rough
|
||||
estimate grows monotonically and hermes-mode fires thread compaction on tiny threads (#100381)."""
|
||||
agent.session_api_calls += 1
|
||||
usage = getattr(turn, "token_usage_last", None)
|
||||
compressor = getattr(agent, "context_compressor", None)
|
||||
@@ -117,6 +120,12 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
compressor.context_length = context_window
|
||||
except Exception:
|
||||
logger.debug("codex app-server usage update failed", exc_info=True)
|
||||
if isinstance(messages, list):
|
||||
from agent.model_metadata import capture_usage_anchor
|
||||
|
||||
anchor = capture_usage_anchor(prompt_tokens, canonical_usage.output_tokens, messages)
|
||||
if anchor is not None:
|
||||
agent._usage_anchor = anchor
|
||||
for key, value in usage_dict.items():
|
||||
setattr(agent, f"session_{key}", getattr(agent, f"session_{key}") + value)
|
||||
cost_result = estimate_usage_cost(
|
||||
@@ -429,7 +438,7 @@ def _finish_codex_turn(agent, turn, messages: List[Dict[str, Any]], *, original_
|
||||
# run_conversation() already bumped _turns_since_memory / _user_turn_count; only _iters_since_skill is ours.
|
||||
agent._iters_since_skill = getattr(agent, "_iters_since_skill", 0) + turn.tool_iterations
|
||||
_record_codex_app_server_compaction(agent, turn)
|
||||
usage_result = _record_codex_app_server_usage(agent, turn)
|
||||
usage_result = _record_codex_app_server_usage(agent, turn, messages=messages)
|
||||
# Skill nudge check AFTER iters were incremented (same as chat_completions).
|
||||
should_review_skills = (0 < agent._skill_nudge_interval <= agent._iters_since_skill
|
||||
and "skill_manage" in agent.valid_tool_names)
|
||||
|
||||
@@ -3812,7 +3812,7 @@ def _compress_context_via_codex_app_server(
|
||||
# An empty usage report must consume the pending verdict, not leave deferral
|
||||
# armed until a later turn; minimal test engines may lack update_from_response.
|
||||
if hasattr(agent.context_compressor, "update_from_response"):
|
||||
_record_codex_app_server_usage(agent, result)
|
||||
_record_codex_app_server_usage(agent, result, messages=messages)
|
||||
_reset_read_dedup_caches(task_id, skills=False)
|
||||
logger.info(
|
||||
"codex app-server compaction done: session=%s thread=%s turn=%s", _sid,
|
||||
|
||||
@@ -208,5 +208,77 @@ class TestCompressionTriggerUsesAnchor:
|
||||
assert anchored is not None and anchored < threshold
|
||||
|
||||
|
||||
class TestCodexAppServerAnchor:
|
||||
"""The codex_app_server runtime bypasses the conversation loop, so its
|
||||
usage recording is the only site that can maintain agent._usage_anchor.
|
||||
Without it, hermes-mode preflight falls back to the rough mirror-transcript
|
||||
heuristic, which grows monotonically (native compaction preserves the
|
||||
mirror) and fires thread compaction on tiny real threads (#100381)."""
|
||||
|
||||
def _agent(self, anchor=None):
|
||||
return SimpleNamespace(
|
||||
_usage_anchor=anchor,
|
||||
session_api_calls=0,
|
||||
session_prompt_tokens=0,
|
||||
session_completion_tokens=0,
|
||||
session_total_tokens=0,
|
||||
session_input_tokens=0,
|
||||
session_output_tokens=0,
|
||||
session_cache_read_tokens=0,
|
||||
session_cache_write_tokens=0,
|
||||
session_reasoning_tokens=0,
|
||||
context_compressor=None,
|
||||
event_callback=None,
|
||||
_session_db=None,
|
||||
model="codex-test-model",
|
||||
provider="openai",
|
||||
base_url=None,
|
||||
)
|
||||
|
||||
def _turn(self, usage):
|
||||
return SimpleNamespace(token_usage_last=usage, model_context_window=None)
|
||||
|
||||
def _usage(self, input_tokens=12_000, output_tokens=100):
|
||||
return {
|
||||
"inputTokens": input_tokens,
|
||||
"cachedInputTokens": 0,
|
||||
"outputTokens": output_tokens,
|
||||
"reasoningOutputTokens": 0,
|
||||
"totalTokens": input_tokens + output_tokens,
|
||||
}
|
||||
|
||||
def test_turn_usage_sets_anchor(self):
|
||||
from agent.codex_runtime import _record_codex_app_server_usage
|
||||
|
||||
messages = _history_with_images(10)
|
||||
agent = self._agent()
|
||||
|
||||
_record_codex_app_server_usage(
|
||||
agent, self._turn(self._usage()), messages=messages
|
||||
)
|
||||
|
||||
anchor = agent._usage_anchor
|
||||
assert anchor is not None
|
||||
assert anchor["prompt_tokens"] == 12_000
|
||||
assert anchor["base_count"] == len(messages)
|
||||
|
||||
# The next turn's preflight estimate anchors on provider truth plus
|
||||
# only the appended delta — not the flat-1500-per-image heuristic.
|
||||
messages.append(_msg("user", "follow-up"))
|
||||
got = _preflight_request_tokens(agent, messages, "")
|
||||
assert 12_000 < got < 12_200
|
||||
|
||||
def test_usage_less_turn_keeps_previous_anchor(self):
|
||||
from agent.codex_runtime import _record_codex_app_server_usage
|
||||
|
||||
messages = _history_with_images(2)
|
||||
prior = capture_usage_anchor(9_000, 50, messages)
|
||||
agent = self._agent(anchor=prior)
|
||||
|
||||
_record_codex_app_server_usage(agent, self._turn(None), messages=messages)
|
||||
|
||||
assert agent._usage_anchor is prior
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(pytest.main([__file__, "-v"]))
|
||||
|
||||
Reference in New Issue
Block a user