fix(compression): anchor codex app-server preflight on reported usage

The codex_app_server runtime bypasses the conversation loop, so the
usage-anchored context accounting captured there never ran: agent._usage_anchor
stayed None forever. Every preflight estimate therefore fell back to the rough
mirror-transcript heuristic, which is deliberately never compacted on this
runtime (_record_codex_app_server_compaction preserves the mirror), so it grows
monotonically while the real thread may be tiny or freshly compacted. With
compression.codex_app_server_auto=hermes that estimate alone tripped the
threshold and fired thread/compact/start on nearly every turn of a long-lived
session (#100381).

Mirror the main loop's post-response capture: _record_codex_app_server_usage
now snapshots the reported thread usage into agent._usage_anchor (base prompt
+ completion exactly as the provider counted them, with estimation confined to
messages appended since). A usage-less turn keeps the previous anchor, and the
post-compaction invalidation site is unchanged.
This commit is contained in:
liuhao1024
2026-09-01 23:26:50 +08:00
committed by Teknium
parent 1cd9106eff
commit a9e56c048a
3 changed files with 85 additions and 4 deletions
+12 -3
View File
@@ -81,9 +81,12 @@ def _queue_token_counts(agent, fail_msg: str, *fail_extra: Any, counts: Callable
logger.debug(fail_msg, agent.session_id, *fail_extra, exc)
def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any]:
"""Translate Codex app-server token usage into Hermes accounting. Prompt bucket = uncached + cached
input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call."""
input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call.
``messages`` (the transcript mirror) lets real usage anchor the next preflight: this runtime bypasses
the main loop's capture, and the mirror is never compacted natively, so without an anchor the rough
estimate grows monotonically and hermes-mode fires thread compaction on tiny threads (#100381)."""
agent.session_api_calls += 1
usage = getattr(turn, "token_usage_last", None)
compressor = getattr(agent, "context_compressor", None)
@@ -117,6 +120,12 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
compressor.context_length = context_window
except Exception:
logger.debug("codex app-server usage update failed", exc_info=True)
if isinstance(messages, list):
from agent.model_metadata import capture_usage_anchor
anchor = capture_usage_anchor(prompt_tokens, canonical_usage.output_tokens, messages)
if anchor is not None:
agent._usage_anchor = anchor
for key, value in usage_dict.items():
setattr(agent, f"session_{key}", getattr(agent, f"session_{key}") + value)
cost_result = estimate_usage_cost(
@@ -429,7 +438,7 @@ def _finish_codex_turn(agent, turn, messages: List[Dict[str, Any]], *, original_
# run_conversation() already bumped _turns_since_memory / _user_turn_count; only _iters_since_skill is ours.
agent._iters_since_skill = getattr(agent, "_iters_since_skill", 0) + turn.tool_iterations
_record_codex_app_server_compaction(agent, turn)
usage_result = _record_codex_app_server_usage(agent, turn)
usage_result = _record_codex_app_server_usage(agent, turn, messages=messages)
# Skill nudge check AFTER iters were incremented (same as chat_completions).
should_review_skills = (0 < agent._skill_nudge_interval <= agent._iters_since_skill
and "skill_manage" in agent.valid_tool_names)
+1 -1
View File
@@ -3812,7 +3812,7 @@ def _compress_context_via_codex_app_server(
# An empty usage report must consume the pending verdict, not leave deferral
# armed until a later turn; minimal test engines may lack update_from_response.
if hasattr(agent.context_compressor, "update_from_response"):
_record_codex_app_server_usage(agent, result)
_record_codex_app_server_usage(agent, result, messages=messages)
_reset_read_dedup_caches(task_id, skills=False)
logger.info(
"codex app-server compaction done: session=%s thread=%s turn=%s", _sid,
+72
View File
@@ -208,5 +208,77 @@ class TestCompressionTriggerUsesAnchor:
assert anchored is not None and anchored < threshold
class TestCodexAppServerAnchor:
"""The codex_app_server runtime bypasses the conversation loop, so its
usage recording is the only site that can maintain agent._usage_anchor.
Without it, hermes-mode preflight falls back to the rough mirror-transcript
heuristic, which grows monotonically (native compaction preserves the
mirror) and fires thread compaction on tiny real threads (#100381)."""
def _agent(self, anchor=None):
return SimpleNamespace(
_usage_anchor=anchor,
session_api_calls=0,
session_prompt_tokens=0,
session_completion_tokens=0,
session_total_tokens=0,
session_input_tokens=0,
session_output_tokens=0,
session_cache_read_tokens=0,
session_cache_write_tokens=0,
session_reasoning_tokens=0,
context_compressor=None,
event_callback=None,
_session_db=None,
model="codex-test-model",
provider="openai",
base_url=None,
)
def _turn(self, usage):
return SimpleNamespace(token_usage_last=usage, model_context_window=None)
def _usage(self, input_tokens=12_000, output_tokens=100):
return {
"inputTokens": input_tokens,
"cachedInputTokens": 0,
"outputTokens": output_tokens,
"reasoningOutputTokens": 0,
"totalTokens": input_tokens + output_tokens,
}
def test_turn_usage_sets_anchor(self):
from agent.codex_runtime import _record_codex_app_server_usage
messages = _history_with_images(10)
agent = self._agent()
_record_codex_app_server_usage(
agent, self._turn(self._usage()), messages=messages
)
anchor = agent._usage_anchor
assert anchor is not None
assert anchor["prompt_tokens"] == 12_000
assert anchor["base_count"] == len(messages)
# The next turn's preflight estimate anchors on provider truth plus
# only the appended delta — not the flat-1500-per-image heuristic.
messages.append(_msg("user", "follow-up"))
got = _preflight_request_tokens(agent, messages, "")
assert 12_000 < got < 12_200
def test_usage_less_turn_keeps_previous_anchor(self):
from agent.codex_runtime import _record_codex_app_server_usage
messages = _history_with_images(2)
prior = capture_usage_anchor(9_000, 50, messages)
agent = self._agent(anchor=prior)
_record_codex_app_server_usage(agent, self._turn(None), messages=messages)
assert agent._usage_anchor is prior
if __name__ == "__main__":
raise SystemExit(pytest.main([__file__, "-v"]))