Files
hermes-agent/tests/agent/test_native_preflight_estimate.py
T
686f6c61 e6b4f3750b fix(agent): count native Responses preflight against pruned wire
Automatic preflight used the full durable transcript even when the
Codex Responses request would prune around a native compaction
checkpoint. That false-triggered a 600s local summary against history
the main request never sent. Estimate the converted, checkpoint-pruned
payload when native compaction is eligible, and keep the generic
estimate as the conservative fallback.
2026-08-27 18:56:21 -07:00

101 lines
3.5 KiB
Python

"""Native Responses preflight must count the checkpoint-pruned wire (#96155)."""
from types import SimpleNamespace
from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens
from agent.model_metadata import estimate_request_tokens_rough
from agent.turn_context import _preflight_request_tokens
def _codex_agent(**over):
agent = SimpleNamespace(
api_mode="codex_responses",
provider="openai-codex",
model="gpt-5.6",
base_url="https://chatgpt.com/backend-api/codex",
_base_url_hostname="chatgpt.com",
_base_url_lower="https://chatgpt.com/backend-api/codex",
codex_responses_native_compaction=True,
compression_enabled=True,
_codex_reasoning_replay_enabled=True,
context_compressor=SimpleNamespace(threshold_tokens=765_000),
tools=None,
)
for key, value in over.items():
setattr(agent, key, value)
return agent
def _history_with_checkpoint():
# Pre-checkpoint assistant/tool rows are dropped from the wire; user
# asks are retained. A durable estimate that counts those dropped rows
# is what falsely tripped local compression in #96155.
pre = []
for i in range(30):
pre.append({"role": "user", "content": f"ask {i}"})
pre.append({"role": "assistant", "content": "working " + ("tool output " * 400)})
pre.append(
{
"role": "tool",
"content": "result " + ("payload " * 400),
"tool_call_id": f"call-{i}",
}
)
checkpoint_turn = {
"role": "assistant",
"content": "checkpointed turn",
"codex_reasoning_items": [
{
"type": "compaction",
"encrypted_content": "blob",
"_issuer_kind": "codex_backend",
}
],
}
tail = [{"role": "user", "content": "follow-up after checkpoint"}]
return pre + [checkpoint_turn] + tail
def test_returns_none_when_api_mode_is_not_codex_responses():
agent = _codex_agent(api_mode="chat_completions")
assert estimate_native_responses_preflight_tokens(agent, _history_with_checkpoint()) is None
def test_returns_none_when_native_compaction_is_disabled():
agent = _codex_agent(codex_responses_native_compaction=False)
assert estimate_native_responses_preflight_tokens(agent, _history_with_checkpoint()) is None
def test_returns_none_when_model_is_outside_gpt56_family():
agent = _codex_agent(model="gpt-5.2")
assert estimate_native_responses_preflight_tokens(agent, _history_with_checkpoint()) is None
def test_pruned_estimate_is_far_below_durable_transcript():
agent = _codex_agent()
messages = _history_with_checkpoint()
generic = estimate_request_tokens_rough(messages)
native = estimate_native_responses_preflight_tokens(agent, messages)
assert native is not None
assert generic > native * 2
assert native < 8_000
def test_preflight_wrapper_uses_pruned_estimate_when_eligible():
agent = _codex_agent()
messages = _history_with_checkpoint()
native = estimate_native_responses_preflight_tokens(agent, messages)
wrapped = _preflight_request_tokens(agent, messages, "")
assert native is not None
assert wrapped == native
def test_preflight_wrapper_falls_back_to_generic_when_ineligible():
agent = _codex_agent(api_mode="chat_completions")
messages = _history_with_checkpoint()
generic = estimate_request_tokens_rough(messages)
assert _preflight_request_tokens(agent, messages, "") == generic