Files
hermes-agent/tests/agent/test_codex_ttfb_watchdog.py
T
AgentLinker cac9db7caf fix(codex): retired stream requests must not synthesize a completed response
When a watchdog (TTFB / stream-idle / stale-call) force-closes a Codex
Responses request, the worker thread can still be draining SSE frames.
`_consume_codex_event_stream` returns `status=terminal_status`, which defaults
to `"completed"`, and its only truncation guard is
`if not saw_terminal and not output`. A mid-stream kill leaves
`saw_terminal=False` but `output`/text non-empty, so the partial text came back
as a `finish_reason=stop` response and got persisted as a finished assistant
turn — a long reply just stops mid-sentence with no error surfaced.

Observed as a long generation dying at `1. Create (6/6)` and never emitting its
end marker, with the truncated text already stored in state.db.

Fix: publish a per-request retirement token so the worker can tell it has been
retired.

- `agent/chat_completion_helpers.py`: `interruptible_api_call` installs
  `agent._active_codex_stream_request_token` before handing off to the worker
  (codex_responses only) and clears it at all four kill sites plus the worker's
  own `finally`. Retirement is cleared BEFORE `_close_request_client_once`,
  which can raise — every other call site wraps it in try/except, and a leaked
  token would let a later worker mistake itself for the owning attempt. The
  request-local `_codex_request_retired` mirror also swallows the transport
  error our own force-close causes, so the worker's local error cannot replace
  the watchdog's retryable TimeoutError (same split as `_request_cancelled`).
- `agent/codex_runtime.py`: `run_codex_stream` captures the token and raises
  `TimeoutError` from `interrupt_check` when it no longer owns the request —
  raising rather than breaking, because a break returns the partial `final`.
  The four stream callbacks also drop post-retirement frames so an abandoned
  attempt cannot stream tokens into the live turn's bubble (the gateway caches
  AIAgent instances per session).

`TimeoutError` is not an httpx / ConnectionError / RuntimeError subclass, so it
passes through the four `except` clauses around the consume call untouched.
No token installed (auxiliary callers such as `handle_max_iterations` drive
`_run_codex_stream` directly) means every check passes — behavior unchanged.

Tests: 5 new cases. Retirement raises instead of returning partial output;
post-retirement deltas stop reaching callbacks; the no-token path keeps its
existing terminal-frame tolerance; the watchdog installs and clears the token;
non-codex api_modes install nothing. A `_LazyCreateStream` helper is needed
because `_FakeCreateStream` materializes events in __init__, which would run
the retirement side effect before consumption starts.
2026-09-01 22:14:06 -07:00

435 lines
15 KiB
Python

"""Regression tests for the Codex time-to-first-byte (TTFB) watchdog.
The chatgpt.com/backend-api/codex endpoint has an intermittent failure mode
where it accepts the connection but never emits a single stream event. The
watchdog in ``interruptible_api_call`` kills such a connection at a short TTFB
cutoff (instead of waiting out the much longer wall-clock stale timeout) so the
retry loop can reconnect promptly. Once any stream event arrives, the TTFB
watchdog is satisfied and a separate idle watchdog handles streams that stop
emitting SSE events.
The "bytes flowing" signal is ``agent._codex_stream_last_event_ts``, set on
*any* event by ``codex_runtime.run_codex_stream`` — so reasoning-only or
tool-call-only turns (which emit no output-text deltas) are not mistaken for a
stall.
"""
from __future__ import annotations
import sys
import time
import types
from types import SimpleNamespace
import pytest
# Stub optional heavy imports so run_agent imports cleanly in isolation.
sys.modules.setdefault("fire", types.SimpleNamespace(Fire=lambda *a, **k: None))
sys.modules.setdefault("firecrawl", types.SimpleNamespace(Firecrawl=object))
sys.modules.setdefault("fal_client", types.SimpleNamespace())
def _make_codex_agent(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
(tmp_path / ".env").write_text("", encoding="utf-8")
(tmp_path / "config.yaml").write_text("{}\n", encoding="utf-8")
from run_agent import AIAgent
agent = AIAgent(
model="gpt-5.5",
provider="openai-codex",
api_key="sk-dummy",
base_url="https://chatgpt.com/backend-api/codex",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
platform="cli",
)
# The watchdog is gated on the codex_responses api_mode; assert/force it so
# the test is robust to detection-logic changes elsewhere.
agent.api_mode = "codex_responses"
monkeypatch.setattr(agent, "_emit_status", lambda *a, **k: None)
# Keep the wall-clock stale timeout high so any early kill is unambiguously
# the TTFB path, not the stale-call path.
monkeypatch.setattr(
agent, "_compute_non_stream_stale_timeout", lambda *a, **k: 60.0
)
return agent
def test_ttfb_includes_silent_hang_hint_for_gpt_5_5(tmp_path, monkeypatch):
"""The no-first-byte watchdog should surface the same actionable hint as the
stale-call timeout path when the model matches the silent-hang heuristic."""
from agent import chat_completion_helpers as h
agent = _make_codex_agent(tmp_path, monkeypatch)
monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "0.4")
closes: list = []
statuses: list[str] = []
dummy_client = SimpleNamespace()
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
monkeypatch.setattr(agent, "_buffer_status", lambda msg: statuses.append(msg))
monkeypatch.setattr(agent, "_emit_status", lambda msg: statuses.append(msg))
monkeypatch.setattr(
agent, "_abort_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
monkeypatch.setattr(
agent, "_close_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
stop = {"flag": False}
def fake_hang(api_kwargs, client=None, on_first_delta=None):
deadline = time.time() + 30
while time.time() < deadline and not stop["flag"] and not agent._interrupt_requested:
time.sleep(0.02)
raise RuntimeError("connection closed")
monkeypatch.setattr(agent, "_run_codex_stream", fake_hang)
try:
with pytest.raises(TimeoutError) as excinfo:
h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": "hi"})
message = str(excinfo.value)
assert "gpt-5.4" in message
assert "gpt-5.3-codex" in message
assert "gpt-5.4-codex" in message
assert "codex_ttfb_kill" in closes
assert statuses, "expected a user-facing watchdog status"
assert any("gpt-5.4" in s and "gpt-5.3-codex" in s for s in statuses)
finally:
stop["flag"] = True
def test_ttfb_installs_and_retires_the_codex_request_token(tmp_path, monkeypatch):
"""The watchdog must publish a per-request token and clear it on the kill.
``run_codex_stream`` reads ``agent._active_codex_stream_request_token`` to
tell whether it is still the owning attempt. Without an install here the
whole retirement guard would be inert, and without the clear on kill a
retired worker would keep normalizing partial deltas into a "completed"
response.
The worker also unwinds with its own local error after the force-close;
that error must not replace the watchdog's retryable ``TimeoutError``.
"""
from agent import chat_completion_helpers as h
agent = _make_codex_agent(tmp_path, monkeypatch)
monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "1")
closes: list = []
seen = {"token_while_running": None}
dummy_client = SimpleNamespace()
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
monkeypatch.setattr(
agent,
"_abort_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
monkeypatch.setattr(
agent,
"_close_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
def fake_stream(api_kwargs, client=None, on_first_delta=None):
seen["token_while_running"] = getattr(
agent, "_active_codex_stream_request_token", None
)
deadline = time.time() + 30
while time.time() < deadline:
if getattr(agent, "_active_codex_stream_request_token", None) is None:
# Retired by the watchdog — mimic the transport unwinding.
raise RuntimeError("retired worker stream ended without terminal")
time.sleep(0.02)
raise RuntimeError("test timed out waiting for retirement")
monkeypatch.setattr(agent, "_run_codex_stream", fake_stream)
with pytest.raises(TimeoutError) as excinfo:
h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": "hi"})
assert seen["token_while_running"] is not None, (
"interruptible_api_call must install a request token before the worker runs"
)
assert "TTFB" in str(excinfo.value)
assert "retired worker" not in str(excinfo.value)
assert "codex_ttfb_kill" in closes
assert getattr(agent, "_active_codex_stream_request_token", None) is None
def test_non_codex_api_mode_installs_no_request_token(tmp_path, monkeypatch):
"""The token is codex_responses-only — other api_modes stay untouched."""
from agent import chat_completion_helpers as h
agent = _make_codex_agent(tmp_path, monkeypatch)
agent.api_mode = "chat_completions"
seen = {"token": "unset"}
dummy_client = SimpleNamespace()
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
def fake_dispatch(_agent, _api_kwargs, *, make_client):
make_client("test")
seen["token"] = getattr(
_agent, "_active_codex_stream_request_token", "absent"
)
return SimpleNamespace(choices=[])
monkeypatch.setattr(h, "_dispatch_nonstreaming_api_request", fake_dispatch)
h.interruptible_api_call(agent, {"model": "gpt-5.5", "messages": []})
assert seen["token"] in (None, "absent")
def test_ttfb_does_not_kill_when_events_flow(tmp_path, monkeypatch):
"""Once a stream event has arrived, a generation that runs past the TTFB
cutoff is NOT killed by the watchdog — it completes normally."""
from agent import chat_completion_helpers as h
agent = _make_codex_agent(tmp_path, monkeypatch)
monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "0.4")
closes: list = []
dummy_client = SimpleNamespace()
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
monkeypatch.setattr(
agent, "_abort_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
monkeypatch.setattr(
agent, "_close_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
sentinel = SimpleNamespace(ok=True)
def fake_stream(api_kwargs, client=None, on_first_delta=None):
# Bytes flowing: mark stream activity right away, then keep generating
# past the 0.4s TTFB cutoff before returning a real response.
agent._codex_stream_last_event_ts = time.time()
if on_first_delta:
on_first_delta()
time.sleep(0.9)
return sentinel
monkeypatch.setattr(agent, "_run_codex_stream", fake_stream)
resp = h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": "hi"})
assert resp is sentinel
assert "codex_ttfb_kill" not in closes
@pytest.mark.parametrize(
"stale_timeout",
[float("inf"), float("-inf"), float("nan")],
)
def test_wait_notice_omits_reconnect_when_all_deadlines_are_non_finite(
stale_timeout,
):
"""A disabled watchdog must not be advertised as a future reconnect."""
from agent import chat_completion_helpers as h
recovery = h._codex_wait_notice_recovery(
stale_timeout=stale_timeout,
ttfb_enabled=False,
ttfb_timeout=float("nan"),
last_event_ts=None,
call_start=100.0,
idle_enabled=False,
idle_timeout=float("nan"),
elapsed=30.0,
)
assert recovery == ""
def test_moa_heartbeat_survives_infinite_stale_timeout(monkeypatch):
"""The full 100-poll MoA heartbeat must leave a healthy call running."""
from agent import chat_completion_helpers as h
notices: list[str] = []
response = SimpleNamespace(ok=True)
agent = SimpleNamespace(
platform="desktop",
api_mode="chat_completions",
provider="moa",
_consecutive_stale_streams=0,
_interrupt_requested=False,
_compute_non_stream_stale_timeout=lambda _kwargs: float("inf"),
_touch_activity=lambda _message: None,
_emit_wait_notice=notices.append,
)
class HeartbeatThread:
"""Keep the synthetic worker alive through one heartbeat."""
def __init__(self, *, target, daemon):
self._polls = 0
self._target = target
def start(self):
pass
def join(self, timeout=None):
pass
def is_alive(self):
self._polls += 1
if self._polls == 101:
self._target()
return False
return True
monkeypatch.setattr(h.threading, "Thread", HeartbeatThread)
monkeypatch.setattr(
h,
"_dispatch_nonstreaming_api_request",
lambda *_args, **_kwargs: response,
)
result = h.interruptible_api_call(agent, {"model": "openai-xai-wide"})
assert result is response
assert len(notices) == 1
assert "waiting on openai-xai-wide" in notices[0]
assert "auto-reconnect" not in notices[0]
def test_wait_notice_formatting_error_does_not_abort_request(monkeypatch):
"""Status construction is fail-open even if its formatter breaks."""
from agent import chat_completion_helpers as h
response = SimpleNamespace(ok=True)
agent = SimpleNamespace(
platform="desktop",
api_mode="chat_completions",
provider="moa",
_consecutive_stale_streams=0,
_interrupt_requested=False,
_compute_non_stream_stale_timeout=lambda _kwargs: float("inf"),
_touch_activity=lambda _message: None,
_emit_wait_notice=lambda _message: None,
)
class HeartbeatThread:
def __init__(self, *, target, daemon):
self._polls = 0
self._target = target
def start(self):
pass
def join(self, timeout=None):
pass
def is_alive(self):
self._polls += 1
if self._polls == 101:
self._target()
return False
return True
monkeypatch.setattr(h.threading, "Thread", HeartbeatThread)
monkeypatch.setattr(
h,
"_dispatch_nonstreaming_api_request",
lambda *_args, **_kwargs: response,
)
monkeypatch.setattr(
h,
"_codex_wait_notice_recovery",
lambda **_kwargs: (_ for _ in ()).throw(ValueError("bad display state")),
)
result = h.interruptible_api_call(agent, {"model": "openai-xai-wide"})
assert result is response
def test_large_codex_request_hard_ceiling_reclaims_silent_stall(tmp_path, monkeypatch):
"""#64507 regression: a large Codex request (TTFB watchdog disabled by the
size gate, stale floor *raised*) that never emits a single byte must still
be reclaimed at a finite hard ceiling — not hang for 13+ minutes while the
worker stays idle and the session shows as active.
Uses the real default TTFB threshold (120s) and asserts the request dies at
the hard ceiling regardless of the size-based TTFB disable.
"""
from agent import chat_completion_helpers as h
agent = _make_codex_agent(tmp_path, monkeypatch)
# Real default TTFB threshold (no HERMES_CODEX_TTFB_* override) → for a
# >10k-token request the no-byte TTFB watchdog is auto-disabled.
monkeypatch.setenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", "3")
closes: list = []
dummy_client = SimpleNamespace()
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
monkeypatch.setattr(
agent, "_abort_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
monkeypatch.setattr(
agent, "_close_request_openai_client",
lambda c, reason=None: closes.append(reason),
)
stop = {"flag": False}
def fake_hang(api_kwargs, client=None, on_first_delta=None):
# No event marker AND no event ever: the exact issue-64507 stall.
deadline = time.time() + 120
while time.time() < deadline and not stop["flag"] and not agent._interrupt_requested:
time.sleep(0.02)
raise RuntimeError("connection closed")
monkeypatch.setattr(agent, "_run_codex_stream", fake_hang)
large_input = "x" * 44_000 # ~11k estimated tokens → TTFB disabled, stale raised
t0 = time.time()
try:
with pytest.raises(TimeoutError) as excinfo:
h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": large_input})
elapsed = time.time() - t0
# Must die at the hard ceiling (3s), nowhere near the raised stale floor.
assert elapsed < 30, f"hard ceiling took {elapsed:.1f}s — stall not reclaimed"
assert "stale_call_kill" in closes, f"stale kill expected, got {closes}"
assert "timed out after" in str(excinfo.value)
assert "with no response" in str(excinfo.value)
finally:
stop["flag"] = True