From 4e548ce7a006210c2c176fccb01fe95242bf89ba Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 08:14:27 -0700 Subject: [PATCH] refactor(agent): compact turn-loop comments/docstrings to invariant statements (AST-identical) --- agent/conversation_loop.py | 3506 ++++++++++-------------------------- agent/turn_context.py | 634 ++----- agent/turn_finalizer.py | 374 ++-- agent/turn_retry_state.py | 60 +- 4 files changed, 1181 insertions(+), 3393 deletions(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 05b19e8b4e..6e27dc30b2 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -1,18 +1,9 @@ """The agent conversation loop — extracted from ``run_agent.AIAgent``. -This is the biggest single chunk pulled out of ``run_agent.py``: the -roughly 3,900-line :func:`run_conversation` body that drives one user -turn through the agent (model call, tool dispatch, retries, fallbacks, -compression, post-turn hooks, background memory/skill review nudges). - -The function takes the parent ``AIAgent`` instance as its first -argument (``agent``) and accesses its state via attribute lookup. -``_ra().AIAgent.run_conversation`` is now a thin forwarder. - -Symbols that production code or tests patch on ``run_agent`` directly -(``handle_function_call``, ``_set_interrupt``, ``OpenAI``, ...) are -resolved through :func:`_ra` so those patches keep working. -""" +``run_conversation(agent, ...)`` drives one user turn (model call, tool dispatch, +retries, fallbacks, compression, post-turn hooks). Symbols that callers patch on +``run_agent`` (``handle_function_call``, ``_set_interrupt``, ``OpenAI``) resolve via +``_ra`` so those patches keep working.""" from __future__ import annotations @@ -68,10 +59,8 @@ from agent.message_sanitization import ( _strip_non_ascii, serialized_messages_bytes, ) -# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py — kept local -# to avoid importing hermes_state at module load time (its module-level -# DEFAULT_DB_PATH = get_hermes_home() / "state.db" breaks tests that -# monkeypatch get_hermes_home to return a str). +# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py; kept local so importing +# hermes_state (module-level DEFAULT_DB_PATH) is not forced at load time. _STALE_MARKER_RE = re.compile(r"^\[[A-Za-z_][A-Za-z0-9_.-]*\]$") from agent.model_metadata import ( MINIMUM_CONTEXT_LENGTH, @@ -119,9 +108,8 @@ logger = logging.getLogger(__name__) _INTERRUPT_SCAFFOLD_MARKER = "[This response was interrupted by a user correction.]" -# One-time wrap-up notice appended when a wall-clock run budget crosses its -# 80% threshold (agent.run_budget_seconds / --run-budget). Mirrors the Codex -# CLI budget wrap-up template: stop new work, deliver from current state. +# One-time wrap-up notice appended when a wall-clock run budget crosses 80% +# (agent.run_budget_seconds / --run-budget): stop new work, deliver current state. RUN_BUDGET_WRAPUP_NOTICE = ( "[SYSTEM NOTICE — run time budget nearly exhausted] " "Run time budget nearly exhausted. Stop new discovery/verification work " @@ -138,19 +126,9 @@ def _midturn_request_pressure_tokens( ) -> int: """Token figure the mid-turn pre-API compression guard compares. - When the upcoming request is eligible for native Responses compaction the - transport will checkpoint-prune the payload before sending, so the generic - durable-history estimate overstates the wire by orders of magnitude on a - compacted session and fires a 600s local compression the main request - never needed (#96995). Mirror the turn-prologue preflight (#96644 / - #96155): use the pruned estimate when native eligibility is proven, the - generic message+tools figure otherwise. - - The native estimator adds the system prompt and tool schemas itself and - its converter skips system-role rows, so passing the assembled - ``api_messages`` (which carries the system row) alongside - ``effective_system`` counts the system prompt exactly once. - """ + Returns the pruned native-Responses estimate when native compaction eligibility is + proven (the generic estimate overstates the wire on compacted sessions, #96995), + else the generic message+tools figure. System prompt is counted exactly once.""" try: from agent.codex_responses_adapter import ( estimate_native_responses_preflight_tokens, @@ -178,16 +156,8 @@ def _midturn_request_pressure_tokens( def _review_input_budget_exhausted(agent: Any) -> bool: """True when a detached review fork has replayed its aggregate input budget. - Only forks carrying an explicit ``_review_input_token_budget`` (the - background-review path, #93057) are gated; every other agent returns - False and is unaffected. ``session_input_tokens`` accumulates from - provider usage after each response, so the check fires at the top of the - NEXT tool-loop iteration — the budget-crossing request is admitted and - completes (its tool writes land), then the loop stops before any further - provider call. This caps the review's total replayed input, complementing - the per-request bound provided by detached in-memory compaction and the - ``_REVIEW_MAX_ITERATIONS`` iteration cap. - """ + Only forks with an explicit ``_review_input_token_budget`` are gated (#93057). Fires + at the top of the NEXT iteration, so the budget-crossing request completes first.""" budget = getattr(agent, "_review_input_token_budget", None) if not isinstance(budget, int) or isinstance(budget, bool) or budget <= 0: return False @@ -198,16 +168,9 @@ def _review_input_budget_exhausted(agent: Any) -> bool: def _maybe_inject_run_budget_wrapup(agent: Any, messages: List[Dict[str, Any]]) -> bool: """Inject the one-time wall-clock wrap-up notice when past 80% of budget. - Cache-safe delivery: the notice is appended to the NEWEST ``role:"tool"`` - message (the same channel /steer uses) — no synthetic user message is - inserted mid-loop and no past context is rewritten, so role alternation - and the prompt-cache prefix survive. Latches ``_run_budget_wrapup_injected`` - only on a successful append, so a first iteration without tool results - retries on the next iteration. Returns True when the notice was injected. - - Dormant unless ``agent.run_budget_seconds`` is set AND the turn stamped - ``_run_budget_started_at`` (see ``turn_context.prepare_conversation_turn``). - """ + Appends to the NEWEST ``role:"tool"`` message (cache-safe, like /steer); latches + ``_run_budget_wrapup_injected`` only on a successful append. Returns True when + injected. Dormant unless ``run_budget_seconds`` + ``_run_budget_started_at`` set.""" budget = getattr(agent, "run_budget_seconds", None) if not budget: return False @@ -247,10 +210,8 @@ def _restore_user_after_reference_handoff( ) -> bool: """Re-append this turn's real user ask when compaction left only a handoff. - Returns True when a restore append happened. The caller has already - established that a reference-only handoff would drive the next model - call (#80622); this helper only decides whether a restorable ask exists. - """ + Returns True when a restore append happened; only decides whether a restorable + ask exists (#80622).""" if user_message is None: return False if isinstance(user_message, str): @@ -289,21 +250,15 @@ def _should_skip_model_call_for_reference_handoff( return True -# Fallback final_response for a turn ended by the sole-handoff skip (#80622). -# Deliberately NOT a replay of the last assistant text: finalize_turn's -# non-assistant-tail chokepoint (#43849) appends final_response as a fresh -# assistant row, so recovering the previous turn's prose here would duplicate -# it in the durable transcript AND re-deliver it to the user as if it were -# this turn's answer. A short status is honest and idempotent. +# Fallback final_response for the sole-handoff skip (#80622). Not a replay of the +# last assistant text: finalize_turn appends final_response as a fresh assistant row. _HANDOFF_SKIP_FINAL_RESPONSE = ( "Context was compacted. The previous response is complete — " "awaiting your next message." ) -# Terminal final_response for a turn ended because context compression hit its -# host progress-aware timeout while the request was still oversized (#98722, -# salvaged from #98741). Sending the unchanged request would only bounce off -# the provider's overflow error and re-enter compression in the same turn. +# Terminal final_response when compression hit its host timeout while the request +# was still oversized; resending would only bounce off the overflow error (#98722). _COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( "Context compression timed out without reducing this conversation. " "No messages were dropped. Start a fresh session with /new, or check " @@ -311,9 +266,8 @@ _COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( ) -# Stable prefix of the local interrupt status string emitted when a turn is -# cancelled while waiting on the provider. Surfaces (ACP, TUI) match on this -# to treat it as cancellation metadata rather than assistant prose. +# Stable prefix of the local interrupt status string; surfaces (ACP, TUI) match on +# it to treat the text as cancellation metadata rather than assistant prose. INTERRUPT_WAITING_FOR_MODEL_PREFIX = "Operation interrupted: waiting for model response (" @@ -326,11 +280,8 @@ def _should_rearm_compression_budget( ) -> bool: """Return True after a provider proves a completed compaction worked. - Rough estimates cannot safely rearm the anti-thrash budget: they can dip - below the threshold while the provider-visible prompt remains too large. - Require the completed-compaction latch plus a positive, normalized prompt - count below the threshold from the next successful provider response. - """ + Rough estimates cannot rearm the anti-thrash budget; require the completed- + compaction latch and a positive normalized prompt count below the threshold.""" return bool( compression_attempts and completed_compaction_pending @@ -339,15 +290,9 @@ def _should_rearm_compression_budget( ) -# Modules that indicate a deterministic local processing error when they -# appear in an exception traceback WITHOUT any API-call module. Used by the -# outer-loop error classifier to avoid retrying bugs that will fail -# identically every time (e.g. TypeError from passing list content into a -# regex helper). IMPORTANT: do NOT include "conversation_loop" or -# "run_agent" here — those are the container modules for the try/except -# itself, so every exception passes through them, which would make -# _hit_local always True and misclassify transient API/network errors as -# non-retryable local bugs. (#66267) +# Modules whose presence in a traceback (without any API-call module) marks a +# deterministic local bug not worth retrying. NEVER add "conversation_loop" or +# "run_agent": every exception passes through them; _hit_local would be True (#66267) _LOCAL_PROCESSING_MODULES = frozenset({ "agent_runtime_helpers", "message_content", @@ -358,33 +303,17 @@ _API_CALL_MODULES = frozenset({ "chat_completion_helpers", }) -# Maximum total outer-loop exceptions tolerated within one user turn before -# the loop gives up (#92450). The turn budget is unlimited by default -# (``max_iterations = sys.maxsize``), so the historical "near the limit" -# guard no longer stops a turn whose outer loop keeps raising: permanent -# failures spun at ~64 retries/s, pegged a core, and overwrote the rotated -# agent.log history (days of diagnostic context) within minutes. The inner -# retry/fallback machinery owns transient API recovery and terminates on its -# own; only exceptions that ESCAPE it reach this bound, so the cap can be -# small. Still scaled down by a tiny explicit ``max_iterations`` so a -# manually bounded budget keeps governing. +# Max outer-loop exceptions per user turn before giving up; only exceptions that +# ESCAPE the inner retry/fallback machinery count, so this can be small (#92450). _MAX_OUTER_LOOP_ERRORS = 8 def _is_interpreter_shutdown_error(exc: Exception) -> bool: """Check if *exc* is a fatal interpreter-shutdown failure. - During teardown, ``concurrent.futures`` refuses new work with - ``RuntimeError: cannot schedule new futures after interpreter shutdown`` - (or the shorter ``... after shutdown`` variant from a plain - ThreadPoolExecutor). Both are documented in #58720. - - Delegates to the shared predicate in ``tools.interpreter_shutdown`` - (same home as cron delivery and concurrent tool submission) so the - shutdown-race bug class has one text-matching site. Keeps the - RuntimeError type gate from the original (#93269): unlike the raw - predicate, a ValueError carrying similar text must not match here. - """ + Delegates to ``tools.interpreter_shutdown`` (one text-matching site for the + shutdown-race bug class) but keeps the RuntimeError type gate: a ValueError + carrying similar text must not match (#93269).""" if isinstance(exc, RuntimeError): from tools.interpreter_shutdown import interpreter_shutting_down @@ -395,13 +324,8 @@ def _is_interpreter_shutdown_error(exc: Exception) -> bool: def _moa_client_consumes_prepared_request(client: Any) -> bool: """True when ``client`` is the in-process MoA facade. - ``_moa_prepared_request`` is a private handshake with - ``MoAChatCompletions.create``, and only that facade exposes ``prepare()``. - Every other chat-completions object raises TypeError on the unexpected - keyword — including the native OpenAI client that credential rotation, - provider fallback and dead-connection cleanup rebuild from - ``_client_kwargs`` while ``agent.provider`` stays ``"moa"``. - """ + Only ``MoAChatCompletions`` exposes ``prepare()``; other clients raise TypeError on + ``_moa_prepared_request`` even while ``agent.provider`` stays ``"moa"``.""" completions = getattr(getattr(client, "chat", None), "completions", None) return callable(getattr(completions, "prepare", None)) @@ -419,11 +343,8 @@ def _join_truncated_parts(parts: List[str]) -> str: def _moa_reference_metrics_for_hook(agent: Any) -> Any: """Per-advisor metrics for post_api_request, or None off the MoA path. - MoA runs N advisor models before its aggregator and returns only the - aggregator's response, so an observability plugin sees one generation for - the whole fan-out. The advisor spend is already computed per slot (see - ``_RefAccounting``); this only carries it across the hook boundary. - """ + MoA returns only the aggregator response, so a plugin sees one generation for + the whole fan-out; this carries the per-slot advisor spend across the hook boundary.""" client = getattr(agent, "client", None) getter = getattr(client, "last_reference_metrics", None) if not callable(getter): @@ -437,40 +358,12 @@ def _moa_reference_metrics_for_hook(agent: Any) -> Any: def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text: str) -> None: """Append a provider-safe checkpoint and correction to the live turn. - Incomplete provider reasoning blocks are not valid replay items (Anthropic - signs them; Responses reasoning items require their following output). - Preserve only the *visible* response text, demoted to ordinary text, then - add the correction as a real user message. This keeps role alternation - valid and leaves every previously cached message byte-for-byte unchanged. - - INVARIANT — raw chain-of-thought must never be serialized into replayable - message content. Streamed reasoning is display-only state: it may be shown - live, but it does not re-enter the transcript as assistant (or user) text. - An assistant turn whose content inlines its own chain-of-thought reads to - Anthropic's output classifier as reasoning-injection/prefill jailbreak, - and because the poisoned checkpoint is persisted and replayed on every - subsequent call, the session dies permanently with deterministic - "Provider returned an empty response" storms that no retry, nudge, or - empty-recovery branch can escape (July 2026: four sessions bricked this - way; every reasoning-free checkpoint that week was untouched — same - mechanism as the ~/.hermes/prefill.json incident, 20/20 blocked with - assistant-exposed CoT vs 0/20 without). The interrupted reasoning was - incomplete by definition; the model regenerates it on the retried turn. - If a future path needs to preserve interrupted thinking, carry it in a - provider-gated reasoning *field*, never in content. - INVARIANT — the scaffolding is provider-replay text, not transcript text. - ``[This response was interrupted by a user correction.]`` and its - ``Visible response before the interruption:`` header exist so the MODEL - understands its own reply was cut off. They are not prose the user wrote - or the agent said. Persisting them into an assistant row's ``content`` or - ``api_content`` made the model treat the scaffold as *its own previous - reply*, echo it, and self-replicate ghost rows across turns (#81841). - Carry the scaffolded form only in the *user correction's* ``api_content`` - sidecar — never on the placeholder assistant row. When nothing was on - screen the placeholder is marked ``display_kind="hidden"`` (empty - content) so every transcript surface drops it, exactly like - compaction-reference rows. - """ + Keeps only the *visible* text (demoted to plain text) then adds the correction as a + real user message, so role alternation holds and cached messages stay byte-identical. + INVARIANT: raw chain-of-thought never enters replayable content — inlined CoT reads + as a prefill jailbreak and bricks the session with empty-response storms. + INVARIANT: the interruption scaffold is replay text, carried only in the user + correction's ``api_content``; an on-screen-empty placeholder is ``display_kind=hidden``.""" visible = agent._strip_think_blocks( getattr(agent, "_current_streamed_assistant_text", "") or "" ).strip() @@ -487,10 +380,9 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text f"{text}" ) - # The normal live tail is user or tool, so an assistant placeholder - # followed by the correction preserves strict alternation. If a transport - # already committed an assistant item, attribute the checkpoint inside the - # user correction instead of creating assistant→assistant. + # The live tail is normally user or tool, so an assistant placeholder + correction + # keeps strict alternation; if the tail is already assistant, fold the checkpoint + # into the user correction instead of creating assistant→assistant. if messages and messages[-1].get("role") == "assistant": # Transcript shows the user's own words; the provider replays the # scaffolded form so it still sees the interrupted context. @@ -499,26 +391,17 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text {"role": "user", "content": text, "api_content": correction}, ) else: - # Placeholder preserves role alternation only. Scaffold bytes must - # never land here — the API replay path substitutes api_content back - # into content, and a scaffold-as-assistant-reply is what the model - # then echoes (#81841 / incomplete #73146 else branch). + # Placeholder preserves role alternation only. Scaffold bytes must never land + # here: api_content is substituted back into content on replay (#81841). placeholder: Dict[str, Any] = { "role": "assistant", "content": visible or "", } if not visible: placeholder["display_kind"] = "hidden" - # Keep the transcript hidden and empty, but give the historical - # API projection a non-empty neutral assistant turn so the - # pre-call sanitizer (repair_empty_non_final_messages) does not - # re-heal this row on every later call (#88955). display_kind is - # stripped before sanitization, while api_content is projected - # back into content for historical assistant rows. Use the - # canonical neutral interruption placeholder, never - # _INTERRUPT_SCAFFOLD_MARKER: replaying the scaffold as assistant - # text made the model echo it and self-replicate ghost rows - # (#81841). + # Hidden row, but a non-empty neutral api_content so the pre-call + # sanitizer does not re-heal it every call (#88955). Never + # _INTERRUPT_SCAFFOLD_MARKER: as assistant text the model echoes it (#81841) from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER placeholder["api_content"] = _INTERRUPTED_PLACEHOLDER @@ -535,11 +418,8 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text def _is_copilot_provider(agent: Any) -> bool: """Delegate to ``AIAgent._is_copilot_provider`` (single owner of the check). - ``agent.provider`` is not always the normalized ``copilot`` slug — - ``/model`` and profile configs can leave the alias ``github-copilot`` (or - ``github``) in place, and a bare ``provider == "copilot"`` gate silently - skips credential recovery for those spellings. - """ + ``agent.provider`` may hold the aliases ``github-copilot`` / ``github``; a bare + ``provider == "copilot"`` gate would skip credential recovery for them.""" try: return bool(agent._is_copilot_provider()) except Exception: @@ -553,21 +433,9 @@ def _is_copilot_provider(agent: Any) -> bool: def _is_stale_copilot_credential_error(status_code: Optional[int], error_message: str) -> bool: """Detect a Copilot 400 that is really a STALE / DEGRADED credential. - Copilot surfaces a stale or degraded credential as an HTTP 400 rather than a - clean 401. Two body markers indicate this class: - - - ``model_not_available_for_integrator`` — the request reached the - restricted ``copilot-language-server`` integrator (the server's fallback - when it receives a raw OAuth token instead of an exchanged API token), - whose model allowlist omits enterprise-only models. - - ``model_not_supported`` / "the requested model is not supported" — the - cached bearer's Copilot entitlement rotated out from under a long-lived - process. - - Matched narrowly (status 400 AND a specific marker) so a genuinely wrong - model name — a real 400 — never triggers the single-shot re-exchange. The - caller enforces copilot-provider scoping and the single-shot guard. - """ + Matches status 400 AND ``model_not_available_for_integrator`` or + ``model_not_supported`` / "the requested model is not supported", so a wrong model + name never triggers the single-shot re-exchange. Caller enforces scoping/guard.""" lowered = (error_message or "").lower() is_400 = status_code == 400 or "error code: 400" in lowered if not is_400: @@ -655,15 +523,10 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str def _maybe_grow_local_window(agent: Any, compressor: Any, request_tokens: int) -> Optional[int]: - """Try growing the managed local model's context window before - compressing. Returns the new window when the ladder granted one, else - None (hold / at native / not a managed local session). + """Try growing the managed local model's context window before compressing. - The window ladder's design order: models launch at their zero-spill - window and grow toward native max as the session needs room; - compression is the move of last resort. Cheap for every non-local - provider: one lowercase compare, no imports. - """ + Returns the new window when the ladder granted one, else None (hold / at native / + not a managed local session). Cheap for non-local providers: one compare.""" provider = (getattr(agent, "provider", "") or "").strip().lower() if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"): return None @@ -688,10 +551,8 @@ def _maybe_grow_local_window(agent: Any, compressor: Any, def _ra(): - """Lazy reference to ``run_agent`` so callers can patch - ``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` / - ``run_agent.OpenAI`` and have those patches reach this code path. - """ + """Lazy ``run_agent`` reference so patches on ``run_agent.handle_function_call`` / + ``run_agent._set_interrupt`` / ``run_agent.OpenAI`` reach this code path.""" import run_agent return run_agent @@ -725,11 +586,8 @@ def _print_nous_entitlement_guidance(agent, capability: str) -> bool: def _system_prompt_for_hooks(api_kwargs: Any, request_messages: Any) -> Any: """System prompt as actually sent to the provider, for observability hooks. - Providers move it out of ``messages``: Anthropic Messages uses a separate - ``system`` kwarg (str or content-block list), the Responses/Codex API uses - top-level ``instructions``; Chat Completions keeps it as ``messages[0]``. - Returns None when the request carries no system prompt. - """ + Checks ``system`` (Anthropic), ``instructions`` (Responses/Codex), then + ``messages[0]``. Returns None when the request carries no system prompt.""" system_prompt = api_kwargs.get("system") if system_prompt is None: system_prompt = api_kwargs.get("instructions") @@ -764,20 +622,11 @@ def _billing_or_entitlement_message( provider_label = (provider or "").strip() or "the selected provider" model_label = (model or "").strip() or "the selected model" - # Anthropic Claude Pro/Max OAuth subscriptions surface exhaustion of the - # metered "extra usage" bucket as a hard 400 ("You're out of extra - # usage"). Point at the exact settings page and note the cycle-reset - # option, since the generic "add credits with that provider" line doesn't - # apply to a subscription — the user waits for the reset or switches to an - # API key. + # Anthropic Pro/Max OAuth surfaces exhaustion of the "extra usage" bucket as a hard + # 400; point at the settings page and cycle reset — "add credits" does not apply. if (provider or "").strip().lower() == "anthropic": - # ``unverified`` (ClassifiedError.billing_unverified, #82154): the - # "out of extra usage" 400 is ambiguous — Anthropic returns the same - # body when its server-side content filter rejects part of the request - # on a subscription OAuth token, so the message reliably misdirects - # diagnosis toward buying quota. Hedge the claim and name the other - # cause. A confirmed verdict (e.g. a real 402 or an API-key credit - # depletion) keeps the assertive wording. + # ``unverified`` (#82154): the "out of extra usage" 400 is also returned for a + # server-side content-filter rejection, so hedge and name the other cause. if unverified: lines = [ ( @@ -812,9 +661,8 @@ def _billing_or_entitlement_message( ] return "\n".join(lines) - # Provider-agnostic billing URL derivation (OpenAI, DeepSeek, xAI, Groq, - # OpenRouter, …) so every text surface — CLI, gateway messaging, TUI - # transcript — shows the same actionable link, not just OpenRouter. + # Provider-agnostic billing URL so every text surface (CLI, gateway, TUI) shows the + # same actionable link, not just OpenRouter. try: from agent.billing_links import build_billing_block @@ -861,9 +709,7 @@ def _billing_terminal_label(summary: str, unverified: bool) -> str: """Terminal-failure prefix for a billing-classified error. ``unverified`` (#82154): the Anthropic "out of extra usage" 400 can be a - content-filter rejection, so the terminal line must not assert billing - exhaustion as fact. - """ + content-filter rejection, so the line must not assert exhaustion as fact.""" if unverified: return ( "Provider reported usage/credit exhaustion (unverified — the same " @@ -885,10 +731,8 @@ def _billing_failure_result( ) -> dict: """Structured terminal result for a billing-classified failure. - Single construction point for the returned terminal response so the - label, guidance, structured block, and ambiguity flag stay consistent - across the non-retryable abort and max-retries paths (#82154). - """ + Single construction point so label, guidance, structured block and ambiguity flag + stay consistent across the non-retryable abort and max-retries paths (#82154).""" unverified = bool(getattr(classified, "billing_unverified", False)) if guidance is None: guidance = _billing_or_entitlement_message( @@ -909,9 +753,8 @@ def _billing_failure_result( "failed": True, "error": summary, "failure_reason": classified.reason.value, - # The classifier's own retry verdict — carried so UI surfaces - # (agent/error_surface.py) show Retry only when a re-run can differ, - # instead of re-deriving retryability from a second taxonomy. + # Classifier's own retry verdict so UI (agent/error_surface.py) shows Retry + # only when a re-run can differ, not re-derived from a second taxonomy. "failure_retryable": bool(classified.retryable), # The billing verdict may rest on an ambiguous body (#82154) — carry # that through the structured result, not just the prose. @@ -963,30 +806,9 @@ def _try_refresh_nous_paid_entitlement_credentials(agent) -> bool: def _restore_or_build_system_prompt(agent, system_message, conversation_history): """Restore the cached system prompt from the session DB or build it fresh. - Mutates ``agent._cached_system_prompt`` and persists a freshly-built - prompt back to the session DB on first build. Extracted from - ``run_conversation`` so the prefix-cache restore path can be tested in - isolation. - - Three-way state distinction for the stored row, surfaced via logs so - silent prefix-cache misses are visible in ``agent.log``: - - * ``missing`` — no session row yet (legitimate first turn). - * ``null`` — row exists, ``system_prompt`` column is NULL. - Legacy session predating system-prompt persistence, or a migration - leftover. Warns when ``conversation_history`` is non-empty. - * ``empty`` — row exists, ``system_prompt`` column is the empty - string. Indicates a previous-turn write that ran but stored - nothing (silent persistence bug). Always warns. - * ``present`` — row exists with a usable prompt → reused verbatim. - - Read or write failures against the session DB log at WARNING (not - DEBUG) so persistent issues (disk full, schema drift, lock contention) - surface without needing verbose mode. This used to be a debug-level - log that silently broke prefix-cache reuse on the gateway path - (which constructs a fresh ``AIAgent`` per turn and depends on this - DB roundtrip). - """ + Mutates ``agent._cached_system_prompt`` and persists a freshly-built prompt on first + build. Row states ``missing``/``null``/``empty``/``present`` are logged and DB + failures log at WARNING so silent prefix-cache misses show in ``agent.log``.""" stored_prompt = None stored_state = "missing" session_row = None @@ -1011,14 +833,9 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) ) if stored_prompt and _stored_prompt_matches_runtime(agent, stored_prompt): - # Bot Chat capability epoch: an eternal bot session must adopt - # user-initiated capability changes (skills/toolsets/MCP/SOUL/roster) - # on the next message, not at /new or compression. The stored prompt - # embeds a fingerprint of the capability surface; a mismatch against - # disk is a deliberate, once-per-change rebuild — the /model - # exception applied to capabilities. Prompts without the stamp - # (every non-Bot-Chat session) never take this branch, and the check - # fails closed to "reuse" so a probe failure can't burn cache. + # Bot Chat capability epoch: the stored prompt embeds a capability fingerprint; + # a mismatch is a deliberate once-per-change rebuild. Unstamped prompts never + # take this branch; probe failures fail closed to "reuse" so cache is kept. _bot_stale = False try: from tools.bot_mode_probe import ( @@ -1036,12 +853,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) pass _bot_stale = stored_prompt_capability_stale(stored_prompt, _home_for_epoch) if not _bot_stale and getattr(agent, "_bot_mode_protocol", True): - # Legacy upgrade: a Bot Chat whose prompt predates the epoch - # mechanism (no stamp, no protocol) gets ONE migration - # rebuild — otherwise pre-existing bots would never learn - # the messaging protocol. Title-gated so ordinary unstamped - # sessions (i.e. all of them) never take this path; the - # rebuilt prompt carries the stamp, so it cannot re-fire. + # Legacy upgrade: a Bot Chat prompt predating the epoch mechanism gets + # ONE title-gated migration rebuild; the stamped result cannot re-fire. _t = str(getattr(agent, "_session_title_hint", "") or "").strip() if not _t and agent._session_db and agent.session_id: try: @@ -1060,10 +873,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) agent.session_id, ) agent._session_title_hint = "Bot Chat" - # The skills index inside the prompt comes from a two-layer cache - # (in-process LRU + disk snapshot) that doesn't watch the skills - # dir; a capability refresh must rebuild THROUGH it or a freshly - # installed skill stays invisible in the new prompt. + # The skills index cache (LRU + disk snapshot) does not watch the skills + # dir; a capability refresh must rebuild THROUGH it or new skills are lost. try: from agent.prompt_builder import clear_skills_system_prompt_cache @@ -1072,10 +883,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) pass agent._cached_system_prompt = agent._build_system_prompt(system_message) agent._bot_capability_refreshed = True - # Persist the refreshed prompt so the NEXT turn restores the new - # bytes verbatim — the cache break is once per capability change, - # never per turn. (on_session_start deliberately not re-fired: - # this is a continuation, not a new session.) + # Persist so the NEXT turn restores the new bytes verbatim (cache break is + # once per capability change). on_session_start not re-fired: continuation. if agent._session_db: try: agent._session_db.update_system_prompt( @@ -1092,9 +901,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # Continuing session — reuse the exact system prompt from the # previous turn so the Anthropic cache prefix matches. agent._cached_system_prompt = stored_prompt - # Same contract for tools[]: a fresh AIAgent for an existing session - # (gateway agent-cache eviction) re-probed every check_fn, so pin the - # array back to the order this session already sent (tools freeze). + # Same contract for tools[]: pin the array to the order this session already + # sent (tools freeze) instead of re-probing every check_fn on a fresh AIAgent. try: saved_tools = session_row.get("tool_names") if session_row else None if saved_tools: @@ -1103,23 +911,14 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) restore_agent_tool_prefix(agent, json.loads(saved_tools)) except Exception: logger.debug("tool prefix restore skipped", exc_info=True) - # Prompt-section callbacks are new-session-only. Recover their frozen - # bytes from the persisted full prompt so a later compression rebuild - # keeps them without evaluating plugin state in this resumed process. + # Prompt-section callbacks are new-session-only; recover their frozen bytes + # from the persisted prompt so a compression rebuild keeps them. from agent.system_prompt import restore_plugin_prompt_sections restore_plugin_prompt_sections(agent, stored_prompt) - # Reconstruct the cross-session-stable prefix for the early cache - # breakpoint. The static prefix is not persisted (only the full - # prompt is), so gateway surfaces that build a fresh AIAgent per - # turn would otherwise lose the two-block system layout after the - # first turn — flip-flopping the wire shape mid-conversation and - # silently degrading to the legacy single-breakpoint layout. - # - # ``reconstruct_static_prefix`` gates on ``_use_prompt_caching`` (so - # non-Anthropic routes skip the rebuild), applies the startswith - # safety gate (stored prompt bytes are never rewritten), and - # fails open to the legacy cache layout. + # The static prefix is not persisted; rebuild it for the early cache breakpoint + # or fresh-per-turn gateway agents fall back to the single-breakpoint layout. + # reconstruct_static_prefix gates on _use_prompt_caching, fails open to legacy. from agent.system_prompt import reconstruct_static_prefix reconstruct_static_prefix(agent, system_message=system_message) @@ -1135,10 +934,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) ) if conversation_history and stored_state in ("null", "empty"): - # Continuing session whose stored prompt is unusable. The - # previous turn's write either never happened or wrote an empty - # string — either way every turn now rebuilds and the prefix - # cache misses every time. + # Continuing session with an unusable stored prompt: every turn now rebuilds + # and the prefix cache misses every time. logger.warning( "Stored system prompt for session %s is %s; rebuilding " "from scratch this turn. Prefix cache will miss until " @@ -1151,9 +948,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # prompt) — build from scratch. agent._cached_system_prompt = agent._build_system_prompt(system_message) - # Plugin hook: on_session_start — fired once when a brand-new - # session is created (not on continuation). Plugins can use this - # to initialise session-scoped state (e.g. warm a memory cache). + # Plugin hook: on_session_start — fired once for a brand-new session, not on + # continuation. try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook _invoke_hook( @@ -1165,12 +961,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception as exc: logger.warning("on_session_start hook failed: %s", exc) - # Cold-start credits seed (L3) — fallback for the first-turn path. The TUI/ - # desktop build seeds at session OPEN (see seed_credits_at_session_start in - # tui_gateway), so this call is usually a no-op there (idempotent: skips when - # _credits_state already exists). For the plain CLI / any path that didn't seed - # at build, it primes credits state from /api/oauth/account (or a fixture) on the - # first turn so depletion / usage-band warnings fire. Fail-open inside the helper. + # Cold-start credits seed (L3) fallback for the first-turn path; TUI/desktop seed at + # session open, so this is idempotent (skips when _credits_state exists). Fail-open. try: from agent.credits_tracker import seed_credits_at_session_start @@ -1178,10 +970,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception: logger.debug("cold-start credits seed failed (fail-open)", exc_info=True) - # Persist the system prompt snapshot in SQLite. Failure here used - # to log at DEBUG, which silently broke prefix-cache reuse on the - # gateway path (fresh AIAgent per turn → reads from this row every - # subsequent turn). + # Persist the system prompt snapshot; the gateway path (fresh AIAgent per turn) + # reads this row every turn, so a failure here breaks prefix-cache reuse. if agent._session_db: try: agent._session_db.update_system_prompt(agent.session_id, agent._cached_system_prompt) @@ -1203,12 +993,8 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: def line_value(label: str) -> str: """Last matching line wins. - Safe ONLY for fields emitted in the volatile tier at the very END of - the prompt (Model / Provider / Platform). User-supplied project - context (AGENTS.md / CLAUDE.md / .cursorrules) is embedded in the - middle context tier, so a last-match scan lets project prose shadow - any field emitted EARLIER — see ``host_info_value``. - """ + Safe ONLY for fields in the volatile tier at the END of the prompt; embedded + project context could shadow earlier fields — see ``host_info_value``.""" prefix = f"{label}:" value = "" for line in prompt.splitlines(): @@ -1219,19 +1005,8 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: def host_info_value(label: str) -> str: """Read a field from the prompt's own host-info block. - The host-info block (``build_environment_hints``) sits in the STABLE - tier, ahead of the embedded project context files. A bare scan of the - whole prompt would therefore match a user's ``AGENTS.md`` that merely - contains a line starting with the same label, comparing runtime state - against project prose. That mismatch never clears, so the check would - reject the stored prompt on EVERY turn — rebuilding the system prompt - each message and destroying the prefix cache for the whole session, - which is far worse than the staleness this function guards against. - - Anchor on the ``User home directory:`` line that immediately precedes - the working-directory line in that block, and take the FIRST such - occurrence, so only Hermes' own emitted block can satisfy the read. - """ + Anchors on the FIRST ``User home directory:`` line so a user's ``AGENTS.md`` row + cannot match; a false mismatch would rebuild the prompt every turn.""" prefix = f"{label}:" lines = prompt.splitlines() for idx, line in enumerate(lines): @@ -1252,19 +1027,15 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: if stored_provider and current_provider and stored_provider != current_provider: return False - # Detect cwd drift: if the stored prompt was built in a different working - # directory, reuse would silently inject a stale path into the prefix cache. - # Compare against resolve_agent_cwd() — the SAME resolver used to build the - # prompt — so gateway/TUI sessions that set TERMINAL_CWD are not falsely - # rejected (they would always differ from the launch dir's os.getcwd()). + # cwd drift check. Compare against resolve_agent_cwd() — the SAME resolver used to + # build the prompt — so TERMINAL_CWD sessions are not falsely rejected. stored_cwd = host_info_value("Current working directory") if stored_cwd: if stored_cwd != str(resolve_agent_cwd()): return False - # Detect runtime-surface drift: the stored prompt records which platform it - # was built for (e.g. "desktop" vs "cli"). Reusing a desktop-built prompt on - # a terminal session (or vice versa) would inject the wrong runtime hints. + # Runtime-surface drift: reusing a desktop-built prompt on a terminal session (or + # vice versa) would inject the wrong runtime hints. stored_platform = line_value("Platform") current_platform = str(getattr(agent, "platform", "") or "").strip() if stored_platform and current_platform and stored_platform != current_platform: @@ -1273,12 +1044,9 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: return True -# The three _get_continuation_prompt variants below, in named-constant form -# so agent.context_compressor's _is_synthetic_compression_user_turn can -# recognize them by content after a crash/interrupt persists one mid-list — -# these rows carry no durable role beyond driving the retry, and SessionDB -# projection strips the _length_continuation_nudge metadata tag that marks -# them in live memory (see agent/context_compressor.py). +# Named constants for the _get_continuation_prompt variants so +# _is_synthetic_compression_user_turn can recognize them by content after a crash +# persists one; SessionDB projection strips the _length_continuation_nudge tag. _LENGTH_CONTINUATION_NETWORK_STUB = ( "[System: The previous response was cut off by a " "network error mid-stream. Continue exactly where " @@ -1290,9 +1058,8 @@ _LENGTH_CONTINUATION_OUTPUT_LIMIT = ( "length limit. Continue exactly where you left off. Do not " "restart or repeat prior text. Finish the answer directly.]" ) -# The dropped-tools variant interpolates the tool name list right after this -# prefix, so it can't be exact-matched — this stable prefix is what -# _is_synthetic_compression_user_turn checks with str.startswith instead. +# The dropped-tools variant interpolates tool names, so +# _is_synthetic_compression_user_turn matches this prefix with str.startswith. _LENGTH_CONTINUATION_DROPPED_TOOLS_PREFIX = "[System: Your previous tool call " @@ -1318,13 +1085,8 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List return _LENGTH_CONTINUATION_OUTPUT_LIMIT -# Continuation nudge for Codex/Responses turns that came back with only -# internal reasoning (no visible content, no tool calls). When the interim -# assistant message also carries no encrypted reasoning items and no -# replayable message items, _chat_messages_to_responses_input emits nothing -# for it — a bare retry would be byte-identical to the request that just -# failed, so the model (observed: grok-4.20 on xai-oauth) deterministically -# repeats the reasoning-only response until the retry budget is exhausted. +# Nudge for Codex/Responses turns that returned only internal reasoning: a bare retry +# would be byte-identical (nothing replayable emitted), so the model repeats it. _CODEX_INCOMPLETE_NUDGE = ( "[System: Your previous response contained only internal reasoning and " "never produced a visible answer or tool call. Do not keep thinking. " @@ -1333,29 +1095,23 @@ _CODEX_INCOMPLETE_NUDGE = ( ) -# Re-prompt sent after a Codex/Responses turn ends with an acknowledgment-only -# reply (no tool calls, no final answer) — named so -# agent.context_compressor's _is_synthetic_compression_user_turn can -# recognize it by content the same way it recognizes _CODEX_INCOMPLETE_NUDGE. +# Re-prompt after an acknowledgment-only Codex/Responses reply; named so +# _is_synthetic_compression_user_turn can recognize it like _CODEX_INCOMPLETE_NUDGE. _CODEX_ACK_CONTINUATION_NUDGE = ( "[System: Continue now. Execute the required tool calls and only " "send your final answer after completing the task.]" ) -# Re-prompt sent when a provider returns finish_reason="tool_calls" with an -# empty tool_calls array (dropped-tool-call recovery, see the retry loop -# below). Named for the same reason as _CODEX_ACK_CONTINUATION_NUDGE — this -# pair is only stripped from the durable transcript once the turn reaches -# finalization; an interrupt/crash mid-retry can still persist it. +# Re-prompt for finish_reason="tool_calls" with empty tool_calls. Named like +# _CODEX_ACK_CONTINUATION_NUDGE: an interrupt mid-retry can persist it. _DROPPED_TOOLCALL_NUDGE_CONTENT = ( "Your previous turn indicated a tool call but none was " "included. Do not narrate a plan or restate intent — issue " "the actual tool call now to continue the task." ) -# Re-prompt sent when the model returns an empty response after executing tool -# calls (#9400). Named for the same reason as the nudges above — its -# _empty_recovery_synthetic metadata flag doesn't survive SessionDB projection. +# Re-prompt for an empty response after tool calls (#9400). Named because its +# _empty_recovery_synthetic metadata flag does not survive SessionDB projection. _EMPTY_TOOL_RESPONSE_NUDGE = ( "You just executed tool calls but returned an " "empty response. Please process the tool " @@ -1363,37 +1119,21 @@ _EMPTY_TOOL_RESPONSE_NUDGE = ( ) -# Shared recovery hint appended to every content-policy refusal message. Both -# the HTTP-200 refusal path (``finish_reason=content_filter``) and the -# exception path (a provider moderation error classified as -# ``content_policy_blocked``) end with the same actionable next steps, so they -# share one trailer to keep the guidance from drifting between the two sites. +# Shared recovery trailer for both content-policy refusal paths (HTTP-200 +# content_filter and the content_policy_blocked exception) so guidance cannot drift. _CONTENT_POLICY_RECOVERY_HINT = ( "Try rephrasing the request, narrowing the context, or " "adding a fallback provider with `hermes fallback add`." ) -# Memo for the send-path tool-call argument canonicalization inside -# run_conversation(). That pass re-canonicalizes the arguments string of -# EVERY historical tool call on EVERY API-call iteration (quadratic in -# session tool-call count), and the api_messages copies share the exact -# argument string objects with the persisted history, so the same strings -# come through unchanged iteration after iteration. -# -# Soundness: canonicalization is a pure, deterministic function of the -# input string (fixed separators, sort_keys=True), so a value-keyed memo -# is exact — equal inputs always produce the canonical form computed the -# first time. Malformed strings raise out of json.loads BEFORE anything -# is stored, so the repair fallback below is never memoized and reruns on -# every occurrence, exactly as before. Bounded FIFO eviction mirrors the -# _MSG_TOKENS_CACHE idiom in agent/model_metadata.py. +# Memo for send-path tool-call argument canonicalization, which re-runs on every +# historical call each iteration. Sound: canonicalization is pure and deterministic; +# malformed strings raise before being stored, so the repair fallback is never memoized. _CANON_ARGS_CACHE: Dict[str, str] = {} _CANON_ARGS_CACHE_MAX = 4096 -# Count bound alone doesn't bound MEMORY: write_file/patch argument strings -# run 100KB+, so 4096 entries could pin ~800MB in a long-lived gateway -# process. The byte budget keeps the memo effective for the common case -# (args ~0.5-2KB) while bounding the worst case. +# Count bound alone does not bound MEMORY: argument strings can run 100KB+, so a byte +# budget bounds the worst case while keeping the memo effective for ~0.5-2KB args. _CANON_ARGS_CACHE_MAX_BYTES = 32 * 1024 * 1024 _canon_args_cache_bytes = 0 @@ -1401,9 +1141,8 @@ _canon_args_cache_bytes = 0 def _canonicalize_tool_call_arguments(arg_str: str) -> str: """Return the canonical wire form of a tool-call arguments JSON string. - Raises whatever ``json.loads`` raises on malformed input; the caller - falls back to ``_repair_tool_call_arguments``, exactly as before. - """ + Raises whatever ``json.loads`` raises on malformed input; the caller falls back to + ``_repair_tool_call_arguments``.""" global _canon_args_cache_bytes cached = _CANON_ARGS_CACHE.get(arg_str) if cached is not None: @@ -1429,34 +1168,9 @@ def _canonicalize_tool_call_arguments(arg_str: str) -> str: def _clone_message_for_send(msg): """Structural clone of a history message for the per-call API copy. - The send path builds ``api_messages`` from the persisted history and - then rewrites the copies in place (canonicalization/repair of tool-call - arguments, surrogate and non-ASCII sanitization, content strips, cache - decoration). A shallow ``msg.copy()`` only decouples TOP-LEVEL fields: - nested containers — ``tool_calls`` entries and their ``function`` dicts, - multimodal ``content`` part lists, ``reasoning_details`` — remain the - SAME objects the persisted history holds, so any in-place write there - silently rewrites the stored transcript (#80498: an unrepairable - ``write_file`` argument string was replaced with ``{}`` in the persisted - turn, destroying the streamed file content). - - Cloning every container (dict/list) recursively while SHARING immutable - leaves (strings, numbers, None) makes every downstream in-place - transform safe by construction — current and future — at container-count - cost, not string-byte cost: big argument strings and base64 image - payloads are shared, never copied. Measured: ~1-5ms per 2000-message - pathological build (20% multimodal, 30% tool calls) vs ~0.4ms for the - shallow copy; compression keeps real request histories far smaller, and - the build runs once per API call — noise next to the call itself. - copy.deepcopy would be equally correct (CPython deepcopy also shares - immutable str) but ~4x slower again and needs its memo machinery; - history messages are JSON-shaped and acyclic (depth < 10 in practice; - a >~1000-deep pathological nest would hit the recursion limit, exactly - as deepcopy would), so cycle handling isn't needed here. Tuples are - shared as leaves: JSON-derived message content never contains tuples, - so a mutable container smuggled inside one is not a reachable shape on - this path. - """ + Clones every dict/list recursively while sharing immutable leaves, so in-place + send-path rewrites can never reach the persisted transcript (#80498). Cheaper than + copy.deepcopy; messages are JSON-shaped and acyclic, tuples are shared as leaves.""" if isinstance(msg, dict): return { k: _clone_message_for_send(v) if isinstance(v, (dict, list)) else v @@ -1473,14 +1187,8 @@ def _clone_message_for_send(msg): def _canonicalize_api_tool_calls(api_messages) -> None: """Canonicalize tool-call argument JSON on the send-path message copy. - Rewrites each message's ``tool_calls`` in place (copy-on-write for the - tool-call dicts it canonicalizes; the persisted history is untouched). - The pass still traverses every message and tool call each iteration; - the memo above bounds the JSON parse/serialize work to one round-trip - per UNIQUE argument string instead of one per string per iteration — - the quadratic part of the cost. The remaining traversal is pointer - chasing and dict copies, cheap next to a json.loads + json.dumps. - """ + Rewrites ``tool_calls`` in place (copy-on-write for the dicts it touches; persisted + history untouched). The memo bounds parse/serialize to one per UNIQUE string.""" for am in api_messages: tcs = am.get("tool_calls") if not tcs: @@ -1496,18 +1204,9 @@ def _canonicalize_api_tool_calls(api_messages) -> None: ), }} except Exception: - # Copy-on-write here too. The send-path build now hands - # this pass structurally-cloned messages (see - # _clone_message_for_send), but this branch keeps its own - # copy as defense in depth: some callers (tests, future - # call sites) pass shallow copies, and assigning into a - # shared ``tc["function"]`` would rewrite the stored - # turn. On the unrepairable path the repair returns "{}", - # so a write-through here replaced the model's real - # arguments with an empty object in the transcript: a - # stream that died mid ``write_file`` lost the file - # content it had already streamed, with only a WARNING - # to show for it (#80498). + # Copy-on-write as defense in depth: callers may pass shallow + # copies, and writing into a shared tc["function"] rewrote the + # stored turn with "{}" on the unrepairable path (#80498). tc = {**tc, "function": { **tc["function"], "arguments": _repair_tool_call_arguments( @@ -1522,18 +1221,9 @@ def _canonicalize_api_tool_calls(api_messages) -> None: def _invalid_tool_name_error_content(name: str, valid_tool_names) -> str: """Error-result content for a tool call whose name isn't a real tool. - A blank/whitespace-only name is not a typo the model can fuzzy-correct - toward a real tool — it is almost always a weak open model echoing - tool-call XML/JSON it saw in file or tool output (#47967: - / payloads in a file prime - mimo/nemotron-class models to emit empty structured calls), or a model - degrading at very large context (observed with gpt-5.6 past ~350K input). - Dumping the full tool catalog in that case feeds the priming loop more - names to mimic and inflates context 3-4x across retries, so send a terse - error that tells the model in-context tool-call syntax is DATA, not a - call to make. A genuinely-wrong-but-nonempty name (an actual typo) still - gets the catalog so the model can self-correct. - """ + A blank name is a model echoing tool-call syntax seen in data, not a typo (#47967); + dumping the catalog feeds that loop, so send a terse error instead. A nonempty wrong + name still gets the catalog so the model can self-correct.""" if not (name or "").strip(): return ( "Tool call rejected: the tool name was empty. " @@ -1556,12 +1246,8 @@ def _content_policy_blocked_result( ) -> Dict[str, Any]: """Build the terminal turn result for a content-policy block. - A content-policy refusal is deterministic for the unchanged prompt, so the - turn ends here (no retry). Both the HTTP-200 refusal handler and the - exception-path handler return the identical shape — a failed, non-completed - turn carrying the user-facing message and a ``content_policy_blocked:`` - prefixed error — so they funnel through this one builder. - """ + Refusals are deterministic for the unchanged prompt, so no retry; both the HTTP-200 + and exception paths return this shape with a ``content_policy_blocked:`` error.""" return { "final_response": final_response, "messages": messages, @@ -1580,23 +1266,9 @@ def _compression_deferred_result( ) -> Dict[str, Any]: """Build the soft turn result for a transiently-deferred compression. - Two transient shapes funnel here, and BOTH must end as a soft defer - (``compression_deferred``), never as ``compression_exhausted``: the - gateway auto-resets (wipes) the session on exhaustion (#9893/#35809). - - * ``reason="lock"`` — another path (a sibling turn, a background review - fork, a manual ``/compress``) holds this session's compression lock, - so every compression pass this turn no-oped and the request still does - not fit. The lock winner is actively shrinking the same session. - * ``reason="transient_block"`` — the compressor is in a timed transient - guard (summary-failure cooldown / structural backoff, e.g. one just - recorded by the host ceiling timeout, #97488). The no-op says nothing - about compressibility; treating it as exhaustion falsely auto-reset - sessions whose compression was merely cooling down. - - ``failed`` stays False so the gateway persists the user turn (transient - branch) and retry-next-message semantics apply. - """ + Both ``reason="lock"`` and ``reason="transient_block"`` must end as + ``compression_deferred``, never ``compression_exhausted`` — the gateway wipes the + session on exhaustion (#9893/#35809). ``failed`` stays False; the turn persists.""" if reason == "transient_block": block = getattr(agent, "_compression_blocked_transient", None) logger.info( @@ -1678,18 +1350,9 @@ def _provider_overflow_exhausted_result( def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool: """Rewrite a cache-decorated system message in place, keeping its blocks. - ``apply_anthropic_cache_control`` runs once per call block, *before* the - retry loop, and splits the system prompt into ``[static prefix, volatile - tail]`` text blocks carrying the cache_control breakpoints. Assigning a bare - string over that list drops both breakpoints, so the failover retry ships - the whole system prompt uncached and re-bills it in full. - - ``rewrite_prompt_model_identity`` only touches the LAST ``Model:`` / - ``Provider:`` lines, and those live in the volatile tail — so the static - prefix stays byte-identical and its cache entry keeps matching. Returns - False when the shape is not one we can safely patch, so the caller falls - back to the plain-string assignment. - """ + Assigning a bare string over the ``[static prefix, volatile tail]`` block list drops + both cache_control breakpoints. Only the LAST ``Model:``/``Provider:`` lines change. + Returns False when the shape cannot be safely patched.""" content = system_message.get("content") if not isinstance(content, list) or not content: return False @@ -1713,18 +1376,9 @@ def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool def _sync_failover_system_message(agent, api_messages, active_system_prompt): """Refresh the in-flight system message after a provider failover. - ``try_activate_fallback`` rewrites the ``Model:``/``Provider:`` identity - lines on ``agent._cached_system_prompt`` (see - ``rewrite_prompt_model_identity``) so the agent reports the model that is - actually answering. But the current call block's ``api_messages`` were - built from the pre-failover prompt, and the retry loop rebuilds - ``api_kwargs`` from that list each iteration — without this sync the - whole turn (and every gateway turn, since fallback re-activates per - message while the primary is down) ships the stale identity. - - Mutates ``api_messages[0]`` in place and returns the prompt to use as - ``active_system_prompt`` for subsequent call-block rebuilds. - """ + ``try_activate_fallback`` rewrites the identity lines on ``_cached_system_prompt``, + but this call block's ``api_messages`` were built pre-failover and are reused each + retry. Mutates ``api_messages[0]`` in place; returns the new ``active_system_prompt``.""" sp = getattr(agent, "_cached_system_prompt", None) if not isinstance(sp, str) or not sp: return active_system_prompt @@ -1738,17 +1392,11 @@ def _sync_failover_system_message(agent, api_messages, active_system_prompt): def _ensure_cached_system_prompt_static(agent, system_message=None) -> None: - """Rebuild ``_cached_system_prompt_static`` when caching becomes active. + """Rebuild ``_cached_system_prompt_static`` when caching becomes active (#72626). - Sessions restored under a cache-off primary skip the static-prefix rebuild - (gated on ``_use_prompt_caching`` at restore time). A later failover to a - cache-on provider would otherwise redecorate with ``static_system_prefix= - None`` and silently fall back to the legacy system-plus-3 layout (#72626). - - Thin wrapper over :func:`agent.system_prompt.reconstruct_static_prefix`, - which memoizes failed rebuilds so this stays cheap on the retry-loop hot - path (it runs at the top of every attempt). - """ + Sessions restored under a cache-off primary skip the static-prefix rebuild; a later + failover to a cache-on provider would otherwise silently fall back to the legacy + system-plus-3 layout. Wraps ``reconstruct_static_prefix`` (memoizes failures).""" from agent.system_prompt import reconstruct_static_prefix reconstruct_static_prefix( @@ -1760,12 +1408,9 @@ def _peel_moa_guidance( messages: List[Dict[str, Any]], guidance: Any, ) -> List[Dict[str, Any]]: - """Remove MoA reference guidance previously attached by ``_attach_reference_guidance``. + """Remove MoA reference guidance attached by ``_attach_reference_guidance``. - Thin wrapper over :func:`agent.moa_loop.peel_reference_guidance` (kept - adjacent to the attach so the forward/inverse shapes evolve together). - Lazy import mirrors the module's other moa_loop touchpoints. - """ + Kept adjacent to the attach so the forward/inverse shapes evolve together.""" from agent.moa_loop import peel_reference_guidance return peel_reference_guidance(messages, guidance) @@ -1781,17 +1426,9 @@ def _redecorate_prompt_cache_for_provider( ) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]] | tuple[List[Dict[str, Any]], Optional[Dict[str, Any]], List[Dict[str, Any]]]: """Strip and re-apply cache_control for the *current* provider policy. - Decoration runs once per call block before the retry loop for the primary - provider. ``try_activate_fallback`` refreshes ``_use_prompt_caching`` / - ``_use_native_cache_layout`` but the nine failover ``continue`` paths reused - the old ``api_messages`` (#72626). Mirror ``_reapply_reasoning_echo_for_provider`` - by reshaping at the top of each retry attempt. - - The source list is the mutated in-flight request (image shrink / ASCII / - reasoning_details recoveries already applied), never a pristine - pre-decoration snapshot. MoA guidance is peeled and rebased without - decoration; the acting aggregator plans its resolved destination later. - """ + Decoration runs once per call block for the primary provider, but failover + ``continue`` paths reuse ``api_messages`` (#72626), so reshape at the top of each + retry from the mutated in-flight request. MoA guidance is peeled and rebased.""" messages: List[Dict[str, Any]] = [ dict(m) if isinstance(m, dict) else m for m in (api_messages or []) ] @@ -1817,9 +1454,8 @@ def _redecorate_prompt_cache_for_provider( return messages, prepared return messages, prepared, planned_tools - # Direct attribute access matches the call-block decoration site — the - # flags are unconditionally initialized on AIAgent, and a getattr - # default here would mask a real init bug as silent cache-off. + # Direct attribute access, not getattr: the flags are always initialized on + # AIAgent, and a default would mask a real init bug as silent cache-off. if agent._use_prompt_caching: _ensure_cached_system_prompt_static(agent, system_message=system_message) static = getattr(agent, "_cached_system_prompt_static", None) @@ -1867,26 +1503,14 @@ def _apply_context_engine_selection( ) -> List[Dict[str, Any]]: """Run the optional per-turn ``ContextEngine.select_context()`` hook. - Returns the (possibly replaced) request message list. The hook is for - context *selection / routing* (retrieval, topic routing, role switching), - which is distinct from compression and fires every turn independent of - ``should_compress()``. - - Fail-open by design: a missing hook, any exception, or an invalid return - value yields the unmodified ``api_messages``. The result is request-only — - persisted conversation history is never mutated here. - """ + Returns the (possibly replaced) request list. Fail-open: a missing hook, exception, + or invalid return yields ``api_messages`` unchanged; history is never mutated.""" engine = getattr(agent, "context_compressor", None) if engine is None or not hasattr(engine, "select_context"): return api_messages - # Skip the no-op base implementation so non-implementing engines — - # including the built-in ContextCompressor — pay nothing per request: - # no history copies below, no call. ``hasattr`` alone is not enough, - # because the ABC defines a default ``select_context`` that every engine - # inherits. Mirrors the base-method short-circuit in - # ``_notify_context_engine_turn_complete``. Lazy import avoids any import - # cycle with agent.context_engine. + # Skip the no-op base ``select_context`` so non-implementing engines pay nothing; + # ``hasattr`` is not enough: the ABC defines a default. Lazy import avoids a cycle. try: from agent.context_engine import ContextEngine as _CE if getattr(engine.select_context, "__func__", None) is _CE.select_context: @@ -1895,16 +1519,8 @@ def _apply_context_engine_selection( pass session_label = getattr(agent, "session_id", None) or "-" - # Pass shallow copies of the reference-only inputs so an engine that - # mutates them in place cannot alter persisted transcript state. Only - # ``request_messages`` (the per-call request list) is meant to be acted on, - # and it may be replaced wholesale via the return value — never mutated in - # place either. ``conversation_messages`` / ``incoming_message`` are - # read-only context; copying enforces the request-only contract rather than - # merely documenting it. Structural clones, not dict(m): a shallow copy - # would leave nested containers (tool_calls, content parts) aliased to - # the persisted history, so an engine writing into them would rewrite - # the transcript (#80498 aliasing class). + # Structural clones: the engine must not be able to write through nested + # containers into persisted history; only the request list is acted on (#80498). _conv_copy = [_clone_message_for_send(m) for m in conversation_messages] \ if conversation_messages is not None else None _incoming_copy = _clone_message_for_send(incoming_message) if isinstance(incoming_message, dict) else incoming_message @@ -1926,11 +1542,8 @@ def _apply_context_engine_selection( if selected is None: return api_messages - # Require a NON-EMPTY list of dicts. An empty list must fall open to the - # original request: ``all([])`` is ``True``, so without the emptiness check - # a ``[]`` returned by a buggy/failing engine would replace a valid request - # with an empty message list that the downstream sanitizers cannot restore, - # reaching the provider as an invalid request instead of failing open. + # Require a NON-EMPTY list of dicts: ``all([])`` is ``True``, so a ``[]`` from a + # buggy engine would otherwise replace the request instead of failing open. if isinstance(selected, list) and selected and all(isinstance(m, dict) for m in selected): return selected @@ -1952,23 +1565,15 @@ def _notify_context_engine_turn_complete( ) -> None: """Notify the active context engine that a user turn has finished. - Calls the optional ``ContextEngine.on_turn_complete()`` observation hook - once per turn, after the assistant/tool loop has produced the finalized - transcript. The complement to ``select_context()`` (pre-request selection): - this lets an engine ingest / index / summarize the completed turn. - - Fail-open: a missing or no-op hook, or any exception, is swallowed. - ``messages`` is passed as a shallow copy so the engine cannot mutate the - persisted transcript. - """ + Fail-open: a missing/no-op hook or any exception is swallowed. ``messages`` is + passed as a copy so the engine cannot mutate the persisted transcript.""" engine = getattr(agent, "context_compressor", None) hook = getattr(engine, "on_turn_complete", None) if engine is None or not callable(hook): return - # Skip the no-op base implementation so non-implementing engines (incl. - # the built-in compressor) pay nothing per turn. Lazy import avoids any - # import cycle with agent.context_engine. + # Skip the no-op base ``on_turn_complete`` so non-implementing engines pay nothing + # per turn. Lazy import avoids an import cycle with agent.context_engine. try: from agent.context_engine import ContextEngine as _CE if getattr(hook, "__func__", None) is _CE.on_turn_complete: @@ -1978,9 +1583,8 @@ def _notify_context_engine_turn_complete( try: hook( - # Structural clones: on_turn_complete receives the PERSISTED - # history; a shallow dict(m) would let a hook write through - # nested containers into the transcript (#80498 aliasing class). + # Structural clones: dict(m) would let a hook write into nested containers + # of the persisted transcript (#80498). [_clone_message_for_send(m) for m in messages], usage=usage, **meta, @@ -2007,38 +1611,17 @@ def run_conversation( persist_user_platform_id: Optional[str] = None, moa_config: Optional[dict[str, Any]] = None, ) -> Dict[str, Any]: - """ - Run a complete conversation with tool calling until completion. + """Run a complete conversation with tool calling until completion. Args: - user_message (str): The user's message/question - system_message (str): Custom system message (optional, overrides ephemeral_system_prompt if provided) - conversation_history (List[Dict]): Previous conversation messages (optional) - task_id (str): Unique identifier for this task to isolate VMs between concurrent tasks (optional, auto-generated if not provided) - stream_callback: Optional callback invoked with each text delta during streaming. - Used by the TTS pipeline to start audio generation before the full response. - When None (default), API calls use the standard non-streaming path. - persist_user_message: Optional clean user message to store in - transcripts/history when user_message contains API-only - synthetic prefixes. - persist_user_timestamp: Optional platform event timestamp to store - as metadata on that persisted user message. - persist_user_display_kind: Optional presentation type for a - synthesized user turn (``auto_continue``, ``model_switch``, …). - Display-only: transcript surfaces render the row as a timeline - event instead of a user bubble, while the model still receives - the message unchanged. - persist_user_display_metadata: Optional payload for that event - (e.g. a delegation's task count). - persist_user_platform_id: Optional platform-side message id (e.g. the - Discord/Telegram message id) to store as metadata on that - persisted user message, so restart drain-window recovery can - dedup an interrupted turn against the transcript. - or queuing follow-up prefetch work. + stream_callback: per-text-delta callback (TTS); None uses the non-streaming path. + persist_user_message: clean text to store when ``user_message`` carries API-only + synthetic prefixes; ``persist_user_timestamp`` / ``persist_user_platform_id`` + are stored as metadata (platform id lets restart drain recovery dedup). + persist_user_display_kind/metadata: display-only event rendering (``auto_continue``, + ``model_switch``); the model still receives the message unchanged. - Returns: - Dict: Complete conversation result with final response and message history - """ + Returns: dict with the final response and message history.""" if moa_config is None: try: from hermes_cli.moa_config import decode_moa_turn @@ -2052,30 +1635,23 @@ def run_conversation( except Exception: pass - # The gateway caches agents across user turns. Compression state is - # per-turn: carrying a prior in-place boundary forward would make a later - # uncompressed result look like a compacted transcript to gateway writers. + # The gateway caches agents across turns; compression state is per-turn, or a stale + # in-place boundary would make a later uncompressed result look compacted. agent._last_compaction_in_place = False agent._last_compression_attempt_recorded = False agent._last_compression_attempt_in_place = None begin_fast_mode_turn(agent, conversation_history) - # Adopt any ~/.hermes/.env credential/base-url edits made since the last - # turn — a Settings save updates .env but not this worker's client, which - # was built at agent init (#67821). No-op when .env is unchanged. + # Adopt ~/.hermes/.env credential/base-url edits made since the last turn — a + # Settings save updates .env, not this worker's client (#67821). No-op if unchanged. try: agent._try_refresh_env_client_credentials() except Exception: logger.debug("per-turn env credential refresh failed", exc_info=True) # ── Per-turn setup (the prologue) ── - # All once-per-turn setup — stdio guarding, retry-counter resets, user - # message sanitization, todo/nudge hydration, system-prompt restore-or- - # build, preflight compression, the ``pre_llm_call`` plugin hook, - # external-memory prefetch, and crash-resilience persistence — lives in - # ``build_turn_context``. It mutates ``agent`` exactly as the inline code - # did and returns the locals the loop below reads back. See - # ``agent/turn_context.py``. + # All once-per-turn setup lives in ``build_turn_context`` (agent/turn_context.py); + # it mutates ``agent`` as the inline code did and returns the locals the loop reads. try: _ctx = build_turn_context( agent, @@ -2101,36 +1677,22 @@ def run_conversation( moa_active=bool(moa_config), ) except PreflightCompressionTimedOut as _preflight_timeout_exc: - # Turn-start fail-closed boundary (#98424): preflight compression hit - # the host's progress-aware timeout while the request was still - # oversized, so no provider call was sent. Convert the typed exception - # into the same typed recovery result the in-loop consumers return - # (salvaged #98741 / PR #99710) instead of letting it escape to the - # surfaces' generic exception handlers — the gateway deliberately - # hides raw exception text from users, which would bury the - # actionable "run /compress and retry" guidance and skip the - # compression_exhausted clean-session recovery contract. + # Preflight compression timed out; no provider call sent (#98424). Return the + # typed recovery result: surfaces hide raw exception text, which would bury the + # actionable guidance and skip the compression_exhausted recovery contract. logger.warning( "Turn-start preflight compression timed out — ending turn with " "typed recovery result: %s", _preflight_timeout_exc, ) - # build_turn_context registered this turn's in-flight tripwire slot - # (note_turn_start) but the early return skips the persist funnel - # that normally clears it — clear it here so the next turn does not - # log a spurious "concurrent turns on one session" warning. The - # inbound user row is intentionally NOT persisted on this path: the - # gateway skips transcript persistence for compression_exhausted - # results to prevent the session-growth loop (#7100), and the - # auto-reset moves future input to a clean session. + # Clear the tripwire slot note_turn_start registered; the early return skips the + # persist funnel that clears it. The user row is deliberately NOT persisted: + # the gateway skips persistence for compression_exhausted results (#7100). from agent.agent_runtime_helpers import note_turn_persisted note_turn_persisted(agent) - # Intentionally NOT _COMPRESSION_TIMEOUT_FINAL_RESPONSE: the boundary's - # exception text carries per-request context (token count, "provider - # call was not sent") that is the actionable guidance this handler - # exists to surface; the in-loop constant describes a different state - # (compression ran and could not reduce). + # Not _COMPRESSION_TIMEOUT_FINAL_RESPONSE — that describes a different state + # (compression ran, could not reduce); the exception text carries the guidance. _final_response = str(_preflight_timeout_exc) return { "final_response": _final_response, @@ -2161,10 +1723,8 @@ def run_conversation( # A configured SessionDB append failure halts only the affected turn. A # cached gateway agent must recover on the next message if storage did. agent._incremental_persistence_failed = False - # Cause of the most recent persistence failure this turn ('locked', - # 'disk', or 'unknown' — see hermes_state.classify_persistence_error). - # Reset alongside the failure flag so a lock-contention diagnosis from a - # previous turn can never leak into this turn's user-facing explanation. + # Cause of the last persistence failure this turn ('locked'/'disk'/'unknown', see + # hermes_state.classify_persistence_error). Reset so a prior diagnosis cannot leak. agent._last_persistence_error_cause = None # Per-turn diagnostic: a failed compression-tip adoption in a previous # turn's flush must not be reported against this turn. @@ -2177,76 +1737,47 @@ def run_conversation( failed = False codex_ack_continuations = 0 length_continue_retries = 0 - # One-shot "continue without thinking" override is turn-scoped: a - # thinking-only truncation arms it right before the continuation restart, - # and build_api_kwargs consumes it on that call. If the turn is - # interrupted/errors between arm and consume, it must not fire on the - # next turn's first request. + # Turn-scoped one-shot: armed by a thinking-only truncation, consumed by + # build_api_kwargs; must not survive an interrupted turn into the next one. agent._ephemeral_reasoning_off = False # Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS. _outer_error_count = 0 truncated_tool_call_retries = 0 truncated_response_parts: List[str] = [] compression_attempts = 0 - # One resolved per-turn compression attempt cap, shared by every site that - # consumes ``compression_attempts``: the pre-API pressure gate, the - # overflow/413 retry handlers, and the post-tool compaction gate. The - # counter is a consecutive unverified/ineffective-attempt backstop: a - # completed compaction rearms it only after a successful provider response - # reports a prompt below the threshold. - # Config-driven via compression.max_attempts (parsed + validated in - # agent_init); default 3 preserves the prior hardcoded behavior for - # objects without the attribute (older pickles / minimal stubs). + # Per-turn compression attempt cap shared by the pre-API gate, 413 handlers and + # post-tool compaction; a consecutive-ineffective-attempt backstop, rearmed only + # after a provider response reports a prompt below threshold. Default 3 if unset. max_compression_attempts = getattr(agent, "max_compression_attempts", 3) _last_preflight_pressure: Optional[int] = None _preflight_compression_blocked = _ctx.preflight_compression_blocked - # A provider overflow is stronger evidence than the rough-estimate - # calibration that normally defers preflight immediately after compaction. - # Keep recovery armed until the rebuilt, complete request is below the - # configured compression threshold. Without this handoff, a compaction - # that drops rows but grows the actual prompt can be sent straight back to - # the provider while awaiting_real_usage_after_compression is true. + # A provider overflow outweighs the rough-estimate calibration that defers preflight + # after compaction: stay armed until the rebuilt request is below the threshold. _provider_overflow_recovery_pending = False - # Armed when a compression host-timeout terminates the turn (#98722, - # salvaged from #98741); finalize below reuses the gateway's existing - # context-recovery contract (error/partial/compression_exhausted). + # Armed when a compression host-timeout ends the turn; finalize reuses the gateway + # context-recovery contract (error/partial/compression_exhausted) (#98722). _compression_timeout_exhausted = False _turn_exit_reason = "unknown" # Diagnostic: why the loop ended - # Last composed answer intentionally held back by a verification gate. If - # that continuation consumes the remaining budget, this is the best - # user-facing result available; it must not be confused with error or - # recovery text produced by unrelated exit paths. + # Last answer held back by a verification gate: if the continuation exhausts the + # budget this is the best user-facing result, distinct from error/recovery text. _pending_verification_response = None - # Tracks whether the pending verification candidate was already streamed - # to the user as interim content. The finalizer uses this to set - # ``_response_was_previewed`` ONLY when the pending candidate is actually - # reused as the final response — not merely because any interim was - # streamed. (#65919 review: response-loss blocker) + # Whether the pending verification candidate was already streamed as interim. + # ``_response_was_previewed`` is set ONLY if it becomes the final response (#65919). _pending_verification_response_previewed = False - # If pre-API compression fires after MoA advisors have produced guidance, - # retain that ephemeral output and rebase it onto the compacted transcript - # on the next loop iteration. This prevents a second advisor fan-out. + # If pre-API compression fires after MoA advisors ran, retain their guidance and + # rebase it onto the compacted transcript next iteration — no second fan-out. pending_moa_prepared_request = None - # Per-turn tally of consecutive successful credential-pool token refreshes, - # keyed by (provider, pool-entry-id). A persistent upstream 401 lets - # ``try_refresh_current()`` "succeed" forever on a single-entry OAuth pool, - # so this tally caps same-entry refreshes and lets the fallback chain take - # over instead of spinning. Reset here so each turn starts fresh. See #26080. + # Per-turn tally of credential-pool refreshes by (provider, pool-entry-id): caps + # same-entry refreshes on a persistent 401 so fallback takes over (#26080). agent._auth_pool_refresh_counts = {} - # Reset the per-turn usage holder forwarded to the context engine's - # on_turn_complete() observation hook. Set after each successful provider - # response (see below); left as None on turns that never reach a response - # (early failure / interrupt) so the hook receives None rather than a - # stale prior turn's usage. + # Per-turn usage forwarded to the context engine's on_turn_complete() hook; left + # None on turns that never reach a response so the hook never sees stale usage. agent._last_turn_usage = None - # Optional opt-in runtime: if api_mode == codex_app_server, hand the - # turn to the codex app-server subprocess (terminal/file ops/patching - # all run inside Codex). Default Hermes path is bypassed entirely. - # See agent/transports/codex_app_server_session.py for the adapter - # and references/codex-app-server-runtime.md for the rationale. + # Opt-in runtime: api_mode == codex_app_server hands the whole turn to the codex + # app-server subprocess (see agent/transports/codex_app_server_session.py). if agent.api_mode == "codex_app_server": return agent._run_codex_app_server_turn( user_message=user_message, @@ -2278,11 +1809,9 @@ def run_conversation( agent._safe_print("\n⚡ Breaking out of tool loop due to interrupt...") break - # Aggregate input budget for detached auxiliary forks (background - # review, #93057): compaction bounds each request; this bounds the - # review as a whole. Fires between iterations — the budget-crossing - # request completed (its tool writes landed), and the tool loop stops - # before the next provider call, mirroring the iteration-budget exit. + # Aggregate input budget for detached auxiliary forks: bounds the whole review, + # not each request. Checked between iterations so the crossing request's writes + # have landed, mirroring the iteration-budget exit (#93057). if _review_input_budget_exhausted(agent): _turn_exit_reason = "review_input_budget_exhausted" if not agent.quiet_mode: @@ -2297,9 +1826,8 @@ def run_conversation( agent._api_call_count = api_call_count agent._touch_activity(f"starting API call #{api_call_count}") - # Grace call: the budget is exhausted but we gave the model one - # more chance. Consume the grace flag so the loop exits after - # this iteration regardless of outcome. + # Grace call: budget exhausted but the model gets one more call. Consume the + # flag so the loop exits after this iteration regardless of outcome. if agent._budget_grace_call: agent._budget_grace_call = False elif not agent.iteration_budget.consume(): @@ -2343,17 +1871,8 @@ def run_conversation( agent._iters_since_skill += 1 # ── Pre-API-call /steer drain ────────────────────────────────── - # If a /steer arrived during the previous API call (while the model - # was thinking), drain it now — before we build api_messages — so - # the model sees the steer text on THIS iteration. Without this, - # steers sent during an API call only land after the NEXT tool batch, - # which may never come if the model returns a final response. - # - # We scan backwards for the last tool-role message in the messages - # list. If found, the steer is appended there. If not (first - # iteration, no tools yet), the steer stays pending for the next - # tool batch — injecting into a user message would break role - # alternation, and there's no tool output to piggyback on. + # Drain a /steer sent during the last API call into the newest tool message so + # it lands THIS iteration. Never put in a user message (breaks alternation). _pre_api_steer = agent._drain_pending_steer() if _pre_api_steer: _injected = False @@ -2394,25 +1913,16 @@ def run_conversation( agent._pending_steer = (existing + "\n" + _pre_api_steer) if existing else _pre_api_steer # ── Wall-clock run-budget wrap-up notice ─────────────────────── - # One-shot: when a run budget (agent.run_budget_seconds / - # --run-budget) is active and 80% of it has elapsed, ask the model - # to wrap up and deliver from the state it already has. Same - # cache-safe channel as /steer (appended to the newest tool - # result); dormant when no budget is set. + # One-shot at 80% of agent.run_budget_seconds: ask the model to wrap up via the + # same cache-safe channel as /steer (newest tool result); off with no budget. if getattr(agent, "run_budget_seconds", None): _maybe_inject_run_budget_wrapup(agent, messages) - # Prepare messages for API call - # If we have an ephemeral system prompt, prepend it to the messages - # Note: Reasoning is embedded in content via tags for trajectory storage. - # However, providers like Moonshot AI require a separate 'reasoning_content' field - # on assistant messages with tool_calls. We handle both cases here. + # Reasoning lives in content via tags for trajectory storage, but some + # providers (Moonshot) also need a 'reasoning_content' field; handle both here. request_logger = getattr(agent, "logger", None) or logging.getLogger(__name__) - # Per-agent validation cursor: skips re-json.loads-ing tool_call - # arguments on history messages already validated in a previous - # iteration. Identity-keyed (strong refs) — compression/undo/repair - # rewriting the list breaks the prefix match and forces a re-scan - # from the divergence point. See sanitize_tool_call_arguments. + # Per-agent validation cursor skips re-parsing tool_call args already validated. + # Identity-keyed; a rewritten list breaks the prefix match and forces a re-scan. _sanitize_cursor = getattr(agent, "_sanitize_args_cursor", None) if _sanitize_cursor is None: _sanitize_cursor = {} @@ -2433,12 +1943,8 @@ def run_conversation( agent.session_id or "-", ) - # Drop legacy ghost rows from the incomplete #73146 else branch BEFORE - # the alternation repair below: a hidden assistant placeholder whose - # content/api_content is the raw interrupt scaffold. Replaying that as - # an assistant message makes the model echo it and self-replicate - # (#81841). Dropping before repair lets repair_message_sequence fix - # any user→user adjacency the filter creates. + # Drop legacy hidden assistant placeholders carrying the raw interrupt scaffold + # before repair: replayed, the model echoes/self-replicates (#81841). messages = [ msg for msg in messages if not ( @@ -2457,16 +1963,9 @@ def run_conversation( ) ] - # Defensive: repair malformed role-alternation before API call. - # Catches cases where the history got wedged into a - # ``tool → user`` or ``user → user`` tail (e.g. after empty- - # response scaffolding was stripped and a new user message - # landed after an orphan tool result). Most providers return - # empty content on malformed sequences, which would otherwise - # retrigger the empty-retry loop indefinitely. - # repair_message_sequence_with_cursor also recomputes the SessionDB - # flush cursor (_last_flushed_db_idx) when repair compacts the list, - # so the turn-end flush doesn't skip the assistant/tool chain (#44837). + # Repair malformed role alternation (tool→user / user→user tails): providers + # return empty content on them and the empty-retry loop spins. The _with_cursor + # variant also recomputes the SessionDB flush cursor after compaction (#44837). from agent.agent_runtime_helpers import ( fill_empty_non_final_wire_payload, repair_message_sequence_with_cursor, @@ -2482,46 +1981,29 @@ def run_conversation( api_messages = [] for idx, msg in enumerate(messages): - # Structural clone, NOT msg.copy(): every in-place transform - # below (canonicalize/repair, surrogate + non-ASCII sanitizers, - # cache decoration) must be unable to reach the persisted - # history through shared nested containers. See - # _clone_message_for_send. + # Structural clone, NOT msg.copy(): in-place transforms below must not reach + # persisted history via nested containers; see _clone_message_for_send. api_msg = _clone_message_for_send(msg) - # api_content is the persistence sidecar carrying the exact bytes - # sent to the API for this message when they differ from the clean - # stored content (see compose_user_api_content in turn_context). - # It is bookkeeping, never a provider field — pop it from EVERY - # outgoing copy. + # api_content is the persistence sidecar of the exact bytes sent to the API; + # bookkeeping, never a provider field — pop it from EVERY outgoing copy. _api_content = api_msg.pop("api_content", None) - # Display-only timeline metadata. Never a provider field — strip - # from every outgoing copy so strict OpenAI-compatible backends - # don't reject the request after a model switch or resumed typed - # event row enters the live history. + # Display-only timeline metadata, never a provider field: strict OpenAI + # backends reject unknown keys once a typed event row enters live history. api_msg.pop("display_kind", None) api_msg.pop("display_metadata", None) - # Durable row identity stamped by _rows_to_conversation so the - # desktop can address a specific persisted message (reactions). - # Bookkeeping, never a provider field — only the chat-completions - # transport strips underscore keys, so drop it centrally here. + # Durable row id from _rows_to_conversation (desktop reactions); only the + # chat-completions transport strips underscore keys, so drop it centrally. api_msg.pop("_row_id", None) - # Inject ephemeral context into the current turn's user message. - # Sources: memory manager prefetch + plugin pre_llm_call hooks - # with target="user_message" (the default). Both are - # API-call-time only — the original message in `messages` is - # never mutated beyond the api_content stamp, so nothing leaks - # into the clean transcript content. + # Inject ephemeral context (memory prefetch + pre_llm_call user hooks) + # at API time only; `messages` is untouched beyond the api_content stamp. if idx == current_turn_user_idx and msg.get("role") == "user": if isinstance(_api_content, str) and _api_content: - # Stamped by the prologue from the same composition — - # reuse it so the persisted sidecar and the wire cannot - # drift, and so every pass this turn sends identical - # bytes (composed from msg["content"], never from a - # previously-injected copy). + # Reuse the prologue's stamp so sidecar and wire cannot drift + # and every pass this turn sends identical bytes. api_msg["content"] = _api_content else: # Callers that bypass the prologue stamping: compose live. @@ -2537,15 +2019,9 @@ def run_conversation( and _api_content and msg.get("role") in ("user", "assistant") ): - # Historical message: replay the exact bytes sent when it was - # live, so the provider prompt-cache prefix stays byte-stable - # instead of diverging at the injection point and - # re-prefilling everything after it. User rows carry the - # prefetch/plugin injection sidecar; user AND assistant rows - # can carry a sanitize-divergence sidecar (content that - # ``get_messages_as_conversation``'s sanitize_context/strip - # would rewrite on reload — see the capture in - # ``_flush_messages_to_session_db``). + # Historical row: replay the exact bytes sent live so the prompt-cache + # prefix stays byte-stable. User rows carry the injection sidecar; user + # and assistant rows may carry a sanitize-divergence sidecar. api_msg["content"] = _api_content # For ALL assistant messages, pass reasoning back to the API @@ -2559,30 +2035,21 @@ def run_conversation( # Remove finish_reason - not accepted by strict APIs (e.g. Mistral) if "finish_reason" in api_msg: api_msg.pop("finish_reason") - # Empty non-final user/assistant turns (#88955 hidden placeholders - # and #96870 stream-death / host-fed empties): once display_kind - # and api_content are stripped, the pre-call sanitizer would - # re-heal the wire copy on every send and flood errors.log. - # Fill the WIRE copy here so the sanitizer has nothing to do. - # Durable history is not mutated. After reasoning copy so a - # thinking-only turn keeps its payload and is not rewritten. + # Fill empty non-final user/assistant wire copies so the pre-call sanitizer + # stops re-healing and flooding errors.log; durable history is untouched. + # After the reasoning copy so thinking-only turns keep payload (#96870). fill_empty_non_final_wire_payload( api_msg, is_final=(idx == len(messages) - 1) ) - # _thinking_prefill survives here intentionally: the drop pass below - # needs it. The transport strips all underscore keys before the wire. - # Strip length-continuation marks; not every transport drops underscore keys. + # _thinking_prefill survives intentionally: the drop pass below needs it. + # Strip length-continuation marks; some transports keep underscore keys. api_msg.pop("_length_continuation_fragment", None) api_msg.pop("_length_continuation_nudge", None) - # Strip Codex Responses API fields (call_id, response_item_id) for - # strict providers like Mistral, Fireworks, etc. that reject unknown fields. - # Uses new dicts so the internal messages list retains the fields - # for Codex Responses compatibility. + # Strip Codex Responses fields (call_id, response_item_id): strict providers + # reject unknown fields. New dicts keep the internal list intact for Codex. if agent._should_sanitize_tool_calls(): - # In MoA mode, agent.model is the virtual preset name - # (e.g. "closed"), not the actual aggregator model. Use - # the resolved aggregator model so Gemini aggregators - # correctly preserve thought_signature (extra_content). + # In MoA mode agent.model is the virtual preset name; use the resolved + # aggregator so Gemini keeps thought_signature (extra_content). _sanitize_model = agent.model if agent.provider == "moa": if moa_config: @@ -2590,12 +2057,8 @@ def run_conversation( if _agg.get("model"): _sanitize_model = _agg["model"] if _sanitize_model == agent.model: - # Virtual-provider mode: no moa_config is threaded - # through run_conversation — the facade resolves the - # preset internally. Ask the facade for the resolved - # aggregator slot from the previous create() instead - # (set before any history replay that could carry - # thought_signature). + # Virtual-provider mode: no moa_config is threaded through; ask + # the facade for the aggregator slot from the previous create(). _moa_client = getattr(agent, "client", None) _agg_slot = getattr(_moa_client, "last_aggregator_slot", None) if _agg_slot and _agg_slot.get("model"): @@ -2605,21 +2068,9 @@ def run_conversation( # The signature field helps maintain reasoning continuity api_messages.append(api_msg) - # Build the final system message: cached prompt + ephemeral system prompt. - # Ephemeral additions are API-call-time only (not persisted to session DB). - # External recall context is injected into the user message, not the system - # prompt, so the stable cache prefix remains unchanged. - # - # NOTE: Plugin context from pre_llm_call hooks is injected into the - # user message (see injection block above), NOT the system prompt. - # This is intentional — system prompt modifications break the prompt - # cache prefix. The system prompt is reserved for Hermes internals. - # - # Hermes invariant: the system prompt is built ONCE per session - # (cached on ``_cached_system_prompt``) and replayed verbatim on - # every turn. ``apply_anthropic_cache_control`` may split its stable - # prefix into content blocks on the wire, but the stored string and - # its byte-stability remain unchanged. + # Final system message = cached prompt + ephemeral additions (API-time only). + # Plugin/recall context goes into the user message, never the system prompt: the + # prompt is built ONCE per session and replayed verbatim (stable cache prefix). effective_system = active_system_prompt or "" if agent.ephemeral_system_prompt: effective_system = (effective_system + "\n\n" + agent.ephemeral_system_prompt).strip() @@ -2635,10 +2086,8 @@ def run_conversation( user_prompt=( original_user_message if isinstance(original_user_message, str) - # Multimodal / decorated content list: extract the - # visible text instead of str()-ing a Python repr of - # the parts (which would leak base64 image payloads - # into the aggregator prompt). + # Multimodal content list: extract visible text rather than + # str()-ing parts, which would leak base64 image payloads. else _flatten_mt(original_user_message) ), api_messages=api_messages, @@ -2666,8 +2115,7 @@ def run_conversation( if isinstance(_base, str): _msg["content"] = _base + "\n\n" + _moa_context elif isinstance(_base, list): - # Multimodal user turn (text + image parts): - # append the MoA context as a trailing text + # Multimodal turn: append MoA context as a trailing text # part instead of silently dropping it. _msg["content"] = [ *_base, @@ -2682,19 +2130,12 @@ def run_conversation( if agent.prefill_messages: sys_offset = 1 if (api_messages and api_messages[0].get("role") == "system") else 0 for idx, pfm in enumerate(agent.prefill_messages): - # Structural clone: the sanitizers below run over - # api_messages in place, and a shallow copy would let them - # write through into agent.prefill_messages' nested - # containers (same aliasing class as the history build). + # Structural clone: the in-place sanitizers below must not write + # through into agent.prefill_messages' nested containers. api_messages.insert(sys_offset + idx, _clone_message_for_send(pfm)) - # Per-turn context selection hook (additive, no-op by default). - # Lets a context engine select/replace which context enters the - # prompt for THIS call only — retrieval, topic routing, role/branch - # switching — distinct from compression and independent of - # should_compress(). Request-only: persisted history is untouched, so - # caching/sanitization below operate on whatever the engine selected. - # Fail-open (see _apply_context_engine_selection). + # Per-turn context selection hook: an engine may select/replace context for THIS + # call only — request-only, fail-open, and independent of should_compress(). _sel_incoming = ( messages[current_turn_user_idx] if 0 <= current_turn_user_idx < len(messages) @@ -2708,18 +2149,12 @@ def run_conversation( logger=request_logger, ) - # Safety net: strip orphaned tool results / add stubs for missing - # results before sending to the API. Runs unconditionally — not - # gated on context_compressor — so orphans from session loading or - # manual message manipulation are always caught. + # Runs unconditionally (not gated on context_compressor) so orphaned tool + # results from session loading or manual message edits are always caught. api_messages = agent._sanitize_api_messages(api_messages) - # One-time repeated-heal escalation notice (#96870): if the sanitizer - # above just crossed the per-session heal threshold, deliver the - # queued notice through the status/warning callback — the normal - # out-of-band delivery channel (gateway status message / CLI print). - # NEVER appended to messages/api_messages: conversation context and - # the cached prompt prefix stay byte-identical. + # One-time repeated-heal notice goes out via the status/warning callback, NEVER + # appended to messages: the cached prompt prefix stays byte-identical (#96870). try: from agent.agent_runtime_helpers import ( consume_pending_sanitizer_heal_notice, @@ -2732,61 +2167,31 @@ def run_conversation( # A notice hiccup must never break the send path. logger.debug("sanitizer heal notice delivery failed", exc_info=True) - # Drop thinking-only assistant turns (reasoning but no visible - # output and no tool_calls) and merge any adjacent user messages - # left behind. Prevents Anthropic 400s ("The final block in an - # assistant message cannot be `thinking`.") and equivalent errors - # from third-party Anthropic-compatible gateways that can't replay - # a thinking-only turn. Runs on the per-call copy only — the - # stored conversation history keeps the reasoning block for the - # UI transcript and session persistence. + # Drop thinking-only assistant turns + merge adjacent users, API copy only: + # Anthropic-style backends 400 on a trailing `thinking` block; history keeps it. api_messages = agent._drop_thinking_only_and_merge_users( api_messages, drop_codex_reasoning_items=agent.api_mode != "codex_responses", ) - # Normalize message whitespace and tool-call JSON for consistent - # prefix matching. Ensures bit-perfect prefixes across turns, - # which enables KV cache reuse on local inference servers - # (llama.cpp, vLLM, Ollama) and improves cache hit rates for - # cloud providers. Operates on api_messages (the API copy) so - # the original conversation history in `messages` is untouched. + # Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns + # (KV-cache reuse on local servers, better cloud cache hits); API copy only. for am in api_messages: if isinstance(am.get("content"), str): am["content"] = am["content"].strip() _canonicalize_api_tool_calls(api_messages) - # Proactively strip any surrogate characters before the API call. - # Models served via Ollama (Kimi K2.5, GLM-5, Qwen) can return - # lone surrogates (U+D800-U+DFFF) that crash json.dumps() inside - # the OpenAI SDK. Sanitizing here prevents the 3-retry cycle. + # Strip lone surrogates (U+D800-U+DFFF) that some Ollama-served models emit; + # they crash json.dumps() inside the OpenAI SDK and trigger the 3-retry cycle. _sanitize_messages_surrogates(api_messages) - # NOTE (empty-content class fix): no send-time pad loop here. The - # single owner for "never send a turn strict wire validation rejects - # as empty" is ``repair_empty_non_final_messages``, which runs inside - # ``_sanitize_api_messages`` above — the unconditional pre-send - # chokepoint shared with the summary path. Its placeholder is - # non-whitespace, so it survives the whitespace-normalization pass - # regardless of ordering (a single-space pad here previously had to - # be sequenced after normalization to survive, forking the concept). + # No send-time pad loop here: ``repair_empty_non_final_messages`` (inside + # ``_sanitize_api_messages``) is the single owner of empty-turn repair, and its + # non-whitespace placeholder survives normalization regardless of ordering. - # Build the request-local cache sections only after every transcript - # mutation. The canonical tool registry stays undecorated. - # - # Runs LAST, after every message mutation above. Marking earlier - # defeats the prefix stability the mutations exist to create: - # ``_apply_cache_marker`` rewrites ``content`` from a plain string - # into a ``[{"type": "text", ...}]`` block, so the marked messages - # no longer match the ``isinstance(content, str)`` test in the - # whitespace-normalization pass and silently keep their raw - # leading/trailing whitespace. A tool result ending in "\n" is - # therefore sent unstripped while it sits in the last-3 window and - # stripped once it rolls out of it — the same message, different - # bytes on consecutive turns, which breaks the prefix match at - # exactly the point the breakpoints were meant to protect. Marking - # last also keeps breakpoints off messages that the orphan sweep or - # the thinking-only drop is about to remove or merge away. + # Build the request-local cache sections LAST, after every transcript mutation; + # the canonical tool registry stays undecorated. Marked ``content`` becomes text + # blocks the whitespace pass skips, so the same row's bytes vary across turns. tools_for_api = agent.tools if agent._use_prompt_caching and agent.provider != "moa": from agent.prompt_caching import ( @@ -2820,12 +2225,9 @@ def run_conversation( api_messages = _initial_cache_plan.messages tools_for_api = _initial_cache_plan.tools - # Build a persistent-MoA request before measuring compression pressure. - # MoA reference output is injected into the aggregator prompt, but it - # is deliberately ephemeral and therefore absent from ``messages``. - # Preparing here makes the pre-API guard measure the exact prompt the - # aggregator will receive; ``create()`` consumes this private prepared - # request later without running the advisors a second time. + # Prepare the persistent-MoA request before measuring compression pressure: the + # ephemeral advisor output is absent from ``messages``; ``create()`` reuses the + # prepared request instead of running the advisors again. _moa_prepared_request = None if agent.provider == "moa": _moa_completions = getattr(getattr(agent.client, "chat", None), "completions", None) @@ -2843,16 +2245,9 @@ def run_conversation( if _moa_prepared_request is not None: api_messages = _moa_prepared_request["messages"] - # One image-stripped message estimate feeds both figures. Was: a - # str(msg) char walk (re-serialized base64 every call) + a second - # messages walk inside estimate_request_tokens_rough. Tools added - # separately (compression needs them: 50+ tools = 20-30K tokens). - # total_chars is a rough (~) proxy — verbose log + hook metric only. - # Charge stale thinking only when the active route actually replays - # it (#84371): on codex_responses the text keys never ship (the - # encrypted item sidecars — charged unconditionally — carry the - # chain), so counting them here re-created the trigger/tail-walk - # disagreement that dead-looped compaction. + # One image-stripped estimate feeds both figures; tools counted separately (50+ + # tools ≈ 20-30K tokens); total_chars is a rough proxy for logs/hooks only. + # Charge stale thinking only when the active route replays it (#84371). from agent.turn_context import _agent_stale_thinking_on_wire if _agent_stale_thinking_on_wire(agent): @@ -2861,32 +2256,21 @@ def run_conversation( approx_tokens = estimate_messages_tokens_rough( api_messages, charge_stale_thinking=False ) - # Route-aware pressure: when the upcoming request is eligible for - # native Responses compaction the transport will checkpoint-prune - # the payload before sending — the generic durable-history figure - # overstates the wire by orders of magnitude on a compacted session - # and fires a 600s local compression the main request never needed - # (#96995, mirroring the turn-prologue preflight #96644/#96155). + # Route-aware: native Responses compaction prunes the wire payload, so the raw + # history figure overstates it and fires needless local compression (#96995). request_pressure_tokens = _midturn_request_pressure_tokens( agent, api_messages, effective_system or "", approx_tokens ) - # Usage-anchored override: when the last provider response's exact - # usage is still valid for the durable transcript, replace the - # whole-history heuristic with anchor + delta-estimate. The anchor's - # prompt_tokens already includes system prompt AND tool schemas as - # the provider counted them, so no tools add-on is needed. Falls - # back to the rough figures above when the anchor is stale/missing - # (first request, post-compaction, usage-less providers). + # Usage-anchored override: real prompt_tokens (incl. system + tool schemas) + + # delta estimate replaces the whole-history heuristic when the anchor is fresh. _anchored_pressure = anchored_context_tokens( messages, getattr(agent, "_usage_anchor", None) ) if _anchored_pressure is not None: request_pressure_tokens = _anchored_pressure total_chars = approx_tokens * 4 - # Stash this request's rough estimate so update_from_response() can - # pair it with the provider's real prompt count — the (rough, real) - # anchor behind should_defer_preflight_to_real_usage()'s projection. - # getattr guard: test doubles built via object.__new__ lack the method. + # Stash the rough estimate so update_from_response() can pair it with the real + # count (should_defer_preflight_to_real_usage). getattr: test doubles lack it. _note_rough = getattr( agent.context_compressor, "note_request_rough_estimate", None ) @@ -2910,23 +2294,9 @@ def run_conversation( pass break - # Pre-API pressure check. The turn-prologue preflight only saw the - # incoming user message; a single turn can then grow by many large - # tool results and leave no output budget before the NEXT call (the - # live 271k/272k Codex failure). The post-response should_compress - # gate at the tool-loop tail uses API-reported last_prompt_tokens, - # which LAGS a just-appended huge tool result — so it misses this - # case. Re-check here against the current request estimate. - # - # Mirror the turn-prologue preflight's guard chain exactly (see - # turn_context.py): (1) defer when the rough estimate is known-noisy - # relative to a recent real provider prompt that fit under threshold - # (schema overhead / post-compaction over-count, #36718); (2) skip - # while a same-session compression-failure cooldown is active; (3) then - # should_compress() — reusing the canonical threshold_tokens (output - # room already reserved by _compute_threshold_tokens) and its summary- - # LLM cooldown + anti-thrash guards (#11529). compression_attempts is a - # hard per-turn backstop shared with the overflow error handlers. + # Pre-API pressure check: tool results grow a turn and last_prompt_tokens lags + # them. Mirror the turn-prologue guard chain: defer on noisy estimate, skip in + # failure cooldown, then should_compress() (#11529). _compressor = agent.context_compressor _preflight_threshold = int( getattr(_compressor, "threshold_tokens", 0) or 0 @@ -2942,16 +2312,11 @@ def run_conversation( _provider_overflow_recovery_pending and not _provider_overflow_preflight ): - # The outer-loop rebuild includes the active system prompt, - # request-only injections, and tool schemas. Once that complete - # request has real output runway again, the provider may be tried. + # The outer-loop rebuild includes system prompt, request-only injections and + # tool schemas; only that full request with output runway may be sent. _provider_overflow_recovery_pending = False - # A previous mid-turn preflight pass deliberately continued the loop so - # API-only context and all sanitization could be rebuilt. Compare that - # fully assembled request with the fully assembled request that caused - # the pass. Raw ``messages`` are not equivalent here: they omit - # api_content/plugin injections, prefills, MoA context, and ephemeral - # system text. + # Compare fully assembled requests, not raw ``messages`` (which omit + # api_content, plugin injections, prefills, MoA context, ephemeral system text). _previous_preflight_pressure = _last_preflight_pressure _last_preflight_pressure = None if ( @@ -2963,10 +2328,8 @@ def run_conversation( _preflight_threshold, ) ): - # Stop proactive retries for this turn without consuming the - # shared overflow-recovery budget. If the provider proves the - # request truly does not fit, its error handler may still compact - # with that stronger signal. + # Stop proactive retries this turn without consuming the shared overflow- + # recovery budget; the provider's error handler may still compact. _preflight_compression_blocked = True logger.warning( "Pre-API compression made insufficient progress: ~%s -> " @@ -2996,20 +2359,14 @@ def run_conversation( and not _compression_cooldown and _compressor.should_compress(request_pressure_tokens) ): - # Managed local runtime: try GROWING the context window before - # compressing (the window ladder's design order — compression is - # the move of last resort, once the window is at the model's - # native max or physics/speed say stop). Only fires for a - # llamacpp-flavored provider whose base_url is the server this - # process supervises; every other provider falls straight - # through to compression, exactly as before. + # Managed local runtime: grow the context window before compressing (last + # resort). Only for a llamacpp provider at the supervised base_url. _grown_window = _maybe_grow_local_window( agent, _compressor, request_pressure_tokens ) if _grown_window: - # The server now grants a bigger window: recalibrate the - # compressor to it and skip compression this pass — the - # request that was over the OLD threshold fits the new one. + # Bigger window granted: recalibrate the compressor and skip compression + # this pass. _compressor.update_model( agent.model, _grown_window, @@ -3022,9 +2379,8 @@ def run_conversation( f"📈 Context window grown to {_grown_window // 1024}K " f"(local model; conversation continues uncompressed)" ) - # This preflight iteration never reached the provider — - # refund the consumed call/budget exactly as the compression - # path below does before ITS continue. + # Never reached the provider — refund the call/budget like the + # compression path does before its continue. api_call_count -= 1 agent._api_call_count = api_call_count agent.iteration_budget.refund() @@ -3032,12 +2388,8 @@ def run_conversation( if _moa_prepared_request is not None: pending_moa_prepared_request = _moa_prepared_request compression_attempts += 1 - # Compression is actually running (block cleared / was never - # blocked) — reset the blocked-overflow warning dedup so a future - # blocked-over-threshold turn can warn again. Mirrors the - # turn-context preflight reset (silent-overflow fix #62625). - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Compression is running: reset the blocked-overflow warning dedup so a + # later blocked turn warns again (#62625). getattr: test doubles lack it. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() @@ -3079,13 +2431,8 @@ def run_conversation( task_id=effective_task_id, ) if context_compression_timed_out(agent): - # Host progress-aware timeout (#98722, salvaged from #98741): - # this preflight iteration never reached the provider. Refund - # its provisional call/budget exactly like a successful - # pre-API compaction, then stop before the unchanged oversized - # request reaches the provider — its overflow error would only - # invoke compression again on the same transcript with the - # wait budget already spent. + # Progress-aware timeout (#98722): never reached the provider — refund + # the call/budget and stop; an overflow retry would only re-compress. api_call_count -= 1 agent._api_call_count = api_call_count agent.iteration_budget.refund() @@ -3098,50 +2445,29 @@ def run_conversation( compression_skipped_due_to_lock(agent) or compression_blocked_transiently(agent) ): - # #69870 lock-skip / #97488 transient-block: this pass - # no-oped for a TEMPORARY reason (another path holds the - # compression lock, or a timed cooldown/backoff guard is - # active). That is a temporary DEFER, not evidence about - # compressibility — refund the attempt (it must not burn the - # shared overflow-recovery budget toward - # compression_exhausted → gateway auto-reset, #9893/#35809) - # and leave the insufficient-progress blocker unarmed. - # Proceed with the current request: if it truly does not - # fit, the provider's 413/overflow handler returns the soft - # compression_deferred result with that stronger signal. + # Temporary DEFER (lock held / cooldown), not evidence about + # compressibility: refund the attempt, leave the progress blocker + # unarmed and proceed (#69870, #97488). compression_attempts -= 1 _last_preflight_pressure = None if pending_moa_prepared_request is _moa_prepared_request: pending_moa_prepared_request = None else: - # Reset retry/empty-response state so the compacted request - # gets a fresh chance instead of inheriting stale recovery - # counters from the pre-compaction history. + # Reset retry/empty-response state so the compacted request gets a fresh + # chance. agent._empty_content_retries = 0 agent._thinking_prefill_retries = 0 agent._last_content_with_tools = None agent._last_content_tools_all_housekeeping = False agent._mute_post_response = False - # Re-baseline the flush cursor for the compaction mode that just - # ran. Legacy session-rotation returns None (the child session has - # not seen the compacted transcript, so the next flush writes it - # whole); in-place compaction returns list(messages) because the - # compacted rows are already persisted under the same session id — - # leaving None there would re-append them, doubling the active - # context and retriggering compression. Mirrors the post-response - # and preflight compaction sites; see - # conversation_history_after_compression(). + # Re-baseline the flush cursor: rotation returns None (child flushes + # whole); in-place returns list(messages) — None would re-append + # persisted rows. See conversation_history_after_compression(). conversation_history = conversation_history_after_compression( agent, messages, conversation_history ) - # This preflight iteration never reaches the provider whether - # we skip the turn (handoff guard below) or re-run the loop — - # refund the consumed call/budget in BOTH cases, mirroring the - # ollama_runtime_context_too_small early-exit above. Without - # the refund on the break path, every skipped turn leaked one - # iteration-budget unit for the agent's lifetime and - # finalize_turn logged an api_call_count including a call that - # was never made. + # Never reaches the provider on skip or re-run — refund the call/budget + # in BOTH cases, else budget leaks and api_call_count over-reports. api_call_count -= 1 agent._api_call_count = api_call_count agent.iteration_budget.refund() @@ -3160,10 +2486,8 @@ def run_conversation( break continue elif _provider_overflow_preflight and _compression_cooldown: - # The provider already proved this request cannot fit, while the - # compressor is temporarily unavailable. Do not send the known- - # oversized request again; let the next user turn retry after the - # cooldown instead of turning this into compression exhaustion. + # Provider proved the request cannot fit and the compressor is unavailable: + # don't resend; let the next user turn retry after cooldown. agent._persist_session(messages, conversation_history) return _compression_deferred_result( agent, @@ -3175,10 +2499,8 @@ def run_conversation( _provider_overflow_preflight and compression_attempts >= max_compression_attempts ): - # Every bounded recovery pass has been consumed and the rebuilt - # request is still over threshold. Fail closed before another - # provider call; llama.cpp can silently truncate an oversized - # retry instead of returning a second actionable overflow error. + # All recovery passes consumed and still over threshold: fail closed — + # llama.cpp may silently truncate an oversized retry. return _provider_overflow_exhausted_result( agent, messages, @@ -3194,12 +2516,8 @@ def run_conversation( and not _defer_preflight(request_pressure_tokens) and _compression_cooldown ): - # Blocked by the summary-LLM cooldown. Surface a deduped warning - # (only when actually over threshold — should_compress_info - # returns a None reason below threshold) so the user isn't left - # with a silently growing context. Mirrors the turn-context - # preflight and the loop-compaction guards (silent-overflow fix - # #62625). + # Summary-LLM cooldown blocks compression: deduped warning only when over + # threshold (should_compress_info reason is None below it) (#62625). _block_reason = None try: _block_reason = _compressor.should_compress_info( @@ -3214,19 +2532,9 @@ def run_conversation( int(getattr(_compressor, "threshold_tokens", 0) or 0), ) elif not agent.compression_enabled and len(messages) > 1: - # Uncompressed session guard (#89297): compression is disabled, so - # nothing shrinks a growing session. Reuse the unconditionally - # computed request estimate (zero marginal cost — this site runs - # before every provider request, covering turn-start AND mid-turn - # tool-result growth) and surface a deduped, actionable warning - # when the request exceeds the model context window. The dedup is - # re-armed by the turn-context preflight once the session is back - # under the window (manual /compress works with compression - # disabled), so the guard warns again on a later re-overflow. - # context_compressor always exists (agent_init constructs it even - # when compression is disabled) and its context_length property - # hard-floors at a positive default — no metadata re-resolution - # needed here. + # Uncompressed session guard (#89297): compression is disabled, so warn + # (deduped) when the request exceeds the context window; the turn-context + # preflight re-arms the dedup. _ctx_len = getattr( getattr(agent, "context_compressor", None), "context_length", None ) @@ -3242,10 +2550,8 @@ def run_conversation( _warn_fn(request_pressure_tokens, _ctx_len) if _provider_overflow_preflight: - # Any other gate that prevented the forced preflight (for example, - # an uncompressible one-message request) must also fail closed. - # Falling through would send a request that the provider already - # proved cannot fit. + # Any other gate blocking the forced preflight (e.g. uncompressible one- + # message request) must fail closed: the request is proven not to fit. return _provider_overflow_exhausted_result( agent, messages, @@ -3296,10 +2602,8 @@ def run_conversation( while retry_count < max_retries: # ── Nous Portal rate limit guard ────────────────────── - # If another session already recorded that Nous is rate- - # limited, skip the API call entirely. Each attempt - # (including SDK-level retries) counts against RPH and - # deepens the rate limit hole. + # Skip the call if another session recorded a rate limit: every attempt + # (incl. SDK retries) counts against RPH. if agent.provider == "nous": try: from agent.nous_rate_guard import ( @@ -3348,22 +2652,15 @@ def run_conversation( try: agent._reset_stream_delivery_tracking() - # Per-attempt first-chunk timestamp, refreshed each attempt so - # a stale value from a previous API call can never leak into - # the post_api_request hook (set again on stream success). + # Per-attempt first-chunk timestamp so a stale value never leaks into + # post_api_request. agent._last_api_first_chunk_at = None - # api_messages is built once, before this retry loop, while the - # primary provider is active. A mid-conversation fallback can - # switch to a require-side provider (DeepSeek / Kimi / MiMo) that - # rejects assistant turns lacking reasoning_content. Re-apply the - # echo-back pad for the *current* provider here (idempotent no-op - # unless the active provider needs it) so the fallback request - # isn't sent with stale, primary-shaped reasoning fields. + # api_messages was built for the primary; a fallback (DeepSeek / Kimi / + # MiMo) may require reasoning_content. Re-apply the echo-back pad + # (idempotent). agent._reapply_reasoning_echo_for_provider(api_messages) - # Same story for prompt-cache decoration (#72626): try_activate_ - # fallback refreshes the policy flags, but the decorated list - # still carries the primary's breakpoints (or none). Strip and - # re-render for the current provider before building kwargs. + # Same for prompt-cache decoration (#72626): strip the primary's + # breakpoints and re-render for the current provider. api_messages, _moa_prepared_request, tools_for_api = ( _redecorate_prompt_cache_for_provider( agent, @@ -3380,15 +2677,9 @@ def run_conversation( api_messages, tools_for_api=tools_for_api, ) - # Outbound-request surrogate chokepoint (#50959): the messages - # were scrubbed above, but the rest of the request body — - # tool/function descriptions (session_search's ±-heavy text is - # the recorded repro), extra_body, system strings routed via - # kwargs — can still carry invalid code points that providers - # reject with a non-retryable HTTP 400 ("invalid unicode code - # point"). One in-place walk here guarantees the entire - # payload json.dumps()-safe regardless of which leaf produced - # the string. Fast no-op when the payload is clean. + # Surrogate chokepoint (#50959): tool descriptions, extra_body and + # kwargs strings can carry invalid code points (HTTP 400). One walk + # makes the payload json.dumps()-safe. _sanitize_structure_surrogates(api_kwargs) if agent._force_ascii_payload: _sanitize_structure_non_ascii(api_kwargs) @@ -3399,17 +2690,14 @@ def run_conversation( is_github_responses=agent._is_copilot_url(), sanitize_harmony_tokens=agent._is_codex_backend(), ) - # OpenRouter response caching replays identical successful - # responses verbatim, including empty completions. An empty- - # response retry must reach the provider instead of replaying - # the response that triggered the retry. + # OpenRouter caching replays identical responses, even empty ones; an + # empty-response retry must bypass the cache. if agent._empty_content_retries > 0 and agent._is_openrouter_url(): _xh = dict(api_kwargs.get("extra_headers") or {}) _xh["X-OpenRouter-Cache"] = "false" api_kwargs["extra_headers"] = _xh - # Copilot x-initiator: the first API call of a user turn is - # marked "user" so Copilot bills a premium request; tool-loop - # follow-ups keep the default "agent" header (#3040). + # Copilot x-initiator: first call of a user turn is "user" (billed + # premium); tool-loop follow-ups keep the default "agent" (#3040). if getattr(agent, "_is_user_initiated_turn", False) and agent._is_copilot_url(): _xh = dict(api_kwargs.get("extra_headers") or {}) _xh["x-initiator"] = "user" @@ -3449,27 +2737,13 @@ def run_conversation( request_messages = api_kwargs.get("input") if not isinstance(request_messages, list): request_messages = api_messages - # Shallow-copy the outer list so plugins that retain the - # reference for async snapshotting don't observe later - # mutations of api_messages. The inner dicts are not - # mutated by the agent loop, so a shallow copy is - # sufficient; a deepcopy would walk every tool result - # and base64 image on every API call. - # - # The ``request_messages`` and ``conversation_history`` - # kwargs below are pre-existing raw passthroughs - # consumed by the bundled langfuse plugin - # (``plugins/observability/langfuse/__init__.py:_coerce_request_messages``). - # They predate ``request`` and are intentionally NOT - # sanitised — secrets are not expected here because - # ``api_kwargs`` is the same object passed to the - # provider client. New consumers should read the - # sanitised view from ``request["body"]["messages"]``. + # Shallow copy: plugins may retain the list; deepcopy is costly. + # ``request_messages``/``conversation_history`` are raw langfuse + # passthroughs. _request_payload = agent._api_request_payload_for_hook(api_kwargs) - # Anthropic (``system``) and Responses/Codex - # (``instructions``) move the system prompt out of - # messages; pass it explicitly for observability - # plugins (Langfuse). + # Anthropic (``system``) and Responses/Codex (``instructions``) + # move the system prompt out of messages; pass it for + # observability. system_prompt_for_hooks = _system_prompt_for_hooks( api_kwargs, request_messages ) @@ -3507,20 +2781,12 @@ def run_conversation( if env_var_enabled("HERMES_DUMP_REQUESTS"): agent._dump_api_request_debug(api_kwargs, reason="preflight") - # This object is private to the in-process MoA facade. Add it - # only after middleware, hooks, and debug dumps so none of them - # attempts to serialize it as part of the provider payload. + # Private to the in-process MoA facade; add after middleware/hooks/debug + # dumps so none serializes it into the provider payload. if _moa_prepared_request is not None and agent.provider == "moa": - # Re-read the live client instead of trusting the one that - # prepared the request above. Credential rotation, provider - # fallback and dead-connection cleanup all rebuild - # agent.client from _client_kwargs between attempts, and - # pending_moa_prepared_request carries a prepared request - # across exactly that boundary. The rebuilt client is a - # native OpenAI client while provider stays "moa", so this - # private key would reach the SDK as an unexpected keyword - # — a non-retryable TypeError that kills every remaining - # turn on the session. + # Re-read the live client: rotation/fallback/cleanup rebuild + # agent.client between attempts; a native OpenAI client rejects this + # key (TypeError). if _moa_client_consumes_prepared_request(agent.client): api_kwargs["_moa_prepared_request"] = _moa_prepared_request else: @@ -3530,17 +2796,9 @@ def run_conversation( type(agent.client).__name__, ) - # Always prefer the streaming path — even without stream - # consumers. Streaming gives us fine-grained health - # checking (90s stale-stream detection, 60s read timeout) - # that the non-streaming path lacks. Without this, - # subagents and other quiet-mode callers can hang - # indefinitely when the provider keeps the connection - # alive with SSE pings but never delivers a response. - # The streaming path is a no-op for callbacks when no - # consumers are registered, and falls back to non- - # streaming automatically if the provider doesn't - # support it. + # Always prefer streaming even without consumers: it gives stale- + # stream/read-timeout health checks that quiet callers otherwise lack. + # Falls back if unsupported. def _stop_spinner(): nonlocal thinking_spinner if thinking_spinner: @@ -3550,37 +2808,26 @@ def run_conversation( agent.thinking_callback("") _use_streaming = True - # Provider signaled "stream not supported" on a previous - # attempt — switch to non-streaming for the rest of this - # session instead of re-failing every retry. + # Provider signaled "stream not supported": stay non-streaming for the + # session. if getattr(agent, "_disable_streaming", False): _use_streaming = False - # An ACP client communicates via subprocess stdio and returns a - # plain SimpleNamespace — not an iterable stream. Keyed on the - # `acp://` scheme rather than one vendor, so any ACP client is - # excluded. Mirror the ACP exclusion used for Responses API - # upgrade (lines ~1083-1085). + # ACP clients (`acp://` scheme, any vendor) return a plain + # SimpleNamespace, not a stream; mirrors the Responses API exclusion. elif ( agent.provider in {"copilot-acp"} or str(agent.base_url or "").lower().startswith("acp://") or str(agent.base_url or "").lower().startswith("acp+tcp://") ): _use_streaming = False - # MoA streams only when a display/TTS consumer is present to - # receive the deltas. MoAChatCompletions.create() honors - # stream=True (runs the references, then returns the aggregator's - # raw token stream) and is reached here because, for provider - # "moa", _create_request_openai_client returns the MoA facade - # itself. Without consumers (quiet mode, subagents, health-check - # probes) we keep the complete-response path: the facade returns a - # whole response when stream is not requested, preserving the - # prior behavior for those callers. + # MoA streams only with a display/TTS consumer + # (MoAChatCompletions.create() honors stream=True); else complete- + # response path. elif agent.provider == "moa" and not agent._has_stream_consumers(): _use_streaming = False elif not agent._has_stream_consumers(): - # No display/TTS consumer. Still prefer streaming for - # health checking, but skip for Mock clients in tests - # (mocks return SimpleNamespace, not stream iterators). + # No consumer: still stream for health checking, except Mock clients + # in tests (SimpleNamespace, not stream iterators). from unittest.mock import Mock if isinstance(getattr(agent, "client", None), Mock): _use_streaming = False @@ -3661,10 +2908,8 @@ def run_conversation( _model_request_active.clear() _redirect_crossed_response = agent._has_pending_redirect() if _redirect_crossed_response: - # The response and redirect can cross on different threads: - # redirect() observed the request as active just before this - # call returned. Discard that now-stale response and rebuild - # from the correction rather than silently losing it. + # Response and redirect can cross threads: discard the now-stale + # response and rebuild from the correction rather than lose it. if thinking_spinner: thinking_spinner.stop("") thinking_spinner = None @@ -3704,9 +2949,8 @@ def run_conversation( response_invalid = True error_details.append("response is None") else: - # Provider returned a terminal failure (e.g. quota exhaustion). - # Treat as invalid so the fallback chain is triggered instead of - # letting the error bubble up outside the retry/fallback loop. + # Terminal provider failure (e.g. quota exhaustion): treat + # as invalid so the fallback chain triggers. _codex_resp_status = str(getattr(response, "status", "") or "").strip().lower() if _codex_resp_status in {"failed", "cancelled"}: _codex_error_obj = getattr(response, "error", None) @@ -3802,9 +3046,8 @@ def run_conversation( # upstream server error, or malformed response. retry_count += 1 - # Eager fallback: empty/malformed responses are a common - # rate-limit symptom. Switch to fallback immediately - # rather than retrying with extended backoff. + # Eager fallback: empty/malformed responses often mean rate limiting + # — switch now instead of extended backoff. if agent._fallback_index < len(agent._fallback_chain): agent._buffer_status("⚠️ Empty/malformed response — switching to fallback...") if agent._try_activate_fallback(): @@ -3914,14 +3157,9 @@ def run_conversation( _backoff_touch_counter = 0 while time.time() < sleep_end: if agent._interrupt_requested: - # A redirect uses the interrupt machinery to cancel - # only the live request. Aborting the retry here - # with clear_interrupt() would DESTROY the pending - # correction and kill the turn with "Operation - # interrupted" — the exact mid-stream steer loss - # users hit when a redirect lands during provider - # backoff. Rebuild from the correction instead, - # mirroring the InterruptedError handler. + # A redirect cancels only the live request; + # clear_interrupt() would DESTROY the pending correction. + # Rebuild from it. if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break @@ -3966,13 +3204,9 @@ def run_conversation( if incomplete_reason is not None: incomplete_reason = str(incomplete_reason).strip().lower() if status == "incomplete" and incomplete_reason in {"max_output_tokens", "length"}: - # Responses API max-output exhaustion is a normal - # Codex incomplete turn. Let the Codex-specific - # continuation path below append the incomplete - # assistant state and retry, instead of routing to - # the generic chat-completions length rollback that - # emits "Response truncated due to output length - # limit" and stops gateway turns. + # Responses API max-output exhaustion is a normal Codex + # incomplete turn: use the Codex continuation path, not the + # length rollback. finish_reason = "incomplete" elif status == "incomplete" and incomplete_reason == "content_filter": finish_reason = "content_filter" @@ -4003,19 +3237,8 @@ def run_conversation( finish_reason = "length" # ── Content-policy refusal (HTTP 200) ────────────────── - # The model — or the provider's safety system — returned a - # *successful* response whose stop/finish reason is a refusal: - # Anthropic ``stop_reason="refusal"`` → ``content_filter``; - # OpenAI / portal ``finish_reason="content_filter"`` or a - # populated ``message.refusal`` (mapped in the chat_completions - # transport); Bedrock ``guardrail_intervened``. The content is - # typically empty, so without this branch the response falls - # through to the empty-response / invalid-response retry loops - # and is mis-surfaced as "rate limited" / "no content after - # retries" — burning paid attempts reproducing a deterministic - # refusal. Surface it clearly and stop. Mirrors the - # exception-based ``content_policy_blocked`` recovery: try a - # configured fallback once, otherwise return the refusal. + # Refusal finish reasons (``content_filter``, ``guardrail_intervened``) + # are deterministic: one fallback try, else return the refusal. if finish_reason == "content_filter": _refusal_transport = agent._get_transport() if agent.api_mode == "anthropic_messages": @@ -4052,9 +3275,8 @@ def run_conversation( if agent.thinking_callback: agent.thinking_callback("") - # Deterministic for the unchanged prompt — never retry. - # Try a configured fallback once (a different model may not - # refuse); otherwise surface the refusal terminally. + # Deterministic for the unchanged prompt — never retry. Try a + # configured fallback once; otherwise surface the refusal. if agent._has_pending_fallback(): agent._buffer_status( "⚠️ Model declined to respond (safety refusal) — trying fallback..." @@ -4119,13 +3341,9 @@ def run_conversation( force=True, ) - # Normalize the truncated response to a single OpenAI-style - # message shape so text-continuation and tool-call retry - # work uniformly across chat_completions, bedrock_converse, - # and anthropic_messages. For Anthropic we use the same - # adapter the agent loop already relies on so the rebuilt - # interim assistant message is byte-identical to what - # would have been appended in the non-truncated path. + # Normalize to one OpenAI-style message so continuation and tool- + # call retry work across transports (Anthropic reuses the loop's + # adapter). _trunc_msg = None _trunc_transport = agent._get_transport() if agent.api_mode == "anthropic_messages": @@ -4140,17 +3358,8 @@ def run_conversation( _trunc_has_tool_calls = bool(getattr(_trunc_msg, "tool_calls", None)) if _trunc_msg else False # ── Detect thinking-budget exhaustion ────────────── - # When the model spends ALL output tokens on reasoning - # and has none left for the response, continuation - # retries are pointless. Detect this early and give a - # targeted error instead of wasting 3 API calls. - # A response is "thinking exhausted" only when the model - # actually produced reasoning blocks but no visible text after - # them. Models that do not use tags (e.g. GLM-4.7 on - # NVIDIA Build, minimax) may return content=None or an empty - # string for unrelated reasons — treat those as normal - # truncations that deserve continuation retries, not as - # thinking-budget exhaustion. + # Only when reasoning blocks exist with no visible text after them; + # content=None from non- models is normal truncation. _has_think_tags = bool( _trunc_content and re.search( r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', @@ -4178,9 +3387,8 @@ def run_conversation( f"no visible response was produced.", force=True, ) - # Return a user-friendly message as the response so - # CLI (response box) and gateway (chat message) both - # display it naturally instead of a suppressed error. + # Return a user-friendly message as the response so CLI and + # gateway display it. _exhaust_response = ( "⚠️ **Thinking Budget Exhausted**\n\n" "The model used all its output tokens on reasoning " @@ -4201,16 +3409,8 @@ def run_conversation( } # ── Detect repetition-dominated truncation (#86581) ── - # A model in a degenerate repetition loop can spend its - # ENTIRE output budget echoing one fragment. The - # continuation nudge below would then stitch the - # pathological fragment into the final response — in the - # #86581 incident one turn produced a 60,698-char - # response delivered as 31 Discord messages. Abort with - # a clear user-facing error instead, mirroring the - # _thinking_exhausted guard above. Reasoning blocks are - # stripped first (repeated scratchpad lines are not - # evidence of a degenerate visible response). + # A repetition loop can burn the whole budget on one fragment; abort + # like _thinking_exhausted (reasoning stripped first). _visible_trunc = ( agent._strip_think_blocks(_trunc_content) if isinstance(_trunc_content, str) @@ -4257,17 +3457,8 @@ def run_conversation( if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}: assistant_message = _trunc_msg # ── Content-filter stream stall → fallback (#32421) ── - # When the provider's output-layer safety filter (e.g. - # MiniMax "output new_sensitive (1027)", Azure - # content_filter) kills the stream mid-delivery, the - # raw error was classified at the swallow point and the - # stub tagged ``_content_filter_terminated``. This - # filter is content-deterministic — continuation - # retries against the SAME primary just re-hit it and - # burn paid attempts (the loop used to give up with - # "Response remained truncated after 3 continuation - # attempts" and never consult the fallback chain). - # Escalate to the configured fallback BEFORE retrying. + # ``_content_filter_terminated`` is content-deterministic; + # escalate to the fallback before retrying the primary. _cf_terminated = getattr( response, "_content_filter_terminated", False ) @@ -4284,10 +3475,8 @@ def run_conversation( "Content filter terminated stream; switching to fallback..." ) if agent._try_activate_fallback(): - # Roll the partial content (if any was already - # appended in a prior continuation pass) back to - # the last clean turn so the fallback provider - # gets a coherent continuation point. + # Roll partial content back to the last clean turn so + # the fallback gets a coherent continuation point. if truncated_response_parts: messages = agent._get_messages_up_to_last_assistant(messages) # Unmark survivors: their text left the stitched partial. @@ -4313,39 +3502,17 @@ def run_conversation( ) if assistant_message is not None and not _trunc_has_tool_calls: length_continue_retries += 1 - # An interim assistant message with NO visible - # content must not be appended — whichever way it - # got that way. An empty partial-stream stub - # (stream dropped before any text was delivered) - # and a response whose whole output budget went to - # reasoning delivered in a separate field (GLM-5.3 - # on ollama-cloud with reasoning_effort=high: - # finish_reason="length", content="", - # completion_tokens == max_tokens) both serialize - # as {"role": "assistant", "content": ""}, and - # strict providers (Moonshot/Kimi via OpenRouter) - # reject empty assistant content with HTTP 400 - # ("message ... with role 'assistant' must not be - # empty") on the very next replay — permanently - # poisoning the session history until the pre-call - # sanitizer "heals" the hole (observed 3+ healings - # per turn). There is no partial text to continue - # from anyway, so only the continuation - # user-message is appended. + # Never append an interim assistant message with NO visible + # content: strict providers reject it (HTTP 400), poisoning + # history. Append only the nudge. _interim_content = getattr(assistant_message, "content", None) _is_empty_partial_stub = ( getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID and not _interim_content ) if not _interim_content and not _is_empty_partial_stub: - # Thinking-only truncation: the model spent the - # entire output cap on reasoning and produced no - # visible text. A continuation with thinking - # ON would re-think the whole context from - # scratch (continuations never replay prior - # reasoning) and re-burn the same budget, so - # the next call drops thinking for one request - # — the answer must be written, not re-derived. + # Thinking-only truncation: continuing with thinking ON + # re-burns the budget, so drop thinking for one request. agent._ephemeral_reasoning_off = True if _interim_content: interim_msg = agent._build_assistant_message(assistant_message, finish_reason) @@ -4396,11 +3563,8 @@ def run_conversation( break partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip() - # The pending one-shot reasoning-off override must - # not leak into the next turn when the 4th - # truncation goes straight to the ceiling exit - # without scheduling a continuation call to - # consume it. + # The one-shot reasoning-off override must not leak into the + # next turn when the ceiling exit skips the consuming call. agent._ephemeral_reasoning_off = False if partial_response: agent._vprint( @@ -4411,12 +3575,8 @@ def run_conversation( ) _ceiling_final = partial_response else: - # Every fragment was empty — e.g. a thinking - # model that spent each attempt's whole cap on - # reasoning (GLM-5.3 on ollama-cloud). Return - # an actionable message instead of an invisible - # None result, which only surfaces as a bare - # error card. + # Every fragment was empty (e.g. reasoning-only model): + # return an actionable message, not a bare None. agent._vprint( f"{agent.log_prefix}⚠️ Response still truncated " f"after {length_continue_retries} continuation attempts — no visible " @@ -4477,9 +3637,8 @@ def run_conversation( if truncated_tool_call_retries < 4: truncated_tool_call_retries += 1 if _is_stub_stall: - # The stream broke mid tool-call (network / - # peer-closed connection), not a real output - # cap — say so instead of "max output tokens". + # Stream broke mid tool-call (network), not a real + # output cap — say so. agent._buffer_vprint( f"⚠️ Stream interrupted mid tool-call — " f"retrying ({truncated_tool_call_retries}/4)..." @@ -4490,11 +3649,8 @@ def run_conversation( f"retrying API call " f"({truncated_tool_call_retries}/4)..." ) - # Boost max_tokens on each retry so the model has - # more room to complete the tool-call JSON. A - # network stall doesn't need a bigger budget, but - # a genuine output-cap truncation does, and the - # boost is harmless for the stall case. + # Boost max_tokens per retry: a real output-cap + # truncation needs it; harmless for a stall. _tc_boost_base = agent.max_tokens if agent.max_tokens else 4096 _tc_boost = _tc_boost_base * (2 ** truncated_tool_call_retries) _tc_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs) @@ -4502,9 +3658,8 @@ def run_conversation( _tc_boost = max(_tc_boost, _tc_requested_cap) _tc_boost_cap = max(32768, _tc_requested_cap or 0) agent._ephemeral_max_output_tokens = min(_tc_boost, _tc_boost_cap) - # Don't append the broken response to messages; - # just re-run the same API call from the current - # message state, giving the model another chance. + # Don't append the broken response; re-run the same call + # from current state. continue agent._flush_status_buffer() if _is_stub_stall: @@ -4524,9 +3679,8 @@ def run_conversation( if _is_stub_stall else "Response truncated due to output length limit" ) - # Prior successful tool batches (or injected tool - # errors) can leave a tool-result tail; this path - # never reaches finalize_turn (#48879 class). + # Prior tool batches can leave a tool-result tail; this path + # never reaches finalize_turn (#48879). close_interrupted_tool_sequence(messages, _final_response) agent._persist_session(messages, conversation_history) return { @@ -4575,16 +3729,11 @@ def run_conversation( provider=agent.provider, api_mode=agent.api_mode, ) - # Aggregator-only usage is retained for cost pricing: MoA - # advisor tokens must be priced at each advisor's OWN model - # rate, not the aggregator's, so they are added as dollars - # (below) rather than folded into the priced usage. + # Aggregator-only usage kept for pricing: advisor tokens are priced + # at each advisor's OWN model rate and added as dollars below. aggregator_usage = canonical_usage - # MoA: fold the reference (advisor) fan-out's token usage - # into this turn's REPORTED token counts. MoA runs advisors - # before the aggregator and returns only the aggregator's - # usage, so without this the entire advisor spend — usually - # the bulk of a MoA turn — is invisible in token counts. + # MoA: fold advisor fan-out usage into REPORTED token counts — only + # aggregator usage is returned, so advisor spend would be invisible. _moa_ref_cost = None _moa_client = getattr(agent, "client", None) if _moa_client is not None and hasattr(_moa_client, "consume_reference_usage"): @@ -4594,14 +3743,9 @@ def run_conversation( canonical_usage = canonical_usage + _ref_usage except Exception as _moa_acct_exc: # pragma: no cover - defensive logger.debug("MoA reference usage accounting failed: %s", _moa_acct_exc) - # Flush the full-turn MoA trace (references + aggregator I/O) - # to disk when moa.save_traces is on. No-op otherwise and - # for non-MoA clients. Uses the live session_id so traces - # land in the right per-session file. On the streaming path - # the aggregator's output wasn't captured inline (its raw - # token stream went to the live consumer), so pass the - # resolved streamed acting text as a fallback — makes the - # trace self-contained instead of only pointing at state.db. + # Flush the full-turn MoA trace when moa.save_traces is on; on the + # streaming path pass the streamed acting text so the trace is self- + # contained. if _moa_client is not None and hasattr(_moa_client, "consume_and_save_trace"): try: _agg_streamed_text = ( @@ -4616,10 +3760,8 @@ def run_conversation( prompt_tokens = canonical_usage.prompt_tokens completion_tokens = canonical_usage.output_tokens total_tokens = canonical_usage.total_tokens - # Forward canonical token + cache buckets so context engines - # can make decisions on cache hit ratios / reasoning costs, - # not just legacy aggregate tokens. Legacy keys stay for - # back-compat with engines that only read prompt/completion/total. + # Forward canonical token + cache buckets for context engines; + # legacy keys stay for back-compat. usage_dict = { "prompt_tokens": prompt_tokens, "completion_tokens": completion_tokens, @@ -4630,10 +3772,9 @@ def run_conversation( "cache_write_tokens": canonical_usage.cache_write_tokens, "reasoning_tokens": canonical_usage.reasoning_tokens, } - # Capture the boundary latch before update_from_response() - # consumes it. Only a real provider prompt count for the - # request immediately following a completed compaction can - # prove that attempt effective and rearm the shared budget. + # Capture the boundary latch before update_from_response() consumes + # it: only the real prompt count right after a compaction rearms the + # budget. _completed_compaction_pending = bool( getattr( agent.context_compressor, @@ -4642,18 +3783,9 @@ def run_conversation( ) ) agent.context_compressor.update_from_response(usage_dict) - # Usage-anchored context accounting: snapshot this - # response's exact provider-reported usage against the - # durable transcript. Later context-size checks anchor on - # this and estimate only the messages appended since, - # instead of re-estimating the whole history with - # heuristics. Main-loop responses ONLY — MoA advisor and - # auxiliary calls never reach this site, so they cannot - # pollute the anchor. A usage-less response leaves the - # previous anchor in place (still valid for its base). - # MoA note: use the pre-fold aggregator usage — the folded - # canonical figure adds advisor fan-out tokens that were - # never part of THIS conversation's prompt. + # Usage-anchored accounting: snapshot exact provider usage against + # the durable transcript; main-loop ONLY. MoA uses pre-fold + # aggregator usage. _new_anchor = capture_usage_anchor( aggregator_usage.prompt_tokens, aggregator_usage.output_tokens, @@ -4661,18 +3793,9 @@ def run_conversation( ) if _new_anchor is not None: agent._usage_anchor = _new_anchor - # Turn-base anchor for display surfaces: the FIRST - # response of a turn carries minimal current-turn - # reasoning replay, so its prompt_tokens approximate - # the durable transcript cost (what the next turn - # inherits). Later same-turn responses inflate - # prompt_tokens with replayed thinking + tool - # scaffolding that evaporates at the turn boundary — - # anchoring the context meter here instead of on the - # last response removes the end-of-turn sawtooth - # (850K mid-loop -> 600K next turn) that users read - # as a broken compaction. Display-only: compression - # trigger math keeps using real last-request usage. + # Anchor the display meter on the turn's FIRST response: + # later same-turn responses inflate prompt_tokens with replayed + # thinking. Display-only; compression math uses real usage. if api_call_count == 1: agent._turn_base_usage_anchor = _new_anchor _compression_threshold = int( @@ -4694,43 +3817,27 @@ def run_conversation( max_compression_attempts, ) compression_attempts = 0 - # Provider-confirmed recovery also invalidates the - # insufficient-progress preflight state: with the - # prompt proven back below the threshold, a prior - # "insufficient progress" verdict (and the stale - # pressure reading it would be compared against) - # describes a request shape that no longer exists. - # Left armed, _preflight_compression_blocked keeps the - # pre-API gate dark for the rest of the turn even - # though the attempt budget was just rearmed, so a - # later pressure spike would grow unchecked until the - # provider's overflow handler fired. + # Confirmed recovery also clears the stale insufficient-progress + # verdict, else _preflight_compression_blocked stays armed all + # turn and a later pressure spike grows unchecked. _preflight_compression_blocked = False _last_preflight_pressure = None - # Stash this response's canonical usage so the post-turn - # on_turn_complete() observation hook can forward it (the - # same dict shape passed to update_from_response). A turn - # may make several API calls; the engine's per-turn signal - # of interest is the cost/size of the latest assembled - # request, so we keep the most recent call's usage. + # Stash canonical usage for on_turn_complete() (same shape as + # update_from_response); keep the latest call's — last request. agent._last_turn_usage = dict(usage_dict) elif getattr( agent.context_compressor, "awaiting_real_usage_after_compression", False, ): - # A response with no usage cannot adjudicate whether the - # prior compaction cleared the threshold. Consume the pending - # verdict now so a much later, unrelated reading is not - # charged to that old compaction, and so preflight deferral - # does not remain latched indefinitely. + # No usage -> cannot adjudicate the prior compaction; consume the + # pending verdict so later readings aren't charged to it and + # preflight deferral isn't latched indefinitely. agent.context_compressor.update_from_response({}) if hasattr(response, 'usage') and response.usage: - # Cache discovered context length after successful call. - # Only persist limits confirmed by the provider (parsed - # from the error message), not guessed probe tiers. + # Persist only provider-confirmed context lengths, not probe tiers. if getattr(agent.context_compressor, "_context_probed", False): ctx = agent.context_compressor.context_length if getattr(agent.context_compressor, "_context_probe_persistable", False): @@ -4770,14 +3877,9 @@ def run_conversation( api_duration, _cache_pct, ) - # On the MoA path, agent.model/provider are the virtual - # preset name ("closed") and "moa", which have no pricing - # entry — estimating against them returns None and silently - # drops the aggregator's own spend, leaving the session cost - # as advisor-fan-out only (a ~50% undercount when the - # aggregator does the full acting loop). Price the aggregator - # turn at its REAL model/provider, read from the MoA client's - # resolved aggregator slot. + # MoA: agent.model/provider are the virtual preset/"moa" with no + # pricing entry, silently dropping aggregator spend. Price at the + # REAL model/provider from the MoA client's aggregator slot. _agg_cost_model = agent.model _agg_cost_provider = agent.provider _agg_cost_base_url = agent.base_url @@ -4805,27 +3907,18 @@ def run_conversation( agent.session_cost_status = cost_result.status agent.session_cost_source = cost_result.source - # Persist token counts to session DB for /insights. - # Do this for every platform with a session_id so non-CLI - # sessions (gateway, cron, delegated runs) cannot lose - # token/accounting data if a higher-level persistence path - # is skipped or fails. Gateway/session-store writes use - # absolute totals, so they safely overwrite these per-call - # deltas instead of double-counting them. + # Persist per-call token deltas for any session_id so non-CLI runs + # can't lose accounting; gateway/session-store writes use absolute + # totals and safely overwrite these deltas. if agent._session_db and agent.session_id: try: - # Ensure the session row exists before attempting UPDATE. - # Under concurrent load (cron/kanban), the initial - # _ensure_db_session() may have failed due to SQLite - # locking. Retry here so per-call token deltas are - # not silently lost (UPDATE on a non-existent row - # affects 0 rows without error). + # Ensure the row exists: under concurrent SQLite load the + # initial _ensure_db_session() may fail, and UPDATE on a + # missing row silently affects 0 rows. if not agent._session_db_created: agent._ensure_db_session() - # Per-call cost delta = aggregator cost + MoA - # advisor cost (each priced at its own rate). Folded - # here so state.db's estimated_cost_usd includes the - # full MoA spend, matching the folded token counts. + # Cost delta = aggregator + MoA advisor cost so state.db's + # estimated_cost_usd matches the folded token counts. _cost_delta = None if cost_result.amount_usd is not None: _cost_delta = float(cost_result.amount_usd) @@ -4834,11 +3927,8 @@ def run_conversation( _cost_delta = (_cost_delta or 0.0) + float(_moa_ref_cost) except (TypeError, ValueError): # pragma: no cover pass - # Enqueued, not written: the background writer - # applies the delta off the turn thread (a cold - # state.db UPDATE here stalled the tool loop for - # up to hundreds of ms per API call). Drained at - # turn finalize via _persist_session. + # Enqueued, not written: a cold state.db UPDATE here stalled + # the tool loop. Drained at finalize via _persist_session. agent._session_db.queue_token_counts( agent.session_id, input_tokens=canonical_usage.input_tokens, @@ -4857,9 +3947,7 @@ def run_conversation( api_call_count=1, ) except Exception as e: - # Log token persistence failures so they're - # visible in agent.log — silent loss here is - # the root cause of undercounted analytics. + # Log failures — silent loss here undercounts analytics. logger.debug( "Token persistence failed (session=%s, tokens=%d): %s", agent.session_id, total_tokens, e, @@ -4868,17 +3956,9 @@ def run_conversation( if agent.verbose_logging: logging.debug(f"Token usage: prompt={usage_dict['prompt_tokens']:,}, completion={usage_dict['completion_tokens']:,}, total={usage_dict['total_tokens']:,}") - # Surface cache hit stats for any provider that reports - # them — not just those where we inject cache_control - # markers. OpenAI/Kimi/DeepSeek/Qwen all do automatic - # server-side prefix caching and return - # ``prompt_tokens_details.cached_tokens``; users - # previously could not see their cache % because this - # line was gated on ``_use_prompt_caching``, which is - # only True for Anthropic-style marker injection. - # ``canonical_usage`` is already normalised from all - # three API shapes (Anthropic / Codex / OpenAI-chat) - # so we can rely on its values directly. + # Report cache stats for any provider that returns + # ``prompt_tokens_details.cached_tokens``, not only when we inject + # cache_control markers. ``canonical_usage`` is already normalised. cached = canonical_usage.cache_read_tokens written = canonical_usage.cache_write_tokens prompt = usage_dict["prompt_tokens"] @@ -4891,14 +3971,9 @@ def run_conversation( ) _retry.has_retried_429 = False # Reset on success - # Note: don't clear the retry buffer here — an "API call - # success" only means we got bytes back, not that we got - # usable content. Empty responses still loop through the - # empty-retry path below; the buffer is cleared when - # genuinely successful content is detected later (~L4127). - # Clear Nous rate limit state on successful request — - # proves the limit has reset and other sessions can - # resume hitting Nous. + # Don't clear the retry buffer: bytes back != usable content; it is + # cleared once genuine content lands. Clearing Nous rate-limit state + # proves the limit reset so other sessions may resume. if agent.provider == "nous": try: from agent.nous_rate_guard import clear_nous_rate_limit @@ -4921,22 +3996,17 @@ def run_conversation( if agent.thinking_callback: agent.thinking_callback("") if agent._has_pending_redirect(): - # redirect() deliberately used the interrupt machinery to - # cancel only this provider request. Keep its correction - # queued, clear the cancellation bit, and let the outer - # loop rebuild a clean request tail. Never materialize - # incomplete signed/encrypted reasoning items. + # redirect() cancelled only this request: keep the correction + # queued, clear the cancellation bit, let the outer loop rebuild. + # Never materialize incomplete signed/encrypted reasoning items. if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break api_elapsed = time.time() - api_start_time agent._vprint(f"{agent.log_prefix}⚡ Interrupted during API call.", force=True) interrupted = True - # Preserve any assistant text already streamed to the user - # before the stop landed. Dropping it leaves history with no - # record of the half-finished reply on screen, so the next turn - # the model "forgets" what it just said — exactly what users hit - # when they stop to redirect mid-response. + # Keep assistant text already streamed before the stop, else the next + # turn has no record of the half-finished reply. _partial = agent._strip_think_blocks( getattr(agent, "_current_streamed_assistant_text", "") or "" ).strip() @@ -4957,35 +4027,21 @@ def run_conversation( if agent.thinking_callback: agent.thinking_callback("") - # ----------------------------------------------------------- - # UnicodeEncodeError recovery. Two common causes: - # 1. Lone surrogates (U+D800..U+DFFF) from clipboard paste - # (Google Docs, rich-text editors) — sanitize and retry. - # 2. ASCII codec on systems with LANG=C or non-UTF-8 locale - # (e.g. Chromebooks) — any non-ASCII character fails. - # Detect via the error message mentioning 'ascii' codec. - # We sanitize messages in-place and may retry twice: - # first to strip surrogates, then once more for pure - # ASCII-only locale sanitization if needed. - # ----------------------------------------------------------- + # UnicodeEncodeError recovery: lone surrogates (clipboard paste) or an + # ASCII codec under a non-UTF-8 locale. Sanitize in-place; at most two + # retries (surrogate strip, then ASCII-only). if isinstance(api_error, UnicodeEncodeError) and getattr(agent, '_unicode_sanitization_passes', 0) < 2: _err_str = str(api_error).lower() _is_ascii_codec = "'ascii'" in _err_str or "ascii" in _err_str - # Detect surrogate errors — utf-8 codec refusing to - # encode U+D800..U+DFFF. The error text is: - # "'utf-8' codec can't encode characters in position - # N-M: surrogates not allowed" + # Surrogate errors: utf-8 refusing U+D800..U+DFFF + # ("surrogates not allowed"). _is_surrogate_error = ( "surrogate" in _err_str or ("'utf-8'" in _err_str and not _is_ascii_codec) ) - # Sanitize surrogates from both the canonical `messages` - # list AND `api_messages` (the API-copy, which may carry - # `reasoning_content`/`reasoning_details` transformed - # from `reasoning` — fields the canonical list doesn't - # have directly). Also clean `api_kwargs` if built and - # `prefill_messages` if present. Mirrors the ASCII - # codec recovery below. + # Sanitize `messages` AND `api_messages` (which may carry + # `reasoning_content`/`reasoning_details`), plus `api_kwargs` and + # `prefill_messages` if present. Mirrors the ASCII recovery below. _surrogates_found = _sanitize_messages_surrogates(messages) if isinstance(api_messages, list): if _sanitize_messages_surrogates(api_messages): @@ -4996,13 +4052,8 @@ def run_conversation( if isinstance(getattr(agent, "prefill_messages", None), list): if _sanitize_messages_surrogates(agent.prefill_messages): _surrogates_found = True - # Gate the retry on the error type, not on whether we - # found anything — _force_ascii_payload / the extended - # surrogate walker above cover all known paths, but a - # new transformed field could still slip through. If - # the error was a surrogate encode failure, always let - # the retry run; the proactive sanitizer at line ~8781 - # runs again on the next iteration. Bounded by + # Gate the retry on the error type, not on whether anything was + # found — a new transformed field could slip through. Bounded by # _unicode_sanitization_passes < 2 (outer guard). if _surrogates_found or _is_surrogate_error: agent._unicode_sanitization_passes += 1 @@ -5017,20 +4068,14 @@ def run_conversation( continue if _is_ascii_codec: agent._force_ascii_payload = True - # ASCII codec: the system encoding can't handle - # non-ASCII characters at all. Sanitize all - # non-ASCII content from messages/tool schemas and retry. - # Sanitize both the canonical `messages` list and - # `api_messages` (the API-copy built before the retry - # loop, which may contain extra fields like - # reasoning_content that are not in `messages`). + # ASCII codec: strip all non-ASCII from messages/tool schemas + # and retry — both `messages` and `api_messages` (which may + # carry extra fields like reasoning_content). _messages_sanitized = _sanitize_messages_non_ascii(messages) if isinstance(api_messages, list): _sanitize_messages_non_ascii(api_messages) - # Also sanitize the last api_kwargs if already built, - # so a leftover non-ASCII value in a transformed field - # (e.g. extra_body, reasoning_content) doesn't survive - # into the next attempt via _build_api_kwargs cache paths. + # Also sanitize the last api_kwargs so a non-ASCII transformed + # field doesn't survive via _build_api_kwargs cache paths. if isinstance(api_kwargs, dict): _sanitize_structure_non_ascii(api_kwargs) _prefill_sanitized = False @@ -5063,27 +4108,21 @@ def run_conversation( if isinstance(_default_headers, dict): _headers_sanitized = _sanitize_structure_non_ascii(_default_headers) - # Sanitize the API key — non-ASCII characters in - # credentials (e.g. ʋ instead of v from a bad - # copy-paste) cause httpx to fail when encoding - # the Authorization header as ASCII. This is the - # most common cause of persistent UnicodeEncodeError - # that survives message/tool sanitization (#6843). + # Sanitize the API key: non-ASCII in credentials makes httpx + # fail encoding the Authorization header — the usual persistent + # cause after message/tool sanitization (#6843). _credential_sanitized = False _raw_key = getattr(agent, "api_key", None) or "" - # Entra ID bearer providers are callables — their - # minted JWTs are always ASCII, so no sanitization - # is needed (and ``_strip_non_ascii`` would crash - # on a callable input). + # Entra ID bearer providers are callables minting ASCII JWTs; + # skip (``_strip_non_ascii`` would crash on a callable). if _raw_key and isinstance(_raw_key, str): _clean_key = _strip_non_ascii(_raw_key) if _clean_key != _raw_key: agent.api_key = _clean_key if isinstance(getattr(agent, "_client_kwargs", None), dict): agent._client_kwargs["api_key"] = _clean_key - # Also update the live client — it holds its - # own copy of api_key which auth_headers reads - # dynamically on every request. + # Also update the live client — auth_headers reads its + # own api_key copy on every request. if getattr(agent, "client", None) is not None and hasattr(agent.client, "api_key"): agent.client.api_key = _clean_key _credential_sanitized = True @@ -5094,14 +4133,9 @@ def run_conversation( force=True, ) - # Always retry on ASCII codec detection — - # _force_ascii_payload guarantees the full - # api_kwargs payload is sanitized on the - # next iteration (line ~8475). Even when - # per-component checks above find nothing - # (e.g. non-ASCII only in api_messages' - # reasoning_content), the flag catches it. - # Bounded by _unicode_sanitization_passes < 2. + # Always retry on ASCII codec detection: _force_ascii_payload + # sanitizes the full api_kwargs next iteration even when + # checks above find nothing. Bounded by passes < 2. agent._unicode_sanitization_passes += 1 _any_sanitized = ( _messages_sanitized @@ -5124,18 +4158,8 @@ def run_conversation( continue # ── Image-rejection recovery ────────────────────────────── - # Some providers (mlx-lm, text-only endpoints, text-only - # fallbacks on multimodal models) reject any message that - # contains image_url content with a 4xx error like - # "Only 'text' content type is supported." On first hit, - # strip all images from the message list, mark the session - # as vision-unsupported, and retry with text only. - # - # Detection is best-effort English phrase matching — a - # locale-translated or heavily-reworded upstream error - # will bypass this guard and fall through to the normal - # error handler. Expand the phrase list when new - # provider wordings are observed in the wild. + # Some providers 4xx on image_url content: strip images, mark session + # vision-unsupported, retry text-only. English phrase match; extend it. _err_body = "" try: _err_body = str(getattr(api_error, "body", None) or @@ -5145,9 +4169,7 @@ def run_conversation( pass _err_status = getattr(api_error, "status_code", None) _looks_like_image_rejection = _looks_like_image_content_rejection(_err_body) - # 4xx-only gate: never interpret 5xx/timeout as "server - # said no to images" — those are transient and must - # route to the normal retry path. + # 4xx-only gate: 5xx/timeouts are transient and take the retry path. _status_ok = _err_status is None or (400 <= int(_err_status) < 500) if ( getattr(agent, "_vision_supported", True) @@ -5167,11 +4189,8 @@ def run_conversation( continue # ── Bedrock AnthropicBedrock SDK streaming failure ── - # The Anthropic SDK's stream accumulator raises RuntimeError - # "Unexpected event order" when Bedrock returns an error event - # before message_start (throttling, overload, validation). - # Fall back to the native Converse API path for the rest of - # this session — it handles these errors gracefully. Ref: #28156. + # SDK raises "Unexpected event order" when Bedrock errors before + # message_start; fall back to native Converse for this session (#28156). if ( isinstance(api_error, RuntimeError) and "unexpected event order" in str(api_error).lower() @@ -5195,15 +4214,8 @@ def run_conversation( error_context = agent._extract_api_error_context(api_error) # ── Interpreter finalization: abandon immediately ── - # The process is exiting (TUI quit, SIGTERM, one-shot done) - # while this turn — typically the post-turn review fork's - # daemon thread — is mid-flight. Retries, credential - # rotation, and fallbacks are all futile ("cannot schedule - # new futures..."), and the buffered ⚠️/❌ retry trace spams - # the shell after the TUI already exited. End the turn with - # a single log line: no print, no traceback, no debug dump, - # no retry. Same class as cron delivery (#55924/#58720) and - # concurrent tool submission — shared predicate. + # Process is exiting mid-flight: retries/rotation/fallbacks are futile + # and the retry trace spams the shell. One log line; shared predicate. from tools.interpreter_shutdown import interpreter_shutting_down if interpreter_shutting_down(api_error): @@ -5287,12 +4299,8 @@ def run_conversation( if recovered_with_pool: continue - # Image-too-large recovery: shrink oversized native image - # parts in-place and retry once. Triggered by Anthropic's - # per-image 5 MB ceiling (400 with "image exceeds 5 MB - # maximum") or any other provider that complains about - # image size. If shrink fails or a second attempt still - # fails, fall through to normal error handling. + # Image-too-large recovery: shrink oversized native image parts + # in-place and retry once; otherwise fall through to normal handling. if ( classified.reason == FailoverReason.image_too_large and not _retry.image_shrink_retry_attempted @@ -5315,13 +4323,9 @@ def run_conversation( "or shrink didn't reduce size; surfacing original error." ) - # Multimodal-tool-content recovery: providers that follow - # the OpenAI spec strictly (tool message content must be a - # string) reject our list-type content with a 400. Strip - # image parts from any list-type tool messages, mark the - # (provider, model) as no-list-tool-content for the rest - # of this session so future tool results preemptively - # downgrade, and retry once. See issue #27344. + # Multimodal-tool-content recovery: strict OpenAI-spec providers 400 + # on list-type tool content. Strip images, mark (provider, model) + # no-list-tool-content for the session, retry once (#27344). if ( classified.reason == FailoverReason.multimodal_tool_content_unsupported and not _retry.multimodal_tool_content_retry_attempted @@ -5340,20 +4344,12 @@ def run_conversation( "messages with image parts found; surfacing original error." ) - # Image-corrupt recovery: the provider decoded the request but - # rejected the image bytes themselves (e.g. xAI's "Invalid PNG - # image." on a re-serialized image part from replayed - # history). Shrinking corrupt bytes doesn't help, so strip the - # image parts and retry once instead of routing through the - # shrink path above. See issue #69078. + # Image-corrupt recovery: provider rejected the image bytes; shrinking + # can't help, so strip image parts and retry once (#69078). if classified.reason == FailoverReason.image_corrupt: - # Strip ONLY the per-call payload copy. api_messages rows - # are shallow copies of canonical history, and the strip - # replaces msg["content"] rather than mutating the shared - # parts list — so canonical messages keep their images. - # A transient provider rejection must not permanently - # erase history (#69104 sweeper review; the copy-on-write - # contract from e762a5a473). + # Strip ONLY the per-call copy: replacing msg["content"] on the + # shallow api_messages rows keeps canonical history's images + # (copy-on-write; transient rejection must not erase history). _imgs_removed = False if isinstance(api_messages, list): _imgs_removed = _strip_images_from_messages(api_messages) @@ -5370,15 +4366,9 @@ def run_conversation( "strip; surfacing original error." ) - # Anthropic OAuth subscription rejected the 1M-context beta - # header ("long context beta is not yet available for this - # subscription"). Disable the beta for the rest of this - # session, rebuild the client, and retry once. 1M-capable - # subscriptions never hit this branch — they accept the - # beta and keep full 1M context. See PR #17680 for the - # original report (we chose reactive recovery over the - # proposed unconditional omit so capable subscriptions - # don't silently lose the capability). + # Anthropic OAuth subscription rejected the 1M-context beta: disable it + # for this session, rebuild the client, retry once. Reactive so capable + # subscriptions keep full 1M context (#17680). if ( classified.reason == FailoverReason.oauth_long_context_beta_forbidden and agent.api_mode == "anthropic_messages" @@ -5431,9 +4421,8 @@ def run_conversation( if agent._try_refresh_nous_client_credentials(force=True): agent._buffer_vprint(f"🔐 Nous agent key refreshed after 401. Retrying request...") continue - # Credential refresh didn't help — show diagnostic info. - # Most common causes: Portal OAuth expired/revoked, - # account out of credits, or agent key blocked. + # Refresh didn't help: likely Portal OAuth expired/revoked, + # no credits, or agent key blocked. from hermes_constants import display_hermes_home as _dhh_fn _dhh = _dhh_fn() _body_text = "" @@ -5478,11 +4467,8 @@ def run_conversation( key = agent._anthropic_api_key print(f"{agent.log_prefix}🔐 Anthropic 401 — authentication failed.") if is_token_provider(key): - # Azure Foundry Entra ID — the bearer token is - # minted per-request by an httpx event hook on a - # custom http_client passed to the SDK. The 401 - # means Azure rejected the JWT (RBAC role missing, - # az login expired, IMDS unreachable, etc.). + # Azure Foundry Entra ID: JWT minted per-request by an httpx + # hook; 401 = Azure rejected it (RBAC, az login, IMDS). print(f"{agent.log_prefix} Auth method: Microsoft Entra ID (httpx event hook)") print(f"{agent.log_prefix} Run `hermes doctor` for credential-chain diagnostics, or") print(f"{agent.log_prefix} `az login` if your developer session expired.") @@ -5500,34 +4486,9 @@ def run_conversation( print(f"{agent.log_prefix} • Legacy cleanup: hermes config set ANTHROPIC_TOKEN \"\"") print(f"{agent.log_prefix} • Clear stale keys: hermes config set ANTHROPIC_API_KEY \"\"") - # Thinking block signature recovery. - # - # Anthropic signs thinking blocks against the full turn - # content. Any upstream mutation (context compression, - # session truncation, message merging) invalidates the - # signature and the API replies HTTP 400 ("invalid - # signature" or "cannot be modified"). Recovery strips - # ``reasoning_details`` so the retry sends no thinking - # blocks at all. One-shot per outer loop. - # - # The strip targets ``api_messages``, which is the - # API-call-time list that ``_build_api_kwargs`` consumes - # on every retry. ``api_messages`` was populated once at - # the start of the turn from shallow copies of - # ``messages``, so mutating it does not touch the - # canonical store. The previous implementation popped - # ``reasoning_details`` from ``messages`` instead, which - # had two problems: ``api_messages`` carried its own - # reference to the field through the shallow copy, so the - # retry's wire payload still included thinking blocks and - # the recovery never reached the API; and the mutation - # persisted into ``state.db`` through any subsequent - # ``_persist_session`` call, permanently corrupting the - # conversation. Future turns would replay the stripped - # state, hit the same 400, and the agent would terminate - # with ``max_retries_exhausted``, often spawning - # cascading compaction-ended sessions chained off the - # corrupted parent. + # Thinking block signature recovery: upstream mutation invalidates + # Anthropic's signature (400). Strip ``reasoning_details`` from + # ``api_messages`` only, never ``messages`` (state.db). One-shot. if ( classified.reason == FailoverReason.thinking_signature and not _retry.thinking_sig_retry_attempted @@ -5552,19 +4513,8 @@ def run_conversation( continue # ── Invalid encrypted reasoning replay recovery ─────── - # OpenAI Responses API surfaces (and some compatible relays) - # return HTTP 400 ``invalid_encrypted_content`` when a - # replayed ``codex_reasoning_items`` blob from a previous - # turn fails verification (provider rotated the encryption - # key, the route doesn't actually persist reasoning state, - # etc.). Recovery: disable replay for the rest of the - # session, strip cached items from history, retry once. - # One-shot — if a second 400 fires we fall through to the - # normal retry/backoff path. Only fires for codex_responses - # mode with at least one assistant message that has cached - # ``codex_reasoning_items``; without replay state, the - # error is unrelated to our cache so the normal retry path - # handles it (the provider is rejecting something else). + # 400 ``invalid_encrypted_content`` on a stale ``codex_reasoning_items`` + # blob: disable replay for the session, strip cached items, retry once. if ( classified.reason == FailoverReason.invalid_encrypted_content and not _retry.invalid_encrypted_content_retry_attempted @@ -5595,14 +4545,8 @@ def run_conversation( continue # ── Native compaction rejection recovery ────────────── - # Provider explicitly rejected the ``context_management`` - # field (structured 400 naming the param). One-shot: turn - # native compaction off for the rest of the session and - # retry — the next _build_api_kwargs re-resolves the gate - # and omits the field, and Hermes' local compression takes - # over as the sole owner. Generic 4xx/5xx/timeouts do NOT - # match (see is_native_compaction_rejection) and take the - # normal retry path. + # Structured 400 naming ``context_management``: disable native + # compaction for the session, retry once; local compression takes over. if ( agent.api_mode == "codex_responses" and not _retry.native_compaction_reject_retry_attempted @@ -5628,14 +4572,8 @@ def run_conversation( continue # ── llama.cpp grammar-parse recovery ────────────────── - # llama.cpp's ``json-schema-to-grammar`` converter rejects - # regex escape classes (``\d``, ``\w``, ``\s``) and most - # ``format`` values in tool schemas. MCP servers emit - # these routinely for date/phone/email params. Recovery: - # strip ``pattern``/``format`` from ``agent.tools`` and - # retry once. We keep the keywords by default so cloud - # providers get the full prompting hints; this branch - # fires only for users on llama.cpp's OAI server. + # ``json-schema-to-grammar`` rejects regex escapes and most ``format`` + # values: strip ``pattern``/``format`` from ``agent.tools``, retry once. if ( classified.reason == FailoverReason.llama_cpp_grammar_pattern and not _retry.llama_cpp_grammar_retry_attempted @@ -5703,10 +4641,8 @@ def run_conversation( agent._buffer_vprint(f" 📋 Details: {_err_body_str}") agent._buffer_vprint(f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens") - # Actionable hint for OpenRouter "no tool endpoints" error. - # Buffered like the rest of the retry trace — surfaced only - # if every retry+fallback exhausts. Avoids spamming users - # who recover automatically via fallback. + # OpenRouter "no tool endpoints" hint, buffered with the retry trace + # so it only surfaces if every retry+fallback exhausts. if ( agent._is_openrouter_url() and "support tool use" in error_msg @@ -5725,12 +4661,8 @@ def run_conversation( f" Check which providers support tools: https://openrouter.ai/models/{_model}" ) - # Actionable hint for a bare 404 on a provider whose catalogue - # uses ``vendor/model`` ids. A model id that lost its prefix - # (e.g. ``nemotron-…`` instead of ``nvidia/nemotron-…``) gets - # a content-free "404 page not found" from the provider that - # never names the model, so it reads like an outage or an auth - # failure. Name the real cause and the exact id to use (#78796). + # Bare 404 on a ``vendor/model`` catalogue usually means the id lost its + # prefix; the provider never names the model, so we do (#78796). if getattr(api_error, "status_code", None) == 404: try: from hermes_cli.model_normalize import suggest_prefixed_model_id @@ -5749,9 +4681,8 @@ def run_conversation( # Check for interrupt before deciding to retry if agent._interrupt_requested: - # Preserve a pending redirect (mid-stream correction): the - # user is steering, not stopping. Rebuild the turn from the - # correction instead of aborting with a dead-end interrupt. + # Preserve a pending redirect: the user is steering, not stopping + # — rebuild the turn from the correction instead of aborting. if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break @@ -5768,32 +4699,12 @@ def run_conversation( "interrupted": True, } - # Check for 413 payload-too-large BEFORE generic 4xx handler. - # A 413 is a payload-size error — the correct response is to - # compress history and retry, not abort immediately. + # Check 413 BEFORE the generic 4xx handler: compress + retry, not abort. status_code = getattr(api_error, "status_code", None) # ── Respect disabled auto-compaction on overflow ────── - # Ported from anomalyco/opencode#30749. When the user has - # turned auto-compaction off (``compression.enabled: false``), - # NO automatic compaction trigger may fire — including the - # provider/request-size overflow recovery paths below - # (long-context-tier 429, 413 payload-too-large, and - # context-overflow). Without this guard the proactive - # threshold path correctly honours the setting (see the - # preflight check and the post-response ``should_compress`` - # gate) but a provider overflow error would still silently - # compress + rotate the session, bypassing the user's - # explicit choice. Surface a terminal error instead so the - # user can compact manually (``/compress``), start fresh - # (``/new``), switch to a larger-context model, or reduce - # attachments. Forced compaction via ``/compress`` - # (``force=True``) is unaffected — it never reaches this loop. - # - # Output-cap errors (max_tokens too large) are NOT input - # overflow — the recovery is a max_tokens-only retry that - # does not require compression. Exempt them from this guard - # so the retry still fires even when compression is disabled. + # ``compression.enabled: false`` forbids every automatic trigger, incl. + # these overflow recovery paths; error out. Output-cap errors exempt. _overflow_reasons = { FailoverReason.long_context_tier, FailoverReason.payload_too_large, @@ -5841,12 +4752,8 @@ def run_conversation( } # ── Anthropic Sonnet long-context tier gate ─────────── - # Anthropic returns HTTP 429 "Extra usage is required for - # long context requests" when a Claude Max (or similar) - # subscription doesn't include the 1M-context tier. This - # is NOT a transient rate limit — retrying or switching - # credentials won't help. Reduce context to 200k (the - # standard tier) and compress. + # 429 "Extra usage is required for long context requests" is a + # subscription-tier limit, not transient: cap at 200k and compress. if classified.reason == FailoverReason.long_context_tier: _reduced_ctx = 200000 compressor = agent.context_compressor @@ -5864,10 +4771,8 @@ def run_conversation( # compressor (plugin engines manage their own). if hasattr(compressor, "_context_probed"): compressor._context_probed = True - # Don't persist — this is a subscription-tier - # limitation, not a model capability. If the - # user later enables extra usage the 1M limit - # should come back automatically. + # Don't persist — subscription-tier limit, not a model + # capability; 1M should return if extra usage is enabled. compressor._context_probe_persistable = False agent._buffer_vprint( f"⚠️ Anthropic long-context tier " @@ -5895,38 +4800,25 @@ def run_conversation( ) ) time.sleep(2) - # Same class as the generic overflow handler below: - # the provider proved the request does not fit the - # (now-reduced) window, and row count alone is not - # proof the rebuilt request does. Recheck the - # complete request before the next provider call. + # Provider proved the request doesn't fit the reduced + # window; row count isn't proof the rebuilt one does. + # Recheck the complete request before the next call. _provider_overflow_recovery_pending = True _retry.restart_with_compressed_messages = True break # Fall through to normal error handling if compression # is exhausted or didn't help. - # Eager fallback for rate-limit errors (429 or quota exhaustion) - # and transport errors (connection failure / timeout / provider - # overloaded). Rate limits and billing: switch immediately — - # the primary provider won't recover within the retry window. - # Transport errors: allow 1 retry first (transient hiccups - # recover), then fall back if the provider is truly unreachable. + # Eager fallback: rate-limit/billing switch immediately (primary won't + # recover in the retry window); transport errors get 1 retry first. is_rate_limited = classified.reason in { FailoverReason.rate_limit, FailoverReason.billing, FailoverReason.upstream_rate_limit, } - # Relay-wrapped output-cap errors: some gateways wrap an - # upstream "[400]: max_tokens (...) exceeds model's maximum - # output tokens (...)" as HTTP 429, which classifies as - # rate_limit. The failure is a deterministic request-shape - # problem — falling back to another provider (or burning - # generic retries) can't fix it, but the output-cap clamp - # below can, in one retry (#72281). Parse once here; the - # result gates both the eager-fallback exemption and the - # widened is_context_length_error entry, and is reused as - # available_out inside the handler. + # Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only + # the max_tokens clamp fixes it (#72281). Parsed once; gates the + # eager-fallback exemption and the overflow entry below. _wrapped_output_cap_budget = ( parse_available_output_tokens_from_error(error_msg) if classified.reason == FailoverReason.rate_limit @@ -5936,13 +4828,9 @@ def run_conversation( FailoverReason.timeout, FailoverReason.overloaded, } - # Z.AI Coding Plan GLM-5.2 overload 429s classify as - # `overloaded` (to spare the credential pool), but `overloaded` - # is excluded from `is_rate_limited` — the gate for the adaptive - # Z.AI backoff below. Detect the overload directly so its - # long-backoff schedule runs, and raise the retry ceiling so the - # long tier (30/60/90/120s) is reachable. See - # zai_coding_overload_retry_ceiling() for the ceiling rationale. + # Z.AI overload 429s classify `overloaded`, which `is_rate_limited` + # excludes. Detect directly so the long backoff runs, and raise the + # ceiling to reach it (see zai_coding_overload_retry_ceiling()). _is_zai_coding_overload = is_zai_coding_overload_error( base_url=str(_base), model=_model, error=api_error ) @@ -5953,14 +4841,9 @@ def run_conversation( or (_is_transport_failure and retry_count >= 2) ) if _should_fallback and agent._fallback_index < len(agent._fallback_chain): - # Don't eagerly fallback if credential pool rotation may - # still recover. See _pool_may_recover_from_rate_limit - # for the single-credential-pool exception. Fixes #11314. - # - # Exception: an upstream-aggregator 429 — the credential - # pool can't help when the *upstream* model (DeepSeek, - # etc.) is throttling OpenRouter, so always fall back to a - # different model regardless of pool state. + # No eager fallback while credential pool rotation may recover + # (_pool_may_recover_from_rate_limit, #11314). Exception: an + # upstream-aggregator 429 — the pool can't help, always fall back. _is_upstream = classified.reason == FailoverReason.upstream_rate_limit pool_may_recover = ( False if _is_upstream @@ -6005,20 +4888,8 @@ def run_conversation( break # ── Auth-failure provider failover ─────────────────────── - # A 401/403 that survives the per-provider credential-refresh - # attempt above (each guarded by its own - # ``*_auth_retry_attempted`` flag) means the active provider's - # credential or endpoint is broken in a way refreshing can't - # fix (revoked OAuth, blocked/expired key, an account pinned to - # a dead/staging endpoint). Previously the loop only printed - # "switch providers manually" advice and fell through, so a - # user with a configured fallback chain kept thrashing on the - # same dead credential every turn instead of failing over. - # Escalate to the fallback chain here, mirroring the rate- - # limit/billing failover above. When no fallback is configured - # (or the chain is exhausted), _try_activate_fallback returns - # False and we fall through to the existing terminal handling - # + provider-specific troubleshooting guidance unchanged. + # A 401/403 surviving credential refresh means a broken credential or + # endpoint: escalate to the fallback chain; False -> terminal handling. if ( classified.is_auth and not _retry.auth_failover_attempted @@ -6039,25 +4910,8 @@ def run_conversation( break # ── Nous Portal: record rate limit & skip retries ───── - # When Nous returns a 429 that is a genuine account- - # level rate limit, record the reset time to a shared - # file so ALL sessions (cron, gateway, auxiliary) know - # not to pile on, then skip further retries -- each - # one burns another RPH request and deepens the hole. - # The retry loop's top-of-iteration guard will catch - # this on the next pass and try fallback or bail. - # - # IMPORTANT: Nous Portal multiplexes multiple upstream - # providers (DeepSeek, Kimi, MiMo, Hermes). A 429 can - # also mean an UPSTREAM provider is out of capacity - # for one specific model -- transient, clears in - # seconds, nothing to do with the caller's quota. - # Tripping the cross-session breaker on that would - # block every Nous model for minutes. We use - # ``is_genuine_nous_rate_limit`` to tell the two - # apart via the 429's own x-ratelimit-* headers and - # the last-known-good state captured on the previous - # successful response. + # A genuine account-level 429 is recorded to a shared file so ALL + # sessions back off; is_genuine_nous_rate_limit excludes upstream 429s. if ( is_rate_limited and agent.provider == "nous" @@ -6094,29 +4948,18 @@ def run_conversation( except Exception: pass if _genuine_nous_rate_limit: - # Re-enter the loop exactly once so the - # top-of-loop Nous guard handles fallback or - # bails cleanly. (Setting retry_count to - # max_retries would make the while condition - # false immediately and the guard would never - # run -- no fallback, generic exhaustion error.) + # Re-enter the loop exactly once so the top-of-loop Nous guard + # runs (retry_count = max_retries would skip it entirely). retry_count = max(0, max_retries - 1) continue - # Upstream capacity 429: fall through to normal - # retry logic. A different model (or the same - # model a moment later) will typically succeed. + # Upstream capacity 429: normal retry logic will typically succeed. is_payload_too_large = ( classified.reason == FailoverReason.payload_too_large ) - # Actionable hint for GitHub Models (Azure) 413 errors. - # The free tier enforces a hard 8K token cap per request, - # which Hermes' system prompt + tool schemas alone exceed. - # Compression can't help — the floor is the system prompt - # itself, not the conversation — so surface a clear "not - # compatible" message instead of looping into three futile - # compression attempts. + # GitHub Models free tier caps requests at 8K tokens, under the system + # prompt + tool schema floor; compression can't help, so say so. if ( status_code == 413 and isinstance(agent.base_url, str) @@ -6166,18 +5009,9 @@ def run_conversation( agent._buffer_status(f"⚠️ Request payload too large (413) — compression attempt {compression_attempts}/{max_compression_attempts}...") original_len = len(messages) - # A 413 is a BYTE-size error, so this branch scores - # progress in BYTES of the serialized messages payload — - # exact and free — never the token estimate. The - # estimator prices every image at a flat per-image token - # cost (see estimate_messages_tokens_rough) so screenshots - # don't trigger premature compaction; that deliberate - # byte-blindness means compaction can free megabytes of - # base64 (real case: two vision results = 96.6% of the - # request body but ~3.7% of the estimate) while the token - # delta stays under any threshold. Token-scored progress - # here burned all attempts on "no progress" and wedged - # the session permanently. (#88960 / #47339) + # A 413 is a BYTE-size error: score progress in payload bytes, + # never the token estimate, which is deliberately byte-blind to + # images and wedged sessions on "no progress" (#88960 / #47339). original_bytes = serialized_messages_bytes(messages) _overflow_input = messages # Option A (LCM issue 441): overhead-aware request size so recovery arms on the @@ -6186,30 +5020,22 @@ def run_conversation( messages, system_message, approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, - # #100661: the provider proved the request does not fit. - # Ignore the summary-failure cooldown for this ONE - # attempt (bounded by max_compression_attempts) instead - # of deferring every turn until the ladder lapses. + # Provider proved the request doesn't fit: ignore the + # summary-failure cooldown for this ONE attempt (#100661). bypass_cooldown=True, ) if messages is _overflow_input and compression_skipped_due_to_lock(agent): - # #69870 lock-skip: the provider proved the request - # does not fit, but this compression pass no-oped only - # because another path holds the session's compression - # lock. Temporary defer, not exhaustion — refund the - # attempt and end the turn softly so the gateway does - # NOT auto-reset the session (#9893/#35809). + # Lock-skip: another path holds the compression lock. A + # temporary defer, not exhaustion — refund the attempt and + # end softly so the gateway does NOT auto-reset (#69870). compression_attempts -= 1 agent._persist_session(messages, conversation_history) return _compression_deferred_result( agent, messages, api_call_count ) if messages is _overflow_input and compression_blocked_transiently(agent): - # #97488 transient-block: compression no-oped because a - # timed guard (host-timeout cooldown / structural - # backoff) is active — a temporary defer, not evidence - # of incompressibility. Never classify it as - # compression_exhausted (gateway auto-reset). + # Transient-block: a timed guard no-oped compression. A + # defer, never compression_exhausted (auto-reset) (#97488). compression_attempts -= 1 agent._persist_session(messages, conversation_history) return _compression_deferred_result( @@ -6220,14 +5046,9 @@ def run_conversation( agent, messages, conversation_history ) - # Re-measure after compression. Same-message-count - # compression (tool-result pruning, in-place summarization) - # can materially reduce request size without reducing the - # message array (#39550), and — the image-dominated case — - # compaction's historical-media aging (#97160) can free - # megabytes of base64 that the token estimate never - # counted. Bytes are the yardstick for a 413; tokens are - # kept only for status display. + # Re-measure: same-count compression and media aging can shrink + # the request without shrinking the array. Bytes are the yardstick + # for a 413; tokens only for status display. new_tokens = estimate_messages_tokens_rough(messages) approx_tokens = new_tokens # update for downstream logging new_bytes = serialized_messages_bytes(messages) @@ -6277,16 +5098,12 @@ def run_conversation( "compression_exhausted": True, } - # Check for context-length errors BEFORE generic 4xx handler. - # The classifier detects context overflow from: explicit error - # messages, generic 400 + large session heuristic (#1630), and - # server disconnect + large session pattern (#2153). + # Check context-length errors BEFORE the generic 4xx handler; the + # classifier also covers 400/disconnect + large-session heuristics. is_context_length_error = ( classified.reason == FailoverReason.context_overflow - # Relay-wrapped output-cap 429s (parsed once above, where - # the eager-fallback exemption is gated) route into the - # output-cap clamp below instead of provider failover or - # generic retries (#72281). + # Relay-wrapped output-cap 429s (parsed above) go to the clamp + # below, not failover or generic retries (#72281). or _wrapped_output_cap_budget is not None ) @@ -6294,26 +5111,14 @@ def run_conversation( compressor = agent.context_compressor old_ctx = compressor.context_length - # ── Distinguish two very different errors ─────────── - # 1. "Prompt too long": the INPUT exceeds the context window. - # Fix: reduce context_length + compress history. - # 2. "max_tokens too large": input is fine, but - # input_tokens + requested max_tokens > context_window. - # Fix: reduce max_tokens (the OUTPUT cap) for this call. - # Do NOT shrink context_length — the window is unchanged. - # - # Note: max_tokens = output token cap (one response). - # context_length = total window (input + output combined). + # Two errors: "prompt too long" = INPUT overflows the window (shrink + # context_length + compress); "max_tokens too large" = input fits + # but input + max_tokens > window (shrink OUTPUT cap only). available_out = parse_available_output_tokens_from_error(error_msg) if available_out is not None: - # This is an output-cap error, not input overflow. - # The provider's available_tokens is the authoritative - # cap for the failed request, so keep it as an upper - # bound. Also estimate the current API request shape - # (system prompt, injected context, tool schemas) because - # Hermes may add API-only content not present in persisted - # messages. Use the smaller budget and apply a small - # safety margin. Do not alter context_length. + # Output-cap error: provider available_tokens is the + # authoritative bound; also estimate the real request shape + # (API-only content), use the smaller minus a margin. request_input_estimate = estimate_request_tokens_rough( api_messages, tools=agent.tools or None, ) @@ -6321,10 +5126,8 @@ def run_conversation( if local_available_out > 0: safe_out = max(1, min(available_out, local_available_out) - 64) else: - # The rough local estimate can overshoot the real - # request size. Fall back to the provider-reported - # budget, which is authoritative for the failed - # request. + # Local estimate can overshoot; fall back to the + # authoritative provider-reported budget. safe_out = max(1, available_out - 64) agent._ephemeral_max_output_tokens = safe_out agent._buffer_vprint( @@ -6354,11 +5157,9 @@ def run_conversation( "failed": True, "compression_exhausted": True, } - # Also compress the message history so the output-cap - # retry does not just spin on max_tokens alone. The - # compressor drops the middle window, freeing enough - # tokens for the total to fit inside context_length. - # (#55546) + # Also compress history so the output-cap retry doesn't spin on + # max_tokens alone; dropping the middle window makes the total + # fit. (#55546) try: original_len = len(messages) original_tokens = estimate_messages_tokens_rough(messages) @@ -6402,14 +5203,9 @@ def run_conversation( _retry.restart_with_compressed_messages = True break - # The error is output-cap-shaped (about max_tokens being - # too large) but the provider's wording didn't let us parse - # the available output budget. Compression CANNOT help here - # — the input already fits; the call fails deterministically - # on the oversized max_tokens. Routing it into compression - # re-sends the same max_tokens, gets the identical 400, and - # death-loops until "cannot compress further" (#55546). - # Fail fast with an actionable message instead of looping. + # Output-cap error with unparseable budget: compression can't help + # (input already fits) and would death-loop on the same 400. Fail + # fast. (#55546) if is_output_cap_error(error_msg): agent._flush_status_buffer() agent._vprint( @@ -6443,12 +5239,9 @@ def run_conversation( "failed": True, } - # Error is about the INPUT being too large. Only reduce - # context_length when the provider explicitly reports the - # real lower limit. If the provider only says "input - # exceeds the context window", keep the configured window - # and try compression; guessing probe tiers can incorrectly - # turn a user-configured 1M window into 256K/128K/64K. + # Input too large: shrink context_length only when the provider + # reports the real limit; else keep the window and compress. Guessed + # probe tiers can turn a configured 1M window into 256K/128K/64K. new_ctx = get_context_length_from_provider_error(error_msg, old_ctx) _provider_lower = (getattr(agent, "provider", "") or "").lower() _base_lower = (getattr(agent, "base_url", "") or "").rstrip("/").lower() @@ -6475,16 +5268,12 @@ def run_conversation( provider=agent.provider, api_mode=agent.api_mode, ) - # Persist an explicit provider-reported limit before - # compression/retry. The next request can be rate - # limited, omit usage, or the process can restart; none - # of those should discard metadata the provider already - # confirmed. Keep the probe flags as a best-effort - # post-success retry if this write cannot complete. + # Persist the provider-reported limit before compression/retry: + # rate limit, missing usage, or restart must not lose confirmed + # metadata. Probe flags remain a fallback if this write fails. save_context_length(agent.model, agent.base_url, new_ctx) - # Context probing flags — only set on built-in - # compressor (plugin engines manage their own). This - # value came from the provider, so it is safe to cache. + # Probe flags only on the built-in compressor (plugin engines + # manage their own); provider-sourced value, so safe to cache. if hasattr(compressor, "_context_probed"): compressor._context_probed = True compressor._context_probe_persistable = True @@ -6523,37 +5312,31 @@ def run_conversation( original_len = len(messages) original_tokens = estimate_messages_tokens_rough(messages) _overflow_input = messages - # Option A (LCM issue 441): pass the OVERHEAD-AWARE request size (msgs + tool - # schemas + system), not the tool-blind message count, so LCM forced-overflow - # recovery arms on the TRUE request that overflowed. See hermes-lcm engine - # _should_force_overflow_recovery. (approx_tokens stays for the status display.) + # Pass the OVERHEAD-AWARE size (msgs + tool schemas + system) so LCM + # forced-overflow recovery arms on the TRUE request; approx_tokens + # stays for status. See hermes-lcm _should_force_overflow_recovery. messages, active_system_prompt = agent._compress_context( messages, system_message, approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, - # #100661: the provider proved the request does not fit. - # Ignore the summary-failure cooldown for this ONE - # attempt (bounded by max_compression_attempts) instead - # of deferring every turn until the ladder lapses. + # Provider proved the request doesn't fit: ignore the + # summary-failure cooldown for this ONE attempt (bounded by + # max_compression_attempts). (#100661) bypass_cooldown=True, ) if messages is _overflow_input and compression_skipped_due_to_lock(agent): - # #69870 lock-skip: the provider proved the request - # does not fit, but this compression pass no-oped only - # because another path holds the session's compression - # lock. Temporary defer, not exhaustion — refund the - # attempt and end the turn softly so the gateway does - # NOT auto-reset the session (#9893/#35809). + # Lock-skip: another path holds the compression lock, so this + # pass no-oped. Temporary defer, not exhaustion — refund the + # attempt, end the turn softly, no auto-reset. (#69870) compression_attempts -= 1 agent._persist_session(messages, conversation_history) return _compression_deferred_result( agent, messages, api_call_count ) if messages is _overflow_input and compression_blocked_transiently(agent): - # #97488 transient-block: a timed guard (host-timeout - # cooldown / structural backoff) no-oped this pass — - # defer softly, never compression_exhausted (which - # would auto-reset the session). + # Transient block: a timed guard (host-timeout cooldown / + # structural backoff) no-oped this pass — defer softly, never + # compression_exhausted (auto-reset). (#97488) compression_attempts -= 1 agent._persist_session(messages, conversation_history) return _compression_deferred_result( @@ -6561,14 +5344,9 @@ def run_conversation( reason="transient_block", ) if context_compression_timed_out(agent): - # Host progress-aware timeout (#98722, salvaged from - # #98741): the provider proved the request does not - # fit, but this recovery pass spent the full wait - # budget without a committed summary. Re-sending the - # unchanged request would bounce off the same overflow - # error and re-enter compression in the same turn. End - # the turn with the typed recovery contract instead — - # transcript intact, no further doomed provider sends. + # Host timeout: recovery spent its wait budget with no committed + # summary. Re-sending would hit the same overflow; end the turn + # via the typed recovery contract. (#98722) agent._persist_session(messages, conversation_history) _final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE return { @@ -6586,10 +5364,9 @@ def run_conversation( agent, messages, conversation_history ) - # Re-estimate tokens after compression. Same-message-count - # compression (tool-result pruning, in-place summarization) - # can materially reduce request size without reducing the - # message array. (#39550) + # Re-estimate after compression: same-message-count compression + # (tool-result pruning, in-place summarization) can shrink the + # request. (#39550) new_tokens = estimate_messages_tokens_rough(messages) approx_tokens = new_tokens # update for downstream logging @@ -6599,10 +5376,9 @@ def run_conversation( elif new_tokens > 0 and new_tokens < original_tokens * 0.95: agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) time.sleep(2) # Brief pause between compression retries - # Rebuild the complete request before the next provider - # call and force normal preflight to honor it. Message - # count alone is not proof that system/tool-inclusive - # token pressure fell. + # Rebuild the full request and force normal preflight to honor + # it; message count alone doesn't prove system/tool-inclusive + # pressure fell. _provider_overflow_recovery_pending = True _retry.restart_with_compressed_messages = True break @@ -6625,61 +5401,30 @@ def run_conversation( "compression_exhausted": True, } - # Check for non-retryable client errors. The classifier - # already accounts for 413, 429, 529 (transient), context - # overflow, and generic-400 heuristics. Local validation - # errors (ValueError, TypeError) are programming bugs. - # Exclude UnicodeEncodeError — it's a ValueError subclass - # but is handled separately by the surrogate sanitization - # path above. Exclude json.JSONDecodeError — also a - # ValueError subclass, but it indicates a transient - # provider/network failure (malformed response body, - # truncated stream, routing layer corruption), not a - # local programming bug, and should be retried (#14782). + # Non-retryable: ValueError/TypeError are local bugs, except + # UnicodeEncodeError (surrogate path above) and json.JSONDecodeError, a + # transient provider/network failure that must be retried (#14782). is_local_validation_error = ( isinstance(api_error, (ValueError, TypeError)) and not isinstance( api_error, (UnicodeEncodeError, json.JSONDecodeError) ) - # ssl.SSLError (and its subclass SSLCertVerificationError) - # inherits from OSError *and* ValueError via Python MRO, - # so the isinstance(ValueError) check above would - # misclassify a TLS transport failure as a local - # programming bug and abort without retrying. Exclude - # ssl.SSLError explicitly so the error classifier's - # retryable=True mapping takes effect instead. + # ssl.SSLError inherits from OSError *and* ValueError, so the + # ValueError check would misclassify a TLS failure as a local bug; + # keep it retryable. and not isinstance(api_error, ssl.SSLError) - # Provider/SDK "NoneType is not iterable" failures are - # shape mismatches from upstream (e.g. chatgpt.com Codex - # backend response.completed.output=null) — not local - # programming bugs. Even after #33042 made our own - # consumer immune, third-party shims and mocked clients - # can still surface this shape via TypeError. Treat - # them as retryable so the error classifier's normal - # retry/fallback path runs instead of killing the turn - # as non-retryable (which left Telegram users staring - # at a bare "Non-retryable error" with no recovery). + # "NoneType is not iterable" TypeErrors are upstream shape + # mismatches (e.g. Codex response.completed.output=null), reachable + # via shims/mocks — retryable so the fallback path runs. and not ( isinstance(api_error, TypeError) and "nonetype" in str(api_error).lower() and "not iterable" in str(api_error).lower() ) ) - # ``FailoverReason.billing`` (HTTP 402) is NOT in this - # exclusion set. By the time we reach this block: - # • credential-pool rotation (line ~2031) has already - # fired for billing and either ``continue``d or - # returned (False, ...) — pool is exhausted or absent. - # • the eager-fallback branch above (line ~2422) also - # fires on billing and ``continue``s if a fallback - # provider is configured. - # Falling through to here means BOTH recovery paths - # gave up. Treating 402 as retryable from this point - # just burns more paid requests against a depleted - # balance with no recovery mechanism left — see #31273 - # (real-world: ~$40 in 48h on a 24/7 gateway). Aborting - # mirrors how 401/403 (also ``should_fallback=True``) - # already behave once their recovery paths have failed. + # ``FailoverReason.billing`` (402) is deliberately NOT excluded: pool + # rotation and eager fallback already gave up, so retrying only burns + # paid requests on a depleted balance. Mirrors 401/403. (#31273) is_client_error = ( is_local_validation_error or ( @@ -6697,17 +5442,9 @@ def run_conversation( ) and not is_context_length_error if is_client_error: - # Copilot self-heal BEFORE fallback: a stale/degraded - # credential surfaces as a 400 - # ``model_not_available_for_integrator`` / - # ``model_not_supported`` (not a clean 401), so the 401 - # refresh path above never fired. Force a fresh token - # exchange + client rebuild and retry once on the SAME - # provider — a fresh 437-char API token routes to the - # correct integrator and the model becomes available again. - # Single-shot guard prevents looping on a genuinely - # unavailable model. Copilot-scoped so other providers' - # real 400s are untouched. + # Copilot self-heal BEFORE fallback: a stale credential yields a 400 + # ``model_not_available_for_integrator`` / ``model_not_supported``, + # not a 401. Fresh token + client rebuild, one retry, SAME provider. if ( _is_copilot_provider(agent) and not _retry.copilot_stale_cred_retry_attempted @@ -6723,12 +5460,9 @@ def run_conversation( ) retry_count = 0 continue - # Try fallback before aborting — a different provider may - # not have the same issue (rate limit, auth, etc.). Only - # announce the attempt when a fallback chain actually - # exists; otherwise "trying fallback..." is a lie and the - # session looks like it's recovering when it's about to - # abort silently (#35314, #17446). + # Try fallback before aborting; announce it only when a fallback + # chain exists, else "trying fallback..." lies before a silent abort + # (#35314). if agent._has_pending_fallback(): if classified.reason == FailoverReason.content_policy_blocked: agent._buffer_status("⚠️ Provider safety filter blocked this request — trying fallback...") @@ -6751,12 +5485,9 @@ def run_conversation( # Terminal — flush buffered context so the user sees # what was tried before the abort. agent._flush_status_buffer() - # Summarize once: Cloudflare/proxy HTML challenge pages and - # other raw provider bodies must be collapsed to a short - # one-liner here, otherwise the full page leaks into the - # returned ``error`` field and downstream consumers deliver - # it verbatim (e.g. a cron failure notification dumped a - # ~60KB Cloudflare challenge page as 31 Discord messages). + # Summarize once: Cloudflare/proxy HTML pages and raw provider + # bodies must be collapsed here or they leak verbatim via the + # ``error`` field. _nonretryable_summary = agent._summarize_api_error(api_error) if classified.reason == FailoverReason.content_policy_blocked: agent._emit_status( @@ -6820,11 +5551,9 @@ def run_conversation( agent._vprint(f"{agent.log_prefix} • Check credits: https://openrouter.ai/settings/credits", force=True) else: agent._vprint(f"{agent.log_prefix} 💡 This type of error won't be fixed by retrying.", force=True) - # Content-policy blocks deserve their own actionable - # guidance — neither "fix your API key" nor "retry won't - # help" tells the user what to actually do. The provider - # has refused this specific prompt, so the recovery is - # either a rephrase or routing to a different model. + # Content-policy blocks get their own guidance: the provider refused + # this prompt, so recovery is a rephrase or another model, not + # key/retry advice. if classified.reason == FailoverReason.content_policy_blocked: agent._vprint( f"{agent.log_prefix} 💡 The provider's safety filter rejected this specific prompt.", @@ -6842,10 +5571,8 @@ def run_conversation( f"{agent.log_prefix} hermes fallback add (interactive picker — same as `hermes model`)", force=True, ) - # TLS certificate failures are environment problems, not - # provider/prompt problems — tell the user exactly which - # knobs fix each common cause. Inspired by Claude Code - # v2.1.199's immediate SSL fix hints. + # TLS certificate failures are environment problems — name the knobs + # that fix each common cause. if classified.reason == FailoverReason.ssl_cert_verification: agent._vprint( f"{agent.log_prefix} 💡 The TLS certificate chain could not be verified. This fails the same", @@ -6880,11 +5607,9 @@ def run_conversation( force=True, ) logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) - # Skip session persistence when the error is likely - # context-overflow related (status 400 + large session). - # Persisting the failed user message would make the - # session even larger, causing the same failure on the - # next attempt. (#1630) + # Skip persistence on likely context-overflow (400 + large session): + # persisting the failed message grows the session and repeats the + # failure. (#1630) if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80): agent._vprint( f"{agent.log_prefix}⚠️ Skipping session persistence " @@ -6906,10 +5631,8 @@ def run_conversation( final_response=_policy_response, error_detail=_nonretryable_summary, ) - # Billing walls are the common non-retryable abort: enrich - # the result with the same structured recovery descriptor as - # the max-retries path so every surface (CLI, TUI, desktop) - # renders one consistent billing signal. + # Billing walls get the same structured recovery descriptor as the + # max-retries path so every surface renders one consistent signal. if classified.reason == FailoverReason.billing: return _billing_failure_result( classified=classified, @@ -6930,19 +5653,16 @@ def run_conversation( } if retry_count >= max_retries: - # Before falling back, try rebuilding the primary - # client once for transient transport errors (stale - # connection pool, TCP reset). Only attempted once - # per API call block. + # Before fallback, rebuild the primary client once for transient + # transport errors (stale pool, TCP reset). Once per API call block. if not _retry.primary_recovery_attempted and agent._try_recover_primary_transport( api_error, retry_count=retry_count, max_retries=max_retries, ): _retry.primary_recovery_attempted = True retry_count = 0 - # Primary transport recovery starts a fresh attempt - # cycle. Re-open fallback state so a follow-on 429 can - # still activate fallback_providers after stale - # pre-recovery fallback/credential-pool bookkeeping. + # Transport recovery starts a fresh attempt cycle: re-open + # fallback state so a follow-on 429 can still activate + # fallback_providers. _retry.has_retried_429 = False agent._fallback_index = 0 agent._fallback_activated = False @@ -6992,11 +5712,9 @@ def run_conversation( agent._emit_status(f"❌ API failed after {max_retries} retries — {_final_summary}") agent._vprint(f"{agent.log_prefix} 💀 Final error: {_final_summary}", force=True) - # Detect SSE stream-drop pattern (e.g. "Network - # connection lost") and surface actionable guidance. - # This typically happens when the model generates a - # very large tool call (write_file with huge content) - # and the proxy/CDN drops the stream mid-response. + # SSE stream-drop (e.g. "Network connection lost"): usually a + # proxy/CDN cutting a very large tool call mid-response; give + # actionable guidance. _is_stream_drop = ( not getattr(api_error, "status_code", None) and any(p in error_msg for p in ( @@ -7021,23 +5739,9 @@ def run_conversation( force=True, ) - # Detect thinking-timeout pattern: a known reasoning model - # hit a transport-layer error before the first content - # token arrived. Distinct from _is_stream_drop above - # (which fires for large file-write stream drops) and - # from any classifier reason that's not a transport - # timeout. Reuses the reasoning-model allowlist from - # agent/reasoning_timeouts.py (Fixes #52217) so the - # trigger is consistent with what the per-model - # stale-timeout floor covers. After the classifier - # override at agent/error_classifier.py:720-738 (this - # PR), transport disconnects on reasoning models route - # to FailoverReason.timeout rather than - # context_overflow, so this branch actually fires. - # Detection and message text live in - # agent.thinking_timeout_guidance so they're - # unit-testable without driving the full retry loop. - # (Part 2 of Fixes #52310.) + # Thinking-timeout: a known reasoning model hit a transport error + # before the first content token. Distinct from _is_stream_drop; + # detection lives in agent.thinking_timeout_guidance. (#52310) from agent.thinking_timeout_guidance import ( is_thinking_timeout, ) @@ -7108,13 +5812,8 @@ def run_conversation( else: _final_response = f"API call failed after {max_retries} retries: {_final_summary}" if _is_thinking_timeout: - # Thinking-timeout guidance overrides the generic - # stream-drop guidance — the latter is wrong for - # this case (it suggests splitting large file - # writes, which isn't what happened). See the - # reasoning-model override at - # agent/error_classifier.py:720-738 and the - # detection block above for context. + # Thinking-timeout guidance overrides stream-drop guidance, + # which would wrongly suggest splitting large file writes. from agent.thinking_timeout_guidance import ( build_thinking_timeout_guidance, ) @@ -7138,11 +5837,9 @@ def run_conversation( "completed": False, "failed": True, "error": _final_summary, - # Surface the classified reason so callers (notably the - # kanban worker path in cli.py) can distinguish a - # transient throttle from a real failure and choose a - # different exit code. ``rate_limit`` / ``billing`` here - # mean "quota wall, not a task error". + # Expose the classified reason so callers (kanban worker in + # cli.py) can tell a quota wall (``rate_limit`` / ``billing``) + # from a task failure. "failure_reason": classified.reason.value, # The classifier's own retry verdict — UI surfaces use # this instead of re-deriving from the reason string. @@ -7163,11 +5860,9 @@ def run_conversation( _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") if _ra_raw: try: - # Cap at 10 minutes. Anthropic Tier 1 input-token - # buckets reset in ~171s, so a 120s cap caused us to - # retry before the actual reset window and re-trip the - # limit. 600s covers all realistic provider reset - # windows while still rejecting pathological values. (#26293) + # Cap at 600s: Anthropic Tier 1 buckets reset in ~171s, + # so a 120s cap retried early and re-tripped the limit. + # (#26293) _retry_after = min(float(_ra_raw), 600) except (TypeError, ValueError): pass @@ -7189,9 +5884,8 @@ def run_conversation( _policy_note = " (Z.AI Coding overload short retry)" _wait_reason = "Provider overloaded" if _is_zai_coding_overload and not is_rate_limited else "Rate limited" _rate_limit_status = f"⏱️ {_wait_reason}. Waiting {wait_time:.1f}s (attempt {retry_count + 1}/{max_retries}){_policy_note}..." - # Normal retries are buffered to avoid noisy transient chatter. Long - # Z.AI Coding waits are different: they can last minutes, so surface - # progress immediately instead of making the TUI look frozen. + # Normal retries are buffered to avoid chatter; long Z.AI Coding + # waits can last minutes, so surface progress immediately. if _backoff_policy == "zai_coding_overload_long": agent._emit_status(_rate_limit_status) else: @@ -7213,9 +5907,9 @@ def run_conversation( _backoff_touch_counter = 0 while time.time() < sleep_end: if agent._interrupt_requested: - # Same preserve-redirect rule as the retry-wait above: - # a steering correction must survive backoff, not die - # as "Operation interrupted". + # Same preserve-redirect rule as the retry-wait above: a + # steering correction must survive backoff, not die as + # "Operation interrupted". if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break @@ -7241,15 +5935,13 @@ def run_conversation( f"{int(sleep_end - time.time())}s remaining" ) if _retry.restart_with_redirected_messages: - # Leave the retry loop — the check right below rebuilds this - # iteration from the correction instead of re-firing the - # stale request. + # Leave the retry loop — the check below rebuilds this iteration + # from the correction instead of re-firing the stale request. break if _retry.restart_with_redirected_messages: - # The cancelled request produced no valid assistant item. Reuse the - # same logical iteration after the outer loop appends the displayed - # partial context and correction to ``messages``. + # Cancelled request produced no valid assistant item: reuse the same logical + # iteration after the outer loop appends partial context + correction. api_call_count -= 1 agent.iteration_budget.refund() _retry.restart_with_redirected_messages = False @@ -7263,9 +5955,8 @@ def run_conversation( if _retry.restart_with_compressed_messages: api_call_count -= 1 agent.iteration_budget.refund() - # Count compression restarts toward the retry limit to prevent - # infinite loops when compression reduces messages but not enough - # to fit the context window. + # Compression restarts count toward the retry limit so a compression that + # shrinks messages but not enough can't loop forever. retry_count += 1 _retry.restart_with_compressed_messages = False if _should_skip_model_call_for_reference_handoff( @@ -7279,15 +5970,9 @@ def run_conversation( final_response = _HANDOFF_SKIP_FINAL_RESPONSE _turn_exit_reason = "compaction_handoff_not_actionable" break - # In-loop compression rebuilt `messages` with fresh compaction - # copies, so the pre-compression current-turn index is stale. - # Re-anchor exactly like the prologue does: a stale index that - # lands on a historical user message would make the live-compose - # fallback inject this turn's prefetch into that message on the - # wire only, diverging the next turn's replayed prefix there. - # Ordered AFTER the handoff guard: the guard may have re-appended - # this turn's real user ask (restore path), and the anchor must - # land on that restored row, not on -1 / a pre-restore index. + # In-loop compression rebuilt `messages`; re-anchor the current-turn index + # like the prologue, AFTER the handoff guard (it may re-append this turn's + # ask). A stale anchor injects prefetch into a historical row. current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) @@ -7295,30 +5980,21 @@ def run_conversation( continue if _retry.restart_with_rebuilt_messages: - # A stream stall or provider failure was escalated to the - # fallback chain (10 activation sites in the retry loop set this - # flag and break here). Re-issue the API call against the - # now-active fallback provider. Refund the budget/count for the - # stalled attempt so the fallback gets a fair turn. + # A stall/failure escalated to the fallback chain: re-issue against the + # active fallback provider, refunding budget/count for the stalled attempt. api_call_count -= 1 agent.iteration_budget.refund() _retry.restart_with_rebuilt_messages = False - # Failover shrank the compressor's context window to the - # fallback's; clear the preflight block so the pre-API preflight - # re-runs against the new threshold before the first fallback - # call (#84733). Hoisted here (the single consumer) so every - # activation site — including ones added later — gets it. + # Failover shrank the compressor window: clear the preflight block so + # preflight re-runs before the first fallback call. Hoisted to the single + # consumer. (#84733) _preflight_compression_blocked = False continue if _retry.restart_with_length_continuation: - # Progressively boost the output token budget on each retry. - # Retry 1 → 2× base, retry 2 → 4× base, retry 3 → 8× base, - # retry 4 → 16× base, then cap at 32 768. - # Applies to all providers via _ephemeral_max_output_tokens. - # If the original request already used a larger provider/model - # default budget, keep that floor so continuation retries do - # not accidentally downshift to a much smaller cap. + # Boost output budget per retry: 2×, 4×, 8×, 16× base, capped at 32 768, via + # _ephemeral_max_output_tokens. Keep a larger original provider/model + # default as the floor so retries never downshift. _boost_base = agent.max_tokens if agent.max_tokens else 4096 _boost = _boost_base * (2 ** length_continue_retries) _requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs) @@ -7328,9 +6004,7 @@ def run_conversation( agent._ephemeral_max_output_tokens = min(_boost, _boost_cap) continue - # Guard: if all retries exhausted without a successful response - # (e.g. repeated context-length errors that exhausted retry_count), - # the `response` variable is still None. Break out cleanly. + # All retries may exhaust with `response` still None; break out cleanly. if response is None: _turn_exit_reason = "all_retries_exhausted_no_response" print(f"{agent.log_prefix}❌ All API retries exhausted with no successful response.") @@ -7346,9 +6020,8 @@ def run_conversation( assistant_message = normalized finish_reason = normalized.finish_reason - # Normalize content to string — some OpenAI-compatible servers - # (llama-server, etc.) return content as a dict or list instead - # of a plain string, which crashes downstream .strip() calls. + # Some OpenAI-compatible servers (llama-server) return content as dict/list, + # which crashes downstream .strip(); normalize to str. if assistant_message.content is not None and not isinstance(assistant_message.content, str): raw = assistant_message.content if isinstance(raw, dict): @@ -7368,12 +6041,8 @@ def run_conversation( assistant_message.content = str(raw) # ── Agent-as-provider projection ────────────────────────────── - # A provider that IS an agent ran its own tools inside its own - # session before we got here: splice that work into the transcript - # as completed call/result rows and tick the skill-review nudge - # with the iterations Hermes never saw. Appended before this turn's - # assistant message, so the order reads call → result → answer. - # No-op for ordinary providers; see agent/provider_projection.py. + # Splice the provider-agent's own tool work in as call/result rows before + # this turn's assistant message; no-op for ordinary providers. splice_provider_projection(agent, response, messages) try: @@ -7402,11 +6071,9 @@ def run_conversation( api_duration=api_duration, started_at=api_start_time, ended_at=_api_ended_at, - # First received stream chunk timestamp (epoch seconds), set by - # interruptible_streaming_api_call from its per-attempt - # stream diagnostics; None when the response was not - # streamed or no chunk arrived. TTFB = - # first_chunk_at - started_at. + # First stream chunk time (epoch s) from + # interruptible_streaming_api_call; None if not streamed / no + # chunk. TTFB = first_chunk_at - started_at. first_chunk_at=getattr( agent, "_last_api_first_chunk_at", None ), @@ -7505,12 +6172,9 @@ def run_conversation( or interim_has_codex_message_items ): last_msg = messages[-1] if messages else None - # Duplicate detection: compare only visible content - # (content + reasoning). Opaque provider state - # (encrypted reasoning items, message item ids/phases) - # drifts per continuation even when the visible output - # is identical, so including it in the comparison defeats - # dedup and causes message storms (#52711). + # Dedup on visible content only (content + reasoning): opaque + # provider state drifts per continuation and would defeat dedup + # (#52711). last_interim_visible = ( agent._interim_assistant_visible_text(last_msg) if isinstance(last_msg, dict) @@ -7533,9 +6197,8 @@ def run_conversation( and same_visible_output ) if visible_duplicate: - # Update replay state in-place so the latest provider - # payload is preserved without re-emitting identical - # user-visible commentary. + # Update replay state in-place: keep the latest provider payload + # without re-emitting identical user-visible commentary. for _key in ( "content", "reasoning", @@ -7546,11 +6209,9 @@ def run_conversation( ): if _key in interim_msg: if _key == "codex_reasoning_items": - # Merge instead of overwrite: a native - # compaction checkpoint captured on the - # earlier incomplete response is the only - # copy — the continuation won't re-emit - # it. See merge_interim_reasoning_items. + # Merge, don't overwrite: the earlier response's + # native compaction checkpoint is the only copy. See + # merge_interim_reasoning_items. from agent.native_compaction import ( merge_interim_reasoning_items, ) @@ -7564,43 +6225,17 @@ def run_conversation( agent._emit_interim_assistant_message(interim_msg) if agent._codex_incomplete_retries < 3: - # When the interim message has nothing the Responses - # input converter will replay (no visible content, no - # encrypted reasoning items, no replayable message - # items — plain-text reasoning only), a bare retry is - # byte-identical to the request that just came back - # incomplete and fails the same way every time - # (observed with grok-4.20 on xai-oauth, whose - # reasoning items lack encrypted_content). Append a - # user-role nudge so the retry actually differs and - # explicitly asks for the final answer. + # If the interim has nothing the Responses converter will replay, a + # bare retry is byte-identical and fails identically; append a + # user-role nudge so the retry differs and asks for the answer. interim_replayable = ( interim_has_content or interim_has_codex_reasoning or interim_has_codex_message_items ) - # A replayable interim is not the same thing as a retry - # that DIFFERS. When the interim replays but carries no - # new instruction, the continuation is byte-identical to - # the request that just failed and returns the same empty - # response until the budget is gone. Live case (gpt-5.6 - # on the Codex backend, Aug 2026): the model answers with - # a server-side ``compaction`` checkpoint and no message. - # The checkpoint lands in ``codex_reasoning_items``, so - # ``interim_replayable`` is True and no nudge is added — - # meanwhile the checkpoint makes the wire converter prune - # every pre-checkpoint item, so all three attempts send - # the same checkpoint + retained user messages and end on - # an empty assistant turn with nothing to answer. The - # provider's own prefix cache reports 99-100% on the - # repeats, and the turn dies with "Codex response - # remained incomplete after 3 continuation attempts", - # losing the whole turn's work. - # - # One bare retry is still worth trying (the model often - # just needs another turn). Once THAT has also come back - # incomplete, a bare retry is proven not to work for this - # turn, so every remaining attempt carries the nudge. + # Replayable ≠ different: an interim holding only a ``compaction`` + # checkpoint in ``codex_reasoning_items`` is replayable yet re-sends + # identically. One bare retry, then always nudge. if not interim_replayable or agent._codex_incomplete_retries >= 2: _last_msg = messages[-1] if messages else None _already_nudged = ( @@ -7608,13 +6243,9 @@ def run_conversation( and _last_msg.get("role") == "user" and _last_msg.get("content") == _CODEX_INCOMPLETE_NUDGE ) - # Alternation guard: the nudge is a user-role message, - # so it may only follow an assistant message. When the - # interim was too empty to append (no content AND no - # reasoning), the last message is still the prior - # user/tool turn — appending the nudge there would - # create a user→user / tool→user sequence that strict - # providers reject. + # Alternation guard: the user-role nudge may only follow an + # assistant message; after a too-empty interim it would create + # user→user / tool→user. _last_is_assistant = ( isinstance(_last_msg, dict) and _last_msg.get("role") == "assistant" @@ -7626,11 +6257,9 @@ def run_conversation( }) if not agent.quiet_mode: agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({agent._codex_incomplete_retries}/3)") - # Surface the continuation on the live spinner/status line - # (CLI/TUI/Desktop) and gateway heartbeat: each of these - # retries can spend minutes waiting on the provider, and - # without a distinct notice the user only sees a generic - # thinking spinner ("infinite thinking", #64434). + # Show the continuation on the spinner/status line and gateway + # heartbeat; these retries can take minutes and otherwise look like + # infinite thinking (#64434). agent._emit_wait_notice( f"↻ model returned reasoning with no final answer — " f"asking it to continue " @@ -7663,12 +6292,9 @@ def run_conversation( args_preview = raw_args[:200] if isinstance(raw_args, str) else repr(raw_args)[:200] logging.debug("Tool call: %s with args: %s...", tc.function.name, args_preview) - # Uniquify duplicate tool-call ids BEFORE any downstream - # consumer (validation error paths, dispatch, history build, - # Responses item-id derivation). Models that reuse one id for - # different calls in a batch otherwise lose the later call's - # result: the pre-API sanitizer keeps only the first - # call/result pair per id. See _uniquify_tool_call_ids. + # Uniquify duplicate tool-call ids BEFORE any downstream consumer: the + # pre-API sanitizer keeps only the first call/result per id. See + # _uniquify_tool_call_ids. agent._uniquify_tool_call_ids(assistant_message.tool_calls) # Validate tool call names - detect model hallucinations @@ -7683,17 +6309,9 @@ def run_conversation( tc.function.name for tc in assistant_message.tool_calls if tc.function.name not in agent.valid_tool_names ] - # Mixed batch: at least one valid call alongside the invalid - # one(s). Degrading models (observed with gpt-5.6 at very - # large context) emit batches like 6 named calls + 1 - # blank-name call; voiding the whole turn throws away real - # work and, across the 3-strike budget, halts sessions that - # were still making progress. Instead: error-result ONLY the - # invalid calls (below, after dedup/cap guardrails) and let - # the valid ones execute. The strike counter only advances - # when a turn contains NO valid call, so a fully-degenerate - # model still halts at 3 while a mostly-coherent one keeps - # working. + # Mixed batch: error-result ONLY the invalid calls and run the valid + # ones; voiding the turn discards real work. Strikes advance only when a + # turn has NO valid call, so a degenerate model still halts at 3. _mixed_invalid_batch = bool(invalid_tool_calls) and any( tc.function.name in agent.valid_tool_names for tc in assistant_message.tool_calls @@ -7724,10 +6342,9 @@ def run_conversation( agent._vprint(f"{agent.log_prefix}❌ Max retries (3) for invalid tool calls exceeded. Stopping as partial.", force=True) agent._invalid_tool_retries = 0 _final_response = f"Model generated invalid tool call: {invalid_preview}" - # Prior <3 retries (or an earlier successful tool batch) - # leave a tool-result tail. Closing it here matches - # interrupt aborts (#48879 / #52592) so the next user - # turn is not tool→user for strict providers. + # Prior retries or an earlier tool batch leave a tool-result + # tail; close it as interrupt aborts do so the next turn is not + # tool→user. (#48879) close_interrupted_tool_sequence(messages, _final_response) agent._persist_session(messages, conversation_history) return { @@ -7783,19 +6400,16 @@ def run_conversation( _mixed_invalid_batch and tc.function.name not in agent.valid_tool_names ): - # This call never executes — it gets an - # invalid-name error result below. Don't let its - # broken args trigger the whole-turn JSON retry. + # This call never executes (invalid-name error result + # below); don't let its broken args trigger the whole-turn + # JSON retry. continue invalid_json_args.append((tc.function.name, str(e))) if invalid_json_args: - # Check if the invalid JSON is due to truncation rather - # than a model formatting mistake. Routers sometimes - # rewrite finish_reason from "length" to "tool_calls", - # hiding the truncation from the length handler above. - # Detect truncation: args that don't end with } or ] - # (after stripping whitespace) are cut off mid-stream. + # Routers may rewrite finish_reason "length" → "tool_calls", hiding + # truncation; args not ending in } or ] (stripped) were cut off + # mid-stream. _truncated = any( not (tc.function.arguments or "").rstrip().endswith(("}", "]")) for tc in assistant_message.tool_calls @@ -7874,11 +6488,9 @@ def run_conversation( assistant_message.tool_calls ) - # Mixed-batch invalid-name handling: collect the invalid - # calls now so the assistant message (built below) keeps - # EVERY call the model emitted — providers require each - # tool_call to have a matching tool result and vice versa — - # while only the valid subset is dispatched for execution. + # Collect invalid calls so the assistant message keeps EVERY emitted + # call (each tool_call needs a matching result) while only valid ones + # dispatch. _invalid_batch_calls = [] if _mixed_invalid_batch: _invalid_batch_calls = [ @@ -7890,11 +6502,9 @@ def run_conversation( turn_content = assistant_message.content or "" - # Some local tool-call templates emit a bare bracketed token - # (for example ``[memory]``) as assistant content alongside a - # function call. It is protocol scaffolding, not an answer. - # Persisting or caching it as visible content lets the empty - # post-tool fallback replay that token forever after compaction (#78148). + # A bare bracketed token (e.g. ``[memory]``) beside a function call is + # protocol scaffolding; persisting it lets the post-tool fallback replay + # it forever (#78148). if ( assistant_message.tool_calls and _STALE_MARKER_RE.fullmatch(turn_content.strip()) @@ -7906,9 +6516,8 @@ def run_conversation( turn_content = "" assistant_msg["content"] = "" - # Classify tools in this turn to determine if they are all housekeeping. - # This classification is needed regardless of whether the turn has visible content, - # because a substantive tool-only turn must invalidate any older housekeeping fallback. + # Classify tools regardless of visible content: a substantive tool-only + # turn must invalidate any older housekeeping fallback. _HOUSEKEEPING_TOOLS = frozenset({ "memory", "todo_list", "skill_manage", "session_search", }) @@ -7917,31 +6526,22 @@ def run_conversation( for tc in assistant_message.tool_calls ) - # If this turn has substantive tools (non-housekeeping), clear any older fallback. - # Prevents a two-turn-old housekeeping narration from being treated as if it belonged - # to the immediately preceding substantive tool turn. + # Substantive tools clear any older fallback so a two-turn-old + # housekeeping narration isn't attributed to the preceding tool turn. if assistant_message.tool_calls and not _all_housekeeping: agent._last_content_with_tools = None agent._last_content_tools_all_housekeeping = False - # Also clear the mute flag: a prior housekeeping turn may - # have set _mute_post_response (line ~4667), and the - # substantive tools in THIS turn should produce visible - # progress output. Without this reset, _vprint suppresses - # tool progress until the no-tool-call branch clears it at - # line ~4834 — after all tools have finished. + # Also clear the mute flag a prior housekeeping turn may have set, + # else _vprint suppresses this turn's tool progress until the + # no-tool-call branch clears it. agent._mute_post_response = False - # If this turn has both content AND tool_calls, capture the content - # as a fallback final response. Common pattern: model delivers its - # answer and calls memory/skill tools as a side-effect in the same - # turn. If the follow-up turn after tools is empty, we use this. + # Content + tool_calls in one turn: keep the content as a fallback final + # response in case the follow-up turn after tools is empty. if turn_content and agent._has_content_after_think_block(turn_content): agent._last_content_with_tools = turn_content - # Only mute subsequent output when EVERY tool call in - # this turn is post-response housekeeping (memory, todo, - # skill_manage, etc.). If any substantive tool is present - # (search_files, read_file, write_file, terminal, ...), - # keep output visible so the user sees progress. + # Mute only when EVERY tool call is post-response housekeeping + # (memory, todo, skill_manage); substantive tools keep output on. agent._last_content_tools_all_housekeeping = _all_housekeeping if _all_housekeeping and agent._has_stream_consumers(): agent._mute_post_response = True @@ -7961,23 +6561,15 @@ def run_conversation( messages.pop() _had_prefill = True - # Reset prefill counter when tool calls follow a prefill - # recovery. Without this, the counter accumulates across - # the whole conversation — a model that intermittently - # empties (empty → prefill → tools → empty → prefill → - # tools) burns both prefill attempts and the third empty - # gets zero recovery. Resetting here treats each tool- - # call success as a fresh start. + # Tool calls after a prefill recovery reset the prefill counter, so + # each tool-call success is a fresh start, not a cumulative burn. if _had_prefill: agent._thinking_prefill_retries = 0 agent._empty_content_retries = 0 - # Successful tool execution — reset the post-tool nudge - # flag so it can fire again if the model goes empty on - # a LATER tool round. + # Re-arm the post-tool nudge so it can fire on a LATER tool round. agent._post_tool_empty_retried = False - # A landed tool call means any earlier dropped-tool-call stall - # was recovered — refresh that budget too so it guards each - # stall independently rather than capping the whole run. + # A landed tool call recovers any dropped-tool-call stall; refresh that + # budget so it guards each stall independently, not the whole run. agent._dropped_toolcall_retries = 0 previous_msg = messages[-1] if messages else None @@ -7996,11 +6588,8 @@ def run_conversation( ) append_message(messages, assistant_msg) - # Mixed batch: error-result the invalid calls and strip them - # from the execution set. The assistant message above keeps - # all calls (each gets a matching tool result — the invalid - # ones get theirs here, the valid ones during execution), so - # provider-side tool_call/result pairing stays intact. + # Mixed batch: error-result invalid calls and drop them from execution. + # The assistant message keeps all calls so tool_call/result pairs hold. if _invalid_batch_calls: for tc in _invalid_batch_calls: append_message(messages, { @@ -8018,10 +6607,8 @@ def run_conversation( _tool_turn_persisted = None try: - # Persist the assistant tool-call turn before any tool - # side effects run. If a destructive tool restarts or - # terminates Hermes mid-turn, resume logic still sees the - # exact tool-call block that already executed. + # Persist the tool-call turn before any tool side effects so resume + # sees the executed block if a destructive tool restarts Hermes. _tool_turn_persisted = agent._flush_messages_to_session_db( messages, conversation_history ) @@ -8039,12 +6626,9 @@ def run_conversation( ) if _tool_turn_persisted is False: - # The canonical append failed. Do not project the row or - # run side-effecting tools from state that exists only in - # this process. Breaking also avoids retrying the same - # unpersisted turn until the iteration budget is exhausted. - # The flush may have classified the cause internally; if - # nothing was recorded, the cause is genuinely unknown. + # Canonical append failed: never project the row or run tools from + # process-only state; break rather than retry the unpersisted turn. + # If the flush recorded no cause, the cause is genuinely unknown. if getattr(agent, "_last_persistence_error_cause", None) is None: agent._last_persistence_error_cause = "unknown" _turn_exit_reason = "session_persistence_failed" @@ -8052,18 +6636,14 @@ def run_conversation( failed = True break - # A UI must never observe an assistant/tool-call row that is - # still only an ephemeral in-memory projection. Emit interim - # commentary only after the canonical SessionDB append above. + # A UI must never observe an assistant/tool-call row that is only an + # in-memory projection: emit interim commentary after the DB append. if not duplicate_previous_interim: agent._emit_interim_assistant_message(assistant_msg) - # Close any open streaming display (response box, reasoning - # box) before tool execution begins. Intermediate turns may - # have streamed early content that opened the response box; - # flushing here prevents it from wrapping tool feed lines. - # Only signal the display callback — TTS (_stream_callback) - # should NOT receive None (it uses None as end-of-stream). + # Flush open streaming boxes before tools so early content doesn't wrap + # tool feed lines. Display callback only — TTS (_stream_callback) must + # NOT receive None (its end-of-stream marker). if agent.stream_delta_callback: try: agent.stream_delta_callback(None) @@ -8073,9 +6653,8 @@ def run_conversation( agent._execute_tool_calls(assistant_message, messages, effective_task_id, api_call_count) if getattr(agent, "_incremental_persistence_failed", False): - # A tool result could not be made canonical. Do not send - # the in-memory result back to the model or project any - # later events from this turn. + # Tool result could not be made canonical: never send the in-memory + # result to the model or project later events from this turn. _turn_exit_reason = "session_persistence_failed" final_response = "" failed = True @@ -8089,11 +6668,8 @@ def run_conversation( f"⚠️ Tool guardrail halted {decision.tool_name}: {decision.code}" ) append_message(messages, {"role": "assistant", "content": final_response}) - # Emit the halt message to the client so it's not - # indistinguishable from a crash. The stream display - # was flushed (callback(None)) before tool execution, - # but the callback is still alive — fire the text - # through it so SSE/TUI clients see the explanation. + # Emit the halt message so it isn't mistaken for a crash; the stream + # callback is still alive, so SSE/TUI clients see the explanation. if final_response: agent._safe_print(f"\n{final_response}\n") if agent.stream_delta_callback: @@ -8104,65 +6680,35 @@ def run_conversation( pass break - # Reset per-turn retry counters after successful tool - # execution so a single truncation doesn't poison the - # entire conversation. + # Reset per-turn retry counters so one truncation can't poison the turn. truncated_tool_call_retries = 0 - # Signal that a paragraph break is needed before the next - # streamed text. We don't emit it immediately because - # multiple consecutive tool iterations would stack up - # redundant blank lines. Instead, _fire_stream_delta() - # will prepend a single "\n\n" the next time real text - # arrives. + # Defer the paragraph break: _fire_stream_delta() prepends one "\n\n" + # when real text arrives, so tool iterations don't stack blank lines. agent._stream_needs_break = True - # Refund the iteration if the ONLY tool(s) called were - # execute_code (programmatic tool calling). These are - # cheap RPC-style calls that shouldn't eat the budget. + # Refund the iteration when the ONLY tool was execute_code (programmatic + # tool calling) — cheap RPC-style calls shouldn't eat the budget. _tc_names = {tc.function.name for tc in assistant_message.tool_calls} if _tc_names == {"execute_code"}: agent.iteration_budget.refund() - # Use real token counts from the API response to decide - # compression. prompt_tokens + completion_tokens is the - # actual context size the provider reported plus the - # assistant turn — a tight lower bound for the next prompt. - # Tool results appended above aren't counted yet, but the - # threshold (default 50%) leaves ample headroom; if tool - # results push past it, the next API call will report the - # real total and trigger compression then. - # - # If last_prompt_tokens is 0 (stale after API disconnect - # or provider returned no usage data), fall back to rough - # estimate to avoid missing compression. Without this, - # a session can grow unbounded after disconnects because - # should_compress(0) never fires. (#2153) + # Decide compression from API-reported prompt tokens (tight lower bound; + # tool results get counted on the next call). If last_prompt_tokens is 0 + # (disconnect / no usage data) fall back to a rough estimate. (#2153) _compressor = agent.context_compressor if _compressor.last_prompt_tokens > 0: - # Only use prompt_tokens — completion/reasoning - # tokens don't consume context window space. - # Thinking models (GLM-5.1, QwQ, DeepSeek R1) - # inflate completion_tokens with reasoning, - # causing premature compression. (#12026) + # Only prompt_tokens: thinking models inflate completion_tokens with + # reasoning that uses no context → premature compression. (#12026) _real_tokens = _compressor.last_prompt_tokens elif _compressor.last_prompt_tokens == -1: - # Compression just ran and no API-reported prompt count - # has arrived yet. Avoid treating a schema-heavy rough - # post-compression estimate as real context pressure. + # Compression just ran, no API prompt count yet: don't treat a rough + # schema-heavy post-compression estimate as real context pressure. _real_tokens = 0 else: - # Include tool schemas — with 50+ tools enabled - # these add 20-30K tokens the messages-only - # estimate misses, which can skip compression - # past the configured threshold (#14695). - # Route-aware (#96995/#97602 class): on a compacted - # native-Codex session the generic durable-history - # figure overstates the wire and would false-trigger - # compression here exactly like the pre-API guard — - # this fallback runs precisely when no provider usage - # is available (post-disconnect / gateway restart), - # the unanchored case from #97602's repro. + # Include tool schemas (20-30K tokens the messages-only estimate + # misses) and stay route-aware: on a compacted native-Codex session + # the generic durable-history figure would false-trigger. (#14695) _real_tokens = _midturn_request_pressure_tokens( agent, messages, @@ -8178,20 +6724,15 @@ def run_conversation( and _compressor.should_compress(_real_tokens) ): compression_attempts += 1 - # Compression is actually running (block cleared / was - # never blocked) — reset the blocked-overflow warning - # dedup so a future blocked-over-threshold turn can warn - # again (silent-overflow fix #62625). - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Compression is running: reset blocked-overflow warning dedup so a + # future blocked turn can warn again. getattr: test doubles lack it. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() agent._safe_print(" ⟳ compacting context…") _post_tool_input = messages - # Route the overhead-aware _real_tokens (computed above) into compression, not - # the bare last_prompt_tokens — which is 0 in the no-usage fallback, hiding the - # true request size from the engine's overflow guard (upstream PR #77169 review). + # Pass overhead-aware _real_tokens, not last_prompt_tokens (0 in + # the no-usage fallback), so the overflow guard sees the true size. messages, active_system_prompt = agent._compress_context( messages, system_message, approx_tokens=_real_tokens, @@ -8201,12 +6742,9 @@ def run_conversation( messages is _post_tool_input and compression_skipped_due_to_lock(agent) ): - # #69870 lock-skip: this pass no-oped because another - # path holds the session's compression lock — a - # temporary defer, not evidence about compressibility. - # Refund the attempt so a lock-loser tool loop does not - # burn the shared per-turn budget toward - # compression_exhausted (#9893/#35809). + # Lock-skip no-op is a temporary defer, not evidence about + # compressibility: refund so a lock-loser loop doesn't burn the + # budget toward compression_exhausted. (#69870) compression_attempts -= 1 else: conversation_history = conversation_history_after_compression( @@ -8225,11 +6763,8 @@ def run_conversation( _turn_exit_reason = "compaction_handoff_not_actionable" break elif agent.compression_enabled: - # Over threshold but compression is blocked (summary-LLM - # cooldown or anti-thrashing). Surface a deduped warning so - # the user isn't left with a silently growing context that - # eventually hits the hard provider limit. Mirrors the - # turn-context preflight guard (silent-overflow fix #62625). + # Over threshold but compression blocked (cooldown/anti-thrash): + # deduped warning so context can't silently overflow. (#62625) _block_reason = None _info = getattr(_compressor, "should_compress_info", None) if _info is not None: @@ -8243,19 +6778,9 @@ def run_conversation( _real_tokens, int(getattr(_compressor, "threshold_tokens", 0) or 0), ) - # Proactive tool-result prune: reclaim re-sent history on - # large-window models long before should_compress() (≈50% of - # the window) would ever fire. Deterministic, no LLM call; - # protects the recent tail. No-op unless proactive_prune_tokens - # is configured and _real_tokens is above it — and even then - # the prune only commits when it reclaims at least - # proactive_prune_min_reclaim_tokens, so prompt-cache breaks - # stay episodic like compression's (the one sanctioned cache - # break) instead of firing every tool iteration. See - # ContextCompressor.prune_tool_results_only. - # getattr guard: plugin context engines predating the hook and - # minimal test doubles (SimpleNamespace compressors) lack the - # method — treat absence as a no-op. + # Proactive tool-result prune (deterministic, no LLM, keeps tail): + # no-op unless proactive_prune_tokens is exceeded; commits only past + # proactive_prune_min_reclaim_tokens so cache breaks stay episodic. _prune = getattr(_compressor, "prune_tool_results_only", None) if callable(_prune): try: @@ -8271,58 +6796,33 @@ def run_conversation( # Standard no-op caller contract: only commit when the # engine returned a NEW list object with a non-zero count. if _pruned_n and _pruned_msgs is not messages: - # Do NOT rebuild conversation_history here. The compressor - # atomically rewrites the active transcript with the durable - # rearm threshold, then stamps every returned row with - # _DB_PERSISTED_MARKER, so the marker-based flush dedup (see - # _flush_messages_to_session_db) prevents duplicate writes. - # Calling - # conversation_history_after_compression (a compaction-only - # helper keyed on the _last_compaction_in_place flag) would be - # a no-op at best, and on a stale in-place flag could seed - # this turn's fresh, not-yet-persisted rows into history_ids - # and skip writing them. + # Do NOT rebuild conversation_history: rows already carry + # _DB_PERSISTED_MARKER, and on a stale in-place flag the + # helper could seed unpersisted rows into history_ids. messages = _pruned_msgs # Save session log incrementally (so progress is visible even if interrupted) agent._session_messages = messages - # Touch activity before continuing so the gateway's - # inactivity monitor never sees a stale timestamp - # between tool completion and the start of the next - # API call. Without this, a tool-call result (which - # takes ~0s to process) followed by slow post-tool - # processing (compression, persist) and a slow - # follow-up API call can exceed the gateway inactivity - # timeout (HERMES_AGENT_TIMEOUT, default 1800s) and the - # gateway kills the session before the next activity - # touch fires (#69559, #69131). + # Touch activity so slow post-tool work plus a slow follow-up API call + # can't exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT). agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}") # Continue loop for next response continue else: - # No tool calls - this is the final response. - # (Dropped tool-call recovery — finish_reason=="tool_calls" with - # an empty tool_calls array — is handled at the finalization - # chokepoint below, after final_msg is built, so it catches - # every path that reaches turn finalization, not just this one.) + # No tool calls — final response. (Dropped tool-call recovery lives at + # the finalization chokepoint below so it catches every path.) final_response = assistant_message.content or "" - # Fix: unmute output when entering the no-tool-call branch - # so the user can see empty-response warnings and recovery - # status messages. _mute_post_response was set during a - # prior housekeeping tool turn and should not silence the - # final response path. + # Unmute: _mute_post_response from a housekeeping tool turn must not + # silence empty-response warnings on the final response path. agent._mute_post_response = False # Check if response only has think block with no actual content after it if not agent._has_content_after_think_block(final_response): - # ── Partial stream recovery ───────────────────── - # If content was already streamed to the user before - # the connection died, use it as the final response - # instead of falling through to prior-turn fallback - # or wasting API calls on retries. + # Partial stream recovery: content streamed before the connection + # died becomes the final response instead of fallback or retries. _partial_streamed = ( getattr(agent, "_current_streamed_assistant_text", "") or "" ) @@ -8339,23 +6839,15 @@ def run_conversation( "as final response" ) final_response = _recovered - # Streaming delivered a fragment, not a confirmed - # final preview. Leave response_previewed false so - # gateway fallback delivery can send the recovered - # text plus the abnormal-turn explanation. + # A streamed fragment isn't a confirmed preview: keep + # response_previewed false so gateway fallback delivery can + # send the text plus the abnormal-turn explanation. agent._response_was_previewed = False break - # If the previous turn already delivered real content alongside - # HOUSEKEEPING tool calls (e.g. "You're welcome!" + memory save), - # the model has nothing more to say. Use the earlier content - # immediately instead of wasting API calls on retries. - # NOTE: Only use this shortcut when ALL tools in that turn were - # housekeeping (memory, todo, etc.). When substantive tools - # were called (terminal, search_files, etc.), the content was - # likely mid-task narration ("I'll scan the directory...") and - # the empty follow-up means the model choked — let the - # post-tool nudge below handle that instead of exiting early. + # Prior turn had real content + ONLY housekeeping tools: model is + # done, reuse it. With substantive tools it was mid-task narration + # and the empty reply is a choke; let the post-tool nudge handle it. fallback = getattr(agent, '_last_content_with_tools', None) if fallback and getattr(agent, '_last_content_tools_all_housekeeping', False): _turn_exit_reason = "fallback_prior_turn_content" @@ -8364,36 +6856,21 @@ def run_conversation( agent._last_content_with_tools = None agent._last_content_tools_all_housekeeping = False agent._empty_content_retries = 0 - # Do NOT modify the assistant message content — the - # old code injected "Calling the X tools..." which - # poisoned the conversation history. Just use the - # fallback text as the final response and break. + # Do NOT modify the assistant message content (injected text + # poisoned history); use the fallback as the response and break. final_response = agent._strip_think_blocks(fallback).strip() agent._response_was_previewed = True break # ── Post-tool-call empty response nudge ─────────── - # The model returned empty after executing tool calls. - # This covers two cases: - # (a) No prior-turn content at all — model went silent - # (b) Prior turn had content + SUBSTANTIVE tools (the - # fallback above was skipped because the content - # was mid-task narration, not a final answer) - # Instead of giving up, nudge the model to continue by - # appending a user-level hint. This is the #9400 case: - # weaker models (mimo-v2-pro, GLM-5, etc.) sometimes - # return empty after tool results instead of continuing - # to the next step. One retry with a nudge usually - # fixes it. + # Empty after tool results (no prior content, or only mid-task + # narration): nudge once via a user-level hint. (#9400) _prior_was_tool = any( m.get("role") == "tool" for m in messages[-5:] # check recent messages ) - # Detect Qwen3/Ollama-style in-content thinking blocks. - # Ollama puts in the content field (not in - # reasoning_content), so _has_structured below would - # miss it. We check here so thinking-only responses - # after tool calls route to prefill instead of nudge. + # Ollama puts in content, not reasoning_content, so + # _has_structured misses it; detect here to route to prefill. _has_inline_thinking = bool( re.search( r'||', @@ -8419,11 +6896,8 @@ def run_conversation( "⚠️ Model returned empty after tool calls — " "nudging to continue" ) - # Append the empty assistant message first so the - # message sequence stays valid: - # tool(result) → assistant("(empty)") → user(nudge) - # Without this, we'd have tool → user which most - # APIs reject as an invalid sequence. + # Append the empty assistant first so the sequence stays valid: + # tool → assistant("(empty)") → user (APIs reject tool→user). _nudge_msg = agent._build_assistant_message(assistant_message, finish_reason) _nudge_msg["content"] = "(empty)" _nudge_msg["_empty_recovery_synthetic"] = True @@ -8436,14 +6910,8 @@ def run_conversation( continue # ── Thinking-only prefill continuation ────────── - # The model produced structured reasoning (via API - # fields) but no visible text content. Rather than - # giving up, append the assistant message as-is and - # continue — the model will see its own reasoning - # on the next turn and produce the text portion. - # Inspired by clawdbot's "incomplete-text" recovery. - # Also covers Qwen3/Ollama in-content blocks - # (detected above as _has_inline_thinking). + # Reasoning but no text: append as-is and continue so the model sees + # its own reasoning and writes text. Covers _has_inline_thinking. _has_structured = bool( getattr(assistant_message, "reasoning", None) or getattr(assistant_message, "reasoning_content", None) @@ -8470,14 +6938,8 @@ def run_conversation( continue # ── Empty response retry ────────────────────── - # Model returned nothing usable. Retry up to 3 - # times before attempting fallback. This covers - # both truly empty responses (no content, no - # reasoning) AND reasoning-only responses after - # prefill exhaustion — models like mimo-v2-pro - # always populate reasoning fields via OpenRouter, - # so the old `not _has_structured` guard blocked - # retries for every reasoning model after prefill. + # Retry up to 3 times before fallback; covers truly empty replies + # AND reasoning-only replies after prefill exhaustion. _truly_empty = not agent._strip_think_blocks( final_response ).strip() @@ -8489,14 +6951,9 @@ def run_conversation( not _has_structured or _prefill_exhausted ) if _empty_candidate: - # NS-503: every empty attempt re-sends the full - # conversation input at full price. Record the - # attempt (usage/finish_reason signature) so - # deterministic empties — e.g. unsignaled - # provider refusals with zero output tokens — - # stop burning paid retries reproducing the - # same empty. Fails open: missing usage or - # any generated tokens keep the full budget. + # Each empty attempt re-bills the full input; record its + # signature so deterministic empties stop burning paid retries. + # Fails open: missing usage or any output keeps the budget. _empty_guard.record_empty_attempt( agent, finish_reason=finish_reason, @@ -8581,11 +7038,7 @@ def run_conversation( ) # ── Exhausted retries — try fallback provider ── - # Before giving up with "(empty)", attempt to - # switch to the next provider in the fallback - # chain. This covers the case where a model - # (e.g. GLM-4.5-Air) consistently returns empty - # due to context degradation or provider issues. + # Before "(empty)", switch to the next provider in the chain. if _truly_empty and agent._fallback_chain: logger.warning( "Empty response after %d retries — " @@ -8610,26 +7063,15 @@ def run_conversation( "now using %s on %s", agent.model, agent.provider, ) - # This site sits directly in the OUTER iteration - # loop (not the retry loop), so `continue` already - # restarts the iteration and re-runs the pre-API - # preflight against the fallback's context window - # (#84733). A `break` here would exit the outer - # loop and end the turn without ever calling the - # fallback. Clear the preflight block so the - # re-run isn't skipped. + # OUTER loop: `continue` re-runs preflight against the + # fallback's window; `break` would end the turn without + # calling the fallback. Clear the preflight block. (#84733) _preflight_compression_blocked = False continue - # Exhausted retries and fallback chain (or no - # fallback configured). Fall through to the - # "(empty)" terminal. - # Surface the buffered retry/fallback trace so the - # user can see what was attempted before "(empty)". - # NS-503: if we know roughly what the empty streak - # cost (each attempt re-billed the full input), say - # so — an unexplained charge for "no answer" is the - # core of the complaint. + # Retries and fallback exhausted — fall through to "(empty)". + # Surface the buffered retry trace and, if known, what the empty + # streak cost (each attempt re-billed the full input). _streak_cost = _empty_guard.streak_cost_usd(agent) if _streak_cost is not None: agent._buffer_status( @@ -8643,11 +7085,8 @@ def run_conversation( agent._drop_trailing_empty_response_scaffolding(messages) assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) assistant_msg["content"] = "(empty)" - # This is a user-facing failure sentinel for the gateway, - # not real assistant content. Persisting it makes later - # "continue" turns replay assistant("(empty)") as if it - # were a meaningful model response, which can keep long - # tool-heavy sessions stuck in empty-response loops. + # Gateway failure sentinel, not content: persisting it lets later + # "continue" turns replay assistant("(empty)") and loop on empties. assistant_msg["_empty_terminal_sentinel"] = True append_message(messages, assistant_msg) @@ -8676,18 +7115,9 @@ def run_conversation( ". No fallback providers configured.") ) - # Deliver a labeled reasoning excerpt instead of a bare - # "(empty)" when the model DID think but never produced - # visible text. This is delivery-only: the persisted - # assistant message above keeps the "(empty)" sentinel - # (its replay semantics prevent empty-response loops), - # and raw chain-of-thought is never promoted to a normal - # answer earlier in the ladder — prefill continuation, - # empty-content retries, and provider fallback all run - # first. Only at this terminal, where the alternative is - # returning nothing, is showing the model's own reasoning - # (clearly labeled as such) strictly more useful. - # Idea credit: PR #48795 (@ligl0325). + # Delivery-only: show labeled reasoning instead of bare "(empty)" + # when the model thought but wrote no text. The persisted row keeps + # the sentinel; reasoning is never promoted earlier in the ladder. if reasoning_text: final_response = ( "⚠️ The model produced only internal reasoning and " @@ -8703,10 +7133,8 @@ def run_conversation( # Reset retry counter/signature on successful content agent._empty_content_retries = 0 agent._thinking_prefill_retries = 0 - # Successful content reached — surface the one-shot fallback - # switch notice (if a fallback activated this turn) before - # dropping the noisy retry buffer, so a provider/model switch - # stays visible even when the fallback succeeds. + # Surface the one-shot fallback switch notice before dropping the retry + # buffer so a provider/model switch stays visible on success. agent._emit_pending_fallback_notice() agent._clear_status_buffer() @@ -8716,15 +7144,9 @@ def run_conversation( ) _ack_mode = intent_ack_continuation_mode(agent) - # Said-continue-but-stopped guard (agent.stall_guards): the - # model ended the turn with no tool calls but its short reply - # TAILS with an announced next action ("Let me now…", - # "I will now…"). Unlike the intent-ack detector below, this - # fires mid-task too (after tool results), which is exactly - # where eval traces show the stall. It reuses the SAME bounded - # continuation path and counter (max 2 per turn), so the - # alternation-safe interim-assistant + user-nudge mechanism — - # not a new parallel one — carries the recovery. + # Said-continue-but-stopped guard: no tool calls but the short reply + # TAILS with an announced next action. Fires mid-task too; reuses the + # SAME bounded continuation path and counter (max 2 per turn). _stall_continue_intent = ( bool(getattr(agent, "_stall_guards", True)) and agent.valid_tool_names @@ -8761,9 +7183,8 @@ def run_conversation( } append_message(messages, continue_msg) agent._session_messages = messages - # An acknowledgment is explicitly non-final. Do not let its - # text suppress iteration-limit summarization if this - # continuation consumes the remaining budget. + # An acknowledgment is non-final: its text must not suppress + # iteration-limit summarization if the continuation exhausts budget. final_response = None continue @@ -8784,17 +7205,8 @@ def run_conversation( final_msg = agent._build_assistant_message(assistant_message, finish_reason) # ── Dropped tool-call recovery (copilot/Claude) ──────── - # Some providers (observed: claude-opus-4.8 / claude-sonnet-4.5 - # on GitHub Copilot, ~2026-07) return finish_reason="tool_calls" - # while the parsed tool_calls array is empty — the model - # signalled it wanted to act but the payload shipped no call. - # Reaching finalization with that mismatch means the turn is - # about to end with the task unstarted (the narration, which may - # be in content or only in the reasoning field, gets treated as - # the final answer). Re-prompt (bounded to 3 CONSECUTIVE stalls; - # the budget resets after any successful tool round) to make the - # model emit the call instead of exiting. finish_reason="stop" - # text finishes never enter this guard. + # finish_reason="tool_calls" with empty tool_calls would end the turn + # unstarted; re-prompt (max 3 CONSECUTIVE stalls, reset per tool round). if ( finish_reason == "tool_calls" and not assistant_message.tool_calls @@ -8811,16 +7223,9 @@ def run_conversation( "↻ Model signaled a tool call but sent none — " f"re-prompting ({agent._dropped_toolcall_retries}/3)" ) - # Both halves of the re-prompt pair are ephemeral recovery - # scaffolding (mirrors the empty-response nudge pattern): - # the interim narration-only assistant turn exists solely to - # keep role alternation valid for the nudge, and the nudge - # exists solely to drive the retry. Flag both so the - # persistence layer never writes them to the durable - # transcript and the finalization pop below can strip an - # unanswered tail pair. A recovered (answered) pair stays - # buried mid-list in live memory but is skipped by the - # flush regardless of position. + # Both halves of the re-prompt pair are ephemeral scaffolding; flag + # them so the flush never persists them and the finalization pop + # can strip an unanswered tail pair. final_msg["_dropped_toolcall_nudge"] = True append_message(messages, final_msg) append_message(messages, { @@ -8832,16 +7237,11 @@ def run_conversation( final_response = None continue - # Reached finalization without the dropped-tool-call mismatch — - # a genuine turn end. Clear the consecutive-stall budget so the - # next turn starts fresh. + # Genuine turn end (no dropped-tool-call mismatch): clear stall budget. agent._dropped_toolcall_retries = 0 - # Pop thinking-only prefill and empty-response retry - # scaffolding before appending either a final response or a - # verification-stop follow-up. These internal turns are only - # for the next API retry and should not become durable - # transcript context. + # Pop prefill / empty-retry scaffolding before the final response or + # verification follow-up; it must not become durable transcript. while ( messages and isinstance(messages[-1], dict) @@ -8877,11 +7277,8 @@ def run_conversation( getattr(agent, "_verification_stop_nudges", 0) + 1 ) final_msg["finish_reason"] = "verification_required" - # The assistant response is real content — persist it and - # emit to the UI as an interim message so the user sees the - # attempted final answer before the verification loop runs. - # Only the nudge is flagged synthetic so it gets stripped - # from the durable transcript (#65919 §7). + # Real content: persist and emit as interim so the user sees the + # attempted answer; only the nudge is flagged synthetic. (#65919) agent._emit_interim_assistant_message(final_msg) append_message(messages, final_msg) try: @@ -8894,18 +7291,12 @@ def run_conversation( "_verification_stop_synthetic": True, }) agent._session_messages = messages - # Run the verification-stop loop silently — the nudge is an - # internal turn that should not add noise to the user's - # terminal. Keep a debug breadcrumb in agent.log for tracing. + # Internal nudge: stay silent on the terminal, debug-log only. logger.debug("verification stop-loop nudge issued (attempt %d)", agent._verification_stop_nudges) - # Keep the attempted answer only as an explicit fallback for - # continuation-budget exhaustion. ``final_response`` itself - # must be cleared so the finalizer can distinguish this gate - # from unrelated error/recovery exits. (#61631) - # Track whether this candidate was already streamed so the - # finalizer can mark the turn previewed only if the - # candidate is actually reused as the final response. + # Keep the answer only as a budget-exhaustion fallback; clear + # ``final_response`` so the finalizer can tell this gate from error + # exits. Mark previewed only if the candidate is reused. (#61631) _pending_verification_response = final_response _pending_verification_response_previewed = ( agent._interim_content_was_streamed(final_response or "") @@ -8913,11 +7304,8 @@ def run_conversation( final_response = None continue - # User verification-loop gate: when the agent edited code this - # turn, let a registered `pre_verify` hook (plugin/shell) keep it - # going one more turn. The shipped guidance is folded into the - # evidence-based verify-on-stop nudge above, so this path has no - # default continuation cost. + # pre_verify hook gate: after code edits a registered hook may keep the + # agent going one more turn; no default continuation cost. _verify_nudge2 = None _edited = sorted(getattr(agent, "_turn_file_mutation_paths", set()) or []) _attempt = getattr(agent, "_pre_verify_nudges", 0) @@ -8949,11 +7337,8 @@ def run_conversation( if _verify_nudge2: agent._pre_verify_nudges = _attempt + 1 final_msg["finish_reason"] = "verify_hook_continue" - # The assistant response is real content — persist it and - # emit to the UI as an interim message so the user sees the - # attempted final answer before the pre_verify loop runs. - # Only the nudge is flagged synthetic so it gets stripped - # from the durable transcript (#65919 §7). + # Real content: persist and emit as interim so the user sees the + # attempted answer; only the nudge is flagged synthetic. (#65919) agent._emit_interim_assistant_message(final_msg) append_message(messages, final_msg) try: @@ -8976,11 +7361,8 @@ def run_conversation( continue # ── Kanban worker terminal-tool stop guard ───────────── - # Workers must end with kanban_complete / kanban_block. - # Models sometimes narrate the next step ("Let me write the - # report") and stop with finish_reason=stop — a clean exit - # that the dispatcher records as protocol_violation. Nudge - # once or twice before allowing that exit. + # Workers must end with kanban_complete / kanban_block; a narrated stop + # is recorded as protocol_violation, so nudge once or twice first. try: from agent.kanban_stop import build_kanban_stop_nudge @@ -9014,10 +7396,8 @@ def run_conversation( "⚠️ Kanban worker tried to exit without " "kanban_complete/kanban_block — nudging to finish" ) - # Same finalizer contract as verify-on-stop: clear - # final_response while continuing so a later budget - # exhaustion path does not treat the narrated stop as - # a completed answer. + # Same finalizer contract as verify-on-stop: clear final_response so + # budget exhaustion doesn't treat the narrated stop as an answer. _pending_verification_response = final_response _pending_verification_response_previewed = ( agent._interim_content_was_streamed(final_response or "") @@ -9026,14 +7406,9 @@ def run_conversation( continue append_message(messages, final_msg) - # Make the completed answer durable before leaving the loop — - # a session torn down before finalize_turn's _persist_session - # otherwise loses a reply the user already saw (#81641). Same - # contract as the tool-call exit (#49045) and the verify exits - # above; _DB_PERSISTED_MARKER keeps _persist_session idempotent. - # Unlike the tool-call exit, failure must NOT abort the turn: - # no side effect follows and _persist_session retries the write. - # Full incident narrative: tests/run_agent/test_81641_*.py. + # Make the answer durable before leaving the loop; _DB_PERSISTED_MARKER + # keeps _persist_session idempotent. Failure must NOT abort the turn: + # _persist_session retries the write. (#81641) try: agent._flush_messages_to_session_db(messages, conversation_history) except Exception: @@ -9050,29 +7425,13 @@ def run_conversation( break except Exception as e: - # Count every escaped exception against the per-turn bound before - # classification — permanent failures must terminate even when the - # turn budget is unlimited (#92450). + # Count every escaped exception before classification so permanent + # failures terminate even with an unlimited turn budget. (#92450) _outer_error_count += 1 - # Phase-aware error classification. The huge outer try/except spans - # both the actual API request and all local post-processing of the - # returned assistant message. Deterministic local bugs (e.g. - # passing a multimodal content list into a regex helper after a - # vision turn or context compaction) should not be retried: they - # will fail identically on every iteration and only burn the - # iteration budget. We classify an error as local by inspecting the - # traceback: if the exception propagated through any of the known - # local post-processing helpers and never entered the interruptible - # API-call helpers, it is almost certainly a local processing bug. - # (#66267) - # - # Interpreter shutdown: if the process is tearing down, every - # executor-backed operation (API call, tool dispatch, memory sync) - # raises ``RuntimeError: cannot schedule new futures after - # interpreter shutdown``. Retrying is pointless — the executor is - # gone for good — and each retry just spams another traceback. - # Break immediately so the turn exits cleanly. (#93217) + # Phase-aware classification: deterministic local post-processing bugs + # (traceback via local helpers, never API helpers) aren't retried (#66267). + # Interpreter shutdown makes every executor op raise: break. (#93217) if sys.is_finalizing() or _is_interpreter_shutdown_error(e): error_msg = ( f"Interpreter is shutting down — cannot continue " @@ -9083,9 +7442,8 @@ def run_conversation( except (OSError, ValueError): pass logger.warning(error_msg) - # Best-effort persist — the executor is dying, so this may - # raise the same RuntimeError. Don't let that mask the - # shutdown exit. finalize_turn will retry the persist. + # Best-effort persist — the dying executor may raise the same error; + # don't let it mask the shutdown exit. finalize_turn retries. try: agent._persist_session(messages, conversation_history) except Exception: @@ -9095,12 +7453,8 @@ def run_conversation( "Session is shutting down. Your conversation can be " "resumed with: hermes --resume " ) - # Don't append the assistant message here — a thinking-prefill - # or interim assistant may already be the tail, and appending - # would create assistant→assistant. finalize_turn handles - # this case safely (lines 341-353: appends only when - # _tail_role != "assistant"), matching the pattern at other - # break sites that set final_response without appending. + # Don't append: a prefill/interim assistant may already be the tail + # (assistant→assistant). finalize_turn appends only when safe. break tb_module_names: set[str] = set() @@ -9122,13 +7476,8 @@ def run_conversation( ) else: error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}" - # The background-review fork sets suppress_status_output=True so - # lifecycle noise never reaches the user's terminal — but this - # bare print() bypassed it, leaking ❌ lines onto the shell after - # the TUI exited. Honor the same contract _vprint enforces - # (quiet_mode -q still shows hard failures, matching force=True - # semantics; suppress_status_output silences them). The - # logger.exception below still captures the full traceback. + # Honor the _vprint contract: suppress_status_output silences hard + # failures; quiet_mode -q still shows them. Traceback is logged below. if getattr(agent, "suppress_status_output", False): logger.error(error_msg) else: @@ -9137,17 +7486,12 @@ def run_conversation( except (OSError, ValueError): logger.error(error_msg) - # Emit the full traceback at ERROR level so it lands in both - # agent.log AND errors.log. Previously this was logged at DEBUG, - # which meant intermittent outer-loop failures were unreproducible - # — users would see a one-line summary on screen with no way to - # recover the call site. logger.exception() includes the - # traceback automatically and emits at ERROR. + # ERROR level with traceback so outer-loop failures land in agent.log + # AND errors.log and stay reproducible. logger.exception("Outer loop error in API call #%d", api_call_count) - # If an assistant message with tool_calls was already appended, - # the API expects a role="tool" result for every tool_call_id. - # Fill in error results for any that weren't answered yet. + # An appended assistant tool_calls message needs a role="tool" result + # per tool_call_id; fill in error results for unanswered ones. for idx in range(len(messages) - 1, -1, -1): msg = messages[idx] if not isinstance(msg, dict): @@ -9172,19 +7516,11 @@ def run_conversation( append_message(messages, err_msg) break - # Non-tool errors don't need a synthetic message injected. - # The error is already printed to the user (line above), and - # the retry loop continues. Injecting a fake user/assistant - # message pollutes history, burns tokens, and risks violating - # role-alternation invariants. + # Non-tool errors are already printed; a synthetic message would pollute + # history and risk breaking role alternation. - # If we're near the limit, break to avoid infinite loops. - # Local processing errors are deterministic — stop immediately - # rather than retrying until the budget is exhausted. Repeated - # outer-loop errors stop after a small per-turn cap: with - # max_iterations now unlimited by default, a permanent failure - # would otherwise spin forever and overwrite the rotated log - # history within minutes (#92450). + # Local errors are deterministic: stop early instead of retrying until the + # budget is gone; a small per-turn cap prevents infinite spinning (#92450). _outer_error_cap = min(_MAX_OUTER_LOOP_ERRORS, max(1, agent.max_iterations)) if ( _is_local_processing_error @@ -9201,17 +7537,11 @@ def run_conversation( else: _turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})" final_response = f"I apologize, but I encountered repeated errors: {error_msg}" - # Don't append the assistant message here — a thinking-prefill - # or interim assistant may already be the tail, and appending - # would create assistant→assistant. finalize_turn handles - # this case safely (lines 341-353: appends only when - # _tail_role != "assistant"), matching the pattern at other - # break sites that set final_response without appending. + # Don't append the assistant message: a prefill/interim assistant may be + # the tail. finalize_turn appends only when _tail_role != "assistant". break - # Post-loop turn finalization extracted to agent/turn_finalizer.finalize_turn - # (god-file decomposition Phase 1 step 4). Behavior-neutral: the assembled - # result dict is returned exactly as before. + # Post-loop finalization lives in agent/turn_finalizer.finalize_turn. result = finalize_turn( agent, final_response=final_response, @@ -9230,10 +7560,8 @@ def run_conversation( _pending_verification_response_previewed=_pending_verification_response_previewed, ) if _compression_timeout_exhausted: - # Reuse the gateway's existing context-recovery contract (#98722, - # salvaged from #98741). The bloated transcript remains intact while - # future input can move to a clean session instead of replaying the - # summarize-timeout loop. + # Reuse the gateway's context-recovery contract: transcript stays intact while + # future input can move to a clean session (#98722). result["error"] = _COMPRESSION_TIMEOUT_FINAL_RESPONSE result["partial"] = True result["compression_exhausted"] = True diff --git a/agent/turn_context.py b/agent/turn_context.py index 691b1de9c0..47bab659c4 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -1,26 +1,9 @@ """Per-turn setup for ``run_conversation`` (the turn prologue). -``run_conversation`` opened with ~470 lines of straight-line setup before the -tool-calling loop ever started: stdio guarding, runtime-main wiring, retry-counter -resets, user-message sanitization, todo/nudge-counter hydration, system-prompt -restore-or-build, session-row creation (before compression, whose DB writes -reference the row), preflight context compression, the ``pre_llm_call`` plugin -hook, external-memory prefetch, and crash-resilience persistence (last, so the -user row is written once with its final ``api_content`` sidecar). - -All of that is *prologue* — it runs once per turn, has no back-references into the -loop, and produces a fixed set of values the loop then consumes. ``TurnContext`` -captures those produced values; ``build_turn_context`` performs the setup work and -returns one. ``run_conversation`` is left to unpack the context and run the loop, -shrinking the orchestrator by the full prologue. - -The builder still mutates ``agent`` heavily (counters, thread id, cached prompt, -session DB) exactly as the inline code did — those side effects are the point. The -``TurnContext`` it returns carries only the *locals* the loop reads back. - -Behavior is identical to the original inline prologue; this is a pure -move-and-name refactor with no semantic change. -""" +``build_turn_context`` runs the once-per-turn setup (stdio guard, sanitization, prompt +restore-or-build, session row, preflight compression, pre_llm_call hook, prefetch, +persistence) mutating ``agent`` exactly as the inline code did, and returns a +``TurnContext`` carrying only the locals the loop reads back.""" from __future__ import annotations @@ -59,17 +42,8 @@ def _preflight_request_tokens( ) -> int: """Token estimate for automatic preflight compression. - When the upcoming request is eligible for native Responses compaction, - count the checkpoint-pruned wire payload rather than the full durable - transcript. Auxiliary compression still uses the generic estimator - (``native_compaction_eligible=False``). - - Usage-anchored fast path: when a provider-reported usage anchor is - valid for ``messages`` (see ``anchored_context_tokens``), it already - covers system prompt + tool schemas + full history EXACTLY as the - provider counted them, with estimation confined to the messages - appended since that response. Prefer it over every heuristic. - """ + Prefers a valid provider usage anchor; on native-compaction-eligible requests counts + the checkpoint-pruned wire payload; otherwise uses the generic estimator.""" anchored = anchored_context_tokens( messages, getattr(agent, "_usage_anchor", None) ) @@ -112,9 +86,7 @@ def _preflight_request_tokens( def _agent_stale_thinking_on_wire(agent: Any) -> bool: """Whether the agent's active route replays stale thinking text (#84371). - Route facts unavailable (test doubles, partially-built agents) default to - ``True`` — the conservative full charge. - """ + Returns ``True`` (conservative full charge) when route facts are unavailable.""" try: from agent.message_sanitization import stale_thinking_reaches_wire @@ -135,20 +107,8 @@ def compose_user_api_content( ) -> Optional[str]: """Compose the API-bound content of the current turn's user message. - Sources: memory-manager prefetch + ``pre_llm_call`` plugin context with - target="user_message" (the default). Both are appended to the *API copy* - of the user message only — the stored content stays clean. - - This is the single source of that composition. The prologue stamps the - result onto the live message as ``api_content`` (persisted alongside the - clean content) and the ``api_messages`` build in ``conversation_loop`` - sends the same helper's output, so the persisted sidecar can never drift - from the bytes on the wire — which is the whole prompt-cache invariant: - what turn N sends must be what turn N+1 replays. - - Returns ``None`` when nothing is injected (multimodal/non-string content, - or no ephemeral context), meaning the message is sent as-is. - """ + Single source for the ``api_content`` sidecar and the wire bytes (never drift). + Returns ``None`` when nothing is injected (message is sent as-is).""" if not isinstance(content, str): return None injections = [] @@ -166,16 +126,8 @@ def compose_user_api_content( def substitute_api_content(api_msg: Dict[str, Any]) -> Optional[str]: """Pop the ``api_content`` sidecar and substitute it into ``content``. - Used at every API-bound message-build site (the ``api_messages`` build in - ``conversation_loop``, the max-iterations summary in - ``chat_completion_helpers``, the chat-completions transport). The sidecar - carries the exact bytes previously sent to the API for this message when - they differ from the clean stored content; substituting it here keeps the - provider prompt-cache prefix byte-stable across turns. - - Returns the popped sidecar string (for callers that need the value for - current-turn composition logic) or ``None`` when absent. - """ + Keeps the provider prompt-cache prefix byte-stable across turns. + Returns the popped sidecar string, or ``None`` when absent.""" sidecar = api_msg.pop("api_content", None) if ( isinstance(sidecar, str) @@ -189,21 +141,12 @@ def substitute_api_content(api_msg: Dict[str, Any]) -> Optional[str]: def drop_stale_api_content(msg: Dict[str, Any]) -> None: """Drop the ``api_content`` sidecar from a message whose content was rewritten. - Called from every content-rewrite path (historical image strip, - merge-summary-into-tail, consecutive-user repair merge, stale-confirmation - redaction). Replaying the pre-rewrite sidecar would resend exactly what - the rewrite removed, so it must be dropped — the cost is one cache - boundary miss, never wrong content. - """ + Replaying it would resend what the rewrite removed; cost is one cache miss.""" msg.pop("api_content", None) def extract_api_content_sidecar(msg: Mapping[str, Any]) -> Optional[str]: - """Extract the ``api_content`` sidecar from a message dict for persistence. - - Shared by the gateway/branch forwarding sites that copy the sidecar into a - new row. Returns the string sidecar or ``None`` when absent/non-string. - """ + """Extract the ``api_content`` sidecar; ``None`` when absent/non-string.""" v = msg.get("api_content") return v if isinstance(v, str) else None @@ -211,14 +154,8 @@ def extract_api_content_sidecar(msg: Mapping[str, Any]) -> Optional[str]: def consume_gateway_turn_context_notes(agent: Any) -> str: """Pop the gateway's per-turn must-deliver notes off the agent (one-shot). - The gateway relocates volatile per-turn facts OUT of the ephemeral system - prompt (auto-reset notes, the first-contact intro, voice-channel changes) - and delivers them on the current user message via the api_content sidecar - instead, so the composed system prompt stays byte-stable turn-over-turn. - It stages the rendered notes on ``agent._gateway_turn_context_notes`` - right before ``run_conversation``; this consumes them so a cached agent - can never replay a stale note on a later turn. - """ + Staged on ``agent._gateway_turn_context_notes``; consuming them keeps the system + prompt byte-stable and prevents a cached agent replaying a stale note.""" notes = getattr(agent, "_gateway_turn_context_notes", "") or "" if hasattr(agent, "_gateway_turn_context_notes"): try: @@ -231,14 +168,8 @@ def consume_gateway_turn_context_notes(agent: Any) -> str: def append_notes_to_multimodal_content(content: Any, notes: str) -> bool: """Deliver must-deliver notes on a multimodal (list) user message. - ``compose_user_api_content`` returns ``None`` for non-string content, so - sidecar-borne facts would silently drop on image/attachment turns. For - gateway must-deliver notes we instead append a text part to the content - list in place — the part becomes durable message content (persisted and - replayed as-is), which keeps the wire and the transcript byte-identical. - - Returns ``True`` when a part was appended. - """ + Appends a durable text part in place, since the sidecar path returns ``None`` + for non-string content. Returns ``True`` when a part was appended.""" if not notes or not isinstance(content, list): return False try: @@ -248,28 +179,13 @@ def append_notes_to_multimodal_content(content: Any, notes: str) -> bool: return False -# Surfaces whose sessions must not be auto-titled. The prologue is shared by -# EVERY agent, not only the ones a human is watching, so membership here is what -# keeps the titler off machine-driven runs: -# -# - cron — the scheduler names its own session after the job in its `finally` -# block, and the opener is the cron delivery hint, not a user's request. -# Titling it writes that scaffolding as the visible name for the whole run and -# bills a side-LLM call per fire, against the same job that sets -# `skip_memory` / `skip_background_review` to avoid exactly that. -# - subagent — a delegated child's session is hidden from every picker, so its -# title is never read. A batch at `max_concurrent_children` would pay N title -# calls for N names nobody sees. +# Surfaces whose sessions must not be auto-titled: cron names its own session and +# its opener is a delivery hint; subagent sessions are hidden from every picker. _UNTITLED_PLATFORMS = frozenset({"cron", "subagent"}) def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: - """Kick off auto-titling for this session's first user message. - - Called from the turn prologue, so every surface a human reads (CLI, gateway, - TUI/desktop, ACP) gets identical behavior without each one re-implementing - the call. Fully defensive: titling is cosmetic and must never break a turn. - """ + """Kick off auto-titling for the session's first user message; never fatal.""" session_db = getattr(agent, "_session_db", None) session_id = getattr(agent, "session_id", None) if not session_db or not session_id: @@ -282,9 +198,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: from agent.message_content import flatten_message_text from agent.title_generator import maybe_auto_title - # The turn's own user message, as text. Multimodal turns flatten to - # their text parts; an image-only turn yields "" and is skipped, since - # there is nothing to title from. + # Turn's user message as text; image-only turns yield "" and are skipped. user_text = "" for msg in reversed(messages or []): if isinstance(msg, dict) and msg.get("role") == "user": @@ -293,9 +207,8 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: if not user_text: return - # The session row is created lazily on the first persist, which happens - # later in the turn. Force it now, or the title write matches zero rows - # and the session stays untitled for the whole turn anyway. + # Session row is created lazily later; force it now or the title write matches + # zero rows. if not getattr(agent, "_session_db_created", False): ensure = getattr(agent, "_ensure_db_session", None) if callable(ensure): @@ -303,9 +216,8 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: if not getattr(agent, "_session_db_created", False): return - # Snapshot the runtime identity; the validator lets the background - # titler skip its LLM call if the user switches models before it fires - # (a stale request would reload an unloaded Ollama model, #19027). + # Snapshot runtime identity so the background titler can skip if the user + # switches models before it fires (#19027). _model = getattr(agent, "model", None) _provider = getattr(agent, "provider", None) @@ -338,19 +250,9 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> int: """Locate this turn's user message after compaction rebuilt ``messages``. - Compression replaces list entries with fresh copies (and may append a - todo-snapshot user message or a restored user turn AFTER the surviving - copy of the current turn's message), so a pre-compression index is - meaningless. Prefer the LAST user message whose content exactly matches - this turn's text — the surviving copy in the common case — so the - injection stamp and the #48677 persist override can't land on a - todo-snapshot or historical row. Fall back to the last *user-originated* - turn when no exact match survives (merge-summary-into-tail rewrites the - content but the trackers still need a live anchor). Compaction handoffs - must never become the fallback anchor (#80622) — they are reference-only - scaffolding, not the active ask. Returns -1 when the list has no - user-originated message at all. - """ + Prefers the LAST user message whose content exactly matches this turn's text, else + the last user-originated turn; compaction handoffs are never the fallback (#80622). + Returns -1 when there is no user-originated message.""" from agent.context_compressor import user_originated_turn_view fallback = -1 @@ -358,9 +260,8 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in msg = messages[i] if not (isinstance(msg, dict) and msg.get("role") == "user"): continue - # Typed synthetic current events still need their physical persistence - # anchor when their raw content is unchanged. They are not eligible - # for the human-only fallback below. + # Typed synthetic current events keep their persistence anchor when raw + # content is unchanged; not eligible for the human-only fallback below. if msg.get("content") == user_message: return i live_view = user_originated_turn_view(msg) @@ -380,28 +281,15 @@ def compression_made_progress( ) -> bool: """Return ``True`` if a compression pass materially reduced the request. - Compression can succeed by summarising message contents — reducing the - estimated request token count — without reducing the message row - count. Treating row count as the sole progress signal false-positives - on size-only wins and surfaces a misleading "Cannot compress further" - failure even when post-compression tokens are well below the model - context window. See issue #39548 for an observed case: 220 → 220 - messages, ~288k → ~183k tokens on a 1M-context model still triggered - auto-reset. - - The token reduction must be *material* (>5%) to count as progress — the - same floor the overflow-handler retry path uses (conversation_loop.py, - #39550) — so a sub-5% wobble doesn't keep the multi-pass loop spinning. - """ + Counts a >5% token reduction as progress even when the row count is unchanged + (size-only wins, #39548); same floor as the overflow-handler retry path.""" if new_len < orig_len: return True return orig_tokens > 0 and new_tokens < orig_tokens * 0.95 -# Back-compat alias: this predicate was module-private until the gateway's -# session-hygiene recovery gate needed the same semantics (#79624). Keeping the -# old name bound means existing callers and any test that patches -# ``_compression_made_progress`` continue to work unchanged. +# Back-compat alias: gateway callers and tests patch ``_compression_made_progress`` +# (#79624). _compression_made_progress = compression_made_progress @@ -425,15 +313,8 @@ def _fail_closed_after_preflight_timeout(agent, request_tokens: int) -> None: def _review_fork_first_request_pending(agent: Any) -> bool: """Whether a detached review fork has yet to send its first provider request. - The background-review fork (issue #93057) replays the parent's FULL - snapshot on its first provider request as a warm prompt-cache read - (same-model cache parity). Compaction must not rewrite the snapshot - before that first request goes out — a compacted transcript would miss - the parent's cached prefix and turn a cheap cached replay into a cold - over-threshold write. Once the first provider response has arrived the - fork's tool loop is its own context, and both compression gates resume. - Dormant for every agent without the attribute. - """ + The fork replays the parent's FULL snapshot as a warm cache read, so compaction must + wait until that first response arrives. Dormant without the attribute (#93057).""" return bool( getattr(agent, "_review_defer_compaction_before_first_response", False) and not getattr(agent, "_turn_received_provider_response", False) @@ -445,11 +326,7 @@ def _compression_warrants_another_preflight_pass( ) -> bool: """Whether an over-threshold request merits another immediate summary. - Row-count progress is enough to prove that a compression boundary was real, - but not enough to justify another expensive pass before trying the provider. - Continue only when the request remains over threshold *and* the previous pass - materially reduced its estimated token pressure (>5%). - """ + Continue only if still over threshold AND the previous pass cut tokens by >5%.""" return ( new_tokens >= threshold_tokens and orig_tokens > 0 @@ -465,21 +342,9 @@ def _should_run_preflight_estimate( ) -> bool: """Cheap gate for the (expensive) full preflight token estimate. - Returns ``True`` when either: - (a) message count exceeds the protected ranges (the historical gate), or - (b) a cheap char-based estimate already crosses the configured threshold - — the few-but-huge case from issue #27405 that the count-only gate - would silently skip (a handful of very large messages never trips - the count condition, so compression was never attempted and the - turn hit a hard context-overflow error). - - Branch (b) uses ``estimate_messages_tokens_rough`` (the shared char-based - estimator) so a single large base64 image isn't mistaken for ~250K tokens. - It intentionally undercounts vs. the full request estimate — it omits the - system prompt and tool schemas — because it is only a *hint* deciding - whether to pay for the authoritative ``estimate_request_tokens_rough``, - which (together with ``should_compress``) makes the real decision. - """ + ``True`` when message count exceeds the protected ranges OR a rough char-based + estimate (``estimate_messages_tokens_rough``, undercounting by design) crosses the + threshold — the few-but-huge case (#27405).""" if len(messages) > protect_first_n + protect_last_n + 1: return True return estimate_messages_tokens_rough(messages) >= threshold_tokens @@ -496,20 +361,9 @@ def _should_idle_compact( ) -> bool: """Decide whether an idle-triggered compaction should run this turn. - Idle compaction is opt-in (``idle_after_seconds <= 0`` disables it). It - fires when a session resumes after a wall-clock gap of at least - ``idle_after_seconds`` since its last activity, so a long-lived thread - that is paused and later resumed compacts its accumulated history up - front instead of re-reading it on every subsequent turn. - - It is orthogonal to the token-threshold trigger: it does NOT require the - context to exceed ``threshold_tokens``. It still skips work when the - context is at or below ``floor_tokens`` (the size compaction would reduce - *to*), so a small idle thread never pays for a summarisation that saves - nothing, and it defers to an active compression-failure cooldown. - - Pure predicate so the policy is unit-testable without a live agent. - """ + Fires after a wall-clock gap of ``idle_after_seconds`` (opt-in, <= 0 disables), + independent of ``threshold_tokens``; skips at/below ``floor_tokens`` and during a + compression-failure cooldown. Pure predicate.""" if not enabled or idle_after_seconds <= 0: return False if idle_gap_seconds < idle_after_seconds: @@ -572,27 +426,18 @@ def build_turn_context( ) -> TurnContext: """Run the once-per-turn setup and return the loop's input context. - The callables/helpers the original prologue referenced from the - ``conversation_loop`` module are passed in explicitly to keep this module - free of an import cycle with ``agent.conversation_loop``. - """ + Helpers are passed in to avoid an import cycle with ``agent.conversation_loop``.""" # Guard stdio against OSError from broken pipes (systemd/headless/daemon). install_safe_stdio() - # Recover a session rotated by another path before binding log/turn ids or - # copying client-supplied history. Everything in this turn must consistently - # belong to the canonical child, including observability metadata. + # Recover a rotated session before binding log/turn ids or copying client history so + # everything in this turn belongs to the canonical child. recovered_history = recover_rotated_compression_session(agent) if recovered_history is not None: conversation_history = recovered_history - # NOTE: the DB session row is created later, AFTER the system prompt is - # restored/built (see _ensure_db_session() below the system-prompt block). - # Creating it here — before _cached_system_prompt is populated — inserts a - # row with system_prompt=NULL on a fresh API/gateway agent that carries - # client-managed history, which then trips the "stored system prompt is - # null; rebuilding from scratch" warning and a needless first-turn prefix - # cache miss. (Issue #45499.) + # NOTE: the DB session row is created later, after the system prompt is built; + # creating it now would persist system_prompt=NULL and cost a cache miss (#45499). # Tag log records on this thread with the session ID for ``hermes logs``. set_session_context(agent.session_id) @@ -608,15 +453,9 @@ def build_turn_context( try: from agent.auxiliary_client import set_runtime_main from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe - # Rotation-stable prompt-cache scope. Memoized per segment on the - # agent, so this is a DB walk at most once per segment — except a - # brand-new session whose row lands later in turn setup - # (_ensure_db_session); that first turn falls back to the physical - # id here and the first build_api_kwargs re-resolves. Stays valid - # through a mid-turn compression rotation because the lineage root - # is by definition rotation-invariant (#79017). Resolved with the - # never-raising variant OUTSIDE the argument list, so a resolution - # failure can only lose the scope — never the whole runtime binding. + # Rotation-stable prompt-cache scope (lineage root), memoized per segment; a + # new session uses the physical id until build_api_kwargs re-resolves (#79017). + # Never-raising variant, outside the argument list: failure loses only scope. _cache_scope = resolve_prompt_cache_scope_safe(agent) or "" set_runtime_main( getattr(agent, "provider", "") or "", @@ -632,31 +471,13 @@ def build_turn_context( except Exception: pass - # Between-turns MCP refresh: an MCP server that finished connecting since - # the previous turn (slow HTTP/OAuth servers routinely take 2-6s on a cold - # connect, missing the bounded startup wait) lands in THIS turn's tool - # snapshot. Timing is cache-safe by construction: it runs in the per-turn - # prologue, before this turn's first API call assembles ``tools=``, so it - # never mutates the prefix of an in-flight turn. ``preserve_prefix`` makes - # the *content* cache-safe too (#100336): a plain rebuild re-derives the - # array from live availability, so a flapping ``check_fn`` silently drops a - # tool and a late arrival splices into sorted position — either one forks - # the tool block and re-prefills the whole history behind it, every turn it - # happens. With the flag the live order is authoritative and the array - # only ever grows. No-op when no MCP servers are registered (the common - # case, gated by the cheap ``has_registered_mcp_tools`` check) or when the - # tool set is unchanged (``refresh_agent_mcp_tools`` diffs by name and - # leaves the snapshot untouched on no-change). + # Between-turns MCP refresh: late-connecting servers land in THIS turn's snapshot, + # before the first API call assembles ``tools=``. ``preserve_prefix`` keeps the + # tool array append-only so a flapping ``check_fn`` can't fork the cache (#100336). try: if not getattr(agent, "_skip_mcp_refresh", False): - # Import-cost gate: ``tools.mcp_tool`` pulls in the whole ``mcp`` - # package (~0.4s measured) even when the user has zero MCP servers - # configured. MCP tools can only be registered by code that has - # already imported ``tools.mcp_tool`` (discovery, /reload-mcp, - # late-binding refresh) — so if it isn't in sys.modules yet, there - # is nothing to refresh and the import can be skipped outright. - # This keeps the no-MCP first turn off the heavy import path - # without changing behavior for MCP users. + # Import-cost gate: MCP tools are only registered by code that already + # imported ``tools.mcp_tool`` (~0.4s); not in sys.modules => nothing to do. import sys as _sys if "tools.mcp_tool" in _sys.modules: from tools.mcp_tool import has_registered_mcp_tools, refresh_agent_mcp_tools @@ -690,9 +511,8 @@ def build_turn_context( agent._relay_pending_turn_id = None agent._current_turn_id = turn_id agent._current_api_request_id = "" - # Tripwire: warn (with both turn ids) when this turn starts before the - # previous turn's turn-end persist — concurrent turns on one session - # interleave transcript writes. Cleared in _persist_session. + # Tripwire: warn when this turn starts before the previous turn-end persist + # (concurrent turns interleave transcript writes). Cleared in _persist_session. from agent.agent_runtime_helpers import note_turn_start note_turn_start(agent, turn_id) @@ -734,9 +554,8 @@ def build_turn_context( # NOTE: _turns_since_memory and _iters_since_skill are NOT reset here. agent.iteration_budget = IterationBudget(agent.max_iterations) - # Wall-clock run budget: per-run_conversation clock. Only stamped when a - # budget is configured so the default path stays clock-free; the wrap-up - # latch resets each turn (one notice per run, not per session). + # Wall-clock run budget: stamped only when configured; the wrap-up latch resets per + # turn (one notice per run). if getattr(agent, "run_budget_seconds", None): agent._run_budget_started_at = time.time() else: @@ -757,11 +576,8 @@ def build_turn_context( # Initialize conversation (copy to avoid mutating the caller's list). messages = list(conversation_history) if conversation_history else [] - # The CLI may already have staged this input outside the history passed to - # ``run_conversation``. Reuse it only when its clean transcript text matches - # this turn; a stale handoff from a failed prior turn must not replace a - # later, different user input. Voice turns compare against their explicit - # clean persistence override rather than the API-only prefixed payload. + # Reuse CLI-staged input only when its clean text matches this turn; a stale + # handoff must not replace later input. Voice turns compare the clean override. pending_cli_message = getattr(agent, "_pending_cli_user_message", None) expected_persist_content = ( persist_user_message if persist_user_message is not None else user_message @@ -771,9 +587,8 @@ def build_turn_context( and pending_cli_message.get("content") == expected_persist_content ): user_msg = pending_cli_message - # The CLI-staged value is the clean transcript text. Restore the - # API-facing variant (for example, a voice-mode prefix) while retaining - # the same dict and any close-path durable marker. + # CLI-staged value is the clean text; restore the API-facing variant (e.g. voice + # prefix) on the same dict, keeping any close-path durable marker. user_msg["content"] = user_message else: user_msg = stamp_message_timestamp( @@ -800,27 +615,16 @@ def build_turn_context( if agent._memory_nudge_interval > 0 and agent._turns_since_memory == 0: agent._turns_since_memory = prior_user_turns % agent._memory_nudge_interval - # Add the current user message after the prompt/session setup has made - # close persistence safe. The handoff above preserves any marker already - # stamped by an earlier close flush. - # - # A synthesized turn (auto-continue recovery note, delegation completion) - # declares how it should READ in a transcript. Stamp that on the live - # message so the crash persist below writes the row already typed. Typing - # it after the turn instead leaves the row untyped for the whole run — and - # forever if the turn crashes — so the raw system note paints as a user - # bubble. The model still receives role/content unchanged; the api_messages - # build strips both fields from every outgoing copy. + # Append the user message now that close persistence is safe. Synthesized turns + # stamp their transcript type so the crash persist writes a typed row; the model + # still receives role/content unchanged (api_messages strips both fields). if persist_user_display_kind: user_msg["display_kind"] = persist_user_display_kind if persist_user_display_metadata: user_msg["display_metadata"] = persist_user_display_metadata - # Stamp the platform-side message id (e.g. the Discord/Telegram message id) - # as metadata on the user turn so it survives the early crash-resilience - # persist below (the turn-start flush). Load-bearing for restart - # drain-window recovery: a recovery pass dedups via - # ``has_platform_message_id`` against this row. + # Stamp the platform message id so it survives the turn-start flush; restart + # drain-window recovery dedups via ``has_platform_message_id`` against this row. if persist_user_platform_id is not None: user_msg["platform_message_id"] = persist_user_platform_id append_message(messages, user_msg) @@ -855,9 +659,8 @@ def build_turn_context( should_review_memory = True agent._turns_since_memory = 0 - # Cosmetic side-signal: detect an affection "reaction" (ily / <3 / good bot) - # and notify the host so it can play hearts. Token-free, never touches the - # conversation, and never fatal — a purely optional UI beat. + # Cosmetic side-signal: detect an affection reaction so the host can play hearts. + # Token-free, never touches the conversation, never fatal. reaction_callback = getattr(agent, "reaction_callback", None) if reaction_callback is not None: try: @@ -882,12 +685,8 @@ def build_turn_context( active_system_prompt = agent._cached_system_prompt - # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" - # session on Bot-Mode-managed installs (same gate as the protocol - # section above). The gate is stable for a session's lifetime, so the - # tool list is byte-identical every turn: prompt-cache safe. Every - # other session (CLI, gateway chats, group-room member sessions, cron, - # subagents) fails the gate and never sees the schema. + # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" session + # (same gate as the protocol section); gate is session-stable, so cache-safe. try: from tools.bot_mode_dm import ensure_message_agent_tool @@ -895,17 +694,9 @@ def build_turn_context( except Exception: logger.debug("message_agent injection skipped", exc_info=True) - # Create the DB session row now that _cached_system_prompt is populated, so - # the persisted snapshot is written non-NULL on the first turn (Issue - # #45499). Idempotent: _ensure_db_session() no-ops once the row exists. - # Must run BEFORE preflight compression: in-place compaction inserts - # message rows referencing this session (archive_and_compact), and - # rotation creates a child with parent_session_id pointing at it — with - # PRAGMA foreign_keys=ON, a missing parent row fails both INSERTs on a - # fresh oversized first turn. The user-turn crash persist itself runs - # LATER (after memory prefetch / pre_llm_call), so the row is written - # once with its final api_content — both steps take the same per-agent - # persist lock as CLI close persistence. + # Create the DB row now (system prompt populated => non-NULL, #45499) and BEFORE + # preflight compression: compaction/rotation INSERTs reference this row under + # PRAGMA foreign_keys=ON. Idempotent; the user-turn crash persist runs later. persist_lock = getattr(agent, "_session_persist_lock", None) try: if persist_lock is None: @@ -920,33 +711,21 @@ def build_turn_context( exc_info=True, ) finally: - # Clear the staged CLI input eagerly (as the pre-refactor code did) - # so a crash in preflight compression — which runs between this row - # create and the late crash-persist below — doesn't leave a stale - # _pending_cli_user_message that the next turn would mistake for a - # fresh staged input. + # Clear staged CLI input eagerly so a crash in preflight compression doesn't + # leave a stale _pending_cli_user_message for the next turn. if not isinstance(pending_cli_message, dict) or pending_cli_message.get("_db_persisted"): agent._pending_cli_user_message = None # ── Idle-triggered compaction (opt-in; ``idle_compact_after_seconds``) ── - # When a session resumes after a long idle gap, compact the accumulated - # history up front so the rest of the conversation does not keep re-reading - # a large stale context on every turn. This fires on elapsed wall-clock time - # rather than size, so it complements (does not replace) the token-threshold - # preflight below. ``_last_activity_ts`` is the last time this turn loop did - # work; nothing has touched it yet this turn, so it measures the gap since - # the previous turn finished. The cheap gap pre-check gates the (more - # expensive) token estimate, mirroring ``_should_run_preflight_estimate``. + # Fires on wall-clock gap since ``_last_activity_ts``, complementing the token + # gate; a cheap gap check gates the estimate (cf. _should_run_preflight_estimate). _idle_after = getattr(agent, "compression_idle_compact_after_seconds", 0) if agent.compression_enabled and _idle_after > 0 and messages: _idle_gap = time.time() - getattr(agent, "_last_activity_ts", time.time()) if _idle_gap >= _idle_after: _compressor = agent.context_compressor - # Route-aware pressure (#96995/#97602 class): on a compacted - # native-Codex session the generic durable-history figure - # overstates the wire by orders of magnitude and would fire an - # idle compaction the next request never needed. Reuse the - # preflight estimator (anchor → native pruned → generic). + # Route-aware pressure: on compacted native-Codex sessions the durable + # figure overstates the wire; reuse the preflight estimator (#96995). _idle_tokens = _preflight_request_tokens( agent, messages, @@ -994,28 +773,21 @@ def build_turn_context( messages, system_message, approx_tokens=_idle_tokens, task_id=effective_task_id, ) - # ``_compress_context`` returns the INPUT list object when it - # skips (per-session lock held by another path, failure - # cooldown, anti-thrash breaker, codex-native routing). Only - # re-baseline + re-anchor after a real compaction — a skip - # must leave the turn's flush baseline and user-message index - # untouched. + # ``_compress_context`` returns the INPUT list object when it skips; + # only re-baseline and re-anchor after a real compaction. if messages is not _idle_input: conversation_history = conversation_history_after_compression( agent, messages, conversation_history ) - # Compaction rebuilt the list, so the index of this turn's - # just-appended user message is stale — re-anchor it the - # same way the preflight path does below. + # Compaction rebuilt the list; re-anchor this turn's user index. current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) agent._persist_user_message_idx = current_turn_user_idx # ── Preflight context compression ── - # Gate the (expensive) full token estimate behind a cheap pre-check. - # See ``_should_run_preflight_estimate`` for the OR semantics that fix - # issue #27405 (a few very large messages slipping past the count gate). + # Cheap pre-check gates the full estimate; see ``_should_run_preflight_estimate`` + # for the OR semantics (#27405). _preflight_compressed = False _preflight_compression_blocked = False agent._turn_received_provider_response = False @@ -1036,10 +808,8 @@ def build_turn_context( active_system_prompt or "", ) _compressor = agent.context_compressor - # getattr guard: minimal compressor doubles (SimpleNamespace in the - # engine-preflight tests) and plugin context engines lack this - # ContextCompressor-only method — absence means no snapshot, and the - # finalizer's rollback stays disarmed for the turn (display-only). + # getattr guard: compressor doubles and plugin engines lack this method — + # absence means no snapshot and the finalizer's rollback stays disarmed. _snapshot_fn = getattr( _compressor, "snapshot_preflight_display_tokens", None ) @@ -1073,13 +843,9 @@ def build_turn_context( ) if not _preflight_deferred: - # Display-only seed (see - # ContextCompressor.maybe_seed_preflight_display_tokens): a real - # provider reading always wins over the rough estimate, and the - # -1 post-compression sentinel (#36718) stays protected. On - # usage-less responses the seed also feeds the tool-loop - # compression gate — the one live path where an inflated seed - # could push compression below the user threshold. + # Display-only seed: a real provider reading wins and the -1 sentinel + # stays protected (#36718). Also feeds the tool-loop gate on usage-less + # responses. _maybe_seed = getattr( _compressor, "maybe_seed_preflight_display_tokens", None ) @@ -1123,12 +889,8 @@ def build_turn_context( else: _should_compress_now = _compressor.should_compress(_preflight_tokens) if not _should_compress_now: - # Context is over threshold but compression is blocked - # (summary-LLM cooldown or anti-thrashing). Ask should_compress_info - # for the human-readable reason so we can surface a warning below. - # getattr guard: minimal compressor doubles (SimpleNamespace in - # the engine-preflight tests) and older plugin engines lack the - # method — absence means no block reason, no warning. + # Over threshold but blocked: ask should_compress_info for the reason + # to surface below. getattr guard: doubles/older engines lack it. _info = getattr(_compressor, "should_compress_info", None) if callable(_info): try: @@ -1136,9 +898,8 @@ def build_turn_context( except Exception: _compress_block_reason = None if _should_compress_now: - # Managed local runtime: growing the window beats compressing — - # the ladder's design order (same seam as the conversation - # loop's pre-API gate; see _maybe_grow_local_window there). + # Managed local runtime: growing the window beats compressing (ladder + # order; same seam as _maybe_grow_local_window in the loop). try: from agent.conversation_loop import _maybe_grow_local_window @@ -1165,11 +926,8 @@ def build_turn_context( ) if _should_compress_now: _preflight_compressed = True - # Compression is actually running (block cleared / was never - # blocked) — reset the dedup so a future blocked-over-threshold - # turn can warn again. Real session boundary. - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Compression is actually running — reset the dedup so a future blocked + # turn can warn again. getattr guard: object.__new__ doubles lack it. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() @@ -1194,9 +952,8 @@ def build_turn_context( ) if _preflight_status: agent._emit_status(_preflight_status) - # Preflight passes honor the same configured per-turn cap - # (compression.max_attempts) as the loop's compression sites; - # default 3 preserves the prior hardcoded behavior. + # Preflight passes honor compression.max_attempts like the loop's sites + # (default 3). _max_preflight_passes = max( 1, int(getattr(agent, "max_compression_attempts", 3) or 3) ) @@ -1212,24 +969,17 @@ def build_turn_context( messages is _preflight_input and compression_skipped_due_to_lock(agent) ): - # #69870 lock-skip: another path holds this session's - # compression lock, so the pass no-oped. That is a - # temporary DEFER, not proof the transcript cannot - # compress — do NOT arm the insufficient-progress - # blocker (the loop's error handlers must keep their - # provider-proven retry budget) and stop preflight - # passes for this turn; the lock winner is shrinking - # the same session concurrently. + # Lock-skip (#69870): another path holds the lock, so this is a + # DEFER, not proof of incompressibility — don't arm the blocker; + # stop preflight passes for this turn. logger.info( "Preflight compression deferred: compression lock " "held by another path (session %s)", agent.session_id or "none", ) break - # Re-estimate now so size-only compression (same row count, - # lower token count — e.g. summarising tool outputs) is - # recognised as progress instead of being misread as - # "Cannot compress further". Fixes #39548. + # Re-estimate so size-only compression (same rows, fewer tokens) + # counts as progress (#39548). _preflight_tokens = _preflight_request_tokens( agent, messages, @@ -1265,29 +1015,21 @@ def build_turn_context( ) break elif _compress_block_reason: - # Context is already over the compression threshold, but compression - # is blocked (summary LLM cooldown or anti-thrashing). Without a - # signal the session keeps growing until the model silently stops - # answering — the conversation hits the hard provider token limit - # with no explanation. Surface a deduped warning so the user can - # take action (/new or /compress) instead of hitting a silent hang. + # Over threshold but compression blocked: surface a deduped warning so + # the user can /new or /compress instead of a silent provider limit. agent._warn_context_overflow_blocked( _compress_block_reason, _preflight_tokens, _compressor.threshold_tokens, ) else: - # Sub-threshold and unblocked — allow the overflow warning to fire - # again next time the context is over threshold but blocked. - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Sub-threshold and unblocked — re-arm the overflow warning. getattr guard: + # object.__new__ test doubles lack the method. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() - # Engine maintenance only when NO skip-branch fired: a failure - # cooldown, deferred estimate, or codex-native route must keep - # the engine hook un-consulted (#20316 contract — the cooldown - # exists precisely because compression recently failed). + # Engine maintenance only when NO skip-branch fired: cooldown, deferred + # estimate, or codex-native route keep the engine hook unconsulted (#20316). if _compression_cooldown or _preflight_deferred or _codex_native_auto: _engine_preflight = None else: @@ -1295,27 +1037,8 @@ def build_turn_context( _compressor, "should_compress_preflight", None ) # ── Engine-driven sub-threshold preflight maintenance (#20316) ── - # None of the threshold-path branches fired (not deferred, no - # failure cooldown, not codex-native, and should_compress() said - # the request is under pressure). Context engines that override - # ``should_compress_preflight()`` (e.g. LCM-style incremental - # leaf-chunk compaction) can still request deferred maintenance - # below the token threshold. The default - # ``ContextEngine.should_compress_preflight()`` returns False, so - # the built-in ``ContextCompressor`` path is byte-identical. - # - # Attempt-cap integration: the engine gets exactly ONE - # ``compress()`` pass per turn. It is mutually exclusive with the - # threshold multi-pass loop above (if/elif), so turn-start - # preflight passes stay bounded by the resolved - # ``compression.max_attempts`` cap (floor 1) in every case. - # - # No-op-blocking integration: a sub-threshold engine pass that - # no-ops says nothing about over-threshold compressibility, so it - # must neither set nor clear ``_preflight_compression_blocked`` - # (#64382) — and being in the ``else`` arm it can never run after - # the threshold loop has proven a retry ineffective. - # (resolved above, gated on no skip-branch having fired) + # Engines overriding ``should_compress_preflight()`` get exactly ONE + # ``compress()`` pass; a no-op never touches _preflight_compression_blocked. _wants_engine_preflight = False if callable(_engine_preflight): try: @@ -1342,12 +1065,8 @@ def build_turn_context( messages, system_message, approx_tokens=_preflight_tokens, task_id=effective_task_id, ) - # ``_compress_context`` returns the INPUT list object on every - # skip path (per-session lock held elsewhere, cooldown, - # anti-thrash breaker, codex-native routing) and an engine may - # legitimately no-op. Only re-baseline the flush history and - # re-anchor the user row after a REAL compaction — a skip must - # leave the turn's bookkeeping untouched. + # ``_compress_context`` returns the INPUT list on every skip path and an + # engine may no-op; re-baseline/re-anchor only after a REAL compaction. if messages is not _engine_input: _preflight_compressed = True conversation_history = conversation_history_after_compression( @@ -1359,16 +1078,8 @@ def build_turn_context( agent._last_content_tools_all_housekeeping = False agent._mute_post_response = False elif not agent.compression_enabled: - # Uncompressed session guard (#89297): when compression is explicitly - # disabled, sessions can grow past the model's context window across - # hundreds of messages with nothing to shrink them. The warning itself - # fires from the conversation loop's pre-API site, which reuses the - # unconditionally computed request estimate at zero marginal cost and - # covers both turn-start and mid-turn growth (every provider request - # passes through it). Here we only RE-ARM the dedup once the session - # is back under the window, so the guard can warn again after the - # user compacts (/compress with force=True works with compression - # disabled) and the context later regrows past the limit. + # Uncompressed session guard (#89297): the warning fires from the loop's + # pre-API site; here we only RE-ARM the dedup once back under the window. _ctx_len = getattr( getattr(agent, "context_compressor", None), "context_length", None ) @@ -1381,16 +1092,12 @@ def build_turn_context( if isinstance(_c, str): _raw_chars += len(_c) elif _c: - # Non-string, non-empty content (multimodal part lists, - # dict payloads) defeats a char count — force the real - # estimate by treating it as over-gate. None/"" (routine - # assistant tool-call rows) contribute nothing. + # Non-string, non-empty content defeats a char count — force the + # real estimate. None/"" contribute nothing. _raw_chars = _ctx_len + 1 break - # Cheap gate: a session whose raw text is under ~1/4 of the - # window (4 chars/token upper bound) cannot be over it — skip - # the estimator. Non-string (multimodal) content defeats a char - # count, so any such message forces the real estimate. + # Cheap gate: raw text under ~1/4 of the window (4 chars/token) cannot + # be over it; non-string (multimodal) content forces the real estimate. if _raw_chars <= _ctx_len: _clear_warn = getattr( agent, "_clear_context_overflow_warn", None @@ -1398,12 +1105,9 @@ def build_turn_context( if callable(_clear_warn): _clear_warn() else: - # Route-aware (#96995/#97602 class): the warn site in the - # conversation loop now measures the checkpoint-pruned wire - # payload on native-Codex sessions, so the re-arm must use - # the same figure — otherwise a compacted session that fits - # on the wire never clears the dedup and future genuine - # overflow warnings stay suppressed. + # Re-arm with the same route-aware (checkpoint-pruned wire) figure the + # warn site measures, else a compacted session never clears the dedup + # and genuine overflow warnings stay suppressed (#96995/#97602). _uncompressed_tokens = _preflight_request_tokens( agent, messages, @@ -1417,13 +1121,9 @@ def build_turn_context( _clear_warn() if _preflight_compressed: - # Compression rebuilt the list (tail messages are fresh compaction - # copies), so the pre-compression index of this turn's user message - # is stale. Re-anchor both index trackers: the api_content stamp - # below, the loop's injection site, and the flush's persist-override - # row (#48677) must all target the surviving dict, not a stale - # position. Exact-content match first so a todo-snapshot user message - # appended after the tail can't steal the anchor. + # Compression rebuilt the list, so the pre-compression user index is stale. + # Re-anchor so the api_content stamp, injection site, and persist-override row + # hit the same dict; exact-content match first so a todo-snapshot can't steal it current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) @@ -1447,9 +1147,8 @@ def build_turn_context( sender_id=getattr(agent, "_user_id", None) or "", ) _ctx_parts: list[str] = [] - # Spill oversized per-hook context to disk so a runaway plugin - # can't inflate every subsequent turn's prompt. Ported from - # openai/codex PR #21069 ("Spill large hook outputs from context"). + # Spill oversized per-hook context to disk so a runaway plugin can't inflate + # every subsequent turn's prompt. try: from tools.hook_output_spill import ( get_spill_config as _spill_cfg, @@ -1483,12 +1182,9 @@ def build_turn_context( except Exception as exc: logger.warning("pre_llm_call hook failed: %s", exc) - # Gateway must-deliver notes (auto-reset note, first-contact intro, - # voice-channel change) ride the same user-message injection channel as - # plugin context so the ephemeral system prompt can stay byte-stable. - # One-shot: staged by the gateway right before this turn, consumed here. - # Multimodal (list) content can't take the string sidecar — append a - # durable text part instead of dropping the fact. + # Gateway must-deliver notes ride the user-message injection channel (one-shot, + # gateway-staged) so the ephemeral system prompt stays byte-stable. Multimodal + # (list) content can't take the string sidecar — append a durable text part. _gateway_notes = consume_gateway_turn_context_notes(agent) if _gateway_notes: _gw_turn_content = ( @@ -1538,10 +1234,8 @@ def build_turn_context( except Exception: pass - # External memory provider: prefetch once before the tool loop. - # - # Skip prefetch on trivial prompts (greetings, acknowledgements) to - # prevent memory-context injection on turns that carry no semantic signal. + # External memory provider: prefetch once before the tool loop. Skipped on + # trivial prompts (greetings, acks) that carry no semantic signal. ext_prefetch_cache = "" if agent._memory_manager: try: @@ -1550,10 +1244,8 @@ def build_turn_context( ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or "" except Exception: pass - # Deterministic, model-independent recall indicator: when memory was - # actually injected this turn, tell the user — don't rely on the model - # to surface it. Rendered by Hermes (via _emit_status), so it always - # shows and can't be silently dropped by the model. + # Deterministic recall indicator: rendered by Hermes via _emit_status when + # memory was injected, so the model can't silently drop it. if ext_prefetch_cache: try: _recall_indicator = agent._memory_manager.describe_recall() @@ -1563,22 +1255,8 @@ def build_turn_context( pass # ── api_content sidecar: persist what you send ── - # The prefetch/plugin context above is injected into the API copy of this - # turn's user message, never into the stored content — so on the next - # turn the message would replay WITHOUT the injection, diverging the - # request prefix at this point and re-prefilling everything after it - # (the whole previous turn's assistant/tool chain). Stamp the exact - # API-bound bytes on the live dict, only when they differ from the clean - # content, so the crash persist below writes both in the same row and - # replay can reproduce the sent prefix byte-for-byte. Guarded by the - # same predicate the api_messages build uses, so the stamped bytes are - # exactly the bytes the loop sends. codex_app_server turns bypass the - # api_messages build entirely (the codex thread gets the plain user - # message), so stamping there would persist bytes that were never sent. - # MoA turns append per-call aggregated reference context to the same API - # copy AFTER this composition, so the stamped bytes would never match the - # wire either — skip the stamp rather than persist provably wrong "exact - # sent bytes" (MoA keeps its pre-sidecar cache behavior). + # Injected context lives only in the API copy; stamp the exact sent bytes on the + # live dict so replay reproduces the prefix. Skipped for codex_app_server/MoA. if ( not moa_active and getattr(agent, "api_mode", None) != "codex_app_server" @@ -1591,14 +1269,9 @@ def build_turn_context( ) if _api_content is not None and _api_content != _turn_user_msg.get("content"): _turn_user_msg["api_content"] = _api_content - # In-place preflight compaction has ALREADY inserted this turn's - # user row (archive_and_compact runs before prefetch/pre_llm_call - # can compose the sidecar), and the crash persist below identity- - # skips every compacted dict (they are all in the rebound - # conversation_history) — so the stamp would never reach the DB. - # Backfill it onto the freshly-inserted row directly. Rotation - # mode needs nothing here: its compacted copies flush to the - # child session after this stamp. + # In-place preflight compaction already inserted this turn's user row and + # the crash persist identity-skips compacted dicts, so backfill the stamp + # onto the row directly. Rotation mode flushes to the child session later. if _preflight_compressed and bool( getattr(agent, "_last_compaction_in_place", False) ): @@ -1618,13 +1291,9 @@ def build_turn_context( exc_info=True, ) - # Crash-resilience: persist the inbound user turn before the first LLM - # call. Runs after preflight compression (which rewrites history anyway) - # and after prefetch/pre_llm_call, so the user row is written once with - # its final api_content instead of being re-written mid-turn. - # Keep row creation and the marker-based append in the same per-agent - # critical section as CLI close persistence, and retry the row create if - # the pre-compression attempt above failed transiently. + # Crash-resilience: persist the inbound user turn once, with final api_content, + # before the first LLM call. Same critical section as CLI close persistence; + # retries the row create if the pre-compression attempt failed transiently. def _ensure_and_persist() -> None: agent._ensure_db_session() agent._persist_session(messages, conversation_history) @@ -1642,20 +1311,13 @@ def build_turn_context( exc_info=True, ) finally: - # Keep an unmarked staged input available to a later close retry if the - # normal persistence attempt failed. Once the marker is present, the - # close path must no longer treat it as a pre-worker UI input. + # Keep an unmarked staged input for a later close retry if persistence failed; + # once marked, the close path must not treat it as a pre-worker UI input. if not isinstance(pending_cli_message, dict) or pending_cli_message.get("_db_persisted"): agent._pending_cli_user_message = None - # Title the session from this user message, now — the row exists and the - # turn has not called the model yet. Titling is derived from the user's - # ask alone, so it runs concurrently with the turn instead of waiting for - # a final response; on a long tool-heavy first turn that is the difference - # between a title in ~1s and a title minutes later (or never, when the - # turn failed before producing one). Fire-and-forget on a daemon thread, - # a no-op once the session has a title, and shared by every surface - # because every surface enters the turn through this prologue. + # Title the session now: the row exists and titling depends only on the user's + # ask, so it runs concurrently with the turn. Daemon thread, no-op once titled. _maybe_title_session_at_turn_start(agent, messages) return TurnContext( diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index 93506f680c..ca52cff981 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -1,24 +1,9 @@ """Post-loop turn finalization for ``run_conversation``. -Extracted from ``agent/conversation_loop.py`` as part of the god-file -decomposition campaign (``~/.hermes/plans/god-file-decomposition.md``, Phase 1 -step 4 — the post-loop ``TurnFinalizer`` seam). ``run_conversation``'s tail -(everything after the main tool-calling ``while`` loop) is lifted here verbatim: -budget-exhaustion summary, trajectory save, session persist, turn diagnostics, -response transforms, result-dict assembly, steer drain, and the memory/skill -review trigger. - -Behavior-neutral: the body is moved unchanged. All ``agent.*`` side effects fire -exactly as before; only the post-loop *locals* are passed in as keyword args, and -the assembled ``result`` dict is returned to ``run_conversation`` which returns it -to the caller. The function is synchronous with a single return — mirroring the -region it replaces (no awaits, no early returns). - -Module ``logger`` is imported lazily inside the body (``from -agent.conversation_loop import logger``) so this module never imports -``agent.conversation_loop`` at import time -> no import cycle, and the log records -keep the exact logger name (``"agent.conversation_loop"``). -""" +Lifted verbatim: budget summary, trajectory save, persist, diagnostics, response +transforms, result assembly, steer drain, memory/skill review. Synchronous, single +return. ``logger`` is imported lazily from ``agent.conversation_loop`` (no cycle, +same logger name).""" from __future__ import annotations @@ -54,11 +39,9 @@ def _fill_assistant_tail_content(agent, tail: dict, final_response) -> None: agent._db_flush_scan_prefix = None -# Verification continuation scaffolding flags: verify-on-stop / pre_verify -# inject a synthetic user nudge to keep the agent going one more turn. -# These nudges must be stripped from returned/live history to avoid -# role-alternation breaks and poisoning the resumed transcript. The -# assistant response is real content and is not flagged. (#65919 §7) +# Verification-continuation nudges (verify-on-stop / pre_verify) must be stripped from +# returned/live history to avoid role-alternation breaks; the assistant response is +# real content and is not flagged. (#65919) _VERIFICATION_CONTINUATION_FLAGS = ( "_verification_stop_synthetic", "_pre_verify_synthetic", @@ -71,14 +54,10 @@ def _record_kanban_budget_exhausted( max_iterations: int, logger: logging.Logger, ) -> None: - """Record a terminal ``timed_out`` outcome for a kanban worker that - exhausted its iteration budget. + """Record a terminal ``timed_out`` outcome for a kanban worker out of budget. - This is a bounded fallback (#87096): the CAS invariant in ``_end_run`` - (``WHERE ended_at IS NULL``) guarantees idempotence — if another path - already closed the run this is a no-op — so it is safe to call from - multiple exit paths. - """ + Idempotent via the ``_end_run`` CAS (``WHERE ended_at IS NULL``): a no-op if + another path already closed the run, so safe from multiple exit paths (#87096).""" try: from hermes_cli import kanban_db as _kb _conn = _kb.connect() @@ -116,10 +95,8 @@ def _record_kanban_budget_exhausted( def _drop_verification_continuation_scaffolding(messages) -> None: """Remove verification-continuation nudge messages from *messages* in place. - Only the synthetic nudges carry these flags, so this strips just the - nudges while preserving the real attempted-final-answer that was - persisted to state.db. - """ + Only the synthetic nudges carry these flags, so the real attempted-final-answer + persisted to state.db survives.""" messages[:] = [ m for m in messages if not (isinstance(m, dict) and any(m.get(f) for f in _VERIFICATION_CONTINUATION_FLAGS)) @@ -153,11 +130,7 @@ def finalize_turn( _pending_verification_response=None, _pending_verification_response_previewed=False, ): - """Run the post-loop finalization and return the turn ``result`` dict. - - Lifted verbatim from ``run_conversation`` (the region after the main agent - loop). See module docstring. - """ + """Run the post-loop finalization and return the turn ``result`` dict.""" from agent.conversation_loop import logger budget_exhausted = ( @@ -179,24 +152,19 @@ def finalize_turn( iteration_limit_fallback = False preserved_verification_fallback = False if continuation_budget_exhausted: - # A verification/continuation gate deliberately withheld a composed - # answer, then consumed the remaining budget before producing a newer - # one. Preserve that exact answer instead of replacing it with another - # fallible model call. The explicit pending value is the provenance - # guard: unrelated error/recovery exits can never enter this branch. + # A verification gate withheld a composed answer, then the budget ran out: + # preserve it rather than make another fallible call. The explicit pending + # value is the provenance guard; unrelated error exits never enter here. final_response = _pending_verification_response - # Mark the turn as previewed only when the reused candidate was - # actually streamed to the user as interim content. (#65919 review: - # response-loss blocker) + # Previewed only if the reused candidate was actually streamed as interim. if _pending_verification_response_previewed: agent._response_was_previewed = True _turn_exit_reason = f"max_iterations_reached({api_call_count}/{agent.max_iterations})" iteration_limit_fallback = True preserved_verification_fallback = True elif final_response is None and budget_fallback_eligible: - # Budget exhausted — ask the model for a summary via one extra - # API call with tools stripped. _handle_max_iterations injects a - # user message and makes a single toolless request. + # Budget exhausted: _handle_max_iterations makes one extra toolless request + # for a summary. _turn_exit_reason = f"max_iterations_reached({api_call_count}/{agent.max_iterations})" agent._emit_status( f"⚠️ Iteration budget exhausted ({api_call_count}/{agent.max_iterations}) " @@ -211,29 +179,18 @@ def finalize_turn( iteration_limit_fallback = True if iteration_limit_fallback: - # If running as a kanban worker, signal the dispatcher that the - # worker could not complete (rather than treating it as a - # protocol violation). This applies whether the user-facing fallback - # came from the summary call or an explicitly pending continuation; - # both exhausted the task budget and must advance the failure circuit. - # - # We route through ``_record_task_failure(outcome="timed_out")`` - # rather than ``kanban_block`` so this counts toward the dispatcher's - # consecutive-failure circuit breaker (#29747 gap 2). + # Kanban worker: signal the dispatcher the worker could not complete. Route + # via ``_record_task_failure(outcome="timed_out")`` (not ``kanban_block``) so + # it counts toward the consecutive-failure circuit breaker (#29747). _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if _kanban_task: _record_kanban_budget_exhausted( _kanban_task, api_call_count, agent.max_iterations, logger, ) elif budget_exhausted: - # Bounded fallback (#87096): budget was exhausted but none of the - # normal fallback paths were eligible (interrupted / failed / - # anomalous exit_reason). If running as a kanban worker we must - # still record a terminal outcome so the task does not remain in - # an ambiguous lifecycle state. The worker's run is closed via - # ``_record_task_failure`` (compare-and-swap receipt path) which - # is a no-op if another path closed it — the CAS invariant in - # ``_end_run`` (``WHERE ended_at IS NULL``) guarantees idempotence. + # Bounded fallback: budget exhausted with no eligible fallback path. A kanban + # worker must still record a terminal outcome; the ``_end_run`` CAS makes it + # idempotent if another path already closed the run (#87096). _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if _kanban_task: _record_kanban_budget_exhausted( @@ -251,18 +208,9 @@ def finalize_turn( ) ) - # Preflight can seed the display count before the provider receives the - # request. Roll that estimate back only when an interrupt wins the race - # before any successful provider response. Compaction state remains owned - # by the real-usage/post-compaction path, including its ``-1`` sentinel. - # Guard rules (test-double density on this path is high): - # - snapshot is type-pinned to a real int — MagicMock agents auto-create - # truthy Mock attributes that must never arm the rollback; - # - the received-response flag is pinned to ``is not True`` — its real - # domain is True/False, and only a literal True means a provider - # response completed; - # - the compressor method gets a getattr+callable guard — SimpleNamespace - # compressor doubles and plugin context engines lack it. + # Roll back the preflight-seeded display count only when an interrupt wins + # before any provider response; compaction state (incl. ``-1``) stays with the + # real-usage path. Type-pinned guards keep MagicMock/SimpleNamespace doubles inert. _preflight_snapshot = getattr( agent, "_turn_preflight_display_snapshot", None ) @@ -281,17 +229,9 @@ def finalize_turn( if callable(_rollback_fn): _rollback_fn(_preflight_snapshot) - # Post-loop cleanup must never lose the response. Trajectory save, - # resource teardown, and session persistence all touch fallible - # surfaces — file I/O / JSON serialization (_save_trajectory), remote - # VM/browser teardown over the network (_cleanup_task_resources), and - # SQLite writes (_persist_session). A raise from any of them used to - # propagate straight out of run_conversation, discarding the partial - # final_response the caller is waiting for (subprocess wrappers saw an - # empty stdout with no traceback — #8049). Each step is now guarded - # independently so one failure can't skip the others, and any errors - # are surfaced on the result dict via ``cleanup_errors`` rather than - # killing the turn. + # Post-loop cleanup must never lose the response: trajectory save, teardown, + # and session persist are guarded independently and errors surface via + # ``cleanup_errors`` rather than killing the turn (#8049). _cleanup_errors = [] # Save trajectory if enabled. ``user_message`` may be a multimodal @@ -309,22 +249,17 @@ def finalize_turn( _cleanup_errors.append(f"cleanup_task_resources: {_cleanup_err}") logger.error("finalize_turn: _cleanup_task_resources failed: %s", _cleanup_err, exc_info=True) - # Persist session to both JSON log and SQLite only after private retry - # scaffolding has been removed. Otherwise a later user "continue" turn - # can replay assistant("(empty)") / recovery nudges and fall into the - # same empty-response loop again. + # Persist only after private retry scaffolding is removed, or a later "continue" + # replays assistant("(empty)") / recovery nudges into the same empty-response loop. try: agent._drop_trailing_empty_response_scaffolding(messages) - # Drop verification-continuation nudges (synthetic user messages) - # from the live history before the tail-assistant check — only the - # nudges need stripping; the assistant candidate persists in - # state.db. (#65919 §7) + # Strip only the synthetic verification nudges before the tail-assistant + # check; the assistant candidate persists in state.db. (#65919) _drop_verification_continuation_scaffolding(messages) - # #95514: an empty terminal completion is not authoritative when the - # stream already delivered text. Recover before persist so a blank - # assistant tail is filled instead of frozen as content=''. + # An empty terminal completion is not authoritative when the stream already + # delivered text; recover before persist so a blank tail isn't frozen (#95514). _recovered_from_stream = False if not interrupted and not failed: _streamed = getattr(agent, "_current_streamed_assistant_text", "") or "" @@ -337,39 +272,16 @@ def finalize_turn( final_response = _streamed _recovered_from_stream = True - # When the turn was interrupted and the last message is a tool - # result, append a synthetic assistant message to close the - # tool-call sequence. Without this, the session persists a - # ``tool → user`` alternation that strict providers (Gemini, - # Claude) reject, causing them to hallucinate a continuation of - # the user's message on the next turn (#48879). - # - # ``_drop_trailing_empty_response_scaffolding`` only rewinds the - # tool tail when an empty-response scaffolding flag is present; a - # clean ``/stop`` interrupt after a successful tool sets no such - # flag, so the tool result survives as the tail and we close it - # here instead. On an interrupt ``final_response`` is typically - # empty, so fall back to an explicit placeholder rather than - # persisting an empty-content assistant turn. + # An interrupt can leave a tool result as the tail (no scaffolding flag rewinds + # it); close the sequence so strict providers don't see ``tool → user``. An + # explicit placeholder is used since final_response is usually empty (#48879). if interrupted: from agent.message_sanitization import close_interrupted_tool_sequence close_interrupted_tool_sequence(messages, final_response) - # Some recovery/fallback paths return a real final_response without - # adding a closing assistant message to the transcript (e.g. the - # partial-stream and prior-turn-content recovery ``break`` sites in - # ``conversation_loop``). If persisted as-is, the durable session can - # end at a tool/user message even though the caller — and the gateway - # platform — already saw a completed assistant response. The next turn - # then replays a user-only backlog and the model re-answers every - # "unanswered" message. Close the durable turn at the source, at the - # single chokepoint every recovery ``break`` flows through, so the - # invariant "delivered final_response ⇒ assistant row in transcript" - # holds regardless of which path produced it. (#43849 / #44100) - # - # Compare content (not just role) so a verification candidate that - # matches the final response is not duplicated at budget - # exhaustion. (#65919 §7) + # Recovery ``break`` sites can return a final_response with no closing + # assistant row; enforce "delivered final_response ⇒ assistant row" here. + # Compare content, not role, so a matching verification candidate isn't dup'd. if final_response and not interrupted: try: _tail = messages[-1] if messages else None @@ -394,66 +306,45 @@ def finalize_turn( ) ) ): - # The tail IS an assistant row, but a *pure tool-call turn* or - # a blank assistant tail whose content was recovered from the - # stream buffer (#95514). Fill that row's content instead of - # appending, so the durable turn ends with the answer without - # creating an assistant→assistant pair. + # Tail is an assistant row (pure tool-call turn or stream-recovered + # blank, #95514): fill its content rather than append a second row. _fill_assistant_tail_content(agent, _tail, final_response) - # The model has completed its request, so replace API-local - # voice/model/skill guidance with the clean user input before writing the - # final durable snapshot and returning the continuation history. Earlier - # turn-start flushes use the DB-only override because their messages are - # still needed for the API request; this finalizer runs after that request - # is complete (#48677 / #63766). + # Request is complete, so replace API-local voice/model/skill guidance with + # the clean user input before the durable snapshot; earlier flushes used the + # DB-only override as their messages were still needed (#48677 / #63766). _apply_override = getattr(agent, "_apply_persist_user_message_override", None) if callable(_apply_override): _apply_override(messages) # ── Post-turn micro-compaction ──────────────────────────── - # After the assistant response is finalized but before the session is - # persisted, run micro-compaction to absorb the oldest uncompacted - # exchange into the rolling summary. This amortizes compression - # across turns rather than batching it into one big pause. + # Absorb the oldest uncompacted exchange into the rolling summary before + # persist, amortizing compression across turns instead of one big pause. if not interrupted and not failed: try: _compressor = getattr(agent, "context_compressor", None) - # Strict `is True` + isinstance gates: plugin context engines - # (and MagicMock compressors in tests) satisfy getattr/duck - # checks with truthy auto-attributes — a bare truthiness check - # here called _micro_compact on a mock and spliced its (empty- - # iterating) return value over the transcript, wiping it. + # Strict `is True` + isinstance gates: plugin context engines and + # MagicMock compressors pass duck checks and would wipe the transcript. if ( _compressor and getattr(_compressor, '_micro_compact_enabled', False) is True and callable(getattr(_compressor, '_micro_compact', None)) and final_response - # compression.checkpoint_required: agent init already - # forces _micro_compact_enabled off, but the compressor - # attribute is plain state a future path could flip on a - # live agent. Micro-compaction has no checkpoint hook in - # its path, so it must never run while the gate is armed. + # Micro-compaction has no checkpoint hook, so it must never run + # while compression.checkpoint_required is armed. and getattr( agent, "compression_checkpoint_required", False ) is not True - # Persistence-isolated agents (background review fork) - # must not micro-compact: the pass burns a real aux-LLM - # call on a throwaway replay transcript, and if the - # compressor ever holds a session_db binding it would - # archive_and_compact the CANONICAL session rows — the - # exact write class _persist_disabled exists to stop. + # Persistence-isolated agents (background review fork) must not + # micro-compact: it burns an aux-LLM call on a throwaway transcript + # and could archive_and_compact the CANONICAL session rows. and not getattr(agent, "_persist_disabled", False) ): _before = len(messages) _compacted = _compressor._micro_compact(messages) - # Micro-compaction defrag rewrites the newest MICRO - # marker's content and pops _db_persisted from the live - # dict in place — the sibling of the pop site above. The - # compressor has no agent reference, so it raises a flag - # for us to invalidate the bounded flush-scan cursor; - # otherwise the rewritten marker row is identity-skipped - # and the stale summary persists to state.db. + # Defrag rewrites the newest MICRO marker in place and pops + # _db_persisted; the compressor flags us to invalidate the flush- + # scan cursor, else the rewritten row is identity-skipped (stale). if getattr( _compressor, "_flush_scan_cursor_invalidated", False ): @@ -475,18 +366,16 @@ def finalize_turn( _cleanup_errors.append(f"persist_session: {_persist_err}") logger.error("finalize_turn: _persist_session failed: %s", _persist_err, exc_info=True) - # The gateway owns a separate in-memory history snapshot. Keep it current - # even when finalization reports a cleanup error: a later prompt must not be - # sent with the pre-turn snapshot while the durable DB already has this turn. + # Keep the gateway's separate in-memory history snapshot current even on + # cleanup error, so a later prompt isn't sent with a pre-turn snapshot. try: agent._session_messages = messages except Exception: pass # ── Turn-exit diagnostic log ───────────────────────────────────── - # Always logged at INFO so agent.log captures WHY every turn ended. - # When the last message is a tool result (agent was mid-work), log - # at WARNING — this is the "just stops" scenario users report. + # Always INFO so agent.log captures WHY every turn ended; WARNING when the last + # message is a tool result (the "just stops" scenario). _last_msg_role = messages[-1].get("role") if messages else None _last_tool_name = None if _last_msg_role == "tool": @@ -527,21 +416,9 @@ def finalize_turn( else: logger.info(_diag_msg, *_diag_args) - # File-mutation verifier footer. - # If one or more ``write_file`` / ``patch`` calls failed during this - # turn and were never superseded by a successful write to the same - # path, append an advisory footer to the assistant response. This - # catches the specific case — reported by Ben Eng (#15524-adjacent) - # — where a model issues a batch of parallel patches, half of them - # fail with "Could not find old_string", and the model summarises - # the turn claiming every file was edited. The user then has to - # manually run ``git status`` to catch the lie. With this footer - # the truth is surfaced on every turn, so over-claiming is - # structurally impossible past the model. - # - # Gate: only applied when a real text response exists for this - # turn and the user didn't interrupt. Empty/interrupted turns - # already have other surface text that shouldn't be augmented. + # File-mutation verifier footer: if ``write_file`` / ``patch`` calls failed and + # were never superseded by a successful write to the same path, append an + # advisory so over-claiming is surfaced. Only on real, uninterrupted responses. if final_response and not interrupted: try: _failed = getattr(agent, "_turn_failed_file_mutations", None) or {} @@ -552,30 +429,16 @@ def finalize_turn( except Exception as _ver_err: logger.debug("file-mutation verifier footer failed: %s", _ver_err) - # Turn-completion explainer. - # When a turn ends abnormally after substantive work — empty content - # after retries, a partial/truncated stream, a still-pending tool - # result, or an iteration/budget limit — the user otherwise gets a - # blank or fragmentary response box with no consolidated reason why - # the agent stopped (#34452). Surface a single user-visible - # explanation derived from ``_turn_exit_reason``, mirroring the - # file-mutation verifier footer pattern above. - # - # Gate carefully so healthy turns stay quiet: - # - ``text_response(...)`` exits never produce an explanation - # (handled inside the formatter), so a terse ``Done.`` is silent. - # - We only ACT when there is no genuinely usable reply this turn: - # an empty response, the "(empty)" terminal sentinel, or a - # suspiciously short partial fragment with no terminating - # punctuation (e.g. "The"). A real short answer keeps its text. + # Turn-completion explainer: on abnormal exits, surface one explanation from + # ``_turn_exit_reason``. Only acts when no usable reply exists (empty, "(empty)", + # or a short unpunctuated fragment); ``text_response(...)`` exits stay silent. if not interrupted: try: if agent._turn_completion_explainer_enabled(): _stripped = (final_response or "").strip() _is_empty_terminal = _stripped == "" or _stripped == "(empty)" - # A short fragment that is not a normal text_response exit - # and lacks sentence-ending punctuation is treated as a - # truncated partial (the "The" case from #34452). + # A short fragment not from a text_response exit and lacking sentence- + # ending punctuation is treated as a truncated partial (#34452). _is_partial_fragment = ( not _is_empty_terminal and not preserved_verification_fallback @@ -601,9 +464,7 @@ def finalize_turn( # the actionable explanation. final_response = _explanation else: - # Keep the partial fragment, append the reason so - # the user sees both what arrived and why it - # stopped. + # Keep the partial fragment and append why it stopped. final_response = ( _stripped + "\n\n" + _explanation ) @@ -613,10 +474,8 @@ def finalize_turn( _response_transformed = False _pre_transform_response = None - # Plugin hook: transform_llm_output - # Fired once per turn after the tool-calling loop completes. - # Plugins can transform the LLM's output text before it's returned. - # First hook to return a string wins; None/empty return leaves text unchanged. + # Plugin hook: transform_llm_output — fired once per turn after the tool loop. + # First hook to return a string wins; None/empty leaves the text unchanged. if final_response and not interrupted: try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -636,10 +495,8 @@ def finalize_turn( except Exception as exc: logger.warning("transform_llm_output hook failed: %s", exc) - # Plugin hook: post_llm_call - # Fired once per turn after the tool-calling loop completes. - # Plugins can use this to persist conversation data (e.g. sync - # to an external memory system). + # Plugin hook: post_llm_call — fired once per turn after the tool loop (e.g. sync + # conversation data to an external memory system). if final_response and not interrupted: try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -657,18 +514,12 @@ def finalize_turn( except Exception as exc: logger.warning("post_llm_call hook failed: %s", exc) - # Context engine observation hook: notify the active engine that this - # turn has finished, with the finalized transcript. Complements the - # per-request select_context() hook (selection before the request; - # observation after the turn). No-op default, fail-open. + # Context engine observation hook (complements per-request select_context()): + # notify the engine the turn finished with the finalized transcript. Fail-open. try: from agent.conversation_loop import _notify_context_engine_turn_complete - # Forward the turn's canonical usage when the host has it. The loop - # stashes the most recent API response's usage dict (the same - # canonical buckets fed to ``update_from_response``) on the agent as - # ``_last_turn_usage``. It is ``None`` on turns that never reached a - # provider response (early failure / interrupt), which is exactly the - # contract: real usage when available, ``None`` otherwise. + # ``_last_turn_usage`` holds the last API response's canonical usage dict, or + # ``None`` on turns that never reached a provider response — by contract. _turn_usage = getattr(agent, "_last_turn_usage", None) _notify_context_engine_turn_complete( agent, @@ -685,15 +536,9 @@ def finalize_turn( except Exception as exc: logger.warning("on_turn_complete notification failed: %s", exc) - # Extract reasoning from the CURRENT turn only. Walk backwards - # but stop at the user message that started this turn — anything - # earlier is from a prior turn and must not leak into the reasoning - # box (confusing stale display; #17055). Within the current turn - # we still want the *most recent* non-empty reasoning: many - # providers (Claude thinking, DeepSeek v4, Codex Responses) emit - # reasoning on the tool-call step and leave the final-answer step - # with reasoning=None, so picking only the last assistant would - # silently drop legitimate same-turn reasoning. + # Reasoning from the CURRENT turn only: stop at this turn's user message + # (#17055), but take the most recent non-empty reasoning since many providers + # emit it on the tool-call step and leave the final step with reasoning=None. last_reasoning = None for msg in reversed(messages): if msg.get("role") == "user": @@ -702,15 +547,9 @@ def finalize_turn( last_reasoning = msg["reasoning"] break - # Class-level surrogate chokepoint (#80366, #55143, #55309, #19819): - # ``final_response`` is often the RAW SDK content - # (``assistant_message.content``), not the sanitized copy stored in - # history by ``build_assistant_message``. Any lone UTF-16 surrogate - # (U+D800–U+DFFF) in it crashes downstream consumers — oneshot stdout - # writes, Telegram's ``utf16_len`` length check, Signal formatting, - # JSON envelope encodes — on every provider (Ollama, NVIDIA NIM, …). - # Scrub once here, where model text leaves the conversation loop, so - # every delivery surface receives valid Unicode. + # Surrogate chokepoint: ``final_response`` may be RAW SDK content, and a lone UTF-16 + # surrogate crashes downstream consumers (stdout, Telegram ``utf16_len``, JSON). + # Scrub once where model text leaves the loop (#80366). if isinstance(final_response, str): final_response = _sanitize_surrogates(final_response) @@ -752,30 +591,27 @@ def finalize_turn( } if agent._tool_guardrail_halt_decision is not None: result["guardrail"] = agent._tool_guardrail_halt_decision.to_metadata() - # Persistence failures already set failed=True + an explanation in - # final_response; also stamp `error` so gateway surfaces status="error" - # (and desktop can toast the cause) instead of a quiet complete frame. + # Persistence failures already set failed=True; also stamp `error` so the gateway + # surfaces status="error" (and desktop can toast) instead of a quiet complete frame. if failed and str(_turn_exit_reason) == "session_persistence_failed": result["error"] = final_response or ( "session storage could not be written — check the state database " "health (`hermes doctor`), then send your message again" ) - # Machine-readable cause for the gateway/desktop: exactly - # 'session_persistence_failed:'. + # Machine-readable cause for the gateway/desktop, exactly + # 'session_persistence_failed:'. # Never clobber a failure_reason another path already stamped. if "failure_reason" not in result: _cause = getattr(agent, "_last_persistence_error_cause", None) result["failure_reason"] = ( "session_persistence_failed:" + (_cause or "unknown") ) - # Surface any post-loop cleanup failures so the caller can distinguish a - # clean turn from one whose trajectory/session/resource teardown raised - # (the response is still returned either way — #8049). + # Surface post-loop cleanup failures so the caller can tell a clean turn from one + # whose teardown raised; the response is returned either way (#8049). if _cleanup_errors: result["cleanup_errors"] = _cleanup_errors - # If a /steer landed after the final assistant turn (no more tool - # batches to drain into), hand it back to the caller so it can be - # delivered as the next user turn instead of being silently lost. + # A /steer landing after the final assistant turn has no tool batch to drain into; + # hand it back so it becomes the next user turn instead of being lost. _leftover_steer = agent._drain_pending_steer() if _leftover_steer: result["pending_steer"] = _leftover_steer @@ -807,11 +643,9 @@ def finalize_turn( messages=messages, ) - # Background memory/skill review — runs AFTER the response is delivered - # so it never competes with the user's task for model attention. - # Suppressed when skip_background_review=True (e.g. cron) — review forks - # spawn another AIAgent (~30K tokens / event) and cron sessions have no - # human-in-the-loop benefit from the review. + # Background memory/skill review runs AFTER delivery so it never competes with the + # user's task. Suppressed by skip_background_review (e.g. cron): the fork costs + # ~30K tokens / event with no human-in-the-loop benefit. if ( final_response and not interrupted @@ -829,16 +663,10 @@ def finalize_turn( except Exception: pass # Background review is best-effort - # Note: Memory provider on_session_end() + shutdown_all() are NOT - # called here — run_conversation() is called once per user message in - # multi-turn sessions. Shutting down after every turn would kill the - # provider before the second message. Actual session-end cleanup is - # handled by the CLI (atexit / /reset) and gateway (session expiry / - # _reset_session). + # Memory provider on_session_end()/shutdown_all() are NOT called here: + # run_conversation() runs once per message; CLI/gateway own session-end cleanup. - # Plugin hook: on_session_end - # Fired at the very end of every run_conversation call. - # Plugins can use this for cleanup, flushing buffers, etc. + # Plugin hook: on_session_end — fired at the end of every run_conversation call. try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook _invoke_hook( diff --git a/agent/turn_retry_state.py b/agent/turn_retry_state.py index 49790c6528..42bc2fddad 100644 --- a/agent/turn_retry_state.py +++ b/agent/turn_retry_state.py @@ -1,28 +1,8 @@ -"""Per-attempt recovery bookkeeping for the conversation turn loop. +"""Per-attempt recovery bookkeeping (``TurnRetryState``) for the conversation turn loop. -The inner retry loop in ``run_conversation`` (``while retry_count < -max_retries``) makes several distinct recovery attempts on a single model API -call: a credential-pool 429 retry, a per-provider OAuth refresh (codex, -anthropic, nous, copilot), a long-context compression restart, a length- -continuation restart, and a handful of format-recovery branches (thinking- -signature stripping, multimodal-tool-content stripping, llama.cpp grammar -fallback, image shrink, invalid-encrypted-content, 1M-beta header). - -Each of those branches is guarded by a one-shot boolean so it fires at most -once per attempt. They used to be ~16 bare ``*_attempted`` / ``has_retried_*`` -/ ``restart_with_*`` locals declared inline before the loop and threaded -through its 2,400-line body. ``TurnRetryState`` collapses them into one object -the loop mutates in place (``state.codex_auth_retry_attempted = True``), giving -the recovery bookkeeping a single named, testable home. - -Loop-control variables (``retry_count``, ``max_retries``, -``max_compression_attempts``) intentionally stay as plain locals — they are the -``while`` mechanics, not recovery bookkeeping, and putting them on the object -would add indirection without clarifying anything. - -This module is dependency-free so it can be unit-tested in isolation and -imported by the turn loop without an import cycle. -""" +Each one-shot recovery branch of the inner retry loop is guarded by a flag here so it +fires at most once per attempt. Loop-control (``retry_count``, ``max_retries``) stays +as plain locals. Dependency-free so it imports without a cycle.""" from __future__ import annotations @@ -33,11 +13,8 @@ from dataclasses import dataclass, fields class TurnRetryState: """One-shot recovery guards + restart signals for a single API-call attempt. - A fresh instance is created for each iteration of the outer turn loop - (once per ``api_call_count``). Each guard fires its recovery branch at most - once; the ``restart_with_*`` signals are read by the loop after the attempt - to decide whether to rebuild the request and retry. - """ + A fresh instance is created per ``api_call_count`` iteration; each guard fires at + most once, and ``restart_with_*`` signals tell the loop to rebuild and retry.""" # ── Per-provider OAuth / credential refresh guards ─────────────────── codex_auth_retry_attempted: bool = False @@ -46,12 +23,8 @@ class TurnRetryState: nous_paid_entitlement_refresh_attempted: bool = False copilot_auth_retry_attempted: bool = False # Copilot surfaces a stale/degraded credential as a 400 - # ``model_not_available_for_integrator`` / ``model_not_supported`` instead - # of a clean 401 (e.g. a raw OAuth token seeded when the token exchange - # degraded at startup, routing the request to the restricted - # ``copilot-language-server`` integrator). Guard a single-shot forced - # re-exchange + client rebuild for that case, separate from the 401 guard - # so both can fire within one attempt if needed. + # ``model_not_available_for_integrator`` / ``model_not_supported``, not a 401. + # Single-shot forced re-exchange + rebuild, separate from the 401 guard. copilot_stale_cred_retry_attempted: bool = False vertex_auth_retry_attempted: bool = False @@ -69,22 +42,19 @@ class TurnRetryState: has_retried_429: bool = False # ── Auth-failure provider failover ─────────────────────────────────── - # Set once we've escalated a persistent 401/403 (after the per-provider - # credential-refresh attempt above failed) to the fallback chain, so we - # don't loop on the same auth failover within one attempt. + # Set once a persistent 401/403 has been escalated to the fallback chain, so + # we don't loop on the same auth failover within one attempt. auth_failover_attempted: bool = False # ── Restart signals (read by the outer loop after the attempt) ─────── restart_with_compressed_messages: bool = False restart_with_length_continuation: bool = False - # Set when a content-filter stream stall (e.g. MiniMax "new_sensitive") - # has been escalated to the fallback chain: the partial-stream content - # was rolled back off ``messages`` and the loop should re-issue the API - # call against the newly-activated provider (#32421). + # Set when a content-filter stream stall (e.g. MiniMax "new_sensitive") was + # escalated to the fallback chain: partial content was rolled back off + # ``messages``; re-issue the call against the new provider (#32421). restart_with_rebuilt_messages: bool = False - # A user correction cancelled the in-flight provider request. The outer - # loop must append a role-safe checkpoint + user message, rebuild the API - # payload, and retry the same logical iteration. + # A user correction cancelled the in-flight request: append a role-safe checkpoint + + # user message, rebuild the payload, and retry the same logical iteration. restart_with_redirected_messages: bool = False def __iter__(self):