diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 05b19e8b4e..b18dba9467 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -1,18 +1,9 @@ """The agent conversation loop — extracted from ``run_agent.AIAgent``. -This is the biggest single chunk pulled out of ``run_agent.py``: the -roughly 3,900-line :func:`run_conversation` body that drives one user -turn through the agent (model call, tool dispatch, retries, fallbacks, -compression, post-turn hooks, background memory/skill review nudges). - -The function takes the parent ``AIAgent`` instance as its first -argument (``agent``) and accesses its state via attribute lookup. -``_ra().AIAgent.run_conversation`` is now a thin forwarder. - -Symbols that production code or tests patch on ``run_agent`` directly -(``handle_function_call``, ``_set_interrupt``, ``OpenAI``, ...) are -resolved through :func:`_ra` so those patches keep working. -""" +``run_conversation(agent, ...)`` drives one user turn (model call, tool dispatch, +retries, fallbacks, compression, post-turn hooks). Symbols that callers patch on +``run_agent`` (``handle_function_call``, ``_set_interrupt``, ``OpenAI``) resolve via +``_ra`` so those patches keep working.""" from __future__ import annotations @@ -28,62 +19,63 @@ from typing import Any, Dict, List, Optional from agent.codex_responses_adapter import _summarize_user_message_for_log from agent.conversation_compression import ( - COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE, - COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE, - COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE, - COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE, - PRE_API_COMPRESSION_STATUS_TEMPLATE, - compression_blocked_transiently, - compression_skipped_due_to_lock, - context_compression_timed_out, - conversation_history_after_compression, + conversation_history_after_compression, # noqa: F401 — resolved lazily by turn_overflow/turn_preflight/turn_recovery (tests patch it here) ) -from agent.context_engine import automatic_compaction_status_message from agent.display import KawaiiSpinner from agent.error_classifier import FailoverReason, classify_api_error from agent.fast_mode import begin_turn as begin_fast_mode_turn from agent.message_metadata import append_message from agent.turn_context import ( + build_api_messages, PreflightCompressionTimedOut, _compression_warrants_another_preflight_pass, - _review_fork_first_request_pending, build_turn_context, - compose_user_api_content, reanchor_current_turn_user_idx, ) from agent.turn_retry_state import TurnRetryState +from agent.turn_usage import record_response_usage +from agent.turn_overflow import recover_from_overflow +from agent.turn_empty_response import recover_empty_response +from agent.turn_stop_gates import apply_stop_gates +from agent.turn_tool_validation import validate_tool_calls +from agent.turn_truncation import ( + continue_codex_incomplete, + handle_content_policy_refusal, + recover_from_truncation, +) +from agent.turn_preflight import compress_after_tool_results, run_preflight_compression +from agent.turn_recovery import ( + route_classified_error, + describe_invalid_response, + validate_response_shape, + compute_error_backoff, + interruptible_backoff_sleep, + log_api_error_attempt, + max_retries_exhausted_result, + nonretryable_client_error_result, + recover_after_classification, + recover_before_classification, +) from agent.runtime_cwd import resolve_agent_cwd from agent.message_sanitization import ( close_interrupted_tool_sequence, _repair_tool_call_arguments, coalesce_tool_call_id, - _sanitize_messages_non_ascii, _sanitize_messages_surrogates, _sanitize_structure_non_ascii, _sanitize_structure_surrogates, _sanitize_surrogates, - _sanitize_tools_non_ascii, - _looks_like_image_content_rejection, - _strip_images_from_messages, - _strip_non_ascii, - serialized_messages_bytes, ) -# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py — kept local -# to avoid importing hermes_state at module load time (its module-level -# DEFAULT_DB_PATH = get_hermes_home() / "state.db" breaks tests that -# monkeypatch get_hermes_home to return a str). +# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py; kept local so importing +# hermes_state (module-level DEFAULT_DB_PATH) is not forced at load time. _STALE_MARKER_RE = re.compile(r"^\[[A-Za-z_][A-Za-z0-9_.-]*\]$") from agent.model_metadata import ( MINIMUM_CONTEXT_LENGTH, _estimate_tools_tokens_rough, anchored_context_tokens, - capture_usage_anchor, estimate_messages_tokens_rough, - estimate_request_tokens_rough, - get_context_length_from_provider_error, - is_output_cap_error, - parse_available_output_tokens_from_error, - save_context_length, + estimate_request_tokens_rough, # noqa: F401 — resolved lazily by turn_overflow/turn_preflight/turn_recovery (tests patch it here) + save_context_length, # noqa: F401 — resolved lazily by agent.turn_overflow (tests patch it here) ) from agent.process_bootstrap import _install_safe_stdio from agent.prompt_caching import ( @@ -94,19 +86,13 @@ from agent.prompt_caching import ( ) from agent.provider_projection import splice_provider_projection from agent.retry_utils import ( - adaptive_rate_limit_backoff, - is_zai_coding_overload_error, + adaptive_rate_limit_backoff, # noqa: F401 — resolved lazily by agent.turn_recovery (tests patch it here) jittered_backoff, - zai_coding_overload_retry_ceiling, ) -from agent.repetition_guard import is_repetition_dominated from agent.trajectory import has_incomplete_scratchpad # Bind before the turn starts so a source-tree swap cannot load a skewed # finalizer at turn end. from agent.turn_finalizer import finalize_turn -from agent.usage_pricing import estimate_usage_cost, normalize_usage -from agent import empty_response_guard as _empty_guard -from hermes_constants import PARTIAL_STREAM_STUB_ID from hermes_logging import set_session_context from tools.skill_provenance import set_current_write_origin from utils import base_url_host_matches, env_var_enabled @@ -119,9 +105,8 @@ logger = logging.getLogger(__name__) _INTERRUPT_SCAFFOLD_MARKER = "[This response was interrupted by a user correction.]" -# One-time wrap-up notice appended when a wall-clock run budget crosses its -# 80% threshold (agent.run_budget_seconds / --run-budget). Mirrors the Codex -# CLI budget wrap-up template: stop new work, deliver from current state. +# One-time wrap-up notice appended when a wall-clock run budget crosses 80% +# (agent.run_budget_seconds / --run-budget): stop new work, deliver current state. RUN_BUDGET_WRAPUP_NOTICE = ( "[SYSTEM NOTICE — run time budget nearly exhausted] " "Run time budget nearly exhausted. Stop new discovery/verification work " @@ -138,19 +123,9 @@ def _midturn_request_pressure_tokens( ) -> int: """Token figure the mid-turn pre-API compression guard compares. - When the upcoming request is eligible for native Responses compaction the - transport will checkpoint-prune the payload before sending, so the generic - durable-history estimate overstates the wire by orders of magnitude on a - compacted session and fires a 600s local compression the main request - never needed (#96995). Mirror the turn-prologue preflight (#96644 / - #96155): use the pruned estimate when native eligibility is proven, the - generic message+tools figure otherwise. - - The native estimator adds the system prompt and tool schemas itself and - its converter skips system-role rows, so passing the assembled - ``api_messages`` (which carries the system row) alongside - ``effective_system`` counts the system prompt exactly once. - """ + Returns the pruned native-Responses estimate when native compaction eligibility is + proven (the generic estimate overstates the wire on compacted sessions, #96995), + else the generic message+tools figure. System prompt is counted exactly once.""" try: from agent.codex_responses_adapter import ( estimate_native_responses_preflight_tokens, @@ -178,16 +153,8 @@ def _midturn_request_pressure_tokens( def _review_input_budget_exhausted(agent: Any) -> bool: """True when a detached review fork has replayed its aggregate input budget. - Only forks carrying an explicit ``_review_input_token_budget`` (the - background-review path, #93057) are gated; every other agent returns - False and is unaffected. ``session_input_tokens`` accumulates from - provider usage after each response, so the check fires at the top of the - NEXT tool-loop iteration — the budget-crossing request is admitted and - completes (its tool writes land), then the loop stops before any further - provider call. This caps the review's total replayed input, complementing - the per-request bound provided by detached in-memory compaction and the - ``_REVIEW_MAX_ITERATIONS`` iteration cap. - """ + Only forks with an explicit ``_review_input_token_budget`` are gated (#93057). Fires + at the top of the NEXT iteration, so the budget-crossing request completes first.""" budget = getattr(agent, "_review_input_token_budget", None) if not isinstance(budget, int) or isinstance(budget, bool) or budget <= 0: return False @@ -198,16 +165,9 @@ def _review_input_budget_exhausted(agent: Any) -> bool: def _maybe_inject_run_budget_wrapup(agent: Any, messages: List[Dict[str, Any]]) -> bool: """Inject the one-time wall-clock wrap-up notice when past 80% of budget. - Cache-safe delivery: the notice is appended to the NEWEST ``role:"tool"`` - message (the same channel /steer uses) — no synthetic user message is - inserted mid-loop and no past context is rewritten, so role alternation - and the prompt-cache prefix survive. Latches ``_run_budget_wrapup_injected`` - only on a successful append, so a first iteration without tool results - retries on the next iteration. Returns True when the notice was injected. - - Dormant unless ``agent.run_budget_seconds`` is set AND the turn stamped - ``_run_budget_started_at`` (see ``turn_context.prepare_conversation_turn``). - """ + Appends to the NEWEST ``role:"tool"`` message (cache-safe, like /steer); latches + ``_run_budget_wrapup_injected`` only on a successful append. Returns True when + injected. Dormant unless ``run_budget_seconds`` + ``_run_budget_started_at`` set.""" budget = getattr(agent, "run_budget_seconds", None) if not budget: return False @@ -247,10 +207,8 @@ def _restore_user_after_reference_handoff( ) -> bool: """Re-append this turn's real user ask when compaction left only a handoff. - Returns True when a restore append happened. The caller has already - established that a reference-only handoff would drive the next model - call (#80622); this helper only decides whether a restorable ask exists. - """ + Returns True when a restore append happened; only decides whether a restorable + ask exists (#80622).""" if user_message is None: return False if isinstance(user_message, str): @@ -289,21 +247,15 @@ def _should_skip_model_call_for_reference_handoff( return True -# Fallback final_response for a turn ended by the sole-handoff skip (#80622). -# Deliberately NOT a replay of the last assistant text: finalize_turn's -# non-assistant-tail chokepoint (#43849) appends final_response as a fresh -# assistant row, so recovering the previous turn's prose here would duplicate -# it in the durable transcript AND re-deliver it to the user as if it were -# this turn's answer. A short status is honest and idempotent. +# Fallback final_response for the sole-handoff skip (#80622). Not a replay of the +# last assistant text: finalize_turn appends final_response as a fresh assistant row. _HANDOFF_SKIP_FINAL_RESPONSE = ( "Context was compacted. The previous response is complete — " "awaiting your next message." ) -# Terminal final_response for a turn ended because context compression hit its -# host progress-aware timeout while the request was still oversized (#98722, -# salvaged from #98741). Sending the unchanged request would only bounce off -# the provider's overflow error and re-enter compression in the same turn. +# Terminal final_response when compression hit its host timeout while the request +# was still oversized; resending would only bounce off the overflow error (#98722). _COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( "Context compression timed out without reducing this conversation. " "No messages were dropped. Start a fresh session with /new, or check " @@ -311,9 +263,8 @@ _COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( ) -# Stable prefix of the local interrupt status string emitted when a turn is -# cancelled while waiting on the provider. Surfaces (ACP, TUI) match on this -# to treat it as cancellation metadata rather than assistant prose. +# Stable prefix of the local interrupt status string; surfaces (ACP, TUI) match on +# it to treat the text as cancellation metadata rather than assistant prose. INTERRUPT_WAITING_FOR_MODEL_PREFIX = "Operation interrupted: waiting for model response (" @@ -326,11 +277,8 @@ def _should_rearm_compression_budget( ) -> bool: """Return True after a provider proves a completed compaction worked. - Rough estimates cannot safely rearm the anti-thrash budget: they can dip - below the threshold while the provider-visible prompt remains too large. - Require the completed-compaction latch plus a positive, normalized prompt - count below the threshold from the next successful provider response. - """ + Rough estimates cannot rearm the anti-thrash budget; require the completed- + compaction latch and a positive normalized prompt count below the threshold.""" return bool( compression_attempts and completed_compaction_pending @@ -339,15 +287,9 @@ def _should_rearm_compression_budget( ) -# Modules that indicate a deterministic local processing error when they -# appear in an exception traceback WITHOUT any API-call module. Used by the -# outer-loop error classifier to avoid retrying bugs that will fail -# identically every time (e.g. TypeError from passing list content into a -# regex helper). IMPORTANT: do NOT include "conversation_loop" or -# "run_agent" here — those are the container modules for the try/except -# itself, so every exception passes through them, which would make -# _hit_local always True and misclassify transient API/network errors as -# non-retryable local bugs. (#66267) +# Modules whose presence in a traceback (without any API-call module) marks a +# deterministic local bug not worth retrying. NEVER add "conversation_loop" or +# "run_agent": every exception passes through them; _hit_local would be True (#66267) _LOCAL_PROCESSING_MODULES = frozenset({ "agent_runtime_helpers", "message_content", @@ -358,33 +300,17 @@ _API_CALL_MODULES = frozenset({ "chat_completion_helpers", }) -# Maximum total outer-loop exceptions tolerated within one user turn before -# the loop gives up (#92450). The turn budget is unlimited by default -# (``max_iterations = sys.maxsize``), so the historical "near the limit" -# guard no longer stops a turn whose outer loop keeps raising: permanent -# failures spun at ~64 retries/s, pegged a core, and overwrote the rotated -# agent.log history (days of diagnostic context) within minutes. The inner -# retry/fallback machinery owns transient API recovery and terminates on its -# own; only exceptions that ESCAPE it reach this bound, so the cap can be -# small. Still scaled down by a tiny explicit ``max_iterations`` so a -# manually bounded budget keeps governing. +# Max outer-loop exceptions per user turn before giving up; only exceptions that +# ESCAPE the inner retry/fallback machinery count, so this can be small (#92450). _MAX_OUTER_LOOP_ERRORS = 8 def _is_interpreter_shutdown_error(exc: Exception) -> bool: """Check if *exc* is a fatal interpreter-shutdown failure. - During teardown, ``concurrent.futures`` refuses new work with - ``RuntimeError: cannot schedule new futures after interpreter shutdown`` - (or the shorter ``... after shutdown`` variant from a plain - ThreadPoolExecutor). Both are documented in #58720. - - Delegates to the shared predicate in ``tools.interpreter_shutdown`` - (same home as cron delivery and concurrent tool submission) so the - shutdown-race bug class has one text-matching site. Keeps the - RuntimeError type gate from the original (#93269): unlike the raw - predicate, a ValueError carrying similar text must not match here. - """ + Delegates to ``tools.interpreter_shutdown`` (one text-matching site for the + shutdown-race bug class) but keeps the RuntimeError type gate: a ValueError + carrying similar text must not match (#93269).""" if isinstance(exc, RuntimeError): from tools.interpreter_shutdown import interpreter_shutting_down @@ -395,13 +321,8 @@ def _is_interpreter_shutdown_error(exc: Exception) -> bool: def _moa_client_consumes_prepared_request(client: Any) -> bool: """True when ``client`` is the in-process MoA facade. - ``_moa_prepared_request`` is a private handshake with - ``MoAChatCompletions.create``, and only that facade exposes ``prepare()``. - Every other chat-completions object raises TypeError on the unexpected - keyword — including the native OpenAI client that credential rotation, - provider fallback and dead-connection cleanup rebuild from - ``_client_kwargs`` while ``agent.provider`` stays ``"moa"``. - """ + Only ``MoAChatCompletions`` exposes ``prepare()``; other clients raise TypeError on + ``_moa_prepared_request`` even while ``agent.provider`` stays ``"moa"``.""" completions = getattr(getattr(client, "chat", None), "completions", None) return callable(getattr(completions, "prepare", None)) @@ -419,11 +340,8 @@ def _join_truncated_parts(parts: List[str]) -> str: def _moa_reference_metrics_for_hook(agent: Any) -> Any: """Per-advisor metrics for post_api_request, or None off the MoA path. - MoA runs N advisor models before its aggregator and returns only the - aggregator's response, so an observability plugin sees one generation for - the whole fan-out. The advisor spend is already computed per slot (see - ``_RefAccounting``); this only carries it across the hook boundary. - """ + MoA returns only the aggregator response, so a plugin sees one generation for + the whole fan-out; this carries the per-slot advisor spend across the hook boundary.""" client = getattr(agent, "client", None) getter = getattr(client, "last_reference_metrics", None) if not callable(getter): @@ -437,40 +355,12 @@ def _moa_reference_metrics_for_hook(agent: Any) -> Any: def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text: str) -> None: """Append a provider-safe checkpoint and correction to the live turn. - Incomplete provider reasoning blocks are not valid replay items (Anthropic - signs them; Responses reasoning items require their following output). - Preserve only the *visible* response text, demoted to ordinary text, then - add the correction as a real user message. This keeps role alternation - valid and leaves every previously cached message byte-for-byte unchanged. - - INVARIANT — raw chain-of-thought must never be serialized into replayable - message content. Streamed reasoning is display-only state: it may be shown - live, but it does not re-enter the transcript as assistant (or user) text. - An assistant turn whose content inlines its own chain-of-thought reads to - Anthropic's output classifier as reasoning-injection/prefill jailbreak, - and because the poisoned checkpoint is persisted and replayed on every - subsequent call, the session dies permanently with deterministic - "Provider returned an empty response" storms that no retry, nudge, or - empty-recovery branch can escape (July 2026: four sessions bricked this - way; every reasoning-free checkpoint that week was untouched — same - mechanism as the ~/.hermes/prefill.json incident, 20/20 blocked with - assistant-exposed CoT vs 0/20 without). The interrupted reasoning was - incomplete by definition; the model regenerates it on the retried turn. - If a future path needs to preserve interrupted thinking, carry it in a - provider-gated reasoning *field*, never in content. - INVARIANT — the scaffolding is provider-replay text, not transcript text. - ``[This response was interrupted by a user correction.]`` and its - ``Visible response before the interruption:`` header exist so the MODEL - understands its own reply was cut off. They are not prose the user wrote - or the agent said. Persisting them into an assistant row's ``content`` or - ``api_content`` made the model treat the scaffold as *its own previous - reply*, echo it, and self-replicate ghost rows across turns (#81841). - Carry the scaffolded form only in the *user correction's* ``api_content`` - sidecar — never on the placeholder assistant row. When nothing was on - screen the placeholder is marked ``display_kind="hidden"`` (empty - content) so every transcript surface drops it, exactly like - compaction-reference rows. - """ + Keeps only the *visible* text (demoted to plain text) then adds the correction as a + real user message, so role alternation holds and cached messages stay byte-identical. + INVARIANT: raw chain-of-thought never enters replayable content — inlined CoT reads + as a prefill jailbreak and bricks the session with empty-response storms. + INVARIANT: the interruption scaffold is replay text, carried only in the user + correction's ``api_content``; an on-screen-empty placeholder is ``display_kind=hidden``.""" visible = agent._strip_think_blocks( getattr(agent, "_current_streamed_assistant_text", "") or "" ).strip() @@ -487,10 +377,9 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text f"{text}" ) - # The normal live tail is user or tool, so an assistant placeholder - # followed by the correction preserves strict alternation. If a transport - # already committed an assistant item, attribute the checkpoint inside the - # user correction instead of creating assistant→assistant. + # The live tail is normally user or tool, so an assistant placeholder + correction + # keeps strict alternation; if the tail is already assistant, fold the checkpoint + # into the user correction instead of creating assistant→assistant. if messages and messages[-1].get("role") == "assistant": # Transcript shows the user's own words; the provider replays the # scaffolded form so it still sees the interrupted context. @@ -499,26 +388,17 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text {"role": "user", "content": text, "api_content": correction}, ) else: - # Placeholder preserves role alternation only. Scaffold bytes must - # never land here — the API replay path substitutes api_content back - # into content, and a scaffold-as-assistant-reply is what the model - # then echoes (#81841 / incomplete #73146 else branch). + # Placeholder preserves role alternation only. Scaffold bytes must never land + # here: api_content is substituted back into content on replay (#81841). placeholder: Dict[str, Any] = { "role": "assistant", "content": visible or "", } if not visible: placeholder["display_kind"] = "hidden" - # Keep the transcript hidden and empty, but give the historical - # API projection a non-empty neutral assistant turn so the - # pre-call sanitizer (repair_empty_non_final_messages) does not - # re-heal this row on every later call (#88955). display_kind is - # stripped before sanitization, while api_content is projected - # back into content for historical assistant rows. Use the - # canonical neutral interruption placeholder, never - # _INTERRUPT_SCAFFOLD_MARKER: replaying the scaffold as assistant - # text made the model echo it and self-replicate ghost rows - # (#81841). + # Hidden row, but a non-empty neutral api_content so the pre-call + # sanitizer does not re-heal it every call (#88955). Never + # _INTERRUPT_SCAFFOLD_MARKER: as assistant text the model echoes it (#81841) from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER placeholder["api_content"] = _INTERRUPTED_PLACEHOLDER @@ -535,11 +415,8 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text def _is_copilot_provider(agent: Any) -> bool: """Delegate to ``AIAgent._is_copilot_provider`` (single owner of the check). - ``agent.provider`` is not always the normalized ``copilot`` slug — - ``/model`` and profile configs can leave the alias ``github-copilot`` (or - ``github``) in place, and a bare ``provider == "copilot"`` gate silently - skips credential recovery for those spellings. - """ + ``agent.provider`` may hold the aliases ``github-copilot`` / ``github``; a bare + ``provider == "copilot"`` gate would skip credential recovery for them.""" try: return bool(agent._is_copilot_provider()) except Exception: @@ -553,21 +430,9 @@ def _is_copilot_provider(agent: Any) -> bool: def _is_stale_copilot_credential_error(status_code: Optional[int], error_message: str) -> bool: """Detect a Copilot 400 that is really a STALE / DEGRADED credential. - Copilot surfaces a stale or degraded credential as an HTTP 400 rather than a - clean 401. Two body markers indicate this class: - - - ``model_not_available_for_integrator`` — the request reached the - restricted ``copilot-language-server`` integrator (the server's fallback - when it receives a raw OAuth token instead of an exchanged API token), - whose model allowlist omits enterprise-only models. - - ``model_not_supported`` / "the requested model is not supported" — the - cached bearer's Copilot entitlement rotated out from under a long-lived - process. - - Matched narrowly (status 400 AND a specific marker) so a genuinely wrong - model name — a real 400 — never triggers the single-shot re-exchange. The - caller enforces copilot-provider scoping and the single-shot guard. - """ + Matches status 400 AND ``model_not_available_for_integrator`` or + ``model_not_supported`` / "the requested model is not supported", so a wrong model + name never triggers the single-shot re-exchange. Caller enforces scoping/guard.""" lowered = (error_message or "").lower() is_400 = status_code == 400 or "error code: 400" in lowered if not is_400: @@ -580,34 +445,6 @@ def _is_stale_copilot_credential_error(status_code: Optional[int], error_message ) -def _image_error_max_dimension(error: Exception) -> Optional[int]: - """Extract a provider-reported image dimension ceiling, if present.""" - parts = [] - for value in ( - error, - getattr(error, "message", None), - getattr(error, "body", None), - ): - if value: - try: - parts.append(str(value)) - except Exception: - pass - text = " ".join(parts).lower() - if "image" not in text or "dimension" not in text or "max allowed size" not in text: - return None - - match = re.search(r"max allowed size(?:\s+for [^:]+)?:\s*(\d{3,5})\s*pixels?", text) - if not match: - return None - try: - max_dimension = int(match.group(1)) - except ValueError: - return None - if 512 <= max_dimension <= 8000: - return max_dimension - return None - def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str]: """Return a user-facing error when Ollama is loaded with too little context.""" @@ -655,15 +492,10 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str def _maybe_grow_local_window(agent: Any, compressor: Any, request_tokens: int) -> Optional[int]: - """Try growing the managed local model's context window before - compressing. Returns the new window when the ladder granted one, else - None (hold / at native / not a managed local session). + """Try growing the managed local model's context window before compressing. - The window ladder's design order: models launch at their zero-spill - window and grow toward native max as the session needs room; - compression is the move of last resort. Cheap for every non-local - provider: one lowercase compare, no imports. - """ + Returns the new window when the ladder granted one, else None (hold / at native / + not a managed local session). Cheap for non-local providers: one compare.""" provider = (getattr(agent, "provider", "") or "").strip().lower() if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"): return None @@ -688,10 +520,8 @@ def _maybe_grow_local_window(agent: Any, compressor: Any, def _ra(): - """Lazy reference to ``run_agent`` so callers can patch - ``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` / - ``run_agent.OpenAI`` and have those patches reach this code path. - """ + """Lazy ``run_agent`` reference so patches on ``run_agent.handle_function_call`` / + ``run_agent._set_interrupt`` / ``run_agent.OpenAI`` reach this code path.""" import run_agent return run_agent @@ -725,11 +555,8 @@ def _print_nous_entitlement_guidance(agent, capability: str) -> bool: def _system_prompt_for_hooks(api_kwargs: Any, request_messages: Any) -> Any: """System prompt as actually sent to the provider, for observability hooks. - Providers move it out of ``messages``: Anthropic Messages uses a separate - ``system`` kwarg (str or content-block list), the Responses/Codex API uses - top-level ``instructions``; Chat Completions keeps it as ``messages[0]``. - Returns None when the request carries no system prompt. - """ + Checks ``system`` (Anthropic), ``instructions`` (Responses/Codex), then + ``messages[0]``. Returns None when the request carries no system prompt.""" system_prompt = api_kwargs.get("system") if system_prompt is None: system_prompt = api_kwargs.get("instructions") @@ -764,20 +591,11 @@ def _billing_or_entitlement_message( provider_label = (provider or "").strip() or "the selected provider" model_label = (model or "").strip() or "the selected model" - # Anthropic Claude Pro/Max OAuth subscriptions surface exhaustion of the - # metered "extra usage" bucket as a hard 400 ("You're out of extra - # usage"). Point at the exact settings page and note the cycle-reset - # option, since the generic "add credits with that provider" line doesn't - # apply to a subscription — the user waits for the reset or switches to an - # API key. + # Anthropic Pro/Max OAuth surfaces exhaustion of the "extra usage" bucket as a hard + # 400; point at the settings page and cycle reset — "add credits" does not apply. if (provider or "").strip().lower() == "anthropic": - # ``unverified`` (ClassifiedError.billing_unverified, #82154): the - # "out of extra usage" 400 is ambiguous — Anthropic returns the same - # body when its server-side content filter rejects part of the request - # on a subscription OAuth token, so the message reliably misdirects - # diagnosis toward buying quota. Hedge the claim and name the other - # cause. A confirmed verdict (e.g. a real 402 or an API-key credit - # depletion) keeps the assertive wording. + # ``unverified`` (#82154): the "out of extra usage" 400 is also returned for a + # server-side content-filter rejection, so hedge and name the other cause. if unverified: lines = [ ( @@ -812,9 +630,8 @@ def _billing_or_entitlement_message( ] return "\n".join(lines) - # Provider-agnostic billing URL derivation (OpenAI, DeepSeek, xAI, Groq, - # OpenRouter, …) so every text surface — CLI, gateway messaging, TUI - # transcript — shows the same actionable link, not just OpenRouter. + # Provider-agnostic billing URL so every text surface (CLI, gateway, TUI) shows the + # same actionable link, not just OpenRouter. try: from agent.billing_links import build_billing_block @@ -861,9 +678,7 @@ def _billing_terminal_label(summary: str, unverified: bool) -> str: """Terminal-failure prefix for a billing-classified error. ``unverified`` (#82154): the Anthropic "out of extra usage" 400 can be a - content-filter rejection, so the terminal line must not assert billing - exhaustion as fact. - """ + content-filter rejection, so the line must not assert exhaustion as fact.""" if unverified: return ( "Provider reported usage/credit exhaustion (unverified — the same " @@ -885,10 +700,8 @@ def _billing_failure_result( ) -> dict: """Structured terminal result for a billing-classified failure. - Single construction point for the returned terminal response so the - label, guidance, structured block, and ambiguity flag stay consistent - across the non-retryable abort and max-retries paths (#82154). - """ + Single construction point so label, guidance, structured block and ambiguity flag + stay consistent across the non-retryable abort and max-retries paths (#82154).""" unverified = bool(getattr(classified, "billing_unverified", False)) if guidance is None: guidance = _billing_or_entitlement_message( @@ -909,9 +722,8 @@ def _billing_failure_result( "failed": True, "error": summary, "failure_reason": classified.reason.value, - # The classifier's own retry verdict — carried so UI surfaces - # (agent/error_surface.py) show Retry only when a re-run can differ, - # instead of re-deriving retryability from a second taxonomy. + # Classifier's own retry verdict so UI (agent/error_surface.py) shows Retry + # only when a re-run can differ, not re-derived from a second taxonomy. "failure_retryable": bool(classified.retryable), # The billing verdict may rest on an ambiguous body (#82154) — carry # that through the structured result, not just the prose. @@ -945,48 +757,13 @@ def _print_billing_or_entitlement_guidance( return True -def _try_refresh_nous_paid_entitlement_credentials(agent) -> bool: - """Refresh Nous runtime credentials after a fresh paid-entitlement check.""" - try: - from hermes_cli.nous_account import get_nous_portal_account_info - - account_info = get_nous_portal_account_info(force_fresh=True) - if account_info.paid_service_access is not True: - return False - return agent._try_refresh_nous_client_credentials( - force=True, - ) - except Exception: - return False - def _restore_or_build_system_prompt(agent, system_message, conversation_history): """Restore the cached system prompt from the session DB or build it fresh. - Mutates ``agent._cached_system_prompt`` and persists a freshly-built - prompt back to the session DB on first build. Extracted from - ``run_conversation`` so the prefix-cache restore path can be tested in - isolation. - - Three-way state distinction for the stored row, surfaced via logs so - silent prefix-cache misses are visible in ``agent.log``: - - * ``missing`` — no session row yet (legitimate first turn). - * ``null`` — row exists, ``system_prompt`` column is NULL. - Legacy session predating system-prompt persistence, or a migration - leftover. Warns when ``conversation_history`` is non-empty. - * ``empty`` — row exists, ``system_prompt`` column is the empty - string. Indicates a previous-turn write that ran but stored - nothing (silent persistence bug). Always warns. - * ``present`` — row exists with a usable prompt → reused verbatim. - - Read or write failures against the session DB log at WARNING (not - DEBUG) so persistent issues (disk full, schema drift, lock contention) - surface without needing verbose mode. This used to be a debug-level - log that silently broke prefix-cache reuse on the gateway path - (which constructs a fresh ``AIAgent`` per turn and depends on this - DB roundtrip). - """ + Mutates ``agent._cached_system_prompt`` and persists a freshly-built prompt on first + build. Row states ``missing``/``null``/``empty``/``present`` are logged and DB + failures log at WARNING so silent prefix-cache misses show in ``agent.log``.""" stored_prompt = None stored_state = "missing" session_row = None @@ -1011,14 +788,9 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) ) if stored_prompt and _stored_prompt_matches_runtime(agent, stored_prompt): - # Bot Chat capability epoch: an eternal bot session must adopt - # user-initiated capability changes (skills/toolsets/MCP/SOUL/roster) - # on the next message, not at /new or compression. The stored prompt - # embeds a fingerprint of the capability surface; a mismatch against - # disk is a deliberate, once-per-change rebuild — the /model - # exception applied to capabilities. Prompts without the stamp - # (every non-Bot-Chat session) never take this branch, and the check - # fails closed to "reuse" so a probe failure can't burn cache. + # Bot Chat capability epoch: the stored prompt embeds a capability fingerprint; + # a mismatch is a deliberate once-per-change rebuild. Unstamped prompts never + # take this branch; probe failures fail closed to "reuse" so cache is kept. _bot_stale = False try: from tools.bot_mode_probe import ( @@ -1036,12 +808,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) pass _bot_stale = stored_prompt_capability_stale(stored_prompt, _home_for_epoch) if not _bot_stale and getattr(agent, "_bot_mode_protocol", True): - # Legacy upgrade: a Bot Chat whose prompt predates the epoch - # mechanism (no stamp, no protocol) gets ONE migration - # rebuild — otherwise pre-existing bots would never learn - # the messaging protocol. Title-gated so ordinary unstamped - # sessions (i.e. all of them) never take this path; the - # rebuilt prompt carries the stamp, so it cannot re-fire. + # Legacy upgrade: a Bot Chat prompt predating the epoch mechanism gets + # ONE title-gated migration rebuild; the stamped result cannot re-fire. _t = str(getattr(agent, "_session_title_hint", "") or "").strip() if not _t and agent._session_db and agent.session_id: try: @@ -1060,10 +828,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) agent.session_id, ) agent._session_title_hint = "Bot Chat" - # The skills index inside the prompt comes from a two-layer cache - # (in-process LRU + disk snapshot) that doesn't watch the skills - # dir; a capability refresh must rebuild THROUGH it or a freshly - # installed skill stays invisible in the new prompt. + # The skills index cache (LRU + disk snapshot) does not watch the skills + # dir; a capability refresh must rebuild THROUGH it or new skills are lost. try: from agent.prompt_builder import clear_skills_system_prompt_cache @@ -1071,11 +837,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception: pass agent._cached_system_prompt = agent._build_system_prompt(system_message) - agent._bot_capability_refreshed = True - # Persist the refreshed prompt so the NEXT turn restores the new - # bytes verbatim — the cache break is once per capability change, - # never per turn. (on_session_start deliberately not re-fired: - # this is a continuation, not a new session.) + # Persist so the NEXT turn restores the new bytes verbatim (cache break is + # once per capability change). on_session_start not re-fired: continuation. if agent._session_db: try: agent._session_db.update_system_prompt( @@ -1092,9 +855,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # Continuing session — reuse the exact system prompt from the # previous turn so the Anthropic cache prefix matches. agent._cached_system_prompt = stored_prompt - # Same contract for tools[]: a fresh AIAgent for an existing session - # (gateway agent-cache eviction) re-probed every check_fn, so pin the - # array back to the order this session already sent (tools freeze). + # Same contract for tools[]: pin the array to the order this session already + # sent (tools freeze) instead of re-probing every check_fn on a fresh AIAgent. try: saved_tools = session_row.get("tool_names") if session_row else None if saved_tools: @@ -1103,23 +865,14 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) restore_agent_tool_prefix(agent, json.loads(saved_tools)) except Exception: logger.debug("tool prefix restore skipped", exc_info=True) - # Prompt-section callbacks are new-session-only. Recover their frozen - # bytes from the persisted full prompt so a later compression rebuild - # keeps them without evaluating plugin state in this resumed process. + # Prompt-section callbacks are new-session-only; recover their frozen bytes + # from the persisted prompt so a compression rebuild keeps them. from agent.system_prompt import restore_plugin_prompt_sections restore_plugin_prompt_sections(agent, stored_prompt) - # Reconstruct the cross-session-stable prefix for the early cache - # breakpoint. The static prefix is not persisted (only the full - # prompt is), so gateway surfaces that build a fresh AIAgent per - # turn would otherwise lose the two-block system layout after the - # first turn — flip-flopping the wire shape mid-conversation and - # silently degrading to the legacy single-breakpoint layout. - # - # ``reconstruct_static_prefix`` gates on ``_use_prompt_caching`` (so - # non-Anthropic routes skip the rebuild), applies the startswith - # safety gate (stored prompt bytes are never rewritten), and - # fails open to the legacy cache layout. + # The static prefix is not persisted; rebuild it for the early cache breakpoint + # or fresh-per-turn gateway agents fall back to the single-breakpoint layout. + # reconstruct_static_prefix gates on _use_prompt_caching, fails open to legacy. from agent.system_prompt import reconstruct_static_prefix reconstruct_static_prefix(agent, system_message=system_message) @@ -1135,10 +888,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) ) if conversation_history and stored_state in ("null", "empty"): - # Continuing session whose stored prompt is unusable. The - # previous turn's write either never happened or wrote an empty - # string — either way every turn now rebuilds and the prefix - # cache misses every time. + # Continuing session with an unusable stored prompt: every turn now rebuilds + # and the prefix cache misses every time. logger.warning( "Stored system prompt for session %s is %s; rebuilding " "from scratch this turn. Prefix cache will miss until " @@ -1151,9 +902,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # prompt) — build from scratch. agent._cached_system_prompt = agent._build_system_prompt(system_message) - # Plugin hook: on_session_start — fired once when a brand-new - # session is created (not on continuation). Plugins can use this - # to initialise session-scoped state (e.g. warm a memory cache). + # Plugin hook: on_session_start — fired once for a brand-new session, not on + # continuation. try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook _invoke_hook( @@ -1165,12 +915,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception as exc: logger.warning("on_session_start hook failed: %s", exc) - # Cold-start credits seed (L3) — fallback for the first-turn path. The TUI/ - # desktop build seeds at session OPEN (see seed_credits_at_session_start in - # tui_gateway), so this call is usually a no-op there (idempotent: skips when - # _credits_state already exists). For the plain CLI / any path that didn't seed - # at build, it primes credits state from /api/oauth/account (or a fixture) on the - # first turn so depletion / usage-band warnings fire. Fail-open inside the helper. + # Cold-start credits seed (L3) fallback for the first-turn path; TUI/desktop seed at + # session open, so this is idempotent (skips when _credits_state exists). Fail-open. try: from agent.credits_tracker import seed_credits_at_session_start @@ -1178,10 +924,8 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception: logger.debug("cold-start credits seed failed (fail-open)", exc_info=True) - # Persist the system prompt snapshot in SQLite. Failure here used - # to log at DEBUG, which silently broke prefix-cache reuse on the - # gateway path (fresh AIAgent per turn → reads from this row every - # subsequent turn). + # Persist the system prompt snapshot; the gateway path (fresh AIAgent per turn) + # reads this row every turn, so a failure here breaks prefix-cache reuse. if agent._session_db: try: agent._session_db.update_system_prompt(agent.session_id, agent._cached_system_prompt) @@ -1203,12 +947,8 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: def line_value(label: str) -> str: """Last matching line wins. - Safe ONLY for fields emitted in the volatile tier at the very END of - the prompt (Model / Provider / Platform). User-supplied project - context (AGENTS.md / CLAUDE.md / .cursorrules) is embedded in the - middle context tier, so a last-match scan lets project prose shadow - any field emitted EARLIER — see ``host_info_value``. - """ + Safe ONLY for fields in the volatile tier at the END of the prompt; embedded + project context could shadow earlier fields — see ``host_info_value``.""" prefix = f"{label}:" value = "" for line in prompt.splitlines(): @@ -1219,19 +959,8 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: def host_info_value(label: str) -> str: """Read a field from the prompt's own host-info block. - The host-info block (``build_environment_hints``) sits in the STABLE - tier, ahead of the embedded project context files. A bare scan of the - whole prompt would therefore match a user's ``AGENTS.md`` that merely - contains a line starting with the same label, comparing runtime state - against project prose. That mismatch never clears, so the check would - reject the stored prompt on EVERY turn — rebuilding the system prompt - each message and destroying the prefix cache for the whole session, - which is far worse than the staleness this function guards against. - - Anchor on the ``User home directory:`` line that immediately precedes - the working-directory line in that block, and take the FIRST such - occurrence, so only Hermes' own emitted block can satisfy the read. - """ + Anchors on the FIRST ``User home directory:`` line so a user's ``AGENTS.md`` row + cannot match; a false mismatch would rebuild the prompt every turn.""" prefix = f"{label}:" lines = prompt.splitlines() for idx, line in enumerate(lines): @@ -1252,19 +981,15 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: if stored_provider and current_provider and stored_provider != current_provider: return False - # Detect cwd drift: if the stored prompt was built in a different working - # directory, reuse would silently inject a stale path into the prefix cache. - # Compare against resolve_agent_cwd() — the SAME resolver used to build the - # prompt — so gateway/TUI sessions that set TERMINAL_CWD are not falsely - # rejected (they would always differ from the launch dir's os.getcwd()). + # cwd drift check. Compare against resolve_agent_cwd() — the SAME resolver used to + # build the prompt — so TERMINAL_CWD sessions are not falsely rejected. stored_cwd = host_info_value("Current working directory") if stored_cwd: if stored_cwd != str(resolve_agent_cwd()): return False - # Detect runtime-surface drift: the stored prompt records which platform it - # was built for (e.g. "desktop" vs "cli"). Reusing a desktop-built prompt on - # a terminal session (or vice versa) would inject the wrong runtime hints. + # Runtime-surface drift: reusing a desktop-built prompt on a terminal session (or + # vice versa) would inject the wrong runtime hints. stored_platform = line_value("Platform") current_platform = str(getattr(agent, "platform", "") or "").strip() if stored_platform and current_platform and stored_platform != current_platform: @@ -1273,12 +998,9 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: return True -# The three _get_continuation_prompt variants below, in named-constant form -# so agent.context_compressor's _is_synthetic_compression_user_turn can -# recognize them by content after a crash/interrupt persists one mid-list — -# these rows carry no durable role beyond driving the retry, and SessionDB -# projection strips the _length_continuation_nudge metadata tag that marks -# them in live memory (see agent/context_compressor.py). +# Named constants for the _get_continuation_prompt variants so +# _is_synthetic_compression_user_turn can recognize them by content after a crash +# persists one; SessionDB projection strips the _length_continuation_nudge tag. _LENGTH_CONTINUATION_NETWORK_STUB = ( "[System: The previous response was cut off by a " "network error mid-stream. Continue exactly where " @@ -1290,9 +1012,8 @@ _LENGTH_CONTINUATION_OUTPUT_LIMIT = ( "length limit. Continue exactly where you left off. Do not " "restart or repeat prior text. Finish the answer directly.]" ) -# The dropped-tools variant interpolates the tool name list right after this -# prefix, so it can't be exact-matched — this stable prefix is what -# _is_synthetic_compression_user_turn checks with str.startswith instead. +# The dropped-tools variant interpolates tool names, so +# _is_synthetic_compression_user_turn matches this prefix with str.startswith. _LENGTH_CONTINUATION_DROPPED_TOOLS_PREFIX = "[System: Your previous tool call " @@ -1318,13 +1039,8 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List return _LENGTH_CONTINUATION_OUTPUT_LIMIT -# Continuation nudge for Codex/Responses turns that came back with only -# internal reasoning (no visible content, no tool calls). When the interim -# assistant message also carries no encrypted reasoning items and no -# replayable message items, _chat_messages_to_responses_input emits nothing -# for it — a bare retry would be byte-identical to the request that just -# failed, so the model (observed: grok-4.20 on xai-oauth) deterministically -# repeats the reasoning-only response until the retry budget is exhausted. +# Nudge for Codex/Responses turns that returned only internal reasoning: a bare retry +# would be byte-identical (nothing replayable emitted), so the model repeats it. _CODEX_INCOMPLETE_NUDGE = ( "[System: Your previous response contained only internal reasoning and " "never produced a visible answer or tool call. Do not keep thinking. " @@ -1333,29 +1049,23 @@ _CODEX_INCOMPLETE_NUDGE = ( ) -# Re-prompt sent after a Codex/Responses turn ends with an acknowledgment-only -# reply (no tool calls, no final answer) — named so -# agent.context_compressor's _is_synthetic_compression_user_turn can -# recognize it by content the same way it recognizes _CODEX_INCOMPLETE_NUDGE. +# Re-prompt after an acknowledgment-only Codex/Responses reply; named so +# _is_synthetic_compression_user_turn can recognize it like _CODEX_INCOMPLETE_NUDGE. _CODEX_ACK_CONTINUATION_NUDGE = ( "[System: Continue now. Execute the required tool calls and only " "send your final answer after completing the task.]" ) -# Re-prompt sent when a provider returns finish_reason="tool_calls" with an -# empty tool_calls array (dropped-tool-call recovery, see the retry loop -# below). Named for the same reason as _CODEX_ACK_CONTINUATION_NUDGE — this -# pair is only stripped from the durable transcript once the turn reaches -# finalization; an interrupt/crash mid-retry can still persist it. +# Re-prompt for finish_reason="tool_calls" with empty tool_calls. Named like +# _CODEX_ACK_CONTINUATION_NUDGE: an interrupt mid-retry can persist it. _DROPPED_TOOLCALL_NUDGE_CONTENT = ( "Your previous turn indicated a tool call but none was " "included. Do not narrate a plan or restate intent — issue " "the actual tool call now to continue the task." ) -# Re-prompt sent when the model returns an empty response after executing tool -# calls (#9400). Named for the same reason as the nudges above — its -# _empty_recovery_synthetic metadata flag doesn't survive SessionDB projection. +# Re-prompt for an empty response after tool calls (#9400). Named because its +# _empty_recovery_synthetic metadata flag does not survive SessionDB projection. _EMPTY_TOOL_RESPONSE_NUDGE = ( "You just executed tool calls but returned an " "empty response. Please process the tool " @@ -1363,37 +1073,21 @@ _EMPTY_TOOL_RESPONSE_NUDGE = ( ) -# Shared recovery hint appended to every content-policy refusal message. Both -# the HTTP-200 refusal path (``finish_reason=content_filter``) and the -# exception path (a provider moderation error classified as -# ``content_policy_blocked``) end with the same actionable next steps, so they -# share one trailer to keep the guidance from drifting between the two sites. +# Shared recovery trailer for both content-policy refusal paths (HTTP-200 +# content_filter and the content_policy_blocked exception) so guidance cannot drift. _CONTENT_POLICY_RECOVERY_HINT = ( "Try rephrasing the request, narrowing the context, or " "adding a fallback provider with `hermes fallback add`." ) -# Memo for the send-path tool-call argument canonicalization inside -# run_conversation(). That pass re-canonicalizes the arguments string of -# EVERY historical tool call on EVERY API-call iteration (quadratic in -# session tool-call count), and the api_messages copies share the exact -# argument string objects with the persisted history, so the same strings -# come through unchanged iteration after iteration. -# -# Soundness: canonicalization is a pure, deterministic function of the -# input string (fixed separators, sort_keys=True), so a value-keyed memo -# is exact — equal inputs always produce the canonical form computed the -# first time. Malformed strings raise out of json.loads BEFORE anything -# is stored, so the repair fallback below is never memoized and reruns on -# every occurrence, exactly as before. Bounded FIFO eviction mirrors the -# _MSG_TOKENS_CACHE idiom in agent/model_metadata.py. +# Memo for send-path tool-call argument canonicalization, which re-runs on every +# historical call each iteration. Sound: canonicalization is pure and deterministic; +# malformed strings raise before being stored, so the repair fallback is never memoized. _CANON_ARGS_CACHE: Dict[str, str] = {} _CANON_ARGS_CACHE_MAX = 4096 -# Count bound alone doesn't bound MEMORY: write_file/patch argument strings -# run 100KB+, so 4096 entries could pin ~800MB in a long-lived gateway -# process. The byte budget keeps the memo effective for the common case -# (args ~0.5-2KB) while bounding the worst case. +# Count bound alone does not bound MEMORY: argument strings can run 100KB+, so a byte +# budget bounds the worst case while keeping the memo effective for ~0.5-2KB args. _CANON_ARGS_CACHE_MAX_BYTES = 32 * 1024 * 1024 _canon_args_cache_bytes = 0 @@ -1401,9 +1095,8 @@ _canon_args_cache_bytes = 0 def _canonicalize_tool_call_arguments(arg_str: str) -> str: """Return the canonical wire form of a tool-call arguments JSON string. - Raises whatever ``json.loads`` raises on malformed input; the caller - falls back to ``_repair_tool_call_arguments``, exactly as before. - """ + Raises whatever ``json.loads`` raises on malformed input; the caller falls back to + ``_repair_tool_call_arguments``.""" global _canon_args_cache_bytes cached = _CANON_ARGS_CACHE.get(arg_str) if cached is not None: @@ -1429,34 +1122,9 @@ def _canonicalize_tool_call_arguments(arg_str: str) -> str: def _clone_message_for_send(msg): """Structural clone of a history message for the per-call API copy. - The send path builds ``api_messages`` from the persisted history and - then rewrites the copies in place (canonicalization/repair of tool-call - arguments, surrogate and non-ASCII sanitization, content strips, cache - decoration). A shallow ``msg.copy()`` only decouples TOP-LEVEL fields: - nested containers — ``tool_calls`` entries and their ``function`` dicts, - multimodal ``content`` part lists, ``reasoning_details`` — remain the - SAME objects the persisted history holds, so any in-place write there - silently rewrites the stored transcript (#80498: an unrepairable - ``write_file`` argument string was replaced with ``{}`` in the persisted - turn, destroying the streamed file content). - - Cloning every container (dict/list) recursively while SHARING immutable - leaves (strings, numbers, None) makes every downstream in-place - transform safe by construction — current and future — at container-count - cost, not string-byte cost: big argument strings and base64 image - payloads are shared, never copied. Measured: ~1-5ms per 2000-message - pathological build (20% multimodal, 30% tool calls) vs ~0.4ms for the - shallow copy; compression keeps real request histories far smaller, and - the build runs once per API call — noise next to the call itself. - copy.deepcopy would be equally correct (CPython deepcopy also shares - immutable str) but ~4x slower again and needs its memo machinery; - history messages are JSON-shaped and acyclic (depth < 10 in practice; - a >~1000-deep pathological nest would hit the recursion limit, exactly - as deepcopy would), so cycle handling isn't needed here. Tuples are - shared as leaves: JSON-derived message content never contains tuples, - so a mutable container smuggled inside one is not a reachable shape on - this path. - """ + Clones every dict/list recursively while sharing immutable leaves, so in-place + send-path rewrites can never reach the persisted transcript (#80498). Cheaper than + copy.deepcopy; messages are JSON-shaped and acyclic, tuples are shared as leaves.""" if isinstance(msg, dict): return { k: _clone_message_for_send(v) if isinstance(v, (dict, list)) else v @@ -1473,14 +1141,8 @@ def _clone_message_for_send(msg): def _canonicalize_api_tool_calls(api_messages) -> None: """Canonicalize tool-call argument JSON on the send-path message copy. - Rewrites each message's ``tool_calls`` in place (copy-on-write for the - tool-call dicts it canonicalizes; the persisted history is untouched). - The pass still traverses every message and tool call each iteration; - the memo above bounds the JSON parse/serialize work to one round-trip - per UNIQUE argument string instead of one per string per iteration — - the quadratic part of the cost. The remaining traversal is pointer - chasing and dict copies, cheap next to a json.loads + json.dumps. - """ + Rewrites ``tool_calls`` in place (copy-on-write for the dicts it touches; persisted + history untouched). The memo bounds parse/serialize to one per UNIQUE string.""" for am in api_messages: tcs = am.get("tool_calls") if not tcs: @@ -1496,18 +1158,9 @@ def _canonicalize_api_tool_calls(api_messages) -> None: ), }} except Exception: - # Copy-on-write here too. The send-path build now hands - # this pass structurally-cloned messages (see - # _clone_message_for_send), but this branch keeps its own - # copy as defense in depth: some callers (tests, future - # call sites) pass shallow copies, and assigning into a - # shared ``tc["function"]`` would rewrite the stored - # turn. On the unrepairable path the repair returns "{}", - # so a write-through here replaced the model's real - # arguments with an empty object in the transcript: a - # stream that died mid ``write_file`` lost the file - # content it had already streamed, with only a WARNING - # to show for it (#80498). + # Copy-on-write as defense in depth: callers may pass shallow + # copies, and writing into a shared tc["function"] rewrote the + # stored turn with "{}" on the unrepairable path (#80498). tc = {**tc, "function": { **tc["function"], "arguments": _repair_tool_call_arguments( @@ -1522,18 +1175,9 @@ def _canonicalize_api_tool_calls(api_messages) -> None: def _invalid_tool_name_error_content(name: str, valid_tool_names) -> str: """Error-result content for a tool call whose name isn't a real tool. - A blank/whitespace-only name is not a typo the model can fuzzy-correct - toward a real tool — it is almost always a weak open model echoing - tool-call XML/JSON it saw in file or tool output (#47967: - / payloads in a file prime - mimo/nemotron-class models to emit empty structured calls), or a model - degrading at very large context (observed with gpt-5.6 past ~350K input). - Dumping the full tool catalog in that case feeds the priming loop more - names to mimic and inflates context 3-4x across retries, so send a terse - error that tells the model in-context tool-call syntax is DATA, not a - call to make. A genuinely-wrong-but-nonempty name (an actual typo) still - gets the catalog so the model can self-correct. - """ + A blank name is a model echoing tool-call syntax seen in data, not a typo (#47967); + dumping the catalog feeds that loop, so send a terse error instead. A nonempty wrong + name still gets the catalog so the model can self-correct.""" if not (name or "").strip(): return ( "Tool call rejected: the tool name was empty. " @@ -1556,12 +1200,8 @@ def _content_policy_blocked_result( ) -> Dict[str, Any]: """Build the terminal turn result for a content-policy block. - A content-policy refusal is deterministic for the unchanged prompt, so the - turn ends here (no retry). Both the HTTP-200 refusal handler and the - exception-path handler return the identical shape — a failed, non-completed - turn carrying the user-facing message and a ``content_policy_blocked:`` - prefixed error — so they funnel through this one builder. - """ + Refusals are deterministic for the unchanged prompt, so no retry; both the HTTP-200 + and exception paths return this shape with a ``content_policy_blocked:`` error.""" return { "final_response": final_response, "messages": messages, @@ -1580,23 +1220,9 @@ def _compression_deferred_result( ) -> Dict[str, Any]: """Build the soft turn result for a transiently-deferred compression. - Two transient shapes funnel here, and BOTH must end as a soft defer - (``compression_deferred``), never as ``compression_exhausted``: the - gateway auto-resets (wipes) the session on exhaustion (#9893/#35809). - - * ``reason="lock"`` — another path (a sibling turn, a background review - fork, a manual ``/compress``) holds this session's compression lock, - so every compression pass this turn no-oped and the request still does - not fit. The lock winner is actively shrinking the same session. - * ``reason="transient_block"`` — the compressor is in a timed transient - guard (summary-failure cooldown / structural backoff, e.g. one just - recorded by the host ceiling timeout, #97488). The no-op says nothing - about compressibility; treating it as exhaustion falsely auto-reset - sessions whose compression was merely cooling down. - - ``failed`` stays False so the gateway persists the user turn (transient - branch) and retry-next-message semantics apply. - """ + Both ``reason="lock"`` and ``reason="transient_block"`` must end as + ``compression_deferred``, never ``compression_exhausted`` — the gateway wipes the + session on exhaustion (#9893/#35809). ``failed`` stays False; the turn persists.""" if reason == "transient_block": block = getattr(agent, "_compression_blocked_transient", None) logger.info( @@ -1678,18 +1304,9 @@ def _provider_overflow_exhausted_result( def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool: """Rewrite a cache-decorated system message in place, keeping its blocks. - ``apply_anthropic_cache_control`` runs once per call block, *before* the - retry loop, and splits the system prompt into ``[static prefix, volatile - tail]`` text blocks carrying the cache_control breakpoints. Assigning a bare - string over that list drops both breakpoints, so the failover retry ships - the whole system prompt uncached and re-bills it in full. - - ``rewrite_prompt_model_identity`` only touches the LAST ``Model:`` / - ``Provider:`` lines, and those live in the volatile tail — so the static - prefix stays byte-identical and its cache entry keeps matching. Returns - False when the shape is not one we can safely patch, so the caller falls - back to the plain-string assignment. - """ + Assigning a bare string over the ``[static prefix, volatile tail]`` block list drops + both cache_control breakpoints. Only the LAST ``Model:``/``Provider:`` lines change. + Returns False when the shape cannot be safely patched.""" content = system_message.get("content") if not isinstance(content, list) or not content: return False @@ -1713,18 +1330,9 @@ def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool def _sync_failover_system_message(agent, api_messages, active_system_prompt): """Refresh the in-flight system message after a provider failover. - ``try_activate_fallback`` rewrites the ``Model:``/``Provider:`` identity - lines on ``agent._cached_system_prompt`` (see - ``rewrite_prompt_model_identity``) so the agent reports the model that is - actually answering. But the current call block's ``api_messages`` were - built from the pre-failover prompt, and the retry loop rebuilds - ``api_kwargs`` from that list each iteration — without this sync the - whole turn (and every gateway turn, since fallback re-activates per - message while the primary is down) ships the stale identity. - - Mutates ``api_messages[0]`` in place and returns the prompt to use as - ``active_system_prompt`` for subsequent call-block rebuilds. - """ + ``try_activate_fallback`` rewrites the identity lines on ``_cached_system_prompt``, + but this call block's ``api_messages`` were built pre-failover and are reused each + retry. Mutates ``api_messages[0]`` in place; returns the new ``active_system_prompt``.""" sp = getattr(agent, "_cached_system_prompt", None) if not isinstance(sp, str) or not sp: return active_system_prompt @@ -1737,18 +1345,24 @@ def _sync_failover_system_message(agent, api_messages, active_system_prompt): return sp +def _arm_fallback_restart(agent, api_messages, active_system_prompt, _retry): + """After ``_try_activate_fallback`` succeeded: sync the system message to the new + provider and arm ``restart_with_rebuilt_messages`` (re-issue against the fallback, + refunding the stalled attempt). Callers also reset ``retry_count`` / + ``compression_attempts`` to 0 and ``break`` the retry loop.""" + active_system_prompt = _sync_failover_system_message( + agent, api_messages, active_system_prompt) + _retry.primary_recovery_attempted = False + _retry.restart_with_rebuilt_messages = True + return active_system_prompt + + def _ensure_cached_system_prompt_static(agent, system_message=None) -> None: - """Rebuild ``_cached_system_prompt_static`` when caching becomes active. + """Rebuild ``_cached_system_prompt_static`` when caching becomes active (#72626). - Sessions restored under a cache-off primary skip the static-prefix rebuild - (gated on ``_use_prompt_caching`` at restore time). A later failover to a - cache-on provider would otherwise redecorate with ``static_system_prefix= - None`` and silently fall back to the legacy system-plus-3 layout (#72626). - - Thin wrapper over :func:`agent.system_prompt.reconstruct_static_prefix`, - which memoizes failed rebuilds so this stays cheap on the retry-loop hot - path (it runs at the top of every attempt). - """ + Sessions restored under a cache-off primary skip the static-prefix rebuild; a later + failover to a cache-on provider would otherwise silently fall back to the legacy + system-plus-3 layout. Wraps ``reconstruct_static_prefix`` (memoizes failures).""" from agent.system_prompt import reconstruct_static_prefix reconstruct_static_prefix( @@ -1760,12 +1374,9 @@ def _peel_moa_guidance( messages: List[Dict[str, Any]], guidance: Any, ) -> List[Dict[str, Any]]: - """Remove MoA reference guidance previously attached by ``_attach_reference_guidance``. + """Remove MoA reference guidance attached by ``_attach_reference_guidance``. - Thin wrapper over :func:`agent.moa_loop.peel_reference_guidance` (kept - adjacent to the attach so the forward/inverse shapes evolve together). - Lazy import mirrors the module's other moa_loop touchpoints. - """ + Kept adjacent to the attach so the forward/inverse shapes evolve together.""" from agent.moa_loop import peel_reference_guidance return peel_reference_guidance(messages, guidance) @@ -1781,17 +1392,9 @@ def _redecorate_prompt_cache_for_provider( ) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]] | tuple[List[Dict[str, Any]], Optional[Dict[str, Any]], List[Dict[str, Any]]]: """Strip and re-apply cache_control for the *current* provider policy. - Decoration runs once per call block before the retry loop for the primary - provider. ``try_activate_fallback`` refreshes ``_use_prompt_caching`` / - ``_use_native_cache_layout`` but the nine failover ``continue`` paths reused - the old ``api_messages`` (#72626). Mirror ``_reapply_reasoning_echo_for_provider`` - by reshaping at the top of each retry attempt. - - The source list is the mutated in-flight request (image shrink / ASCII / - reasoning_details recoveries already applied), never a pristine - pre-decoration snapshot. MoA guidance is peeled and rebased without - decoration; the acting aggregator plans its resolved destination later. - """ + Decoration runs once per call block for the primary provider, but failover + ``continue`` paths reuse ``api_messages`` (#72626), so reshape at the top of each + retry from the mutated in-flight request. MoA guidance is peeled and rebased.""" messages: List[Dict[str, Any]] = [ dict(m) if isinstance(m, dict) else m for m in (api_messages or []) ] @@ -1817,9 +1420,8 @@ def _redecorate_prompt_cache_for_provider( return messages, prepared return messages, prepared, planned_tools - # Direct attribute access matches the call-block decoration site — the - # flags are unconditionally initialized on AIAgent, and a getattr - # default here would mask a real init bug as silent cache-off. + # Direct attribute access, not getattr: the flags are always initialized on + # AIAgent, and a default would mask a real init bug as silent cache-off. if agent._use_prompt_caching: _ensure_cached_system_prompt_static(agent, system_message=system_message) static = getattr(agent, "_cached_system_prompt_static", None) @@ -1867,26 +1469,14 @@ def _apply_context_engine_selection( ) -> List[Dict[str, Any]]: """Run the optional per-turn ``ContextEngine.select_context()`` hook. - Returns the (possibly replaced) request message list. The hook is for - context *selection / routing* (retrieval, topic routing, role switching), - which is distinct from compression and fires every turn independent of - ``should_compress()``. - - Fail-open by design: a missing hook, any exception, or an invalid return - value yields the unmodified ``api_messages``. The result is request-only — - persisted conversation history is never mutated here. - """ + Returns the (possibly replaced) request list. Fail-open: a missing hook, exception, + or invalid return yields ``api_messages`` unchanged; history is never mutated.""" engine = getattr(agent, "context_compressor", None) if engine is None or not hasattr(engine, "select_context"): return api_messages - # Skip the no-op base implementation so non-implementing engines — - # including the built-in ContextCompressor — pay nothing per request: - # no history copies below, no call. ``hasattr`` alone is not enough, - # because the ABC defines a default ``select_context`` that every engine - # inherits. Mirrors the base-method short-circuit in - # ``_notify_context_engine_turn_complete``. Lazy import avoids any import - # cycle with agent.context_engine. + # Skip the no-op base ``select_context`` so non-implementing engines pay nothing; + # ``hasattr`` is not enough: the ABC defines a default. Lazy import avoids a cycle. try: from agent.context_engine import ContextEngine as _CE if getattr(engine.select_context, "__func__", None) is _CE.select_context: @@ -1895,16 +1485,8 @@ def _apply_context_engine_selection( pass session_label = getattr(agent, "session_id", None) or "-" - # Pass shallow copies of the reference-only inputs so an engine that - # mutates them in place cannot alter persisted transcript state. Only - # ``request_messages`` (the per-call request list) is meant to be acted on, - # and it may be replaced wholesale via the return value — never mutated in - # place either. ``conversation_messages`` / ``incoming_message`` are - # read-only context; copying enforces the request-only contract rather than - # merely documenting it. Structural clones, not dict(m): a shallow copy - # would leave nested containers (tool_calls, content parts) aliased to - # the persisted history, so an engine writing into them would rewrite - # the transcript (#80498 aliasing class). + # Structural clones: the engine must not be able to write through nested + # containers into persisted history; only the request list is acted on (#80498). _conv_copy = [_clone_message_for_send(m) for m in conversation_messages] \ if conversation_messages is not None else None _incoming_copy = _clone_message_for_send(incoming_message) if isinstance(incoming_message, dict) else incoming_message @@ -1926,11 +1508,8 @@ def _apply_context_engine_selection( if selected is None: return api_messages - # Require a NON-EMPTY list of dicts. An empty list must fall open to the - # original request: ``all([])`` is ``True``, so without the emptiness check - # a ``[]`` returned by a buggy/failing engine would replace a valid request - # with an empty message list that the downstream sanitizers cannot restore, - # reaching the provider as an invalid request instead of failing open. + # Require a NON-EMPTY list of dicts: ``all([])`` is ``True``, so a ``[]`` from a + # buggy engine would otherwise replace the request instead of failing open. if isinstance(selected, list) and selected and all(isinstance(m, dict) for m in selected): return selected @@ -1952,23 +1531,15 @@ def _notify_context_engine_turn_complete( ) -> None: """Notify the active context engine that a user turn has finished. - Calls the optional ``ContextEngine.on_turn_complete()`` observation hook - once per turn, after the assistant/tool loop has produced the finalized - transcript. The complement to ``select_context()`` (pre-request selection): - this lets an engine ingest / index / summarize the completed turn. - - Fail-open: a missing or no-op hook, or any exception, is swallowed. - ``messages`` is passed as a shallow copy so the engine cannot mutate the - persisted transcript. - """ + Fail-open: a missing/no-op hook or any exception is swallowed. ``messages`` is + passed as a copy so the engine cannot mutate the persisted transcript.""" engine = getattr(agent, "context_compressor", None) hook = getattr(engine, "on_turn_complete", None) if engine is None or not callable(hook): return - # Skip the no-op base implementation so non-implementing engines (incl. - # the built-in compressor) pay nothing per turn. Lazy import avoids any - # import cycle with agent.context_engine. + # Skip the no-op base ``on_turn_complete`` so non-implementing engines pay nothing + # per turn. Lazy import avoids an import cycle with agent.context_engine. try: from agent.context_engine import ContextEngine as _CE if getattr(hook, "__func__", None) is _CE.on_turn_complete: @@ -1978,9 +1549,8 @@ def _notify_context_engine_turn_complete( try: hook( - # Structural clones: on_turn_complete receives the PERSISTED - # history; a shallow dict(m) would let a hook write through - # nested containers into the transcript (#80498 aliasing class). + # Structural clones: dict(m) would let a hook write into nested containers + # of the persisted transcript (#80498). [_clone_message_for_send(m) for m in messages], usage=usage, **meta, @@ -2007,38 +1577,17 @@ def run_conversation( persist_user_platform_id: Optional[str] = None, moa_config: Optional[dict[str, Any]] = None, ) -> Dict[str, Any]: - """ - Run a complete conversation with tool calling until completion. + """Run a complete conversation with tool calling until completion. Args: - user_message (str): The user's message/question - system_message (str): Custom system message (optional, overrides ephemeral_system_prompt if provided) - conversation_history (List[Dict]): Previous conversation messages (optional) - task_id (str): Unique identifier for this task to isolate VMs between concurrent tasks (optional, auto-generated if not provided) - stream_callback: Optional callback invoked with each text delta during streaming. - Used by the TTS pipeline to start audio generation before the full response. - When None (default), API calls use the standard non-streaming path. - persist_user_message: Optional clean user message to store in - transcripts/history when user_message contains API-only - synthetic prefixes. - persist_user_timestamp: Optional platform event timestamp to store - as metadata on that persisted user message. - persist_user_display_kind: Optional presentation type for a - synthesized user turn (``auto_continue``, ``model_switch``, …). - Display-only: transcript surfaces render the row as a timeline - event instead of a user bubble, while the model still receives - the message unchanged. - persist_user_display_metadata: Optional payload for that event - (e.g. a delegation's task count). - persist_user_platform_id: Optional platform-side message id (e.g. the - Discord/Telegram message id) to store as metadata on that - persisted user message, so restart drain-window recovery can - dedup an interrupted turn against the transcript. - or queuing follow-up prefetch work. + stream_callback: per-text-delta callback (TTS); None uses the non-streaming path. + persist_user_message: clean text to store when ``user_message`` carries API-only + synthetic prefixes; ``persist_user_timestamp`` / ``persist_user_platform_id`` + are stored as metadata (platform id lets restart drain recovery dedup). + persist_user_display_kind/metadata: display-only event rendering (``auto_continue``, + ``model_switch``); the model still receives the message unchanged. - Returns: - Dict: Complete conversation result with final response and message history - """ + Returns: dict with the final response and message history.""" if moa_config is None: try: from hermes_cli.moa_config import decode_moa_turn @@ -2052,30 +1601,23 @@ def run_conversation( except Exception: pass - # The gateway caches agents across user turns. Compression state is - # per-turn: carrying a prior in-place boundary forward would make a later - # uncompressed result look like a compacted transcript to gateway writers. + # The gateway caches agents across turns; compression state is per-turn, or a stale + # in-place boundary would make a later uncompressed result look compacted. agent._last_compaction_in_place = False agent._last_compression_attempt_recorded = False agent._last_compression_attempt_in_place = None begin_fast_mode_turn(agent, conversation_history) - # Adopt any ~/.hermes/.env credential/base-url edits made since the last - # turn — a Settings save updates .env but not this worker's client, which - # was built at agent init (#67821). No-op when .env is unchanged. + # Adopt ~/.hermes/.env credential/base-url edits made since the last turn — a + # Settings save updates .env, not this worker's client (#67821). No-op if unchanged. try: agent._try_refresh_env_client_credentials() except Exception: logger.debug("per-turn env credential refresh failed", exc_info=True) # ── Per-turn setup (the prologue) ── - # All once-per-turn setup — stdio guarding, retry-counter resets, user - # message sanitization, todo/nudge hydration, system-prompt restore-or- - # build, preflight compression, the ``pre_llm_call`` plugin hook, - # external-memory prefetch, and crash-resilience persistence — lives in - # ``build_turn_context``. It mutates ``agent`` exactly as the inline code - # did and returns the locals the loop below reads back. See - # ``agent/turn_context.py``. + # All once-per-turn setup lives in ``build_turn_context`` (agent/turn_context.py); + # it mutates ``agent`` as the inline code did and returns the locals the loop reads. try: _ctx = build_turn_context( agent, @@ -2101,36 +1643,22 @@ def run_conversation( moa_active=bool(moa_config), ) except PreflightCompressionTimedOut as _preflight_timeout_exc: - # Turn-start fail-closed boundary (#98424): preflight compression hit - # the host's progress-aware timeout while the request was still - # oversized, so no provider call was sent. Convert the typed exception - # into the same typed recovery result the in-loop consumers return - # (salvaged #98741 / PR #99710) instead of letting it escape to the - # surfaces' generic exception handlers — the gateway deliberately - # hides raw exception text from users, which would bury the - # actionable "run /compress and retry" guidance and skip the - # compression_exhausted clean-session recovery contract. + # Preflight compression timed out; no provider call sent (#98424). Return the + # typed recovery result: surfaces hide raw exception text, which would bury the + # actionable guidance and skip the compression_exhausted recovery contract. logger.warning( "Turn-start preflight compression timed out — ending turn with " "typed recovery result: %s", _preflight_timeout_exc, ) - # build_turn_context registered this turn's in-flight tripwire slot - # (note_turn_start) but the early return skips the persist funnel - # that normally clears it — clear it here so the next turn does not - # log a spurious "concurrent turns on one session" warning. The - # inbound user row is intentionally NOT persisted on this path: the - # gateway skips transcript persistence for compression_exhausted - # results to prevent the session-growth loop (#7100), and the - # auto-reset moves future input to a clean session. + # Clear the tripwire slot note_turn_start registered; the early return skips the + # persist funnel that clears it. The user row is deliberately NOT persisted: + # the gateway skips persistence for compression_exhausted results (#7100). from agent.agent_runtime_helpers import note_turn_persisted note_turn_persisted(agent) - # Intentionally NOT _COMPRESSION_TIMEOUT_FINAL_RESPONSE: the boundary's - # exception text carries per-request context (token count, "provider - # call was not sent") that is the actionable guidance this handler - # exists to surface; the in-loop constant describes a different state - # (compression ran and could not reduce). + # Not _COMPRESSION_TIMEOUT_FINAL_RESPONSE — that describes a different state + # (compression ran, could not reduce); the exception text carries the guidance. _final_response = str(_preflight_timeout_exc) return { "final_response": _final_response, @@ -2161,10 +1689,8 @@ def run_conversation( # A configured SessionDB append failure halts only the affected turn. A # cached gateway agent must recover on the next message if storage did. agent._incremental_persistence_failed = False - # Cause of the most recent persistence failure this turn ('locked', - # 'disk', or 'unknown' — see hermes_state.classify_persistence_error). - # Reset alongside the failure flag so a lock-contention diagnosis from a - # previous turn can never leak into this turn's user-facing explanation. + # Cause of the last persistence failure this turn ('locked'/'disk'/'unknown', see + # hermes_state.classify_persistence_error). Reset so a prior diagnosis cannot leak. agent._last_persistence_error_cause = None # Per-turn diagnostic: a failed compression-tip adoption in a previous # turn's flush must not be reported against this turn. @@ -2177,76 +1703,47 @@ def run_conversation( failed = False codex_ack_continuations = 0 length_continue_retries = 0 - # One-shot "continue without thinking" override is turn-scoped: a - # thinking-only truncation arms it right before the continuation restart, - # and build_api_kwargs consumes it on that call. If the turn is - # interrupted/errors between arm and consume, it must not fire on the - # next turn's first request. + # Turn-scoped one-shot: armed by a thinking-only truncation, consumed by + # build_api_kwargs; must not survive an interrupted turn into the next one. agent._ephemeral_reasoning_off = False # Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS. _outer_error_count = 0 truncated_tool_call_retries = 0 truncated_response_parts: List[str] = [] compression_attempts = 0 - # One resolved per-turn compression attempt cap, shared by every site that - # consumes ``compression_attempts``: the pre-API pressure gate, the - # overflow/413 retry handlers, and the post-tool compaction gate. The - # counter is a consecutive unverified/ineffective-attempt backstop: a - # completed compaction rearms it only after a successful provider response - # reports a prompt below the threshold. - # Config-driven via compression.max_attempts (parsed + validated in - # agent_init); default 3 preserves the prior hardcoded behavior for - # objects without the attribute (older pickles / minimal stubs). + # Per-turn compression attempt cap shared by the pre-API gate, 413 handlers and + # post-tool compaction; a consecutive-ineffective-attempt backstop, rearmed only + # after a provider response reports a prompt below threshold. Default 3 if unset. max_compression_attempts = getattr(agent, "max_compression_attempts", 3) _last_preflight_pressure: Optional[int] = None _preflight_compression_blocked = _ctx.preflight_compression_blocked - # A provider overflow is stronger evidence than the rough-estimate - # calibration that normally defers preflight immediately after compaction. - # Keep recovery armed until the rebuilt, complete request is below the - # configured compression threshold. Without this handoff, a compaction - # that drops rows but grows the actual prompt can be sent straight back to - # the provider while awaiting_real_usage_after_compression is true. + # A provider overflow outweighs the rough-estimate calibration that defers preflight + # after compaction: stay armed until the rebuilt request is below the threshold. _provider_overflow_recovery_pending = False - # Armed when a compression host-timeout terminates the turn (#98722, - # salvaged from #98741); finalize below reuses the gateway's existing - # context-recovery contract (error/partial/compression_exhausted). + # Armed when a compression host-timeout ends the turn; finalize reuses the gateway + # context-recovery contract (error/partial/compression_exhausted) (#98722). _compression_timeout_exhausted = False _turn_exit_reason = "unknown" # Diagnostic: why the loop ended - # Last composed answer intentionally held back by a verification gate. If - # that continuation consumes the remaining budget, this is the best - # user-facing result available; it must not be confused with error or - # recovery text produced by unrelated exit paths. + # Last answer held back by a verification gate: if the continuation exhausts the + # budget this is the best user-facing result, distinct from error/recovery text. _pending_verification_response = None - # Tracks whether the pending verification candidate was already streamed - # to the user as interim content. The finalizer uses this to set - # ``_response_was_previewed`` ONLY when the pending candidate is actually - # reused as the final response — not merely because any interim was - # streamed. (#65919 review: response-loss blocker) + # Whether the pending verification candidate was already streamed as interim. + # ``_response_was_previewed`` is set ONLY if it becomes the final response (#65919). _pending_verification_response_previewed = False - # If pre-API compression fires after MoA advisors have produced guidance, - # retain that ephemeral output and rebase it onto the compacted transcript - # on the next loop iteration. This prevents a second advisor fan-out. + # If pre-API compression fires after MoA advisors ran, retain their guidance and + # rebase it onto the compacted transcript next iteration — no second fan-out. pending_moa_prepared_request = None - # Per-turn tally of consecutive successful credential-pool token refreshes, - # keyed by (provider, pool-entry-id). A persistent upstream 401 lets - # ``try_refresh_current()`` "succeed" forever on a single-entry OAuth pool, - # so this tally caps same-entry refreshes and lets the fallback chain take - # over instead of spinning. Reset here so each turn starts fresh. See #26080. + # Per-turn tally of credential-pool refreshes by (provider, pool-entry-id): caps + # same-entry refreshes on a persistent 401 so fallback takes over (#26080). agent._auth_pool_refresh_counts = {} - # Reset the per-turn usage holder forwarded to the context engine's - # on_turn_complete() observation hook. Set after each successful provider - # response (see below); left as None on turns that never reach a response - # (early failure / interrupt) so the hook receives None rather than a - # stale prior turn's usage. + # Per-turn usage forwarded to the context engine's on_turn_complete() hook; left + # None on turns that never reach a response so the hook never sees stale usage. agent._last_turn_usage = None - # Optional opt-in runtime: if api_mode == codex_app_server, hand the - # turn to the codex app-server subprocess (terminal/file ops/patching - # all run inside Codex). Default Hermes path is bypassed entirely. - # See agent/transports/codex_app_server_session.py for the adapter - # and references/codex-app-server-runtime.md for the rationale. + # Opt-in runtime: api_mode == codex_app_server hands the whole turn to the codex + # app-server subprocess (see agent/transports/codex_app_server_session.py). if agent.api_mode == "codex_app_server": return agent._run_codex_app_server_turn( user_message=user_message, @@ -2278,11 +1775,9 @@ def run_conversation( agent._safe_print("\n⚡ Breaking out of tool loop due to interrupt...") break - # Aggregate input budget for detached auxiliary forks (background - # review, #93057): compaction bounds each request; this bounds the - # review as a whole. Fires between iterations — the budget-crossing - # request completed (its tool writes landed), and the tool loop stops - # before the next provider call, mirroring the iteration-budget exit. + # Aggregate input budget for detached auxiliary forks: bounds the whole review, + # not each request. Checked between iterations so the crossing request's writes + # have landed, mirroring the iteration-budget exit (#93057). if _review_input_budget_exhausted(agent): _turn_exit_reason = "review_input_budget_exhausted" if not agent.quiet_mode: @@ -2297,9 +1792,8 @@ def run_conversation( agent._api_call_count = api_call_count agent._touch_activity(f"starting API call #{api_call_count}") - # Grace call: the budget is exhausted but we gave the model one - # more chance. Consume the grace flag so the loop exits after - # this iteration regardless of outcome. + # Grace call: budget exhausted but the model gets one more call. Consume the + # flag so the loop exits after this iteration regardless of outcome. if agent._budget_grace_call: agent._budget_grace_call = False elif not agent.iteration_budget.consume(): @@ -2343,17 +1837,8 @@ def run_conversation( agent._iters_since_skill += 1 # ── Pre-API-call /steer drain ────────────────────────────────── - # If a /steer arrived during the previous API call (while the model - # was thinking), drain it now — before we build api_messages — so - # the model sees the steer text on THIS iteration. Without this, - # steers sent during an API call only land after the NEXT tool batch, - # which may never come if the model returns a final response. - # - # We scan backwards for the last tool-role message in the messages - # list. If found, the steer is appended there. If not (first - # iteration, no tools yet), the steer stays pending for the next - # tool batch — injecting into a user message would break role - # alternation, and there's no tool output to piggyback on. + # Drain a /steer sent during the last API call into the newest tool message so + # it lands THIS iteration. Never put in a user message (breaks alternation). _pre_api_steer = agent._drain_pending_steer() if _pre_api_steer: _injected = False @@ -2394,25 +1879,16 @@ def run_conversation( agent._pending_steer = (existing + "\n" + _pre_api_steer) if existing else _pre_api_steer # ── Wall-clock run-budget wrap-up notice ─────────────────────── - # One-shot: when a run budget (agent.run_budget_seconds / - # --run-budget) is active and 80% of it has elapsed, ask the model - # to wrap up and deliver from the state it already has. Same - # cache-safe channel as /steer (appended to the newest tool - # result); dormant when no budget is set. + # One-shot at 80% of agent.run_budget_seconds: ask the model to wrap up via the + # same cache-safe channel as /steer (newest tool result); off with no budget. if getattr(agent, "run_budget_seconds", None): _maybe_inject_run_budget_wrapup(agent, messages) - # Prepare messages for API call - # If we have an ephemeral system prompt, prepend it to the messages - # Note: Reasoning is embedded in content via tags for trajectory storage. - # However, providers like Moonshot AI require a separate 'reasoning_content' field - # on assistant messages with tool_calls. We handle both cases here. + # Reasoning lives in content via tags for trajectory storage, but some + # providers (Moonshot) also need a 'reasoning_content' field; handle both here. request_logger = getattr(agent, "logger", None) or logging.getLogger(__name__) - # Per-agent validation cursor: skips re-json.loads-ing tool_call - # arguments on history messages already validated in a previous - # iteration. Identity-keyed (strong refs) — compression/undo/repair - # rewriting the list breaks the prefix match and forces a re-scan - # from the divergence point. See sanitize_tool_call_arguments. + # Per-agent validation cursor skips re-parsing tool_call args already validated. + # Identity-keyed; a rewritten list breaks the prefix match and forces a re-scan. _sanitize_cursor = getattr(agent, "_sanitize_args_cursor", None) if _sanitize_cursor is None: _sanitize_cursor = {} @@ -2433,12 +1909,8 @@ def run_conversation( agent.session_id or "-", ) - # Drop legacy ghost rows from the incomplete #73146 else branch BEFORE - # the alternation repair below: a hidden assistant placeholder whose - # content/api_content is the raw interrupt scaffold. Replaying that as - # an assistant message makes the model echo it and self-replicate - # (#81841). Dropping before repair lets repair_message_sequence fix - # any user→user adjacency the filter creates. + # Drop legacy hidden assistant placeholders carrying the raw interrupt scaffold + # before repair: replayed, the model echoes/self-replicates (#81841). messages = [ msg for msg in messages if not ( @@ -2457,20 +1929,10 @@ def run_conversation( ) ] - # Defensive: repair malformed role-alternation before API call. - # Catches cases where the history got wedged into a - # ``tool → user`` or ``user → user`` tail (e.g. after empty- - # response scaffolding was stripped and a new user message - # landed after an orphan tool result). Most providers return - # empty content on malformed sequences, which would otherwise - # retrigger the empty-retry loop indefinitely. - # repair_message_sequence_with_cursor also recomputes the SessionDB - # flush cursor (_last_flushed_db_idx) when repair compacts the list, - # so the turn-end flush doesn't skip the assistant/tool chain (#44837). - from agent.agent_runtime_helpers import ( - fill_empty_non_final_wire_payload, - repair_message_sequence_with_cursor, - ) + # Repair malformed role alternation (tool→user / user→user tails): providers + # return empty content on them and the empty-retry loop spins. The _with_cursor + # variant also recomputes the SessionDB flush cursor after compaction (#44837). + from agent.agent_runtime_helpers import repair_message_sequence_with_cursor repaired_seq = repair_message_sequence_with_cursor(agent, messages) if repaired_seq > 0: request_logger.info( @@ -2479,152 +1941,15 @@ def run_conversation( agent.session_id or "-", ) - api_messages = [] - for idx, msg in enumerate(messages): - - # Structural clone, NOT msg.copy(): every in-place transform - # below (canonicalize/repair, surrogate + non-ASCII sanitizers, - # cache decoration) must be unable to reach the persisted - # history through shared nested containers. See - # _clone_message_for_send. - api_msg = _clone_message_for_send(msg) - - # api_content is the persistence sidecar carrying the exact bytes - # sent to the API for this message when they differ from the clean - # stored content (see compose_user_api_content in turn_context). - # It is bookkeeping, never a provider field — pop it from EVERY - # outgoing copy. - _api_content = api_msg.pop("api_content", None) - - # Display-only timeline metadata. Never a provider field — strip - # from every outgoing copy so strict OpenAI-compatible backends - # don't reject the request after a model switch or resumed typed - # event row enters the live history. - api_msg.pop("display_kind", None) - api_msg.pop("display_metadata", None) - - # Durable row identity stamped by _rows_to_conversation so the - # desktop can address a specific persisted message (reactions). - # Bookkeeping, never a provider field — only the chat-completions - # transport strips underscore keys, so drop it centrally here. - api_msg.pop("_row_id", None) - - # Inject ephemeral context into the current turn's user message. - # Sources: memory manager prefetch + plugin pre_llm_call hooks - # with target="user_message" (the default). Both are - # API-call-time only — the original message in `messages` is - # never mutated beyond the api_content stamp, so nothing leaks - # into the clean transcript content. - if idx == current_turn_user_idx and msg.get("role") == "user": - if isinstance(_api_content, str) and _api_content: - # Stamped by the prologue from the same composition — - # reuse it so the persisted sidecar and the wire cannot - # drift, and so every pass this turn sends identical - # bytes (composed from msg["content"], never from a - # previously-injected copy). - api_msg["content"] = _api_content - else: - # Callers that bypass the prologue stamping: compose live. - _composed = compose_user_api_content( - api_msg.get("content", ""), - _ext_prefetch_cache, - _plugin_user_context, - ) - if _composed is not None: - api_msg["content"] = _composed - elif ( - isinstance(_api_content, str) - and _api_content - and msg.get("role") in ("user", "assistant") - ): - # Historical message: replay the exact bytes sent when it was - # live, so the provider prompt-cache prefix stays byte-stable - # instead of diverging at the injection point and - # re-prefilling everything after it. User rows carry the - # prefetch/plugin injection sidecar; user AND assistant rows - # can carry a sanitize-divergence sidecar (content that - # ``get_messages_as_conversation``'s sanitize_context/strip - # would rewrite on reload — see the capture in - # ``_flush_messages_to_session_db``). - api_msg["content"] = _api_content - - # For ALL assistant messages, pass reasoning back to the API - # This ensures multi-turn reasoning context is preserved - agent._copy_reasoning_content_for_api(msg, api_msg) - - # Remove 'reasoning' field - it's for trajectory storage only - # We've copied it to 'reasoning_content' for the API above - if "reasoning" in api_msg: - api_msg.pop("reasoning") - # Remove finish_reason - not accepted by strict APIs (e.g. Mistral) - if "finish_reason" in api_msg: - api_msg.pop("finish_reason") - # Empty non-final user/assistant turns (#88955 hidden placeholders - # and #96870 stream-death / host-fed empties): once display_kind - # and api_content are stripped, the pre-call sanitizer would - # re-heal the wire copy on every send and flood errors.log. - # Fill the WIRE copy here so the sanitizer has nothing to do. - # Durable history is not mutated. After reasoning copy so a - # thinking-only turn keeps its payload and is not rewritten. - fill_empty_non_final_wire_payload( - api_msg, is_final=(idx == len(messages) - 1) - ) - # _thinking_prefill survives here intentionally: the drop pass below - # needs it. The transport strips all underscore keys before the wire. - # Strip length-continuation marks; not every transport drops underscore keys. - api_msg.pop("_length_continuation_fragment", None) - api_msg.pop("_length_continuation_nudge", None) - # Strip Codex Responses API fields (call_id, response_item_id) for - # strict providers like Mistral, Fireworks, etc. that reject unknown fields. - # Uses new dicts so the internal messages list retains the fields - # for Codex Responses compatibility. - if agent._should_sanitize_tool_calls(): - # In MoA mode, agent.model is the virtual preset name - # (e.g. "closed"), not the actual aggregator model. Use - # the resolved aggregator model so Gemini aggregators - # correctly preserve thought_signature (extra_content). - _sanitize_model = agent.model - if agent.provider == "moa": - if moa_config: - _agg = moa_config.get("aggregator") or {} - if _agg.get("model"): - _sanitize_model = _agg["model"] - if _sanitize_model == agent.model: - # Virtual-provider mode: no moa_config is threaded - # through run_conversation — the facade resolves the - # preset internally. Ask the facade for the resolved - # aggregator slot from the previous create() instead - # (set before any history replay that could carry - # thought_signature). - _moa_client = getattr(agent, "client", None) - _agg_slot = getattr(_moa_client, "last_aggregator_slot", None) - if _agg_slot and _agg_slot.get("model"): - _sanitize_model = _agg_slot["model"] - agent._sanitize_tool_calls_for_strict_api(api_msg, model=_sanitize_model) - # Keep 'reasoning_details' - OpenRouter uses this for multi-turn reasoning context - # The signature field helps maintain reasoning continuity - api_messages.append(api_msg) - - # Build the final system message: cached prompt + ephemeral system prompt. - # Ephemeral additions are API-call-time only (not persisted to session DB). - # External recall context is injected into the user message, not the system - # prompt, so the stable cache prefix remains unchanged. - # - # NOTE: Plugin context from pre_llm_call hooks is injected into the - # user message (see injection block above), NOT the system prompt. - # This is intentional — system prompt modifications break the prompt - # cache prefix. The system prompt is reserved for Hermes internals. - # - # Hermes invariant: the system prompt is built ONCE per session - # (cached on ``_cached_system_prompt``) and replayed verbatim on - # every turn. ``apply_anthropic_cache_control`` may split its stable - # prefix into content blocks on the wire, but the stored string and - # its byte-stability remain unchanged. - effective_system = active_system_prompt or "" - if agent.ephemeral_system_prompt: - effective_system = (effective_system + "\n\n" + agent.ephemeral_system_prompt).strip() - if effective_system: - api_messages = [{"role": "system", "content": effective_system}] + api_messages + api_messages, effective_system = build_api_messages( + agent, + messages, + current_turn_user_idx=current_turn_user_idx, + ext_prefetch_cache=_ext_prefetch_cache, + plugin_user_context=_plugin_user_context, + moa_config=moa_config, + active_system_prompt=active_system_prompt, + ) if moa_config: try: @@ -2635,10 +1960,8 @@ def run_conversation( user_prompt=( original_user_message if isinstance(original_user_message, str) - # Multimodal / decorated content list: extract the - # visible text instead of str()-ing a Python repr of - # the parts (which would leak base64 image payloads - # into the aggregator prompt). + # Multimodal content list: extract visible text rather than + # str()-ing parts, which would leak base64 image payloads. else _flatten_mt(original_user_message) ), api_messages=api_messages, @@ -2666,8 +1989,7 @@ def run_conversation( if isinstance(_base, str): _msg["content"] = _base + "\n\n" + _moa_context elif isinstance(_base, list): - # Multimodal user turn (text + image parts): - # append the MoA context as a trailing text + # Multimodal turn: append MoA context as a trailing text # part instead of silently dropping it. _msg["content"] = [ *_base, @@ -2682,19 +2004,12 @@ def run_conversation( if agent.prefill_messages: sys_offset = 1 if (api_messages and api_messages[0].get("role") == "system") else 0 for idx, pfm in enumerate(agent.prefill_messages): - # Structural clone: the sanitizers below run over - # api_messages in place, and a shallow copy would let them - # write through into agent.prefill_messages' nested - # containers (same aliasing class as the history build). + # Structural clone: the in-place sanitizers below must not write + # through into agent.prefill_messages' nested containers. api_messages.insert(sys_offset + idx, _clone_message_for_send(pfm)) - # Per-turn context selection hook (additive, no-op by default). - # Lets a context engine select/replace which context enters the - # prompt for THIS call only — retrieval, topic routing, role/branch - # switching — distinct from compression and independent of - # should_compress(). Request-only: persisted history is untouched, so - # caching/sanitization below operate on whatever the engine selected. - # Fail-open (see _apply_context_engine_selection). + # Per-turn context selection hook: an engine may select/replace context for THIS + # call only — request-only, fail-open, and independent of should_compress(). _sel_incoming = ( messages[current_turn_user_idx] if 0 <= current_turn_user_idx < len(messages) @@ -2708,18 +2023,12 @@ def run_conversation( logger=request_logger, ) - # Safety net: strip orphaned tool results / add stubs for missing - # results before sending to the API. Runs unconditionally — not - # gated on context_compressor — so orphans from session loading or - # manual message manipulation are always caught. + # Runs unconditionally (not gated on context_compressor) so orphaned tool + # results from session loading or manual message edits are always caught. api_messages = agent._sanitize_api_messages(api_messages) - # One-time repeated-heal escalation notice (#96870): if the sanitizer - # above just crossed the per-session heal threshold, deliver the - # queued notice through the status/warning callback — the normal - # out-of-band delivery channel (gateway status message / CLI print). - # NEVER appended to messages/api_messages: conversation context and - # the cached prompt prefix stay byte-identical. + # One-time repeated-heal notice goes out via the status/warning callback, NEVER + # appended to messages: the cached prompt prefix stays byte-identical (#96870). try: from agent.agent_runtime_helpers import ( consume_pending_sanitizer_heal_notice, @@ -2732,61 +2041,31 @@ def run_conversation( # A notice hiccup must never break the send path. logger.debug("sanitizer heal notice delivery failed", exc_info=True) - # Drop thinking-only assistant turns (reasoning but no visible - # output and no tool_calls) and merge any adjacent user messages - # left behind. Prevents Anthropic 400s ("The final block in an - # assistant message cannot be `thinking`.") and equivalent errors - # from third-party Anthropic-compatible gateways that can't replay - # a thinking-only turn. Runs on the per-call copy only — the - # stored conversation history keeps the reasoning block for the - # UI transcript and session persistence. + # Drop thinking-only assistant turns + merge adjacent users, API copy only: + # Anthropic-style backends 400 on a trailing `thinking` block; history keeps it. api_messages = agent._drop_thinking_only_and_merge_users( api_messages, drop_codex_reasoning_items=agent.api_mode != "codex_responses", ) - # Normalize message whitespace and tool-call JSON for consistent - # prefix matching. Ensures bit-perfect prefixes across turns, - # which enables KV cache reuse on local inference servers - # (llama.cpp, vLLM, Ollama) and improves cache hit rates for - # cloud providers. Operates on api_messages (the API copy) so - # the original conversation history in `messages` is untouched. + # Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns + # (KV-cache reuse on local servers, better cloud cache hits); API copy only. for am in api_messages: if isinstance(am.get("content"), str): am["content"] = am["content"].strip() _canonicalize_api_tool_calls(api_messages) - # Proactively strip any surrogate characters before the API call. - # Models served via Ollama (Kimi K2.5, GLM-5, Qwen) can return - # lone surrogates (U+D800-U+DFFF) that crash json.dumps() inside - # the OpenAI SDK. Sanitizing here prevents the 3-retry cycle. + # Strip lone surrogates (U+D800-U+DFFF) that some Ollama-served models emit; + # they crash json.dumps() inside the OpenAI SDK and trigger the 3-retry cycle. _sanitize_messages_surrogates(api_messages) - # NOTE (empty-content class fix): no send-time pad loop here. The - # single owner for "never send a turn strict wire validation rejects - # as empty" is ``repair_empty_non_final_messages``, which runs inside - # ``_sanitize_api_messages`` above — the unconditional pre-send - # chokepoint shared with the summary path. Its placeholder is - # non-whitespace, so it survives the whitespace-normalization pass - # regardless of ordering (a single-space pad here previously had to - # be sequenced after normalization to survive, forking the concept). + # No send-time pad loop here: ``repair_empty_non_final_messages`` (inside + # ``_sanitize_api_messages``) is the single owner of empty-turn repair, and its + # non-whitespace placeholder survives normalization regardless of ordering. - # Build the request-local cache sections only after every transcript - # mutation. The canonical tool registry stays undecorated. - # - # Runs LAST, after every message mutation above. Marking earlier - # defeats the prefix stability the mutations exist to create: - # ``_apply_cache_marker`` rewrites ``content`` from a plain string - # into a ``[{"type": "text", ...}]`` block, so the marked messages - # no longer match the ``isinstance(content, str)`` test in the - # whitespace-normalization pass and silently keep their raw - # leading/trailing whitespace. A tool result ending in "\n" is - # therefore sent unstripped while it sits in the last-3 window and - # stripped once it rolls out of it — the same message, different - # bytes on consecutive turns, which breaks the prefix match at - # exactly the point the breakpoints were meant to protect. Marking - # last also keeps breakpoints off messages that the orphan sweep or - # the thinking-only drop is about to remove or merge away. + # Build the request-local cache sections LAST, after every transcript mutation; + # the canonical tool registry stays undecorated. Marked ``content`` becomes text + # blocks the whitespace pass skips, so the same row's bytes vary across turns. tools_for_api = agent.tools if agent._use_prompt_caching and agent.provider != "moa": from agent.prompt_caching import ( @@ -2820,12 +2099,9 @@ def run_conversation( api_messages = _initial_cache_plan.messages tools_for_api = _initial_cache_plan.tools - # Build a persistent-MoA request before measuring compression pressure. - # MoA reference output is injected into the aggregator prompt, but it - # is deliberately ephemeral and therefore absent from ``messages``. - # Preparing here makes the pre-API guard measure the exact prompt the - # aggregator will receive; ``create()`` consumes this private prepared - # request later without running the advisors a second time. + # Prepare the persistent-MoA request before measuring compression pressure: the + # ephemeral advisor output is absent from ``messages``; ``create()`` reuses the + # prepared request instead of running the advisors again. _moa_prepared_request = None if agent.provider == "moa": _moa_completions = getattr(getattr(agent.client, "chat", None), "completions", None) @@ -2843,16 +2119,9 @@ def run_conversation( if _moa_prepared_request is not None: api_messages = _moa_prepared_request["messages"] - # One image-stripped message estimate feeds both figures. Was: a - # str(msg) char walk (re-serialized base64 every call) + a second - # messages walk inside estimate_request_tokens_rough. Tools added - # separately (compression needs them: 50+ tools = 20-30K tokens). - # total_chars is a rough (~) proxy — verbose log + hook metric only. - # Charge stale thinking only when the active route actually replays - # it (#84371): on codex_responses the text keys never ship (the - # encrypted item sidecars — charged unconditionally — carry the - # chain), so counting them here re-created the trigger/tail-walk - # disagreement that dead-looped compaction. + # One image-stripped estimate feeds both figures; tools counted separately (50+ + # tools ≈ 20-30K tokens); total_chars is a rough proxy for logs/hooks only. + # Charge stale thinking only when the active route replays it (#84371). from agent.turn_context import _agent_stale_thinking_on_wire if _agent_stale_thinking_on_wire(agent): @@ -2861,32 +2130,21 @@ def run_conversation( approx_tokens = estimate_messages_tokens_rough( api_messages, charge_stale_thinking=False ) - # Route-aware pressure: when the upcoming request is eligible for - # native Responses compaction the transport will checkpoint-prune - # the payload before sending — the generic durable-history figure - # overstates the wire by orders of magnitude on a compacted session - # and fires a 600s local compression the main request never needed - # (#96995, mirroring the turn-prologue preflight #96644/#96155). + # Route-aware: native Responses compaction prunes the wire payload, so the raw + # history figure overstates it and fires needless local compression (#96995). request_pressure_tokens = _midturn_request_pressure_tokens( agent, api_messages, effective_system or "", approx_tokens ) - # Usage-anchored override: when the last provider response's exact - # usage is still valid for the durable transcript, replace the - # whole-history heuristic with anchor + delta-estimate. The anchor's - # prompt_tokens already includes system prompt AND tool schemas as - # the provider counted them, so no tools add-on is needed. Falls - # back to the rough figures above when the anchor is stale/missing - # (first request, post-compaction, usage-less providers). + # Usage-anchored override: real prompt_tokens (incl. system + tool schemas) + + # delta estimate replaces the whole-history heuristic when the anchor is fresh. _anchored_pressure = anchored_context_tokens( messages, getattr(agent, "_usage_anchor", None) ) if _anchored_pressure is not None: request_pressure_tokens = _anchored_pressure total_chars = approx_tokens * 4 - # Stash this request's rough estimate so update_from_response() can - # pair it with the provider's real prompt count — the (rough, real) - # anchor behind should_defer_preflight_to_real_usage()'s projection. - # getattr guard: test doubles built via object.__new__ lack the method. + # Stash the rough estimate so update_from_response() can pair it with the real + # count (should_defer_preflight_to_real_usage). getattr: test doubles lack it. _note_rough = getattr( agent.context_compressor, "note_request_rough_estimate", None ) @@ -2910,23 +2168,9 @@ def run_conversation( pass break - # Pre-API pressure check. The turn-prologue preflight only saw the - # incoming user message; a single turn can then grow by many large - # tool results and leave no output budget before the NEXT call (the - # live 271k/272k Codex failure). The post-response should_compress - # gate at the tool-loop tail uses API-reported last_prompt_tokens, - # which LAGS a just-appended huge tool result — so it misses this - # case. Re-check here against the current request estimate. - # - # Mirror the turn-prologue preflight's guard chain exactly (see - # turn_context.py): (1) defer when the rough estimate is known-noisy - # relative to a recent real provider prompt that fit under threshold - # (schema overhead / post-compaction over-count, #36718); (2) skip - # while a same-session compression-failure cooldown is active; (3) then - # should_compress() — reusing the canonical threshold_tokens (output - # room already reserved by _compute_threshold_tokens) and its summary- - # LLM cooldown + anti-thrash guards (#11529). compression_attempts is a - # hard per-turn backstop shared with the overflow error handlers. + # Pre-API pressure check: tool results grow a turn and last_prompt_tokens lags + # them. Mirror the turn-prologue guard chain: defer on noisy estimate, skip in + # failure cooldown, then should_compress() (#11529). _compressor = agent.context_compressor _preflight_threshold = int( getattr(_compressor, "threshold_tokens", 0) or 0 @@ -2942,16 +2186,11 @@ def run_conversation( _provider_overflow_recovery_pending and not _provider_overflow_preflight ): - # The outer-loop rebuild includes the active system prompt, - # request-only injections, and tool schemas. Once that complete - # request has real output runway again, the provider may be tried. + # The outer-loop rebuild includes system prompt, request-only injections and + # tool schemas; only that full request with output runway may be sent. _provider_overflow_recovery_pending = False - # A previous mid-turn preflight pass deliberately continued the loop so - # API-only context and all sanitization could be rebuilt. Compare that - # fully assembled request with the fully assembled request that caused - # the pass. Raw ``messages`` are not equivalent here: they omit - # api_content/plugin injections, prefills, MoA context, and ephemeral - # system text. + # Compare fully assembled requests, not raw ``messages`` (which omit + # api_content, plugin injections, prefills, MoA context, ephemeral system text). _previous_preflight_pressure = _last_preflight_pressure _last_preflight_pressure = None if ( @@ -2963,10 +2202,8 @@ def run_conversation( _preflight_threshold, ) ): - # Stop proactive retries for this turn without consuming the - # shared overflow-recovery budget. If the provider proves the - # request truly does not fit, its error handler may still compact - # with that stronger signal. + # Stop proactive retries this turn without consuming the shared overflow- + # recovery budget; the provider's error handler may still compact. _preflight_compression_blocked = True logger.warning( "Pre-API compression made insufficient progress: ~%s -> " @@ -2977,283 +2214,47 @@ def run_conversation( _defer_preflight = getattr( _compressor, "should_defer_preflight_to_real_usage", lambda _t: False ) - _compression_cooldown = getattr( - _compressor, "get_active_compression_failure_cooldown", lambda: None - )() - if ( - agent.compression_enabled - and not _review_fork_first_request_pending(agent) - and len(messages) > 1 - and compression_attempts < max_compression_attempts - and ( - not _preflight_compression_blocked - or _provider_overflow_preflight - ) - and ( - not _defer_preflight(request_pressure_tokens) - or _provider_overflow_preflight - ) - and not _compression_cooldown - and _compressor.should_compress(request_pressure_tokens) - ): - # Managed local runtime: try GROWING the context window before - # compressing (the window ladder's design order — compression is - # the move of last resort, once the window is at the model's - # native max or physics/speed say stop). Only fires for a - # llamacpp-flavored provider whose base_url is the server this - # process supervises; every other provider falls straight - # through to compression, exactly as before. - _grown_window = _maybe_grow_local_window( - agent, _compressor, request_pressure_tokens - ) - if _grown_window: - # The server now grants a bigger window: recalibrate the - # compressor to it and skip compression this pass — the - # request that was over the OLD threshold fits the new one. - _compressor.update_model( - agent.model, - _grown_window, - base_url=getattr(agent, "base_url", "") or "", - api_key=getattr(agent, "api_key", "") or "", - provider=getattr(agent, "provider", "") or "", - api_mode=getattr(agent, "api_mode", "") or "", - ) - agent._buffer_status( - f"📈 Context window grown to {_grown_window // 1024}K " - f"(local model; conversation continues uncompressed)" - ) - # This preflight iteration never reached the provider — - # refund the consumed call/budget exactly as the compression - # path below does before ITS continue. - api_call_count -= 1 - agent._api_call_count = api_call_count - agent.iteration_budget.refund() - continue - if _moa_prepared_request is not None: - pending_moa_prepared_request = _moa_prepared_request - compression_attempts += 1 - # Compression is actually running (block cleared / was never - # blocked) — reset the blocked-overflow warning dedup so a future - # blocked-over-threshold turn can warn again. Mirrors the - # turn-context preflight reset (silent-overflow fix #62625). - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. - _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) - if callable(_clear_warn): - _clear_warn() - logger.info( - "Pre-API compression: ~%s request tokens >= %s threshold " - "(context=%s, attempt=%s/%s)", - f"{request_pressure_tokens:,}", - f"{int(getattr(_compressor, 'threshold_tokens', 0) or 0):,}", - f"{int(getattr(_compressor, 'context_length', 0) or 0):,}" - if getattr(_compressor, "context_length", 0) else "unknown", - compression_attempts, - max_compression_attempts, - ) - _pre_api_status = automatic_compaction_status_message( - _compressor, - phase="pre_api", - default_message=PRE_API_COMPRESSION_STATUS_TEMPLATE.format( - tokens=request_pressure_tokens - ), - approx_tokens=request_pressure_tokens, - threshold_tokens=int( - getattr(_compressor, "threshold_tokens", 0) or 0 - ), - context_length=int( - getattr(_compressor, "context_length", 0) or 0 - ), - model=agent.model, - attempt=compression_attempts, - max_attempts=max_compression_attempts, - ) - if _pre_api_status: - agent._emit_status(_pre_api_status) - _last_preflight_pressure = request_pressure_tokens - _pre_api_input = messages - messages, active_system_prompt = agent._compress_context( - messages, - system_message, - approx_tokens=request_pressure_tokens, - task_id=effective_task_id, - ) - if context_compression_timed_out(agent): - # Host progress-aware timeout (#98722, salvaged from #98741): - # this preflight iteration never reached the provider. Refund - # its provisional call/budget exactly like a successful - # pre-API compaction, then stop before the unchanged oversized - # request reaches the provider — its overflow error would only - # invoke compression again on the same transcript with the - # wait budget already spent. - api_call_count -= 1 - agent._api_call_count = api_call_count - agent.iteration_budget.refund() - final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE - failed = True - _compression_timeout_exhausted = True - _turn_exit_reason = "context_compression_timeout" - break - if messages is _pre_api_input and ( - compression_skipped_due_to_lock(agent) - or compression_blocked_transiently(agent) - ): - # #69870 lock-skip / #97488 transient-block: this pass - # no-oped for a TEMPORARY reason (another path holds the - # compression lock, or a timed cooldown/backoff guard is - # active). That is a temporary DEFER, not evidence about - # compressibility — refund the attempt (it must not burn the - # shared overflow-recovery budget toward - # compression_exhausted → gateway auto-reset, #9893/#35809) - # and leave the insufficient-progress blocker unarmed. - # Proceed with the current request: if it truly does not - # fit, the provider's 413/overflow handler returns the soft - # compression_deferred result with that stronger signal. - compression_attempts -= 1 - _last_preflight_pressure = None - if pending_moa_prepared_request is _moa_prepared_request: - pending_moa_prepared_request = None - else: - # Reset retry/empty-response state so the compacted request - # gets a fresh chance instead of inheriting stale recovery - # counters from the pre-compaction history. - agent._empty_content_retries = 0 - agent._thinking_prefill_retries = 0 - agent._last_content_with_tools = None - agent._last_content_tools_all_housekeeping = False - agent._mute_post_response = False - # Re-baseline the flush cursor for the compaction mode that just - # ran. Legacy session-rotation returns None (the child session has - # not seen the compacted transcript, so the next flush writes it - # whole); in-place compaction returns list(messages) because the - # compacted rows are already persisted under the same session id — - # leaving None there would re-append them, doubling the active - # context and retriggering compression. Mirrors the post-response - # and preflight compaction sites; see - # conversation_history_after_compression(). - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - # This preflight iteration never reaches the provider whether - # we skip the turn (handoff guard below) or re-run the loop — - # refund the consumed call/budget in BOTH cases, mirroring the - # ollama_runtime_context_too_small early-exit above. Without - # the refund on the break path, every skipped turn leaked one - # iteration-budget unit for the agent's lifetime and - # finalize_turn logged an api_call_count including a call that - # was never made. - api_call_count -= 1 - agent._api_call_count = api_call_count - agent.iteration_budget.refund() - if _should_skip_model_call_for_reference_handoff( - messages, user_message - ): - # Reference-only handoff must not become the active turn - # after a completed assistant response (#80622). - logger.info( - "Skipping post-compaction model call: reference-only " - "handoff would be the sole active user turn (#80622)" - ) - if not final_response: - final_response = _HANDOFF_SKIP_FINAL_RESPONSE - _turn_exit_reason = "compaction_handoff_not_actionable" - break - continue - elif _provider_overflow_preflight and _compression_cooldown: - # The provider already proved this request cannot fit, while the - # compressor is temporarily unavailable. Do not send the known- - # oversized request again; let the next user turn retry after the - # cooldown instead of turning this into compression exhaustion. - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, - messages, - api_call_count, - reason="transient_block", - ) - elif ( - _provider_overflow_preflight - and compression_attempts >= max_compression_attempts - ): - # Every bounded recovery pass has been consumed and the rebuilt - # request is still over threshold. Fail closed before another - # provider call; llama.cpp can silently truncate an oversized - # retry instead of returning a second actionable overflow error. - return _provider_overflow_exhausted_result( - agent, - messages, - conversation_history, - api_call_count, - request_pressure_tokens, - max_compression_attempts, - ) - elif ( - agent.compression_enabled - and len(messages) > 1 - and compression_attempts < max_compression_attempts - and not _defer_preflight(request_pressure_tokens) - and _compression_cooldown - ): - # Blocked by the summary-LLM cooldown. Surface a deduped warning - # (only when actually over threshold — should_compress_info - # returns a None reason below threshold) so the user isn't left - # with a silently growing context. Mirrors the turn-context - # preflight and the loop-compaction guards (silent-overflow fix - # #62625). - _block_reason = None - try: - _block_reason = _compressor.should_compress_info( - request_pressure_tokens - )[1] - except Exception: - _block_reason = None - if _block_reason: - agent._warn_context_overflow_blocked( - _block_reason, - request_pressure_tokens, - int(getattr(_compressor, "threshold_tokens", 0) or 0), - ) - elif not agent.compression_enabled and len(messages) > 1: - # Uncompressed session guard (#89297): compression is disabled, so - # nothing shrinks a growing session. Reuse the unconditionally - # computed request estimate (zero marginal cost — this site runs - # before every provider request, covering turn-start AND mid-turn - # tool-result growth) and surface a deduped, actionable warning - # when the request exceeds the model context window. The dedup is - # re-armed by the turn-context preflight once the session is back - # under the window (manual /compress works with compression - # disabled), so the guard warns again on a later re-overflow. - # context_compressor always exists (agent_init constructs it even - # when compression is disabled) and its context_length property - # hard-floors at a positive default — no metadata re-resolution - # needed here. - _ctx_len = getattr( - getattr(agent, "context_compressor", None), "context_length", None - ) - if ( - isinstance(_ctx_len, int) - and _ctx_len > 0 - and request_pressure_tokens > _ctx_len - ): - _warn_fn = getattr( - agent, "_warn_uncompressed_context_overflow", None - ) - if callable(_warn_fn): - _warn_fn(request_pressure_tokens, _ctx_len) - - if _provider_overflow_preflight: - # Any other gate that prevented the forced preflight (for example, - # an uncompressible one-message request) must also fail closed. - # Falling through would send a request that the provider already - # proved cannot fit. - return _provider_overflow_exhausted_result( - agent, - messages, - conversation_history, - api_call_count, - request_pressure_tokens, - max_compression_attempts, - ) + _pf = run_preflight_compression( + agent, + compressor=_compressor, + request_pressure_tokens=request_pressure_tokens, + provider_overflow_preflight=_provider_overflow_preflight, + preflight_compression_blocked=_preflight_compression_blocked, + defer_preflight=_defer_preflight, + moa_prepared_request=_moa_prepared_request, + pending_moa_prepared_request=pending_moa_prepared_request, + messages=messages, + system_message=system_message, + user_message=user_message, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + api_call_count=api_call_count, + compression_attempts=compression_attempts, + max_compression_attempts=max_compression_attempts, + effective_task_id=effective_task_id, + final_response=final_response, + failed=failed, + compression_timeout_exhausted=_compression_timeout_exhausted, + turn_exit_reason=_turn_exit_reason, + ) + messages = _pf.messages + active_system_prompt = _pf.active_system_prompt + conversation_history = _pf.conversation_history + api_call_count = _pf.api_call_count + compression_attempts = _pf.compression_attempts + pending_moa_prepared_request = _pf.pending_moa_prepared_request + final_response = _pf.final_response + failed = _pf.failed + _compression_timeout_exhausted = _pf.compression_timeout_exhausted + _turn_exit_reason = _pf.turn_exit_reason + if _pf.last_preflight_pressure is not None: + _last_preflight_pressure = _pf.last_preflight_pressure + if _pf.action == "return": + return _pf.result + if _pf.action == "break": + break + if _pf.action == "continue": + continue # Thinking spinner for quiet mode (animated during API call) thinking_spinner = None @@ -3296,10 +2297,8 @@ def run_conversation( while retry_count < max_retries: # ── Nous Portal rate limit guard ────────────────────── - # If another session already recorded that Nous is rate- - # limited, skip the API call entirely. Each attempt - # (including SDK-level retries) counts against RPH and - # deepens the rate limit hole. + # Skip the call if another session recorded a rate limit: every attempt + # (incl. SDK retries) counts against RPH. if agent.provider == "nous": try: from agent.nous_rate_guard import ( @@ -3317,12 +2316,10 @@ def run_conversation( ) agent._buffer_status(f"⏳ {_nous_msg}") if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) retry_count = 0 compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True break # No fallback available — surface buffered context # so user sees the rate-limit message that led here. @@ -3348,22 +2345,15 @@ def run_conversation( try: agent._reset_stream_delivery_tracking() - # Per-attempt first-chunk timestamp, refreshed each attempt so - # a stale value from a previous API call can never leak into - # the post_api_request hook (set again on stream success). + # Per-attempt first-chunk timestamp so a stale value never leaks into + # post_api_request. agent._last_api_first_chunk_at = None - # api_messages is built once, before this retry loop, while the - # primary provider is active. A mid-conversation fallback can - # switch to a require-side provider (DeepSeek / Kimi / MiMo) that - # rejects assistant turns lacking reasoning_content. Re-apply the - # echo-back pad for the *current* provider here (idempotent no-op - # unless the active provider needs it) so the fallback request - # isn't sent with stale, primary-shaped reasoning fields. + # api_messages was built for the primary; a fallback (DeepSeek / Kimi / + # MiMo) may require reasoning_content. Re-apply the echo-back pad + # (idempotent). agent._reapply_reasoning_echo_for_provider(api_messages) - # Same story for prompt-cache decoration (#72626): try_activate_ - # fallback refreshes the policy flags, but the decorated list - # still carries the primary's breakpoints (or none). Strip and - # re-render for the current provider before building kwargs. + # Same for prompt-cache decoration (#72626): strip the primary's + # breakpoints and re-render for the current provider. api_messages, _moa_prepared_request, tools_for_api = ( _redecorate_prompt_cache_for_provider( agent, @@ -3380,15 +2370,9 @@ def run_conversation( api_messages, tools_for_api=tools_for_api, ) - # Outbound-request surrogate chokepoint (#50959): the messages - # were scrubbed above, but the rest of the request body — - # tool/function descriptions (session_search's ±-heavy text is - # the recorded repro), extra_body, system strings routed via - # kwargs — can still carry invalid code points that providers - # reject with a non-retryable HTTP 400 ("invalid unicode code - # point"). One in-place walk here guarantees the entire - # payload json.dumps()-safe regardless of which leaf produced - # the string. Fast no-op when the payload is clean. + # Surrogate chokepoint (#50959): tool descriptions, extra_body and + # kwargs strings can carry invalid code points (HTTP 400). One walk + # makes the payload json.dumps()-safe. _sanitize_structure_surrogates(api_kwargs) if agent._force_ascii_payload: _sanitize_structure_non_ascii(api_kwargs) @@ -3399,17 +2383,14 @@ def run_conversation( is_github_responses=agent._is_copilot_url(), sanitize_harmony_tokens=agent._is_codex_backend(), ) - # OpenRouter response caching replays identical successful - # responses verbatim, including empty completions. An empty- - # response retry must reach the provider instead of replaying - # the response that triggered the retry. + # OpenRouter caching replays identical responses, even empty ones; an + # empty-response retry must bypass the cache. if agent._empty_content_retries > 0 and agent._is_openrouter_url(): _xh = dict(api_kwargs.get("extra_headers") or {}) _xh["X-OpenRouter-Cache"] = "false" api_kwargs["extra_headers"] = _xh - # Copilot x-initiator: the first API call of a user turn is - # marked "user" so Copilot bills a premium request; tool-loop - # follow-ups keep the default "agent" header (#3040). + # Copilot x-initiator: first call of a user turn is "user" (billed + # premium); tool-loop follow-ups keep the default "agent" (#3040). if getattr(agent, "_is_user_initiated_turn", False) and agent._is_copilot_url(): _xh = dict(api_kwargs.get("extra_headers") or {}) _xh["x-initiator"] = "user" @@ -3449,27 +2430,13 @@ def run_conversation( request_messages = api_kwargs.get("input") if not isinstance(request_messages, list): request_messages = api_messages - # Shallow-copy the outer list so plugins that retain the - # reference for async snapshotting don't observe later - # mutations of api_messages. The inner dicts are not - # mutated by the agent loop, so a shallow copy is - # sufficient; a deepcopy would walk every tool result - # and base64 image on every API call. - # - # The ``request_messages`` and ``conversation_history`` - # kwargs below are pre-existing raw passthroughs - # consumed by the bundled langfuse plugin - # (``plugins/observability/langfuse/__init__.py:_coerce_request_messages``). - # They predate ``request`` and are intentionally NOT - # sanitised — secrets are not expected here because - # ``api_kwargs`` is the same object passed to the - # provider client. New consumers should read the - # sanitised view from ``request["body"]["messages"]``. + # Shallow copy: plugins may retain the list; deepcopy is costly. + # ``request_messages``/``conversation_history`` are raw langfuse + # passthroughs. _request_payload = agent._api_request_payload_for_hook(api_kwargs) - # Anthropic (``system``) and Responses/Codex - # (``instructions``) move the system prompt out of - # messages; pass it explicitly for observability - # plugins (Langfuse). + # Anthropic (``system``) and Responses/Codex (``instructions``) + # move the system prompt out of messages; pass it for + # observability. system_prompt_for_hooks = _system_prompt_for_hooks( api_kwargs, request_messages ) @@ -3507,20 +2474,12 @@ def run_conversation( if env_var_enabled("HERMES_DUMP_REQUESTS"): agent._dump_api_request_debug(api_kwargs, reason="preflight") - # This object is private to the in-process MoA facade. Add it - # only after middleware, hooks, and debug dumps so none of them - # attempts to serialize it as part of the provider payload. + # Private to the in-process MoA facade; add after middleware/hooks/debug + # dumps so none serializes it into the provider payload. if _moa_prepared_request is not None and agent.provider == "moa": - # Re-read the live client instead of trusting the one that - # prepared the request above. Credential rotation, provider - # fallback and dead-connection cleanup all rebuild - # agent.client from _client_kwargs between attempts, and - # pending_moa_prepared_request carries a prepared request - # across exactly that boundary. The rebuilt client is a - # native OpenAI client while provider stays "moa", so this - # private key would reach the SDK as an unexpected keyword - # — a non-retryable TypeError that kills every remaining - # turn on the session. + # Re-read the live client: rotation/fallback/cleanup rebuild + # agent.client between attempts; a native OpenAI client rejects this + # key (TypeError). if _moa_client_consumes_prepared_request(agent.client): api_kwargs["_moa_prepared_request"] = _moa_prepared_request else: @@ -3530,17 +2489,9 @@ def run_conversation( type(agent.client).__name__, ) - # Always prefer the streaming path — even without stream - # consumers. Streaming gives us fine-grained health - # checking (90s stale-stream detection, 60s read timeout) - # that the non-streaming path lacks. Without this, - # subagents and other quiet-mode callers can hang - # indefinitely when the provider keeps the connection - # alive with SSE pings but never delivers a response. - # The streaming path is a no-op for callbacks when no - # consumers are registered, and falls back to non- - # streaming automatically if the provider doesn't - # support it. + # Always prefer streaming even without consumers: it gives stale- + # stream/read-timeout health checks that quiet callers otherwise lack. + # Falls back if unsupported. def _stop_spinner(): nonlocal thinking_spinner if thinking_spinner: @@ -3550,37 +2501,26 @@ def run_conversation( agent.thinking_callback("") _use_streaming = True - # Provider signaled "stream not supported" on a previous - # attempt — switch to non-streaming for the rest of this - # session instead of re-failing every retry. + # Provider signaled "stream not supported": stay non-streaming for the + # session. if getattr(agent, "_disable_streaming", False): _use_streaming = False - # An ACP client communicates via subprocess stdio and returns a - # plain SimpleNamespace — not an iterable stream. Keyed on the - # `acp://` scheme rather than one vendor, so any ACP client is - # excluded. Mirror the ACP exclusion used for Responses API - # upgrade (lines ~1083-1085). + # ACP clients (`acp://` scheme, any vendor) return a plain + # SimpleNamespace, not a stream; mirrors the Responses API exclusion. elif ( agent.provider in {"copilot-acp"} or str(agent.base_url or "").lower().startswith("acp://") or str(agent.base_url or "").lower().startswith("acp+tcp://") ): _use_streaming = False - # MoA streams only when a display/TTS consumer is present to - # receive the deltas. MoAChatCompletions.create() honors - # stream=True (runs the references, then returns the aggregator's - # raw token stream) and is reached here because, for provider - # "moa", _create_request_openai_client returns the MoA facade - # itself. Without consumers (quiet mode, subagents, health-check - # probes) we keep the complete-response path: the facade returns a - # whole response when stream is not requested, preserving the - # prior behavior for those callers. + # MoA streams only with a display/TTS consumer + # (MoAChatCompletions.create() honors stream=True); else complete- + # response path. elif agent.provider == "moa" and not agent._has_stream_consumers(): _use_streaming = False elif not agent._has_stream_consumers(): - # No display/TTS consumer. Still prefer streaming for - # health checking, but skip for Mock clients in tests - # (mocks return SimpleNamespace, not stream iterators). + # No consumer: still stream for health checking, except Mock clients + # in tests (SimpleNamespace, not stream iterators). from unittest.mock import Mock if isinstance(getattr(agent, "client", None), Mock): _use_streaming = False @@ -3661,10 +2601,8 @@ def run_conversation( _model_request_active.clear() _redirect_crossed_response = agent._has_pending_redirect() if _redirect_crossed_response: - # The response and redirect can cross on different threads: - # redirect() observed the request as active just before this - # call returned. Discard that now-stale response and rebuild - # from the correction rather than silently losing it. + # Response and redirect can cross threads: discard the now-stale + # response and rebuild from the correction rather than lose it. if thinking_spinner: thinking_spinner.stop("") thinking_spinner = None @@ -3695,84 +2633,7 @@ def run_conversation( logging.debug(f"API Response received - Model: {resp_model}, Usage: {response.usage if hasattr(response, 'usage') else 'N/A'}") # Validate response shape before proceeding - response_invalid = False - error_details = [] - if agent.api_mode == "codex_responses": - _ct_v = agent._get_transport() - if not _ct_v.validate_response(response): - if response is None: - response_invalid = True - error_details.append("response is None") - else: - # Provider returned a terminal failure (e.g. quota exhaustion). - # Treat as invalid so the fallback chain is triggered instead of - # letting the error bubble up outside the retry/fallback loop. - _codex_resp_status = str(getattr(response, "status", "") or "").strip().lower() - if _codex_resp_status in {"failed", "cancelled"}: - _codex_error_obj = getattr(response, "error", None) - _codex_error_msg = ( - _codex_error_obj.get("message") if isinstance(_codex_error_obj, dict) - else str(_codex_error_obj) if _codex_error_obj - else f"Responses API returned status '{_codex_resp_status}'" - ) - logger.warning( - "Codex response status='%s' (error=%s). Routing to fallback. %s", - _codex_resp_status, _codex_error_msg, - agent._client_log_context(), - ) - response_invalid = True - error_details.append(f"response.status={_codex_resp_status}: {_codex_error_msg}") - else: - # output_text fallback: stream backfill may have failed - # but normalize can still recover from output_text - _out_text = getattr(response, "output_text", None) - _out_text_stripped = _out_text.strip() if isinstance(_out_text, str) else "" - if _out_text_stripped: - logger.debug( - "Codex response.output is empty but output_text is present " - "(%d chars); deferring to normalization.", - len(_out_text_stripped), - ) - else: - _resp_status = getattr(response, "status", None) - _resp_incomplete = getattr(response, "incomplete_details", None) - logger.warning( - "Codex response.output is empty after stream backfill " - "(status=%s, incomplete_details=%s, model=%s). %s", - _resp_status, _resp_incomplete, - getattr(response, "model", None), - f"api_mode={agent.api_mode} provider={agent.provider}", - ) - response_invalid = True - error_details.append("response.output is empty") - elif agent.api_mode == "anthropic_messages": - _tv = agent._get_transport() - if not _tv.validate_response(response): - response_invalid = True - if response is None: - error_details.append("response is None") - else: - error_details.append("response.content invalid (not a non-empty list)") - elif agent.api_mode == "bedrock_converse": - _btv = agent._get_transport() - if not _btv.validate_response(response): - response_invalid = True - if response is None: - error_details.append("response is None") - else: - error_details.append("Bedrock response invalid (no output or choices)") - else: - _ctv = agent._get_transport() - if not _ctv.validate_response(response): - response_invalid = True - if response is None: - error_details.append("response is None") - elif not hasattr(response, 'choices'): - error_details.append("response has no 'choices' attribute") - elif response.choices is None: - error_details.append("response.choices is None") - else: - error_details.append("response.choices is empty") + response_invalid, error_details = validate_response_shape(agent, response) if response_invalid: agent._invoke_api_request_error_hook( @@ -3802,74 +2663,20 @@ def run_conversation( # upstream server error, or malformed response. retry_count += 1 - # Eager fallback: empty/malformed responses are a common - # rate-limit symptom. Switch to fallback immediately - # rather than retrying with extended backoff. + # Eager fallback: empty/malformed responses often mean rate limiting + # — switch now instead of extended backoff. if agent._fallback_index < len(agent._fallback_chain): agent._buffer_status("⚠️ Empty/malformed response — switching to fallback...") if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) retry_count = 0 compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True break - # Check for error field in response (some providers include this) - error_msg = "Unknown" - provider_name = "Unknown" - if response and hasattr(response, 'error') and response.error: - error_msg = str(response.error) - # Try to extract provider from error metadata - if hasattr(response.error, 'metadata') and response.error.metadata: - provider_name = response.error.metadata.get('provider_name', 'Unknown') - elif response and hasattr(response, 'message') and response.message: - error_msg = str(response.message) - - # Try to get provider from model field (OpenRouter often returns actual model used) - if provider_name == "Unknown" and response and hasattr(response, 'model') and response.model: - provider_name = f"model={response.model}" - - # Check for x-openrouter-provider or similar metadata - if provider_name == "Unknown" and response: - # Log all response attributes for debugging - resp_attrs = {k: str(v)[:100] for k, v in vars(response).items() if not k.startswith('_')} - if agent.verbose_logging: - logging.debug(f"Response attributes for invalid response: {resp_attrs}") - - # Extract error code from response for contextual diagnostics - _resp_error_code = None - if response and hasattr(response, 'error') and response.error: - _code_raw = getattr(response.error, 'code', None) - if _code_raw is None and isinstance(response.error, dict): - _code_raw = response.error.get('code') - if _code_raw is not None: - try: - _resp_error_code = int(_code_raw) - except (TypeError, ValueError): - pass - - # Build a human-readable failure hint from the error code - # and response time, instead of always assuming rate limiting. - if _resp_error_code == 524: - _failure_hint = f"upstream provider timed out (Cloudflare 524, {api_duration:.0f}s)" - elif _resp_error_code == 504: - _failure_hint = f"upstream gateway timeout (504, {api_duration:.0f}s)" - elif _resp_error_code == 429: - _failure_hint = "rate limited by upstream provider (429)" - elif _resp_error_code in {500, 502}: - _failure_hint = f"upstream server error ({_resp_error_code}, {api_duration:.0f}s)" - elif _resp_error_code in {503, 529}: - _failure_hint = f"upstream provider overloaded ({_resp_error_code})" - elif _resp_error_code is not None: - _failure_hint = f"upstream error (code {_resp_error_code}, {api_duration:.0f}s)" - elif api_duration < 10: - _failure_hint = f"fast response ({api_duration:.1f}s) — likely rate limited" - elif api_duration > 60: - _failure_hint = f"slow response ({api_duration:.0f}s) — likely upstream timeout" - else: - _failure_hint = f"response time {api_duration:.1f}s" + error_msg, provider_name, _failure_hint = describe_invalid_response( + agent, response, api_duration + ) agent._buffer_vprint(f"⚠️ Invalid API response (attempt {retry_count}/{max_retries}): {', '.join(error_details)}") agent._buffer_vprint(f" 🏢 Provider: {provider_name}") @@ -3882,12 +2689,10 @@ def run_conversation( if agent._has_pending_fallback(): agent._buffer_status(f"⚠️ Max retries ({max_retries}) for invalid responses — trying fallback...") if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) retry_count = 0 compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True break # Terminal — flush buffered retry trace so user sees what happened. agent._flush_status_buffer() @@ -3909,43 +2714,20 @@ def run_conversation( agent._buffer_vprint(f"⏳ Retrying in {wait_time:.1f}s ({_failure_hint})...") logger.warning("Invalid API response (retry %d/%d): %s | Provider: %s", retry_count, max_retries, ', '.join(error_details), provider_name) - # Sleep in small increments to stay responsive to interrupts - sleep_end = time.time() + wait_time - _backoff_touch_counter = 0 - while time.time() < sleep_end: - if agent._interrupt_requested: - # A redirect uses the interrupt machinery to cancel - # only the live request. Aborting the retry here - # with clear_interrupt() would DESTROY the pending - # correction and kill the turn with "Operation - # interrupted" — the exact mid-stream steer loss - # users hit when a redirect lands during provider - # backoff. Rebuild from the correction instead, - # mirroring the InterruptedError handler. - if agent.clear_interrupt(preserve_redirect=True): - _retry.restart_with_redirected_messages = True - break - agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during retry wait, aborting.", force=True) - _interrupt_text = f"Operation interrupted during retry ({_failure_hint}, attempt {retry_count}/{max_retries})." - close_interrupted_tool_sequence(messages, _interrupt_text) - agent._persist_session(messages, conversation_history) - agent.clear_interrupt() - return { - "final_response": _interrupt_text, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "interrupted": True, - } - time.sleep(0.2) - # Touch activity every ~30s so the gateway's inactivity - # monitor knows we're alive during backoff waits. - _backoff_touch_counter += 1 - if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s - agent._touch_activity( - f"retry backoff ({retry_count}/{max_retries}), " - f"{int(sleep_end - time.time())}s remaining" - ) + # A redirect cancels only the live request; the helper preserves the + # pending correction (restart_with_redirected_messages) instead of + # destroying it with clear_interrupt(). + _interrupted = interruptible_backoff_sleep( + agent, wait_time, _retry, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + abort_message="Interrupt detected during retry wait, aborting.", + interrupt_text=f"Operation interrupted during retry ({_failure_hint}, attempt {retry_count}/{max_retries}).", + activity_label=f"retry backoff ({retry_count}/{max_retries})", + ) + if _interrupted is not None: + return _interrupted if _retry.restart_with_redirected_messages: break # rebuild this iteration from the correction continue # Retry the API call @@ -3966,13 +2748,9 @@ def run_conversation( if incomplete_reason is not None: incomplete_reason = str(incomplete_reason).strip().lower() if status == "incomplete" and incomplete_reason in {"max_output_tokens", "length"}: - # Responses API max-output exhaustion is a normal - # Codex incomplete turn. Let the Codex-specific - # continuation path below append the incomplete - # assistant state and retry, instead of routing to - # the generic chat-completions length rollback that - # emits "Response truncated due to output length - # limit" and stops gateway turns. + # Responses API max-output exhaustion is a normal Codex + # incomplete turn: use the Codex continuation path, not the + # length rollback. finish_reason = "incomplete" elif status == "incomplete" and incomplete_reason == "content_filter": finish_reason = "content_filter" @@ -4003,902 +2781,86 @@ def run_conversation( finish_reason = "length" # ── Content-policy refusal (HTTP 200) ────────────────── - # The model — or the provider's safety system — returned a - # *successful* response whose stop/finish reason is a refusal: - # Anthropic ``stop_reason="refusal"`` → ``content_filter``; - # OpenAI / portal ``finish_reason="content_filter"`` or a - # populated ``message.refusal`` (mapped in the chat_completions - # transport); Bedrock ``guardrail_intervened``. The content is - # typically empty, so without this branch the response falls - # through to the empty-response / invalid-response retry loops - # and is mis-surfaced as "rate limited" / "no content after - # retries" — burning paid attempts reproducing a deterministic - # refusal. Surface it clearly and stop. Mirrors the - # exception-based ``content_policy_blocked`` recovery: try a - # configured fallback once, otherwise return the refusal. + # Refusal finish reasons (``content_filter``, ``guardrail_intervened``) + # are deterministic: one fallback try, else return the refusal. if finish_reason == "content_filter": - _refusal_transport = agent._get_transport() - if agent.api_mode == "anthropic_messages": - _refusal_result = _refusal_transport.normalize_response( - response, strip_tool_prefix=agent._is_anthropic_oauth - ) - else: - _refusal_result = _refusal_transport.normalize_response(response) - _refusal_text = (getattr(_refusal_result, "content", None) or "").strip() - # Some refusals carry the explanation only in the reasoning - # channel; fall back to it so the user sees *something*. - if not _refusal_text: - _refusal_text = (agent._extract_reasoning(_refusal_result) or "").strip() - - agent._invoke_api_request_error_hook( - task_id=effective_task_id, + _rv = handle_content_policy_refusal( + agent, + response, + _retry, + thinking_spinner=thinking_spinner, + messages=messages, + api_messages=api_messages, + api_kwargs=api_kwargs, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + api_call_count=api_call_count, + effective_task_id=effective_task_id, turn_id=turn_id, api_request_id=api_request_id, - api_call_count=api_call_count, api_start_time=api_start_time, - api_kwargs=api_kwargs, - error_type="ContentPolicyBlocked", - error_message=_refusal_text or "model declined to respond (content_filter)", - status_code=None, retry_count=retry_count, max_retries=max_retries, - retryable=False, - reason=FailoverReason.content_policy_blocked.value, - ) - - if thinking_spinner: - thinking_spinner.stop("") - thinking_spinner = None - if agent.thinking_callback: - agent.thinking_callback("") - - # Deterministic for the unchanged prompt — never retry. - # Try a configured fallback once (a different model may not - # refuse); otherwise surface the refusal terminally. - if agent._has_pending_fallback(): - agent._buffer_status( - "⚠️ Model declined to respond (safety refusal) — trying fallback..." - ) - if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) - retry_count = 0 - compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True - break - - agent._flush_status_buffer() - _refusal_log = ( - _refusal_text[:500] + "..." - if len(_refusal_text) > 500 - else _refusal_text - ) - logger.warning( - "%sModel declined to respond (finish_reason=content_filter). " - "model=%s provider=%s refusal=%s", - agent.log_prefix, agent.model, agent.provider, - _refusal_log or "(no text)", - ) - agent._emit_status( - "⚠️ The model declined to respond to this request (safety refusal)." - ) - - _refusal_detail = ( - f"Model's explanation: {_refusal_text}" - if _refusal_text - else "The model returned no explanation." - ) - _refusal_response = ( - "⚠️ The model declined to respond to this request " - "(safety refusal — not a Hermes/gateway failure).\n\n" - f"{_refusal_detail}\n\n" - f"{_CONTENT_POLICY_RECOVERY_HINT}" - ) - - agent._cleanup_task_resources(effective_task_id) - agent._persist_session(messages, conversation_history) - return _content_policy_blocked_result( - messages, - api_call_count, - final_response=_refusal_response, - error_detail=_refusal_text or "model declined (content_filter)", ) + thinking_spinner = None + active_system_prompt = _rv.active_system_prompt + if _rv.action == "return": + return _rv.result + retry_count = 0 + compression_attempts = 0 + break if finish_reason == "length": - if getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID: - agent._vprint( - f"{agent.log_prefix}⚠️ Response truncated — stream " - f"ended before completion", - force=True, - ) - else: - agent._vprint( - f"{agent.log_prefix}⚠️ Response truncated " - f"(finish_reason='length') - model hit max output tokens", - force=True, - ) - - # Normalize the truncated response to a single OpenAI-style - # message shape so text-continuation and tool-call retry - # work uniformly across chat_completions, bedrock_converse, - # and anthropic_messages. For Anthropic we use the same - # adapter the agent loop already relies on so the rebuilt - # interim assistant message is byte-identical to what - # would have been appended in the non-truncated path. - _trunc_msg = None - _trunc_transport = agent._get_transport() - if agent.api_mode == "anthropic_messages": - _trunc_result = _trunc_transport.normalize_response( - response, strip_tool_prefix=agent._is_anthropic_oauth - ) - else: - _trunc_result = _trunc_transport.normalize_response(response) - _trunc_msg = _trunc_result - - _trunc_content = getattr(_trunc_msg, "content", None) if _trunc_msg else None - _trunc_has_tool_calls = bool(getattr(_trunc_msg, "tool_calls", None)) if _trunc_msg else False - - # ── Detect thinking-budget exhaustion ────────────── - # When the model spends ALL output tokens on reasoning - # and has none left for the response, continuation - # retries are pointless. Detect this early and give a - # targeted error instead of wasting 3 API calls. - # A response is "thinking exhausted" only when the model - # actually produced reasoning blocks but no visible text after - # them. Models that do not use tags (e.g. GLM-4.7 on - # NVIDIA Build, minimax) may return content=None or an empty - # string for unrelated reasons — treat those as normal - # truncations that deserve continuation retries, not as - # thinking-budget exhaustion. - _has_think_tags = bool( - _trunc_content and re.search( - r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', - _trunc_content, - re.IGNORECASE, - ) + _tv = recover_from_truncation( + agent, + response, + finish_reason, + _retry, + messages=messages, + conversation_history=conversation_history, + api_kwargs=api_kwargs, + api_call_count=api_call_count, + effective_task_id=effective_task_id, + current_turn_user_idx=current_turn_user_idx, + length_continue_retries=length_continue_retries, + truncated_response_parts=truncated_response_parts, + truncated_tool_call_retries=truncated_tool_call_retries, + retry_count=retry_count, + compression_attempts=compression_attempts, ) - _thinking_exhausted = ( - not _trunc_has_tool_calls - and _has_think_tags - and ( - (_trunc_content is not None and not agent._has_content_after_think_block(_trunc_content)) - or _trunc_content is None - ) - ) - - if _thinking_exhausted: - _exhaust_error = ( - "Model used all output tokens on reasoning with none left " - "for the response. Try lowering reasoning effort or " - "increasing max_tokens." - ) - agent._vprint( - f"{agent.log_prefix}💭 Reasoning exhausted the output token budget — " - f"no visible response was produced.", - force=True, - ) - # Return a user-friendly message as the response so - # CLI (response box) and gateway (chat message) both - # display it naturally instead of a suppressed error. - _exhaust_response = ( - "⚠️ **Thinking Budget Exhausted**\n\n" - "The model used all its output tokens on reasoning " - "and had none left for the actual response.\n\n" - "To fix this:\n" - "→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n" - "→ Or switch to a larger/non-reasoning model with `/model`" - ) - agent._cleanup_task_resources(effective_task_id) - agent._persist_session(messages, conversation_history) - return { - "final_response": _exhaust_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": _exhaust_error, - } - - # ── Detect repetition-dominated truncation (#86581) ── - # A model in a degenerate repetition loop can spend its - # ENTIRE output budget echoing one fragment. The - # continuation nudge below would then stitch the - # pathological fragment into the final response — in the - # #86581 incident one turn produced a 60,698-char - # response delivered as 31 Discord messages. Abort with - # a clear user-facing error instead, mirroring the - # _thinking_exhausted guard above. Reasoning blocks are - # stripped first (repeated scratchpad lines are not - # evidence of a degenerate visible response). - _visible_trunc = ( - agent._strip_think_blocks(_trunc_content) - if isinstance(_trunc_content, str) - else _trunc_content - ) - _repetition_dominated = ( - not _trunc_has_tool_calls - and bool(_visible_trunc) - and is_repetition_dominated(_visible_trunc) - ) - if _repetition_dominated: - _rep_error = ( - "Model output entered a repetition loop and was " - "truncated mid-loop; refusing to continue a " - "degenerate response." - ) - agent._vprint( - f"{agent.log_prefix}🔁 Response dominated by " - f"repeated text — stopping instead of " - f"continuing a degenerate response.", - force=True, - ) - _rep_response = ( - "⚠️ **Response Stopped — Repetition Detected**\n\n" - "The model fell into a repetition loop while " - "writing this response, so continuing would only " - "produce more repeated text. The partial response " - "was discarded.\n\n" - "→ Switch to a different model with `/model`\n" - "→ Or resend your message (your conversation " - "history is preserved)" - ) - agent._cleanup_task_resources(effective_task_id) - agent._persist_session(messages, conversation_history) - return { - "final_response": _rep_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": _rep_error, - } - - if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}: - assistant_message = _trunc_msg - # ── Content-filter stream stall → fallback (#32421) ── - # When the provider's output-layer safety filter (e.g. - # MiniMax "output new_sensitive (1027)", Azure - # content_filter) kills the stream mid-delivery, the - # raw error was classified at the swallow point and the - # stub tagged ``_content_filter_terminated``. This - # filter is content-deterministic — continuation - # retries against the SAME primary just re-hit it and - # burn paid attempts (the loop used to give up with - # "Response remained truncated after 3 continuation - # attempts" and never consult the fallback chain). - # Escalate to the configured fallback BEFORE retrying. - _cf_terminated = getattr( - response, "_content_filter_terminated", False - ) - if ( - _cf_terminated - and agent._fallback_index < len(agent._fallback_chain) - ): - agent._vprint( - f"{agent.log_prefix}🛡️ Content filter terminated " - f"stream — activating fallback provider...", - force=True, - ) - agent._emit_status( - "Content filter terminated stream; switching to fallback..." - ) - if agent._try_activate_fallback(): - # Roll the partial content (if any was already - # appended in a prior continuation pass) back to - # the last clean turn so the fallback provider - # gets a coherent continuation point. - if truncated_response_parts: - messages = agent._get_messages_up_to_last_assistant(messages) - # Unmark survivors: their text left the stitched partial. - for _frag in messages: - if isinstance(_frag, dict): - _frag.pop("_length_continuation_fragment", None) - _frag.pop("_length_continuation_nudge", None) - agent._session_messages = messages - length_continue_retries = 0 - truncated_response_parts = [] - retry_count = 0 - compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True - break - # No fallback available — fall through to normal - # continuation (best-effort, may loop). - agent._vprint( - f"{agent.log_prefix}⚠️ No fallback provider " - f"configured — retrying with same provider " - f"(may re-hit filter)...", - force=True, - ) - if assistant_message is not None and not _trunc_has_tool_calls: - length_continue_retries += 1 - # An interim assistant message with NO visible - # content must not be appended — whichever way it - # got that way. An empty partial-stream stub - # (stream dropped before any text was delivered) - # and a response whose whole output budget went to - # reasoning delivered in a separate field (GLM-5.3 - # on ollama-cloud with reasoning_effort=high: - # finish_reason="length", content="", - # completion_tokens == max_tokens) both serialize - # as {"role": "assistant", "content": ""}, and - # strict providers (Moonshot/Kimi via OpenRouter) - # reject empty assistant content with HTTP 400 - # ("message ... with role 'assistant' must not be - # empty") on the very next replay — permanently - # poisoning the session history until the pre-call - # sanitizer "heals" the hole (observed 3+ healings - # per turn). There is no partial text to continue - # from anyway, so only the continuation - # user-message is appended. - _interim_content = getattr(assistant_message, "content", None) - _is_empty_partial_stub = ( - getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID - and not _interim_content - ) - if not _interim_content and not _is_empty_partial_stub: - # Thinking-only truncation: the model spent the - # entire output cap on reasoning and produced no - # visible text. A continuation with thinking - # ON would re-think the whole context from - # scratch (continuations never replay prior - # reasoning) and re-burn the same budget, so - # the next call drops thinking for one request - # — the answer must be written, not re-derived. - agent._ephemeral_reasoning_off = True - if _interim_content: - interim_msg = agent._build_assistant_message(assistant_message, finish_reason) - # Marked so the ceiling exit can drop the fragment trail. - interim_msg["_length_continuation_fragment"] = True - append_message(messages, interim_msg) - truncated_response_parts.append(_interim_content) - - if length_continue_retries < 4: - _is_partial_stream_stub = ( - getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID - ) - _dropped_tools = getattr( - response, "_dropped_tool_names", None - ) - - if _is_partial_stream_stub and _dropped_tools: - _tool_list = ", ".join(_dropped_tools[:3]) - agent._vprint( - f"{agent.log_prefix}↻ Stream interrupted mid " - f"tool-call ({_tool_list}) — requesting " - f"chunked retry " - f"({length_continue_retries}/4)..." - ) - elif _is_partial_stream_stub: - agent._vprint( - f"{agent.log_prefix}↻ Stream interrupted — " - f"requesting continuation " - f"({length_continue_retries}/4)..." - ) - else: - agent._vprint( - f"{agent.log_prefix}↻ Requesting continuation " - f"({length_continue_retries}/4)..." - ) - - _continue_content = _get_continuation_prompt( - _is_partial_stream_stub, _dropped_tools - ) - continue_msg = { - "role": "user", - "content": _continue_content, - "_length_continuation_nudge": True, - } - append_message(messages, continue_msg) - agent._session_messages = messages - _retry.restart_with_length_continuation = True - break - - partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip() - # The pending one-shot reasoning-off override must - # not leak into the next turn when the 4th - # truncation goes straight to the ceiling exit - # without scheduling a continuation call to - # consume it. - agent._ephemeral_reasoning_off = False - if partial_response: - agent._vprint( - f"{agent.log_prefix}⚠️ Response still truncated " - f"after {length_continue_retries} continuation attempts — keeping the " - f"partial response received so far.", - force=True, - ) - _ceiling_final = partial_response - else: - # Every fragment was empty — e.g. a thinking - # model that spent each attempt's whole cap on - # reasoning (GLM-5.3 on ollama-cloud). Return - # an actionable message instead of an invisible - # None result, which only surfaces as a bare - # error card. - agent._vprint( - f"{agent.log_prefix}⚠️ Response still truncated " - f"after {length_continue_retries} continuation attempts — no visible " - f"text was produced.", - force=True, - ) - _ceiling_final = ( - "⚠️ **No visible answer was produced.** The " - "model hit its output-token limit on every " - "continuation attempt — its reasoning " - "consumed the entire budget each time.\n\n" - "To fix this:\n" - "→ Lower reasoning effort: `/reasoning low` " - "or `/reasoning none`\n" - "→ Or raise max_tokens for this model" - ) - # Unanswered continue nudges made every later turn re-truncate. - _turn_start = ( - current_turn_user_idx + 1 - if isinstance(current_turn_user_idx, int) - and current_turn_user_idx >= 0 - else 0 - ) - messages[_turn_start:] = [ - m for m in messages[_turn_start:] - if not ( - isinstance(m, dict) - and ( - m.get("_length_continuation_fragment") - or m.get("_length_continuation_nudge") - ) - ) - ] - if partial_response: - append_message(messages, { - "role": "assistant", - "content": partial_response, - "finish_reason": "length", - }) - agent._session_messages = messages - agent._cleanup_task_resources(effective_task_id) - agent._persist_session(messages, conversation_history) - return { - "final_response": _ceiling_final, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": "Response remained truncated after 4 continuation attempts", - } - - if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}: - assistant_message = _trunc_msg - if assistant_message is not None and _trunc_has_tool_calls: - _is_stub_stall = ( - getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID - ) - if truncated_tool_call_retries < 4: - truncated_tool_call_retries += 1 - if _is_stub_stall: - # The stream broke mid tool-call (network / - # peer-closed connection), not a real output - # cap — say so instead of "max output tokens". - agent._buffer_vprint( - f"⚠️ Stream interrupted mid tool-call — " - f"retrying ({truncated_tool_call_retries}/4)..." - ) - else: - agent._buffer_vprint( - f"⚠️ Truncated tool call detected — " - f"retrying API call " - f"({truncated_tool_call_retries}/4)..." - ) - # Boost max_tokens on each retry so the model has - # more room to complete the tool-call JSON. A - # network stall doesn't need a bigger budget, but - # a genuine output-cap truncation does, and the - # boost is harmless for the stall case. - _tc_boost_base = agent.max_tokens if agent.max_tokens else 4096 - _tc_boost = _tc_boost_base * (2 ** truncated_tool_call_retries) - _tc_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs) - if _tc_requested_cap is not None: - _tc_boost = max(_tc_boost, _tc_requested_cap) - _tc_boost_cap = max(32768, _tc_requested_cap or 0) - agent._ephemeral_max_output_tokens = min(_tc_boost, _tc_boost_cap) - # Don't append the broken response to messages; - # just re-run the same API call from the current - # message state, giving the model another chance. - continue - agent._flush_status_buffer() - if _is_stub_stall: - agent._vprint( - f"{agent.log_prefix}⚠️ Stream kept dropping mid tool-call after 4 retries — the action was not executed.", - force=True, - ) - else: - agent._vprint( - f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.", - force=True, - ) - agent._cleanup_task_resources(effective_task_id) - _final_response = ( - "Stream repeatedly dropped mid tool-call (network); " - "the tool was not executed" - if _is_stub_stall - else "Response truncated due to output length limit" - ) - # Prior successful tool batches (or injected tool - # errors) can leave a tool-result tail; this path - # never reaches finalize_turn (#48879 class). - close_interrupted_tool_sequence(messages, _final_response) - agent._persist_session(messages, conversation_history) - return { - "final_response": _final_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": _final_response, - } - - # If we have prior messages, roll back to last complete state - if len(messages) > 1: - agent._vprint(f"{agent.log_prefix} ⏪ Rolling back to last complete assistant turn") - rolled_back_messages = agent._get_messages_up_to_last_assistant(messages) - - agent._cleanup_task_resources(effective_task_id) - agent._persist_session(messages, conversation_history) - - return { - "final_response": "Response truncated due to output length limit", - "messages": rolled_back_messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": "Response truncated due to output length limit" - } - else: - # First message was truncated - mark as failed - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ First response truncated - cannot recover", force=True) - agent._persist_session(messages, conversation_history) - return { - "final_response": "First response truncated due to output length limit", - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "failed": True, - "error": "First response truncated due to output length limit" - } + messages = _tv.messages + length_continue_retries = _tv.length_continue_retries + truncated_response_parts = _tv.truncated_response_parts + truncated_tool_call_retries = _tv.truncated_tool_call_retries + retry_count = _tv.retry_count + compression_attempts = _tv.compression_attempts + if _tv.action == "return": + return _tv.result + if _tv.action == "break": + break + if _tv.action == "continue": + continue - # Track actual token usage from response for context management - if hasattr(response, 'usage') and response.usage: - canonical_usage = normalize_usage( - response.usage, - provider=agent.provider, - api_mode=agent.api_mode, - ) - # Aggregator-only usage is retained for cost pricing: MoA - # advisor tokens must be priced at each advisor's OWN model - # rate, not the aggregator's, so they are added as dollars - # (below) rather than folded into the priced usage. - aggregator_usage = canonical_usage - # MoA: fold the reference (advisor) fan-out's token usage - # into this turn's REPORTED token counts. MoA runs advisors - # before the aggregator and returns only the aggregator's - # usage, so without this the entire advisor spend — usually - # the bulk of a MoA turn — is invisible in token counts. - _moa_ref_cost = None - _moa_client = getattr(agent, "client", None) - if _moa_client is not None and hasattr(_moa_client, "consume_reference_usage"): - try: - _ref_usage, _moa_ref_cost = _moa_client.consume_reference_usage() - if _ref_usage is not None: - canonical_usage = canonical_usage + _ref_usage - except Exception as _moa_acct_exc: # pragma: no cover - defensive - logger.debug("MoA reference usage accounting failed: %s", _moa_acct_exc) - # Flush the full-turn MoA trace (references + aggregator I/O) - # to disk when moa.save_traces is on. No-op otherwise and - # for non-MoA clients. Uses the live session_id so traces - # land in the right per-session file. On the streaming path - # the aggregator's output wasn't captured inline (its raw - # token stream went to the live consumer), so pass the - # resolved streamed acting text as a fallback — makes the - # trace self-contained instead of only pointing at state.db. - if _moa_client is not None and hasattr(_moa_client, "consume_and_save_trace"): - try: - _agg_streamed_text = ( - getattr(agent, "_current_streamed_assistant_text", "") or "" - ) - _moa_client.consume_and_save_trace( - agent.session_id, - aggregator_output_fallback=_agg_streamed_text or None, - ) - except Exception as _moa_trace_exc: # pragma: no cover - defensive - logger.debug("MoA trace flush failed: %s", _moa_trace_exc) - prompt_tokens = canonical_usage.prompt_tokens - completion_tokens = canonical_usage.output_tokens - total_tokens = canonical_usage.total_tokens - # Forward canonical token + cache buckets so context engines - # can make decisions on cache hit ratios / reasoning costs, - # not just legacy aggregate tokens. Legacy keys stay for - # back-compat with engines that only read prompt/completion/total. - usage_dict = { - "prompt_tokens": prompt_tokens, - "completion_tokens": completion_tokens, - "total_tokens": total_tokens, - "input_tokens": canonical_usage.input_tokens, - "output_tokens": canonical_usage.output_tokens, - "cache_read_tokens": canonical_usage.cache_read_tokens, - "cache_write_tokens": canonical_usage.cache_write_tokens, - "reasoning_tokens": canonical_usage.reasoning_tokens, - } - # Capture the boundary latch before update_from_response() - # consumes it. Only a real provider prompt count for the - # request immediately following a completed compaction can - # prove that attempt effective and rearm the shared budget. - _completed_compaction_pending = bool( - getattr( - agent.context_compressor, - "_verify_compaction_cleared_threshold", - False, - ) - ) - agent.context_compressor.update_from_response(usage_dict) - # Usage-anchored context accounting: snapshot this - # response's exact provider-reported usage against the - # durable transcript. Later context-size checks anchor on - # this and estimate only the messages appended since, - # instead of re-estimating the whole history with - # heuristics. Main-loop responses ONLY — MoA advisor and - # auxiliary calls never reach this site, so they cannot - # pollute the anchor. A usage-less response leaves the - # previous anchor in place (still valid for its base). - # MoA note: use the pre-fold aggregator usage — the folded - # canonical figure adds advisor fan-out tokens that were - # never part of THIS conversation's prompt. - _new_anchor = capture_usage_anchor( - aggregator_usage.prompt_tokens, - aggregator_usage.output_tokens, - messages, - ) - if _new_anchor is not None: - agent._usage_anchor = _new_anchor - # Turn-base anchor for display surfaces: the FIRST - # response of a turn carries minimal current-turn - # reasoning replay, so its prompt_tokens approximate - # the durable transcript cost (what the next turn - # inherits). Later same-turn responses inflate - # prompt_tokens with replayed thinking + tool - # scaffolding that evaporates at the turn boundary — - # anchoring the context meter here instead of on the - # last response removes the end-of-turn sawtooth - # (850K mid-loop -> 600K next turn) that users read - # as a broken compaction. Display-only: compression - # trigger math keeps using real last-request usage. - if api_call_count == 1: - agent._turn_base_usage_anchor = _new_anchor - _compression_threshold = int( - getattr(agent.context_compressor, "threshold_tokens", 0) - or 0 - ) - if _should_rearm_compression_budget( - compression_attempts, - completed_compaction_pending=_completed_compaction_pending, - prompt_tokens=prompt_tokens, - threshold_tokens=_compression_threshold, - ): - logger.info( - "Compression budget rearmed after provider-confirmed " - "recovery: prompt=%s < threshold=%s (attempts were %s/%s)", - f"{prompt_tokens:,}", - f"{_compression_threshold:,}", - compression_attempts, - max_compression_attempts, - ) - compression_attempts = 0 - # Provider-confirmed recovery also invalidates the - # insufficient-progress preflight state: with the - # prompt proven back below the threshold, a prior - # "insufficient progress" verdict (and the stale - # pressure reading it would be compared against) - # describes a request shape that no longer exists. - # Left armed, _preflight_compression_blocked keeps the - # pre-API gate dark for the rest of the turn even - # though the attempt budget was just rearmed, so a - # later pressure spike would grow unchecked until the - # provider's overflow handler fired. - _preflight_compression_blocked = False - _last_preflight_pressure = None - - # Stash this response's canonical usage so the post-turn - # on_turn_complete() observation hook can forward it (the - # same dict shape passed to update_from_response). A turn - # may make several API calls; the engine's per-turn signal - # of interest is the cost/size of the latest assembled - # request, so we keep the most recent call's usage. - agent._last_turn_usage = dict(usage_dict) - elif getattr( - agent.context_compressor, - "awaiting_real_usage_after_compression", - False, - ): - # A response with no usage cannot adjudicate whether the - # prior compaction cleared the threshold. Consume the pending - # verdict now so a much later, unrelated reading is not - # charged to that old compaction, and so preflight deferral - # does not remain latched indefinitely. - agent.context_compressor.update_from_response({}) - - if hasattr(response, 'usage') and response.usage: - # Cache discovered context length after successful call. - # Only persist limits confirmed by the provider (parsed - # from the error message), not guessed probe tiers. - if getattr(agent.context_compressor, "_context_probed", False): - ctx = agent.context_compressor.context_length - if getattr(agent.context_compressor, "_context_probe_persistable", False): - save_context_length(agent.model, agent.base_url, ctx) - agent._safe_print(f"{agent.log_prefix}💾 Cached context length: {ctx:,} tokens for {agent.model}") - agent.context_compressor._context_probed = False - agent.context_compressor._context_probe_persistable = False - - agent.session_prompt_tokens += prompt_tokens - agent.session_completion_tokens += completion_tokens - agent.session_total_tokens += total_tokens - agent.session_api_calls += 1 - agent.session_input_tokens += canonical_usage.input_tokens - agent.session_output_tokens += canonical_usage.output_tokens - agent.session_cache_read_tokens += canonical_usage.cache_read_tokens - agent.session_cache_write_tokens += canonical_usage.cache_write_tokens - agent.session_reasoning_tokens += canonical_usage.reasoning_tokens - # Rolling history for status-bar averages (last 10). - try: - hist = getattr(agent, "_api_latency_history", None) - if hist is not None: - hist.append(float(api_duration)) - ohist = getattr(agent, "_api_output_history", None) - if ohist is not None: - ohist.append(int(canonical_usage.output_tokens or 0)) - except Exception: - pass - - # Log API call details for debugging/observability - _cache_pct = "" - if canonical_usage.cache_read_tokens and prompt_tokens: - _cache_pct = f" cache={canonical_usage.cache_read_tokens}/{prompt_tokens} ({100*canonical_usage.cache_read_tokens/prompt_tokens:.0f}%)" - logger.info( - "API call #%d: model=%s provider=%s in=%d out=%d total=%d latency=%.1fs%s", - agent.session_api_calls, agent.model, agent.provider or "unknown", - prompt_tokens, completion_tokens, total_tokens, - api_duration, _cache_pct, - ) - - # On the MoA path, agent.model/provider are the virtual - # preset name ("closed") and "moa", which have no pricing - # entry — estimating against them returns None and silently - # drops the aggregator's own spend, leaving the session cost - # as advisor-fan-out only (a ~50% undercount when the - # aggregator does the full acting loop). Price the aggregator - # turn at its REAL model/provider, read from the MoA client's - # resolved aggregator slot. - _agg_cost_model = agent.model - _agg_cost_provider = agent.provider - _agg_cost_base_url = agent.base_url - _agg_slot = getattr(_moa_client, "last_aggregator_slot", None) if _moa_client is not None else None - if _agg_slot and _agg_slot.get("model"): - _agg_cost_model = _agg_slot["model"] - _agg_cost_provider = _agg_slot.get("provider") or agent.provider - _agg_cost_base_url = _agg_slot.get("base_url") or agent.base_url - cost_result = estimate_usage_cost( - _agg_cost_model, - aggregator_usage, - provider=_agg_cost_provider, - base_url=_agg_cost_base_url, - api_key=getattr(agent, "api_key", ""), - ) - if cost_result.amount_usd is not None: - agent.session_estimated_cost_usd += float(cost_result.amount_usd) - # Add MoA advisor cost (already priced per-advisor at each - # advisor's own model rate) on top of the aggregator cost. - if _moa_ref_cost is not None: - try: - agent.session_estimated_cost_usd += float(_moa_ref_cost) - except (TypeError, ValueError): # pragma: no cover - defensive - pass - agent.session_cost_status = cost_result.status - agent.session_cost_source = cost_result.source - - # Persist token counts to session DB for /insights. - # Do this for every platform with a session_id so non-CLI - # sessions (gateway, cron, delegated runs) cannot lose - # token/accounting data if a higher-level persistence path - # is skipped or fails. Gateway/session-store writes use - # absolute totals, so they safely overwrite these per-call - # deltas instead of double-counting them. - if agent._session_db and agent.session_id: - try: - # Ensure the session row exists before attempting UPDATE. - # Under concurrent load (cron/kanban), the initial - # _ensure_db_session() may have failed due to SQLite - # locking. Retry here so per-call token deltas are - # not silently lost (UPDATE on a non-existent row - # affects 0 rows without error). - if not agent._session_db_created: - agent._ensure_db_session() - # Per-call cost delta = aggregator cost + MoA - # advisor cost (each priced at its own rate). Folded - # here so state.db's estimated_cost_usd includes the - # full MoA spend, matching the folded token counts. - _cost_delta = None - if cost_result.amount_usd is not None: - _cost_delta = float(cost_result.amount_usd) - if _moa_ref_cost is not None: - try: - _cost_delta = (_cost_delta or 0.0) + float(_moa_ref_cost) - except (TypeError, ValueError): # pragma: no cover - pass - # Enqueued, not written: the background writer - # applies the delta off the turn thread (a cold - # state.db UPDATE here stalled the tool loop for - # up to hundreds of ms per API call). Drained at - # turn finalize via _persist_session. - agent._session_db.queue_token_counts( - agent.session_id, - input_tokens=canonical_usage.input_tokens, - output_tokens=canonical_usage.output_tokens, - cache_read_tokens=canonical_usage.cache_read_tokens, - cache_write_tokens=canonical_usage.cache_write_tokens, - reasoning_tokens=canonical_usage.reasoning_tokens, - estimated_cost_usd=_cost_delta, - cost_status=cost_result.status, - cost_source=cost_result.source, - billing_provider=agent.provider, - billing_base_url=agent.base_url, - billing_mode="subscription_included" - if cost_result.status == "included" else None, - model=agent.model, - api_call_count=1, - ) - except Exception as e: - # Log token persistence failures so they're - # visible in agent.log — silent loss here is - # the root cause of undercounted analytics. - logger.debug( - "Token persistence failed (session=%s, tokens=%d): %s", - agent.session_id, total_tokens, e, - ) - - if agent.verbose_logging: - logging.debug(f"Token usage: prompt={usage_dict['prompt_tokens']:,}, completion={usage_dict['completion_tokens']:,}, total={usage_dict['total_tokens']:,}") - - # Surface cache hit stats for any provider that reports - # them — not just those where we inject cache_control - # markers. OpenAI/Kimi/DeepSeek/Qwen all do automatic - # server-side prefix caching and return - # ``prompt_tokens_details.cached_tokens``; users - # previously could not see their cache % because this - # line was gated on ``_use_prompt_caching``, which is - # only True for Anthropic-style marker injection. - # ``canonical_usage`` is already normalised from all - # three API shapes (Anthropic / Codex / OpenAI-chat) - # so we can rely on its values directly. - cached = canonical_usage.cache_read_tokens - written = canonical_usage.cache_write_tokens - prompt = usage_dict["prompt_tokens"] - if (cached or written) and not agent.quiet_mode: - hit_pct = (cached / prompt * 100) if prompt > 0 else 0 - agent._vprint( - f"{agent.log_prefix} 💾 Cache: " - f"{cached:,}/{prompt:,} tokens " - f"({hit_pct:.0f}% hit, {written:,} written)" - ) + # Fold provider usage into compressor / anchors / session counters / state.db + # (agent/turn_usage.py). A rearmed budget also clears the preflight-block latch. + _usage_outcome = record_response_usage( + agent, + response, + messages=messages, + api_call_count=api_call_count, + api_duration=api_duration, + compression_attempts=compression_attempts, + max_compression_attempts=max_compression_attempts, + ) + compression_attempts = _usage_outcome.compression_attempts + if _usage_outcome.rearmed: + _preflight_compression_blocked = False + _last_preflight_pressure = None _retry.has_retried_429 = False # Reset on success - # Note: don't clear the retry buffer here — an "API call - # success" only means we got bytes back, not that we got - # usable content. Empty responses still loop through the - # empty-retry path below; the buffer is cleared when - # genuinely successful content is detected later (~L4127). - # Clear Nous rate limit state on successful request — - # proves the limit has reset and other sessions can - # resume hitting Nous. + # Don't clear the retry buffer: bytes back != usable content; it is + # cleared once genuine content lands. Clearing Nous rate-limit state + # proves the limit reset so other sessions may resume. if agent.provider == "nous": try: from agent.nous_rate_guard import clear_nous_rate_limit @@ -4921,22 +2883,17 @@ def run_conversation( if agent.thinking_callback: agent.thinking_callback("") if agent._has_pending_redirect(): - # redirect() deliberately used the interrupt machinery to - # cancel only this provider request. Keep its correction - # queued, clear the cancellation bit, and let the outer - # loop rebuild a clean request tail. Never materialize - # incomplete signed/encrypted reasoning items. + # redirect() cancelled only this request: keep the correction + # queued, clear the cancellation bit, let the outer loop rebuild. + # Never materialize incomplete signed/encrypted reasoning items. if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break api_elapsed = time.time() - api_start_time agent._vprint(f"{agent.log_prefix}⚡ Interrupted during API call.", force=True) interrupted = True - # Preserve any assistant text already streamed to the user - # before the stop landed. Dropping it leaves history with no - # record of the half-finished reply on screen, so the next turn - # the model "forgets" what it just said — exactly what users hit - # when they stop to redirect mid-response. + # Keep assistant text already streamed before the stop, else the next + # turn has no record of the half-finished reply. _partial = agent._strip_think_blocks( getattr(agent, "_current_streamed_assistant_text", "") or "" ).strip() @@ -4957,253 +2914,25 @@ def run_conversation( if agent.thinking_callback: agent.thinking_callback("") - # ----------------------------------------------------------- - # UnicodeEncodeError recovery. Two common causes: - # 1. Lone surrogates (U+D800..U+DFFF) from clipboard paste - # (Google Docs, rich-text editors) — sanitize and retry. - # 2. ASCII codec on systems with LANG=C or non-UTF-8 locale - # (e.g. Chromebooks) — any non-ASCII character fails. - # Detect via the error message mentioning 'ascii' codec. - # We sanitize messages in-place and may retry twice: - # first to strip surrogates, then once more for pure - # ASCII-only locale sanitization if needed. - # ----------------------------------------------------------- - if isinstance(api_error, UnicodeEncodeError) and getattr(agent, '_unicode_sanitization_passes', 0) < 2: - _err_str = str(api_error).lower() - _is_ascii_codec = "'ascii'" in _err_str or "ascii" in _err_str - # Detect surrogate errors — utf-8 codec refusing to - # encode U+D800..U+DFFF. The error text is: - # "'utf-8' codec can't encode characters in position - # N-M: surrogates not allowed" - _is_surrogate_error = ( - "surrogate" in _err_str - or ("'utf-8'" in _err_str and not _is_ascii_codec) - ) - # Sanitize surrogates from both the canonical `messages` - # list AND `api_messages` (the API-copy, which may carry - # `reasoning_content`/`reasoning_details` transformed - # from `reasoning` — fields the canonical list doesn't - # have directly). Also clean `api_kwargs` if built and - # `prefill_messages` if present. Mirrors the ASCII - # codec recovery below. - _surrogates_found = _sanitize_messages_surrogates(messages) - if isinstance(api_messages, list): - if _sanitize_messages_surrogates(api_messages): - _surrogates_found = True - if isinstance(api_kwargs, dict): - if _sanitize_structure_surrogates(api_kwargs): - _surrogates_found = True - if isinstance(getattr(agent, "prefill_messages", None), list): - if _sanitize_messages_surrogates(agent.prefill_messages): - _surrogates_found = True - # Gate the retry on the error type, not on whether we - # found anything — _force_ascii_payload / the extended - # surrogate walker above cover all known paths, but a - # new transformed field could still slip through. If - # the error was a surrogate encode failure, always let - # the retry run; the proactive sanitizer at line ~8781 - # runs again on the next iteration. Bounded by - # _unicode_sanitization_passes < 2 (outer guard). - if _surrogates_found or _is_surrogate_error: - agent._unicode_sanitization_passes += 1 - if _surrogates_found: - agent._buffer_vprint( - "⚠️ Stripped invalid surrogate characters from messages. Retrying..." - ) - else: - agent._buffer_vprint( - "⚠️ Surrogate encoding error — retrying after full-payload sanitization..." - ) - continue - if _is_ascii_codec: - agent._force_ascii_payload = True - # ASCII codec: the system encoding can't handle - # non-ASCII characters at all. Sanitize all - # non-ASCII content from messages/tool schemas and retry. - # Sanitize both the canonical `messages` list and - # `api_messages` (the API-copy built before the retry - # loop, which may contain extra fields like - # reasoning_content that are not in `messages`). - _messages_sanitized = _sanitize_messages_non_ascii(messages) - if isinstance(api_messages, list): - _sanitize_messages_non_ascii(api_messages) - # Also sanitize the last api_kwargs if already built, - # so a leftover non-ASCII value in a transformed field - # (e.g. extra_body, reasoning_content) doesn't survive - # into the next attempt via _build_api_kwargs cache paths. - if isinstance(api_kwargs, dict): - _sanitize_structure_non_ascii(api_kwargs) - _prefill_sanitized = False - if isinstance(getattr(agent, "prefill_messages", None), list): - _prefill_sanitized = _sanitize_messages_non_ascii(agent.prefill_messages) - - _tools_sanitized = False - if isinstance(getattr(agent, "tools", None), list): - _tools_sanitized = _sanitize_tools_non_ascii(agent.tools) - - _system_sanitized = False - if isinstance(active_system_prompt, str): - _sanitized_system = _strip_non_ascii(active_system_prompt) - if _sanitized_system != active_system_prompt: - active_system_prompt = _sanitized_system - agent._cached_system_prompt = _sanitized_system - _system_sanitized = True - if isinstance(getattr(agent, "ephemeral_system_prompt", None), str): - _sanitized_ephemeral = _strip_non_ascii(agent.ephemeral_system_prompt) - if _sanitized_ephemeral != agent.ephemeral_system_prompt: - agent.ephemeral_system_prompt = _sanitized_ephemeral - _system_sanitized = True - - _headers_sanitized = False - _default_headers = ( - agent._client_kwargs.get("default_headers") - if isinstance(getattr(agent, "_client_kwargs", None), dict) - else None - ) - if isinstance(_default_headers, dict): - _headers_sanitized = _sanitize_structure_non_ascii(_default_headers) - - # Sanitize the API key — non-ASCII characters in - # credentials (e.g. ʋ instead of v from a bad - # copy-paste) cause httpx to fail when encoding - # the Authorization header as ASCII. This is the - # most common cause of persistent UnicodeEncodeError - # that survives message/tool sanitization (#6843). - _credential_sanitized = False - _raw_key = getattr(agent, "api_key", None) or "" - # Entra ID bearer providers are callables — their - # minted JWTs are always ASCII, so no sanitization - # is needed (and ``_strip_non_ascii`` would crash - # on a callable input). - if _raw_key and isinstance(_raw_key, str): - _clean_key = _strip_non_ascii(_raw_key) - if _clean_key != _raw_key: - agent.api_key = _clean_key - if isinstance(getattr(agent, "_client_kwargs", None), dict): - agent._client_kwargs["api_key"] = _clean_key - # Also update the live client — it holds its - # own copy of api_key which auth_headers reads - # dynamically on every request. - if getattr(agent, "client", None) is not None and hasattr(agent.client, "api_key"): - agent.client.api_key = _clean_key - _credential_sanitized = True - agent._vprint( - f"{agent.log_prefix}⚠️ API key contained non-ASCII characters " - f"(bad copy-paste?) — stripped them. If auth fails, " - f"re-copy the key from your provider's dashboard.", - force=True, - ) - - # Always retry on ASCII codec detection — - # _force_ascii_payload guarantees the full - # api_kwargs payload is sanitized on the - # next iteration (line ~8475). Even when - # per-component checks above find nothing - # (e.g. non-ASCII only in api_messages' - # reasoning_content), the flag catches it. - # Bounded by _unicode_sanitization_passes < 2. - agent._unicode_sanitization_passes += 1 - _any_sanitized = ( - _messages_sanitized - or _prefill_sanitized - or _tools_sanitized - or _system_sanitized - or _headers_sanitized - or _credential_sanitized - ) - if _any_sanitized: - agent._vprint( - f"{agent.log_prefix}⚠️ System encoding is ASCII — stripped non-ASCII characters from request payload. Retrying...", - force=True, - ) - else: - agent._vprint( - f"{agent.log_prefix}⚠️ System encoding is ASCII — enabling full-payload sanitization for retry...", - force=True, - ) - continue - - # ── Image-rejection recovery ────────────────────────────── - # Some providers (mlx-lm, text-only endpoints, text-only - # fallbacks on multimodal models) reject any message that - # contains image_url content with a 4xx error like - # "Only 'text' content type is supported." On first hit, - # strip all images from the message list, mark the session - # as vision-unsupported, and retry with text only. - # - # Detection is best-effort English phrase matching — a - # locale-translated or heavily-reworded upstream error - # will bypass this guard and fall through to the normal - # error handler. Expand the phrase list when new - # provider wordings are observed in the wild. - _err_body = "" - try: - _err_body = str(getattr(api_error, "body", None) or - getattr(api_error, "message", None) or - str(api_error)) - except Exception: - pass - _err_status = getattr(api_error, "status_code", None) - _looks_like_image_rejection = _looks_like_image_content_rejection(_err_body) - # 4xx-only gate: never interpret 5xx/timeout as "server - # said no to images" — those are transient and must - # route to the normal retry path. - _status_ok = _err_status is None or (400 <= int(_err_status) < 500) - if ( - getattr(agent, "_vision_supported", True) - and _looks_like_image_rejection - and _status_ok - ): - agent._vision_supported = False - _imgs_removed = _strip_images_from_messages(messages) - if isinstance(api_messages, list): - _strip_images_from_messages(api_messages) - agent._vprint( - f"{agent.log_prefix}⚠️ Server rejected image content — " - f"switching to text-only mode for this session" - + (". Stripped images from history and retrying." if _imgs_removed else "."), - force=True, - ) - continue - - # ── Bedrock AnthropicBedrock SDK streaming failure ── - # The Anthropic SDK's stream accumulator raises RuntimeError - # "Unexpected event order" when Bedrock returns an error event - # before message_start (throttling, overload, validation). - # Fall back to the native Converse API path for the rest of - # this session — it handles these errors gracefully. Ref: #28156. - if ( - isinstance(api_error, RuntimeError) - and "unexpected event order" in str(api_error).lower() - and getattr(agent, "provider", "") == "bedrock" - and agent.api_mode == "anthropic_messages" - and not getattr(agent, "_bedrock_converse_fallback_attempted", False) - ): - agent._bedrock_converse_fallback_attempted = True - agent.api_mode = "bedrock_converse" - agent._bedrock_region = getattr(agent, "_bedrock_region", None) or "us-east-1" - agent.client = None # Drop the AnthropicBedrock client - agent._client_kwargs = {} - agent._vprint( - f"{agent.log_prefix}⚠️ AnthropicBedrock SDK streaming failed — " - f"falling back to native Converse API for this session.", - force=True, - ) + # Pre-classification recovery (encoding sanitization, image rejection, + # Bedrock SDK streaming fallback) — see agent/turn_recovery.py. + _recovered, active_system_prompt = recover_before_classification( + agent, + api_error, + messages=messages, + api_messages=api_messages, + api_kwargs=api_kwargs, + active_system_prompt=active_system_prompt, + ) + if _recovered: continue status_code = getattr(api_error, "status_code", None) error_context = agent._extract_api_error_context(api_error) # ── Interpreter finalization: abandon immediately ── - # The process is exiting (TUI quit, SIGTERM, one-shot done) - # while this turn — typically the post-turn review fork's - # daemon thread — is mid-flight. Retries, credential - # rotation, and fallbacks are all futile ("cannot schedule - # new futures..."), and the buffered ⚠️/❌ retry trace spams - # the shell after the TUI already exited. End the turn with - # a single log line: no print, no traceback, no debug dump, - # no retry. Same class as cron delivery (#55924/#58720) and - # concurrent tool submission — shared predicate. + # Process is exiting mid-flight: retries/rotation/fallbacks are futile + # and the retry trace spams the shell. One log line; shared predicate. from tools.interpreter_shutdown import interpreter_shutting_down if interpreter_shutting_down(api_error): @@ -5260,498 +2989,43 @@ def run_conversation( reason=classified.reason.value, ) - if ( - classified.reason == FailoverReason.billing - and _is_nous_inference_route( - getattr(agent, "provider", "") or "", - getattr(agent, "base_url", "") or "", - ) - and not _retry.nous_paid_entitlement_refresh_attempted - ): - _retry.nous_paid_entitlement_refresh_attempted = True - if _try_refresh_nous_paid_entitlement_credentials(agent): - agent._vprint( - f"{agent.log_prefix}🔐 Nous paid access verified — " - "refreshed runtime credentials and retrying request...", - force=True, - ) - continue - - recovered_with_pool, _retry.has_retried_429 = agent._recover_with_credential_pool( + # One-shot post-classification recovery chain (entitlement refresh, credential + # pool, image/multimodal strips, per-provider 401 refresh, format-recovery + # strips) — see agent/turn_recovery.py. + _recovered, recovered_with_pool = recover_after_classification( + agent, + api_error, + classified, + _retry, status_code=status_code, - has_retried_429=_retry.has_retried_429, - classified_reason=classified.reason, error_context=error_context, - billing_unverified=classified.billing_unverified, + messages=messages, + api_messages=api_messages, ) - if recovered_with_pool: + if _recovered: continue - # Image-too-large recovery: shrink oversized native image - # parts in-place and retry once. Triggered by Anthropic's - # per-image 5 MB ceiling (400 with "image exceeds 5 MB - # maximum") or any other provider that complains about - # image size. If shrink fails or a second attempt still - # fails, fall through to normal error handling. - if ( - classified.reason == FailoverReason.image_too_large - and not _retry.image_shrink_retry_attempted - ): - _retry.image_shrink_retry_attempted = True - image_max_dimension = _image_error_max_dimension(api_error) or 8000 - if agent._try_shrink_image_parts_in_messages( - api_messages, - max_dimension=image_max_dimension, - ): - agent._vprint( - f"{agent.log_prefix}📐 Image(s) exceeded provider size limit — " - f"shrank and retrying...", - force=True, - ) - continue - else: - logger.info( - "image-shrink recovery: no data-URL image parts found " - "or shrink didn't reduce size; surfacing original error." - ) - - # Multimodal-tool-content recovery: providers that follow - # the OpenAI spec strictly (tool message content must be a - # string) reject our list-type content with a 400. Strip - # image parts from any list-type tool messages, mark the - # (provider, model) as no-list-tool-content for the rest - # of this session so future tool results preemptively - # downgrade, and retry once. See issue #27344. - if ( - classified.reason == FailoverReason.multimodal_tool_content_unsupported - and not _retry.multimodal_tool_content_retry_attempted - ): - _retry.multimodal_tool_content_retry_attempted = True - if agent._try_strip_image_parts_from_tool_messages(api_messages): - agent._vprint( - f"{agent.log_prefix}📐 Provider rejected list-type tool content — " - f"downgraded screenshots to text and retrying...", - force=True, - ) - continue - else: - logger.info( - "multimodal-tool-content recovery: no list-type tool " - "messages with image parts found; surfacing original error." - ) - - # Image-corrupt recovery: the provider decoded the request but - # rejected the image bytes themselves (e.g. xAI's "Invalid PNG - # image." on a re-serialized image part from replayed - # history). Shrinking corrupt bytes doesn't help, so strip the - # image parts and retry once instead of routing through the - # shrink path above. See issue #69078. - if classified.reason == FailoverReason.image_corrupt: - # Strip ONLY the per-call payload copy. api_messages rows - # are shallow copies of canonical history, and the strip - # replaces msg["content"] rather than mutating the shared - # parts list — so canonical messages keep their images. - # A transient provider rejection must not permanently - # erase history (#69104 sweeper review; the copy-on-write - # contract from e762a5a473). - _imgs_removed = False - if isinstance(api_messages, list): - _imgs_removed = _strip_images_from_messages(api_messages) - if _imgs_removed: - agent._vprint( - f"{agent.log_prefix}⚠️ Provider rejected a corrupted image — " - f"stripped images from the retry payload and retrying...", - force=True, - ) - continue - else: - logger.info( - "image-corrupt recovery: no image parts found to " - "strip; surfacing original error." - ) - - # Anthropic OAuth subscription rejected the 1M-context beta - # header ("long context beta is not yet available for this - # subscription"). Disable the beta for the rest of this - # session, rebuild the client, and retry once. 1M-capable - # subscriptions never hit this branch — they accept the - # beta and keep full 1M context. See PR #17680 for the - # original report (we chose reactive recovery over the - # proposed unconditional omit so capable subscriptions - # don't silently lose the capability). - if ( - classified.reason == FailoverReason.oauth_long_context_beta_forbidden - and agent.api_mode == "anthropic_messages" - and agent._is_anthropic_oauth - and not _retry.oauth_1m_beta_retry_attempted - ): - _retry.oauth_1m_beta_retry_attempted = True - if not getattr(agent, "_oauth_1m_beta_disabled", False): - agent._oauth_1m_beta_disabled = True - try: - agent._anthropic_client.close() - except Exception: - pass - agent._rebuild_anthropic_client() - agent._vprint( - f"{agent.log_prefix}🔕 OAuth subscription doesn't support " - f"the 1M-context beta — disabled for this session and retrying...", - force=True, - ) - continue - - if ( - agent.api_mode == "codex_responses" - and agent.provider in {"openai-codex", "xai-oauth"} - and status_code == 401 - and not _retry.codex_auth_retry_attempted - ): - _retry.codex_auth_retry_attempted = True - if agent._try_refresh_codex_client_credentials(force=True): - _label = "xAI OAuth" if agent.provider == "xai-oauth" else "Codex" - agent._buffer_vprint(f"🔐 {_label} auth refreshed after 401. Retrying request...") - continue - if ( - agent.api_mode == "chat_completions" - and agent.provider == "vertex" - and status_code == 401 - and not _retry.vertex_auth_retry_attempted - ): - _retry.vertex_auth_retry_attempted = True - if agent._try_refresh_vertex_client_credentials(): - agent._buffer_vprint("🔐 Vertex AI token refreshed after 401. Retrying request...") - continue - if ( - agent.api_mode in ("chat_completions", "anthropic_messages") - and agent.provider == "nous" - and status_code == 401 - and not _retry.nous_auth_retry_attempted - ): - _retry.nous_auth_retry_attempted = True - if agent._try_refresh_nous_client_credentials(force=True): - agent._buffer_vprint(f"🔐 Nous agent key refreshed after 401. Retrying request...") - continue - # Credential refresh didn't help — show diagnostic info. - # Most common causes: Portal OAuth expired/revoked, - # account out of credits, or agent key blocked. - from hermes_constants import display_hermes_home as _dhh_fn - _dhh = _dhh_fn() - _body_text = "" - try: - _body = getattr(api_error, "body", None) or getattr(api_error, "response", None) - if _body is not None: - _body_text = str(_body)[:200] - except Exception: - pass - print(f"{agent.log_prefix}🔐 Nous 401 — Portal authentication failed.") - if _body_text: - print(f"{agent.log_prefix} Response: {_body_text}") - if not _print_nous_entitlement_guidance(agent, "Nous model access"): - print(f"{agent.log_prefix} Most likely: Portal OAuth expired, account out of credits, or agent key revoked.") - print(f"{agent.log_prefix} Troubleshooting:") - print(f"{agent.log_prefix} • Re-authenticate: hermes auth add nous") - print(f"{agent.log_prefix} • Check credits / billing: https://portal.nousresearch.com") - print(f"{agent.log_prefix} • Verify stored credentials: {_dhh}/auth.json") - print(f"{agent.log_prefix} • Switch providers temporarily: /model --provider openrouter") - if ( - _is_copilot_provider(agent) - and status_code == 401 - and not _retry.copilot_auth_retry_attempted - ): - _retry.copilot_auth_retry_attempted = True - if agent._try_refresh_copilot_client_credentials(): - agent._buffer_vprint("🔐 Copilot credentials refreshed after 401. Retrying request...") - continue - if ( - agent.api_mode == "anthropic_messages" - and status_code == 401 - and hasattr(agent, '_anthropic_api_key') - and not _retry.anthropic_auth_retry_attempted - ): - _retry.anthropic_auth_retry_attempted = True - from agent.anthropic_adapter import _is_oauth_token - from agent.azure_identity_adapter import is_token_provider - if agent._try_refresh_anthropic_client_credentials(): - print(f"{agent.log_prefix}🔐 Anthropic credentials refreshed after 401. Retrying request...") - continue - # Credential refresh didn't help — show diagnostic info - key = agent._anthropic_api_key - print(f"{agent.log_prefix}🔐 Anthropic 401 — authentication failed.") - if is_token_provider(key): - # Azure Foundry Entra ID — the bearer token is - # minted per-request by an httpx event hook on a - # custom http_client passed to the SDK. The 401 - # means Azure rejected the JWT (RBAC role missing, - # az login expired, IMDS unreachable, etc.). - print(f"{agent.log_prefix} Auth method: Microsoft Entra ID (httpx event hook)") - print(f"{agent.log_prefix} Run `hermes doctor` for credential-chain diagnostics, or") - print(f"{agent.log_prefix} `az login` if your developer session expired.") - else: - auth_method = "Bearer (OAuth/setup-token)" if _is_oauth_token(key) else "x-api-key (API key)" - print(f"{agent.log_prefix} Auth method: {auth_method}") - print(f"{agent.log_prefix} Token prefix: {key[:12]}..." if isinstance(key, str) and len(key) > 12 else f"{agent.log_prefix} Token: (empty or short)") - print(f"{agent.log_prefix} Troubleshooting:") - from hermes_constants import display_hermes_home as _dhh_fn - _dhh = _dhh_fn() - print(f"{agent.log_prefix} • Check ANTHROPIC_TOKEN in {_dhh}/.env for Hermes-managed OAuth/setup tokens") - print(f"{agent.log_prefix} • Check ANTHROPIC_API_KEY in {_dhh}/.env for API keys or legacy token values") - print(f"{agent.log_prefix} • For API keys: verify at https://platform.claude.com/settings/keys") - print(f"{agent.log_prefix} • For Claude Code: run 'claude /login' to refresh, then retry") - print(f"{agent.log_prefix} • Legacy cleanup: hermes config set ANTHROPIC_TOKEN \"\"") - print(f"{agent.log_prefix} • Clear stale keys: hermes config set ANTHROPIC_API_KEY \"\"") - - # Thinking block signature recovery. - # - # Anthropic signs thinking blocks against the full turn - # content. Any upstream mutation (context compression, - # session truncation, message merging) invalidates the - # signature and the API replies HTTP 400 ("invalid - # signature" or "cannot be modified"). Recovery strips - # ``reasoning_details`` so the retry sends no thinking - # blocks at all. One-shot per outer loop. - # - # The strip targets ``api_messages``, which is the - # API-call-time list that ``_build_api_kwargs`` consumes - # on every retry. ``api_messages`` was populated once at - # the start of the turn from shallow copies of - # ``messages``, so mutating it does not touch the - # canonical store. The previous implementation popped - # ``reasoning_details`` from ``messages`` instead, which - # had two problems: ``api_messages`` carried its own - # reference to the field through the shallow copy, so the - # retry's wire payload still included thinking blocks and - # the recovery never reached the API; and the mutation - # persisted into ``state.db`` through any subsequent - # ``_persist_session`` call, permanently corrupting the - # conversation. Future turns would replay the stripped - # state, hit the same 400, and the agent would terminate - # with ``max_retries_exhausted``, often spawning - # cascading compaction-ended sessions chained off the - # corrupted parent. - if ( - classified.reason == FailoverReason.thinking_signature - and not _retry.thinking_sig_retry_attempted - ): - _retry.thinking_sig_retry_attempted = True - _api_stripped = 0 - for _m in api_messages: - if isinstance(_m, dict) and "reasoning_details" in _m: - _m.pop("reasoning_details", None) - _api_stripped += 1 - agent._vprint( - f"{agent.log_prefix}⚠️ Thinking block signature invalid, " - f"stripped reasoning_details from api_messages for retry...", - force=True, - ) - logger.warning( - "%sThinking block signature recovery: stripped " - "reasoning_details from %d api_messages " - "(canonical messages unchanged)", - agent.log_prefix, _api_stripped, - ) - continue - - # ── Invalid encrypted reasoning replay recovery ─────── - # OpenAI Responses API surfaces (and some compatible relays) - # return HTTP 400 ``invalid_encrypted_content`` when a - # replayed ``codex_reasoning_items`` blob from a previous - # turn fails verification (provider rotated the encryption - # key, the route doesn't actually persist reasoning state, - # etc.). Recovery: disable replay for the rest of the - # session, strip cached items from history, retry once. - # One-shot — if a second 400 fires we fall through to the - # normal retry/backoff path. Only fires for codex_responses - # mode with at least one assistant message that has cached - # ``codex_reasoning_items``; without replay state, the - # error is unrelated to our cache so the normal retry path - # handles it (the provider is rejecting something else). - if ( - classified.reason == FailoverReason.invalid_encrypted_content - and not _retry.invalid_encrypted_content_retry_attempted - and agent.api_mode == "codex_responses" - and bool(getattr(agent, "_codex_reasoning_replay_enabled", True)) - and any( - isinstance(_m, dict) - and _m.get("role") == "assistant" - and isinstance(_m.get("codex_reasoning_items"), list) - and _m.get("codex_reasoning_items") - for _m in messages - ) - ): - _retry.invalid_encrypted_content_retry_attempted = True - replay_stats = agent._disable_codex_reasoning_replay(messages) - agent._vprint( - f"{agent.log_prefix}⚠️ Encrypted reasoning replay was rejected by the provider — " - f"disabled replay and stripped {replay_stats['items']} item(s) from " - f"{replay_stats['messages']} message(s), retrying...", - force=True, - ) - logger.warning( - "%sInvalid encrypted reasoning recovery: disabled replay and stripped %d items from %d messages", - agent.log_prefix, - replay_stats["items"], - replay_stats["messages"], - ) - continue - - # ── Native compaction rejection recovery ────────────── - # Provider explicitly rejected the ``context_management`` - # field (structured 400 naming the param). One-shot: turn - # native compaction off for the rest of the session and - # retry — the next _build_api_kwargs re-resolves the gate - # and omits the field, and Hermes' local compression takes - # over as the sole owner. Generic 4xx/5xx/timeouts do NOT - # match (see is_native_compaction_rejection) and take the - # normal retry path. - if ( - agent.api_mode == "codex_responses" - and not _retry.native_compaction_reject_retry_attempted - and bool(getattr(agent, "codex_responses_native_compaction", False)) - ): - from agent.native_compaction import is_native_compaction_rejection - if is_native_compaction_rejection( - api_error, getattr(api_error, "status_code", None) - ): - _retry.native_compaction_reject_retry_attempted = True - agent.codex_responses_native_compaction = False - agent._vprint( - f"{agent.log_prefix}⚠️ Provider rejected native compaction " - f"(context_management) — disabled for this session, " - f"local compression stays active. Retrying...", - force=True, - ) - logger.warning( - "%sNative compaction rejection recovery: disabled " - "codex_responses_native for this session and retrying", - agent.log_prefix, - ) - continue - - # ── llama.cpp grammar-parse recovery ────────────────── - # llama.cpp's ``json-schema-to-grammar`` converter rejects - # regex escape classes (``\d``, ``\w``, ``\s``) and most - # ``format`` values in tool schemas. MCP servers emit - # these routinely for date/phone/email params. Recovery: - # strip ``pattern``/``format`` from ``agent.tools`` and - # retry once. We keep the keywords by default so cloud - # providers get the full prompting hints; this branch - # fires only for users on llama.cpp's OAI server. - if ( - classified.reason == FailoverReason.llama_cpp_grammar_pattern - and not _retry.llama_cpp_grammar_retry_attempted - ): - _retry.llama_cpp_grammar_retry_attempted = True - try: - from tools.schema_sanitizer import strip_pattern_and_format - _, _stripped = strip_pattern_and_format(agent.tools) - except Exception as _strip_exc: # pragma: no cover — defensive - logger.warning( - "%sllama.cpp grammar recovery: strip helper failed: %s", - agent.log_prefix, _strip_exc, - ) - _stripped = 0 - if _stripped: - agent._vprint( - f"{agent.log_prefix}⚠️ llama.cpp rejected tool schema grammar — " - f"stripped {_stripped} pattern/format keyword(s), retrying...", - force=True, - ) - logger.warning( - "%sllama.cpp grammar recovery: stripped %d " - "pattern/format keyword(s) from tool schemas", - agent.log_prefix, _stripped, - ) - continue - # No keywords found to strip — fall through to normal - # retry path rather than loop forever on the same error. - logger.warning( - "%sllama.cpp grammar error but no pattern/format " - "keywords to strip — falling through to normal retry", - agent.log_prefix, - ) - retry_count += 1 elapsed_time = time.time() - api_start_time agent._touch_activity( f"API error recovery (attempt {retry_count}/{max_retries})" ) - error_type = type(api_error).__name__ - error_msg = str(api_error).lower() - _error_summary = agent._summarize_api_error(api_error) - logger.warning( - "API call failed (attempt %s/%s) error_type=%s %s summary=%s", - retry_count, - max_retries, - error_type, - agent._client_log_context(), - _error_summary, + error_type, error_msg, _provider, _base, _model = log_api_error_attempt( + agent, + api_error, + retry_count=retry_count, + max_retries=max_retries, + status_code=status_code, + elapsed_time=elapsed_time, + api_messages=api_messages, + approx_tokens=approx_tokens, ) - _provider = getattr(agent, "provider", "unknown") - _base = getattr(agent, "base_url", "unknown") - _model = getattr(agent, "model", "unknown") - _status_code_str = f" [HTTP {status_code}]" if status_code else "" - agent._buffer_vprint(f"⚠️ API call failed (attempt {retry_count}/{max_retries}): {error_type}{_status_code_str}") - agent._buffer_vprint(f" 🔌 Provider: {_provider} Model: {_model}") - agent._buffer_vprint(f" 🌐 Endpoint: {_base}") - agent._buffer_vprint(f" 📝 Error: {_error_summary}") - if status_code and status_code < 500: - _err_body = getattr(api_error, "body", None) - _err_body_str = str(_err_body)[:300] if _err_body else None - if _err_body_str: - agent._buffer_vprint(f" 📋 Details: {_err_body_str}") - agent._buffer_vprint(f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens") - - # Actionable hint for OpenRouter "no tool endpoints" error. - # Buffered like the rest of the retry trace — surfaced only - # if every retry+fallback exhausts. Avoids spamming users - # who recover automatically via fallback. - if ( - agent._is_openrouter_url() - and "support tool use" in error_msg - ): - agent._buffer_vprint( - f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings." - ) - if agent.providers_allowed: - agent._buffer_vprint( - " Your provider_routing.only restriction is filtering out tool-capable providers." - ) - agent._buffer_vprint( - " Try removing the restriction or adding providers that support tools for this model." - ) - agent._buffer_vprint( - f" Check which providers support tools: https://openrouter.ai/models/{_model}" - ) - - # Actionable hint for a bare 404 on a provider whose catalogue - # uses ``vendor/model`` ids. A model id that lost its prefix - # (e.g. ``nemotron-…`` instead of ``nvidia/nemotron-…``) gets - # a content-free "404 page not found" from the provider that - # never names the model, so it reads like an outage or an auth - # failure. Name the real cause and the exact id to use (#78796). - if getattr(api_error, "status_code", None) == 404: - try: - from hermes_cli.model_normalize import suggest_prefixed_model_id - - _suggestion = suggest_prefixed_model_id(_provider, _model) - except Exception: - _suggestion = None - if _suggestion: - agent._buffer_vprint( - f" 💡 Model '{_model}' is not a valid id for provider {_provider} — " - f"it is missing its vendor prefix." - ) - agent._buffer_vprint( - f" Did you mean '{_suggestion}'? Re-pick it with `hermes model`." - ) - # Check for interrupt before deciding to retry if agent._interrupt_requested: - # Preserve a pending redirect (mid-stream correction): the - # user is steering, not stopping. Rebuild the turn from the - # correction instead of aborting with a dead-end interrupt. + # Preserve a pending redirect: the user is steering, not stopping + # — rebuild the turn from the correction instead of aborting. if agent.clear_interrupt(preserve_redirect=True): _retry.restart_with_redirected_messages = True break @@ -5768,918 +3042,105 @@ def run_conversation( "interrupted": True, } - # Check for 413 payload-too-large BEFORE generic 4xx handler. - # A 413 is a payload-size error — the correct response is to - # compress history and retry, not abort immediately. - status_code = getattr(api_error, "status_code", None) - - # ── Respect disabled auto-compaction on overflow ────── - # Ported from anomalyco/opencode#30749. When the user has - # turned auto-compaction off (``compression.enabled: false``), - # NO automatic compaction trigger may fire — including the - # provider/request-size overflow recovery paths below - # (long-context-tier 429, 413 payload-too-large, and - # context-overflow). Without this guard the proactive - # threshold path correctly honours the setting (see the - # preflight check and the post-response ``should_compress`` - # gate) but a provider overflow error would still silently - # compress + rotate the session, bypassing the user's - # explicit choice. Surface a terminal error instead so the - # user can compact manually (``/compress``), start fresh - # (``/new``), switch to a larger-context model, or reduce - # attachments. Forced compaction via ``/compress`` - # (``force=True``) is unaffected — it never reaches this loop. - # - # Output-cap errors (max_tokens too large) are NOT input - # overflow — the recovery is a max_tokens-only retry that - # does not require compression. Exempt them from this guard - # so the retry still fires even when compression is disabled. - _overflow_reasons = { - FailoverReason.long_context_tier, - FailoverReason.payload_too_large, - FailoverReason.context_overflow, - } - _is_output_cap_error = ( - is_output_cap_error(error_msg) - or parse_available_output_tokens_from_error(error_msg) is not None + _ce = route_classified_error( + agent, + api_error, + classified, + _retry, + error_msg=error_msg, + error_context=error_context, + recovered_with_pool=recovered_with_pool, + base_url=_base, + model=_model, + messages=messages, + api_messages=api_messages, + system_message=system_message, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + retry_count=retry_count, + max_retries=max_retries, + compression_attempts=compression_attempts, + max_compression_attempts=max_compression_attempts, + api_call_count=api_call_count, + effective_task_id=effective_task_id, ) - if ( - classified.reason in _overflow_reasons - and not getattr(agent, "compression_enabled", True) - and not _is_output_cap_error - ): - agent._flush_status_buffer() - agent._vprint( - f"{agent.log_prefix}❌ Context overflow, but auto-compaction is disabled " - f"(compression.enabled: false).", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} 💡 Run /compress to compact manually, /new to start fresh, " - f"switch to a larger-context model, or reduce attachments.", - force=True, - ) - logger.error( - f"{agent.log_prefix}Context overflow ({classified.reason.value}) with " - f"auto-compaction disabled — not compressing." - ) - agent._persist_session(messages, conversation_history) - _final_response = ( - "Context overflow and auto-compaction is disabled " - "(compression.enabled: false). Run /compress to compact manually, " - "/new to start fresh, or switch to a larger-context model." - ) - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compaction_disabled": True, - } + status_code = _ce.status_code + messages = _ce.messages + active_system_prompt = _ce.active_system_prompt + conversation_history = _ce.conversation_history + retry_count = _ce.retry_count + max_retries = _ce.max_retries + compression_attempts = _ce.compression_attempts + is_rate_limited = _ce.is_rate_limited + _wrapped_output_cap_budget = _ce.wrapped_output_cap_budget + _is_zai_coding_overload = _ce.is_zai_coding_overload + if _ce.provider_overflow_recovery_pending: + _provider_overflow_recovery_pending = True + if _ce.action == "return": + return _ce.result + if _ce.action == "break": + break + if _ce.action == "continue": + continue - # ── Anthropic Sonnet long-context tier gate ─────────── - # Anthropic returns HTTP 429 "Extra usage is required for - # long context requests" when a Claude Max (or similar) - # subscription doesn't include the 1M-context tier. This - # is NOT a transient rate limit — retrying or switching - # credentials won't help. Reduce context to 200k (the - # standard tier) and compress. - if classified.reason == FailoverReason.long_context_tier: - _reduced_ctx = 200000 - compressor = agent.context_compressor - old_ctx = compressor.context_length - if old_ctx > _reduced_ctx: - compressor.update_model( - model=agent.model, - context_length=_reduced_ctx, - base_url=agent.base_url, - api_key=getattr(agent, "api_key", ""), - provider=agent.provider, - api_mode=agent.api_mode, - ) - # Context probing flags — only set on built-in - # compressor (plugin engines manage their own). - if hasattr(compressor, "_context_probed"): - compressor._context_probed = True - # Don't persist — this is a subscription-tier - # limitation, not a model capability. If the - # user later enables extra usage the 1M limit - # should come back automatically. - compressor._context_probe_persistable = False - agent._buffer_vprint( - f"⚠️ Anthropic long-context tier " - f"requires extra usage — reducing context: " - f"{old_ctx:,} → {_reduced_ctx:,} tokens" - ) - - compression_attempts += 1 - if compression_attempts <= max_compression_attempts: - original_len = len(messages) - # Option A (LCM issue 441): overhead-aware request size so recovery arms on - # the true request (msgs + tools + system), not the tool-blind message count. - messages, active_system_prompt = agent._compress_context( - messages, system_message, - approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), - task_id=effective_task_id, - ) - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - if len(messages) < original_len or old_ctx > _reduced_ctx: - agent._buffer_status( - COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE.format( - new_ctx=_reduced_ctx, old_ctx=old_ctx - ) - ) - time.sleep(2) - # Same class as the generic overflow handler below: - # the provider proved the request does not fit the - # (now-reduced) window, and row count alone is not - # proof the rebuilt request does. Recheck the - # complete request before the next provider call. - _provider_overflow_recovery_pending = True - _retry.restart_with_compressed_messages = True - break - # Fall through to normal error handling if compression - # is exhausted or didn't help. - - # Eager fallback for rate-limit errors (429 or quota exhaustion) - # and transport errors (connection failure / timeout / provider - # overloaded). Rate limits and billing: switch immediately — - # the primary provider won't recover within the retry window. - # Transport errors: allow 1 retry first (transient hiccups - # recover), then fall back if the provider is truly unreachable. - is_rate_limited = classified.reason in { - FailoverReason.rate_limit, - FailoverReason.billing, - FailoverReason.upstream_rate_limit, - } - # Relay-wrapped output-cap errors: some gateways wrap an - # upstream "[400]: max_tokens (...) exceeds model's maximum - # output tokens (...)" as HTTP 429, which classifies as - # rate_limit. The failure is a deterministic request-shape - # problem — falling back to another provider (or burning - # generic retries) can't fix it, but the output-cap clamp - # below can, in one retry (#72281). Parse once here; the - # result gates both the eager-fallback exemption and the - # widened is_context_length_error entry, and is reused as - # available_out inside the handler. - _wrapped_output_cap_budget = ( - parse_available_output_tokens_from_error(error_msg) - if classified.reason == FailoverReason.rate_limit - else None + _ov = recover_from_overflow( + agent, + api_error, + classified, + _retry, + status_code=status_code, + error_msg=error_msg, + wrapped_output_cap_budget=_wrapped_output_cap_budget, + messages=messages, + api_messages=api_messages, + system_message=system_message, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + approx_tokens=approx_tokens, + compression_attempts=compression_attempts, + max_compression_attempts=max_compression_attempts, + api_call_count=api_call_count, + effective_task_id=effective_task_id, ) - _is_transport_failure = classified.reason in { - FailoverReason.timeout, - FailoverReason.overloaded, - } - # Z.AI Coding Plan GLM-5.2 overload 429s classify as - # `overloaded` (to spare the credential pool), but `overloaded` - # is excluded from `is_rate_limited` — the gate for the adaptive - # Z.AI backoff below. Detect the overload directly so its - # long-backoff schedule runs, and raise the retry ceiling so the - # long tier (30/60/90/120s) is reachable. See - # zai_coding_overload_retry_ceiling() for the ceiling rationale. - _is_zai_coding_overload = is_zai_coding_overload_error( - base_url=str(_base), model=_model, error=api_error - ) - if _is_zai_coding_overload: - max_retries = max(max_retries, zai_coding_overload_retry_ceiling()) - _should_fallback = ( - (is_rate_limited and _wrapped_output_cap_budget is None) - or (_is_transport_failure and retry_count >= 2) - ) - if _should_fallback and agent._fallback_index < len(agent._fallback_chain): - # Don't eagerly fallback if credential pool rotation may - # still recover. See _pool_may_recover_from_rate_limit - # for the single-credential-pool exception. Fixes #11314. - # - # Exception: an upstream-aggregator 429 — the credential - # pool can't help when the *upstream* model (DeepSeek, - # etc.) is throttling OpenRouter, so always fall back to a - # different model regardless of pool state. - _is_upstream = classified.reason == FailoverReason.upstream_rate_limit - pool_may_recover = ( - False if _is_upstream - else _ra()._pool_may_recover_from_rate_limit( - agent._credential_pool, - ) - ) - if not pool_may_recover: - if _is_upstream: - _upstream_name = (classified.error_context or {}).get( - "upstream_provider", "aggregator" - ) - agent._buffer_status( - f"⚠️ Upstream {_upstream_name} rate-limited — " - "switching to fallback model..." - ) - elif classified.reason == FailoverReason.billing: - if classified.billing_unverified: - # Ambiguous body (#82154) — don't assert billing. - agent._buffer_status( - "⚠️ Provider reported usage/credit exhaustion " - "(unverified — may be a content-filter rejection) " - "— switching to fallback provider..." - ) - else: - agent._buffer_status( - "⚠️ Billing or credits exhausted — switching to fallback provider..." - ) - elif _is_transport_failure: - agent._buffer_status( - "⚠️ Provider unreachable — switching to fallback provider..." - ) - else: - agent._buffer_status("⚠️ Rate limited — switching to fallback provider...") - if agent._try_activate_fallback(reason=classified.reason): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) - retry_count = 0 - compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True - break + messages = _ov.messages + active_system_prompt = _ov.active_system_prompt + conversation_history = _ov.conversation_history + approx_tokens = _ov.approx_tokens + compression_attempts = _ov.compression_attempts + is_context_length_error = _ov.is_context_length_error + if _ov.provider_overflow_recovery_pending: + _provider_overflow_recovery_pending = True + if _ov.action == "return": + return _ov.result + if _ov.action == "break": + break + if _ov.action == "continue": + continue - # ── Auth-failure provider failover ─────────────────────── - # A 401/403 that survives the per-provider credential-refresh - # attempt above (each guarded by its own - # ``*_auth_retry_attempted`` flag) means the active provider's - # credential or endpoint is broken in a way refreshing can't - # fix (revoked OAuth, blocked/expired key, an account pinned to - # a dead/staging endpoint). Previously the loop only printed - # "switch providers manually" advice and fell through, so a - # user with a configured fallback chain kept thrashing on the - # same dead credential every turn instead of failing over. - # Escalate to the fallback chain here, mirroring the rate- - # limit/billing failover above. When no fallback is configured - # (or the chain is exhausted), _try_activate_fallback returns - # False and we fall through to the existing terminal handling - # + provider-specific troubleshooting guidance unchanged. - if ( - classified.is_auth - and not _retry.auth_failover_attempted - and agent._fallback_index < len(agent._fallback_chain) - ): - _retry.auth_failover_attempted = True - agent._buffer_status( - "🔐 Authentication failed and could not be refreshed — " - "switching to fallback provider..." - ) - if agent._try_activate_fallback(reason=classified.reason): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) - retry_count = 0 - compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True - break - - # ── Nous Portal: record rate limit & skip retries ───── - # When Nous returns a 429 that is a genuine account- - # level rate limit, record the reset time to a shared - # file so ALL sessions (cron, gateway, auxiliary) know - # not to pile on, then skip further retries -- each - # one burns another RPH request and deepens the hole. - # The retry loop's top-of-iteration guard will catch - # this on the next pass and try fallback or bail. - # - # IMPORTANT: Nous Portal multiplexes multiple upstream - # providers (DeepSeek, Kimi, MiMo, Hermes). A 429 can - # also mean an UPSTREAM provider is out of capacity - # for one specific model -- transient, clears in - # seconds, nothing to do with the caller's quota. - # Tripping the cross-session breaker on that would - # block every Nous model for minutes. We use - # ``is_genuine_nous_rate_limit`` to tell the two - # apart via the 429's own x-ratelimit-* headers and - # the last-known-good state captured on the previous - # successful response. - if ( - is_rate_limited - and agent.provider == "nous" - and classified.reason == FailoverReason.rate_limit - and not recovered_with_pool - ): - _genuine_nous_rate_limit = False - try: - from agent.nous_rate_guard import ( - is_genuine_nous_rate_limit, - record_nous_rate_limit, - ) - _err_resp = getattr(api_error, "response", None) - _err_hdrs = ( - getattr(_err_resp, "headers", None) - if _err_resp else None - ) - _genuine_nous_rate_limit = is_genuine_nous_rate_limit( - headers=_err_hdrs, - last_known_state=agent._rate_limit_state, - ) - if _genuine_nous_rate_limit: - record_nous_rate_limit( - headers=_err_hdrs, - error_context=error_context, - ) - else: - logger.info( - "Nous 429 looks like upstream capacity " - "(no exhausted bucket in headers or " - "last-known state) -- not tripping " - "cross-session breaker." - ) - except Exception: - pass - if _genuine_nous_rate_limit: - # Re-enter the loop exactly once so the - # top-of-loop Nous guard handles fallback or - # bails cleanly. (Setting retry_count to - # max_retries would make the while condition - # false immediately and the guard would never - # run -- no fallback, generic exhaustion error.) - retry_count = max(0, max_retries - 1) - continue - # Upstream capacity 429: fall through to normal - # retry logic. A different model (or the same - # model a moment later) will typically succeed. - - is_payload_too_large = ( - classified.reason == FailoverReason.payload_too_large - ) - - # Actionable hint for GitHub Models (Azure) 413 errors. - # The free tier enforces a hard 8K token cap per request, - # which Hermes' system prompt + tool schemas alone exceed. - # Compression can't help — the floor is the system prompt - # itself, not the conversation — so surface a clear "not - # compatible" message instead of looping into three futile - # compression attempts. - if ( - status_code == 413 - and isinstance(agent.base_url, str) - and base_url_host_matches(agent.base_url, "models.inference.ai.azure.com") - ): - agent._vprint( - f"{agent.log_prefix} 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} request at ~8K tokens. Hermes' system prompt + tool schemas baseline", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} exceeds that floor, so this endpoint cannot run an agentic loop.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} Use the `copilot` provider with a Copilot subscription token (`hermes", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} setup` → GitHub Copilot), or pick any other provider.", - force=True, - ) - - if is_payload_too_large: - compression_attempts += 1 - if compression_attempts > max_compression_attempts: - # Terminal — surface the buffered retry trace. - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached for payload-too-large error.", force=True) - agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error("%s413 compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) - agent._persist_session(messages, conversation_history) - _final_response = f"Request payload too large: max compression attempts ({max_compression_attempts}) reached." - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - } - agent._buffer_status(f"⚠️ Request payload too large (413) — compression attempt {compression_attempts}/{max_compression_attempts}...") - - original_len = len(messages) - # A 413 is a BYTE-size error, so this branch scores - # progress in BYTES of the serialized messages payload — - # exact and free — never the token estimate. The - # estimator prices every image at a flat per-image token - # cost (see estimate_messages_tokens_rough) so screenshots - # don't trigger premature compaction; that deliberate - # byte-blindness means compaction can free megabytes of - # base64 (real case: two vision results = 96.6% of the - # request body but ~3.7% of the estimate) while the token - # delta stays under any threshold. Token-scored progress - # here burned all attempts on "no progress" and wedged - # the session permanently. (#88960 / #47339) - original_bytes = serialized_messages_bytes(messages) - _overflow_input = messages - # Option A (LCM issue 441): overhead-aware request size so recovery arms on the - # true request (msgs + tools + system), not the tool-blind message count. - messages, active_system_prompt = agent._compress_context( - messages, system_message, - approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), - task_id=effective_task_id, - # #100661: the provider proved the request does not fit. - # Ignore the summary-failure cooldown for this ONE - # attempt (bounded by max_compression_attempts) instead - # of deferring every turn until the ladder lapses. - bypass_cooldown=True, - ) - if messages is _overflow_input and compression_skipped_due_to_lock(agent): - # #69870 lock-skip: the provider proved the request - # does not fit, but this compression pass no-oped only - # because another path holds the session's compression - # lock. Temporary defer, not exhaustion — refund the - # attempt and end the turn softly so the gateway does - # NOT auto-reset the session (#9893/#35809). - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count - ) - if messages is _overflow_input and compression_blocked_transiently(agent): - # #97488 transient-block: compression no-oped because a - # timed guard (host-timeout cooldown / structural - # backoff) is active — a temporary defer, not evidence - # of incompressibility. Never classify it as - # compression_exhausted (gateway auto-reset). - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count, - reason="transient_block", - ) - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - - # Re-measure after compression. Same-message-count - # compression (tool-result pruning, in-place summarization) - # can materially reduce request size without reducing the - # message array (#39550), and — the image-dominated case — - # compaction's historical-media aging (#97160) can free - # megabytes of base64 that the token estimate never - # counted. Bytes are the yardstick for a 413; tokens are - # kept only for status display. - new_tokens = estimate_messages_tokens_rough(messages) - approx_tokens = new_tokens # update for downstream logging - new_bytes = serialized_messages_bytes(messages) - - made_progress = ( - len(messages) < original_len - or (new_bytes > 0 and new_bytes < original_bytes * 0.95) - ) - if made_progress: - if len(messages) < original_len: - agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) - else: - agent._buffer_status( - f"🗜️ Compressed {original_bytes:,} → {new_bytes:,} " - f"payload bytes, retrying..." - ) - time.sleep(2) # Brief pause between compression retries - _retry.restart_with_compressed_messages = True - break - else: - if agent._try_strip_image_parts_from_tool_messages( - api_messages, - remember_model=False, - ): - agent._buffer_status( - "📐 Compression could not reduce the request further — " - "removed retained vision payloads and retrying..." - ) - continue - - # Terminal — surface buffered context so the user - # sees what compression attempts were made. - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Payload too large and cannot compress further.", force=True) - agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error("%s413 payload too large. Cannot compress further.", agent.log_prefix) - agent._persist_session(messages, conversation_history) - _final_response = "Request payload too large (413). Cannot compress further." - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - } - - # Check for context-length errors BEFORE generic 4xx handler. - # The classifier detects context overflow from: explicit error - # messages, generic 400 + large session heuristic (#1630), and - # server disconnect + large session pattern (#2153). - is_context_length_error = ( - classified.reason == FailoverReason.context_overflow - # Relay-wrapped output-cap 429s (parsed once above, where - # the eager-fallback exemption is gated) route into the - # output-cap clamp below instead of provider failover or - # generic retries (#72281). - or _wrapped_output_cap_budget is not None - ) - - if is_context_length_error: - compressor = agent.context_compressor - old_ctx = compressor.context_length - - # ── Distinguish two very different errors ─────────── - # 1. "Prompt too long": the INPUT exceeds the context window. - # Fix: reduce context_length + compress history. - # 2. "max_tokens too large": input is fine, but - # input_tokens + requested max_tokens > context_window. - # Fix: reduce max_tokens (the OUTPUT cap) for this call. - # Do NOT shrink context_length — the window is unchanged. - # - # Note: max_tokens = output token cap (one response). - # context_length = total window (input + output combined). - available_out = parse_available_output_tokens_from_error(error_msg) - if available_out is not None: - # This is an output-cap error, not input overflow. - # The provider's available_tokens is the authoritative - # cap for the failed request, so keep it as an upper - # bound. Also estimate the current API request shape - # (system prompt, injected context, tool schemas) because - # Hermes may add API-only content not present in persisted - # messages. Use the smaller budget and apply a small - # safety margin. Do not alter context_length. - request_input_estimate = estimate_request_tokens_rough( - api_messages, tools=agent.tools or None, - ) - local_available_out = old_ctx - request_input_estimate - if local_available_out > 0: - safe_out = max(1, min(available_out, local_available_out) - 64) - else: - # The rough local estimate can overshoot the real - # request size. Fall back to the provider-reported - # budget, which is authoritative for the failed - # request. - safe_out = max(1, available_out - 64) - agent._ephemeral_max_output_tokens = safe_out - agent._buffer_vprint( - f"⚠️ Output cap too large for current prompt — " - f"retrying with max_tokens={safe_out:,} " - f"(provider_available={available_out:,}, " - f"estimated_request_tokens={request_input_estimate:,}; " - f"context_length unchanged at {old_ctx:,})" - ) - # Still count against compression_attempts so we don't - # loop forever if the error keeps recurring. - compression_attempts += 1 - if compression_attempts > max_compression_attempts: - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) - agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) - agent._persist_session(messages, conversation_history) - _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - } - # Also compress the message history so the output-cap - # retry does not just spin on max_tokens alone. The - # compressor drops the middle window, freeing enough - # tokens for the total to fit inside context_length. - # (#55546) - try: - original_len = len(messages) - original_tokens = estimate_messages_tokens_rough(messages) - _overflow_input = messages - messages, active_system_prompt = agent._compress_context( - messages, system_message, - approx_tokens=request_input_estimate, - task_id=effective_task_id, - bypass_cooldown=True, # #100661 provider-proven overflow - ) - if messages is _overflow_input and compression_skipped_due_to_lock(agent): - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count - ) - if messages is _overflow_input and compression_blocked_transiently(agent): - # #97488: timed transient guard — defer, never - # exhaustion (gateway auto-reset). - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count, - reason="transient_block", - ) - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - new_tokens = estimate_messages_tokens_rough(messages) - if len(messages) < original_len: - agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) - elif new_tokens > 0 and new_tokens < original_tokens * 0.95: - agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) - except Exception: - # Compression must never turn an output-cap error - # fatal — fall through and retry on max_tokens alone. - logger.warning( - "%sOutput-cap compression hit an error; retrying on max_tokens only.", - agent.log_prefix, - ) - _retry.restart_with_compressed_messages = True - break - - # The error is output-cap-shaped (about max_tokens being - # too large) but the provider's wording didn't let us parse - # the available output budget. Compression CANNOT help here - # — the input already fits; the call fails deterministically - # on the oversized max_tokens. Routing it into compression - # re-sends the same max_tokens, gets the identical 400, and - # death-loops until "cannot compress further" (#55546). - # Fail fast with an actionable message instead of looping. - if is_output_cap_error(error_msg): - agent._flush_status_buffer() - agent._vprint( - f"{agent.log_prefix}❌ The provider rejected the request because " - f"max_tokens exceeds its output cap for this model.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} 💡 Lower model.max_tokens in your config.yaml to " - f"at or below the model's max-output limit. " - f"(This is an output-cap error, not a context overflow — " - f"compression cannot fix it.)", - force=True, - ) - logger.error( - f"{agent.log_prefix}Output-cap error not routed into compression " - f"(max_tokens over provider cap): {error_msg[:200]}" - ) - agent._persist_session(messages, conversation_history) - _final_response = ( - "max_tokens exceeds the provider's output cap for this model. " - "Lower model.max_tokens in config.yaml." - ) - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - } - - # Error is about the INPUT being too large. Only reduce - # context_length when the provider explicitly reports the - # real lower limit. If the provider only says "input - # exceeds the context window", keep the configured window - # and try compression; guessing probe tiers can incorrectly - # turn a user-configured 1M window into 256K/128K/64K. - new_ctx = get_context_length_from_provider_error(error_msg, old_ctx) - _provider_lower = (getattr(agent, "provider", "") or "").lower() - _base_lower = (getattr(agent, "base_url", "") or "").rstrip("/").lower() - is_minimax_provider = ( - _provider_lower in {"minimax", "minimax-cn"} - or _base_lower.startswith(( - "https://api.minimax.io/anthropic", - "https://api.minimaxi.com/anthropic", - )) - ) - minimax_delta_only_overflow = ( - is_minimax_provider - and new_ctx is None - and "context window exceeds limit (" in error_msg - ) - - if new_ctx is not None: - agent._buffer_vprint(f"Context limit detected from API: {new_ctx:,} tokens (was {old_ctx:,})") - compressor.update_model( - model=agent.model, - context_length=new_ctx, - base_url=agent.base_url, - api_key=getattr(agent, "api_key", ""), - provider=agent.provider, - api_mode=agent.api_mode, - ) - # Persist an explicit provider-reported limit before - # compression/retry. The next request can be rate - # limited, omit usage, or the process can restart; none - # of those should discard metadata the provider already - # confirmed. Keep the probe flags as a best-effort - # post-success retry if this write cannot complete. - save_context_length(agent.model, agent.base_url, new_ctx) - # Context probing flags — only set on built-in - # compressor (plugin engines manage their own). This - # value came from the provider, so it is safe to cache. - if hasattr(compressor, "_context_probed"): - compressor._context_probed = True - compressor._context_probe_persistable = True - agent._buffer_vprint(f"⚠️ Context length exceeded — using provider limit: {old_ctx:,} → {new_ctx:,} tokens") - elif minimax_delta_only_overflow: - agent._buffer_vprint( - f"Provider reported overflow amount only; " - f"keeping context_length at {old_ctx:,} tokens and compressing." - ) - else: - agent._buffer_vprint( - f"⚠️ Context length exceeded, but provider did not report a max context length; " - f"keeping context_length at {old_ctx:,} tokens and compressing." - ) - - compression_attempts += 1 - if compression_attempts > max_compression_attempts: - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) - agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) - agent._persist_session(messages, conversation_history) - _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - } - agent._buffer_status(COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE.format(tokens=approx_tokens, attempt=compression_attempts, cap=max_compression_attempts)) - - original_len = len(messages) - original_tokens = estimate_messages_tokens_rough(messages) - _overflow_input = messages - # Option A (LCM issue 441): pass the OVERHEAD-AWARE request size (msgs + tool - # schemas + system), not the tool-blind message count, so LCM forced-overflow - # recovery arms on the TRUE request that overflowed. See hermes-lcm engine - # _should_force_overflow_recovery. (approx_tokens stays for the status display.) - messages, active_system_prompt = agent._compress_context( - messages, system_message, - approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), - task_id=effective_task_id, - # #100661: the provider proved the request does not fit. - # Ignore the summary-failure cooldown for this ONE - # attempt (bounded by max_compression_attempts) instead - # of deferring every turn until the ladder lapses. - bypass_cooldown=True, - ) - if messages is _overflow_input and compression_skipped_due_to_lock(agent): - # #69870 lock-skip: the provider proved the request - # does not fit, but this compression pass no-oped only - # because another path holds the session's compression - # lock. Temporary defer, not exhaustion — refund the - # attempt and end the turn softly so the gateway does - # NOT auto-reset the session (#9893/#35809). - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count - ) - if messages is _overflow_input and compression_blocked_transiently(agent): - # #97488 transient-block: a timed guard (host-timeout - # cooldown / structural backoff) no-oped this pass — - # defer softly, never compression_exhausted (which - # would auto-reset the session). - compression_attempts -= 1 - agent._persist_session(messages, conversation_history) - return _compression_deferred_result( - agent, messages, api_call_count, - reason="transient_block", - ) - if context_compression_timed_out(agent): - # Host progress-aware timeout (#98722, salvaged from - # #98741): the provider proved the request does not - # fit, but this recovery pass spent the full wait - # budget without a committed summary. Re-sending the - # unchanged request would bounce off the same overflow - # error and re-enter compression in the same turn. End - # the turn with the typed recovery contract instead — - # transcript intact, no further doomed provider sends. - agent._persist_session(messages, conversation_history) - _final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - "turn_exit_reason": "context_compression_timeout", - } - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - - # Re-estimate tokens after compression. Same-message-count - # compression (tool-result pruning, in-place summarization) - # can materially reduce request size without reducing the - # message array. (#39550) - new_tokens = estimate_messages_tokens_rough(messages) - approx_tokens = new_tokens # update for downstream logging - - if len(messages) < original_len or (new_tokens > 0 and new_tokens < original_tokens * 0.95) or (new_ctx and new_ctx < old_ctx): - if len(messages) < original_len: - agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) - elif new_tokens > 0 and new_tokens < original_tokens * 0.95: - agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) - time.sleep(2) # Brief pause between compression retries - # Rebuild the complete request before the next provider - # call and force normal preflight to honor it. Message - # count alone is not proof that system/tool-inclusive - # token pressure fell. - _provider_overflow_recovery_pending = True - _retry.restart_with_compressed_messages = True - break - else: - # Can't compress further and already at minimum tier - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Context length exceeded and cannot compress further.", force=True) - agent._vprint(f"{agent.log_prefix} 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.", force=True) - logger.error("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}") - agent._persist_session(messages, conversation_history) - _final_response = f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further." - return { - "final_response": _final_response, - "messages": messages, - "completed": False, - "api_calls": api_call_count, - "error": _final_response, - "partial": True, - "failed": True, - "compression_exhausted": True, - } - - # Check for non-retryable client errors. The classifier - # already accounts for 413, 429, 529 (transient), context - # overflow, and generic-400 heuristics. Local validation - # errors (ValueError, TypeError) are programming bugs. - # Exclude UnicodeEncodeError — it's a ValueError subclass - # but is handled separately by the surrogate sanitization - # path above. Exclude json.JSONDecodeError — also a - # ValueError subclass, but it indicates a transient - # provider/network failure (malformed response body, - # truncated stream, routing layer corruption), not a - # local programming bug, and should be retried (#14782). + # Non-retryable: ValueError/TypeError are local bugs, except + # UnicodeEncodeError (surrogate path above) and json.JSONDecodeError, a + # transient provider/network failure that must be retried (#14782). is_local_validation_error = ( isinstance(api_error, (ValueError, TypeError)) and not isinstance( api_error, (UnicodeEncodeError, json.JSONDecodeError) ) - # ssl.SSLError (and its subclass SSLCertVerificationError) - # inherits from OSError *and* ValueError via Python MRO, - # so the isinstance(ValueError) check above would - # misclassify a TLS transport failure as a local - # programming bug and abort without retrying. Exclude - # ssl.SSLError explicitly so the error classifier's - # retryable=True mapping takes effect instead. + # ssl.SSLError inherits from OSError *and* ValueError, so the + # ValueError check would misclassify a TLS failure as a local bug; + # keep it retryable. and not isinstance(api_error, ssl.SSLError) - # Provider/SDK "NoneType is not iterable" failures are - # shape mismatches from upstream (e.g. chatgpt.com Codex - # backend response.completed.output=null) — not local - # programming bugs. Even after #33042 made our own - # consumer immune, third-party shims and mocked clients - # can still surface this shape via TypeError. Treat - # them as retryable so the error classifier's normal - # retry/fallback path runs instead of killing the turn - # as non-retryable (which left Telegram users staring - # at a bare "Non-retryable error" with no recovery). + # "NoneType is not iterable" TypeErrors are upstream shape + # mismatches (e.g. Codex response.completed.output=null), reachable + # via shims/mocks — retryable so the fallback path runs. and not ( isinstance(api_error, TypeError) and "nonetype" in str(api_error).lower() and "not iterable" in str(api_error).lower() ) ) - # ``FailoverReason.billing`` (HTTP 402) is NOT in this - # exclusion set. By the time we reach this block: - # • credential-pool rotation (line ~2031) has already - # fired for billing and either ``continue``d or - # returned (False, ...) — pool is exhausted or absent. - # • the eager-fallback branch above (line ~2422) also - # fires on billing and ``continue``s if a fallback - # provider is configured. - # Falling through to here means BOTH recovery paths - # gave up. Treating 402 as retryable from this point - # just burns more paid requests against a depleted - # balance with no recovery mechanism left — see #31273 - # (real-world: ~$40 in 48h on a 24/7 gateway). Aborting - # mirrors how 401/403 (also ``should_fallback=True``) - # already behave once their recovery paths have failed. + # ``FailoverReason.billing`` (402) is deliberately NOT excluded: pool + # rotation and eager fallback already gave up, so retrying only burns + # paid requests on a depleted balance. Mirrors 401/403. (#31273) is_client_error = ( is_local_validation_error or ( @@ -6697,17 +3158,9 @@ def run_conversation( ) and not is_context_length_error if is_client_error: - # Copilot self-heal BEFORE fallback: a stale/degraded - # credential surfaces as a 400 - # ``model_not_available_for_integrator`` / - # ``model_not_supported`` (not a clean 401), so the 401 - # refresh path above never fired. Force a fresh token - # exchange + client rebuild and retry once on the SAME - # provider — a fresh 437-char API token routes to the - # correct integrator and the model becomes available again. - # Single-shot guard prevents looping on a genuinely - # unavailable model. Copilot-scoped so other providers' - # real 400s are untouched. + # Copilot self-heal BEFORE fallback: a stale credential yields a 400 + # ``model_not_available_for_integrator`` / ``model_not_supported``, + # not a 401. Fresh token + client rebuild, one retry, SAME provider. if ( _is_copilot_provider(agent) and not _retry.copilot_stale_cred_retry_attempted @@ -6723,12 +3176,9 @@ def run_conversation( ) retry_count = 0 continue - # Try fallback before aborting — a different provider may - # not have the same issue (rate limit, auth, etc.). Only - # announce the attempt when a fallback chain actually - # exists; otherwise "trying fallback..." is a lie and the - # session looks like it's recovering when it's about to - # abort silently (#35314, #17446). + # Try fallback before aborting; announce it only when a fallback + # chain exists, else "trying fallback..." lies before a silent abort + # (#35314). if agent._has_pending_fallback(): if classified.reason == FailoverReason.content_policy_blocked: agent._buffer_status("⚠️ Provider safety filter blocked this request — trying fallback...") @@ -6737,212 +3187,38 @@ def run_conversation( else: agent._buffer_status(f"⚠️ Non-retryable error (HTTP {status_code}) — trying fallback...") if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) retry_count = 0 compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True break - if api_kwargs is not None: - agent._dump_api_request_debug( - api_kwargs, reason="non_retryable_client_error", error=api_error, - ) - # Terminal — flush buffered context so the user sees - # what was tried before the abort. - agent._flush_status_buffer() - # Summarize once: Cloudflare/proxy HTML challenge pages and - # other raw provider bodies must be collapsed to a short - # one-liner here, otherwise the full page leaks into the - # returned ``error`` field and downstream consumers deliver - # it verbatim (e.g. a cron failure notification dumped a - # ~60KB Cloudflare challenge page as 31 Discord messages). - _nonretryable_summary = agent._summarize_api_error(api_error) - if classified.reason == FailoverReason.content_policy_blocked: - agent._emit_status( - f"❌ Provider safety filter blocked this request: " - f"{_nonretryable_summary}" - ) - elif classified.reason == FailoverReason.ssl_cert_verification: - agent._emit_status( - f"❌ TLS certificate verification failed: " - f"{_nonretryable_summary}" - ) - else: - agent._emit_status( - f"❌ Non-retryable error (HTTP {status_code}): " - f"{_nonretryable_summary}" - ) - agent._vprint(f"{agent.log_prefix}❌ Non-retryable client error (HTTP {status_code}). Aborting.", force=True) - agent._vprint(f"{agent.log_prefix} 🔌 Provider: {_provider} Model: {_model}", force=True) - agent._vprint(f"{agent.log_prefix} 🌐 Endpoint: {_base}", force=True) - # Actionable guidance for common auth errors - if classified.is_auth or classified.reason == FailoverReason.billing: - if classified.reason == FailoverReason.billing and _print_billing_or_entitlement_guidance( - agent, - capability="model access", - provider=_provider, - base_url=str(_base), - model=_model, - unverified=classified.billing_unverified, - ): - pass - elif _provider == "nous" and _print_nous_entitlement_guidance( - agent, - "Nous model access", - ): - pass - elif _provider in {"openai-codex", "xai-oauth", "nous"} and status_code == 401: - if _provider == "openai-codex": - agent._vprint(f"{agent.log_prefix} 💡 Codex OAuth token was rejected (HTTP 401). Your token may have been", force=True) - agent._vprint(f"{agent.log_prefix} refreshed by another client (Codex CLI, VS Code). To fix:", force=True) - agent._vprint(f"{agent.log_prefix} 1. Run `codex` in your terminal to generate fresh tokens.", force=True) - agent._vprint(f"{agent.log_prefix} 2. Then run `hermes auth` to re-authenticate.", force=True) - elif _provider == "xai-oauth": - agent._vprint(f"{agent.log_prefix} 💡 xAI OAuth token was rejected (HTTP 401). To fix:", force=True) - agent._vprint(f"{agent.log_prefix} re-authenticate with xAI Grok OAuth (SuperGrok / Premium+) from `hermes model`.", force=True) - else: # nous - agent._vprint(f"{agent.log_prefix} 💡 Nous Portal OAuth token was rejected (HTTP 401). Your token may be", force=True) - agent._vprint(f"{agent.log_prefix} expired, revoked, or your account may be out of credits. To fix:", force=True) - agent._vprint(f"{agent.log_prefix} 1. Re-authenticate: hermes portal", force=True) - agent._vprint(f"{agent.log_prefix} 2. Check your portal account: https://portal.nousresearch.com", force=True) - # ``:free`` is OpenRouter slug syntax; Nous Portal will reject - # the model name even after a successful re-auth. - if isinstance(_model, str) and _model.endswith(":free"): - agent._vprint(f"{agent.log_prefix} ⚠️ Note: `{_model}` looks like an OpenRouter slug (`:free` suffix).", force=True) - agent._vprint(f"{agent.log_prefix} Nous Portal won't recognize that model name. Either switch to a", force=True) - agent._vprint(f"{agent.log_prefix} Nous catalog model, or run `/model openrouter:{_model}` to use OpenRouter.", force=True) - else: - agent._vprint(f"{agent.log_prefix} 💡 Your API key was rejected by the provider. Check:", force=True) - agent._vprint(f"{agent.log_prefix} • Is the key valid? Run: hermes setup", force=True) - agent._vprint(f"{agent.log_prefix} • Does your account have access to {_model}?", force=True) - if base_url_host_matches(str(_base), "openrouter.ai"): - agent._vprint(f"{agent.log_prefix} • Check credits: https://openrouter.ai/settings/credits", force=True) - else: - agent._vprint(f"{agent.log_prefix} 💡 This type of error won't be fixed by retrying.", force=True) - # Content-policy blocks deserve their own actionable - # guidance — neither "fix your API key" nor "retry won't - # help" tells the user what to actually do. The provider - # has refused this specific prompt, so the recovery is - # either a rephrase or routing to a different model. - if classified.reason == FailoverReason.content_policy_blocked: - agent._vprint( - f"{agent.log_prefix} 💡 The provider's safety filter rejected this specific prompt.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} • Try rephrasing the request, narrowing the context, or splitting into smaller steps.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} • Configure a fallback provider so future blocks route automatically:", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} hermes fallback add (interactive picker — same as `hermes model`)", - force=True, - ) - # TLS certificate failures are environment problems, not - # provider/prompt problems — tell the user exactly which - # knobs fix each common cause. Inspired by Claude Code - # v2.1.199's immediate SSL fix hints. - if classified.reason == FailoverReason.ssl_cert_verification: - agent._vprint( - f"{agent.log_prefix} 💡 The TLS certificate chain could not be verified. This fails the same", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} way on every retry — fix the environment, then try again:", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} • Corporate TLS-inspecting proxy? Point Python at its CA bundle:", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} export SSL_CERT_FILE=/path/to/corp-ca.pem (also REQUESTS_CA_BUNDLE)", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} • Missing/stale system CA store? Install/refresh it:", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} pip install --upgrade certifi (macOS: run 'Install Certificates.command')", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} • Self-signed local endpoint (llama.cpp, LM Studio, vLLM)? Use http://", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} for localhost, or add the server's cert to your trust store.", - force=True, - ) - logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) - # Skip session persistence when the error is likely - # context-overflow related (status 400 + large session). - # Persisting the failed user message would make the - # session even larger, causing the same failure on the - # next attempt. (#1630) - if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80): - agent._vprint( - f"{agent.log_prefix}⚠️ Skipping session persistence " - f"for large failed session to prevent growth loop.", - force=True, - ) - else: - agent._persist_session(messages, conversation_history) - if classified.reason == FailoverReason.content_policy_blocked: - _policy_response = ( - "⚠️ The model provider's safety filter blocked this request " - "(not a Hermes/gateway failure).\n\n" - f"Provider message: {_nonretryable_summary}\n\n" - f"{_CONTENT_POLICY_RECOVERY_HINT}" - ) - return _content_policy_blocked_result( - messages, - api_call_count, - final_response=_policy_response, - error_detail=_nonretryable_summary, - ) - # Billing walls are the common non-retryable abort: enrich - # the result with the same structured recovery descriptor as - # the max-retries path so every surface (CLI, TUI, desktop) - # renders one consistent billing signal. - if classified.reason == FailoverReason.billing: - return _billing_failure_result( - classified=classified, - summary=_nonretryable_summary, - messages=messages, - api_call_count=api_call_count, - provider=_provider, - base_url=_base, - model=_model, - ) - return { - "final_response": _nonretryable_summary, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "failed": True, - "error": _nonretryable_summary, - } + return nonretryable_client_error_result( + agent, + api_error, + classified, + status_code=status_code, + api_kwargs=api_kwargs, + api_messages=api_messages, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + approx_tokens=approx_tokens, + provider=_provider, + base_url=_base, + model=_model, + ) if retry_count >= max_retries: - # Before falling back, try rebuilding the primary - # client once for transient transport errors (stale - # connection pool, TCP reset). Only attempted once - # per API call block. + # Before fallback, rebuild the primary client once for transient + # transport errors (stale pool, TCP reset). Once per API call block. if not _retry.primary_recovery_attempted and agent._try_recover_primary_transport( api_error, retry_count=retry_count, max_retries=max_retries, ): _retry.primary_recovery_attempted = True retry_count = 0 - # Primary transport recovery starts a fresh attempt - # cycle. Re-open fallback state so a follow-on 429 can - # still activate fallback_providers after stale - # pre-recovery fallback/credential-pool bookkeeping. + # Transport recovery starts a fresh attempt cycle: re-open + # fallback state so a follow-on 429 can still activate + # fallback_providers. _retry.has_retried_429 = False agent._fallback_index = 0 agent._fallback_activated = False @@ -6951,305 +3227,60 @@ def run_conversation( if agent._has_pending_fallback(): agent._buffer_status(f"⚠️ Max retries ({max_retries}) exhausted — trying fallback...") if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) retry_count = 0 compression_attempts = 0 - _retry.primary_recovery_attempted = False - _retry.restart_with_rebuilt_messages = True break - # Terminal — flush buffered retry/fallback trace. - agent._flush_status_buffer() - _final_summary = agent._summarize_api_error(api_error) - _billing_guidance = "" - if classified.reason == FailoverReason.billing: - if classified.billing_unverified: - # Ambiguous body (#82154) — hedge the terminal line. - agent._emit_status( - "❌ Provider reported usage/credit exhaustion " - f"(unverified — may be a content-filter rejection) — {_final_summary}" - ) - else: - agent._emit_status(f"❌ Billing or credits exhausted — {_final_summary}") - _billing_guidance = _billing_or_entitlement_message( - capability="model access", - provider=_provider, - base_url=str(_base), - model=_model, - unverified=classified.billing_unverified, - ) - _print_billing_or_entitlement_guidance( - agent, - capability="model access", - provider=_provider, - base_url=str(_base), - model=_model, - unverified=classified.billing_unverified, - ) - elif is_rate_limited: - agent._emit_status(f"❌ Rate limited after {max_retries} retries — {_final_summary}") - else: - agent._emit_status(f"❌ API failed after {max_retries} retries — {_final_summary}") - agent._vprint(f"{agent.log_prefix} 💀 Final error: {_final_summary}", force=True) - - # Detect SSE stream-drop pattern (e.g. "Network - # connection lost") and surface actionable guidance. - # This typically happens when the model generates a - # very large tool call (write_file with huge content) - # and the proxy/CDN drops the stream mid-response. - _is_stream_drop = ( - not getattr(api_error, "status_code", None) - and any(p in error_msg for p in ( - "connection lost", "connection reset", - "connection closed", "network connection", - "network error", "terminated", - )) - ) - if _is_stream_drop: - agent._vprint( - f"{agent.log_prefix} 💡 The provider's stream " - f"connection keeps dropping. This often happens " - f"when the model tries to write a very large " - f"file in a single tool call.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} Try asking the model " - f"to use execute_code with Python's open() for " - f"large files, or to write the file in smaller " - f"sections.", - force=True, - ) - - # Detect thinking-timeout pattern: a known reasoning model - # hit a transport-layer error before the first content - # token arrived. Distinct from _is_stream_drop above - # (which fires for large file-write stream drops) and - # from any classifier reason that's not a transport - # timeout. Reuses the reasoning-model allowlist from - # agent/reasoning_timeouts.py (Fixes #52217) so the - # trigger is consistent with what the per-model - # stale-timeout floor covers. After the classifier - # override at agent/error_classifier.py:720-738 (this - # PR), transport disconnects on reasoning models route - # to FailoverReason.timeout rather than - # context_overflow, so this branch actually fires. - # Detection and message text live in - # agent.thinking_timeout_guidance so they're - # unit-testable without driving the full retry loop. - # (Part 2 of Fixes #52310.) - from agent.thinking_timeout_guidance import ( - is_thinking_timeout, - ) - _is_thinking_timeout = is_thinking_timeout( + return max_retries_exhausted_result( + agent, + api_error, classified, - _model, - error_msg, - ) - if _is_thinking_timeout: - agent._vprint( - f"{agent.log_prefix} 💡 The model's thinking " - f"phase exceeded the upstream proxy's idle " - f"timeout before the first content token " - f"arrived. This is a known issue with " - f"reasoning models behind cloud gateways " - f"(NVIDIA NIM, OpenAI, Anthropic, DeepSeek).", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} Workarounds in priority order:", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} 1. Set " - f"`providers.{_provider}.models.{_model}.stale_timeout_seconds: 900` " - f"in `~/.hermes/config.yaml` to extend the per-call " - f"timeout. (Hermes's built-in floor is 600s for " - f"known reasoning models — if you still see this " - f"after raising, the upstream cap is even shorter.)", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} 2. Lower `reasoning_budget` or set " - f"`reasoning_effort: medium` on this model if the provider supports it.", - force=True, - ) - agent._vprint( - f"{agent.log_prefix} 3. Use a smaller / faster reasoning " - f"model if the task doesn't require deep thinking.", - force=True, - ) - - logger.error( - "%sAPI call failed after %s retries. %s | provider=%s model=%s msgs=%s tokens=~%s", - agent.log_prefix, max_retries, _final_summary, - _provider, _model, len(api_messages), f"{approx_tokens:,}", - ) - if api_kwargs is not None: - agent._dump_api_request_debug( - api_kwargs, reason="max_retries_exhausted", error=api_error, - ) - agent._persist_session(messages, conversation_history) - _billing_block = None - _billing_unverified = False - if classified.reason == FailoverReason.billing: - _billing_unverified = classified.billing_unverified - _final_response = _billing_terminal_label( - _final_summary, _billing_unverified - ) - if _billing_guidance: - _final_response += f"\n\n{_billing_guidance}" - # Structured recovery descriptor so every surface renders - # the same link + label from one signal (see helper). - _billing_block = _billing_block_dict( - _provider, _base, _model, _billing_guidance, - unverified=_billing_unverified, - ) - else: - _final_response = f"API call failed after {max_retries} retries: {_final_summary}" - if _is_thinking_timeout: - # Thinking-timeout guidance overrides the generic - # stream-drop guidance — the latter is wrong for - # this case (it suggests splitting large file - # writes, which isn't what happened). See the - # reasoning-model override at - # agent/error_classifier.py:720-738 and the - # detection block above for context. - from agent.thinking_timeout_guidance import ( - build_thinking_timeout_guidance, - ) - _final_response += build_thinking_timeout_guidance( - provider=_provider, - model=_model, - ) - elif _is_stream_drop: - _final_response += ( - "\n\nThe provider's stream connection keeps " - "dropping — this often happens when generating " - "very large tool call responses (e.g. write_file " - "with long content). Try asking me to use " - "execute_code with Python's open() for large " - "files, or to write in smaller sections." - ) - return { - "final_response": _final_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "failed": True, - "error": _final_summary, - # Surface the classified reason so callers (notably the - # kanban worker path in cli.py) can distinguish a - # transient throttle from a real failure and choose a - # different exit code. ``rate_limit`` / ``billing`` here - # mean "quota wall, not a task error". - "failure_reason": classified.reason.value, - # The classifier's own retry verdict — UI surfaces use - # this instead of re-deriving from the reason string. - "failure_retryable": bool(classified.retryable), - # True when the billing verdict rests on an ambiguous - # body (#82154) — may be a content-filter rejection. - "billing_unverified": _billing_unverified, - # Present only for billing walls: structured recovery - # descriptor (provider, billing_url, is_nous, message). - "billing_block": _billing_block, - } - - # For rate limits, respect the Retry-After header if present - _retry_after = None - if is_rate_limited: - _resp_headers = getattr(getattr(api_error, "response", None), "headers", None) - if _resp_headers and hasattr(_resp_headers, "get"): - _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") - if _ra_raw: - try: - # Cap at 10 minutes. Anthropic Tier 1 input-token - # buckets reset in ~171s, so a 120s cap caused us to - # retry before the actual reset window and re-trip the - # limit. 600s covers all realistic provider reset - # windows while still rejecting pathological values. (#26293) - _retry_after = min(float(_ra_raw), 600) - except (TypeError, ValueError): - pass - wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) - _backoff_policy = None - if (is_rate_limited or _is_zai_coding_overload) and not _retry_after: - wait_time, _backoff_policy = adaptive_rate_limit_backoff( - retry_count, - base_url=str(_base), + max_retries=max_retries, + is_rate_limited=is_rate_limited, + error_msg=error_msg, + api_kwargs=api_kwargs, + api_messages=api_messages, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + approx_tokens=approx_tokens, + provider=_provider, + base_url=_base, model=_model, - error=api_error, - default_wait=wait_time, ) - if is_rate_limited or _is_zai_coding_overload: - _policy_note = "" - if _backoff_policy == "zai_coding_overload_long": - _policy_note = " (Z.AI Coding overload adaptive long backoff)" - elif _backoff_policy == "zai_coding_overload_short": - _policy_note = " (Z.AI Coding overload short retry)" - _wait_reason = "Provider overloaded" if _is_zai_coding_overload and not is_rate_limited else "Rate limited" - _rate_limit_status = f"⏱️ {_wait_reason}. Waiting {wait_time:.1f}s (attempt {retry_count + 1}/{max_retries}){_policy_note}..." - # Normal retries are buffered to avoid noisy transient chatter. Long - # Z.AI Coding waits are different: they can last minutes, so surface - # progress immediately instead of making the TUI look frozen. - if _backoff_policy == "zai_coding_overload_long": - agent._emit_status(_rate_limit_status) - else: - agent._buffer_status(_rate_limit_status) - else: - agent._buffer_status(f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})...") - logger.warning( - "Retrying API call in %ss (attempt %s/%s) %s policy=%s error=%s", - wait_time, - retry_count, - max_retries, - agent._client_log_context(), - _backoff_policy or "default", + + wait_time = compute_error_backoff( + agent, api_error, + retry_count=retry_count, + max_retries=max_retries, + is_rate_limited=is_rate_limited, + is_zai_coding_overload=_is_zai_coding_overload, + base_url=_base, + model=_model, ) - # Sleep in small increments so we can respond to interrupts quickly - # instead of blocking the entire wait_time in one sleep() call - sleep_end = time.time() + wait_time - _backoff_touch_counter = 0 - while time.time() < sleep_end: - if agent._interrupt_requested: - # Same preserve-redirect rule as the retry-wait above: - # a steering correction must survive backoff, not die - # as "Operation interrupted". - if agent.clear_interrupt(preserve_redirect=True): - _retry.restart_with_redirected_messages = True - break - agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during retry wait, aborting.", force=True) - _interrupt_text = f"Operation interrupted: retrying API call after error (retry {retry_count}/{max_retries})." - close_interrupted_tool_sequence(messages, _interrupt_text) - agent._persist_session(messages, conversation_history) - agent.clear_interrupt() - return { - "final_response": _interrupt_text, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "interrupted": True, - } - time.sleep(0.2) # Check interrupt every 200ms - # Touch activity every ~30s so the gateway's inactivity - # monitor knows we're alive during backoff waits. - _backoff_touch_counter += 1 - if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s - agent._touch_activity( - f"error retry backoff ({retry_count}/{max_retries}), " - f"{int(sleep_end - time.time())}s remaining" - ) + # Same preserve-redirect rule as the invalid-response wait: a steering + # correction must survive backoff, not die as "Operation interrupted". + _interrupted = interruptible_backoff_sleep( + agent, wait_time, _retry, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + abort_message="Interrupt detected during retry wait, aborting.", + interrupt_text=f"Operation interrupted: retrying API call after error (retry {retry_count}/{max_retries}).", + activity_label=f"error retry backoff ({retry_count}/{max_retries})", + ) + if _interrupted is not None: + return _interrupted if _retry.restart_with_redirected_messages: - # Leave the retry loop — the check right below rebuilds this - # iteration from the correction instead of re-firing the - # stale request. + # Leave the retry loop — the check below rebuilds this iteration + # from the correction instead of re-firing the stale request. break if _retry.restart_with_redirected_messages: - # The cancelled request produced no valid assistant item. Reuse the - # same logical iteration after the outer loop appends the displayed - # partial context and correction to ``messages``. + # Cancelled request produced no valid assistant item: reuse the same logical + # iteration after the outer loop appends partial context + correction. api_call_count -= 1 agent.iteration_budget.refund() _retry.restart_with_redirected_messages = False @@ -7263,9 +3294,8 @@ def run_conversation( if _retry.restart_with_compressed_messages: api_call_count -= 1 agent.iteration_budget.refund() - # Count compression restarts toward the retry limit to prevent - # infinite loops when compression reduces messages but not enough - # to fit the context window. + # Compression restarts count toward the retry limit so a compression that + # shrinks messages but not enough can't loop forever. retry_count += 1 _retry.restart_with_compressed_messages = False if _should_skip_model_call_for_reference_handoff( @@ -7279,15 +3309,9 @@ def run_conversation( final_response = _HANDOFF_SKIP_FINAL_RESPONSE _turn_exit_reason = "compaction_handoff_not_actionable" break - # In-loop compression rebuilt `messages` with fresh compaction - # copies, so the pre-compression current-turn index is stale. - # Re-anchor exactly like the prologue does: a stale index that - # lands on a historical user message would make the live-compose - # fallback inject this turn's prefetch into that message on the - # wire only, diverging the next turn's replayed prefix there. - # Ordered AFTER the handoff guard: the guard may have re-appended - # this turn's real user ask (restore path), and the anchor must - # land on that restored row, not on -1 / a pre-restore index. + # In-loop compression rebuilt `messages`; re-anchor the current-turn index + # like the prologue, AFTER the handoff guard (it may re-append this turn's + # ask). A stale anchor injects prefetch into a historical row. current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) @@ -7295,30 +3319,21 @@ def run_conversation( continue if _retry.restart_with_rebuilt_messages: - # A stream stall or provider failure was escalated to the - # fallback chain (10 activation sites in the retry loop set this - # flag and break here). Re-issue the API call against the - # now-active fallback provider. Refund the budget/count for the - # stalled attempt so the fallback gets a fair turn. + # A stall/failure escalated to the fallback chain: re-issue against the + # active fallback provider, refunding budget/count for the stalled attempt. api_call_count -= 1 agent.iteration_budget.refund() _retry.restart_with_rebuilt_messages = False - # Failover shrank the compressor's context window to the - # fallback's; clear the preflight block so the pre-API preflight - # re-runs against the new threshold before the first fallback - # call (#84733). Hoisted here (the single consumer) so every - # activation site — including ones added later — gets it. + # Failover shrank the compressor window: clear the preflight block so + # preflight re-runs before the first fallback call. Hoisted to the single + # consumer. (#84733) _preflight_compression_blocked = False continue if _retry.restart_with_length_continuation: - # Progressively boost the output token budget on each retry. - # Retry 1 → 2× base, retry 2 → 4× base, retry 3 → 8× base, - # retry 4 → 16× base, then cap at 32 768. - # Applies to all providers via _ephemeral_max_output_tokens. - # If the original request already used a larger provider/model - # default budget, keep that floor so continuation retries do - # not accidentally downshift to a much smaller cap. + # Boost output budget per retry: 2×, 4×, 8×, 16× base, capped at 32 768, via + # _ephemeral_max_output_tokens. Keep a larger original provider/model + # default as the floor so retries never downshift. _boost_base = agent.max_tokens if agent.max_tokens else 4096 _boost = _boost_base * (2 ** length_continue_retries) _requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs) @@ -7328,9 +3343,7 @@ def run_conversation( agent._ephemeral_max_output_tokens = min(_boost, _boost_cap) continue - # Guard: if all retries exhausted without a successful response - # (e.g. repeated context-length errors that exhausted retry_count), - # the `response` variable is still None. Break out cleanly. + # All retries may exhaust with `response` still None; break out cleanly. if response is None: _turn_exit_reason = "all_retries_exhausted_no_response" print(f"{agent.log_prefix}❌ All API retries exhausted with no successful response.") @@ -7346,9 +3359,8 @@ def run_conversation( assistant_message = normalized finish_reason = normalized.finish_reason - # Normalize content to string — some OpenAI-compatible servers - # (llama-server, etc.) return content as a dict or list instead - # of a plain string, which crashes downstream .strip() calls. + # Some OpenAI-compatible servers (llama-server) return content as dict/list, + # which crashes downstream .strip(); normalize to str. if assistant_message.content is not None and not isinstance(assistant_message.content, str): raw = assistant_message.content if isinstance(raw, dict): @@ -7368,12 +3380,8 @@ def run_conversation( assistant_message.content = str(raw) # ── Agent-as-provider projection ────────────────────────────── - # A provider that IS an agent ran its own tools inside its own - # session before we got here: splice that work into the transcript - # as completed call/result rows and tick the skill-review nudge - # with the iterations Hermes never saw. Appended before this turn's - # assistant message, so the order reads call → result → answer. - # No-op for ordinary providers; see agent/provider_projection.py. + # Splice the provider-agent's own tool work in as call/result rows before + # this turn's assistant message; no-op for ordinary providers. splice_provider_projection(agent, response, messages) try: @@ -7402,11 +3410,9 @@ def run_conversation( api_duration=api_duration, started_at=api_start_time, ended_at=_api_ended_at, - # First received stream chunk timestamp (epoch seconds), set by - # interruptible_streaming_api_call from its per-attempt - # stream diagnostics; None when the response was not - # streamed or no chunk arrived. TTFB = - # first_chunk_at - started_at. + # First stream chunk time (epoch s) from + # interruptible_streaming_api_call; None if not streamed / no + # chunk. TTFB = first_chunk_at - started_at. first_chunk_at=getattr( agent, "_last_api_first_chunk_at", None ), @@ -7490,165 +3496,17 @@ def run_conversation( agent._incomplete_scratchpad_retries = 0 if agent.api_mode == "codex_responses" and finish_reason == "incomplete": - agent._codex_incomplete_retries += 1 - - interim_msg = agent._build_assistant_message(assistant_message, finish_reason) - interim_has_content = bool((interim_msg.get("content") or "").strip()) - interim_has_reasoning = bool(interim_msg.get("reasoning", "").strip()) if isinstance(interim_msg.get("reasoning"), str) else False - interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items")) - interim_has_codex_message_items = bool(interim_msg.get("codex_message_items")) - - if ( - interim_has_content - or interim_has_reasoning - or interim_has_codex_reasoning - or interim_has_codex_message_items - ): - last_msg = messages[-1] if messages else None - # Duplicate detection: compare only visible content - # (content + reasoning). Opaque provider state - # (encrypted reasoning items, message item ids/phases) - # drifts per continuation even when the visible output - # is identical, so including it in the comparison defeats - # dedup and causes message storms (#52711). - last_interim_visible = ( - agent._interim_assistant_visible_text(last_msg) - if isinstance(last_msg, dict) - else "" - ) - current_interim_visible = agent._interim_assistant_visible_text(interim_msg) - if last_interim_visible or current_interim_visible: - same_visible_output = last_interim_visible == current_interim_visible - else: - # Preserve the existing reasoning-only behavior when - # neither response has text eligible for interim delivery. - same_visible_output = ( - (last_msg.get("content") or "") == (interim_msg.get("content") or "") - and (last_msg.get("reasoning") or "") == (interim_msg.get("reasoning") or "") - ) if isinstance(last_msg, dict) else False - visible_duplicate = ( - isinstance(last_msg, dict) - and last_msg.get("role") == "assistant" - and last_msg.get("finish_reason") == "incomplete" - and same_visible_output - ) - if visible_duplicate: - # Update replay state in-place so the latest provider - # payload is preserved without re-emitting identical - # user-visible commentary. - for _key in ( - "content", - "reasoning", - "reasoning_content", - "reasoning_details", - "codex_reasoning_items", - "codex_message_items", - ): - if _key in interim_msg: - if _key == "codex_reasoning_items": - # Merge instead of overwrite: a native - # compaction checkpoint captured on the - # earlier incomplete response is the only - # copy — the continuation won't re-emit - # it. See merge_interim_reasoning_items. - from agent.native_compaction import ( - merge_interim_reasoning_items, - ) - last_msg[_key] = merge_interim_reasoning_items( - last_msg.get(_key), interim_msg[_key] - ) - else: - last_msg[_key] = interim_msg[_key] - else: - append_message(messages, interim_msg) - agent._emit_interim_assistant_message(interim_msg) - - if agent._codex_incomplete_retries < 3: - # When the interim message has nothing the Responses - # input converter will replay (no visible content, no - # encrypted reasoning items, no replayable message - # items — plain-text reasoning only), a bare retry is - # byte-identical to the request that just came back - # incomplete and fails the same way every time - # (observed with grok-4.20 on xai-oauth, whose - # reasoning items lack encrypted_content). Append a - # user-role nudge so the retry actually differs and - # explicitly asks for the final answer. - interim_replayable = ( - interim_has_content - or interim_has_codex_reasoning - or interim_has_codex_message_items - ) - # A replayable interim is not the same thing as a retry - # that DIFFERS. When the interim replays but carries no - # new instruction, the continuation is byte-identical to - # the request that just failed and returns the same empty - # response until the budget is gone. Live case (gpt-5.6 - # on the Codex backend, Aug 2026): the model answers with - # a server-side ``compaction`` checkpoint and no message. - # The checkpoint lands in ``codex_reasoning_items``, so - # ``interim_replayable`` is True and no nudge is added — - # meanwhile the checkpoint makes the wire converter prune - # every pre-checkpoint item, so all three attempts send - # the same checkpoint + retained user messages and end on - # an empty assistant turn with nothing to answer. The - # provider's own prefix cache reports 99-100% on the - # repeats, and the turn dies with "Codex response - # remained incomplete after 3 continuation attempts", - # losing the whole turn's work. - # - # One bare retry is still worth trying (the model often - # just needs another turn). Once THAT has also come back - # incomplete, a bare retry is proven not to work for this - # turn, so every remaining attempt carries the nudge. - if not interim_replayable or agent._codex_incomplete_retries >= 2: - _last_msg = messages[-1] if messages else None - _already_nudged = ( - isinstance(_last_msg, dict) - and _last_msg.get("role") == "user" - and _last_msg.get("content") == _CODEX_INCOMPLETE_NUDGE - ) - # Alternation guard: the nudge is a user-role message, - # so it may only follow an assistant message. When the - # interim was too empty to append (no content AND no - # reasoning), the last message is still the prior - # user/tool turn — appending the nudge there would - # create a user→user / tool→user sequence that strict - # providers reject. - _last_is_assistant = ( - isinstance(_last_msg, dict) - and _last_msg.get("role") == "assistant" - ) - if not _already_nudged and _last_is_assistant: - append_message(messages, { - "role": "user", - "content": _CODEX_INCOMPLETE_NUDGE, - }) - if not agent.quiet_mode: - agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({agent._codex_incomplete_retries}/3)") - # Surface the continuation on the live spinner/status line - # (CLI/TUI/Desktop) and gateway heartbeat: each of these - # retries can spend minutes waiting on the provider, and - # without a distinct notice the user only sees a generic - # thinking spinner ("infinite thinking", #64434). - agent._emit_wait_notice( - f"↻ model returned reasoning with no final answer — " - f"asking it to continue " - f"({agent._codex_incomplete_retries}/3)" - ) - agent._session_messages = messages - continue - - agent._codex_incomplete_retries = 0 - agent._persist_session(messages, conversation_history) - return { - "final_response": "Codex response remained incomplete after 3 continuation attempts", - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": "Codex response remained incomplete after 3 continuation attempts", - } + _codex_result = continue_codex_incomplete( + agent, + assistant_message, + finish_reason, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + ) + if _codex_result is not None: + return _codex_result + continue elif hasattr(agent, "_codex_incomplete_retries"): agent._codex_incomplete_retries = 0 @@ -7663,208 +3521,20 @@ def run_conversation( args_preview = raw_args[:200] if isinstance(raw_args, str) else repr(raw_args)[:200] logging.debug("Tool call: %s with args: %s...", tc.function.name, args_preview) - # Uniquify duplicate tool-call ids BEFORE any downstream - # consumer (validation error paths, dispatch, history build, - # Responses item-id derivation). Models that reuse one id for - # different calls in a batch otherwise lose the later call's - # result: the pre-API sanitizer keeps only the first - # call/result pair per id. See _uniquify_tool_call_ids. - agent._uniquify_tool_call_ids(assistant_message.tool_calls) - - # Validate tool call names - detect model hallucinations - # Repair mismatched tool names before validating - for tc in assistant_message.tool_calls: - if tc.function.name not in agent.valid_tool_names: - repaired = agent._repair_tool_call(tc.function.name) - if repaired: - print(f"{agent.log_prefix}🔧 Auto-repaired tool name: '{tc.function.name}' -> '{repaired}'") - tc.function.name = repaired - invalid_tool_calls = [ - tc.function.name for tc in assistant_message.tool_calls - if tc.function.name not in agent.valid_tool_names - ] - # Mixed batch: at least one valid call alongside the invalid - # one(s). Degrading models (observed with gpt-5.6 at very - # large context) emit batches like 6 named calls + 1 - # blank-name call; voiding the whole turn throws away real - # work and, across the 3-strike budget, halts sessions that - # were still making progress. Instead: error-result ONLY the - # invalid calls (below, after dedup/cap guardrails) and let - # the valid ones execute. The strike counter only advances - # when a turn contains NO valid call, so a fully-degenerate - # model still halts at 3 while a mostly-coherent one keeps - # working. - _mixed_invalid_batch = bool(invalid_tool_calls) and any( - tc.function.name in agent.valid_tool_names - for tc in assistant_message.tool_calls + _tvv = validate_tool_calls( + agent, + assistant_message, + finish_reason, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + effective_task_id=effective_task_id, ) - if _mixed_invalid_batch: - agent._invalid_tool_retries = 0 - invalid_name = invalid_tool_calls[0] - invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name - _n_valid = sum( - 1 for tc in assistant_message.tool_calls - if tc.function.name in agent.valid_tool_names - ) - agent._buffer_vprint( - f"⚠️ Unknown tool '{invalid_preview}' in batch — erroring that call, " - f"executing {_n_valid} valid call(s)" - ) - elif invalid_tool_calls: - # Track retries for invalid tool calls - agent._invalid_tool_retries += 1 - - # Return helpful error to model — model can agent-correct next turn - invalid_name = invalid_tool_calls[0] - invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name - agent._buffer_vprint(f"⚠️ Unknown tool '{invalid_preview}' — sending error to model for agent-correction ({agent._invalid_tool_retries}/3)") - - if agent._invalid_tool_retries >= 3: - agent._flush_status_buffer() - agent._vprint(f"{agent.log_prefix}❌ Max retries (3) for invalid tool calls exceeded. Stopping as partial.", force=True) - agent._invalid_tool_retries = 0 - _final_response = f"Model generated invalid tool call: {invalid_preview}" - # Prior <3 retries (or an earlier successful tool batch) - # leave a tool-result tail. Closing it here matches - # interrupt aborts (#48879 / #52592) so the next user - # turn is not tool→user for strict providers. - close_interrupted_tool_sequence(messages, _final_response) - agent._persist_session(messages, conversation_history) - return { - "final_response": _final_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": _final_response - } - - assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) - append_message(messages, assistant_msg) - for tc in assistant_message.tool_calls: - _tc_name = tc.function.name - if _tc_name not in agent.valid_tool_names: - # See _invalid_tool_name_error_content for the - # blank-name anti-priming rationale (#47967). - content = _invalid_tool_name_error_content( - _tc_name, agent.valid_tool_names - ) - else: - content = "Skipped: another tool call in this turn used an invalid name. Please retry this tool call." - append_message(messages, { - "role": "tool", - "name": tc.function.name, - "tool_call_id": coalesce_tool_call_id(tc), - "content": content, - }) + _mixed_invalid_batch = _tvv.mixed_invalid_batch + if _tvv.action == "return": + return _tvv.result + if _tvv.action == "continue": continue - # Reset retry counter on successful tool call validation - agent._invalid_tool_retries = 0 - - # Validate tool call arguments are valid JSON - # Handle empty strings as empty objects (common model quirk) - invalid_json_args = [] - for tc in assistant_message.tool_calls: - args = tc.function.arguments - if isinstance(args, (dict, list)): - tc.function.arguments = json.dumps(args) - continue - if args is not None and not isinstance(args, str): - tc.function.arguments = str(args) - args = tc.function.arguments - # Treat empty/whitespace strings as empty object - if not args or not args.strip(): - tc.function.arguments = "{}" - continue - try: - json.loads(args) - except json.JSONDecodeError as e: - if ( - _mixed_invalid_batch - and tc.function.name not in agent.valid_tool_names - ): - # This call never executes — it gets an - # invalid-name error result below. Don't let its - # broken args trigger the whole-turn JSON retry. - continue - invalid_json_args.append((tc.function.name, str(e))) - - if invalid_json_args: - # Check if the invalid JSON is due to truncation rather - # than a model formatting mistake. Routers sometimes - # rewrite finish_reason from "length" to "tool_calls", - # hiding the truncation from the length handler above. - # Detect truncation: args that don't end with } or ] - # (after stripping whitespace) are cut off mid-stream. - _truncated = any( - not (tc.function.arguments or "").rstrip().endswith(("}", "]")) - for tc in assistant_message.tool_calls - if tc.function.name in {n for n, _ in invalid_json_args} - ) - if _truncated: - agent._vprint( - f"{agent.log_prefix}⚠️ Truncated tool call arguments detected " - f"(finish_reason={finish_reason!r}) — refusing to execute.", - force=True, - ) - agent._invalid_json_retries = 0 - agent._cleanup_task_resources(effective_task_id) - _final_response = "Response truncated due to output length limit" - # Same tool-tail close as interrupt / invalid-tool - # exhaustion — this path never reaches finalize_turn. - close_interrupted_tool_sequence(messages, _final_response) - agent._persist_session(messages, conversation_history) - return { - "final_response": _final_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "partial": True, - "error": _final_response, - } - - # Track retries for invalid JSON arguments - agent._invalid_json_retries += 1 - - tool_name, error_msg = invalid_json_args[0] - agent._buffer_vprint(f"⚠️ Invalid JSON in tool call arguments for '{tool_name}': {error_msg}") - - if agent._invalid_json_retries < 3: - agent._buffer_vprint(f"🔄 Retrying API call ({agent._invalid_json_retries}/3)...") - # Don't add anything to messages, just retry the API call - continue - else: - # Instead of returning partial, inject tool error results so the model can recover. - # Using tool results (not user messages) preserves role alternation. - agent._buffer_vprint("⚠️ Injecting recovery tool results for invalid JSON...") - agent._invalid_json_retries = 0 # Reset for next attempt - - # Append the assistant message with its (broken) tool_calls - recovery_assistant = agent._build_assistant_message(assistant_message, finish_reason) - append_message(messages, recovery_assistant) - - # Respond with tool error results for each tool call - invalid_names = {name for name, _ in invalid_json_args} - for tc in assistant_message.tool_calls: - if tc.function.name in invalid_names: - err = next(e for n, e in invalid_json_args if n == tc.function.name) - tool_result = ( - f"Error: Invalid JSON arguments. {err}. " - f"For tools with no required parameters, use an empty object: {{}}. " - f"Please retry with valid JSON." - ) - else: - tool_result = "Skipped: other tool call in this response had invalid JSON." - append_message(messages, { - "role": "tool", - "name": tc.function.name, - "tool_call_id": coalesce_tool_call_id(tc), - "content": tool_result, - }) - continue - - # Reset retry counter on successful JSON validation - agent._invalid_json_retries = 0 # ── Post-call guardrails ────────────────────────── assistant_message.tool_calls = agent._cap_delegate_task_calls( @@ -7874,11 +3544,9 @@ def run_conversation( assistant_message.tool_calls ) - # Mixed-batch invalid-name handling: collect the invalid - # calls now so the assistant message (built below) keeps - # EVERY call the model emitted — providers require each - # tool_call to have a matching tool result and vice versa — - # while only the valid subset is dispatched for execution. + # Collect invalid calls so the assistant message keeps EVERY emitted + # call (each tool_call needs a matching result) while only valid ones + # dispatch. _invalid_batch_calls = [] if _mixed_invalid_batch: _invalid_batch_calls = [ @@ -7890,11 +3558,9 @@ def run_conversation( turn_content = assistant_message.content or "" - # Some local tool-call templates emit a bare bracketed token - # (for example ``[memory]``) as assistant content alongside a - # function call. It is protocol scaffolding, not an answer. - # Persisting or caching it as visible content lets the empty - # post-tool fallback replay that token forever after compaction (#78148). + # A bare bracketed token (e.g. ``[memory]``) beside a function call is + # protocol scaffolding; persisting it lets the post-tool fallback replay + # it forever (#78148). if ( assistant_message.tool_calls and _STALE_MARKER_RE.fullmatch(turn_content.strip()) @@ -7906,9 +3572,8 @@ def run_conversation( turn_content = "" assistant_msg["content"] = "" - # Classify tools in this turn to determine if they are all housekeeping. - # This classification is needed regardless of whether the turn has visible content, - # because a substantive tool-only turn must invalidate any older housekeeping fallback. + # Classify tools regardless of visible content: a substantive tool-only + # turn must invalidate any older housekeeping fallback. _HOUSEKEEPING_TOOLS = frozenset({ "memory", "todo_list", "skill_manage", "session_search", }) @@ -7917,31 +3582,22 @@ def run_conversation( for tc in assistant_message.tool_calls ) - # If this turn has substantive tools (non-housekeeping), clear any older fallback. - # Prevents a two-turn-old housekeeping narration from being treated as if it belonged - # to the immediately preceding substantive tool turn. + # Substantive tools clear any older fallback so a two-turn-old + # housekeeping narration isn't attributed to the preceding tool turn. if assistant_message.tool_calls and not _all_housekeeping: agent._last_content_with_tools = None agent._last_content_tools_all_housekeeping = False - # Also clear the mute flag: a prior housekeeping turn may - # have set _mute_post_response (line ~4667), and the - # substantive tools in THIS turn should produce visible - # progress output. Without this reset, _vprint suppresses - # tool progress until the no-tool-call branch clears it at - # line ~4834 — after all tools have finished. + # Also clear the mute flag a prior housekeeping turn may have set, + # else _vprint suppresses this turn's tool progress until the + # no-tool-call branch clears it. agent._mute_post_response = False - # If this turn has both content AND tool_calls, capture the content - # as a fallback final response. Common pattern: model delivers its - # answer and calls memory/skill tools as a side-effect in the same - # turn. If the follow-up turn after tools is empty, we use this. + # Content + tool_calls in one turn: keep the content as a fallback final + # response in case the follow-up turn after tools is empty. if turn_content and agent._has_content_after_think_block(turn_content): agent._last_content_with_tools = turn_content - # Only mute subsequent output when EVERY tool call in - # this turn is post-response housekeeping (memory, todo, - # skill_manage, etc.). If any substantive tool is present - # (search_files, read_file, write_file, terminal, ...), - # keep output visible so the user sees progress. + # Mute only when EVERY tool call is post-response housekeeping + # (memory, todo, skill_manage); substantive tools keep output on. agent._last_content_tools_all_housekeeping = _all_housekeeping if _all_housekeeping and agent._has_stream_consumers(): agent._mute_post_response = True @@ -7961,23 +3617,15 @@ def run_conversation( messages.pop() _had_prefill = True - # Reset prefill counter when tool calls follow a prefill - # recovery. Without this, the counter accumulates across - # the whole conversation — a model that intermittently - # empties (empty → prefill → tools → empty → prefill → - # tools) burns both prefill attempts and the third empty - # gets zero recovery. Resetting here treats each tool- - # call success as a fresh start. + # Tool calls after a prefill recovery reset the prefill counter, so + # each tool-call success is a fresh start, not a cumulative burn. if _had_prefill: agent._thinking_prefill_retries = 0 agent._empty_content_retries = 0 - # Successful tool execution — reset the post-tool nudge - # flag so it can fire again if the model goes empty on - # a LATER tool round. + # Re-arm the post-tool nudge so it can fire on a LATER tool round. agent._post_tool_empty_retried = False - # A landed tool call means any earlier dropped-tool-call stall - # was recovered — refresh that budget too so it guards each - # stall independently rather than capping the whole run. + # A landed tool call recovers any dropped-tool-call stall; refresh that + # budget so it guards each stall independently, not the whole run. agent._dropped_toolcall_retries = 0 previous_msg = messages[-1] if messages else None @@ -7996,11 +3644,8 @@ def run_conversation( ) append_message(messages, assistant_msg) - # Mixed batch: error-result the invalid calls and strip them - # from the execution set. The assistant message above keeps - # all calls (each gets a matching tool result — the invalid - # ones get theirs here, the valid ones during execution), so - # provider-side tool_call/result pairing stays intact. + # Mixed batch: error-result invalid calls and drop them from execution. + # The assistant message keeps all calls so tool_call/result pairs hold. if _invalid_batch_calls: for tc in _invalid_batch_calls: append_message(messages, { @@ -8018,10 +3663,8 @@ def run_conversation( _tool_turn_persisted = None try: - # Persist the assistant tool-call turn before any tool - # side effects run. If a destructive tool restarts or - # terminates Hermes mid-turn, resume logic still sees the - # exact tool-call block that already executed. + # Persist the tool-call turn before any tool side effects so resume + # sees the executed block if a destructive tool restarts Hermes. _tool_turn_persisted = agent._flush_messages_to_session_db( messages, conversation_history ) @@ -8039,12 +3682,9 @@ def run_conversation( ) if _tool_turn_persisted is False: - # The canonical append failed. Do not project the row or - # run side-effecting tools from state that exists only in - # this process. Breaking also avoids retrying the same - # unpersisted turn until the iteration budget is exhausted. - # The flush may have classified the cause internally; if - # nothing was recorded, the cause is genuinely unknown. + # Canonical append failed: never project the row or run tools from + # process-only state; break rather than retry the unpersisted turn. + # If the flush recorded no cause, the cause is genuinely unknown. if getattr(agent, "_last_persistence_error_cause", None) is None: agent._last_persistence_error_cause = "unknown" _turn_exit_reason = "session_persistence_failed" @@ -8052,18 +3692,14 @@ def run_conversation( failed = True break - # A UI must never observe an assistant/tool-call row that is - # still only an ephemeral in-memory projection. Emit interim - # commentary only after the canonical SessionDB append above. + # A UI must never observe an assistant/tool-call row that is only an + # in-memory projection: emit interim commentary after the DB append. if not duplicate_previous_interim: agent._emit_interim_assistant_message(assistant_msg) - # Close any open streaming display (response box, reasoning - # box) before tool execution begins. Intermediate turns may - # have streamed early content that opened the response box; - # flushing here prevents it from wrapping tool feed lines. - # Only signal the display callback — TTS (_stream_callback) - # should NOT receive None (it uses None as end-of-stream). + # Flush open streaming boxes before tools so early content doesn't wrap + # tool feed lines. Display callback only — TTS (_stream_callback) must + # NOT receive None (its end-of-stream marker). if agent.stream_delta_callback: try: agent.stream_delta_callback(None) @@ -8073,9 +3709,8 @@ def run_conversation( agent._execute_tool_calls(assistant_message, messages, effective_task_id, api_call_count) if getattr(agent, "_incremental_persistence_failed", False): - # A tool result could not be made canonical. Do not send - # the in-memory result back to the model or project any - # later events from this turn. + # Tool result could not be made canonical: never send the in-memory + # result to the model or project later events from this turn. _turn_exit_reason = "session_persistence_failed" final_response = "" failed = True @@ -8089,11 +3724,8 @@ def run_conversation( f"⚠️ Tool guardrail halted {decision.tool_name}: {decision.code}" ) append_message(messages, {"role": "assistant", "content": final_response}) - # Emit the halt message to the client so it's not - # indistinguishable from a crash. The stream display - # was flushed (callback(None)) before tool execution, - # but the callback is still alive — fire the text - # through it so SSE/TUI clients see the explanation. + # Emit the halt message so it isn't mistaken for a crash; the stream + # callback is still alive, so SSE/TUI clients see the explanation. if final_response: agent._safe_print(f"\n{final_response}\n") if agent.stream_delta_callback: @@ -8104,609 +3736,90 @@ def run_conversation( pass break - # Reset per-turn retry counters after successful tool - # execution so a single truncation doesn't poison the - # entire conversation. + # Reset per-turn retry counters so one truncation can't poison the turn. truncated_tool_call_retries = 0 - # Signal that a paragraph break is needed before the next - # streamed text. We don't emit it immediately because - # multiple consecutive tool iterations would stack up - # redundant blank lines. Instead, _fire_stream_delta() - # will prepend a single "\n\n" the next time real text - # arrives. + # Defer the paragraph break: _fire_stream_delta() prepends one "\n\n" + # when real text arrives, so tool iterations don't stack blank lines. agent._stream_needs_break = True - # Refund the iteration if the ONLY tool(s) called were - # execute_code (programmatic tool calling). These are - # cheap RPC-style calls that shouldn't eat the budget. + # Refund the iteration when the ONLY tool was execute_code (programmatic + # tool calling) — cheap RPC-style calls shouldn't eat the budget. _tc_names = {tc.function.name for tc in assistant_message.tool_calls} if _tc_names == {"execute_code"}: agent.iteration_budget.refund() - # Use real token counts from the API response to decide - # compression. prompt_tokens + completion_tokens is the - # actual context size the provider reported plus the - # assistant turn — a tight lower bound for the next prompt. - # Tool results appended above aren't counted yet, but the - # threshold (default 50%) leaves ample headroom; if tool - # results push past it, the next API call will report the - # real total and trigger compression then. - # - # If last_prompt_tokens is 0 (stale after API disconnect - # or provider returned no usage data), fall back to rough - # estimate to avoid missing compression. Without this, - # a session can grow unbounded after disconnects because - # should_compress(0) never fires. (#2153) - _compressor = agent.context_compressor - if _compressor.last_prompt_tokens > 0: - # Only use prompt_tokens — completion/reasoning - # tokens don't consume context window space. - # Thinking models (GLM-5.1, QwQ, DeepSeek R1) - # inflate completion_tokens with reasoning, - # causing premature compression. (#12026) - _real_tokens = _compressor.last_prompt_tokens - elif _compressor.last_prompt_tokens == -1: - # Compression just ran and no API-reported prompt count - # has arrived yet. Avoid treating a schema-heavy rough - # post-compression estimate as real context pressure. - _real_tokens = 0 - else: - # Include tool schemas — with 50+ tools enabled - # these add 20-30K tokens the messages-only - # estimate misses, which can skip compression - # past the configured threshold (#14695). - # Route-aware (#96995/#97602 class): on a compacted - # native-Codex session the generic durable-history - # figure overstates the wire and would false-trigger - # compression here exactly like the pre-API guard — - # this fallback runs precisely when no provider usage - # is available (post-disconnect / gateway restart), - # the unanchored case from #97602's repro. - _real_tokens = _midturn_request_pressure_tokens( - agent, - messages, - active_system_prompt or "", - estimate_request_tokens_rough( - messages, tools=agent.tools or None - ), - ) - - if ( - agent.compression_enabled - and compression_attempts < max_compression_attempts - and _compressor.should_compress(_real_tokens) - ): - compression_attempts += 1 - # Compression is actually running (block cleared / was - # never blocked) — reset the blocked-overflow warning - # dedup so a future blocked-over-threshold turn can warn - # again (silent-overflow fix #62625). - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. - _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) - if callable(_clear_warn): - _clear_warn() - agent._safe_print(" ⟳ compacting context…") - _post_tool_input = messages - # Route the overhead-aware _real_tokens (computed above) into compression, not - # the bare last_prompt_tokens — which is 0 in the no-usage fallback, hiding the - # true request size from the engine's overflow guard (upstream PR #77169 review). - messages, active_system_prompt = agent._compress_context( - messages, system_message, - approx_tokens=_real_tokens, - task_id=effective_task_id, - ) - if ( - messages is _post_tool_input - and compression_skipped_due_to_lock(agent) - ): - # #69870 lock-skip: this pass no-oped because another - # path holds the session's compression lock — a - # temporary defer, not evidence about compressibility. - # Refund the attempt so a lock-loser tool loop does not - # burn the shared per-turn budget toward - # compression_exhausted (#9893/#35809). - compression_attempts -= 1 - else: - conversation_history = conversation_history_after_compression( - agent, messages, conversation_history - ) - if _should_skip_model_call_for_reference_handoff( - messages, user_message - ): - logger.info( - "Skipping post-tool compaction model call: " - "reference-only handoff would be the sole " - "active user turn (#80622)" - ) - if not final_response: - final_response = _HANDOFF_SKIP_FINAL_RESPONSE - _turn_exit_reason = "compaction_handoff_not_actionable" - break - elif agent.compression_enabled: - # Over threshold but compression is blocked (summary-LLM - # cooldown or anti-thrashing). Surface a deduped warning so - # the user isn't left with a silently growing context that - # eventually hits the hard provider limit. Mirrors the - # turn-context preflight guard (silent-overflow fix #62625). - _block_reason = None - _info = getattr(_compressor, "should_compress_info", None) - if _info is not None: - try: - _block_reason = _info(_real_tokens)[1] - except Exception: - _block_reason = None - if _block_reason: - agent._warn_context_overflow_blocked( - _block_reason, - _real_tokens, - int(getattr(_compressor, "threshold_tokens", 0) or 0), - ) - # Proactive tool-result prune: reclaim re-sent history on - # large-window models long before should_compress() (≈50% of - # the window) would ever fire. Deterministic, no LLM call; - # protects the recent tail. No-op unless proactive_prune_tokens - # is configured and _real_tokens is above it — and even then - # the prune only commits when it reclaims at least - # proactive_prune_min_reclaim_tokens, so prompt-cache breaks - # stay episodic like compression's (the one sanctioned cache - # break) instead of firing every tool iteration. See - # ContextCompressor.prune_tool_results_only. - # getattr guard: plugin context engines predating the hook and - # minimal test doubles (SimpleNamespace compressors) lack the - # method — treat absence as a no-op. - _prune = getattr(_compressor, "prune_tool_results_only", None) - if callable(_prune): - try: - _pruned_msgs, _pruned_n = _prune( - messages, current_tokens=_real_tokens - ) - except Exception: - logger.debug( - "proactive tool-result prune failed; skipping", - exc_info=True, - ) - _pruned_msgs, _pruned_n = messages, 0 - # Standard no-op caller contract: only commit when the - # engine returned a NEW list object with a non-zero count. - if _pruned_n and _pruned_msgs is not messages: - # Do NOT rebuild conversation_history here. The compressor - # atomically rewrites the active transcript with the durable - # rearm threshold, then stamps every returned row with - # _DB_PERSISTED_MARKER, so the marker-based flush dedup (see - # _flush_messages_to_session_db) prevents duplicate writes. - # Calling - # conversation_history_after_compression (a compaction-only - # helper keyed on the _last_compaction_in_place flag) would be - # a no-op at best, and on a stale in-place flag could seed - # this turn's fresh, not-yet-persisted rows into history_ids - # and skip writing them. - messages = _pruned_msgs + _ptc = compress_after_tool_results( + agent, + messages=messages, + system_message=system_message, + user_message=user_message, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + compression_attempts=compression_attempts, + max_compression_attempts=max_compression_attempts, + effective_task_id=effective_task_id, + final_response=final_response, + turn_exit_reason=_turn_exit_reason, + ) + messages = _ptc.messages + active_system_prompt = _ptc.active_system_prompt + conversation_history = _ptc.conversation_history + compression_attempts = _ptc.compression_attempts + final_response = _ptc.final_response + _turn_exit_reason = _ptc.turn_exit_reason + if _ptc.end_turn: + break # Save session log incrementally (so progress is visible even if interrupted) agent._session_messages = messages - # Touch activity before continuing so the gateway's - # inactivity monitor never sees a stale timestamp - # between tool completion and the start of the next - # API call. Without this, a tool-call result (which - # takes ~0s to process) followed by slow post-tool - # processing (compression, persist) and a slow - # follow-up API call can exceed the gateway inactivity - # timeout (HERMES_AGENT_TIMEOUT, default 1800s) and the - # gateway kills the session before the next activity - # touch fires (#69559, #69131). + # Touch activity so slow post-tool work plus a slow follow-up API call + # can't exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT). agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}") # Continue loop for next response continue else: - # No tool calls - this is the final response. - # (Dropped tool-call recovery — finish_reason=="tool_calls" with - # an empty tool_calls array — is handled at the finalization - # chokepoint below, after final_msg is built, so it catches - # every path that reaches turn finalization, not just this one.) + # No tool calls — final response. (Dropped tool-call recovery lives at + # the finalization chokepoint below so it catches every path.) final_response = assistant_message.content or "" - # Fix: unmute output when entering the no-tool-call branch - # so the user can see empty-response warnings and recovery - # status messages. _mute_post_response was set during a - # prior housekeeping tool turn and should not silence the - # final response path. + # Unmute: _mute_post_response from a housekeeping tool turn must not + # silence empty-response warnings on the final response path. agent._mute_post_response = False # Check if response only has think block with no actual content after it if not agent._has_content_after_think_block(final_response): - # ── Partial stream recovery ───────────────────── - # If content was already streamed to the user before - # the connection died, use it as the final response - # instead of falling through to prior-turn fallback - # or wasting API calls on retries. - _partial_streamed = ( - getattr(agent, "_current_streamed_assistant_text", "") or "" + _ev = recover_empty_response( + agent, + assistant_message, + response, + finish_reason, + final_response=final_response, + messages=messages, + api_messages=api_messages, + conversation_history=conversation_history, + active_system_prompt=active_system_prompt, + api_call_count=api_call_count, + turn_exit_reason=_turn_exit_reason, + preflight_compression_blocked=_preflight_compression_blocked, ) - if agent._has_content_after_think_block(_partial_streamed): - _turn_exit_reason = "partial_stream_recovery" - _recovered = agent._strip_think_blocks(_partial_streamed).strip() - logger.info( - "Partial stream content delivered (%d chars) " - "— using as final response", - len(_recovered), - ) - agent._emit_status( - "↻ Stream interrupted — using delivered content " - "as final response" - ) - final_response = _recovered - # Streaming delivered a fragment, not a confirmed - # final preview. Leave response_previewed false so - # gateway fallback delivery can send the recovered - # text plus the abnormal-turn explanation. - agent._response_was_previewed = False + final_response = _ev.final_response + _turn_exit_reason = _ev.turn_exit_reason + active_system_prompt = _ev.active_system_prompt + _preflight_compression_blocked = _ev.preflight_compression_blocked + if _ev.action == "return": + return _ev.result + if _ev.action == "break": break - - # If the previous turn already delivered real content alongside - # HOUSEKEEPING tool calls (e.g. "You're welcome!" + memory save), - # the model has nothing more to say. Use the earlier content - # immediately instead of wasting API calls on retries. - # NOTE: Only use this shortcut when ALL tools in that turn were - # housekeeping (memory, todo, etc.). When substantive tools - # were called (terminal, search_files, etc.), the content was - # likely mid-task narration ("I'll scan the directory...") and - # the empty follow-up means the model choked — let the - # post-tool nudge below handle that instead of exiting early. - fallback = getattr(agent, '_last_content_with_tools', None) - if fallback and getattr(agent, '_last_content_tools_all_housekeeping', False): - _turn_exit_reason = "fallback_prior_turn_content" - logger.info("Empty follow-up after tool calls — using prior turn content as final response") - agent._emit_status("↻ Empty response after tool calls — using earlier content as final answer") - agent._last_content_with_tools = None - agent._last_content_tools_all_housekeeping = False - agent._empty_content_retries = 0 - # Do NOT modify the assistant message content — the - # old code injected "Calling the X tools..." which - # poisoned the conversation history. Just use the - # fallback text as the final response and break. - final_response = agent._strip_think_blocks(fallback).strip() - agent._response_was_previewed = True - break - - # ── Post-tool-call empty response nudge ─────────── - # The model returned empty after executing tool calls. - # This covers two cases: - # (a) No prior-turn content at all — model went silent - # (b) Prior turn had content + SUBSTANTIVE tools (the - # fallback above was skipped because the content - # was mid-task narration, not a final answer) - # Instead of giving up, nudge the model to continue by - # appending a user-level hint. This is the #9400 case: - # weaker models (mimo-v2-pro, GLM-5, etc.) sometimes - # return empty after tool results instead of continuing - # to the next step. One retry with a nudge usually - # fixes it. - _prior_was_tool = any( - m.get("role") == "tool" - for m in messages[-5:] # check recent messages - ) - # Detect Qwen3/Ollama-style in-content thinking blocks. - # Ollama puts in the content field (not in - # reasoning_content), so _has_structured below would - # miss it. We check here so thinking-only responses - # after tool calls route to prefill instead of nudge. - _has_inline_thinking = bool( - re.search( - r'||', - final_response or "", - re.IGNORECASE, - ) - ) - if ( - _prior_was_tool - and not getattr(agent, "_post_tool_empty_retried", False) - and not _has_inline_thinking # thinking model still working — let prefill handle - ): - agent._post_tool_empty_retried = True - # Clear stale narration so it doesn't resurface - # on a later empty response after the nudge. - agent._last_content_with_tools = None - agent._last_content_tools_all_housekeeping = False - logger.info( - "Empty response after tool calls — nudging model " - "to continue processing" - ) - agent._buffer_status( - "⚠️ Model returned empty after tool calls — " - "nudging to continue" - ) - # Append the empty assistant message first so the - # message sequence stays valid: - # tool(result) → assistant("(empty)") → user(nudge) - # Without this, we'd have tool → user which most - # APIs reject as an invalid sequence. - _nudge_msg = agent._build_assistant_message(assistant_message, finish_reason) - _nudge_msg["content"] = "(empty)" - _nudge_msg["_empty_recovery_synthetic"] = True - append_message(messages, _nudge_msg) - append_message(messages, { - "role": "user", - "content": _EMPTY_TOOL_RESPONSE_NUDGE, - "_empty_recovery_synthetic": True, - }) - continue - - # ── Thinking-only prefill continuation ────────── - # The model produced structured reasoning (via API - # fields) but no visible text content. Rather than - # giving up, append the assistant message as-is and - # continue — the model will see its own reasoning - # on the next turn and produce the text portion. - # Inspired by clawdbot's "incomplete-text" recovery. - # Also covers Qwen3/Ollama in-content blocks - # (detected above as _has_inline_thinking). - _has_structured = bool( - getattr(assistant_message, "reasoning", None) - or getattr(assistant_message, "reasoning_content", None) - or getattr(assistant_message, "reasoning_details", None) - or _has_inline_thinking - ) - if _has_structured and agent._thinking_prefill_retries < 2: - agent._thinking_prefill_retries += 1 - logger.info( - "Thinking-only response (no visible content) — " - "prefilling to continue (%d/2)", - agent._thinking_prefill_retries, - ) - agent._buffer_status( - f"↻ Thinking-only response — prefilling to continue " - f"({agent._thinking_prefill_retries}/2)" - ) - interim_msg = agent._build_assistant_message( - assistant_message, "incomplete" - ) - interim_msg["_thinking_prefill"] = True - append_message(messages, interim_msg) - agent._session_messages = messages - continue - - # ── Empty response retry ────────────────────── - # Model returned nothing usable. Retry up to 3 - # times before attempting fallback. This covers - # both truly empty responses (no content, no - # reasoning) AND reasoning-only responses after - # prefill exhaustion — models like mimo-v2-pro - # always populate reasoning fields via OpenRouter, - # so the old `not _has_structured` guard blocked - # retries for every reasoning model after prefill. - _truly_empty = not agent._strip_think_blocks( - final_response - ).strip() - _prefill_exhausted = ( - _has_structured - and agent._thinking_prefill_retries >= 2 - ) - _empty_candidate = _truly_empty and ( - not _has_structured or _prefill_exhausted - ) - if _empty_candidate: - # NS-503: every empty attempt re-sends the full - # conversation input at full price. Record the - # attempt (usage/finish_reason signature) so - # deterministic empties — e.g. unsignaled - # provider refusals with zero output tokens — - # stop burning paid retries reproducing the - # same empty. Fails open: missing usage or - # any generated tokens keep the full budget. - _empty_guard.record_empty_attempt( - agent, - finish_reason=finish_reason, - response=response, - ) - _empty_retry_budget = ( - _empty_guard.empty_retry_budget(agent, response) - if _empty_candidate - else _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET - ) - _deterministic_empty = _empty_candidate and ( - _empty_guard.deterministic_empty(agent) - ) - if ( - _empty_candidate - and agent._empty_content_retries < _empty_retry_budget - and not _deterministic_empty - ): - agent._empty_content_retries += 1 - wait_time = jittered_backoff( - agent._empty_content_retries, - base_delay=5.0, - max_delay=60.0, - ) - logger.warning( - "Empty response (no content or reasoning) — " - "retry %d/%d in %.1fs (model=%s)", - agent._empty_content_retries, - _empty_retry_budget, wait_time, agent.model, - ) - _budget_note = ( - " — high-cost request, reduced retry budget" - if _empty_retry_budget < _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET - else "" - ) - agent._buffer_status( - f"⚠️ Empty response from model — retrying " - f"({agent._empty_content_retries}/{_empty_retry_budget}) " - f"in {wait_time:.0f}s{_budget_note}" - ) - # Sleep in small increments to stay responsive to interrupts - sleep_end = time.time() + wait_time - _backoff_touch_counter = 0 - while time.time() < sleep_end: - if agent._interrupt_requested: - agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during empty-response retry wait, aborting.", force=True) - _interrupt_text = ( - f"Operation interrupted: retrying empty response from model " - f"(retry {agent._empty_content_retries}/{_empty_retry_budget})." - ) - close_interrupted_tool_sequence(messages, _interrupt_text) - agent._persist_session(messages, conversation_history) - agent.clear_interrupt() - return { - "final_response": _interrupt_text, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "interrupted": True, - } - time.sleep(0.2) - _backoff_touch_counter += 1 - if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s - agent._touch_activity( - f"empty response retry backoff ({agent._empty_content_retries}/{_empty_retry_budget}), " - f"{int(sleep_end - time.time())}s remaining" - ) - continue - - if _truly_empty and _deterministic_empty: - logger.warning( - "Deterministic empty response detected " - "(consecutive zero-output completions, " - "model=%s provider=%s finish_reason=%s) — " - "skipping remaining retries", - agent.model, agent.provider, finish_reason, - ) - agent._buffer_status( - "⚠️ Model is deterministically returning empty " - "(zero output tokens) — skipping further retries " - "to avoid repeat charges" - ) - - # ── Exhausted retries — try fallback provider ── - # Before giving up with "(empty)", attempt to - # switch to the next provider in the fallback - # chain. This covers the case where a model - # (e.g. GLM-4.5-Air) consistently returns empty - # due to context degradation or provider issues. - if _truly_empty and agent._fallback_chain: - logger.warning( - "Empty response after %d retries — " - "attempting fallback (model=%s, provider=%s)", - agent._empty_content_retries, agent.model, - agent.provider, - ) - agent._buffer_status( - "⚠️ Model returning empty responses — " - "switching to fallback provider..." - ) - if agent._try_activate_fallback(): - active_system_prompt = _sync_failover_system_message( - agent, api_messages, active_system_prompt) - agent._empty_content_retries = 0 - agent._buffer_status( - f"↻ Switched to fallback: {agent.model} " - f"({agent.provider})" - ) - logger.info( - "Fallback activated after empty responses: " - "now using %s on %s", - agent.model, agent.provider, - ) - # This site sits directly in the OUTER iteration - # loop (not the retry loop), so `continue` already - # restarts the iteration and re-runs the pre-API - # preflight against the fallback's context window - # (#84733). A `break` here would exit the outer - # loop and end the turn without ever calling the - # fallback. Clear the preflight block so the - # re-run isn't skipped. - _preflight_compression_blocked = False - continue - - # Exhausted retries and fallback chain (or no - # fallback configured). Fall through to the - # "(empty)" terminal. - # Surface the buffered retry/fallback trace so the - # user can see what was attempted before "(empty)". - # NS-503: if we know roughly what the empty streak - # cost (each attempt re-billed the full input), say - # so — an unexplained charge for "no answer" is the - # core of the complaint. - _streak_cost = _empty_guard.streak_cost_usd(agent) - if _streak_cost is not None: - agent._buffer_status( - f"ℹ️ Estimated cost of these empty attempts: " - f"~${_streak_cost:.2f} (input tokens are billed " - f"per attempt even when no answer is produced)" - ) - agent._flush_status_buffer() - _turn_exit_reason = "empty_response_exhausted" - reasoning_text = agent._extract_reasoning(assistant_message) - agent._drop_trailing_empty_response_scaffolding(messages) - assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) - assistant_msg["content"] = "(empty)" - # This is a user-facing failure sentinel for the gateway, - # not real assistant content. Persisting it makes later - # "continue" turns replay assistant("(empty)") as if it - # were a meaningful model response, which can keep long - # tool-heavy sessions stuck in empty-response loops. - assistant_msg["_empty_terminal_sentinel"] = True - append_message(messages, assistant_msg) - - if reasoning_text: - reasoning_preview = reasoning_text[:500] + "..." if len(reasoning_text) > 500 else reasoning_text - logger.warning( - "Reasoning-only response (no visible content) " - "after exhausting retries and fallback. " - "Reasoning: %s", reasoning_preview, - ) - agent._emit_status( - "⚠️ Model produced reasoning but no visible " - "response after all retries. Returning empty." - ) - else: - logger.warning( - "Empty response (no content or reasoning) " - "after %d retries. No fallback available. " - "model=%s provider=%s", - agent._empty_content_retries, agent.model, - agent.provider, - ) - agent._emit_status( - "❌ Model returned no content after all retries" - + (" and fallback attempts." if agent._fallback_chain else - ". No fallback providers configured.") - ) - - # Deliver a labeled reasoning excerpt instead of a bare - # "(empty)" when the model DID think but never produced - # visible text. This is delivery-only: the persisted - # assistant message above keeps the "(empty)" sentinel - # (its replay semantics prevent empty-response loops), - # and raw chain-of-thought is never promoted to a normal - # answer earlier in the ladder — prefill continuation, - # empty-content retries, and provider fallback all run - # first. Only at this terminal, where the alternative is - # returning nothing, is showing the model's own reasoning - # (clearly labeled as such) strictly more useful. - # Idea credit: PR #48795 (@ligl0325). - if reasoning_text: - final_response = ( - "⚠️ The model produced only internal reasoning and " - "no final answer, despite retries" - + (" and fallback" if agent._fallback_chain else "") - + ". Its last reasoning, which may contain the " - "answer:\n\n" + reasoning_preview - ) - else: - final_response = "(empty)" - break + continue # Reset retry counter/signature on successful content agent._empty_content_retries = 0 agent._thinking_prefill_retries = 0 - # Successful content reached — surface the one-shot fallback - # switch notice (if a fallback activated this turn) before - # dropping the noisy retry buffer, so a provider/model switch - # stays visible even when the fallback succeeds. + # Surface the one-shot fallback switch notice before dropping the retry + # buffer so a provider/model switch stays visible on success. agent._emit_pending_fallback_notice() agent._clear_status_buffer() @@ -8716,15 +3829,9 @@ def run_conversation( ) _ack_mode = intent_ack_continuation_mode(agent) - # Said-continue-but-stopped guard (agent.stall_guards): the - # model ended the turn with no tool calls but its short reply - # TAILS with an announced next action ("Let me now…", - # "I will now…"). Unlike the intent-ack detector below, this - # fires mid-task too (after tool results), which is exactly - # where eval traces show the stall. It reuses the SAME bounded - # continuation path and counter (max 2 per turn), so the - # alternation-safe interim-assistant + user-nudge mechanism — - # not a new parallel one — carries the recovery. + # Said-continue-but-stopped guard: no tool calls but the short reply + # TAILS with an announced next action. Fires mid-task too; reuses the + # SAME bounded continuation path and counter (max 2 per turn). _stall_continue_intent = ( bool(getattr(agent, "_stall_guards", True)) and agent.valid_tool_names @@ -8761,9 +3868,8 @@ def run_conversation( } append_message(messages, continue_msg) agent._session_messages = messages - # An acknowledgment is explicitly non-final. Do not let its - # text suppress iteration-limit summarization if this - # continuation consumes the remaining budget. + # An acknowledgment is non-final: its text must not suppress + # iteration-limit summarization if the continuation exhausts budget. final_response = None continue @@ -8784,17 +3890,8 @@ def run_conversation( final_msg = agent._build_assistant_message(assistant_message, finish_reason) # ── Dropped tool-call recovery (copilot/Claude) ──────── - # Some providers (observed: claude-opus-4.8 / claude-sonnet-4.5 - # on GitHub Copilot, ~2026-07) return finish_reason="tool_calls" - # while the parsed tool_calls array is empty — the model - # signalled it wanted to act but the payload shipped no call. - # Reaching finalization with that mismatch means the turn is - # about to end with the task unstarted (the narration, which may - # be in content or only in the reasoning field, gets treated as - # the final answer). Re-prompt (bounded to 3 CONSECUTIVE stalls; - # the budget resets after any successful tool round) to make the - # model emit the call instead of exiting. finish_reason="stop" - # text finishes never enter this guard. + # finish_reason="tool_calls" with empty tool_calls would end the turn + # unstarted; re-prompt (max 3 CONSECUTIVE stalls, reset per tool round). if ( finish_reason == "tool_calls" and not assistant_message.tool_calls @@ -8811,16 +3908,9 @@ def run_conversation( "↻ Model signaled a tool call but sent none — " f"re-prompting ({agent._dropped_toolcall_retries}/3)" ) - # Both halves of the re-prompt pair are ephemeral recovery - # scaffolding (mirrors the empty-response nudge pattern): - # the interim narration-only assistant turn exists solely to - # keep role alternation valid for the nudge, and the nudge - # exists solely to drive the retry. Flag both so the - # persistence layer never writes them to the durable - # transcript and the finalization pop below can strip an - # unanswered tail pair. A recovered (answered) pair stays - # buried mid-list in live memory but is skipped by the - # flush regardless of position. + # Both halves of the re-prompt pair are ephemeral scaffolding; flag + # them so the flush never persists them and the finalization pop + # can strip an unanswered tail pair. final_msg["_dropped_toolcall_nudge"] = True append_message(messages, final_msg) append_message(messages, { @@ -8832,16 +3922,11 @@ def run_conversation( final_response = None continue - # Reached finalization without the dropped-tool-call mismatch — - # a genuine turn end. Clear the consecutive-stall budget so the - # next turn starts fresh. + # Genuine turn end (no dropped-tool-call mismatch): clear stall budget. agent._dropped_toolcall_retries = 0 - # Pop thinking-only prefill and empty-response retry - # scaffolding before appending either a final response or a - # verification-stop follow-up. These internal turns are only - # for the next API retry and should not become durable - # transcript context. + # Pop prefill / empty-retry scaffolding before the final response or + # verification follow-up; it must not become durable transcript. while ( messages and isinstance(messages[-1], dict) @@ -8854,186 +3939,25 @@ def run_conversation( ): messages.pop() - try: - from agent.verification_stop import ( - build_verify_on_stop_nudge, - verify_on_stop_enabled, - ) - - if verify_on_stop_enabled(): - _verify_nudge = build_verify_on_stop_nudge( - session_id=getattr(agent, "session_id", None), - changed_paths=getattr(agent, "_turn_file_mutation_paths", set()), - attempts=getattr(agent, "_verification_stop_nudges", 0), - ) - else: - _verify_nudge = None - except Exception: - logger.debug("verification stop-loop check failed", exc_info=True) - _verify_nudge = None - - if _verify_nudge: - agent._verification_stop_nudges = ( - getattr(agent, "_verification_stop_nudges", 0) + 1 - ) - final_msg["finish_reason"] = "verification_required" - # The assistant response is real content — persist it and - # emit to the UI as an interim message so the user sees the - # attempted final answer before the verification loop runs. - # Only the nudge is flagged synthetic so it gets stripped - # from the durable transcript (#65919 §7). - agent._emit_interim_assistant_message(final_msg) - append_message(messages, final_msg) - try: - agent._flush_messages_to_session_db(messages, conversation_history) - except Exception: - logger.debug("verify-on-stop interim flush failed", exc_info=True) - append_message(messages, { - "role": "user", - "content": _verify_nudge, - "_verification_stop_synthetic": True, - }) - agent._session_messages = messages - # Run the verification-stop loop silently — the nudge is an - # internal turn that should not add noise to the user's - # terminal. Keep a debug breadcrumb in agent.log for tracing. - logger.debug("verification stop-loop nudge issued (attempt %d)", - agent._verification_stop_nudges) - # Keep the attempted answer only as an explicit fallback for - # continuation-budget exhaustion. ``final_response`` itself - # must be cleared so the finalizer can distinguish this gate - # from unrelated error/recovery exits. (#61631) - # Track whether this candidate was already streamed so the - # finalizer can mark the turn previewed only if the - # candidate is actually reused as the final response. - _pending_verification_response = final_response - _pending_verification_response_previewed = ( - agent._interim_content_was_streamed(final_response or "") - ) - final_response = None - continue - - # User verification-loop gate: when the agent edited code this - # turn, let a registered `pre_verify` hook (plugin/shell) keep it - # going one more turn. The shipped guidance is folded into the - # evidence-based verify-on-stop nudge above, so this path has no - # default continuation cost. - _verify_nudge2 = None - _edited = sorted(getattr(agent, "_turn_file_mutation_paths", set()) or []) - _attempt = getattr(agent, "_pre_verify_nudges", 0) - try: - from agent.verify_hooks import max_verify_nudges - from hermes_cli.lifecycle import has_hook - from hermes_cli.plugins import get_pre_verify_continue_message - - if _edited and has_hook("pre_verify") and _attempt < max_verify_nudges(): - # Posture is fixed for the session — resolve once + cache. - coding = getattr(agent, "_resolved_is_coding", None) - if coding is None: - from agent.coding_context import is_coding_context - coding = bool(is_coding_context(platform=getattr(agent, "platform", "") or "")) - agent._resolved_is_coding = coding - _verify_nudge2 = get_pre_verify_continue_message( - session_id=getattr(agent, "session_id", None) or "", - platform=getattr(agent, "platform", "") or "", - model=getattr(agent, "model", "") or "", - coding=coding, - attempt=_attempt, - final_response=final_response, - changed_paths=_edited, - ) - except Exception: - logger.debug("pre_verify hook check failed", exc_info=True) - _verify_nudge2 = None - - if _verify_nudge2: - agent._pre_verify_nudges = _attempt + 1 - final_msg["finish_reason"] = "verify_hook_continue" - # The assistant response is real content — persist it and - # emit to the UI as an interim message so the user sees the - # attempted final answer before the pre_verify loop runs. - # Only the nudge is flagged synthetic so it gets stripped - # from the durable transcript (#65919 §7). - agent._emit_interim_assistant_message(final_msg) - append_message(messages, final_msg) - try: - agent._flush_messages_to_session_db(messages, conversation_history) - except Exception: - logger.debug("pre_verify interim flush failed", exc_info=True) - append_message(messages, { - "role": "user", - "content": _verify_nudge2, - "_pre_verify_synthetic": True, - }) - agent._session_messages = messages - logger.debug("pre_verify nudge issued (attempt %d)", - agent._pre_verify_nudges) - _pending_verification_response = final_response - _pending_verification_response_previewed = ( - agent._interim_content_was_streamed(final_response or "") - ) - final_response = None - continue - - # ── Kanban worker terminal-tool stop guard ───────────── - # Workers must end with kanban_complete / kanban_block. - # Models sometimes narrate the next step ("Let me write the - # report") and stop with finish_reason=stop — a clean exit - # that the dispatcher records as protocol_violation. Nudge - # once or twice before allowing that exit. - try: - from agent.kanban_stop import build_kanban_stop_nudge - - _kanban_nudge = build_kanban_stop_nudge( - messages=messages, - attempts=getattr(agent, "_kanban_stop_nudges", 0), - ) - except Exception: - logger.debug("kanban stop-loop check failed", exc_info=True) - _kanban_nudge = None - - if _kanban_nudge: - agent._kanban_stop_nudges = ( - getattr(agent, "_kanban_stop_nudges", 0) + 1 - ) - final_msg["finish_reason"] = "kanban_terminal_required" - final_msg["_kanban_stop_synthetic"] = True - append_message(messages, final_msg) - append_message(messages, { - "role": "user", - "content": _kanban_nudge, - "_kanban_stop_synthetic": True, - }) - agent._session_messages = messages - logger.info( - "kanban stop-loop nudge issued (attempt %d) task=%s", - agent._kanban_stop_nudges, - os.environ.get("HERMES_KANBAN_TASK", ""), - ) - agent._emit_status( - "⚠️ Kanban worker tried to exit without " - "kanban_complete/kanban_block — nudging to finish" - ) - # Same finalizer contract as verify-on-stop: clear - # final_response while continuing so a later budget - # exhaustion path does not treat the narrated stop as - # a completed answer. - _pending_verification_response = final_response - _pending_verification_response_previewed = ( - agent._interim_content_was_streamed(final_response or "") - ) + _sg = apply_stop_gates( + agent, + final_msg, + final_response=final_response, + messages=messages, + conversation_history=conversation_history, + pending_verification_response=_pending_verification_response, + pending_verification_response_previewed=_pending_verification_response_previewed, + ) + _pending_verification_response = _sg.pending_verification_response + _pending_verification_response_previewed = _sg.pending_verification_response_previewed + if _sg.continue_turn: final_response = None continue append_message(messages, final_msg) - # Make the completed answer durable before leaving the loop — - # a session torn down before finalize_turn's _persist_session - # otherwise loses a reply the user already saw (#81641). Same - # contract as the tool-call exit (#49045) and the verify exits - # above; _DB_PERSISTED_MARKER keeps _persist_session idempotent. - # Unlike the tool-call exit, failure must NOT abort the turn: - # no side effect follows and _persist_session retries the write. - # Full incident narrative: tests/run_agent/test_81641_*.py. + # Make the answer durable before leaving the loop; _DB_PERSISTED_MARKER + # keeps _persist_session idempotent. Failure must NOT abort the turn: + # _persist_session retries the write. (#81641) try: agent._flush_messages_to_session_db(messages, conversation_history) except Exception: @@ -9050,29 +3974,13 @@ def run_conversation( break except Exception as e: - # Count every escaped exception against the per-turn bound before - # classification — permanent failures must terminate even when the - # turn budget is unlimited (#92450). + # Count every escaped exception before classification so permanent + # failures terminate even with an unlimited turn budget. (#92450) _outer_error_count += 1 - # Phase-aware error classification. The huge outer try/except spans - # both the actual API request and all local post-processing of the - # returned assistant message. Deterministic local bugs (e.g. - # passing a multimodal content list into a regex helper after a - # vision turn or context compaction) should not be retried: they - # will fail identically on every iteration and only burn the - # iteration budget. We classify an error as local by inspecting the - # traceback: if the exception propagated through any of the known - # local post-processing helpers and never entered the interruptible - # API-call helpers, it is almost certainly a local processing bug. - # (#66267) - # - # Interpreter shutdown: if the process is tearing down, every - # executor-backed operation (API call, tool dispatch, memory sync) - # raises ``RuntimeError: cannot schedule new futures after - # interpreter shutdown``. Retrying is pointless — the executor is - # gone for good — and each retry just spams another traceback. - # Break immediately so the turn exits cleanly. (#93217) + # Phase-aware classification: deterministic local post-processing bugs + # (traceback via local helpers, never API helpers) aren't retried (#66267). + # Interpreter shutdown makes every executor op raise: break. (#93217) if sys.is_finalizing() or _is_interpreter_shutdown_error(e): error_msg = ( f"Interpreter is shutting down — cannot continue " @@ -9083,9 +3991,8 @@ def run_conversation( except (OSError, ValueError): pass logger.warning(error_msg) - # Best-effort persist — the executor is dying, so this may - # raise the same RuntimeError. Don't let that mask the - # shutdown exit. finalize_turn will retry the persist. + # Best-effort persist — the dying executor may raise the same error; + # don't let it mask the shutdown exit. finalize_turn retries. try: agent._persist_session(messages, conversation_history) except Exception: @@ -9095,12 +4002,8 @@ def run_conversation( "Session is shutting down. Your conversation can be " "resumed with: hermes --resume " ) - # Don't append the assistant message here — a thinking-prefill - # or interim assistant may already be the tail, and appending - # would create assistant→assistant. finalize_turn handles - # this case safely (lines 341-353: appends only when - # _tail_role != "assistant"), matching the pattern at other - # break sites that set final_response without appending. + # Don't append: a prefill/interim assistant may already be the tail + # (assistant→assistant). finalize_turn appends only when safe. break tb_module_names: set[str] = set() @@ -9122,13 +4025,8 @@ def run_conversation( ) else: error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}" - # The background-review fork sets suppress_status_output=True so - # lifecycle noise never reaches the user's terminal — but this - # bare print() bypassed it, leaking ❌ lines onto the shell after - # the TUI exited. Honor the same contract _vprint enforces - # (quiet_mode -q still shows hard failures, matching force=True - # semantics; suppress_status_output silences them). The - # logger.exception below still captures the full traceback. + # Honor the _vprint contract: suppress_status_output silences hard + # failures; quiet_mode -q still shows them. Traceback is logged below. if getattr(agent, "suppress_status_output", False): logger.error(error_msg) else: @@ -9137,17 +4035,12 @@ def run_conversation( except (OSError, ValueError): logger.error(error_msg) - # Emit the full traceback at ERROR level so it lands in both - # agent.log AND errors.log. Previously this was logged at DEBUG, - # which meant intermittent outer-loop failures were unreproducible - # — users would see a one-line summary on screen with no way to - # recover the call site. logger.exception() includes the - # traceback automatically and emits at ERROR. + # ERROR level with traceback so outer-loop failures land in agent.log + # AND errors.log and stay reproducible. logger.exception("Outer loop error in API call #%d", api_call_count) - # If an assistant message with tool_calls was already appended, - # the API expects a role="tool" result for every tool_call_id. - # Fill in error results for any that weren't answered yet. + # An appended assistant tool_calls message needs a role="tool" result + # per tool_call_id; fill in error results for unanswered ones. for idx in range(len(messages) - 1, -1, -1): msg = messages[idx] if not isinstance(msg, dict): @@ -9172,19 +4065,11 @@ def run_conversation( append_message(messages, err_msg) break - # Non-tool errors don't need a synthetic message injected. - # The error is already printed to the user (line above), and - # the retry loop continues. Injecting a fake user/assistant - # message pollutes history, burns tokens, and risks violating - # role-alternation invariants. + # Non-tool errors are already printed; a synthetic message would pollute + # history and risk breaking role alternation. - # If we're near the limit, break to avoid infinite loops. - # Local processing errors are deterministic — stop immediately - # rather than retrying until the budget is exhausted. Repeated - # outer-loop errors stop after a small per-turn cap: with - # max_iterations now unlimited by default, a permanent failure - # would otherwise spin forever and overwrite the rotated log - # history within minutes (#92450). + # Local errors are deterministic: stop early instead of retrying until the + # budget is gone; a small per-turn cap prevents infinite spinning (#92450). _outer_error_cap = min(_MAX_OUTER_LOOP_ERRORS, max(1, agent.max_iterations)) if ( _is_local_processing_error @@ -9201,17 +4086,11 @@ def run_conversation( else: _turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})" final_response = f"I apologize, but I encountered repeated errors: {error_msg}" - # Don't append the assistant message here — a thinking-prefill - # or interim assistant may already be the tail, and appending - # would create assistant→assistant. finalize_turn handles - # this case safely (lines 341-353: appends only when - # _tail_role != "assistant"), matching the pattern at other - # break sites that set final_response without appending. + # Don't append the assistant message: a prefill/interim assistant may be + # the tail. finalize_turn appends only when _tail_role != "assistant". break - # Post-loop turn finalization extracted to agent/turn_finalizer.finalize_turn - # (god-file decomposition Phase 1 step 4). Behavior-neutral: the assembled - # result dict is returned exactly as before. + # Post-loop finalization lives in agent/turn_finalizer.finalize_turn. result = finalize_turn( agent, final_response=final_response, @@ -9230,10 +4109,8 @@ def run_conversation( _pending_verification_response_previewed=_pending_verification_response_previewed, ) if _compression_timeout_exhausted: - # Reuse the gateway's existing context-recovery contract (#98722, - # salvaged from #98741). The bloated transcript remains intact while - # future input can move to a clean session instead of replaying the - # summarize-timeout loop. + # Reuse the gateway's context-recovery contract: transcript stays intact while + # future input can move to a clean session (#98722). result["error"] = _COMPRESSION_TIMEOUT_FINAL_RESPONSE result["partial"] = True result["compression_exhausted"] = True diff --git a/agent/turn_context.py b/agent/turn_context.py index 691b1de9c0..655d668281 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -1,26 +1,9 @@ """Per-turn setup for ``run_conversation`` (the turn prologue). -``run_conversation`` opened with ~470 lines of straight-line setup before the -tool-calling loop ever started: stdio guarding, runtime-main wiring, retry-counter -resets, user-message sanitization, todo/nudge-counter hydration, system-prompt -restore-or-build, session-row creation (before compression, whose DB writes -reference the row), preflight context compression, the ``pre_llm_call`` plugin -hook, external-memory prefetch, and crash-resilience persistence (last, so the -user row is written once with its final ``api_content`` sidecar). - -All of that is *prologue* — it runs once per turn, has no back-references into the -loop, and produces a fixed set of values the loop then consumes. ``TurnContext`` -captures those produced values; ``build_turn_context`` performs the setup work and -returns one. ``run_conversation`` is left to unpack the context and run the loop, -shrinking the orchestrator by the full prologue. - -The builder still mutates ``agent`` heavily (counters, thread id, cached prompt, -session DB) exactly as the inline code did — those side effects are the point. The -``TurnContext`` it returns carries only the *locals* the loop reads back. - -Behavior is identical to the original inline prologue; this is a pure -move-and-name refactor with no semantic change. -""" +``build_turn_context`` runs the once-per-turn setup (stdio guard, sanitization, prompt +restore-or-build, session row, preflight compression, pre_llm_call hook, prefetch, +persistence) mutating ``agent`` exactly as the inline code did, and returns a +``TurnContext`` carrying only the locals the loop reads back.""" from __future__ import annotations @@ -29,7 +12,7 @@ import threading import time import uuid from dataclasses import dataclass -from typing import Any, Dict, List, Mapping, Optional +from typing import Any, Dict, List, Mapping, Optional, Tuple from agent.conversation_compression import ( IDLE_COMPACTION_STATUS_TEMPLATE, @@ -59,17 +42,8 @@ def _preflight_request_tokens( ) -> int: """Token estimate for automatic preflight compression. - When the upcoming request is eligible for native Responses compaction, - count the checkpoint-pruned wire payload rather than the full durable - transcript. Auxiliary compression still uses the generic estimator - (``native_compaction_eligible=False``). - - Usage-anchored fast path: when a provider-reported usage anchor is - valid for ``messages`` (see ``anchored_context_tokens``), it already - covers system prompt + tool schemas + full history EXACTLY as the - provider counted them, with estimation confined to the messages - appended since that response. Prefer it over every heuristic. - """ + Prefers a valid provider usage anchor; on native-compaction-eligible requests counts + the checkpoint-pruned wire payload; otherwise uses the generic estimator.""" anchored = anchored_context_tokens( messages, getattr(agent, "_usage_anchor", None) ) @@ -112,9 +86,7 @@ def _preflight_request_tokens( def _agent_stale_thinking_on_wire(agent: Any) -> bool: """Whether the agent's active route replays stale thinking text (#84371). - Route facts unavailable (test doubles, partially-built agents) default to - ``True`` — the conservative full charge. - """ + Returns ``True`` (conservative full charge) when route facts are unavailable.""" try: from agent.message_sanitization import stale_thinking_reaches_wire @@ -135,20 +107,9 @@ def compose_user_api_content( ) -> Optional[str]: """Compose the API-bound content of the current turn's user message. - Sources: memory-manager prefetch + ``pre_llm_call`` plugin context with - target="user_message" (the default). Both are appended to the *API copy* - of the user message only — the stored content stays clean. - - This is the single source of that composition. The prologue stamps the - result onto the live message as ``api_content`` (persisted alongside the - clean content) and the ``api_messages`` build in ``conversation_loop`` - sends the same helper's output, so the persisted sidecar can never drift - from the bytes on the wire — which is the whole prompt-cache invariant: - what turn N sends must be what turn N+1 replays. - - Returns ``None`` when nothing is injected (multimodal/non-string content, - or no ephemeral context), meaning the message is sent as-is. - """ + Single source for the ``api_content`` sidecar and the wire bytes, so they never + drift — the prompt-cache invariant: what turn N sends is what turn N+1 replays. + Returns ``None`` when nothing is injected (message is sent as-is).""" if not isinstance(content, str): return None injections = [] @@ -166,16 +127,8 @@ def compose_user_api_content( def substitute_api_content(api_msg: Dict[str, Any]) -> Optional[str]: """Pop the ``api_content`` sidecar and substitute it into ``content``. - Used at every API-bound message-build site (the ``api_messages`` build in - ``conversation_loop``, the max-iterations summary in - ``chat_completion_helpers``, the chat-completions transport). The sidecar - carries the exact bytes previously sent to the API for this message when - they differ from the clean stored content; substituting it here keeps the - provider prompt-cache prefix byte-stable across turns. - - Returns the popped sidecar string (for callers that need the value for - current-turn composition logic) or ``None`` when absent. - """ + Keeps the provider prompt-cache prefix byte-stable across turns. + Returns the popped sidecar string, or ``None`` when absent.""" sidecar = api_msg.pop("api_content", None) if ( isinstance(sidecar, str) @@ -189,21 +142,12 @@ def substitute_api_content(api_msg: Dict[str, Any]) -> Optional[str]: def drop_stale_api_content(msg: Dict[str, Any]) -> None: """Drop the ``api_content`` sidecar from a message whose content was rewritten. - Called from every content-rewrite path (historical image strip, - merge-summary-into-tail, consecutive-user repair merge, stale-confirmation - redaction). Replaying the pre-rewrite sidecar would resend exactly what - the rewrite removed, so it must be dropped — the cost is one cache - boundary miss, never wrong content. - """ + Replaying it would resend what the rewrite removed; cost is one cache miss.""" msg.pop("api_content", None) def extract_api_content_sidecar(msg: Mapping[str, Any]) -> Optional[str]: - """Extract the ``api_content`` sidecar from a message dict for persistence. - - Shared by the gateway/branch forwarding sites that copy the sidecar into a - new row. Returns the string sidecar or ``None`` when absent/non-string. - """ + """Extract the ``api_content`` sidecar; ``None`` when absent/non-string.""" v = msg.get("api_content") return v if isinstance(v, str) else None @@ -211,14 +155,8 @@ def extract_api_content_sidecar(msg: Mapping[str, Any]) -> Optional[str]: def consume_gateway_turn_context_notes(agent: Any) -> str: """Pop the gateway's per-turn must-deliver notes off the agent (one-shot). - The gateway relocates volatile per-turn facts OUT of the ephemeral system - prompt (auto-reset notes, the first-contact intro, voice-channel changes) - and delivers them on the current user message via the api_content sidecar - instead, so the composed system prompt stays byte-stable turn-over-turn. - It stages the rendered notes on ``agent._gateway_turn_context_notes`` - right before ``run_conversation``; this consumes them so a cached agent - can never replay a stale note on a later turn. - """ + Staged on ``agent._gateway_turn_context_notes``; consuming them keeps the system + prompt byte-stable and prevents a cached agent replaying a stale note.""" notes = getattr(agent, "_gateway_turn_context_notes", "") or "" if hasattr(agent, "_gateway_turn_context_notes"): try: @@ -231,14 +169,8 @@ def consume_gateway_turn_context_notes(agent: Any) -> str: def append_notes_to_multimodal_content(content: Any, notes: str) -> bool: """Deliver must-deliver notes on a multimodal (list) user message. - ``compose_user_api_content`` returns ``None`` for non-string content, so - sidecar-borne facts would silently drop on image/attachment turns. For - gateway must-deliver notes we instead append a text part to the content - list in place — the part becomes durable message content (persisted and - replayed as-is), which keeps the wire and the transcript byte-identical. - - Returns ``True`` when a part was appended. - """ + Appends a durable text part in place, since the sidecar path returns ``None`` + for non-string content. Returns ``True`` when a part was appended.""" if not notes or not isinstance(content, list): return False try: @@ -248,28 +180,13 @@ def append_notes_to_multimodal_content(content: Any, notes: str) -> bool: return False -# Surfaces whose sessions must not be auto-titled. The prologue is shared by -# EVERY agent, not only the ones a human is watching, so membership here is what -# keeps the titler off machine-driven runs: -# -# - cron — the scheduler names its own session after the job in its `finally` -# block, and the opener is the cron delivery hint, not a user's request. -# Titling it writes that scaffolding as the visible name for the whole run and -# bills a side-LLM call per fire, against the same job that sets -# `skip_memory` / `skip_background_review` to avoid exactly that. -# - subagent — a delegated child's session is hidden from every picker, so its -# title is never read. A batch at `max_concurrent_children` would pay N title -# calls for N names nobody sees. +# Surfaces whose sessions must not be auto-titled: cron names its own session and +# its opener is a delivery hint; subagent sessions are hidden from every picker. _UNTITLED_PLATFORMS = frozenset({"cron", "subagent"}) def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: - """Kick off auto-titling for this session's first user message. - - Called from the turn prologue, so every surface a human reads (CLI, gateway, - TUI/desktop, ACP) gets identical behavior without each one re-implementing - the call. Fully defensive: titling is cosmetic and must never break a turn. - """ + """Kick off auto-titling for the session's first user message; never fatal.""" session_db = getattr(agent, "_session_db", None) session_id = getattr(agent, "session_id", None) if not session_db or not session_id: @@ -282,9 +199,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: from agent.message_content import flatten_message_text from agent.title_generator import maybe_auto_title - # The turn's own user message, as text. Multimodal turns flatten to - # their text parts; an image-only turn yields "" and is skipped, since - # there is nothing to title from. + # Turn's user message as text; image-only turns yield "" and are skipped. user_text = "" for msg in reversed(messages or []): if isinstance(msg, dict) and msg.get("role") == "user": @@ -293,9 +208,8 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: if not user_text: return - # The session row is created lazily on the first persist, which happens - # later in the turn. Force it now, or the title write matches zero rows - # and the session stays untitled for the whole turn anyway. + # Session row is created lazily later; force it now or the title write matches + # zero rows. if not getattr(agent, "_session_db_created", False): ensure = getattr(agent, "_ensure_db_session", None) if callable(ensure): @@ -303,9 +217,8 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: if not getattr(agent, "_session_db_created", False): return - # Snapshot the runtime identity; the validator lets the background - # titler skip its LLM call if the user switches models before it fires - # (a stale request would reload an unloaded Ollama model, #19027). + # Snapshot runtime identity so the background titler can skip if the user + # switches models before it fires (#19027). _model = getattr(agent, "model", None) _provider = getattr(agent, "provider", None) @@ -338,19 +251,9 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> int: """Locate this turn's user message after compaction rebuilt ``messages``. - Compression replaces list entries with fresh copies (and may append a - todo-snapshot user message or a restored user turn AFTER the surviving - copy of the current turn's message), so a pre-compression index is - meaningless. Prefer the LAST user message whose content exactly matches - this turn's text — the surviving copy in the common case — so the - injection stamp and the #48677 persist override can't land on a - todo-snapshot or historical row. Fall back to the last *user-originated* - turn when no exact match survives (merge-summary-into-tail rewrites the - content but the trackers still need a live anchor). Compaction handoffs - must never become the fallback anchor (#80622) — they are reference-only - scaffolding, not the active ask. Returns -1 when the list has no - user-originated message at all. - """ + Prefers the LAST user message whose content exactly matches this turn's text, else + the last user-originated turn; compaction handoffs are never the fallback (#80622). + Returns -1 when there is no user-originated message.""" from agent.context_compressor import user_originated_turn_view fallback = -1 @@ -358,9 +261,8 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in msg = messages[i] if not (isinstance(msg, dict) and msg.get("role") == "user"): continue - # Typed synthetic current events still need their physical persistence - # anchor when their raw content is unchanged. They are not eligible - # for the human-only fallback below. + # Typed synthetic current events keep their persistence anchor when raw + # content is unchanged; not eligible for the human-only fallback below. if msg.get("content") == user_message: return i live_view = user_originated_turn_view(msg) @@ -380,28 +282,15 @@ def compression_made_progress( ) -> bool: """Return ``True`` if a compression pass materially reduced the request. - Compression can succeed by summarising message contents — reducing the - estimated request token count — without reducing the message row - count. Treating row count as the sole progress signal false-positives - on size-only wins and surfaces a misleading "Cannot compress further" - failure even when post-compression tokens are well below the model - context window. See issue #39548 for an observed case: 220 → 220 - messages, ~288k → ~183k tokens on a 1M-context model still triggered - auto-reset. - - The token reduction must be *material* (>5%) to count as progress — the - same floor the overflow-handler retry path uses (conversation_loop.py, - #39550) — so a sub-5% wobble doesn't keep the multi-pass loop spinning. - """ + Counts a >5% token reduction as progress even when the row count is unchanged + (size-only wins, #39548); same floor as the overflow-handler retry path.""" if new_len < orig_len: return True return orig_tokens > 0 and new_tokens < orig_tokens * 0.95 -# Back-compat alias: this predicate was module-private until the gateway's -# session-hygiene recovery gate needed the same semantics (#79624). Keeping the -# old name bound means existing callers and any test that patches -# ``_compression_made_progress`` continue to work unchanged. +# Back-compat alias: gateway callers and tests patch ``_compression_made_progress`` +# (#79624). _compression_made_progress = compression_made_progress @@ -425,15 +314,8 @@ def _fail_closed_after_preflight_timeout(agent, request_tokens: int) -> None: def _review_fork_first_request_pending(agent: Any) -> bool: """Whether a detached review fork has yet to send its first provider request. - The background-review fork (issue #93057) replays the parent's FULL - snapshot on its first provider request as a warm prompt-cache read - (same-model cache parity). Compaction must not rewrite the snapshot - before that first request goes out — a compacted transcript would miss - the parent's cached prefix and turn a cheap cached replay into a cold - over-threshold write. Once the first provider response has arrived the - fork's tool loop is its own context, and both compression gates resume. - Dormant for every agent without the attribute. - """ + The fork replays the parent's FULL snapshot as a warm cache read, so compaction must + wait until that first response arrives. Dormant without the attribute (#93057).""" return bool( getattr(agent, "_review_defer_compaction_before_first_response", False) and not getattr(agent, "_turn_received_provider_response", False) @@ -445,11 +327,7 @@ def _compression_warrants_another_preflight_pass( ) -> bool: """Whether an over-threshold request merits another immediate summary. - Row-count progress is enough to prove that a compression boundary was real, - but not enough to justify another expensive pass before trying the provider. - Continue only when the request remains over threshold *and* the previous pass - materially reduced its estimated token pressure (>5%). - """ + Continue only if still over threshold AND the previous pass cut tokens by >5%.""" return ( new_tokens >= threshold_tokens and orig_tokens > 0 @@ -465,21 +343,10 @@ def _should_run_preflight_estimate( ) -> bool: """Cheap gate for the (expensive) full preflight token estimate. - Returns ``True`` when either: - (a) message count exceeds the protected ranges (the historical gate), or - (b) a cheap char-based estimate already crosses the configured threshold - — the few-but-huge case from issue #27405 that the count-only gate - would silently skip (a handful of very large messages never trips - the count condition, so compression was never attempted and the - turn hit a hard context-overflow error). - - Branch (b) uses ``estimate_messages_tokens_rough`` (the shared char-based - estimator) so a single large base64 image isn't mistaken for ~250K tokens. - It intentionally undercounts vs. the full request estimate — it omits the - system prompt and tool schemas — because it is only a *hint* deciding - whether to pay for the authoritative ``estimate_request_tokens_rough``, - which (together with ``should_compress``) makes the real decision. - """ + ``True`` when message count exceeds the protected ranges OR a rough char-based + estimate crosses the threshold — the few-but-huge case (#27405). The estimator + undercounts by design (omits system/tools) so one large base64 image is not + mistaken for ~250K tokens.""" if len(messages) > protect_first_n + protect_last_n + 1: return True return estimate_messages_tokens_rough(messages) >= threshold_tokens @@ -496,20 +363,9 @@ def _should_idle_compact( ) -> bool: """Decide whether an idle-triggered compaction should run this turn. - Idle compaction is opt-in (``idle_after_seconds <= 0`` disables it). It - fires when a session resumes after a wall-clock gap of at least - ``idle_after_seconds`` since its last activity, so a long-lived thread - that is paused and later resumed compacts its accumulated history up - front instead of re-reading it on every subsequent turn. - - It is orthogonal to the token-threshold trigger: it does NOT require the - context to exceed ``threshold_tokens``. It still skips work when the - context is at or below ``floor_tokens`` (the size compaction would reduce - *to*), so a small idle thread never pays for a summarisation that saves - nothing, and it defers to an active compression-failure cooldown. - - Pure predicate so the policy is unit-testable without a live agent. - """ + Fires after a wall-clock gap of ``idle_after_seconds`` (opt-in, <= 0 disables), + independent of ``threshold_tokens``; skips at/below ``floor_tokens`` and during a + compression-failure cooldown. Pure predicate.""" if not enabled or idle_after_seconds <= 0: return False if idle_gap_seconds < idle_after_seconds: @@ -572,27 +428,18 @@ def build_turn_context( ) -> TurnContext: """Run the once-per-turn setup and return the loop's input context. - The callables/helpers the original prologue referenced from the - ``conversation_loop`` module are passed in explicitly to keep this module - free of an import cycle with ``agent.conversation_loop``. - """ + Helpers are passed in to avoid an import cycle with ``agent.conversation_loop``.""" # Guard stdio against OSError from broken pipes (systemd/headless/daemon). install_safe_stdio() - # Recover a session rotated by another path before binding log/turn ids or - # copying client-supplied history. Everything in this turn must consistently - # belong to the canonical child, including observability metadata. + # Recover a rotated session before binding log/turn ids or copying client history so + # everything in this turn belongs to the canonical child. recovered_history = recover_rotated_compression_session(agent) if recovered_history is not None: conversation_history = recovered_history - # NOTE: the DB session row is created later, AFTER the system prompt is - # restored/built (see _ensure_db_session() below the system-prompt block). - # Creating it here — before _cached_system_prompt is populated — inserts a - # row with system_prompt=NULL on a fresh API/gateway agent that carries - # client-managed history, which then trips the "stored system prompt is - # null; rebuilding from scratch" warning and a needless first-turn prefix - # cache miss. (Issue #45499.) + # NOTE: the DB session row is created later, after the system prompt is built; + # creating it now would persist system_prompt=NULL and cost a cache miss (#45499). # Tag log records on this thread with the session ID for ``hermes logs``. set_session_context(agent.session_id) @@ -608,15 +455,9 @@ def build_turn_context( try: from agent.auxiliary_client import set_runtime_main from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe - # Rotation-stable prompt-cache scope. Memoized per segment on the - # agent, so this is a DB walk at most once per segment — except a - # brand-new session whose row lands later in turn setup - # (_ensure_db_session); that first turn falls back to the physical - # id here and the first build_api_kwargs re-resolves. Stays valid - # through a mid-turn compression rotation because the lineage root - # is by definition rotation-invariant (#79017). Resolved with the - # never-raising variant OUTSIDE the argument list, so a resolution - # failure can only lose the scope — never the whole runtime binding. + # Rotation-stable prompt-cache scope (lineage root), memoized per segment; a + # new session uses the physical id until build_api_kwargs re-resolves (#79017). + # Never-raising variant, outside the argument list: failure loses only scope. _cache_scope = resolve_prompt_cache_scope_safe(agent) or "" set_runtime_main( getattr(agent, "provider", "") or "", @@ -632,31 +473,13 @@ def build_turn_context( except Exception: pass - # Between-turns MCP refresh: an MCP server that finished connecting since - # the previous turn (slow HTTP/OAuth servers routinely take 2-6s on a cold - # connect, missing the bounded startup wait) lands in THIS turn's tool - # snapshot. Timing is cache-safe by construction: it runs in the per-turn - # prologue, before this turn's first API call assembles ``tools=``, so it - # never mutates the prefix of an in-flight turn. ``preserve_prefix`` makes - # the *content* cache-safe too (#100336): a plain rebuild re-derives the - # array from live availability, so a flapping ``check_fn`` silently drops a - # tool and a late arrival splices into sorted position — either one forks - # the tool block and re-prefills the whole history behind it, every turn it - # happens. With the flag the live order is authoritative and the array - # only ever grows. No-op when no MCP servers are registered (the common - # case, gated by the cheap ``has_registered_mcp_tools`` check) or when the - # tool set is unchanged (``refresh_agent_mcp_tools`` diffs by name and - # leaves the snapshot untouched on no-change). + # Between-turns MCP refresh: late-connecting servers land in THIS turn's snapshot, + # before the first API call assembles ``tools=``. ``preserve_prefix`` keeps the + # tool array append-only so a flapping ``check_fn`` can't fork the cache (#100336). try: if not getattr(agent, "_skip_mcp_refresh", False): - # Import-cost gate: ``tools.mcp_tool`` pulls in the whole ``mcp`` - # package (~0.4s measured) even when the user has zero MCP servers - # configured. MCP tools can only be registered by code that has - # already imported ``tools.mcp_tool`` (discovery, /reload-mcp, - # late-binding refresh) — so if it isn't in sys.modules yet, there - # is nothing to refresh and the import can be skipped outright. - # This keeps the no-MCP first turn off the heavy import path - # without changing behavior for MCP users. + # Import-cost gate: MCP tools are only registered by code that already + # imported ``tools.mcp_tool`` (~0.4s); not in sys.modules => nothing to do. import sys as _sys if "tools.mcp_tool" in _sys.modules: from tools.mcp_tool import has_registered_mcp_tools, refresh_agent_mcp_tools @@ -690,9 +513,8 @@ def build_turn_context( agent._relay_pending_turn_id = None agent._current_turn_id = turn_id agent._current_api_request_id = "" - # Tripwire: warn (with both turn ids) when this turn starts before the - # previous turn's turn-end persist — concurrent turns on one session - # interleave transcript writes. Cleared in _persist_session. + # Tripwire: warn when this turn starts before the previous turn-end persist + # (concurrent turns interleave transcript writes). Cleared in _persist_session. from agent.agent_runtime_helpers import note_turn_start note_turn_start(agent, turn_id) @@ -734,9 +556,8 @@ def build_turn_context( # NOTE: _turns_since_memory and _iters_since_skill are NOT reset here. agent.iteration_budget = IterationBudget(agent.max_iterations) - # Wall-clock run budget: per-run_conversation clock. Only stamped when a - # budget is configured so the default path stays clock-free; the wrap-up - # latch resets each turn (one notice per run, not per session). + # Wall-clock run budget: stamped only when configured; the wrap-up latch resets per + # turn (one notice per run). if getattr(agent, "run_budget_seconds", None): agent._run_budget_started_at = time.time() else: @@ -757,11 +578,8 @@ def build_turn_context( # Initialize conversation (copy to avoid mutating the caller's list). messages = list(conversation_history) if conversation_history else [] - # The CLI may already have staged this input outside the history passed to - # ``run_conversation``. Reuse it only when its clean transcript text matches - # this turn; a stale handoff from a failed prior turn must not replace a - # later, different user input. Voice turns compare against their explicit - # clean persistence override rather than the API-only prefixed payload. + # Reuse CLI-staged input only when its clean text matches this turn; a stale + # handoff must not replace later input. Voice turns compare the clean override. pending_cli_message = getattr(agent, "_pending_cli_user_message", None) expected_persist_content = ( persist_user_message if persist_user_message is not None else user_message @@ -771,9 +589,8 @@ def build_turn_context( and pending_cli_message.get("content") == expected_persist_content ): user_msg = pending_cli_message - # The CLI-staged value is the clean transcript text. Restore the - # API-facing variant (for example, a voice-mode prefix) while retaining - # the same dict and any close-path durable marker. + # CLI-staged value is the clean text; restore the API-facing variant (e.g. voice + # prefix) on the same dict, keeping any close-path durable marker. user_msg["content"] = user_message else: user_msg = stamp_message_timestamp( @@ -800,27 +617,16 @@ def build_turn_context( if agent._memory_nudge_interval > 0 and agent._turns_since_memory == 0: agent._turns_since_memory = prior_user_turns % agent._memory_nudge_interval - # Add the current user message after the prompt/session setup has made - # close persistence safe. The handoff above preserves any marker already - # stamped by an earlier close flush. - # - # A synthesized turn (auto-continue recovery note, delegation completion) - # declares how it should READ in a transcript. Stamp that on the live - # message so the crash persist below writes the row already typed. Typing - # it after the turn instead leaves the row untyped for the whole run — and - # forever if the turn crashes — so the raw system note paints as a user - # bubble. The model still receives role/content unchanged; the api_messages - # build strips both fields from every outgoing copy. + # Append the user message now that close persistence is safe. Synthesized turns + # stamp their transcript type so the crash persist writes a typed row; the model + # still receives role/content unchanged (api_messages strips both fields). if persist_user_display_kind: user_msg["display_kind"] = persist_user_display_kind if persist_user_display_metadata: user_msg["display_metadata"] = persist_user_display_metadata - # Stamp the platform-side message id (e.g. the Discord/Telegram message id) - # as metadata on the user turn so it survives the early crash-resilience - # persist below (the turn-start flush). Load-bearing for restart - # drain-window recovery: a recovery pass dedups via - # ``has_platform_message_id`` against this row. + # Stamp the platform message id so it survives the turn-start flush; restart + # drain-window recovery dedups via ``has_platform_message_id`` against this row. if persist_user_platform_id is not None: user_msg["platform_message_id"] = persist_user_platform_id append_message(messages, user_msg) @@ -855,9 +661,8 @@ def build_turn_context( should_review_memory = True agent._turns_since_memory = 0 - # Cosmetic side-signal: detect an affection "reaction" (ily / <3 / good bot) - # and notify the host so it can play hearts. Token-free, never touches the - # conversation, and never fatal — a purely optional UI beat. + # Cosmetic side-signal: detect an affection reaction so the host can play hearts. + # Token-free, never touches the conversation, never fatal. reaction_callback = getattr(agent, "reaction_callback", None) if reaction_callback is not None: try: @@ -882,12 +687,8 @@ def build_turn_context( active_system_prompt = agent._cached_system_prompt - # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" - # session on Bot-Mode-managed installs (same gate as the protocol - # section above). The gate is stable for a session's lifetime, so the - # tool list is byte-identical every turn: prompt-cache safe. Every - # other session (CLI, gateway chats, group-room member sessions, cron, - # subagents) fails the gate and never sees the schema. + # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" session + # (same gate as the protocol section); gate is session-stable, so cache-safe. try: from tools.bot_mode_dm import ensure_message_agent_tool @@ -895,17 +696,9 @@ def build_turn_context( except Exception: logger.debug("message_agent injection skipped", exc_info=True) - # Create the DB session row now that _cached_system_prompt is populated, so - # the persisted snapshot is written non-NULL on the first turn (Issue - # #45499). Idempotent: _ensure_db_session() no-ops once the row exists. - # Must run BEFORE preflight compression: in-place compaction inserts - # message rows referencing this session (archive_and_compact), and - # rotation creates a child with parent_session_id pointing at it — with - # PRAGMA foreign_keys=ON, a missing parent row fails both INSERTs on a - # fresh oversized first turn. The user-turn crash persist itself runs - # LATER (after memory prefetch / pre_llm_call), so the row is written - # once with its final api_content — both steps take the same per-agent - # persist lock as CLI close persistence. + # Create the DB row now (system prompt populated => non-NULL, #45499) and BEFORE + # preflight compression: compaction/rotation INSERTs reference this row under + # PRAGMA foreign_keys=ON. Idempotent; the user-turn crash persist runs later. persist_lock = getattr(agent, "_session_persist_lock", None) try: if persist_lock is None: @@ -920,33 +713,21 @@ def build_turn_context( exc_info=True, ) finally: - # Clear the staged CLI input eagerly (as the pre-refactor code did) - # so a crash in preflight compression — which runs between this row - # create and the late crash-persist below — doesn't leave a stale - # _pending_cli_user_message that the next turn would mistake for a - # fresh staged input. + # Clear staged CLI input eagerly so a crash in preflight compression doesn't + # leave a stale _pending_cli_user_message for the next turn. if not isinstance(pending_cli_message, dict) or pending_cli_message.get("_db_persisted"): agent._pending_cli_user_message = None # ── Idle-triggered compaction (opt-in; ``idle_compact_after_seconds``) ── - # When a session resumes after a long idle gap, compact the accumulated - # history up front so the rest of the conversation does not keep re-reading - # a large stale context on every turn. This fires on elapsed wall-clock time - # rather than size, so it complements (does not replace) the token-threshold - # preflight below. ``_last_activity_ts`` is the last time this turn loop did - # work; nothing has touched it yet this turn, so it measures the gap since - # the previous turn finished. The cheap gap pre-check gates the (more - # expensive) token estimate, mirroring ``_should_run_preflight_estimate``. + # Fires on wall-clock gap since ``_last_activity_ts``, complementing the token + # gate; a cheap gap check gates the estimate (cf. _should_run_preflight_estimate). _idle_after = getattr(agent, "compression_idle_compact_after_seconds", 0) if agent.compression_enabled and _idle_after > 0 and messages: _idle_gap = time.time() - getattr(agent, "_last_activity_ts", time.time()) if _idle_gap >= _idle_after: _compressor = agent.context_compressor - # Route-aware pressure (#96995/#97602 class): on a compacted - # native-Codex session the generic durable-history figure - # overstates the wire by orders of magnitude and would fire an - # idle compaction the next request never needed. Reuse the - # preflight estimator (anchor → native pruned → generic). + # Route-aware pressure: on compacted native-Codex sessions the durable + # figure overstates the wire; reuse the preflight estimator (#96995). _idle_tokens = _preflight_request_tokens( agent, messages, @@ -994,28 +775,21 @@ def build_turn_context( messages, system_message, approx_tokens=_idle_tokens, task_id=effective_task_id, ) - # ``_compress_context`` returns the INPUT list object when it - # skips (per-session lock held by another path, failure - # cooldown, anti-thrash breaker, codex-native routing). Only - # re-baseline + re-anchor after a real compaction — a skip - # must leave the turn's flush baseline and user-message index - # untouched. + # ``_compress_context`` returns the INPUT list object when it skips; + # only re-baseline and re-anchor after a real compaction. if messages is not _idle_input: conversation_history = conversation_history_after_compression( agent, messages, conversation_history ) - # Compaction rebuilt the list, so the index of this turn's - # just-appended user message is stale — re-anchor it the - # same way the preflight path does below. + # Compaction rebuilt the list; re-anchor this turn's user index. current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) agent._persist_user_message_idx = current_turn_user_idx # ── Preflight context compression ── - # Gate the (expensive) full token estimate behind a cheap pre-check. - # See ``_should_run_preflight_estimate`` for the OR semantics that fix - # issue #27405 (a few very large messages slipping past the count gate). + # Cheap pre-check gates the full estimate; see ``_should_run_preflight_estimate`` + # for the OR semantics (#27405). _preflight_compressed = False _preflight_compression_blocked = False agent._turn_received_provider_response = False @@ -1036,10 +810,8 @@ def build_turn_context( active_system_prompt or "", ) _compressor = agent.context_compressor - # getattr guard: minimal compressor doubles (SimpleNamespace in the - # engine-preflight tests) and plugin context engines lack this - # ContextCompressor-only method — absence means no snapshot, and the - # finalizer's rollback stays disarmed for the turn (display-only). + # getattr guard: compressor doubles and plugin engines lack this method — + # absence means no snapshot and the finalizer's rollback stays disarmed. _snapshot_fn = getattr( _compressor, "snapshot_preflight_display_tokens", None ) @@ -1073,13 +845,9 @@ def build_turn_context( ) if not _preflight_deferred: - # Display-only seed (see - # ContextCompressor.maybe_seed_preflight_display_tokens): a real - # provider reading always wins over the rough estimate, and the - # -1 post-compression sentinel (#36718) stays protected. On - # usage-less responses the seed also feeds the tool-loop - # compression gate — the one live path where an inflated seed - # could push compression below the user threshold. + # Display-only seed: a real provider reading wins and the -1 sentinel + # stays protected (#36718). Also feeds the tool-loop gate on usage-less + # responses. _maybe_seed = getattr( _compressor, "maybe_seed_preflight_display_tokens", None ) @@ -1123,12 +891,8 @@ def build_turn_context( else: _should_compress_now = _compressor.should_compress(_preflight_tokens) if not _should_compress_now: - # Context is over threshold but compression is blocked - # (summary-LLM cooldown or anti-thrashing). Ask should_compress_info - # for the human-readable reason so we can surface a warning below. - # getattr guard: minimal compressor doubles (SimpleNamespace in - # the engine-preflight tests) and older plugin engines lack the - # method — absence means no block reason, no warning. + # Over threshold but blocked: ask should_compress_info for the reason + # to surface below. getattr guard: doubles/older engines lack it. _info = getattr(_compressor, "should_compress_info", None) if callable(_info): try: @@ -1136,9 +900,8 @@ def build_turn_context( except Exception: _compress_block_reason = None if _should_compress_now: - # Managed local runtime: growing the window beats compressing — - # the ladder's design order (same seam as the conversation - # loop's pre-API gate; see _maybe_grow_local_window there). + # Managed local runtime: growing the window beats compressing (ladder + # order; same seam as _maybe_grow_local_window in the loop). try: from agent.conversation_loop import _maybe_grow_local_window @@ -1165,11 +928,8 @@ def build_turn_context( ) if _should_compress_now: _preflight_compressed = True - # Compression is actually running (block cleared / was never - # blocked) — reset the dedup so a future blocked-over-threshold - # turn can warn again. Real session boundary. - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Compression is actually running — reset the dedup so a future blocked + # turn can warn again. getattr guard: object.__new__ doubles lack it. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() @@ -1194,9 +954,8 @@ def build_turn_context( ) if _preflight_status: agent._emit_status(_preflight_status) - # Preflight passes honor the same configured per-turn cap - # (compression.max_attempts) as the loop's compression sites; - # default 3 preserves the prior hardcoded behavior. + # Preflight passes honor compression.max_attempts like the loop's sites + # (default 3). _max_preflight_passes = max( 1, int(getattr(agent, "max_compression_attempts", 3) or 3) ) @@ -1212,24 +971,17 @@ def build_turn_context( messages is _preflight_input and compression_skipped_due_to_lock(agent) ): - # #69870 lock-skip: another path holds this session's - # compression lock, so the pass no-oped. That is a - # temporary DEFER, not proof the transcript cannot - # compress — do NOT arm the insufficient-progress - # blocker (the loop's error handlers must keep their - # provider-proven retry budget) and stop preflight - # passes for this turn; the lock winner is shrinking - # the same session concurrently. + # Lock-skip (#69870): another path holds the lock, so this is a + # DEFER, not proof of incompressibility — don't arm the blocker; + # stop preflight passes for this turn. logger.info( "Preflight compression deferred: compression lock " "held by another path (session %s)", agent.session_id or "none", ) break - # Re-estimate now so size-only compression (same row count, - # lower token count — e.g. summarising tool outputs) is - # recognised as progress instead of being misread as - # "Cannot compress further". Fixes #39548. + # Re-estimate so size-only compression (same rows, fewer tokens) + # counts as progress (#39548). _preflight_tokens = _preflight_request_tokens( agent, messages, @@ -1265,29 +1017,21 @@ def build_turn_context( ) break elif _compress_block_reason: - # Context is already over the compression threshold, but compression - # is blocked (summary LLM cooldown or anti-thrashing). Without a - # signal the session keeps growing until the model silently stops - # answering — the conversation hits the hard provider token limit - # with no explanation. Surface a deduped warning so the user can - # take action (/new or /compress) instead of hitting a silent hang. + # Over threshold but compression blocked: surface a deduped warning so + # the user can /new or /compress instead of a silent provider limit. agent._warn_context_overflow_blocked( _compress_block_reason, _preflight_tokens, _compressor.threshold_tokens, ) else: - # Sub-threshold and unblocked — allow the overflow warning to fire - # again next time the context is over threshold but blocked. - # getattr guard: test doubles built via object.__new__ lack the - # method (gateway test-double pitfall) — treat absence as no-op. + # Sub-threshold and unblocked — re-arm the overflow warning. getattr guard: + # object.__new__ test doubles lack the method. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() - # Engine maintenance only when NO skip-branch fired: a failure - # cooldown, deferred estimate, or codex-native route must keep - # the engine hook un-consulted (#20316 contract — the cooldown - # exists precisely because compression recently failed). + # Engine maintenance only when NO skip-branch fired: cooldown, deferred + # estimate, or codex-native route keep the engine hook unconsulted (#20316). if _compression_cooldown or _preflight_deferred or _codex_native_auto: _engine_preflight = None else: @@ -1295,27 +1039,8 @@ def build_turn_context( _compressor, "should_compress_preflight", None ) # ── Engine-driven sub-threshold preflight maintenance (#20316) ── - # None of the threshold-path branches fired (not deferred, no - # failure cooldown, not codex-native, and should_compress() said - # the request is under pressure). Context engines that override - # ``should_compress_preflight()`` (e.g. LCM-style incremental - # leaf-chunk compaction) can still request deferred maintenance - # below the token threshold. The default - # ``ContextEngine.should_compress_preflight()`` returns False, so - # the built-in ``ContextCompressor`` path is byte-identical. - # - # Attempt-cap integration: the engine gets exactly ONE - # ``compress()`` pass per turn. It is mutually exclusive with the - # threshold multi-pass loop above (if/elif), so turn-start - # preflight passes stay bounded by the resolved - # ``compression.max_attempts`` cap (floor 1) in every case. - # - # No-op-blocking integration: a sub-threshold engine pass that - # no-ops says nothing about over-threshold compressibility, so it - # must neither set nor clear ``_preflight_compression_blocked`` - # (#64382) — and being in the ``else`` arm it can never run after - # the threshold loop has proven a retry ineffective. - # (resolved above, gated on no skip-branch having fired) + # Engines overriding ``should_compress_preflight()`` get exactly ONE + # ``compress()`` pass; a no-op never touches _preflight_compression_blocked. _wants_engine_preflight = False if callable(_engine_preflight): try: @@ -1342,12 +1067,8 @@ def build_turn_context( messages, system_message, approx_tokens=_preflight_tokens, task_id=effective_task_id, ) - # ``_compress_context`` returns the INPUT list object on every - # skip path (per-session lock held elsewhere, cooldown, - # anti-thrash breaker, codex-native routing) and an engine may - # legitimately no-op. Only re-baseline the flush history and - # re-anchor the user row after a REAL compaction — a skip must - # leave the turn's bookkeeping untouched. + # ``_compress_context`` returns the INPUT list on every skip path and an + # engine may no-op; re-baseline/re-anchor only after a REAL compaction. if messages is not _engine_input: _preflight_compressed = True conversation_history = conversation_history_after_compression( @@ -1359,16 +1080,8 @@ def build_turn_context( agent._last_content_tools_all_housekeeping = False agent._mute_post_response = False elif not agent.compression_enabled: - # Uncompressed session guard (#89297): when compression is explicitly - # disabled, sessions can grow past the model's context window across - # hundreds of messages with nothing to shrink them. The warning itself - # fires from the conversation loop's pre-API site, which reuses the - # unconditionally computed request estimate at zero marginal cost and - # covers both turn-start and mid-turn growth (every provider request - # passes through it). Here we only RE-ARM the dedup once the session - # is back under the window, so the guard can warn again after the - # user compacts (/compress with force=True works with compression - # disabled) and the context later regrows past the limit. + # Uncompressed session guard (#89297): the warning fires from the loop's + # pre-API site; here we only RE-ARM the dedup once back under the window. _ctx_len = getattr( getattr(agent, "context_compressor", None), "context_length", None ) @@ -1381,16 +1094,12 @@ def build_turn_context( if isinstance(_c, str): _raw_chars += len(_c) elif _c: - # Non-string, non-empty content (multimodal part lists, - # dict payloads) defeats a char count — force the real - # estimate by treating it as over-gate. None/"" (routine - # assistant tool-call rows) contribute nothing. + # Non-string, non-empty content defeats a char count — force the + # real estimate. None/"" contribute nothing. _raw_chars = _ctx_len + 1 break - # Cheap gate: a session whose raw text is under ~1/4 of the - # window (4 chars/token upper bound) cannot be over it — skip - # the estimator. Non-string (multimodal) content defeats a char - # count, so any such message forces the real estimate. + # Cheap gate: raw text under ~1/4 of the window (4 chars/token) cannot + # be over it; non-string (multimodal) content forces the real estimate. if _raw_chars <= _ctx_len: _clear_warn = getattr( agent, "_clear_context_overflow_warn", None @@ -1398,12 +1107,9 @@ def build_turn_context( if callable(_clear_warn): _clear_warn() else: - # Route-aware (#96995/#97602 class): the warn site in the - # conversation loop now measures the checkpoint-pruned wire - # payload on native-Codex sessions, so the re-arm must use - # the same figure — otherwise a compacted session that fits - # on the wire never clears the dedup and future genuine - # overflow warnings stay suppressed. + # Re-arm with the same route-aware (checkpoint-pruned wire) figure the + # warn site measures, else a compacted session never clears the dedup + # and genuine overflow warnings stay suppressed (#96995/#97602). _uncompressed_tokens = _preflight_request_tokens( agent, messages, @@ -1417,13 +1123,9 @@ def build_turn_context( _clear_warn() if _preflight_compressed: - # Compression rebuilt the list (tail messages are fresh compaction - # copies), so the pre-compression index of this turn's user message - # is stale. Re-anchor both index trackers: the api_content stamp - # below, the loop's injection site, and the flush's persist-override - # row (#48677) must all target the surviving dict, not a stale - # position. Exact-content match first so a todo-snapshot user message - # appended after the tail can't steal the anchor. + # Compression rebuilt the list, so the pre-compression user index is stale. + # Re-anchor so the api_content stamp, injection site, and persist-override row + # hit the same dict; exact-content match first so a todo-snapshot can't steal it current_turn_user_idx = reanchor_current_turn_user_idx( messages, user_message ) @@ -1447,9 +1149,8 @@ def build_turn_context( sender_id=getattr(agent, "_user_id", None) or "", ) _ctx_parts: list[str] = [] - # Spill oversized per-hook context to disk so a runaway plugin - # can't inflate every subsequent turn's prompt. Ported from - # openai/codex PR #21069 ("Spill large hook outputs from context"). + # Spill oversized per-hook context to disk so a runaway plugin can't inflate + # every subsequent turn's prompt. try: from tools.hook_output_spill import ( get_spill_config as _spill_cfg, @@ -1483,12 +1184,9 @@ def build_turn_context( except Exception as exc: logger.warning("pre_llm_call hook failed: %s", exc) - # Gateway must-deliver notes (auto-reset note, first-contact intro, - # voice-channel change) ride the same user-message injection channel as - # plugin context so the ephemeral system prompt can stay byte-stable. - # One-shot: staged by the gateway right before this turn, consumed here. - # Multimodal (list) content can't take the string sidecar — append a - # durable text part instead of dropping the fact. + # Gateway must-deliver notes ride the user-message injection channel (one-shot, + # gateway-staged) so the ephemeral system prompt stays byte-stable. Multimodal + # (list) content can't take the string sidecar — append a durable text part. _gateway_notes = consume_gateway_turn_context_notes(agent) if _gateway_notes: _gw_turn_content = ( @@ -1538,10 +1236,8 @@ def build_turn_context( except Exception: pass - # External memory provider: prefetch once before the tool loop. - # - # Skip prefetch on trivial prompts (greetings, acknowledgements) to - # prevent memory-context injection on turns that carry no semantic signal. + # External memory provider: prefetch once before the tool loop. Skipped on + # trivial prompts (greetings, acks) that carry no semantic signal. ext_prefetch_cache = "" if agent._memory_manager: try: @@ -1550,10 +1246,8 @@ def build_turn_context( ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or "" except Exception: pass - # Deterministic, model-independent recall indicator: when memory was - # actually injected this turn, tell the user — don't rely on the model - # to surface it. Rendered by Hermes (via _emit_status), so it always - # shows and can't be silently dropped by the model. + # Deterministic recall indicator: rendered by Hermes via _emit_status when + # memory was injected, so the model can't silently drop it. if ext_prefetch_cache: try: _recall_indicator = agent._memory_manager.describe_recall() @@ -1563,22 +1257,8 @@ def build_turn_context( pass # ── api_content sidecar: persist what you send ── - # The prefetch/plugin context above is injected into the API copy of this - # turn's user message, never into the stored content — so on the next - # turn the message would replay WITHOUT the injection, diverging the - # request prefix at this point and re-prefilling everything after it - # (the whole previous turn's assistant/tool chain). Stamp the exact - # API-bound bytes on the live dict, only when they differ from the clean - # content, so the crash persist below writes both in the same row and - # replay can reproduce the sent prefix byte-for-byte. Guarded by the - # same predicate the api_messages build uses, so the stamped bytes are - # exactly the bytes the loop sends. codex_app_server turns bypass the - # api_messages build entirely (the codex thread gets the plain user - # message), so stamping there would persist bytes that were never sent. - # MoA turns append per-call aggregated reference context to the same API - # copy AFTER this composition, so the stamped bytes would never match the - # wire either — skip the stamp rather than persist provably wrong "exact - # sent bytes" (MoA keeps its pre-sidecar cache behavior). + # Injected context lives only in the API copy; stamp the exact sent bytes on the + # live dict so replay reproduces the prefix. Skipped for codex_app_server/MoA. if ( not moa_active and getattr(agent, "api_mode", None) != "codex_app_server" @@ -1591,14 +1271,9 @@ def build_turn_context( ) if _api_content is not None and _api_content != _turn_user_msg.get("content"): _turn_user_msg["api_content"] = _api_content - # In-place preflight compaction has ALREADY inserted this turn's - # user row (archive_and_compact runs before prefetch/pre_llm_call - # can compose the sidecar), and the crash persist below identity- - # skips every compacted dict (they are all in the rebound - # conversation_history) — so the stamp would never reach the DB. - # Backfill it onto the freshly-inserted row directly. Rotation - # mode needs nothing here: its compacted copies flush to the - # child session after this stamp. + # In-place preflight compaction already inserted this turn's user row and + # the crash persist identity-skips compacted dicts, so backfill the stamp + # onto the row directly. Rotation mode flushes to the child session later. if _preflight_compressed and bool( getattr(agent, "_last_compaction_in_place", False) ): @@ -1618,13 +1293,9 @@ def build_turn_context( exc_info=True, ) - # Crash-resilience: persist the inbound user turn before the first LLM - # call. Runs after preflight compression (which rewrites history anyway) - # and after prefetch/pre_llm_call, so the user row is written once with - # its final api_content instead of being re-written mid-turn. - # Keep row creation and the marker-based append in the same per-agent - # critical section as CLI close persistence, and retry the row create if - # the pre-compression attempt above failed transiently. + # Crash-resilience: persist the inbound user turn once, with final api_content, + # before the first LLM call. Same critical section as CLI close persistence; + # retries the row create if the pre-compression attempt failed transiently. def _ensure_and_persist() -> None: agent._ensure_db_session() agent._persist_session(messages, conversation_history) @@ -1642,20 +1313,13 @@ def build_turn_context( exc_info=True, ) finally: - # Keep an unmarked staged input available to a later close retry if the - # normal persistence attempt failed. Once the marker is present, the - # close path must no longer treat it as a pre-worker UI input. + # Keep an unmarked staged input for a later close retry if persistence failed; + # once marked, the close path must not treat it as a pre-worker UI input. if not isinstance(pending_cli_message, dict) or pending_cli_message.get("_db_persisted"): agent._pending_cli_user_message = None - # Title the session from this user message, now — the row exists and the - # turn has not called the model yet. Titling is derived from the user's - # ask alone, so it runs concurrently with the turn instead of waiting for - # a final response; on a long tool-heavy first turn that is the difference - # between a title in ~1s and a title minutes later (or never, when the - # turn failed before producing one). Fire-and-forget on a daemon thread, - # a no-op once the session has a title, and shared by every surface - # because every surface enters the turn through this prologue. + # Title the session now: the row exists and titling depends only on the user's + # ask, so it runs concurrently with the turn. Daemon thread, no-op once titled. _maybe_title_session_at_turn_start(agent, messages) return TurnContext( @@ -1672,3 +1336,128 @@ def build_turn_context( ext_prefetch_cache=ext_prefetch_cache, preflight_compression_blocked=_preflight_compression_blocked, ) + + +def build_api_messages( + agent: Any, + messages: List[Dict[str, Any]], + *, + current_turn_user_idx: Any, + ext_prefetch_cache: Any, + plugin_user_context: Any, + moa_config: Any, + active_system_prompt: Any, +) -> Tuple[List[Dict[str, Any]], str]: + """Build the wire copy of ``messages`` for one API call plus the effective system + message. Returns ``(api_messages, effective_system)``. + + Prompt-cache invariant: historical user/assistant rows replay their ``api_content`` + sidecar (the exact bytes sent live) so the prefix stays byte-stable; the current + user turn reuses the prologue's stamp (or composes live when a caller bypassed the + prologue). Ephemeral context (prefetch, ``pre_llm_call`` hooks, ``ephemeral_system_prompt``) + is added at API time only — ``messages`` stays untouched beyond the sidecar stamp, + and the system prompt is built ONCE per session and replayed verbatim.""" + from agent.agent_runtime_helpers import fill_empty_non_final_wire_payload + from agent.conversation_loop import _clone_message_for_send + + _ext_prefetch_cache = ext_prefetch_cache + _plugin_user_context = plugin_user_context + api_messages = [] + for idx, msg in enumerate(messages): + + # Structural clone, NOT msg.copy(): in-place transforms below must not reach + # persisted history via nested containers; see _clone_message_for_send. + api_msg = _clone_message_for_send(msg) + + # api_content is the persistence sidecar of the exact bytes sent to the API; + # bookkeeping, never a provider field — pop it from EVERY outgoing copy. + _api_content = api_msg.pop("api_content", None) + + # Display-only timeline metadata, never a provider field: strict OpenAI + # backends reject unknown keys once a typed event row enters live history. + api_msg.pop("display_kind", None) + api_msg.pop("display_metadata", None) + + # Durable row id from _rows_to_conversation (desktop reactions); only the + # chat-completions transport strips underscore keys, so drop it centrally. + api_msg.pop("_row_id", None) + + # Inject ephemeral context (memory prefetch + pre_llm_call user hooks) + # at API time only; `messages` is untouched beyond the api_content stamp. + if idx == current_turn_user_idx and msg.get("role") == "user": + if isinstance(_api_content, str) and _api_content: + # Reuse the prologue's stamp so sidecar and wire cannot drift + # and every pass this turn sends identical bytes. + api_msg["content"] = _api_content + else: + # Callers that bypass the prologue stamping: compose live. + _composed = compose_user_api_content( + api_msg.get("content", ""), + _ext_prefetch_cache, + _plugin_user_context, + ) + if _composed is not None: + api_msg["content"] = _composed + elif ( + isinstance(_api_content, str) + and _api_content + and msg.get("role") in ("user", "assistant") + ): + # Historical row: replay the exact bytes sent live so the prompt-cache + # prefix stays byte-stable. User rows carry the injection sidecar; user + # and assistant rows may carry a sanitize-divergence sidecar. + api_msg["content"] = _api_content + + # For ALL assistant messages, pass reasoning back to the API + # This ensures multi-turn reasoning context is preserved + agent._copy_reasoning_content_for_api(msg, api_msg) + + # Remove 'reasoning' field - it's for trajectory storage only + # We've copied it to 'reasoning_content' for the API above + if "reasoning" in api_msg: + api_msg.pop("reasoning") + # Remove finish_reason - not accepted by strict APIs (e.g. Mistral) + if "finish_reason" in api_msg: + api_msg.pop("finish_reason") + # Fill empty non-final user/assistant wire copies so the pre-call sanitizer + # stops re-healing and flooding errors.log; durable history is untouched. + # After the reasoning copy so thinking-only turns keep payload (#96870). + fill_empty_non_final_wire_payload( + api_msg, is_final=(idx == len(messages) - 1) + ) + # _thinking_prefill survives intentionally: the drop pass below needs it. + # Strip length-continuation marks; some transports keep underscore keys. + api_msg.pop("_length_continuation_fragment", None) + api_msg.pop("_length_continuation_nudge", None) + # Strip Codex Responses fields (call_id, response_item_id): strict providers + # reject unknown fields. New dicts keep the internal list intact for Codex. + if agent._should_sanitize_tool_calls(): + # In MoA mode agent.model is the virtual preset name; use the resolved + # aggregator so Gemini keeps thought_signature (extra_content). + _sanitize_model = agent.model + if agent.provider == "moa": + if moa_config: + _agg = moa_config.get("aggregator") or {} + if _agg.get("model"): + _sanitize_model = _agg["model"] + if _sanitize_model == agent.model: + # Virtual-provider mode: no moa_config is threaded through; ask + # the facade for the aggregator slot from the previous create(). + _moa_client = getattr(agent, "client", None) + _agg_slot = getattr(_moa_client, "last_aggregator_slot", None) + if _agg_slot and _agg_slot.get("model"): + _sanitize_model = _agg_slot["model"] + agent._sanitize_tool_calls_for_strict_api(api_msg, model=_sanitize_model) + # Keep 'reasoning_details' - OpenRouter uses this for multi-turn reasoning context + # The signature field helps maintain reasoning continuity + api_messages.append(api_msg) + + # Final system message = cached prompt + ephemeral additions (API-time only). + # Plugin/recall context goes into the user message, never the system prompt: the + # prompt is built ONCE per session and replayed verbatim (stable cache prefix). + effective_system = active_system_prompt or "" + if agent.ephemeral_system_prompt: + effective_system = (effective_system + "\n\n" + agent.ephemeral_system_prompt).strip() + if effective_system: + api_messages = [{"role": "system", "content": effective_system}] + api_messages + return api_messages, effective_system diff --git a/agent/turn_empty_response.py b/agent/turn_empty_response.py new file mode 100644 index 0000000000..a83a1b557f --- /dev/null +++ b/agent/turn_empty_response.py @@ -0,0 +1,377 @@ +"""Empty / thinking-only final-response recovery ladder for the conversation turn loop. + +Extracted from ``run_conversation``. Runs when the model returned no visible text after +```` blocks. Ladder order is load-bearing: partial-stream recovery → reuse prior +turn content (housekeeping tools only) → one post-tool-call nudge (#9400) → thinking-only +prefill continuation (×2) → empty-response retries (budgeted, deterministic-empty +short-circuit) → fallback provider → terminal ``(empty)`` sentinel. Nothing here imports +``agent.conversation_loop`` at module level (cycle); loop-internal helpers resolve lazily. +""" + +from __future__ import annotations + +import logging +import re +from dataclasses import dataclass +from typing import Any, Dict, List, Optional + +from agent import empty_response_guard as _empty_guard +from agent.message_metadata import append_message +from agent.turn_recovery import interruptible_backoff_sleep + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class EmptyResponseVerdict: + """Outcome of ``recover_empty_response``. + + ``action``: ``"break"`` (turn is done — ``final_response`` is set), ``"continue"`` + (re-enter the OUTER turn loop: a nudge/prefill row was appended, a retry wait + elapsed, or a fallback was activated and preflight must re-run), ``"return"`` + (interrupted during a retry wait — return ``result``) or ``"fallthrough"`` + (unreachable: every path exits; kept for the contract).""" + + action: str + result: Optional[Dict[str, Any]] + final_response: Any + turn_exit_reason: Any + active_system_prompt: Any + preflight_compression_blocked: bool + + +def recover_empty_response( + agent: Any, + assistant_message: Any, + response: Any, + finish_reason: str, + *, + final_response: Any, + messages: List[Dict[str, Any]], + api_messages: Any, + conversation_history: Any, + active_system_prompt: Any, + api_call_count: int, + turn_exit_reason: Any, + preflight_compression_blocked: bool, +) -> EmptyResponseVerdict: + """Recover from a final response with no visible content (see module docstring for + the ladder). Role alternation is preserved: the post-tool nudge appends the empty + assistant row BEFORE the user-level hint (APIs reject tool→user). Reasoning is + surfaced only at the terminal step, for delivery — the persisted row keeps the + ``(empty)`` sentinel.""" + from agent.conversation_loop import ( + _EMPTY_TOOL_RESPONSE_NUDGE, + _sync_failover_system_message, + jittered_backoff, + ) + + _turn_exit_reason = turn_exit_reason + _preflight_compression_blocked = preflight_compression_blocked + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> EmptyResponseVerdict: + return EmptyResponseVerdict( + action=action, + result=result, + final_response=final_response, + turn_exit_reason=_turn_exit_reason, + active_system_prompt=active_system_prompt, + preflight_compression_blocked=_preflight_compression_blocked, + ) + + # Partial stream recovery: content streamed before the connection + # died becomes the final response instead of fallback or retries. + _partial_streamed = ( + getattr(agent, "_current_streamed_assistant_text", "") or "" + ) + if agent._has_content_after_think_block(_partial_streamed): + _turn_exit_reason = "partial_stream_recovery" + _recovered = agent._strip_think_blocks(_partial_streamed).strip() + logger.info( + "Partial stream content delivered (%d chars) " + "— using as final response", + len(_recovered), + ) + agent._emit_status( + "↻ Stream interrupted — using delivered content " + "as final response" + ) + final_response = _recovered + # A streamed fragment isn't a confirmed preview: keep + # response_previewed false so gateway fallback delivery can + # send the text plus the abnormal-turn explanation. + agent._response_was_previewed = False + return _verdict("break") + + # Prior turn had real content + ONLY housekeeping tools: model is + # done, reuse it. With substantive tools it was mid-task narration + # and the empty reply is a choke; let the post-tool nudge handle it. + fallback = getattr(agent, '_last_content_with_tools', None) + if fallback and getattr(agent, '_last_content_tools_all_housekeeping', False): + _turn_exit_reason = "fallback_prior_turn_content" + logger.info("Empty follow-up after tool calls — using prior turn content as final response") + agent._emit_status("↻ Empty response after tool calls — using earlier content as final answer") + agent._last_content_with_tools = None + agent._last_content_tools_all_housekeeping = False + agent._empty_content_retries = 0 + # Do NOT modify the assistant message content (injected text + # poisoned history); use the fallback as the response and break. + final_response = agent._strip_think_blocks(fallback).strip() + agent._response_was_previewed = True + return _verdict("break") + + # ── Post-tool-call empty response nudge ─────────── + # Empty after tool results (no prior content, or only mid-task + # narration): nudge once via a user-level hint. (#9400) + _prior_was_tool = any( + m.get("role") == "tool" + for m in messages[-5:] # check recent messages + ) + # Ollama puts in content, not reasoning_content, so + # _has_structured misses it; detect here to route to prefill. + _has_inline_thinking = bool( + re.search( + r'||', + final_response or "", + re.IGNORECASE, + ) + ) + if ( + _prior_was_tool + and not getattr(agent, "_post_tool_empty_retried", False) + and not _has_inline_thinking # thinking model still working — let prefill handle + ): + agent._post_tool_empty_retried = True + # Clear stale narration so it doesn't resurface + # on a later empty response after the nudge. + agent._last_content_with_tools = None + agent._last_content_tools_all_housekeeping = False + logger.info( + "Empty response after tool calls — nudging model " + "to continue processing" + ) + agent._buffer_status( + "⚠️ Model returned empty after tool calls — " + "nudging to continue" + ) + # Append the empty assistant first so the sequence stays valid: + # tool → assistant("(empty)") → user (APIs reject tool→user). + _nudge_msg = agent._build_assistant_message(assistant_message, finish_reason) + _nudge_msg["content"] = "(empty)" + _nudge_msg["_empty_recovery_synthetic"] = True + append_message(messages, _nudge_msg) + append_message(messages, { + "role": "user", + "content": _EMPTY_TOOL_RESPONSE_NUDGE, + "_empty_recovery_synthetic": True, + }) + return _verdict("continue") + + # ── Thinking-only prefill continuation ────────── + # Reasoning but no text: append as-is and continue so the model sees + # its own reasoning and writes text. Covers _has_inline_thinking. + _has_structured = bool( + getattr(assistant_message, "reasoning", None) + or getattr(assistant_message, "reasoning_content", None) + or getattr(assistant_message, "reasoning_details", None) + or _has_inline_thinking + ) + if _has_structured and agent._thinking_prefill_retries < 2: + agent._thinking_prefill_retries += 1 + logger.info( + "Thinking-only response (no visible content) — " + "prefilling to continue (%d/2)", + agent._thinking_prefill_retries, + ) + agent._buffer_status( + f"↻ Thinking-only response — prefilling to continue " + f"({agent._thinking_prefill_retries}/2)" + ) + interim_msg = agent._build_assistant_message( + assistant_message, "incomplete" + ) + interim_msg["_thinking_prefill"] = True + append_message(messages, interim_msg) + agent._session_messages = messages + return _verdict("continue") + + # ── Empty response retry ────────────────────── + # Retry up to 3 times before fallback; covers truly empty replies + # AND reasoning-only replies after prefill exhaustion. + _truly_empty = not agent._strip_think_blocks( + final_response + ).strip() + _prefill_exhausted = ( + _has_structured + and agent._thinking_prefill_retries >= 2 + ) + _empty_candidate = _truly_empty and ( + not _has_structured or _prefill_exhausted + ) + if _empty_candidate: + # Each empty attempt re-bills the full input; record its + # signature so deterministic empties stop burning paid retries. + # Fails open: missing usage or any output keeps the budget. + _empty_guard.record_empty_attempt( + agent, + finish_reason=finish_reason, + response=response, + ) + _empty_retry_budget = ( + _empty_guard.empty_retry_budget(agent, response) + if _empty_candidate + else _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET + ) + _deterministic_empty = _empty_candidate and ( + _empty_guard.deterministic_empty(agent) + ) + if ( + _empty_candidate + and agent._empty_content_retries < _empty_retry_budget + and not _deterministic_empty + ): + agent._empty_content_retries += 1 + wait_time = jittered_backoff( + agent._empty_content_retries, + base_delay=5.0, + max_delay=60.0, + ) + logger.warning( + "Empty response (no content or reasoning) — " + "retry %d/%d in %.1fs (model=%s)", + agent._empty_content_retries, + _empty_retry_budget, wait_time, agent.model, + ) + _budget_note = ( + " — high-cost request, reduced retry budget" + if _empty_retry_budget < _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET + else "" + ) + agent._buffer_status( + f"⚠️ Empty response from model — retrying " + f"({agent._empty_content_retries}/{_empty_retry_budget}) " + f"in {wait_time:.0f}s{_budget_note}" + ) + _interrupted = interruptible_backoff_sleep( + agent, wait_time, None, + messages=messages, + conversation_history=conversation_history, + api_call_count=api_call_count, + abort_message="Interrupt detected during empty-response retry wait, aborting.", + interrupt_text=( + f"Operation interrupted: retrying empty response from model " + f"(retry {agent._empty_content_retries}/{_empty_retry_budget})." + ), + activity_label=f"empty response retry backoff ({agent._empty_content_retries}/{_empty_retry_budget})", + ) + if _interrupted is not None: + return _verdict("return", _interrupted) + return _verdict("continue") + + if _truly_empty and _deterministic_empty: + logger.warning( + "Deterministic empty response detected " + "(consecutive zero-output completions, " + "model=%s provider=%s finish_reason=%s) — " + "skipping remaining retries", + agent.model, agent.provider, finish_reason, + ) + agent._buffer_status( + "⚠️ Model is deterministically returning empty " + "(zero output tokens) — skipping further retries " + "to avoid repeat charges" + ) + + # ── Exhausted retries — try fallback provider ── + # Before "(empty)", switch to the next provider in the chain. + if _truly_empty and agent._fallback_chain: + logger.warning( + "Empty response after %d retries — " + "attempting fallback (model=%s, provider=%s)", + agent._empty_content_retries, agent.model, + agent.provider, + ) + agent._buffer_status( + "⚠️ Model returning empty responses — " + "switching to fallback provider..." + ) + if agent._try_activate_fallback(): + active_system_prompt = _sync_failover_system_message( + agent, api_messages, active_system_prompt) + agent._empty_content_retries = 0 + agent._buffer_status( + f"↻ Switched to fallback: {agent.model} " + f"({agent.provider})" + ) + logger.info( + "Fallback activated after empty responses: " + "now using %s on %s", + agent.model, agent.provider, + ) + # OUTER loop: `continue` re-runs preflight against the + # fallback's window; `break` would end the turn without + # calling the fallback. Clear the preflight block. (#84733) + _preflight_compression_blocked = False + return _verdict("continue") + + # Retries and fallback exhausted — fall through to "(empty)". + # Surface the buffered retry trace and, if known, what the empty + # streak cost (each attempt re-billed the full input). + _streak_cost = _empty_guard.streak_cost_usd(agent) + if _streak_cost is not None: + agent._buffer_status( + f"ℹ️ Estimated cost of these empty attempts: " + f"~${_streak_cost:.2f} (input tokens are billed " + f"per attempt even when no answer is produced)" + ) + agent._flush_status_buffer() + _turn_exit_reason = "empty_response_exhausted" + reasoning_text = agent._extract_reasoning(assistant_message) + agent._drop_trailing_empty_response_scaffolding(messages) + assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) + assistant_msg["content"] = "(empty)" + # Gateway failure sentinel, not content: persisting it lets later + # "continue" turns replay assistant("(empty)") and loop on empties. + assistant_msg["_empty_terminal_sentinel"] = True + append_message(messages, assistant_msg) + + if reasoning_text: + reasoning_preview = reasoning_text[:500] + "..." if len(reasoning_text) > 500 else reasoning_text + logger.warning( + "Reasoning-only response (no visible content) " + "after exhausting retries and fallback. " + "Reasoning: %s", reasoning_preview, + ) + agent._emit_status( + "⚠️ Model produced reasoning but no visible " + "response after all retries. Returning empty." + ) + else: + logger.warning( + "Empty response (no content or reasoning) " + "after %d retries. No fallback available. " + "model=%s provider=%s", + agent._empty_content_retries, agent.model, + agent.provider, + ) + agent._emit_status( + "❌ Model returned no content after all retries" + + (" and fallback attempts." if agent._fallback_chain else + ". No fallback providers configured.") + ) + + # Delivery-only: show labeled reasoning instead of bare "(empty)" + # when the model thought but wrote no text. The persisted row keeps + # the sentinel; reasoning is never promoted earlier in the ladder. + if reasoning_text: + final_response = ( + "⚠️ The model produced only internal reasoning and " + "no final answer, despite retries" + + (" and fallback" if agent._fallback_chain else "") + + ". Its last reasoning, which may contain the " + "answer:\n\n" + reasoning_preview + ) + else: + final_response = "(empty)" + return _verdict("break") + return _verdict("fallthrough") diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index 93506f680c..ca52cff981 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -1,24 +1,9 @@ """Post-loop turn finalization for ``run_conversation``. -Extracted from ``agent/conversation_loop.py`` as part of the god-file -decomposition campaign (``~/.hermes/plans/god-file-decomposition.md``, Phase 1 -step 4 — the post-loop ``TurnFinalizer`` seam). ``run_conversation``'s tail -(everything after the main tool-calling ``while`` loop) is lifted here verbatim: -budget-exhaustion summary, trajectory save, session persist, turn diagnostics, -response transforms, result-dict assembly, steer drain, and the memory/skill -review trigger. - -Behavior-neutral: the body is moved unchanged. All ``agent.*`` side effects fire -exactly as before; only the post-loop *locals* are passed in as keyword args, and -the assembled ``result`` dict is returned to ``run_conversation`` which returns it -to the caller. The function is synchronous with a single return — mirroring the -region it replaces (no awaits, no early returns). - -Module ``logger`` is imported lazily inside the body (``from -agent.conversation_loop import logger``) so this module never imports -``agent.conversation_loop`` at import time -> no import cycle, and the log records -keep the exact logger name (``"agent.conversation_loop"``). -""" +Lifted verbatim: budget summary, trajectory save, persist, diagnostics, response +transforms, result assembly, steer drain, memory/skill review. Synchronous, single +return. ``logger`` is imported lazily from ``agent.conversation_loop`` (no cycle, +same logger name).""" from __future__ import annotations @@ -54,11 +39,9 @@ def _fill_assistant_tail_content(agent, tail: dict, final_response) -> None: agent._db_flush_scan_prefix = None -# Verification continuation scaffolding flags: verify-on-stop / pre_verify -# inject a synthetic user nudge to keep the agent going one more turn. -# These nudges must be stripped from returned/live history to avoid -# role-alternation breaks and poisoning the resumed transcript. The -# assistant response is real content and is not flagged. (#65919 §7) +# Verification-continuation nudges (verify-on-stop / pre_verify) must be stripped from +# returned/live history to avoid role-alternation breaks; the assistant response is +# real content and is not flagged. (#65919) _VERIFICATION_CONTINUATION_FLAGS = ( "_verification_stop_synthetic", "_pre_verify_synthetic", @@ -71,14 +54,10 @@ def _record_kanban_budget_exhausted( max_iterations: int, logger: logging.Logger, ) -> None: - """Record a terminal ``timed_out`` outcome for a kanban worker that - exhausted its iteration budget. + """Record a terminal ``timed_out`` outcome for a kanban worker out of budget. - This is a bounded fallback (#87096): the CAS invariant in ``_end_run`` - (``WHERE ended_at IS NULL``) guarantees idempotence — if another path - already closed the run this is a no-op — so it is safe to call from - multiple exit paths. - """ + Idempotent via the ``_end_run`` CAS (``WHERE ended_at IS NULL``): a no-op if + another path already closed the run, so safe from multiple exit paths (#87096).""" try: from hermes_cli import kanban_db as _kb _conn = _kb.connect() @@ -116,10 +95,8 @@ def _record_kanban_budget_exhausted( def _drop_verification_continuation_scaffolding(messages) -> None: """Remove verification-continuation nudge messages from *messages* in place. - Only the synthetic nudges carry these flags, so this strips just the - nudges while preserving the real attempted-final-answer that was - persisted to state.db. - """ + Only the synthetic nudges carry these flags, so the real attempted-final-answer + persisted to state.db survives.""" messages[:] = [ m for m in messages if not (isinstance(m, dict) and any(m.get(f) for f in _VERIFICATION_CONTINUATION_FLAGS)) @@ -153,11 +130,7 @@ def finalize_turn( _pending_verification_response=None, _pending_verification_response_previewed=False, ): - """Run the post-loop finalization and return the turn ``result`` dict. - - Lifted verbatim from ``run_conversation`` (the region after the main agent - loop). See module docstring. - """ + """Run the post-loop finalization and return the turn ``result`` dict.""" from agent.conversation_loop import logger budget_exhausted = ( @@ -179,24 +152,19 @@ def finalize_turn( iteration_limit_fallback = False preserved_verification_fallback = False if continuation_budget_exhausted: - # A verification/continuation gate deliberately withheld a composed - # answer, then consumed the remaining budget before producing a newer - # one. Preserve that exact answer instead of replacing it with another - # fallible model call. The explicit pending value is the provenance - # guard: unrelated error/recovery exits can never enter this branch. + # A verification gate withheld a composed answer, then the budget ran out: + # preserve it rather than make another fallible call. The explicit pending + # value is the provenance guard; unrelated error exits never enter here. final_response = _pending_verification_response - # Mark the turn as previewed only when the reused candidate was - # actually streamed to the user as interim content. (#65919 review: - # response-loss blocker) + # Previewed only if the reused candidate was actually streamed as interim. if _pending_verification_response_previewed: agent._response_was_previewed = True _turn_exit_reason = f"max_iterations_reached({api_call_count}/{agent.max_iterations})" iteration_limit_fallback = True preserved_verification_fallback = True elif final_response is None and budget_fallback_eligible: - # Budget exhausted — ask the model for a summary via one extra - # API call with tools stripped. _handle_max_iterations injects a - # user message and makes a single toolless request. + # Budget exhausted: _handle_max_iterations makes one extra toolless request + # for a summary. _turn_exit_reason = f"max_iterations_reached({api_call_count}/{agent.max_iterations})" agent._emit_status( f"⚠️ Iteration budget exhausted ({api_call_count}/{agent.max_iterations}) " @@ -211,29 +179,18 @@ def finalize_turn( iteration_limit_fallback = True if iteration_limit_fallback: - # If running as a kanban worker, signal the dispatcher that the - # worker could not complete (rather than treating it as a - # protocol violation). This applies whether the user-facing fallback - # came from the summary call or an explicitly pending continuation; - # both exhausted the task budget and must advance the failure circuit. - # - # We route through ``_record_task_failure(outcome="timed_out")`` - # rather than ``kanban_block`` so this counts toward the dispatcher's - # consecutive-failure circuit breaker (#29747 gap 2). + # Kanban worker: signal the dispatcher the worker could not complete. Route + # via ``_record_task_failure(outcome="timed_out")`` (not ``kanban_block``) so + # it counts toward the consecutive-failure circuit breaker (#29747). _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if _kanban_task: _record_kanban_budget_exhausted( _kanban_task, api_call_count, agent.max_iterations, logger, ) elif budget_exhausted: - # Bounded fallback (#87096): budget was exhausted but none of the - # normal fallback paths were eligible (interrupted / failed / - # anomalous exit_reason). If running as a kanban worker we must - # still record a terminal outcome so the task does not remain in - # an ambiguous lifecycle state. The worker's run is closed via - # ``_record_task_failure`` (compare-and-swap receipt path) which - # is a no-op if another path closed it — the CAS invariant in - # ``_end_run`` (``WHERE ended_at IS NULL``) guarantees idempotence. + # Bounded fallback: budget exhausted with no eligible fallback path. A kanban + # worker must still record a terminal outcome; the ``_end_run`` CAS makes it + # idempotent if another path already closed the run (#87096). _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if _kanban_task: _record_kanban_budget_exhausted( @@ -251,18 +208,9 @@ def finalize_turn( ) ) - # Preflight can seed the display count before the provider receives the - # request. Roll that estimate back only when an interrupt wins the race - # before any successful provider response. Compaction state remains owned - # by the real-usage/post-compaction path, including its ``-1`` sentinel. - # Guard rules (test-double density on this path is high): - # - snapshot is type-pinned to a real int — MagicMock agents auto-create - # truthy Mock attributes that must never arm the rollback; - # - the received-response flag is pinned to ``is not True`` — its real - # domain is True/False, and only a literal True means a provider - # response completed; - # - the compressor method gets a getattr+callable guard — SimpleNamespace - # compressor doubles and plugin context engines lack it. + # Roll back the preflight-seeded display count only when an interrupt wins + # before any provider response; compaction state (incl. ``-1``) stays with the + # real-usage path. Type-pinned guards keep MagicMock/SimpleNamespace doubles inert. _preflight_snapshot = getattr( agent, "_turn_preflight_display_snapshot", None ) @@ -281,17 +229,9 @@ def finalize_turn( if callable(_rollback_fn): _rollback_fn(_preflight_snapshot) - # Post-loop cleanup must never lose the response. Trajectory save, - # resource teardown, and session persistence all touch fallible - # surfaces — file I/O / JSON serialization (_save_trajectory), remote - # VM/browser teardown over the network (_cleanup_task_resources), and - # SQLite writes (_persist_session). A raise from any of them used to - # propagate straight out of run_conversation, discarding the partial - # final_response the caller is waiting for (subprocess wrappers saw an - # empty stdout with no traceback — #8049). Each step is now guarded - # independently so one failure can't skip the others, and any errors - # are surfaced on the result dict via ``cleanup_errors`` rather than - # killing the turn. + # Post-loop cleanup must never lose the response: trajectory save, teardown, + # and session persist are guarded independently and errors surface via + # ``cleanup_errors`` rather than killing the turn (#8049). _cleanup_errors = [] # Save trajectory if enabled. ``user_message`` may be a multimodal @@ -309,22 +249,17 @@ def finalize_turn( _cleanup_errors.append(f"cleanup_task_resources: {_cleanup_err}") logger.error("finalize_turn: _cleanup_task_resources failed: %s", _cleanup_err, exc_info=True) - # Persist session to both JSON log and SQLite only after private retry - # scaffolding has been removed. Otherwise a later user "continue" turn - # can replay assistant("(empty)") / recovery nudges and fall into the - # same empty-response loop again. + # Persist only after private retry scaffolding is removed, or a later "continue" + # replays assistant("(empty)") / recovery nudges into the same empty-response loop. try: agent._drop_trailing_empty_response_scaffolding(messages) - # Drop verification-continuation nudges (synthetic user messages) - # from the live history before the tail-assistant check — only the - # nudges need stripping; the assistant candidate persists in - # state.db. (#65919 §7) + # Strip only the synthetic verification nudges before the tail-assistant + # check; the assistant candidate persists in state.db. (#65919) _drop_verification_continuation_scaffolding(messages) - # #95514: an empty terminal completion is not authoritative when the - # stream already delivered text. Recover before persist so a blank - # assistant tail is filled instead of frozen as content=''. + # An empty terminal completion is not authoritative when the stream already + # delivered text; recover before persist so a blank tail isn't frozen (#95514). _recovered_from_stream = False if not interrupted and not failed: _streamed = getattr(agent, "_current_streamed_assistant_text", "") or "" @@ -337,39 +272,16 @@ def finalize_turn( final_response = _streamed _recovered_from_stream = True - # When the turn was interrupted and the last message is a tool - # result, append a synthetic assistant message to close the - # tool-call sequence. Without this, the session persists a - # ``tool → user`` alternation that strict providers (Gemini, - # Claude) reject, causing them to hallucinate a continuation of - # the user's message on the next turn (#48879). - # - # ``_drop_trailing_empty_response_scaffolding`` only rewinds the - # tool tail when an empty-response scaffolding flag is present; a - # clean ``/stop`` interrupt after a successful tool sets no such - # flag, so the tool result survives as the tail and we close it - # here instead. On an interrupt ``final_response`` is typically - # empty, so fall back to an explicit placeholder rather than - # persisting an empty-content assistant turn. + # An interrupt can leave a tool result as the tail (no scaffolding flag rewinds + # it); close the sequence so strict providers don't see ``tool → user``. An + # explicit placeholder is used since final_response is usually empty (#48879). if interrupted: from agent.message_sanitization import close_interrupted_tool_sequence close_interrupted_tool_sequence(messages, final_response) - # Some recovery/fallback paths return a real final_response without - # adding a closing assistant message to the transcript (e.g. the - # partial-stream and prior-turn-content recovery ``break`` sites in - # ``conversation_loop``). If persisted as-is, the durable session can - # end at a tool/user message even though the caller — and the gateway - # platform — already saw a completed assistant response. The next turn - # then replays a user-only backlog and the model re-answers every - # "unanswered" message. Close the durable turn at the source, at the - # single chokepoint every recovery ``break`` flows through, so the - # invariant "delivered final_response ⇒ assistant row in transcript" - # holds regardless of which path produced it. (#43849 / #44100) - # - # Compare content (not just role) so a verification candidate that - # matches the final response is not duplicated at budget - # exhaustion. (#65919 §7) + # Recovery ``break`` sites can return a final_response with no closing + # assistant row; enforce "delivered final_response ⇒ assistant row" here. + # Compare content, not role, so a matching verification candidate isn't dup'd. if final_response and not interrupted: try: _tail = messages[-1] if messages else None @@ -394,66 +306,45 @@ def finalize_turn( ) ) ): - # The tail IS an assistant row, but a *pure tool-call turn* or - # a blank assistant tail whose content was recovered from the - # stream buffer (#95514). Fill that row's content instead of - # appending, so the durable turn ends with the answer without - # creating an assistant→assistant pair. + # Tail is an assistant row (pure tool-call turn or stream-recovered + # blank, #95514): fill its content rather than append a second row. _fill_assistant_tail_content(agent, _tail, final_response) - # The model has completed its request, so replace API-local - # voice/model/skill guidance with the clean user input before writing the - # final durable snapshot and returning the continuation history. Earlier - # turn-start flushes use the DB-only override because their messages are - # still needed for the API request; this finalizer runs after that request - # is complete (#48677 / #63766). + # Request is complete, so replace API-local voice/model/skill guidance with + # the clean user input before the durable snapshot; earlier flushes used the + # DB-only override as their messages were still needed (#48677 / #63766). _apply_override = getattr(agent, "_apply_persist_user_message_override", None) if callable(_apply_override): _apply_override(messages) # ── Post-turn micro-compaction ──────────────────────────── - # After the assistant response is finalized but before the session is - # persisted, run micro-compaction to absorb the oldest uncompacted - # exchange into the rolling summary. This amortizes compression - # across turns rather than batching it into one big pause. + # Absorb the oldest uncompacted exchange into the rolling summary before + # persist, amortizing compression across turns instead of one big pause. if not interrupted and not failed: try: _compressor = getattr(agent, "context_compressor", None) - # Strict `is True` + isinstance gates: plugin context engines - # (and MagicMock compressors in tests) satisfy getattr/duck - # checks with truthy auto-attributes — a bare truthiness check - # here called _micro_compact on a mock and spliced its (empty- - # iterating) return value over the transcript, wiping it. + # Strict `is True` + isinstance gates: plugin context engines and + # MagicMock compressors pass duck checks and would wipe the transcript. if ( _compressor and getattr(_compressor, '_micro_compact_enabled', False) is True and callable(getattr(_compressor, '_micro_compact', None)) and final_response - # compression.checkpoint_required: agent init already - # forces _micro_compact_enabled off, but the compressor - # attribute is plain state a future path could flip on a - # live agent. Micro-compaction has no checkpoint hook in - # its path, so it must never run while the gate is armed. + # Micro-compaction has no checkpoint hook, so it must never run + # while compression.checkpoint_required is armed. and getattr( agent, "compression_checkpoint_required", False ) is not True - # Persistence-isolated agents (background review fork) - # must not micro-compact: the pass burns a real aux-LLM - # call on a throwaway replay transcript, and if the - # compressor ever holds a session_db binding it would - # archive_and_compact the CANONICAL session rows — the - # exact write class _persist_disabled exists to stop. + # Persistence-isolated agents (background review fork) must not + # micro-compact: it burns an aux-LLM call on a throwaway transcript + # and could archive_and_compact the CANONICAL session rows. and not getattr(agent, "_persist_disabled", False) ): _before = len(messages) _compacted = _compressor._micro_compact(messages) - # Micro-compaction defrag rewrites the newest MICRO - # marker's content and pops _db_persisted from the live - # dict in place — the sibling of the pop site above. The - # compressor has no agent reference, so it raises a flag - # for us to invalidate the bounded flush-scan cursor; - # otherwise the rewritten marker row is identity-skipped - # and the stale summary persists to state.db. + # Defrag rewrites the newest MICRO marker in place and pops + # _db_persisted; the compressor flags us to invalidate the flush- + # scan cursor, else the rewritten row is identity-skipped (stale). if getattr( _compressor, "_flush_scan_cursor_invalidated", False ): @@ -475,18 +366,16 @@ def finalize_turn( _cleanup_errors.append(f"persist_session: {_persist_err}") logger.error("finalize_turn: _persist_session failed: %s", _persist_err, exc_info=True) - # The gateway owns a separate in-memory history snapshot. Keep it current - # even when finalization reports a cleanup error: a later prompt must not be - # sent with the pre-turn snapshot while the durable DB already has this turn. + # Keep the gateway's separate in-memory history snapshot current even on + # cleanup error, so a later prompt isn't sent with a pre-turn snapshot. try: agent._session_messages = messages except Exception: pass # ── Turn-exit diagnostic log ───────────────────────────────────── - # Always logged at INFO so agent.log captures WHY every turn ended. - # When the last message is a tool result (agent was mid-work), log - # at WARNING — this is the "just stops" scenario users report. + # Always INFO so agent.log captures WHY every turn ended; WARNING when the last + # message is a tool result (the "just stops" scenario). _last_msg_role = messages[-1].get("role") if messages else None _last_tool_name = None if _last_msg_role == "tool": @@ -527,21 +416,9 @@ def finalize_turn( else: logger.info(_diag_msg, *_diag_args) - # File-mutation verifier footer. - # If one or more ``write_file`` / ``patch`` calls failed during this - # turn and were never superseded by a successful write to the same - # path, append an advisory footer to the assistant response. This - # catches the specific case — reported by Ben Eng (#15524-adjacent) - # — where a model issues a batch of parallel patches, half of them - # fail with "Could not find old_string", and the model summarises - # the turn claiming every file was edited. The user then has to - # manually run ``git status`` to catch the lie. With this footer - # the truth is surfaced on every turn, so over-claiming is - # structurally impossible past the model. - # - # Gate: only applied when a real text response exists for this - # turn and the user didn't interrupt. Empty/interrupted turns - # already have other surface text that shouldn't be augmented. + # File-mutation verifier footer: if ``write_file`` / ``patch`` calls failed and + # were never superseded by a successful write to the same path, append an + # advisory so over-claiming is surfaced. Only on real, uninterrupted responses. if final_response and not interrupted: try: _failed = getattr(agent, "_turn_failed_file_mutations", None) or {} @@ -552,30 +429,16 @@ def finalize_turn( except Exception as _ver_err: logger.debug("file-mutation verifier footer failed: %s", _ver_err) - # Turn-completion explainer. - # When a turn ends abnormally after substantive work — empty content - # after retries, a partial/truncated stream, a still-pending tool - # result, or an iteration/budget limit — the user otherwise gets a - # blank or fragmentary response box with no consolidated reason why - # the agent stopped (#34452). Surface a single user-visible - # explanation derived from ``_turn_exit_reason``, mirroring the - # file-mutation verifier footer pattern above. - # - # Gate carefully so healthy turns stay quiet: - # - ``text_response(...)`` exits never produce an explanation - # (handled inside the formatter), so a terse ``Done.`` is silent. - # - We only ACT when there is no genuinely usable reply this turn: - # an empty response, the "(empty)" terminal sentinel, or a - # suspiciously short partial fragment with no terminating - # punctuation (e.g. "The"). A real short answer keeps its text. + # Turn-completion explainer: on abnormal exits, surface one explanation from + # ``_turn_exit_reason``. Only acts when no usable reply exists (empty, "(empty)", + # or a short unpunctuated fragment); ``text_response(...)`` exits stay silent. if not interrupted: try: if agent._turn_completion_explainer_enabled(): _stripped = (final_response or "").strip() _is_empty_terminal = _stripped == "" or _stripped == "(empty)" - # A short fragment that is not a normal text_response exit - # and lacks sentence-ending punctuation is treated as a - # truncated partial (the "The" case from #34452). + # A short fragment not from a text_response exit and lacking sentence- + # ending punctuation is treated as a truncated partial (#34452). _is_partial_fragment = ( not _is_empty_terminal and not preserved_verification_fallback @@ -601,9 +464,7 @@ def finalize_turn( # the actionable explanation. final_response = _explanation else: - # Keep the partial fragment, append the reason so - # the user sees both what arrived and why it - # stopped. + # Keep the partial fragment and append why it stopped. final_response = ( _stripped + "\n\n" + _explanation ) @@ -613,10 +474,8 @@ def finalize_turn( _response_transformed = False _pre_transform_response = None - # Plugin hook: transform_llm_output - # Fired once per turn after the tool-calling loop completes. - # Plugins can transform the LLM's output text before it's returned. - # First hook to return a string wins; None/empty return leaves text unchanged. + # Plugin hook: transform_llm_output — fired once per turn after the tool loop. + # First hook to return a string wins; None/empty leaves the text unchanged. if final_response and not interrupted: try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -636,10 +495,8 @@ def finalize_turn( except Exception as exc: logger.warning("transform_llm_output hook failed: %s", exc) - # Plugin hook: post_llm_call - # Fired once per turn after the tool-calling loop completes. - # Plugins can use this to persist conversation data (e.g. sync - # to an external memory system). + # Plugin hook: post_llm_call — fired once per turn after the tool loop (e.g. sync + # conversation data to an external memory system). if final_response and not interrupted: try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -657,18 +514,12 @@ def finalize_turn( except Exception as exc: logger.warning("post_llm_call hook failed: %s", exc) - # Context engine observation hook: notify the active engine that this - # turn has finished, with the finalized transcript. Complements the - # per-request select_context() hook (selection before the request; - # observation after the turn). No-op default, fail-open. + # Context engine observation hook (complements per-request select_context()): + # notify the engine the turn finished with the finalized transcript. Fail-open. try: from agent.conversation_loop import _notify_context_engine_turn_complete - # Forward the turn's canonical usage when the host has it. The loop - # stashes the most recent API response's usage dict (the same - # canonical buckets fed to ``update_from_response``) on the agent as - # ``_last_turn_usage``. It is ``None`` on turns that never reached a - # provider response (early failure / interrupt), which is exactly the - # contract: real usage when available, ``None`` otherwise. + # ``_last_turn_usage`` holds the last API response's canonical usage dict, or + # ``None`` on turns that never reached a provider response — by contract. _turn_usage = getattr(agent, "_last_turn_usage", None) _notify_context_engine_turn_complete( agent, @@ -685,15 +536,9 @@ def finalize_turn( except Exception as exc: logger.warning("on_turn_complete notification failed: %s", exc) - # Extract reasoning from the CURRENT turn only. Walk backwards - # but stop at the user message that started this turn — anything - # earlier is from a prior turn and must not leak into the reasoning - # box (confusing stale display; #17055). Within the current turn - # we still want the *most recent* non-empty reasoning: many - # providers (Claude thinking, DeepSeek v4, Codex Responses) emit - # reasoning on the tool-call step and leave the final-answer step - # with reasoning=None, so picking only the last assistant would - # silently drop legitimate same-turn reasoning. + # Reasoning from the CURRENT turn only: stop at this turn's user message + # (#17055), but take the most recent non-empty reasoning since many providers + # emit it on the tool-call step and leave the final step with reasoning=None. last_reasoning = None for msg in reversed(messages): if msg.get("role") == "user": @@ -702,15 +547,9 @@ def finalize_turn( last_reasoning = msg["reasoning"] break - # Class-level surrogate chokepoint (#80366, #55143, #55309, #19819): - # ``final_response`` is often the RAW SDK content - # (``assistant_message.content``), not the sanitized copy stored in - # history by ``build_assistant_message``. Any lone UTF-16 surrogate - # (U+D800–U+DFFF) in it crashes downstream consumers — oneshot stdout - # writes, Telegram's ``utf16_len`` length check, Signal formatting, - # JSON envelope encodes — on every provider (Ollama, NVIDIA NIM, …). - # Scrub once here, where model text leaves the conversation loop, so - # every delivery surface receives valid Unicode. + # Surrogate chokepoint: ``final_response`` may be RAW SDK content, and a lone UTF-16 + # surrogate crashes downstream consumers (stdout, Telegram ``utf16_len``, JSON). + # Scrub once where model text leaves the loop (#80366). if isinstance(final_response, str): final_response = _sanitize_surrogates(final_response) @@ -752,30 +591,27 @@ def finalize_turn( } if agent._tool_guardrail_halt_decision is not None: result["guardrail"] = agent._tool_guardrail_halt_decision.to_metadata() - # Persistence failures already set failed=True + an explanation in - # final_response; also stamp `error` so gateway surfaces status="error" - # (and desktop can toast the cause) instead of a quiet complete frame. + # Persistence failures already set failed=True; also stamp `error` so the gateway + # surfaces status="error" (and desktop can toast) instead of a quiet complete frame. if failed and str(_turn_exit_reason) == "session_persistence_failed": result["error"] = final_response or ( "session storage could not be written — check the state database " "health (`hermes doctor`), then send your message again" ) - # Machine-readable cause for the gateway/desktop: exactly - # 'session_persistence_failed:'. + # Machine-readable cause for the gateway/desktop, exactly + # 'session_persistence_failed:'. # Never clobber a failure_reason another path already stamped. if "failure_reason" not in result: _cause = getattr(agent, "_last_persistence_error_cause", None) result["failure_reason"] = ( "session_persistence_failed:" + (_cause or "unknown") ) - # Surface any post-loop cleanup failures so the caller can distinguish a - # clean turn from one whose trajectory/session/resource teardown raised - # (the response is still returned either way — #8049). + # Surface post-loop cleanup failures so the caller can tell a clean turn from one + # whose teardown raised; the response is returned either way (#8049). if _cleanup_errors: result["cleanup_errors"] = _cleanup_errors - # If a /steer landed after the final assistant turn (no more tool - # batches to drain into), hand it back to the caller so it can be - # delivered as the next user turn instead of being silently lost. + # A /steer landing after the final assistant turn has no tool batch to drain into; + # hand it back so it becomes the next user turn instead of being lost. _leftover_steer = agent._drain_pending_steer() if _leftover_steer: result["pending_steer"] = _leftover_steer @@ -807,11 +643,9 @@ def finalize_turn( messages=messages, ) - # Background memory/skill review — runs AFTER the response is delivered - # so it never competes with the user's task for model attention. - # Suppressed when skip_background_review=True (e.g. cron) — review forks - # spawn another AIAgent (~30K tokens / event) and cron sessions have no - # human-in-the-loop benefit from the review. + # Background memory/skill review runs AFTER delivery so it never competes with the + # user's task. Suppressed by skip_background_review (e.g. cron): the fork costs + # ~30K tokens / event with no human-in-the-loop benefit. if ( final_response and not interrupted @@ -829,16 +663,10 @@ def finalize_turn( except Exception: pass # Background review is best-effort - # Note: Memory provider on_session_end() + shutdown_all() are NOT - # called here — run_conversation() is called once per user message in - # multi-turn sessions. Shutting down after every turn would kill the - # provider before the second message. Actual session-end cleanup is - # handled by the CLI (atexit / /reset) and gateway (session expiry / - # _reset_session). + # Memory provider on_session_end()/shutdown_all() are NOT called here: + # run_conversation() runs once per message; CLI/gateway own session-end cleanup. - # Plugin hook: on_session_end - # Fired at the very end of every run_conversation call. - # Plugins can use this for cleanup, flushing buffers, etc. + # Plugin hook: on_session_end — fired at the end of every run_conversation call. try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook _invoke_hook( diff --git a/agent/turn_overflow.py b/agent/turn_overflow.py new file mode 100644 index 0000000000..2ff14926f4 --- /dev/null +++ b/agent/turn_overflow.py @@ -0,0 +1,560 @@ +"""Overflow recovery for the conversation turn loop: 413 payload-too-large and +context-length errors after ``classify_api_error``. + +Extracted from ``run_conversation``'s ``except`` branch. Each path either compresses and +signals a restart, defers softly (compression lock / transient block), or ends the turn +with a typed result. Nothing here imports ``agent.conversation_loop`` at module level +(cycle); loop-internal helpers and the token estimators that tests patch on the loop +module are imported lazily inside the handler so they keep resolving through the loop. +""" + +from __future__ import annotations + +import logging +import time +from dataclasses import dataclass +from typing import Any, Dict, List, Optional + +from agent.conversation_compression import ( + COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE, + COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE, + COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE, + compression_blocked_transiently, + compression_skipped_due_to_lock, + context_compression_timed_out, +) +from agent.error_classifier import FailoverReason +from agent.message_sanitization import serialized_messages_bytes +from agent.model_metadata import ( + get_context_length_from_provider_error, + is_output_cap_error, + parse_available_output_tokens_from_error, +) +from agent.turn_retry_state import TurnRetryState +from utils import base_url_host_matches + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class OverflowVerdict: + """Outcome of ``recover_from_overflow``. + + ``action`` is one of ``"return"`` (end the turn with ``result``), ``"break"`` + (restart the API call — a ``_retry.restart_with_*`` flag is set), ``"continue"`` + (retry the call immediately) or ``"fallthrough"`` (not an overflow error, or + overflow recovery declined — continue generic error handling). The remaining + fields are the loop locals the handler may have rebound.""" + + action: str + result: Optional[Dict[str, Any]] + messages: List[Dict[str, Any]] + active_system_prompt: Any + conversation_history: Any + approx_tokens: int + compression_attempts: int + provider_overflow_recovery_pending: bool + is_context_length_error: bool + + +def recover_from_overflow( + agent: Any, + api_error: Exception, + classified: Any, + _retry: TurnRetryState, + *, + status_code: Optional[int], + error_msg: str, + wrapped_output_cap_budget: Optional[int], + messages: List[Dict[str, Any]], + api_messages: Any, + system_message: Any, + active_system_prompt: Any, + conversation_history: Any, + approx_tokens: int, + compression_attempts: int, + max_compression_attempts: int, + api_call_count: int, + effective_task_id: Any, +) -> OverflowVerdict: + """413 payload-too-large and context-length recovery (compress + retry, output-cap + clamp, provider-reported context limit, GitHub Models free-tier hint). Order is + load-bearing: 413 is checked BEFORE the generic 4xx handler, and context-length + errors (incl. relay-wrapped output-cap 429s) BEFORE non-retryable client errors. + Compression progress is scored in payload BYTES for 413 (never the byte-blind token + estimate) and in tokens/message count for context overflow.""" + # Token estimators + loop-internal helpers resolve through the loop module so + # existing ``patch("agent.conversation_loop.X")`` mocks keep intercepting. + from agent.conversation_loop import ( + _COMPRESSION_TIMEOUT_FINAL_RESPONSE, + _compression_deferred_result, + conversation_history_after_compression, + estimate_messages_tokens_rough, + estimate_request_tokens_rough, + save_context_length, + ) + + _provider_overflow_recovery_pending = False + is_context_length_error = False + _wrapped_output_cap_budget = wrapped_output_cap_budget + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> OverflowVerdict: + return OverflowVerdict( + action=action, + result=result, + messages=messages, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + approx_tokens=approx_tokens, + compression_attempts=compression_attempts, + provider_overflow_recovery_pending=_provider_overflow_recovery_pending, + is_context_length_error=is_context_length_error, + ) + + is_payload_too_large = ( + classified.reason == FailoverReason.payload_too_large + ) + + # GitHub Models free tier caps requests at 8K tokens, under the system + # prompt + tool schema floor; compression can't help, so say so. + if ( + status_code == 413 + and isinstance(agent.base_url, str) + and base_url_host_matches(agent.base_url, "models.inference.ai.azure.com") + ): + agent._vprint( + f"{agent.log_prefix} 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} request at ~8K tokens. Hermes' system prompt + tool schemas baseline", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} exceeds that floor, so this endpoint cannot run an agentic loop.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} Use the `copilot` provider with a Copilot subscription token (`hermes", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} setup` → GitHub Copilot), or pick any other provider.", + force=True, + ) + + if is_payload_too_large: + compression_attempts += 1 + if compression_attempts > max_compression_attempts: + # Terminal — surface the buffered retry trace. + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached for payload-too-large error.", force=True) + agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) + logger.error("%s413 compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) + agent._persist_session(messages, conversation_history) + _final_response = f"Request payload too large: max compression attempts ({max_compression_attempts}) reached." + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + }) + agent._buffer_status(f"⚠️ Request payload too large (413) — compression attempt {compression_attempts}/{max_compression_attempts}...") + + original_len = len(messages) + # A 413 is a BYTE-size error: score progress in payload bytes, + # never the token estimate, which is deliberately byte-blind to + # images and wedged sessions on "no progress" (#88960 / #47339). + original_bytes = serialized_messages_bytes(messages) + _overflow_input = messages + # Option A (LCM issue 441): overhead-aware request size so recovery arms on the + # true request (msgs + tools + system), not the tool-blind message count. + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), + task_id=effective_task_id, + # Provider proved the request doesn't fit: ignore the + # summary-failure cooldown for this ONE attempt (#100661). + bypass_cooldown=True, + ) + if messages is _overflow_input and compression_skipped_due_to_lock(agent): + # Lock-skip: another path holds the compression lock. A + # temporary defer, not exhaustion — refund the attempt and + # end softly so the gateway does NOT auto-reset (#69870). + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count + )) + if messages is _overflow_input and compression_blocked_transiently(agent): + # Transient-block: a timed guard no-oped compression. A + # defer, never compression_exhausted (auto-reset) (#97488). + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count, + reason="transient_block", + )) + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + + # Re-measure: same-count compression and media aging can shrink + # the request without shrinking the array. Bytes are the yardstick + # for a 413; tokens only for status display. + new_tokens = estimate_messages_tokens_rough(messages) + approx_tokens = new_tokens # update for downstream logging + new_bytes = serialized_messages_bytes(messages) + + made_progress = ( + len(messages) < original_len + or (new_bytes > 0 and new_bytes < original_bytes * 0.95) + ) + if made_progress: + if len(messages) < original_len: + agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) + else: + agent._buffer_status( + f"🗜️ Compressed {original_bytes:,} → {new_bytes:,} " + f"payload bytes, retrying..." + ) + time.sleep(2) # Brief pause between compression retries + _retry.restart_with_compressed_messages = True + return _verdict("break") + else: + if agent._try_strip_image_parts_from_tool_messages( + api_messages, + remember_model=False, + ): + agent._buffer_status( + "📐 Compression could not reduce the request further — " + "removed retained vision payloads and retrying..." + ) + return _verdict("continue") + + # Terminal — surface buffered context so the user + # sees what compression attempts were made. + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Payload too large and cannot compress further.", force=True) + agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) + logger.error("%s413 payload too large. Cannot compress further.", agent.log_prefix) + agent._persist_session(messages, conversation_history) + _final_response = "Request payload too large (413). Cannot compress further." + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + }) + + # Check context-length errors BEFORE the generic 4xx handler; the + # classifier also covers 400/disconnect + large-session heuristics. + is_context_length_error = ( + classified.reason == FailoverReason.context_overflow + # Relay-wrapped output-cap 429s (parsed above) go to the clamp + # below, not failover or generic retries (#72281). + or _wrapped_output_cap_budget is not None + ) + + if is_context_length_error: + compressor = agent.context_compressor + old_ctx = compressor.context_length + + # Two errors: "prompt too long" = INPUT overflows the window (shrink + # context_length + compress); "max_tokens too large" = input fits + # but input + max_tokens > window (shrink OUTPUT cap only). + available_out = parse_available_output_tokens_from_error(error_msg) + if available_out is not None: + # Output-cap error: provider available_tokens is the + # authoritative bound; also estimate the real request shape + # (API-only content), use the smaller minus a margin. + request_input_estimate = estimate_request_tokens_rough( + api_messages, tools=agent.tools or None, + ) + local_available_out = old_ctx - request_input_estimate + if local_available_out > 0: + safe_out = max(1, min(available_out, local_available_out) - 64) + else: + # Local estimate can overshoot; fall back to the + # authoritative provider-reported budget. + safe_out = max(1, available_out - 64) + agent._ephemeral_max_output_tokens = safe_out + agent._buffer_vprint( + f"⚠️ Output cap too large for current prompt — " + f"retrying with max_tokens={safe_out:,} " + f"(provider_available={available_out:,}, " + f"estimated_request_tokens={request_input_estimate:,}; " + f"context_length unchanged at {old_ctx:,})" + ) + # Still count against compression_attempts so we don't + # loop forever if the error keeps recurring. + compression_attempts += 1 + if compression_attempts > max_compression_attempts: + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) + agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) + logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) + agent._persist_session(messages, conversation_history) + _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + }) + # Also compress history so the output-cap retry doesn't spin on + # max_tokens alone; dropping the middle window makes the total + # fit. (#55546) + try: + original_len = len(messages) + original_tokens = estimate_messages_tokens_rough(messages) + _overflow_input = messages + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=request_input_estimate, + task_id=effective_task_id, + bypass_cooldown=True, # #100661 provider-proven overflow + ) + if messages is _overflow_input and compression_skipped_due_to_lock(agent): + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count + )) + if messages is _overflow_input and compression_blocked_transiently(agent): + # #97488: timed transient guard — defer, never + # exhaustion (gateway auto-reset). + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count, + reason="transient_block", + )) + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + new_tokens = estimate_messages_tokens_rough(messages) + if len(messages) < original_len: + agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) + elif new_tokens > 0 and new_tokens < original_tokens * 0.95: + agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) + except Exception: + # Compression must never turn an output-cap error + # fatal — fall through and retry on max_tokens alone. + logger.warning( + "%sOutput-cap compression hit an error; retrying on max_tokens only.", + agent.log_prefix, + ) + _retry.restart_with_compressed_messages = True + return _verdict("break") + + # Output-cap error with unparseable budget: compression can't help + # (input already fits) and would death-loop on the same 400. Fail + # fast. (#55546) + if is_output_cap_error(error_msg): + agent._flush_status_buffer() + agent._vprint( + f"{agent.log_prefix}❌ The provider rejected the request because " + f"max_tokens exceeds its output cap for this model.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} 💡 Lower model.max_tokens in your config.yaml to " + f"at or below the model's max-output limit. " + f"(This is an output-cap error, not a context overflow — " + f"compression cannot fix it.)", + force=True, + ) + logger.error( + f"{agent.log_prefix}Output-cap error not routed into compression " + f"(max_tokens over provider cap): {error_msg[:200]}" + ) + agent._persist_session(messages, conversation_history) + _final_response = ( + "max_tokens exceeds the provider's output cap for this model. " + "Lower model.max_tokens in config.yaml." + ) + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + }) + + # Input too large: shrink context_length only when the provider + # reports the real limit; else keep the window and compress. Guessed + # probe tiers can turn a configured 1M window into 256K/128K/64K. + new_ctx = get_context_length_from_provider_error(error_msg, old_ctx) + _provider_lower = (getattr(agent, "provider", "") or "").lower() + _base_lower = (getattr(agent, "base_url", "") or "").rstrip("/").lower() + is_minimax_provider = ( + _provider_lower in {"minimax", "minimax-cn"} + or _base_lower.startswith(( + "https://api.minimax.io/anthropic", + "https://api.minimaxi.com/anthropic", + )) + ) + minimax_delta_only_overflow = ( + is_minimax_provider + and new_ctx is None + and "context window exceeds limit (" in error_msg + ) + + if new_ctx is not None: + agent._buffer_vprint(f"Context limit detected from API: {new_ctx:,} tokens (was {old_ctx:,})") + compressor.update_model( + model=agent.model, + context_length=new_ctx, + base_url=agent.base_url, + api_key=getattr(agent, "api_key", ""), + provider=agent.provider, + api_mode=agent.api_mode, + ) + # Persist the provider-reported limit before compression/retry: + # rate limit, missing usage, or restart must not lose confirmed + # metadata. Probe flags remain a fallback if this write fails. + save_context_length(agent.model, agent.base_url, new_ctx) + # Probe flags only on the built-in compressor (plugin engines + # manage their own); provider-sourced value, so safe to cache. + if hasattr(compressor, "_context_probed"): + compressor._context_probed = True + compressor._context_probe_persistable = True + agent._buffer_vprint(f"⚠️ Context length exceeded — using provider limit: {old_ctx:,} → {new_ctx:,} tokens") + elif minimax_delta_only_overflow: + agent._buffer_vprint( + f"Provider reported overflow amount only; " + f"keeping context_length at {old_ctx:,} tokens and compressing." + ) + else: + agent._buffer_vprint( + f"⚠️ Context length exceeded, but provider did not report a max context length; " + f"keeping context_length at {old_ctx:,} tokens and compressing." + ) + + compression_attempts += 1 + if compression_attempts > max_compression_attempts: + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) + agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) + logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) + agent._persist_session(messages, conversation_history) + _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + }) + agent._buffer_status(COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE.format(tokens=approx_tokens, attempt=compression_attempts, cap=max_compression_attempts)) + + original_len = len(messages) + original_tokens = estimate_messages_tokens_rough(messages) + _overflow_input = messages + # Pass the OVERHEAD-AWARE size (msgs + tool schemas + system) so LCM + # forced-overflow recovery arms on the TRUE request; approx_tokens + # stays for status. See hermes-lcm _should_force_overflow_recovery. + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), + task_id=effective_task_id, + # Provider proved the request doesn't fit: ignore the + # summary-failure cooldown for this ONE attempt (bounded by + # max_compression_attempts). (#100661) + bypass_cooldown=True, + ) + if messages is _overflow_input and compression_skipped_due_to_lock(agent): + # Lock-skip: another path holds the compression lock, so this + # pass no-oped. Temporary defer, not exhaustion — refund the + # attempt, end the turn softly, no auto-reset. (#69870) + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count + )) + if messages is _overflow_input and compression_blocked_transiently(agent): + # Transient block: a timed guard (host-timeout cooldown / + # structural backoff) no-oped this pass — defer softly, never + # compression_exhausted (auto-reset). (#97488) + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, messages, api_call_count, + reason="transient_block", + )) + if context_compression_timed_out(agent): + # Host timeout: recovery spent its wait budget with no committed + # summary. Re-sending would hit the same overflow; end the turn + # via the typed recovery contract. (#98722) + agent._persist_session(messages, conversation_history) + _final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + "turn_exit_reason": "context_compression_timeout", + }) + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + + # Re-estimate after compression: same-message-count compression + # (tool-result pruning, in-place summarization) can shrink the + # request. (#39550) + new_tokens = estimate_messages_tokens_rough(messages) + approx_tokens = new_tokens # update for downstream logging + + if len(messages) < original_len or (new_tokens > 0 and new_tokens < original_tokens * 0.95) or (new_ctx and new_ctx < old_ctx): + if len(messages) < original_len: + agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) + elif new_tokens > 0 and new_tokens < original_tokens * 0.95: + agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) + time.sleep(2) # Brief pause between compression retries + # Rebuild the full request and force normal preflight to honor + # it; message count alone doesn't prove system/tool-inclusive + # pressure fell. + _provider_overflow_recovery_pending = True + _retry.restart_with_compressed_messages = True + return _verdict("break") + else: + # Can't compress further and already at minimum tier + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Context length exceeded and cannot compress further.", force=True) + agent._vprint(f"{agent.log_prefix} 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.", force=True) + logger.error("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}") + agent._persist_session(messages, conversation_history) + _final_response = f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further." + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compression_exhausted": True, + }) + return _verdict("fallthrough") diff --git a/agent/turn_preflight.py b/agent/turn_preflight.py new file mode 100644 index 0000000000..173bb0a445 --- /dev/null +++ b/agent/turn_preflight.py @@ -0,0 +1,506 @@ +"""Pre-API preflight compression gate for the conversation turn loop. + +Extracted from ``run_conversation``. Runs once per API call after the request pressure +is measured: grow a managed llama.cpp window (last resort), or compress when over +threshold (deferring on noisy estimates, in failure cooldown, or when the review fork's +first request is pending), handle the provider-proven overflow re-check fail-closed, +and emit the deduped blocked/uncompressed overflow warnings. Nothing here imports +``agent.conversation_loop`` at module level (cycle); loop-internal helpers resolve lazily. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any, Dict, List, Optional + +from agent.context_engine import automatic_compaction_status_message +from agent.conversation_compression import ( + PRE_API_COMPRESSION_STATUS_TEMPLATE, + compression_blocked_transiently, + compression_skipped_due_to_lock, + context_compression_timed_out, + conversation_history_after_compression, +) +from agent.turn_context import _review_fork_first_request_pending + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class PreflightVerdict: + """Outcome of ``run_preflight_compression``. + + ``action``: ``"proceed"`` (make the API call), ``"continue"`` (window grown or + history compacted — the call/budget was refunded, re-enter the turn loop and + re-measure), ``"break"`` (turn ends: compression timeout or non-actionable + compaction handoff — ``final_response``/``failed``/``turn_exit_reason`` set) or + ``"return"`` (typed deferred/exhausted result in ``result``). The remaining fields + are the loop locals the gate may have rebound.""" + + action: str + result: Optional[Dict[str, Any]] + messages: List[Dict[str, Any]] + active_system_prompt: Any + conversation_history: Any + api_call_count: int + compression_attempts: int + pending_moa_prepared_request: Any + last_preflight_pressure: Optional[int] + final_response: Any + failed: bool + compression_timeout_exhausted: bool + turn_exit_reason: Any + + +def run_preflight_compression( + agent: Any, + *, + compressor: Any, + request_pressure_tokens: int, + provider_overflow_preflight: bool, + preflight_compression_blocked: bool, + defer_preflight: Any, + moa_prepared_request: Any, + pending_moa_prepared_request: Any, + messages: List[Dict[str, Any]], + system_message: Any, + user_message: Any, + active_system_prompt: Any, + conversation_history: Any, + api_call_count: int, + compression_attempts: int, + max_compression_attempts: int, + effective_task_id: Any, + final_response: Any, + failed: bool, + compression_timeout_exhausted: bool, + turn_exit_reason: Any, +) -> PreflightVerdict: + """Mirror of the turn-prologue guard chain (defer on noisy estimate → skip in failure + cooldown → ``should_compress``), #11529. A compression pass that never reaches the + provider refunds the call/budget in every branch (skip, re-run, timeout) so + ``api_call_count`` never over-reports; a lock/transient skip refunds the attempt and + leaves the progress blocker unarmed (#69870, #97488). A forced provider-overflow + preflight that any gate blocks fails closed (llama.cpp may silently truncate).""" + from agent.conversation_loop import ( + _COMPRESSION_TIMEOUT_FINAL_RESPONSE, + _HANDOFF_SKIP_FINAL_RESPONSE, + _compression_deferred_result, + _maybe_grow_local_window, + _provider_overflow_exhausted_result, + _should_skip_model_call_for_reference_handoff, + ) + + _compressor = compressor + _provider_overflow_preflight = provider_overflow_preflight + _preflight_compression_blocked = preflight_compression_blocked + _defer_preflight = defer_preflight + _moa_prepared_request = moa_prepared_request + _last_preflight_pressure = None + _compression_timeout_exhausted = compression_timeout_exhausted + _turn_exit_reason = turn_exit_reason + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> PreflightVerdict: + return PreflightVerdict( + action=action, + result=result, + messages=messages, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + api_call_count=api_call_count, + compression_attempts=compression_attempts, + pending_moa_prepared_request=pending_moa_prepared_request, + last_preflight_pressure=_last_preflight_pressure, + final_response=final_response, + failed=failed, + compression_timeout_exhausted=_compression_timeout_exhausted, + turn_exit_reason=_turn_exit_reason, + ) + + _compression_cooldown = getattr( + _compressor, "get_active_compression_failure_cooldown", lambda: None + )() + if ( + agent.compression_enabled + and not _review_fork_first_request_pending(agent) + and len(messages) > 1 + and compression_attempts < max_compression_attempts + and ( + not _preflight_compression_blocked + or _provider_overflow_preflight + ) + and ( + not _defer_preflight(request_pressure_tokens) + or _provider_overflow_preflight + ) + and not _compression_cooldown + and _compressor.should_compress(request_pressure_tokens) + ): + # Managed local runtime: grow the context window before compressing (last + # resort). Only for a llamacpp provider at the supervised base_url. + _grown_window = _maybe_grow_local_window( + agent, _compressor, request_pressure_tokens + ) + if _grown_window: + # Bigger window granted: recalibrate the compressor and skip compression + # this pass. + _compressor.update_model( + agent.model, + _grown_window, + base_url=getattr(agent, "base_url", "") or "", + api_key=getattr(agent, "api_key", "") or "", + provider=getattr(agent, "provider", "") or "", + api_mode=getattr(agent, "api_mode", "") or "", + ) + agent._buffer_status( + f"📈 Context window grown to {_grown_window // 1024}K " + f"(local model; conversation continues uncompressed)" + ) + # Never reached the provider — refund the call/budget like the + # compression path does before its continue. + api_call_count -= 1 + agent._api_call_count = api_call_count + agent.iteration_budget.refund() + return _verdict("continue") + if _moa_prepared_request is not None: + pending_moa_prepared_request = _moa_prepared_request + compression_attempts += 1 + # Compression is running: reset the blocked-overflow warning dedup so a + # later blocked turn warns again (#62625). getattr: test doubles lack it. + _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) + if callable(_clear_warn): + _clear_warn() + logger.info( + "Pre-API compression: ~%s request tokens >= %s threshold " + "(context=%s, attempt=%s/%s)", + f"{request_pressure_tokens:,}", + f"{int(getattr(_compressor, 'threshold_tokens', 0) or 0):,}", + f"{int(getattr(_compressor, 'context_length', 0) or 0):,}" + if getattr(_compressor, "context_length", 0) else "unknown", + compression_attempts, + max_compression_attempts, + ) + _pre_api_status = automatic_compaction_status_message( + _compressor, + phase="pre_api", + default_message=PRE_API_COMPRESSION_STATUS_TEMPLATE.format( + tokens=request_pressure_tokens + ), + approx_tokens=request_pressure_tokens, + threshold_tokens=int( + getattr(_compressor, "threshold_tokens", 0) or 0 + ), + context_length=int( + getattr(_compressor, "context_length", 0) or 0 + ), + model=agent.model, + attempt=compression_attempts, + max_attempts=max_compression_attempts, + ) + if _pre_api_status: + agent._emit_status(_pre_api_status) + _last_preflight_pressure = request_pressure_tokens + _pre_api_input = messages + messages, active_system_prompt = agent._compress_context( + messages, + system_message, + approx_tokens=request_pressure_tokens, + task_id=effective_task_id, + ) + if context_compression_timed_out(agent): + # Progress-aware timeout (#98722): never reached the provider — refund + # the call/budget and stop; an overflow retry would only re-compress. + api_call_count -= 1 + agent._api_call_count = api_call_count + agent.iteration_budget.refund() + final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE + failed = True + _compression_timeout_exhausted = True + _turn_exit_reason = "context_compression_timeout" + return _verdict("break") + if messages is _pre_api_input and ( + compression_skipped_due_to_lock(agent) + or compression_blocked_transiently(agent) + ): + # Temporary DEFER (lock held / cooldown), not evidence about + # compressibility: refund the attempt, leave the progress blocker + # unarmed and proceed (#69870, #97488). + compression_attempts -= 1 + _last_preflight_pressure = None + if pending_moa_prepared_request is _moa_prepared_request: + pending_moa_prepared_request = None + else: + # Reset retry/empty-response state so the compacted request gets a fresh + # chance. + agent._empty_content_retries = 0 + agent._thinking_prefill_retries = 0 + agent._last_content_with_tools = None + agent._last_content_tools_all_housekeeping = False + agent._mute_post_response = False + # Re-baseline the flush cursor: rotation returns None (child flushes + # whole); in-place returns list(messages) — None would re-append + # persisted rows. See conversation_history_after_compression(). + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + # Never reaches the provider on skip or re-run — refund the call/budget + # in BOTH cases, else budget leaks and api_call_count over-reports. + api_call_count -= 1 + agent._api_call_count = api_call_count + agent.iteration_budget.refund() + if _should_skip_model_call_for_reference_handoff( + messages, user_message + ): + # Reference-only handoff must not become the active turn + # after a completed assistant response (#80622). + logger.info( + "Skipping post-compaction model call: reference-only " + "handoff would be the sole active user turn (#80622)" + ) + if not final_response: + final_response = _HANDOFF_SKIP_FINAL_RESPONSE + _turn_exit_reason = "compaction_handoff_not_actionable" + return _verdict("break") + return _verdict("continue") + elif _provider_overflow_preflight and _compression_cooldown: + # Provider proved the request cannot fit and the compressor is unavailable: + # don't resend; let the next user turn retry after cooldown. + agent._persist_session(messages, conversation_history) + return _verdict("return", _compression_deferred_result( + agent, + messages, + api_call_count, + reason="transient_block", + )) + elif ( + _provider_overflow_preflight + and compression_attempts >= max_compression_attempts + ): + # All recovery passes consumed and still over threshold: fail closed — + # llama.cpp may silently truncate an oversized retry. + return _verdict("return", _provider_overflow_exhausted_result( + agent, + messages, + conversation_history, + api_call_count, + request_pressure_tokens, + max_compression_attempts, + )) + elif ( + agent.compression_enabled + and len(messages) > 1 + and compression_attempts < max_compression_attempts + and not _defer_preflight(request_pressure_tokens) + and _compression_cooldown + ): + # Summary-LLM cooldown blocks compression: deduped warning only when over + # threshold (should_compress_info reason is None below it) (#62625). + _block_reason = None + try: + _block_reason = _compressor.should_compress_info( + request_pressure_tokens + )[1] + except Exception: + _block_reason = None + if _block_reason: + agent._warn_context_overflow_blocked( + _block_reason, + request_pressure_tokens, + int(getattr(_compressor, "threshold_tokens", 0) or 0), + ) + elif not agent.compression_enabled and len(messages) > 1: + # Uncompressed session guard (#89297): compression is disabled, so warn + # (deduped) when the request exceeds the context window; the turn-context + # preflight re-arms the dedup. + _ctx_len = getattr( + getattr(agent, "context_compressor", None), "context_length", None + ) + if ( + isinstance(_ctx_len, int) + and _ctx_len > 0 + and request_pressure_tokens > _ctx_len + ): + _warn_fn = getattr( + agent, "_warn_uncompressed_context_overflow", None + ) + if callable(_warn_fn): + _warn_fn(request_pressure_tokens, _ctx_len) + + if _provider_overflow_preflight: + # Any other gate blocking the forced preflight (e.g. uncompressible one- + # message request) must fail closed: the request is proven not to fit. + return _verdict("return", _provider_overflow_exhausted_result( + agent, + messages, + conversation_history, + api_call_count, + request_pressure_tokens, + max_compression_attempts, + )) + return _verdict("proceed") + + +@dataclass +class PostToolCompressionVerdict: + """``end_turn`` True → a reference-only compaction handoff would be the sole active + user turn (#80622): stop without another model call (``final_response`` / + ``turn_exit_reason`` set).""" + + end_turn: bool + messages: List[Dict[str, Any]] + active_system_prompt: Any + conversation_history: Any + compression_attempts: int + final_response: Any + turn_exit_reason: Any + + +def compress_after_tool_results( + agent: Any, + *, + messages: List[Dict[str, Any]], + system_message: Any, + user_message: Any, + active_system_prompt: Any, + conversation_history: Any, + compression_attempts: int, + max_compression_attempts: int, + effective_task_id: Any, + final_response: Any, + turn_exit_reason: Any, +) -> PostToolCompressionVerdict: + """Post-tool-call compression decision. Pressure comes from API-reported + ``prompt_tokens`` (a tight lower bound; thinking models inflate completion tokens, + #12026), ``0`` right after compression (no real count yet), else the route-aware + overhead-inclusive estimate (#14695). Over threshold but blocked → deduped warning + (#62625) plus the deterministic tool-result-only prune, committed only when the + engine returns a NEW list (never rebuild ``conversation_history`` for it).""" + from agent.conversation_loop import ( + _HANDOFF_SKIP_FINAL_RESPONSE, + _midturn_request_pressure_tokens, + _should_skip_model_call_for_reference_handoff, + estimate_request_tokens_rough, + ) + + _turn_exit_reason = turn_exit_reason + + def _verdict(end_turn: bool) -> PostToolCompressionVerdict: + return PostToolCompressionVerdict( + end_turn=end_turn, + messages=messages, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + compression_attempts=compression_attempts, + final_response=final_response, + turn_exit_reason=_turn_exit_reason, + ) + + # Decide compression from API-reported prompt tokens (tight lower bound; + # tool results get counted on the next call). If last_prompt_tokens is 0 + # (disconnect / no usage data) fall back to a rough estimate. (#2153) + _compressor = agent.context_compressor + if _compressor.last_prompt_tokens > 0: + # Only prompt_tokens: thinking models inflate completion_tokens with + # reasoning that uses no context → premature compression. (#12026) + _real_tokens = _compressor.last_prompt_tokens + elif _compressor.last_prompt_tokens == -1: + # Compression just ran, no API prompt count yet: don't treat a rough + # schema-heavy post-compression estimate as real context pressure. + _real_tokens = 0 + else: + # Include tool schemas (20-30K tokens the messages-only estimate + # misses) and stay route-aware: on a compacted native-Codex session + # the generic durable-history figure would false-trigger. (#14695) + _real_tokens = _midturn_request_pressure_tokens( + agent, + messages, + active_system_prompt or "", + estimate_request_tokens_rough( + messages, tools=agent.tools or None + ), + ) + + if ( + agent.compression_enabled + and compression_attempts < max_compression_attempts + and _compressor.should_compress(_real_tokens) + ): + compression_attempts += 1 + # Compression is running: reset blocked-overflow warning dedup so a + # future blocked turn can warn again. getattr: test doubles lack it. + _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) + if callable(_clear_warn): + _clear_warn() + agent._safe_print(" ⟳ compacting context…") + _post_tool_input = messages + # Pass overhead-aware _real_tokens, not last_prompt_tokens (0 in + # the no-usage fallback), so the overflow guard sees the true size. + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=_real_tokens, + task_id=effective_task_id, + ) + if ( + messages is _post_tool_input + and compression_skipped_due_to_lock(agent) + ): + # Lock-skip no-op is a temporary defer, not evidence about + # compressibility: refund so a lock-loser loop doesn't burn the + # budget toward compression_exhausted. (#69870) + compression_attempts -= 1 + else: + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + if _should_skip_model_call_for_reference_handoff( + messages, user_message + ): + logger.info( + "Skipping post-tool compaction model call: " + "reference-only handoff would be the sole " + "active user turn (#80622)" + ) + if not final_response: + final_response = _HANDOFF_SKIP_FINAL_RESPONSE + _turn_exit_reason = "compaction_handoff_not_actionable" + return _verdict(True) + elif agent.compression_enabled: + # Over threshold but compression blocked (cooldown/anti-thrash): + # deduped warning so context can't silently overflow. (#62625) + _block_reason = None + _info = getattr(_compressor, "should_compress_info", None) + if _info is not None: + try: + _block_reason = _info(_real_tokens)[1] + except Exception: + _block_reason = None + if _block_reason: + agent._warn_context_overflow_blocked( + _block_reason, + _real_tokens, + int(getattr(_compressor, "threshold_tokens", 0) or 0), + ) + # Proactive tool-result prune (deterministic, no LLM, keeps tail): + # no-op unless proactive_prune_tokens is exceeded; commits only past + # proactive_prune_min_reclaim_tokens so cache breaks stay episodic. + _prune = getattr(_compressor, "prune_tool_results_only", None) + if callable(_prune): + try: + _pruned_msgs, _pruned_n = _prune( + messages, current_tokens=_real_tokens + ) + except Exception: + logger.debug( + "proactive tool-result prune failed; skipping", + exc_info=True, + ) + _pruned_msgs, _pruned_n = messages, 0 + # Standard no-op caller contract: only commit when the + # engine returned a NEW list object with a non-zero count. + if _pruned_n and _pruned_msgs is not messages: + # Do NOT rebuild conversation_history: rows already carry + # _DB_PERSISTED_MARKER, and on a stale in-place flag the + # helper could seed unpersisted rows into history_ids. + messages = _pruned_msgs + return _verdict(False) diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py new file mode 100644 index 0000000000..6295b165c7 --- /dev/null +++ b/agent/turn_recovery.py @@ -0,0 +1,1761 @@ +"""Recovery-branch handlers for the conversation turn's inner retry loop. + +``run_conversation`` wraps every model API call in ``while retry_count < max_retries``. +When the call raises, a long chain of one-shot recovery branches runs before the generic +retry/backoff path: payload sanitization (surrogates / ASCII codec), image rejection, +per-provider 401 credential refresh, format-recovery strips (thinking signatures, +encrypted reasoning, native compaction, llama.cpp grammar), etc. Each handler here owns +one contiguous chain and returns a verdict the loop acts on: + +* ``True`` → the request was repaired in place; the loop ``continue``s (re-issues the + call with the same ``retry_count``, exactly as the inline ``continue`` did). +* ``False`` → nothing applied; the loop falls through to the generic retry path. + +One-shot guards live on ``TurnRetryState`` (``agent/turn_retry_state.py``); handlers set +them exactly where the inline code did. Handlers mutate ``agent`` / ``messages`` / +``api_messages`` in place — side effects are the point; only locals moved. + +Logger name stays ``agent.conversation_loop`` so caplog pins and log routing are +unchanged. Nothing here imports ``agent.conversation_loop`` at module level (cycle); +the few loop-internal helpers are passed in or imported lazily. +""" + +from __future__ import annotations + +import logging +import re +import time +from dataclasses import dataclass +from typing import Any, Dict, List, Optional, Tuple + +from agent.conversation_compression import COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE +from agent.model_metadata import is_output_cap_error, parse_available_output_tokens_from_error +from agent.retry_utils import is_zai_coding_overload_error, zai_coding_overload_retry_ceiling +from agent.error_classifier import FailoverReason +from agent.message_sanitization import ( + _looks_like_image_content_rejection, + _sanitize_messages_non_ascii, + _sanitize_messages_surrogates, + _sanitize_structure_non_ascii, + _sanitize_structure_surrogates, + _sanitize_tools_non_ascii, + _strip_images_from_messages, + _strip_non_ascii, + close_interrupted_tool_sequence, +) +from agent.turn_retry_state import TurnRetryState +from utils import base_url_host_matches + +logger = logging.getLogger("agent.conversation_loop") + + +def _image_error_max_dimension(error: Exception) -> Optional[int]: + """Extract a provider-reported image dimension ceiling, if present.""" + parts = [] + for value in ( + error, + getattr(error, "message", None), + getattr(error, "body", None), + ): + if value: + try: + parts.append(str(value)) + except Exception: + pass + text = " ".join(parts).lower() + if "image" not in text or "dimension" not in text or "max allowed size" not in text: + return None + + match = re.search(r"max allowed size(?:\s+for [^:]+)?:\s*(\d{3,5})\s*pixels?", text) + if not match: + return None + try: + max_dimension = int(match.group(1)) + except ValueError: + return None + if 512 <= max_dimension <= 8000: + return max_dimension + return None + + +def _try_refresh_nous_paid_entitlement_credentials(agent) -> bool: + """Refresh Nous runtime credentials after a fresh paid-entitlement check.""" + try: + from hermes_cli.nous_account import get_nous_portal_account_info + + account_info = get_nous_portal_account_info(force_fresh=True) + if account_info.paid_service_access is not True: + return False + return agent._try_refresh_nous_client_credentials( + force=True, + ) + except Exception: + return False + + +def recover_before_classification( + agent: Any, + api_error: Exception, + *, + messages: List[Dict[str, Any]], + api_messages: Any, + api_kwargs: Any, + active_system_prompt: Any, +) -> Tuple[bool, Any]: + """Recovery branches that run BEFORE ``classify_api_error``: UnicodeEncodeError + sanitization (surrogates, then ASCII codec), provider image-content rejection + (switch session to text-only), and the Bedrock AnthropicBedrock SDK streaming + fallback. Returns ``(retry_now, active_system_prompt)``; the prompt may be + ASCII-sanitized in place.""" + # UnicodeEncodeError recovery: lone surrogates (clipboard paste) or an + # ASCII codec under a non-UTF-8 locale. Sanitize in-place; at most two + # retries (surrogate strip, then ASCII-only). + if isinstance(api_error, UnicodeEncodeError) and getattr(agent, '_unicode_sanitization_passes', 0) < 2: + _err_str = str(api_error).lower() + _is_ascii_codec = "'ascii'" in _err_str or "ascii" in _err_str + # Surrogate errors: utf-8 refusing U+D800..U+DFFF + # ("surrogates not allowed"). + _is_surrogate_error = ( + "surrogate" in _err_str + or ("'utf-8'" in _err_str and not _is_ascii_codec) + ) + # Sanitize `messages` AND `api_messages` (which may carry + # `reasoning_content`/`reasoning_details`), plus `api_kwargs` and + # `prefill_messages` if present. Mirrors the ASCII recovery below. + _surrogates_found = _sanitize_messages_surrogates(messages) + if isinstance(api_messages, list): + if _sanitize_messages_surrogates(api_messages): + _surrogates_found = True + if isinstance(api_kwargs, dict): + if _sanitize_structure_surrogates(api_kwargs): + _surrogates_found = True + if isinstance(getattr(agent, "prefill_messages", None), list): + if _sanitize_messages_surrogates(agent.prefill_messages): + _surrogates_found = True + # Gate the retry on the error type, not on whether anything was + # found — a new transformed field could slip through. Bounded by + # _unicode_sanitization_passes < 2 (outer guard). + if _surrogates_found or _is_surrogate_error: + agent._unicode_sanitization_passes += 1 + if _surrogates_found: + agent._buffer_vprint( + "⚠️ Stripped invalid surrogate characters from messages. Retrying..." + ) + else: + agent._buffer_vprint( + "⚠️ Surrogate encoding error — retrying after full-payload sanitization..." + ) + return True, active_system_prompt + if _is_ascii_codec: + agent._force_ascii_payload = True + # ASCII codec: strip all non-ASCII from messages/tool schemas + # and retry — both `messages` and `api_messages` (which may + # carry extra fields like reasoning_content). + _messages_sanitized = _sanitize_messages_non_ascii(messages) + if isinstance(api_messages, list): + _sanitize_messages_non_ascii(api_messages) + # Also sanitize the last api_kwargs so a non-ASCII transformed + # field doesn't survive via _build_api_kwargs cache paths. + if isinstance(api_kwargs, dict): + _sanitize_structure_non_ascii(api_kwargs) + _prefill_sanitized = False + if isinstance(getattr(agent, "prefill_messages", None), list): + _prefill_sanitized = _sanitize_messages_non_ascii(agent.prefill_messages) + + _tools_sanitized = False + if isinstance(getattr(agent, "tools", None), list): + _tools_sanitized = _sanitize_tools_non_ascii(agent.tools) + + _system_sanitized = False + if isinstance(active_system_prompt, str): + _sanitized_system = _strip_non_ascii(active_system_prompt) + if _sanitized_system != active_system_prompt: + active_system_prompt = _sanitized_system + agent._cached_system_prompt = _sanitized_system + _system_sanitized = True + if isinstance(getattr(agent, "ephemeral_system_prompt", None), str): + _sanitized_ephemeral = _strip_non_ascii(agent.ephemeral_system_prompt) + if _sanitized_ephemeral != agent.ephemeral_system_prompt: + agent.ephemeral_system_prompt = _sanitized_ephemeral + _system_sanitized = True + + _headers_sanitized = False + _default_headers = ( + agent._client_kwargs.get("default_headers") + if isinstance(getattr(agent, "_client_kwargs", None), dict) + else None + ) + if isinstance(_default_headers, dict): + _headers_sanitized = _sanitize_structure_non_ascii(_default_headers) + + # Sanitize the API key: non-ASCII in credentials makes httpx + # fail encoding the Authorization header — the usual persistent + # cause after message/tool sanitization (#6843). + _credential_sanitized = False + _raw_key = getattr(agent, "api_key", None) or "" + # Entra ID bearer providers are callables minting ASCII JWTs; + # skip (``_strip_non_ascii`` would crash on a callable). + if _raw_key and isinstance(_raw_key, str): + _clean_key = _strip_non_ascii(_raw_key) + if _clean_key != _raw_key: + agent.api_key = _clean_key + if isinstance(getattr(agent, "_client_kwargs", None), dict): + agent._client_kwargs["api_key"] = _clean_key + # Also update the live client — auth_headers reads its + # own api_key copy on every request. + if getattr(agent, "client", None) is not None and hasattr(agent.client, "api_key"): + agent.client.api_key = _clean_key + _credential_sanitized = True + agent._vprint( + f"{agent.log_prefix}⚠️ API key contained non-ASCII characters " + f"(bad copy-paste?) — stripped them. If auth fails, " + f"re-copy the key from your provider's dashboard.", + force=True, + ) + + # Always retry on ASCII codec detection: _force_ascii_payload + # sanitizes the full api_kwargs next iteration even when + # checks above find nothing. Bounded by passes < 2. + agent._unicode_sanitization_passes += 1 + _any_sanitized = ( + _messages_sanitized + or _prefill_sanitized + or _tools_sanitized + or _system_sanitized + or _headers_sanitized + or _credential_sanitized + ) + if _any_sanitized: + agent._vprint( + f"{agent.log_prefix}⚠️ System encoding is ASCII — stripped non-ASCII characters from request payload. Retrying...", + force=True, + ) + else: + agent._vprint( + f"{agent.log_prefix}⚠️ System encoding is ASCII — enabling full-payload sanitization for retry...", + force=True, + ) + return True, active_system_prompt + + # ── Image-rejection recovery ────────────────────────────── + # Some providers 4xx on image_url content: strip images, mark session + # vision-unsupported, retry text-only. English phrase match; extend it. + _err_body = "" + try: + _err_body = str(getattr(api_error, "body", None) or + getattr(api_error, "message", None) or + str(api_error)) + except Exception: + pass + _err_status = getattr(api_error, "status_code", None) + _looks_like_image_rejection = _looks_like_image_content_rejection(_err_body) + # 4xx-only gate: 5xx/timeouts are transient and take the retry path. + _status_ok = _err_status is None or (400 <= int(_err_status) < 500) + if ( + getattr(agent, "_vision_supported", True) + and _looks_like_image_rejection + and _status_ok + ): + agent._vision_supported = False + _imgs_removed = _strip_images_from_messages(messages) + if isinstance(api_messages, list): + _strip_images_from_messages(api_messages) + agent._vprint( + f"{agent.log_prefix}⚠️ Server rejected image content — " + f"switching to text-only mode for this session" + + (". Stripped images from history and retrying." if _imgs_removed else "."), + force=True, + ) + return True, active_system_prompt + + # ── Bedrock AnthropicBedrock SDK streaming failure ── + # SDK raises "Unexpected event order" when Bedrock errors before + # message_start; fall back to native Converse for this session (#28156). + if ( + isinstance(api_error, RuntimeError) + and "unexpected event order" in str(api_error).lower() + and getattr(agent, "provider", "") == "bedrock" + and agent.api_mode == "anthropic_messages" + and not getattr(agent, "_bedrock_converse_fallback_attempted", False) + ): + agent._bedrock_converse_fallback_attempted = True + agent.api_mode = "bedrock_converse" + agent._bedrock_region = getattr(agent, "_bedrock_region", None) or "us-east-1" + agent.client = None # Drop the AnthropicBedrock client + agent._client_kwargs = {} + agent._vprint( + f"{agent.log_prefix}⚠️ AnthropicBedrock SDK streaming failed — " + f"falling back to native Converse API for this session.", + force=True, + ) + return True, active_system_prompt + return False, active_system_prompt + + +def recover_after_classification( + agent: Any, + api_error: Exception, + classified: Any, + _retry: TurnRetryState, + *, + status_code: Optional[int], + error_context: Any, + messages: List[Dict[str, Any]], + api_messages: Any, +) -> Tuple[bool, bool]: + """One-shot recovery chain that runs AFTER ``classify_api_error`` and before the + generic retry path. Order is load-bearing (each branch may ``return`` early): + Nous paid-entitlement refresh → credential-pool rotation → image shrink → + multimodal-tool-content strip → corrupt-image strip → Anthropic OAuth 1M-beta + disable → per-provider 401 credential refresh (codex/xai, vertex, nous, copilot, + anthropic, with user-facing diagnostics when refresh fails) → thinking-signature + strip → invalid-encrypted-content replay disable → native-compaction reject → + llama.cpp grammar strip. Returns ``(retry_now, recovered_with_pool)``; + ``recovered_with_pool`` is read later by the Nous rate-limit guard.""" + # Shared with the billing/entitlement helpers that stay in the loop module; + # lazy so this module never imports agent.conversation_loop at load time. + from agent.conversation_loop import ( + _is_copilot_provider, + _is_nous_inference_route, + _print_nous_entitlement_guidance, + ) + + if ( + classified.reason == FailoverReason.billing + and _is_nous_inference_route( + getattr(agent, "provider", "") or "", + getattr(agent, "base_url", "") or "", + ) + and not _retry.nous_paid_entitlement_refresh_attempted + ): + _retry.nous_paid_entitlement_refresh_attempted = True + if _try_refresh_nous_paid_entitlement_credentials(agent): + agent._vprint( + f"{agent.log_prefix}🔐 Nous paid access verified — " + "refreshed runtime credentials and retrying request...", + force=True, + ) + return True, False + + recovered_with_pool, _retry.has_retried_429 = agent._recover_with_credential_pool( + status_code=status_code, + has_retried_429=_retry.has_retried_429, + classified_reason=classified.reason, + error_context=error_context, + billing_unverified=classified.billing_unverified, + ) + if recovered_with_pool: + return True, recovered_with_pool + + # Image-too-large recovery: shrink oversized native image parts + # in-place and retry once; otherwise fall through to normal handling. + if ( + classified.reason == FailoverReason.image_too_large + and not _retry.image_shrink_retry_attempted + ): + _retry.image_shrink_retry_attempted = True + image_max_dimension = _image_error_max_dimension(api_error) or 8000 + if agent._try_shrink_image_parts_in_messages( + api_messages, + max_dimension=image_max_dimension, + ): + agent._vprint( + f"{agent.log_prefix}📐 Image(s) exceeded provider size limit — " + f"shrank and retrying...", + force=True, + ) + return True, recovered_with_pool + else: + logger.info( + "image-shrink recovery: no data-URL image parts found " + "or shrink didn't reduce size; surfacing original error." + ) + + # Multimodal-tool-content recovery: strict OpenAI-spec providers 400 + # on list-type tool content. Strip images, mark (provider, model) + # no-list-tool-content for the session, retry once (#27344). + if ( + classified.reason == FailoverReason.multimodal_tool_content_unsupported + and not _retry.multimodal_tool_content_retry_attempted + ): + _retry.multimodal_tool_content_retry_attempted = True + if agent._try_strip_image_parts_from_tool_messages(api_messages): + agent._vprint( + f"{agent.log_prefix}📐 Provider rejected list-type tool content — " + f"downgraded screenshots to text and retrying...", + force=True, + ) + return True, recovered_with_pool + else: + logger.info( + "multimodal-tool-content recovery: no list-type tool " + "messages with image parts found; surfacing original error." + ) + + # Image-corrupt recovery: provider rejected the image bytes; shrinking + # can't help, so strip image parts and retry once (#69078). + if classified.reason == FailoverReason.image_corrupt: + # Strip ONLY the per-call copy: replacing msg["content"] on the + # shallow api_messages rows keeps canonical history's images + # (copy-on-write; transient rejection must not erase history). + _imgs_removed = False + if isinstance(api_messages, list): + _imgs_removed = _strip_images_from_messages(api_messages) + if _imgs_removed: + agent._vprint( + f"{agent.log_prefix}⚠️ Provider rejected a corrupted image — " + f"stripped images from the retry payload and retrying...", + force=True, + ) + return True, recovered_with_pool + else: + logger.info( + "image-corrupt recovery: no image parts found to " + "strip; surfacing original error." + ) + + # Anthropic OAuth subscription rejected the 1M-context beta: disable it + # for this session, rebuild the client, retry once. Reactive so capable + # subscriptions keep full 1M context (#17680). + if ( + classified.reason == FailoverReason.oauth_long_context_beta_forbidden + and agent.api_mode == "anthropic_messages" + and agent._is_anthropic_oauth + and not _retry.oauth_1m_beta_retry_attempted + ): + _retry.oauth_1m_beta_retry_attempted = True + if not getattr(agent, "_oauth_1m_beta_disabled", False): + agent._oauth_1m_beta_disabled = True + try: + agent._anthropic_client.close() + except Exception: + pass + agent._rebuild_anthropic_client() + agent._vprint( + f"{agent.log_prefix}🔕 OAuth subscription doesn't support " + f"the 1M-context beta — disabled for this session and retrying...", + force=True, + ) + return True, recovered_with_pool + + if ( + agent.api_mode == "codex_responses" + and agent.provider in {"openai-codex", "xai-oauth"} + and status_code == 401 + and not _retry.codex_auth_retry_attempted + ): + _retry.codex_auth_retry_attempted = True + if agent._try_refresh_codex_client_credentials(force=True): + _label = "xAI OAuth" if agent.provider == "xai-oauth" else "Codex" + agent._buffer_vprint(f"🔐 {_label} auth refreshed after 401. Retrying request...") + return True, recovered_with_pool + if ( + agent.api_mode == "chat_completions" + and agent.provider == "vertex" + and status_code == 401 + and not _retry.vertex_auth_retry_attempted + ): + _retry.vertex_auth_retry_attempted = True + if agent._try_refresh_vertex_client_credentials(): + agent._buffer_vprint("🔐 Vertex AI token refreshed after 401. Retrying request...") + return True, recovered_with_pool + if ( + agent.api_mode in ("chat_completions", "anthropic_messages") + and agent.provider == "nous" + and status_code == 401 + and not _retry.nous_auth_retry_attempted + ): + _retry.nous_auth_retry_attempted = True + if agent._try_refresh_nous_client_credentials(force=True): + agent._buffer_vprint("🔐 Nous agent key refreshed after 401. Retrying request...") + return True, recovered_with_pool + # Refresh didn't help: likely Portal OAuth expired/revoked, + # no credits, or agent key blocked. + from hermes_constants import display_hermes_home as _dhh_fn + _dhh = _dhh_fn() + _body_text = "" + try: + _body = getattr(api_error, "body", None) or getattr(api_error, "response", None) + if _body is not None: + _body_text = str(_body)[:200] + except Exception: + pass + print(f"{agent.log_prefix}🔐 Nous 401 — Portal authentication failed.") + if _body_text: + print(f"{agent.log_prefix} Response: {_body_text}") + if not _print_nous_entitlement_guidance(agent, "Nous model access"): + print(f"{agent.log_prefix} Most likely: Portal OAuth expired, account out of credits, or agent key revoked.") + print(f"{agent.log_prefix} Troubleshooting:") + print(f"{agent.log_prefix} • Re-authenticate: hermes auth add nous") + print(f"{agent.log_prefix} • Check credits / billing: https://portal.nousresearch.com") + print(f"{agent.log_prefix} • Verify stored credentials: {_dhh}/auth.json") + print(f"{agent.log_prefix} • Switch providers temporarily: /model --provider openrouter") + if ( + _is_copilot_provider(agent) + and status_code == 401 + and not _retry.copilot_auth_retry_attempted + ): + _retry.copilot_auth_retry_attempted = True + if agent._try_refresh_copilot_client_credentials(): + agent._buffer_vprint("🔐 Copilot credentials refreshed after 401. Retrying request...") + return True, recovered_with_pool + if ( + agent.api_mode == "anthropic_messages" + and status_code == 401 + and hasattr(agent, '_anthropic_api_key') + and not _retry.anthropic_auth_retry_attempted + ): + _retry.anthropic_auth_retry_attempted = True + from agent.anthropic_adapter import _is_oauth_token + from agent.azure_identity_adapter import is_token_provider + if agent._try_refresh_anthropic_client_credentials(): + print(f"{agent.log_prefix}🔐 Anthropic credentials refreshed after 401. Retrying request...") + return True, recovered_with_pool + # Credential refresh didn't help — show diagnostic info + key = agent._anthropic_api_key + print(f"{agent.log_prefix}🔐 Anthropic 401 — authentication failed.") + if is_token_provider(key): + # Azure Foundry Entra ID: JWT minted per-request by an httpx + # hook; 401 = Azure rejected it (RBAC, az login, IMDS). + print(f"{agent.log_prefix} Auth method: Microsoft Entra ID (httpx event hook)") + print(f"{agent.log_prefix} Run `hermes doctor` for credential-chain diagnostics, or") + print(f"{agent.log_prefix} `az login` if your developer session expired.") + else: + auth_method = "Bearer (OAuth/setup-token)" if _is_oauth_token(key) else "x-api-key (API key)" + print(f"{agent.log_prefix} Auth method: {auth_method}") + print(f"{agent.log_prefix} Token prefix: {key[:12]}..." if isinstance(key, str) and len(key) > 12 else f"{agent.log_prefix} Token: (empty or short)") + print(f"{agent.log_prefix} Troubleshooting:") + from hermes_constants import display_hermes_home as _dhh_fn + _dhh = _dhh_fn() + print(f"{agent.log_prefix} • Check ANTHROPIC_TOKEN in {_dhh}/.env for Hermes-managed OAuth/setup tokens") + print(f"{agent.log_prefix} • Check ANTHROPIC_API_KEY in {_dhh}/.env for API keys or legacy token values") + print(f"{agent.log_prefix} • For API keys: verify at https://platform.claude.com/settings/keys") + print(f"{agent.log_prefix} • For Claude Code: run 'claude /login' to refresh, then retry") + print(f"{agent.log_prefix} • Legacy cleanup: hermes config set ANTHROPIC_TOKEN \"\"") + print(f"{agent.log_prefix} • Clear stale keys: hermes config set ANTHROPIC_API_KEY \"\"") + + # Thinking block signature recovery: upstream mutation invalidates + # Anthropic's signature (400). Strip ``reasoning_details`` from + # ``api_messages`` only, never ``messages`` (state.db). One-shot. + if ( + classified.reason == FailoverReason.thinking_signature + and not _retry.thinking_sig_retry_attempted + ): + _retry.thinking_sig_retry_attempted = True + _api_stripped = 0 + for _m in api_messages: + if isinstance(_m, dict) and "reasoning_details" in _m: + _m.pop("reasoning_details", None) + _api_stripped += 1 + agent._vprint( + f"{agent.log_prefix}⚠️ Thinking block signature invalid, " + f"stripped reasoning_details from api_messages for retry...", + force=True, + ) + logger.warning( + "%sThinking block signature recovery: stripped " + "reasoning_details from %d api_messages " + "(canonical messages unchanged)", + agent.log_prefix, _api_stripped, + ) + return True, recovered_with_pool + + # ── Invalid encrypted reasoning replay recovery ─────── + # 400 ``invalid_encrypted_content`` on a stale ``codex_reasoning_items`` + # blob: disable replay for the session, strip cached items, retry once. + if ( + classified.reason == FailoverReason.invalid_encrypted_content + and not _retry.invalid_encrypted_content_retry_attempted + and agent.api_mode == "codex_responses" + and bool(getattr(agent, "_codex_reasoning_replay_enabled", True)) + and any( + isinstance(_m, dict) + and _m.get("role") == "assistant" + and isinstance(_m.get("codex_reasoning_items"), list) + and _m.get("codex_reasoning_items") + for _m in messages + ) + ): + _retry.invalid_encrypted_content_retry_attempted = True + replay_stats = agent._disable_codex_reasoning_replay(messages) + agent._vprint( + f"{agent.log_prefix}⚠️ Encrypted reasoning replay was rejected by the provider — " + f"disabled replay and stripped {replay_stats['items']} item(s) from " + f"{replay_stats['messages']} message(s), retrying...", + force=True, + ) + logger.warning( + "%sInvalid encrypted reasoning recovery: disabled replay and stripped %d items from %d messages", + agent.log_prefix, + replay_stats["items"], + replay_stats["messages"], + ) + return True, recovered_with_pool + + # ── Native compaction rejection recovery ────────────── + # Structured 400 naming ``context_management``: disable native + # compaction for the session, retry once; local compression takes over. + if ( + agent.api_mode == "codex_responses" + and not _retry.native_compaction_reject_retry_attempted + and bool(getattr(agent, "codex_responses_native_compaction", False)) + ): + from agent.native_compaction import is_native_compaction_rejection + if is_native_compaction_rejection( + api_error, getattr(api_error, "status_code", None) + ): + _retry.native_compaction_reject_retry_attempted = True + agent.codex_responses_native_compaction = False + agent._vprint( + f"{agent.log_prefix}⚠️ Provider rejected native compaction " + f"(context_management) — disabled for this session, " + f"local compression stays active. Retrying...", + force=True, + ) + logger.warning( + "%sNative compaction rejection recovery: disabled " + "codex_responses_native for this session and retrying", + agent.log_prefix, + ) + return True, recovered_with_pool + + # ── llama.cpp grammar-parse recovery ────────────────── + # ``json-schema-to-grammar`` rejects regex escapes and most ``format`` + # values: strip ``pattern``/``format`` from ``agent.tools``, retry once. + if ( + classified.reason == FailoverReason.llama_cpp_grammar_pattern + and not _retry.llama_cpp_grammar_retry_attempted + ): + _retry.llama_cpp_grammar_retry_attempted = True + try: + from tools.schema_sanitizer import strip_pattern_and_format + _, _stripped = strip_pattern_and_format(agent.tools) + except Exception as _strip_exc: # pragma: no cover — defensive + logger.warning( + "%sllama.cpp grammar recovery: strip helper failed: %s", + agent.log_prefix, _strip_exc, + ) + _stripped = 0 + if _stripped: + agent._vprint( + f"{agent.log_prefix}⚠️ llama.cpp rejected tool schema grammar — " + f"stripped {_stripped} pattern/format keyword(s), retrying...", + force=True, + ) + logger.warning( + "%sllama.cpp grammar recovery: stripped %d " + "pattern/format keyword(s) from tool schemas", + agent.log_prefix, _stripped, + ) + return True, recovered_with_pool + # No keywords found to strip — fall through to normal + # retry path rather than loop forever on the same error. + logger.warning( + "%sllama.cpp grammar error but no pattern/format " + "keywords to strip — falling through to normal retry", + agent.log_prefix, + ) + return False, recovered_with_pool + + +def nonretryable_client_error_result( + agent: Any, + api_error: Exception, + classified: Any, + *, + status_code: Optional[int], + api_kwargs: Any, + api_messages: Any, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_call_count: int, + approx_tokens: int, + provider: Any, + base_url: Any, + model: Any, +) -> Dict[str, Any]: + """Terminal path for a non-retryable 4xx once fallback is exhausted: dump the + request for debugging, flush the buffered retry trace, print actionable auth / + billing / content-policy / TLS guidance, persist the session (skipped for likely + context-overflow 400s so the failure does not grow the session, #1630) and build + the failed-turn result dict.""" + # Result/guidance helpers stay in the loop module (tests import + patch them + # there); lazy import avoids a load-time cycle. + from agent.conversation_loop import ( + _CONTENT_POLICY_RECOVERY_HINT, + _billing_failure_result, + _content_policy_blocked_result, + _print_billing_or_entitlement_guidance, + _print_nous_entitlement_guidance, + ) + + if api_kwargs is not None: + agent._dump_api_request_debug( + api_kwargs, reason="non_retryable_client_error", error=api_error, + ) + # Terminal — flush buffered context so the user sees + # what was tried before the abort. + agent._flush_status_buffer() + # Summarize once: Cloudflare/proxy HTML pages and raw provider + # bodies must be collapsed here or they leak verbatim via the + # ``error`` field. + _nonretryable_summary = agent._summarize_api_error(api_error) + if classified.reason == FailoverReason.content_policy_blocked: + agent._emit_status( + f"❌ Provider safety filter blocked this request: " + f"{_nonretryable_summary}" + ) + elif classified.reason == FailoverReason.ssl_cert_verification: + agent._emit_status( + f"❌ TLS certificate verification failed: " + f"{_nonretryable_summary}" + ) + else: + agent._emit_status( + f"❌ Non-retryable error (HTTP {status_code}): " + f"{_nonretryable_summary}" + ) + agent._vprint(f"{agent.log_prefix}❌ Non-retryable client error (HTTP {status_code}). Aborting.", force=True) + agent._vprint(f"{agent.log_prefix} 🔌 Provider: {provider} Model: {model}", force=True) + agent._vprint(f"{agent.log_prefix} 🌐 Endpoint: {base_url}", force=True) + # Actionable guidance for common auth errors + if classified.is_auth or classified.reason == FailoverReason.billing: + if classified.reason == FailoverReason.billing and _print_billing_or_entitlement_guidance( + agent, + capability="model access", + provider=provider, + base_url=str(base_url), + model=model, + unverified=classified.billing_unverified, + ): + pass + elif provider == "nous" and _print_nous_entitlement_guidance( + agent, + "Nous model access", + ): + pass + elif provider in {"openai-codex", "xai-oauth", "nous"} and status_code == 401: + if provider == "openai-codex": + agent._vprint(f"{agent.log_prefix} 💡 Codex OAuth token was rejected (HTTP 401). Your token may have been", force=True) + agent._vprint(f"{agent.log_prefix} refreshed by another client (Codex CLI, VS Code). To fix:", force=True) + agent._vprint(f"{agent.log_prefix} 1. Run `codex` in your terminal to generate fresh tokens.", force=True) + agent._vprint(f"{agent.log_prefix} 2. Then run `hermes auth` to re-authenticate.", force=True) + elif provider == "xai-oauth": + agent._vprint(f"{agent.log_prefix} 💡 xAI OAuth token was rejected (HTTP 401). To fix:", force=True) + agent._vprint(f"{agent.log_prefix} re-authenticate with xAI Grok OAuth (SuperGrok / Premium+) from `hermes model`.", force=True) + else: # nous + agent._vprint(f"{agent.log_prefix} 💡 Nous Portal OAuth token was rejected (HTTP 401). Your token may be", force=True) + agent._vprint(f"{agent.log_prefix} expired, revoked, or your account may be out of credits. To fix:", force=True) + agent._vprint(f"{agent.log_prefix} 1. Re-authenticate: hermes portal", force=True) + agent._vprint(f"{agent.log_prefix} 2. Check your portal account: https://portal.nousresearch.com", force=True) + # ``:free`` is OpenRouter slug syntax; Nous Portal will reject + # the model name even after a successful re-auth. + if isinstance(model, str) and model.endswith(":free"): + agent._vprint(f"{agent.log_prefix} ⚠️ Note: `{model}` looks like an OpenRouter slug (`:free` suffix).", force=True) + agent._vprint(f"{agent.log_prefix} Nous Portal won't recognize that model name. Either switch to a", force=True) + agent._vprint(f"{agent.log_prefix} Nous catalog model, or run `/model openrouter:{model}` to use OpenRouter.", force=True) + else: + agent._vprint(f"{agent.log_prefix} 💡 Your API key was rejected by the provider. Check:", force=True) + agent._vprint(f"{agent.log_prefix} • Is the key valid? Run: hermes setup", force=True) + agent._vprint(f"{agent.log_prefix} • Does your account have access to {model}?", force=True) + if base_url_host_matches(str(base_url), "openrouter.ai"): + agent._vprint(f"{agent.log_prefix} • Check credits: https://openrouter.ai/settings/credits", force=True) + else: + agent._vprint(f"{agent.log_prefix} 💡 This type of error won't be fixed by retrying.", force=True) + # Content-policy blocks get their own guidance: the provider refused + # this prompt, so recovery is a rephrase or another model, not + # key/retry advice. + if classified.reason == FailoverReason.content_policy_blocked: + agent._vprint( + f"{agent.log_prefix} 💡 The provider's safety filter rejected this specific prompt.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} • Try rephrasing the request, narrowing the context, or splitting into smaller steps.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} • Configure a fallback provider so future blocks route automatically:", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} hermes fallback add (interactive picker — same as `hermes model`)", + force=True, + ) + # TLS certificate failures are environment problems — name the knobs + # that fix each common cause. + if classified.reason == FailoverReason.ssl_cert_verification: + agent._vprint( + f"{agent.log_prefix} 💡 The TLS certificate chain could not be verified. This fails the same", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} way on every retry — fix the environment, then try again:", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} • Corporate TLS-inspecting proxy? Point Python at its CA bundle:", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} export SSL_CERT_FILE=/path/to/corp-ca.pem (also REQUESTS_CA_BUNDLE)", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} • Missing/stale system CA store? Install/refresh it:", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} pip install --upgrade certifi (macOS: run 'Install Certificates.command')", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} • Self-signed local endpoint (llama.cpp, LM Studio, vLLM)? Use http://", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} for localhost, or add the server's cert to your trust store.", + force=True, + ) + logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) + # Skip persistence on likely context-overflow (400 + large session): + # persisting the failed message grows the session and repeats the + # failure. (#1630) + if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80): + agent._vprint( + f"{agent.log_prefix}⚠️ Skipping session persistence " + f"for large failed session to prevent growth loop.", + force=True, + ) + else: + agent._persist_session(messages, conversation_history) + if classified.reason == FailoverReason.content_policy_blocked: + _policy_response = ( + "⚠️ The model provider's safety filter blocked this request " + "(not a Hermes/gateway failure).\n\n" + f"Provider message: {_nonretryable_summary}\n\n" + f"{_CONTENT_POLICY_RECOVERY_HINT}" + ) + return _content_policy_blocked_result( + messages, + api_call_count, + final_response=_policy_response, + error_detail=_nonretryable_summary, + ) + # Billing walls get the same structured recovery descriptor as the + # max-retries path so every surface renders one consistent signal. + if classified.reason == FailoverReason.billing: + return _billing_failure_result( + classified=classified, + summary=_nonretryable_summary, + messages=messages, + api_call_count=api_call_count, + provider=provider, + base_url=base_url, + model=model, + ) + return { + "final_response": _nonretryable_summary, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "failed": True, + "error": _nonretryable_summary, + } + + +def max_retries_exhausted_result( + agent: Any, + api_error: Exception, + classified: Any, + *, + max_retries: int, + is_rate_limited: bool, + error_msg: str, + api_kwargs: Any, + api_messages: Any, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_call_count: int, + approx_tokens: int, + provider: Any, + base_url: Any, + model: Any, +) -> Dict[str, Any]: + """Terminal path once ``retry_count >= max_retries`` and transport recovery + + fallback both failed: flush the buffered trace, emit the billing / rate-limit / + generic status line, print stream-drop or thinking-timeout guidance (the latter + wins, #52310), persist, and build the failed-turn result dict carrying the + classified ``failure_reason`` / ``failure_retryable`` / ``billing_block``.""" + # Result/guidance helpers stay in the loop module (tests import + patch them + # there); lazy import avoids a load-time cycle. + from agent.conversation_loop import ( + _billing_block_dict, + _billing_or_entitlement_message, + _billing_terminal_label, + _print_billing_or_entitlement_guidance, + ) + + # Terminal — flush buffered retry/fallback trace. + agent._flush_status_buffer() + _final_summary = agent._summarize_api_error(api_error) + _billing_guidance = "" + if classified.reason == FailoverReason.billing: + if classified.billing_unverified: + # Ambiguous body (#82154) — hedge the terminal line. + agent._emit_status( + "❌ Provider reported usage/credit exhaustion " + f"(unverified — may be a content-filter rejection) — {_final_summary}" + ) + else: + agent._emit_status(f"❌ Billing or credits exhausted — {_final_summary}") + _billing_guidance = _billing_or_entitlement_message( + capability="model access", + provider=provider, + base_url=str(base_url), + model=model, + unverified=classified.billing_unverified, + ) + _print_billing_or_entitlement_guidance( + agent, + capability="model access", + provider=provider, + base_url=str(base_url), + model=model, + unverified=classified.billing_unverified, + ) + elif is_rate_limited: + agent._emit_status(f"❌ Rate limited after {max_retries} retries — {_final_summary}") + else: + agent._emit_status(f"❌ API failed after {max_retries} retries — {_final_summary}") + agent._vprint(f"{agent.log_prefix} 💀 Final error: {_final_summary}", force=True) + + # SSE stream-drop (e.g. "Network connection lost"): usually a + # proxy/CDN cutting a very large tool call mid-response; give + # actionable guidance. + _is_stream_drop = ( + not getattr(api_error, "status_code", None) + and any(p in error_msg for p in ( + "connection lost", "connection reset", + "connection closed", "network connection", + "network error", "terminated", + )) + ) + if _is_stream_drop: + agent._vprint( + f"{agent.log_prefix} 💡 The provider's stream " + f"connection keeps dropping. This often happens " + f"when the model tries to write a very large " + f"file in a single tool call.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} Try asking the model " + f"to use execute_code with Python's open() for " + f"large files, or to write the file in smaller " + f"sections.", + force=True, + ) + + # Thinking-timeout: a known reasoning model hit a transport error + # before the first content token. Distinct from _is_stream_drop; + # detection lives in agent.thinking_timeout_guidance. (#52310) + from agent.thinking_timeout_guidance import ( + is_thinking_timeout, + ) + _is_thinking_timeout = is_thinking_timeout( + classified, + model, + error_msg, + ) + if _is_thinking_timeout: + agent._vprint( + f"{agent.log_prefix} 💡 The model's thinking " + f"phase exceeded the upstream proxy's idle " + f"timeout before the first content token " + f"arrived. This is a known issue with " + f"reasoning models behind cloud gateways " + f"(NVIDIA NIM, OpenAI, Anthropic, DeepSeek).", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} Workarounds in priority order:", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} 1. Set " + f"`providers.{provider}.models.{model}.stale_timeout_seconds: 900` " + f"in `~/.hermes/config.yaml` to extend the per-call " + f"timeout. (Hermes's built-in floor is 600s for " + f"known reasoning models — if you still see this " + f"after raising, the upstream cap is even shorter.)", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} 2. Lower `reasoning_budget` or set " + f"`reasoning_effort: medium` on this model if the provider supports it.", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} 3. Use a smaller / faster reasoning " + f"model if the task doesn't require deep thinking.", + force=True, + ) + + logger.error( + "%sAPI call failed after %s retries. %s | provider=%s model=%s msgs=%s tokens=~%s", + agent.log_prefix, max_retries, _final_summary, + provider, model, len(api_messages), f"{approx_tokens:,}", + ) + if api_kwargs is not None: + agent._dump_api_request_debug( + api_kwargs, reason="max_retries_exhausted", error=api_error, + ) + agent._persist_session(messages, conversation_history) + _billing_block = None + _billing_unverified = False + if classified.reason == FailoverReason.billing: + _billing_unverified = classified.billing_unverified + _final_response = _billing_terminal_label( + _final_summary, _billing_unverified + ) + if _billing_guidance: + _final_response += f"\n\n{_billing_guidance}" + # Structured recovery descriptor so every surface renders + # the same link + label from one signal (see helper). + _billing_block = _billing_block_dict( + provider, base_url, model, _billing_guidance, + unverified=_billing_unverified, + ) + else: + _final_response = f"API call failed after {max_retries} retries: {_final_summary}" + if _is_thinking_timeout: + # Thinking-timeout guidance overrides stream-drop guidance, + # which would wrongly suggest splitting large file writes. + from agent.thinking_timeout_guidance import ( + build_thinking_timeout_guidance, + ) + _final_response += build_thinking_timeout_guidance( + provider=provider, + model=model, + ) + elif _is_stream_drop: + _final_response += ( + "\n\nThe provider's stream connection keeps " + "dropping — this often happens when generating " + "very large tool call responses (e.g. write_file " + "with long content). Try asking me to use " + "execute_code with Python's open() for large " + "files, or to write in smaller sections." + ) + return { + "final_response": _final_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "failed": True, + "error": _final_summary, + # Expose the classified reason so callers (kanban worker in + # cli.py) can tell a quota wall (``rate_limit`` / ``billing``) + # from a task failure. + "failure_reason": classified.reason.value, + # The classifier's own retry verdict — UI surfaces use + # this instead of re-deriving from the reason string. + "failure_retryable": bool(classified.retryable), + # True when the billing verdict rests on an ambiguous + # body (#82154) — may be a content-filter rejection. + "billing_unverified": _billing_unverified, + # Present only for billing walls: structured recovery + # descriptor (provider, billing_url, is_nous, message). + "billing_block": _billing_block, + } + + +def log_api_error_attempt( + agent: Any, + api_error: Exception, + *, + retry_count: int, + max_retries: int, + status_code: Optional[int], + elapsed_time: float, + api_messages: Any, + approx_tokens: int, +) -> Tuple[str, str, Any, Any, Any]: + """Log one failed API attempt: the ``API call failed`` warning plus the buffered + retry trace (provider/endpoint/error/4xx body/elapsed), the OpenRouter + "no tool endpoints" hint and the bare-404 missing-vendor-prefix hint (#78796). + The buffer only surfaces if every retry+fallback exhausts. + + Returns ``(error_type, error_msg, provider, base_url, model)`` — the loop's + ``error_type`` / ``error_msg`` / ``_provider`` / ``_base`` / ``_model`` locals.""" + error_type = type(api_error).__name__ + error_msg = str(api_error).lower() + _error_summary = agent._summarize_api_error(api_error) + logger.warning( + "API call failed (attempt %s/%s) error_type=%s %s summary=%s", + retry_count, + max_retries, + error_type, + agent._client_log_context(), + _error_summary, + ) + + _provider = getattr(agent, "provider", "unknown") + _base = getattr(agent, "base_url", "unknown") + _model = getattr(agent, "model", "unknown") + _status_code_str = f" [HTTP {status_code}]" if status_code else "" + agent._buffer_vprint(f"⚠️ API call failed (attempt {retry_count}/{max_retries}): {error_type}{_status_code_str}") + agent._buffer_vprint(f" 🔌 Provider: {_provider} Model: {_model}") + agent._buffer_vprint(f" 🌐 Endpoint: {_base}") + agent._buffer_vprint(f" 📝 Error: {_error_summary}") + if status_code and status_code < 500: + _err_body = getattr(api_error, "body", None) + _err_body_str = str(_err_body)[:300] if _err_body else None + if _err_body_str: + agent._buffer_vprint(f" 📋 Details: {_err_body_str}") + agent._buffer_vprint(f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens") + + # OpenRouter "no tool endpoints" hint, buffered with the retry trace + # so it only surfaces if every retry+fallback exhausts. + if ( + agent._is_openrouter_url() + and "support tool use" in error_msg + ): + agent._buffer_vprint( + f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings." + ) + if agent.providers_allowed: + agent._buffer_vprint( + " Your provider_routing.only restriction is filtering out tool-capable providers." + ) + agent._buffer_vprint( + " Try removing the restriction or adding providers that support tools for this model." + ) + agent._buffer_vprint( + f" Check which providers support tools: https://openrouter.ai/models/{_model}" + ) + + # Bare 404 on a ``vendor/model`` catalogue usually means the id lost its + # prefix; the provider never names the model, so we do (#78796). + if getattr(api_error, "status_code", None) == 404: + try: + from hermes_cli.model_normalize import suggest_prefixed_model_id + + _suggestion = suggest_prefixed_model_id(_provider, _model) + except Exception: + _suggestion = None + if _suggestion: + agent._buffer_vprint( + f" 💡 Model '{_model}' is not a valid id for provider {_provider} — " + f"it is missing its vendor prefix." + ) + agent._buffer_vprint( + f" Did you mean '{_suggestion}'? Re-pick it with `hermes model`." + ) + return error_type, error_msg, _provider, _base, _model + + +def interruptible_backoff_sleep( + agent: Any, + wait_time: float, + _retry: Optional[TurnRetryState], + *, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_call_count: int, + abort_message: str, + interrupt_text: str, + activity_label: str, +) -> Optional[Dict[str, Any]]: + """Sleep ``wait_time`` in 200 ms slices so interrupts are honoured promptly, touching + activity every ~30 s so the gateway's inactivity monitor knows we are alive. + + On interrupt: when ``_retry`` is given and a redirect is pending, preserve it + (``clear_interrupt(preserve_redirect=True)``), set + ``_retry.restart_with_redirected_messages`` and return ``None`` — the caller + rebuilds the turn from the correction. Otherwise close any open tool sequence, + persist, clear the interrupt and return the ``interrupted`` result dict. + Returns ``None`` when the wait completed.""" + sleep_end = time.time() + wait_time + _touch_counter = 0 + while time.time() < sleep_end: + if agent._interrupt_requested: + if _retry is not None and agent.clear_interrupt(preserve_redirect=True): + _retry.restart_with_redirected_messages = True + return None + agent._vprint(f"{agent.log_prefix}⚡ {abort_message}", force=True) + close_interrupted_tool_sequence(messages, interrupt_text) + agent._persist_session(messages, conversation_history) + agent.clear_interrupt() + return { + "final_response": interrupt_text, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "interrupted": True, + } + time.sleep(0.2) + _touch_counter += 1 + if _touch_counter % 150 == 0: # 150 × 0.2s = 30s + agent._touch_activity( + f"{activity_label}, {int(sleep_end - time.time())}s remaining" + ) + return None + + +def compute_error_backoff( + agent: Any, + api_error: Exception, + *, + retry_count: int, + max_retries: int, + is_rate_limited: bool, + is_zai_coding_overload: bool, + base_url: Any, + model: Any, +) -> float: + """Pick the wait before the next API retry and announce it. + + Retry-After header wins for rate limits (capped at 600s: Anthropic Tier 1 buckets + reset in ~171s, so a 120s cap retried early and re-tripped the limit, #26293); + otherwise jittered exponential backoff, replaced by the adaptive rate-limit policy + for 429s / Z.AI overloads. Normal retries are buffered to avoid chatter; long Z.AI + Coding waits can last minutes, so those surface immediately.""" + # Resolved through the loop module so tests that patch + # ``agent.conversation_loop.jittered_backoff`` / ``adaptive_rate_limit_backoff`` + # (incl. the run_agent conftest fast-backoff fixture) keep intercepting. + from agent.conversation_loop import adaptive_rate_limit_backoff, jittered_backoff + + _retry_after = None + if is_rate_limited: + _resp_headers = getattr(getattr(api_error, "response", None), "headers", None) + if _resp_headers and hasattr(_resp_headers, "get"): + _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") + if _ra_raw: + try: + _retry_after = min(float(_ra_raw), 600) + except (TypeError, ValueError): + pass + wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) + _backoff_policy = None + if (is_rate_limited or is_zai_coding_overload) and not _retry_after: + wait_time, _backoff_policy = adaptive_rate_limit_backoff( + retry_count, + base_url=str(base_url), + model=model, + error=api_error, + default_wait=wait_time, + ) + if is_rate_limited or is_zai_coding_overload: + _policy_note = "" + if _backoff_policy == "zai_coding_overload_long": + _policy_note = " (Z.AI Coding overload adaptive long backoff)" + elif _backoff_policy == "zai_coding_overload_short": + _policy_note = " (Z.AI Coding overload short retry)" + _wait_reason = "Provider overloaded" if is_zai_coding_overload and not is_rate_limited else "Rate limited" + _rate_limit_status = f"⏱️ {_wait_reason}. Waiting {wait_time:.1f}s (attempt {retry_count + 1}/{max_retries}){_policy_note}..." + if _backoff_policy == "zai_coding_overload_long": + agent._emit_status(_rate_limit_status) + else: + agent._buffer_status(_rate_limit_status) + else: + agent._buffer_status(f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})...") + logger.warning( + "Retrying API call in %ss (attempt %s/%s) %s policy=%s error=%s", + wait_time, + retry_count, + max_retries, + agent._client_log_context(), + _backoff_policy or "default", + api_error, + ) + return wait_time + + +def validate_response_shape(agent: Any, response: Any) -> Tuple[bool, List[str]]: + """Validate the raw provider response per api_mode via the transport's + ``validate_response``. Returns ``(response_invalid, error_details)``; a Codex + ``failed``/``cancelled`` status (terminal provider failure, e.g. quota exhaustion) + is treated as invalid so the fallback chain triggers, while an empty Codex + ``output`` with a non-empty ``output_text`` is deferred to normalization.""" + response_invalid = False + error_details = [] + if agent.api_mode == "codex_responses": + _ct_v = agent._get_transport() + if not _ct_v.validate_response(response): + if response is None: + response_invalid = True + error_details.append("response is None") + else: + # Terminal provider failure (e.g. quota exhaustion): treat + # as invalid so the fallback chain triggers. + _codex_resp_status = str(getattr(response, "status", "") or "").strip().lower() + if _codex_resp_status in {"failed", "cancelled"}: + _codex_error_obj = getattr(response, "error", None) + _codex_error_msg = ( + _codex_error_obj.get("message") if isinstance(_codex_error_obj, dict) + else str(_codex_error_obj) if _codex_error_obj + else f"Responses API returned status '{_codex_resp_status}'" + ) + logger.warning( + "Codex response status='%s' (error=%s). Routing to fallback. %s", + _codex_resp_status, _codex_error_msg, + agent._client_log_context(), + ) + response_invalid = True + error_details.append(f"response.status={_codex_resp_status}: {_codex_error_msg}") + else: + # output_text fallback: stream backfill may have failed + # but normalize can still recover from output_text + _out_text = getattr(response, "output_text", None) + _out_text_stripped = _out_text.strip() if isinstance(_out_text, str) else "" + if _out_text_stripped: + logger.debug( + "Codex response.output is empty but output_text is present " + "(%d chars); deferring to normalization.", + len(_out_text_stripped), + ) + else: + _resp_status = getattr(response, "status", None) + _resp_incomplete = getattr(response, "incomplete_details", None) + logger.warning( + "Codex response.output is empty after stream backfill " + "(status=%s, incomplete_details=%s, model=%s). %s", + _resp_status, _resp_incomplete, + getattr(response, "model", None), + f"api_mode={agent.api_mode} provider={agent.provider}", + ) + response_invalid = True + error_details.append("response.output is empty") + elif agent.api_mode == "anthropic_messages": + _tv = agent._get_transport() + if not _tv.validate_response(response): + response_invalid = True + if response is None: + error_details.append("response is None") + else: + error_details.append("response.content invalid (not a non-empty list)") + elif agent.api_mode == "bedrock_converse": + _btv = agent._get_transport() + if not _btv.validate_response(response): + response_invalid = True + if response is None: + error_details.append("response is None") + else: + error_details.append("Bedrock response invalid (no output or choices)") + else: + _ctv = agent._get_transport() + if not _ctv.validate_response(response): + response_invalid = True + if response is None: + error_details.append("response is None") + elif not hasattr(response, 'choices'): + error_details.append("response has no 'choices' attribute") + elif response.choices is None: + error_details.append("response.choices is None") + else: + error_details.append("response.choices is empty") + return response_invalid, error_details + + +def describe_invalid_response(agent: Any, response: Any, api_duration: float) -> Tuple[str, str, str]: + """Diagnostics for an empty/malformed response: ``(error_msg, provider_name, + failure_hint)``. The hint is derived from the provider error code (524/504/429/ + 5xx) and the response time, instead of always assuming rate limiting.""" + # Check for error field in response (some providers include this) + error_msg = "Unknown" + provider_name = "Unknown" + if response and hasattr(response, 'error') and response.error: + error_msg = str(response.error) + # Try to extract provider from error metadata + if hasattr(response.error, 'metadata') and response.error.metadata: + provider_name = response.error.metadata.get('provider_name', 'Unknown') + elif response and hasattr(response, 'message') and response.message: + error_msg = str(response.message) + + # Try to get provider from model field (OpenRouter often returns actual model used) + if provider_name == "Unknown" and response and hasattr(response, 'model') and response.model: + provider_name = f"model={response.model}" + + # Check for x-openrouter-provider or similar metadata + if provider_name == "Unknown" and response: + # Log all response attributes for debugging + resp_attrs = {k: str(v)[:100] for k, v in vars(response).items() if not k.startswith('_')} + if agent.verbose_logging: + logging.debug(f"Response attributes for invalid response: {resp_attrs}") + + # Extract error code from response for contextual diagnostics + _resp_error_code = None + if response and hasattr(response, 'error') and response.error: + _code_raw = getattr(response.error, 'code', None) + if _code_raw is None and isinstance(response.error, dict): + _code_raw = response.error.get('code') + if _code_raw is not None: + try: + _resp_error_code = int(_code_raw) + except (TypeError, ValueError): + pass + + # Build a human-readable failure hint from the error code + # and response time, instead of always assuming rate limiting. + if _resp_error_code == 524: + _failure_hint = f"upstream provider timed out (Cloudflare 524, {api_duration:.0f}s)" + elif _resp_error_code == 504: + _failure_hint = f"upstream gateway timeout (504, {api_duration:.0f}s)" + elif _resp_error_code == 429: + _failure_hint = "rate limited by upstream provider (429)" + elif _resp_error_code in {500, 502}: + _failure_hint = f"upstream server error ({_resp_error_code}, {api_duration:.0f}s)" + elif _resp_error_code in {503, 529}: + _failure_hint = f"upstream provider overloaded ({_resp_error_code})" + elif _resp_error_code is not None: + _failure_hint = f"upstream error (code {_resp_error_code}, {api_duration:.0f}s)" + elif api_duration < 10: + _failure_hint = f"fast response ({api_duration:.1f}s) — likely rate limited" + elif api_duration > 60: + _failure_hint = f"slow response ({api_duration:.0f}s) — likely upstream timeout" + else: + _failure_hint = f"response time {api_duration:.1f}s" + return error_msg, provider_name, _failure_hint + + +@dataclass +class ClassifiedErrorVerdict: + """Outcome of ``route_classified_error``. + + ``action``: ``"return"`` (terminal result), ``"break"`` (restart armed on + ``_retry``), ``"continue"`` (re-enter the retry loop; Nous guard re-check) or + ``"fallthrough"`` (proceed to overflow / client-error / backoff handling). The + remaining fields are loop locals the router may have rebound or computed for the + later steps (``is_rate_limited``, ``wrapped_output_cap_budget``, + ``is_zai_coding_overload`` feed the overflow entry, client-error gate and backoff).""" + + action: str + result: Optional[Dict[str, Any]] + status_code: Optional[int] + messages: List[Dict[str, Any]] + active_system_prompt: Any + conversation_history: Any + retry_count: int + max_retries: int + compression_attempts: int + provider_overflow_recovery_pending: bool + is_rate_limited: bool + wrapped_output_cap_budget: Optional[int] + is_zai_coding_overload: bool + + +def route_classified_error( + agent: Any, + api_error: Exception, + classified: Any, + _retry: TurnRetryState, + *, + error_msg: str, + error_context: Any, + recovered_with_pool: bool, + base_url: Any, + model: Any, + messages: List[Dict[str, Any]], + api_messages: Any, + system_message: Any, + active_system_prompt: Any, + conversation_history: Any, + retry_count: int, + max_retries: int, + compression_attempts: int, + max_compression_attempts: int, + api_call_count: int, + effective_task_id: Any, +) -> ClassifiedErrorVerdict: + """Ordered recovery steps between classification logging and overflow handling: + compaction-disabled overflow → terminal error (output-cap errors exempt); + Anthropic long-context tier 429 → cap at 200k and compress; eager fallback for + rate-limit/billing (immediately) and transport failures (after 1 retry), unless + credential-pool rotation may still recover (#11314; upstream-aggregator 429s always + fall back); persistent 401/403 → fallback chain once; genuine Nous 429 → record to + the cross-session breaker and re-enter the loop exactly once so the top-of-loop guard + runs. Order is load-bearing.""" + from agent.conversation_loop import ( + _arm_fallback_restart, + _ra, + conversation_history_after_compression, + estimate_request_tokens_rough, + ) + + _base = base_url + _model = model + _provider_overflow_recovery_pending = False + is_rate_limited = False + _wrapped_output_cap_budget = None + _is_zai_coding_overload = False + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> ClassifiedErrorVerdict: + return ClassifiedErrorVerdict( + action=action, + result=result, + status_code=status_code, + messages=messages, + active_system_prompt=active_system_prompt, + conversation_history=conversation_history, + retry_count=retry_count, + max_retries=max_retries, + compression_attempts=compression_attempts, + provider_overflow_recovery_pending=_provider_overflow_recovery_pending, + is_rate_limited=is_rate_limited, + wrapped_output_cap_budget=_wrapped_output_cap_budget, + is_zai_coding_overload=_is_zai_coding_overload, + ) + + # Check 413 BEFORE the generic 4xx handler: compress + retry, not abort. + status_code = getattr(api_error, "status_code", None) + + # ── Respect disabled auto-compaction on overflow ────── + # ``compression.enabled: false`` forbids every automatic trigger, incl. + # these overflow recovery paths; error out. Output-cap errors exempt. + _overflow_reasons = { + FailoverReason.long_context_tier, + FailoverReason.payload_too_large, + FailoverReason.context_overflow, + } + _is_output_cap_error = ( + is_output_cap_error(error_msg) + or parse_available_output_tokens_from_error(error_msg) is not None + ) + if ( + classified.reason in _overflow_reasons + and not getattr(agent, "compression_enabled", True) + and not _is_output_cap_error + ): + agent._flush_status_buffer() + agent._vprint( + f"{agent.log_prefix}❌ Context overflow, but auto-compaction is disabled " + f"(compression.enabled: false).", + force=True, + ) + agent._vprint( + f"{agent.log_prefix} 💡 Run /compress to compact manually, /new to start fresh, " + f"switch to a larger-context model, or reduce attachments.", + force=True, + ) + logger.error( + f"{agent.log_prefix}Context overflow ({classified.reason.value}) with " + f"auto-compaction disabled — not compressing." + ) + agent._persist_session(messages, conversation_history) + _final_response = ( + "Context overflow and auto-compaction is disabled " + "(compression.enabled: false). Run /compress to compact manually, " + "/new to start fresh, or switch to a larger-context model." + ) + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "completed": False, + "api_calls": api_call_count, + "error": _final_response, + "partial": True, + "failed": True, + "compaction_disabled": True, + }) + + # ── Anthropic Sonnet long-context tier gate ─────────── + # 429 "Extra usage is required for long context requests" is a + # subscription-tier limit, not transient: cap at 200k and compress. + if classified.reason == FailoverReason.long_context_tier: + _reduced_ctx = 200000 + compressor = agent.context_compressor + old_ctx = compressor.context_length + if old_ctx > _reduced_ctx: + compressor.update_model( + model=agent.model, + context_length=_reduced_ctx, + base_url=agent.base_url, + api_key=getattr(agent, "api_key", ""), + provider=agent.provider, + api_mode=agent.api_mode, + ) + # Context probing flags — only set on built-in + # compressor (plugin engines manage their own). + if hasattr(compressor, "_context_probed"): + compressor._context_probed = True + # Don't persist — subscription-tier limit, not a model + # capability; 1M should return if extra usage is enabled. + compressor._context_probe_persistable = False + agent._buffer_vprint( + f"⚠️ Anthropic long-context tier " + f"requires extra usage — reducing context: " + f"{old_ctx:,} → {_reduced_ctx:,} tokens" + ) + + compression_attempts += 1 + if compression_attempts <= max_compression_attempts: + original_len = len(messages) + # Option A (LCM issue 441): overhead-aware request size so recovery arms on + # the true request (msgs + tools + system), not the tool-blind message count. + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), + task_id=effective_task_id, + ) + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + if len(messages) < original_len or old_ctx > _reduced_ctx: + agent._buffer_status( + COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE.format( + new_ctx=_reduced_ctx, old_ctx=old_ctx + ) + ) + time.sleep(2) + # Provider proved the request doesn't fit the reduced + # window; row count isn't proof the rebuilt one does. + # Recheck the complete request before the next call. + _provider_overflow_recovery_pending = True + _retry.restart_with_compressed_messages = True + return _verdict("break") + # Fall through to normal error handling if compression + # is exhausted or didn't help. + + # Eager fallback: rate-limit/billing switch immediately (primary won't + # recover in the retry window); transport errors get 1 retry first. + is_rate_limited = classified.reason in { + FailoverReason.rate_limit, + FailoverReason.billing, + FailoverReason.upstream_rate_limit, + } + # Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only + # the max_tokens clamp fixes it (#72281). Parsed once; gates the + # eager-fallback exemption and the overflow entry below. + _wrapped_output_cap_budget = ( + parse_available_output_tokens_from_error(error_msg) + if classified.reason == FailoverReason.rate_limit + else None + ) + _is_transport_failure = classified.reason in { + FailoverReason.timeout, + FailoverReason.overloaded, + } + # Z.AI overload 429s classify `overloaded`, which `is_rate_limited` + # excludes. Detect directly so the long backoff runs, and raise the + # ceiling to reach it (see zai_coding_overload_retry_ceiling()). + _is_zai_coding_overload = is_zai_coding_overload_error( + base_url=str(_base), model=_model, error=api_error + ) + if _is_zai_coding_overload: + max_retries = max(max_retries, zai_coding_overload_retry_ceiling()) + _should_fallback = ( + (is_rate_limited and _wrapped_output_cap_budget is None) + or (_is_transport_failure and retry_count >= 2) + ) + if _should_fallback and agent._fallback_index < len(agent._fallback_chain): + # No eager fallback while credential pool rotation may recover + # (_pool_may_recover_from_rate_limit, #11314). Exception: an + # upstream-aggregator 429 — the pool can't help, always fall back. + _is_upstream = classified.reason == FailoverReason.upstream_rate_limit + pool_may_recover = ( + False if _is_upstream + else _ra()._pool_may_recover_from_rate_limit( + agent._credential_pool, + ) + ) + if not pool_may_recover: + if _is_upstream: + _upstream_name = (classified.error_context or {}).get( + "upstream_provider", "aggregator" + ) + agent._buffer_status( + f"⚠️ Upstream {_upstream_name} rate-limited — " + "switching to fallback model..." + ) + elif classified.reason == FailoverReason.billing: + if classified.billing_unverified: + # Ambiguous body (#82154) — don't assert billing. + agent._buffer_status( + "⚠️ Provider reported usage/credit exhaustion " + "(unverified — may be a content-filter rejection) " + "— switching to fallback provider..." + ) + else: + agent._buffer_status( + "⚠️ Billing or credits exhausted — switching to fallback provider..." + ) + elif _is_transport_failure: + agent._buffer_status( + "⚠️ Provider unreachable — switching to fallback provider..." + ) + else: + agent._buffer_status("⚠️ Rate limited — switching to fallback provider...") + if agent._try_activate_fallback(reason=classified.reason): + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) + retry_count = 0 + compression_attempts = 0 + return _verdict("break") + + # ── Auth-failure provider failover ─────────────────────── + # A 401/403 surviving credential refresh means a broken credential or + # endpoint: escalate to the fallback chain; False -> terminal handling. + if ( + classified.is_auth + and not _retry.auth_failover_attempted + and agent._fallback_index < len(agent._fallback_chain) + ): + _retry.auth_failover_attempted = True + agent._buffer_status( + "🔐 Authentication failed and could not be refreshed — " + "switching to fallback provider..." + ) + if agent._try_activate_fallback(reason=classified.reason): + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) + retry_count = 0 + compression_attempts = 0 + return _verdict("break") + + # ── Nous Portal: record rate limit & skip retries ───── + # A genuine account-level 429 is recorded to a shared file so ALL + # sessions back off; is_genuine_nous_rate_limit excludes upstream 429s. + if ( + is_rate_limited + and agent.provider == "nous" + and classified.reason == FailoverReason.rate_limit + and not recovered_with_pool + ): + _genuine_nous_rate_limit = False + try: + from agent.nous_rate_guard import ( + is_genuine_nous_rate_limit, + record_nous_rate_limit, + ) + _err_resp = getattr(api_error, "response", None) + _err_hdrs = ( + getattr(_err_resp, "headers", None) + if _err_resp else None + ) + _genuine_nous_rate_limit = is_genuine_nous_rate_limit( + headers=_err_hdrs, + last_known_state=agent._rate_limit_state, + ) + if _genuine_nous_rate_limit: + record_nous_rate_limit( + headers=_err_hdrs, + error_context=error_context, + ) + else: + logger.info( + "Nous 429 looks like upstream capacity " + "(no exhausted bucket in headers or " + "last-known state) -- not tripping " + "cross-session breaker." + ) + except Exception: + pass + if _genuine_nous_rate_limit: + # Re-enter the loop exactly once so the top-of-loop Nous guard + # runs (retry_count = max_retries would skip it entirely). + retry_count = max(0, max_retries - 1) + return _verdict("continue") + # Upstream capacity 429: normal retry logic will typically succeed. + return _verdict("fallthrough") diff --git a/agent/turn_retry_state.py b/agent/turn_retry_state.py index 49790c6528..42bc2fddad 100644 --- a/agent/turn_retry_state.py +++ b/agent/turn_retry_state.py @@ -1,28 +1,8 @@ -"""Per-attempt recovery bookkeeping for the conversation turn loop. +"""Per-attempt recovery bookkeeping (``TurnRetryState``) for the conversation turn loop. -The inner retry loop in ``run_conversation`` (``while retry_count < -max_retries``) makes several distinct recovery attempts on a single model API -call: a credential-pool 429 retry, a per-provider OAuth refresh (codex, -anthropic, nous, copilot), a long-context compression restart, a length- -continuation restart, and a handful of format-recovery branches (thinking- -signature stripping, multimodal-tool-content stripping, llama.cpp grammar -fallback, image shrink, invalid-encrypted-content, 1M-beta header). - -Each of those branches is guarded by a one-shot boolean so it fires at most -once per attempt. They used to be ~16 bare ``*_attempted`` / ``has_retried_*`` -/ ``restart_with_*`` locals declared inline before the loop and threaded -through its 2,400-line body. ``TurnRetryState`` collapses them into one object -the loop mutates in place (``state.codex_auth_retry_attempted = True``), giving -the recovery bookkeeping a single named, testable home. - -Loop-control variables (``retry_count``, ``max_retries``, -``max_compression_attempts``) intentionally stay as plain locals — they are the -``while`` mechanics, not recovery bookkeeping, and putting them on the object -would add indirection without clarifying anything. - -This module is dependency-free so it can be unit-tested in isolation and -imported by the turn loop without an import cycle. -""" +Each one-shot recovery branch of the inner retry loop is guarded by a flag here so it +fires at most once per attempt. Loop-control (``retry_count``, ``max_retries``) stays +as plain locals. Dependency-free so it imports without a cycle.""" from __future__ import annotations @@ -33,11 +13,8 @@ from dataclasses import dataclass, fields class TurnRetryState: """One-shot recovery guards + restart signals for a single API-call attempt. - A fresh instance is created for each iteration of the outer turn loop - (once per ``api_call_count``). Each guard fires its recovery branch at most - once; the ``restart_with_*`` signals are read by the loop after the attempt - to decide whether to rebuild the request and retry. - """ + A fresh instance is created per ``api_call_count`` iteration; each guard fires at + most once, and ``restart_with_*`` signals tell the loop to rebuild and retry.""" # ── Per-provider OAuth / credential refresh guards ─────────────────── codex_auth_retry_attempted: bool = False @@ -46,12 +23,8 @@ class TurnRetryState: nous_paid_entitlement_refresh_attempted: bool = False copilot_auth_retry_attempted: bool = False # Copilot surfaces a stale/degraded credential as a 400 - # ``model_not_available_for_integrator`` / ``model_not_supported`` instead - # of a clean 401 (e.g. a raw OAuth token seeded when the token exchange - # degraded at startup, routing the request to the restricted - # ``copilot-language-server`` integrator). Guard a single-shot forced - # re-exchange + client rebuild for that case, separate from the 401 guard - # so both can fire within one attempt if needed. + # ``model_not_available_for_integrator`` / ``model_not_supported``, not a 401. + # Single-shot forced re-exchange + rebuild, separate from the 401 guard. copilot_stale_cred_retry_attempted: bool = False vertex_auth_retry_attempted: bool = False @@ -69,22 +42,19 @@ class TurnRetryState: has_retried_429: bool = False # ── Auth-failure provider failover ─────────────────────────────────── - # Set once we've escalated a persistent 401/403 (after the per-provider - # credential-refresh attempt above failed) to the fallback chain, so we - # don't loop on the same auth failover within one attempt. + # Set once a persistent 401/403 has been escalated to the fallback chain, so + # we don't loop on the same auth failover within one attempt. auth_failover_attempted: bool = False # ── Restart signals (read by the outer loop after the attempt) ─────── restart_with_compressed_messages: bool = False restart_with_length_continuation: bool = False - # Set when a content-filter stream stall (e.g. MiniMax "new_sensitive") - # has been escalated to the fallback chain: the partial-stream content - # was rolled back off ``messages`` and the loop should re-issue the API - # call against the newly-activated provider (#32421). + # Set when a content-filter stream stall (e.g. MiniMax "new_sensitive") was + # escalated to the fallback chain: partial content was rolled back off + # ``messages``; re-issue the call against the new provider (#32421). restart_with_rebuilt_messages: bool = False - # A user correction cancelled the in-flight provider request. The outer - # loop must append a role-safe checkpoint + user message, rebuild the API - # payload, and retry the same logical iteration. + # A user correction cancelled the in-flight request: append a role-safe checkpoint + + # user message, rebuild the payload, and retry the same logical iteration. restart_with_redirected_messages: bool = False def __iter__(self): diff --git a/agent/turn_stop_gates.py b/agent/turn_stop_gates.py new file mode 100644 index 0000000000..7260566d75 --- /dev/null +++ b/agent/turn_stop_gates.py @@ -0,0 +1,207 @@ +"""Text-response stop gates for the conversation turn loop. + +Extracted from ``run_conversation``. When the model stops with a text answer, three +gates may instead append the answer as an interim row plus a synthetic user-role nudge +and continue the turn: verify-on-stop (#65919), the ``pre_verify`` plugin hook after code +edits, and the kanban worker terminal-tool guard. Each keeps the candidate answer as a +budget-exhaustion fallback (``pending_verification_response``) and clears +``final_response`` so the finalizer can tell this gate from error exits (#61631). +Nothing here imports ``agent.conversation_loop`` at module level (cycle). +""" + +from __future__ import annotations + +import logging +import os +from dataclasses import dataclass +from typing import Any, Dict, List + +from agent.message_metadata import append_message + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class StopGateVerdict: + """``continue_turn`` True → a nudge was appended; re-enter the turn loop with + ``final_response=None`` and the pending-verification fields updated.""" + + continue_turn: bool + final_response: Any + pending_verification_response: Any + pending_verification_response_previewed: Any + + +def apply_stop_gates( + agent: Any, + final_msg: Dict[str, Any], + *, + final_response: Any, + messages: List[Dict[str, Any]], + conversation_history: Any, + pending_verification_response: Any, + pending_verification_response_previewed: Any, +) -> StopGateVerdict: + """Run verify-on-stop → pre_verify hook → kanban stop guard, in that order. Nudges + are user-role rows appended only after the assistant answer row, so role alternation + holds. Hook lookups are imported lazily from their origin modules (tests patch them + there).""" + _pending_verification_response = pending_verification_response + _pending_verification_response_previewed = pending_verification_response_previewed + + def _verdict(continue_turn: bool) -> StopGateVerdict: + return StopGateVerdict( + continue_turn=continue_turn, + final_response=None if continue_turn else final_response, + pending_verification_response=_pending_verification_response, + pending_verification_response_previewed=_pending_verification_response_previewed, + ) + + try: + from agent.verification_stop import ( + build_verify_on_stop_nudge, + verify_on_stop_enabled, + ) + + if verify_on_stop_enabled(): + _verify_nudge = build_verify_on_stop_nudge( + session_id=getattr(agent, "session_id", None), + changed_paths=getattr(agent, "_turn_file_mutation_paths", set()), + attempts=getattr(agent, "_verification_stop_nudges", 0), + ) + else: + _verify_nudge = None + except Exception: + logger.debug("verification stop-loop check failed", exc_info=True) + _verify_nudge = None + + if _verify_nudge: + agent._verification_stop_nudges = ( + getattr(agent, "_verification_stop_nudges", 0) + 1 + ) + final_msg["finish_reason"] = "verification_required" + # Real content: persist and emit as interim so the user sees the + # attempted answer; only the nudge is flagged synthetic. (#65919) + agent._emit_interim_assistant_message(final_msg) + append_message(messages, final_msg) + try: + agent._flush_messages_to_session_db(messages, conversation_history) + except Exception: + logger.debug("verify-on-stop interim flush failed", exc_info=True) + append_message(messages, { + "role": "user", + "content": _verify_nudge, + "_verification_stop_synthetic": True, + }) + agent._session_messages = messages + # Internal nudge: stay silent on the terminal, debug-log only. + logger.debug("verification stop-loop nudge issued (attempt %d)", + agent._verification_stop_nudges) + # Keep the answer only as a budget-exhaustion fallback; clear + # ``final_response`` so the finalizer can tell this gate from error + # exits. Mark previewed only if the candidate is reused. (#61631) + _pending_verification_response = final_response + _pending_verification_response_previewed = ( + agent._interim_content_was_streamed(final_response or "") + ) + return _verdict(True) + + # pre_verify hook gate: after code edits a registered hook may keep the + # agent going one more turn; no default continuation cost. + _verify_nudge2 = None + _edited = sorted(getattr(agent, "_turn_file_mutation_paths", set()) or []) + _attempt = getattr(agent, "_pre_verify_nudges", 0) + try: + from agent.verify_hooks import max_verify_nudges + from hermes_cli.lifecycle import has_hook + from hermes_cli.plugins import get_pre_verify_continue_message + + if _edited and has_hook("pre_verify") and _attempt < max_verify_nudges(): + # Posture is fixed for the session — resolve once + cache. + coding = getattr(agent, "_resolved_is_coding", None) + if coding is None: + from agent.coding_context import is_coding_context + coding = bool(is_coding_context(platform=getattr(agent, "platform", "") or "")) + agent._resolved_is_coding = coding + _verify_nudge2 = get_pre_verify_continue_message( + session_id=getattr(agent, "session_id", None) or "", + platform=getattr(agent, "platform", "") or "", + model=getattr(agent, "model", "") or "", + coding=coding, + attempt=_attempt, + final_response=final_response, + changed_paths=_edited, + ) + except Exception: + logger.debug("pre_verify hook check failed", exc_info=True) + _verify_nudge2 = None + + if _verify_nudge2: + agent._pre_verify_nudges = _attempt + 1 + final_msg["finish_reason"] = "verify_hook_continue" + # Real content: persist and emit as interim so the user sees the + # attempted answer; only the nudge is flagged synthetic. (#65919) + agent._emit_interim_assistant_message(final_msg) + append_message(messages, final_msg) + try: + agent._flush_messages_to_session_db(messages, conversation_history) + except Exception: + logger.debug("pre_verify interim flush failed", exc_info=True) + append_message(messages, { + "role": "user", + "content": _verify_nudge2, + "_pre_verify_synthetic": True, + }) + agent._session_messages = messages + logger.debug("pre_verify nudge issued (attempt %d)", + agent._pre_verify_nudges) + _pending_verification_response = final_response + _pending_verification_response_previewed = ( + agent._interim_content_was_streamed(final_response or "") + ) + return _verdict(True) + + # ── Kanban worker terminal-tool stop guard ───────────── + # Workers must end with kanban_complete / kanban_block; a narrated stop + # is recorded as protocol_violation, so nudge once or twice first. + try: + from agent.kanban_stop import build_kanban_stop_nudge + + _kanban_nudge = build_kanban_stop_nudge( + messages=messages, + attempts=getattr(agent, "_kanban_stop_nudges", 0), + ) + except Exception: + logger.debug("kanban stop-loop check failed", exc_info=True) + _kanban_nudge = None + + if _kanban_nudge: + agent._kanban_stop_nudges = ( + getattr(agent, "_kanban_stop_nudges", 0) + 1 + ) + final_msg["finish_reason"] = "kanban_terminal_required" + final_msg["_kanban_stop_synthetic"] = True + append_message(messages, final_msg) + append_message(messages, { + "role": "user", + "content": _kanban_nudge, + "_kanban_stop_synthetic": True, + }) + agent._session_messages = messages + logger.info( + "kanban stop-loop nudge issued (attempt %d) task=%s", + agent._kanban_stop_nudges, + os.environ.get("HERMES_KANBAN_TASK", ""), + ) + agent._emit_status( + "⚠️ Kanban worker tried to exit without " + "kanban_complete/kanban_block — nudging to finish" + ) + # Same finalizer contract as verify-on-stop: clear final_response so + # budget exhaustion doesn't treat the narrated stop as an answer. + _pending_verification_response = final_response + _pending_verification_response_previewed = ( + agent._interim_content_was_streamed(final_response or "") + ) + return _verdict(True) + return _verdict(False) diff --git a/agent/turn_tool_validation.py b/agent/turn_tool_validation.py new file mode 100644 index 0000000000..4d4a5aeaac --- /dev/null +++ b/agent/turn_tool_validation.py @@ -0,0 +1,248 @@ +"""Tool-call validation for the conversation turn loop: unknown tool names (with +auto-repair and the 3-strike partial exit) and malformed JSON arguments (retry, then +recovery tool results). + +Extracted from ``run_conversation``. Role alternation is preserved on every path: an +invalid batch is answered with tool-role error results (never a user message), and +the exits close any open tool-result tail (#48879). Nothing here imports +``agent.conversation_loop`` at module level (cycle); loop-internal helpers resolve lazily. +""" + +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass +from typing import Any, Dict, List, Optional + +from agent.message_metadata import append_message +from agent.message_sanitization import close_interrupted_tool_sequence, coalesce_tool_call_id + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class ToolValidationVerdict: + """Outcome of ``validate_tool_calls``. + + ``action``: ``"ok"`` (dispatch the calls), ``"continue"`` (re-issue the API call — + error results / retry state were recorded) or ``"return"`` (terminal partial + result in ``result``). ``mixed_invalid_batch`` is True when the batch contains BOTH + valid and unknown tool names: only the invalid calls get error results, the valid + ones run.""" + + action: str + result: Optional[Dict[str, Any]] + mixed_invalid_batch: bool + + +def validate_tool_calls( + agent: Any, + assistant_message: Any, + finish_reason: str, + *, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_call_count: int, + effective_task_id: Any, +) -> ToolValidationVerdict: + """Validate ``assistant_message.tool_calls`` in place (ids uniquified, names + repaired, dict/empty args normalized to JSON strings). Strikes for invalid names + advance only when a turn has NO valid call, so a degenerate model still halts at + 3; args cut off mid-stream (routers rewrite ``length`` → ``tool_calls``) are refused + outright rather than retried.""" + from agent.conversation_loop import _invalid_tool_name_error_content + + _mixed_invalid_batch = False + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> ToolValidationVerdict: + return ToolValidationVerdict(action=action, result=result, mixed_invalid_batch=_mixed_invalid_batch) + + # Uniquify duplicate tool-call ids BEFORE any downstream consumer: the + # pre-API sanitizer keeps only the first call/result per id. See + # _uniquify_tool_call_ids. + agent._uniquify_tool_call_ids(assistant_message.tool_calls) + + # Validate tool call names - detect model hallucinations + # Repair mismatched tool names before validating + for tc in assistant_message.tool_calls: + if tc.function.name not in agent.valid_tool_names: + repaired = agent._repair_tool_call(tc.function.name) + if repaired: + print(f"{agent.log_prefix}🔧 Auto-repaired tool name: '{tc.function.name}' -> '{repaired}'") + tc.function.name = repaired + invalid_tool_calls = [ + tc.function.name for tc in assistant_message.tool_calls + if tc.function.name not in agent.valid_tool_names + ] + # Mixed batch: error-result ONLY the invalid calls and run the valid + # ones; voiding the turn discards real work. Strikes advance only when a + # turn has NO valid call, so a degenerate model still halts at 3. + _mixed_invalid_batch = bool(invalid_tool_calls) and any( + tc.function.name in agent.valid_tool_names + for tc in assistant_message.tool_calls + ) + if _mixed_invalid_batch: + agent._invalid_tool_retries = 0 + invalid_name = invalid_tool_calls[0] + invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name + _n_valid = sum( + 1 for tc in assistant_message.tool_calls + if tc.function.name in agent.valid_tool_names + ) + agent._buffer_vprint( + f"⚠️ Unknown tool '{invalid_preview}' in batch — erroring that call, " + f"executing {_n_valid} valid call(s)" + ) + elif invalid_tool_calls: + # Track retries for invalid tool calls + agent._invalid_tool_retries += 1 + + # Return helpful error to model — model can agent-correct next turn + invalid_name = invalid_tool_calls[0] + invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name + agent._buffer_vprint(f"⚠️ Unknown tool '{invalid_preview}' — sending error to model for agent-correction ({agent._invalid_tool_retries}/3)") + + if agent._invalid_tool_retries >= 3: + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ Max retries (3) for invalid tool calls exceeded. Stopping as partial.", force=True) + agent._invalid_tool_retries = 0 + _final_response = f"Model generated invalid tool call: {invalid_preview}" + # Prior retries or an earlier tool batch leave a tool-result + # tail; close it as interrupt aborts do so the next turn is not + # tool→user. (#48879) + close_interrupted_tool_sequence(messages, _final_response) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": _final_response + }) + + assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) + append_message(messages, assistant_msg) + for tc in assistant_message.tool_calls: + _tc_name = tc.function.name + if _tc_name not in agent.valid_tool_names: + # See _invalid_tool_name_error_content for the + # blank-name anti-priming rationale (#47967). + content = _invalid_tool_name_error_content( + _tc_name, agent.valid_tool_names + ) + else: + content = "Skipped: another tool call in this turn used an invalid name. Please retry this tool call." + append_message(messages, { + "role": "tool", + "name": tc.function.name, + "tool_call_id": coalesce_tool_call_id(tc), + "content": content, + }) + return _verdict("continue") + # Reset retry counter on successful tool call validation + agent._invalid_tool_retries = 0 + + # Validate tool call arguments are valid JSON + # Handle empty strings as empty objects (common model quirk) + invalid_json_args = [] + for tc in assistant_message.tool_calls: + args = tc.function.arguments + if isinstance(args, (dict, list)): + tc.function.arguments = json.dumps(args) + continue + if args is not None and not isinstance(args, str): + tc.function.arguments = str(args) + args = tc.function.arguments + # Treat empty/whitespace strings as empty object + if not args or not args.strip(): + tc.function.arguments = "{}" + continue + try: + json.loads(args) + except json.JSONDecodeError as e: + if ( + _mixed_invalid_batch + and tc.function.name not in agent.valid_tool_names + ): + # This call never executes (invalid-name error result + # below); don't let its broken args trigger the whole-turn + # JSON retry. + continue + invalid_json_args.append((tc.function.name, str(e))) + + if invalid_json_args: + # Routers may rewrite finish_reason "length" → "tool_calls", hiding + # truncation; args not ending in } or ] (stripped) were cut off + # mid-stream. + _truncated = any( + not (tc.function.arguments or "").rstrip().endswith(("}", "]")) + for tc in assistant_message.tool_calls + if tc.function.name in {n for n, _ in invalid_json_args} + ) + if _truncated: + agent._vprint( + f"{agent.log_prefix}⚠️ Truncated tool call arguments detected " + f"(finish_reason={finish_reason!r}) — refusing to execute.", + force=True, + ) + agent._invalid_json_retries = 0 + agent._cleanup_task_resources(effective_task_id) + _final_response = "Response truncated due to output length limit" + # Same tool-tail close as interrupt / invalid-tool + # exhaustion — this path never reaches finalize_turn. + close_interrupted_tool_sequence(messages, _final_response) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": _final_response, + }) + + # Track retries for invalid JSON arguments + agent._invalid_json_retries += 1 + + tool_name, error_msg = invalid_json_args[0] + agent._buffer_vprint(f"⚠️ Invalid JSON in tool call arguments for '{tool_name}': {error_msg}") + + if agent._invalid_json_retries < 3: + agent._buffer_vprint(f"🔄 Retrying API call ({agent._invalid_json_retries}/3)...") + # Don't add anything to messages, just retry the API call + return _verdict("continue") + else: + # Instead of returning partial, inject tool error results so the model can recover. + # Using tool results (not user messages) preserves role alternation. + agent._buffer_vprint("⚠️ Injecting recovery tool results for invalid JSON...") + agent._invalid_json_retries = 0 # Reset for next attempt + + # Append the assistant message with its (broken) tool_calls + recovery_assistant = agent._build_assistant_message(assistant_message, finish_reason) + append_message(messages, recovery_assistant) + + # Respond with tool error results for each tool call + invalid_names = {name for name, _ in invalid_json_args} + for tc in assistant_message.tool_calls: + if tc.function.name in invalid_names: + err = next(e for n, e in invalid_json_args if n == tc.function.name) + tool_result = ( + f"Error: Invalid JSON arguments. {err}. " + f"For tools with no required parameters, use an empty object: {{}}. " + f"Please retry with valid JSON." + ) + else: + tool_result = "Skipped: other tool call in this response had invalid JSON." + append_message(messages, { + "role": "tool", + "name": tc.function.name, + "tool_call_id": coalesce_tool_call_id(tc), + "content": tool_result, + }) + return _verdict("continue") + + # Reset retry counter on successful JSON validation + agent._invalid_json_retries = 0 + return _verdict("ok") diff --git a/agent/turn_truncation.py b/agent/turn_truncation.py new file mode 100644 index 0000000000..0f6d58cea0 --- /dev/null +++ b/agent/turn_truncation.py @@ -0,0 +1,748 @@ +"""Truncation recovery (``finish_reason == "length"``) for the conversation turn loop. + +Extracted from ``run_conversation``. Handles thinking-budget exhaustion, repetition- +dominated truncation (#86581), content-filter stream stalls escalated to the fallback +chain (#32421), text continuation nudges (up to 4, with the ceiling exit that drops the +fragment trail), truncated tool-call retries with max_tokens boosts, and the final +roll-back. Nothing here imports ``agent.conversation_loop`` at module level (cycle); +loop-internal helpers are imported lazily so tests patching them on the loop keep working. +""" + +from __future__ import annotations + +import logging +import re +from dataclasses import dataclass +from typing import Any, Dict, List, Optional + +from agent.error_classifier import FailoverReason +from agent.message_metadata import append_message +from agent.message_sanitization import close_interrupted_tool_sequence +from agent.repetition_guard import is_repetition_dominated +from agent.turn_retry_state import TurnRetryState +from hermes_constants import PARTIAL_STREAM_STUB_ID + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class TruncationVerdict: + """Outcome of ``recover_from_truncation``. + + ``action``: ``"return"`` (end the turn with ``result``), ``"break"`` (a + ``_retry.restart_with_*`` flag is set — restart the API call), ``"continue"`` + (re-issue the same call immediately) or ``"fallthrough"`` (unreachable in practice: + every path exits, kept for the contract). The remaining fields are the loop locals + the handler may have rebound.""" + + action: str + result: Optional[Dict[str, Any]] + messages: List[Dict[str, Any]] + length_continue_retries: int + truncated_response_parts: List[str] + truncated_tool_call_retries: int + retry_count: int + compression_attempts: int + + +def recover_from_truncation( + agent: Any, + response: Any, + finish_reason: str, + _retry: TurnRetryState, + *, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_kwargs: Any, + api_call_count: int, + effective_task_id: Any, + current_turn_user_idx: Any, + length_continue_retries: int, + truncated_response_parts: List[str], + truncated_tool_call_retries: int, + retry_count: int, + compression_attempts: int, +) -> TruncationVerdict: + """Recover from a truncated response. Order is load-bearing: thinking exhaustion and + repetition abort BEFORE any continuation; a content-filter stall escalates to the + fallback chain BEFORE the primary is retried; text continuation (no tool calls) then + truncated tool-call retry; finally roll back to the last complete assistant turn. + Never appends an interim assistant row with NO visible content (strict providers + reject it with 400) — only the continuation nudge.""" + from agent.conversation_loop import _get_continuation_prompt, _join_truncated_parts + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> TruncationVerdict: + return TruncationVerdict( + action=action, + result=result, + messages=messages, + length_continue_retries=length_continue_retries, + truncated_response_parts=truncated_response_parts, + truncated_tool_call_retries=truncated_tool_call_retries, + retry_count=retry_count, + compression_attempts=compression_attempts, + ) + + if getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID: + agent._vprint( + f"{agent.log_prefix}⚠️ Response truncated — stream " + f"ended before completion", + force=True, + ) + else: + agent._vprint( + f"{agent.log_prefix}⚠️ Response truncated " + f"(finish_reason='length') - model hit max output tokens", + force=True, + ) + + # Normalize to one OpenAI-style message so continuation and tool- + # call retry work across transports (Anthropic reuses the loop's + # adapter). + _trunc_msg = None + _trunc_transport = agent._get_transport() + if agent.api_mode == "anthropic_messages": + _trunc_result = _trunc_transport.normalize_response( + response, strip_tool_prefix=agent._is_anthropic_oauth + ) + else: + _trunc_result = _trunc_transport.normalize_response(response) + _trunc_msg = _trunc_result + + _trunc_content = getattr(_trunc_msg, "content", None) if _trunc_msg else None + _trunc_has_tool_calls = bool(getattr(_trunc_msg, "tool_calls", None)) if _trunc_msg else False + + # ── Detect thinking-budget exhaustion ────────────── + # Only when reasoning blocks exist with no visible text after them; + # content=None from non- models is normal truncation. + _has_think_tags = bool( + _trunc_content and re.search( + r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', + _trunc_content, + re.IGNORECASE, + ) + ) + _thinking_exhausted = ( + not _trunc_has_tool_calls + and _has_think_tags + and ( + (_trunc_content is not None and not agent._has_content_after_think_block(_trunc_content)) + or _trunc_content is None + ) + ) + + if _thinking_exhausted: + _exhaust_error = ( + "Model used all output tokens on reasoning with none left " + "for the response. Try lowering reasoning effort or " + "increasing max_tokens." + ) + agent._vprint( + f"{agent.log_prefix}💭 Reasoning exhausted the output token budget — " + f"no visible response was produced.", + force=True, + ) + # Return a user-friendly message as the response so CLI and + # gateway display it. + _exhaust_response = ( + "⚠️ **Thinking Budget Exhausted**\n\n" + "The model used all its output tokens on reasoning " + "and had none left for the actual response.\n\n" + "To fix this:\n" + "→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n" + "→ Or switch to a larger/non-reasoning model with `/model`" + ) + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _exhaust_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": _exhaust_error, + }) + + # ── Detect repetition-dominated truncation (#86581) ── + # A repetition loop can burn the whole budget on one fragment; abort + # like _thinking_exhausted (reasoning stripped first). + _visible_trunc = ( + agent._strip_think_blocks(_trunc_content) + if isinstance(_trunc_content, str) + else _trunc_content + ) + _repetition_dominated = ( + not _trunc_has_tool_calls + and bool(_visible_trunc) + and is_repetition_dominated(_visible_trunc) + ) + if _repetition_dominated: + _rep_error = ( + "Model output entered a repetition loop and was " + "truncated mid-loop; refusing to continue a " + "degenerate response." + ) + agent._vprint( + f"{agent.log_prefix}🔁 Response dominated by " + f"repeated text — stopping instead of " + f"continuing a degenerate response.", + force=True, + ) + _rep_response = ( + "⚠️ **Response Stopped — Repetition Detected**\n\n" + "The model fell into a repetition loop while " + "writing this response, so continuing would only " + "produce more repeated text. The partial response " + "was discarded.\n\n" + "→ Switch to a different model with `/model`\n" + "→ Or resend your message (your conversation " + "history is preserved)" + ) + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _rep_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": _rep_error, + }) + + if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}: + assistant_message = _trunc_msg + # ── Content-filter stream stall → fallback (#32421) ── + # ``_content_filter_terminated`` is content-deterministic; + # escalate to the fallback before retrying the primary. + _cf_terminated = getattr( + response, "_content_filter_terminated", False + ) + if ( + _cf_terminated + and agent._fallback_index < len(agent._fallback_chain) + ): + agent._vprint( + f"{agent.log_prefix}🛡️ Content filter terminated " + f"stream — activating fallback provider...", + force=True, + ) + agent._emit_status( + "Content filter terminated stream; switching to fallback..." + ) + if agent._try_activate_fallback(): + # Roll partial content back to the last clean turn so + # the fallback gets a coherent continuation point. + if truncated_response_parts: + messages = agent._get_messages_up_to_last_assistant(messages) + # Unmark survivors: their text left the stitched partial. + for _frag in messages: + if isinstance(_frag, dict): + _frag.pop("_length_continuation_fragment", None) + _frag.pop("_length_continuation_nudge", None) + agent._session_messages = messages + length_continue_retries = 0 + truncated_response_parts = [] + retry_count = 0 + compression_attempts = 0 + _retry.primary_recovery_attempted = False + _retry.restart_with_rebuilt_messages = True + return _verdict("break") + # No fallback available — fall through to normal + # continuation (best-effort, may loop). + agent._vprint( + f"{agent.log_prefix}⚠️ No fallback provider " + f"configured — retrying with same provider " + f"(may re-hit filter)...", + force=True, + ) + if assistant_message is not None and not _trunc_has_tool_calls: + length_continue_retries += 1 + # Never append an interim assistant message with NO visible + # content: strict providers reject it (HTTP 400), poisoning + # history. Append only the nudge. + _interim_content = getattr(assistant_message, "content", None) + _is_empty_partial_stub = ( + getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID + and not _interim_content + ) + if not _interim_content and not _is_empty_partial_stub: + # Thinking-only truncation: continuing with thinking ON + # re-burns the budget, so drop thinking for one request. + agent._ephemeral_reasoning_off = True + if _interim_content: + interim_msg = agent._build_assistant_message(assistant_message, finish_reason) + # Marked so the ceiling exit can drop the fragment trail. + interim_msg["_length_continuation_fragment"] = True + append_message(messages, interim_msg) + truncated_response_parts.append(_interim_content) + + if length_continue_retries < 4: + _is_partial_stream_stub = ( + getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID + ) + _dropped_tools = getattr( + response, "_dropped_tool_names", None + ) + + if _is_partial_stream_stub and _dropped_tools: + _tool_list = ", ".join(_dropped_tools[:3]) + agent._vprint( + f"{agent.log_prefix}↻ Stream interrupted mid " + f"tool-call ({_tool_list}) — requesting " + f"chunked retry " + f"({length_continue_retries}/4)..." + ) + elif _is_partial_stream_stub: + agent._vprint( + f"{agent.log_prefix}↻ Stream interrupted — " + f"requesting continuation " + f"({length_continue_retries}/4)..." + ) + else: + agent._vprint( + f"{agent.log_prefix}↻ Requesting continuation " + f"({length_continue_retries}/4)..." + ) + + _continue_content = _get_continuation_prompt( + _is_partial_stream_stub, _dropped_tools + ) + continue_msg = { + "role": "user", + "content": _continue_content, + "_length_continuation_nudge": True, + } + append_message(messages, continue_msg) + agent._session_messages = messages + _retry.restart_with_length_continuation = True + return _verdict("break") + + partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip() + # The one-shot reasoning-off override must not leak into the + # next turn when the ceiling exit skips the consuming call. + agent._ephemeral_reasoning_off = False + if partial_response: + agent._vprint( + f"{agent.log_prefix}⚠️ Response still truncated " + f"after {length_continue_retries} continuation attempts — keeping the " + f"partial response received so far.", + force=True, + ) + _ceiling_final = partial_response + else: + # Every fragment was empty (e.g. reasoning-only model): + # return an actionable message, not a bare None. + agent._vprint( + f"{agent.log_prefix}⚠️ Response still truncated " + f"after {length_continue_retries} continuation attempts — no visible " + f"text was produced.", + force=True, + ) + _ceiling_final = ( + "⚠️ **No visible answer was produced.** The " + "model hit its output-token limit on every " + "continuation attempt — its reasoning " + "consumed the entire budget each time.\n\n" + "To fix this:\n" + "→ Lower reasoning effort: `/reasoning low` " + "or `/reasoning none`\n" + "→ Or raise max_tokens for this model" + ) + # Unanswered continue nudges made every later turn re-truncate. + _turn_start = ( + current_turn_user_idx + 1 + if isinstance(current_turn_user_idx, int) + and current_turn_user_idx >= 0 + else 0 + ) + messages[_turn_start:] = [ + m for m in messages[_turn_start:] + if not ( + isinstance(m, dict) + and ( + m.get("_length_continuation_fragment") + or m.get("_length_continuation_nudge") + ) + ) + ] + if partial_response: + append_message(messages, { + "role": "assistant", + "content": partial_response, + "finish_reason": "length", + }) + agent._session_messages = messages + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _ceiling_final, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": "Response remained truncated after 4 continuation attempts", + }) + + if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}: + assistant_message = _trunc_msg + if assistant_message is not None and _trunc_has_tool_calls: + _is_stub_stall = ( + getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID + ) + if truncated_tool_call_retries < 4: + truncated_tool_call_retries += 1 + if _is_stub_stall: + # Stream broke mid tool-call (network), not a real + # output cap — say so. + agent._buffer_vprint( + f"⚠️ Stream interrupted mid tool-call — " + f"retrying ({truncated_tool_call_retries}/4)..." + ) + else: + agent._buffer_vprint( + f"⚠️ Truncated tool call detected — " + f"retrying API call " + f"({truncated_tool_call_retries}/4)..." + ) + # Boost max_tokens per retry: a real output-cap + # truncation needs it; harmless for a stall. + _tc_boost_base = agent.max_tokens if agent.max_tokens else 4096 + _tc_boost = _tc_boost_base * (2 ** truncated_tool_call_retries) + _tc_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs) + if _tc_requested_cap is not None: + _tc_boost = max(_tc_boost, _tc_requested_cap) + _tc_boost_cap = max(32768, _tc_requested_cap or 0) + agent._ephemeral_max_output_tokens = min(_tc_boost, _tc_boost_cap) + # Don't append the broken response; re-run the same call + # from current state. + return _verdict("continue") + agent._flush_status_buffer() + if _is_stub_stall: + agent._vprint( + f"{agent.log_prefix}⚠️ Stream kept dropping mid tool-call after 4 retries — the action was not executed.", + force=True, + ) + else: + agent._vprint( + f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.", + force=True, + ) + agent._cleanup_task_resources(effective_task_id) + _final_response = ( + "Stream repeatedly dropped mid tool-call (network); " + "the tool was not executed" + if _is_stub_stall + else "Response truncated due to output length limit" + ) + # Prior tool batches can leave a tool-result tail; this path + # never reaches finalize_turn (#48879). + close_interrupted_tool_sequence(messages, _final_response) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": _final_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": _final_response, + }) + + # If we have prior messages, roll back to last complete state + if len(messages) > 1: + agent._vprint(f"{agent.log_prefix} ⏪ Rolling back to last complete assistant turn") + rolled_back_messages = agent._get_messages_up_to_last_assistant(messages) + + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + + return _verdict("return", { + "final_response": "Response truncated due to output length limit", + "messages": rolled_back_messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": "Response truncated due to output length limit" + }) + else: + # First message was truncated - mark as failed + agent._flush_status_buffer() + agent._vprint(f"{agent.log_prefix}❌ First response truncated - cannot recover", force=True) + agent._persist_session(messages, conversation_history) + return _verdict("return", { + "final_response": "First response truncated due to output length limit", + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "failed": True, + "error": "First response truncated due to output length limit" + }) + return _verdict("fallthrough") + + +def continue_codex_incomplete( + agent: Any, + assistant_message: Any, + finish_reason: str, + *, + messages: List[Dict[str, Any]], + conversation_history: Any, + api_call_count: int, +) -> Optional[Dict[str, Any]]: + """Codex Responses ``status=incomplete`` continuation (max 3 per turn). + + Appends the interim assistant message (deduped on visible content only — opaque + provider state drifts per continuation, #52711; ``codex_reasoning_items`` are merged, + not overwritten, because the earlier response holds the only native-compaction + checkpoint) and, when a bare retry would be byte-identical, a user-role nudge — only + after an assistant row, to preserve role alternation. Returns ``None`` to continue + the turn loop, or the terminal ``partial`` result once retries are exhausted.""" + from agent.conversation_loop import _CODEX_INCOMPLETE_NUDGE + + agent._codex_incomplete_retries += 1 + + interim_msg = agent._build_assistant_message(assistant_message, finish_reason) + interim_has_content = bool((interim_msg.get("content") or "").strip()) + interim_has_reasoning = bool(interim_msg.get("reasoning", "").strip()) if isinstance(interim_msg.get("reasoning"), str) else False + interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items")) + interim_has_codex_message_items = bool(interim_msg.get("codex_message_items")) + + if ( + interim_has_content + or interim_has_reasoning + or interim_has_codex_reasoning + or interim_has_codex_message_items + ): + last_msg = messages[-1] if messages else None + # Dedup on visible content only (content + reasoning): opaque + # provider state drifts per continuation and would defeat dedup + # (#52711). + last_interim_visible = ( + agent._interim_assistant_visible_text(last_msg) + if isinstance(last_msg, dict) + else "" + ) + current_interim_visible = agent._interim_assistant_visible_text(interim_msg) + if last_interim_visible or current_interim_visible: + same_visible_output = last_interim_visible == current_interim_visible + else: + # Preserve the existing reasoning-only behavior when + # neither response has text eligible for interim delivery. + same_visible_output = ( + (last_msg.get("content") or "") == (interim_msg.get("content") or "") + and (last_msg.get("reasoning") or "") == (interim_msg.get("reasoning") or "") + ) if isinstance(last_msg, dict) else False + visible_duplicate = ( + isinstance(last_msg, dict) + and last_msg.get("role") == "assistant" + and last_msg.get("finish_reason") == "incomplete" + and same_visible_output + ) + if visible_duplicate: + # Update replay state in-place: keep the latest provider payload + # without re-emitting identical user-visible commentary. + for _key in ( + "content", + "reasoning", + "reasoning_content", + "reasoning_details", + "codex_reasoning_items", + "codex_message_items", + ): + if _key in interim_msg: + if _key == "codex_reasoning_items": + # Merge, don't overwrite: the earlier response's + # native compaction checkpoint is the only copy. See + # merge_interim_reasoning_items. + from agent.native_compaction import ( + merge_interim_reasoning_items, + ) + last_msg[_key] = merge_interim_reasoning_items( + last_msg.get(_key), interim_msg[_key] + ) + else: + last_msg[_key] = interim_msg[_key] + else: + append_message(messages, interim_msg) + agent._emit_interim_assistant_message(interim_msg) + + if agent._codex_incomplete_retries < 3: + # If the interim has nothing the Responses converter will replay, a + # bare retry is byte-identical and fails identically; append a + # user-role nudge so the retry differs and asks for the answer. + interim_replayable = ( + interim_has_content + or interim_has_codex_reasoning + or interim_has_codex_message_items + ) + # Replayable ≠ different: an interim holding only a ``compaction`` + # checkpoint in ``codex_reasoning_items`` is replayable yet re-sends + # identically. One bare retry, then always nudge. + if not interim_replayable or agent._codex_incomplete_retries >= 2: + _last_msg = messages[-1] if messages else None + _already_nudged = ( + isinstance(_last_msg, dict) + and _last_msg.get("role") == "user" + and _last_msg.get("content") == _CODEX_INCOMPLETE_NUDGE + ) + # Alternation guard: the user-role nudge may only follow an + # assistant message; after a too-empty interim it would create + # user→user / tool→user. + _last_is_assistant = ( + isinstance(_last_msg, dict) + and _last_msg.get("role") == "assistant" + ) + if not _already_nudged and _last_is_assistant: + append_message(messages, { + "role": "user", + "content": _CODEX_INCOMPLETE_NUDGE, + }) + if not agent.quiet_mode: + agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({agent._codex_incomplete_retries}/3)") + # Show the continuation on the spinner/status line and gateway + # heartbeat; these retries can take minutes and otherwise look like + # infinite thinking (#64434). + agent._emit_wait_notice( + f"↻ model returned reasoning with no final answer — " + f"asking it to continue " + f"({agent._codex_incomplete_retries}/3)" + ) + agent._session_messages = messages + return None + + agent._codex_incomplete_retries = 0 + agent._persist_session(messages, conversation_history) + return { + "final_response": "Codex response remained incomplete after 3 continuation attempts", + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "partial": True, + "error": "Codex response remained incomplete after 3 continuation attempts", + } + + +@dataclass +class RefusalVerdict: + """Outcome of ``handle_content_policy_refusal``: ``"break"`` (fallback activated — + restart armed on ``_retry``; caller resets retry/compression counters) or + ``"return"`` (the typed content-policy result in ``result``). ``active_system_prompt`` + is the possibly re-synced system prompt.""" + + action: str + result: Optional[Dict[str, Any]] + active_system_prompt: Any + + +def handle_content_policy_refusal( + agent: Any, + response: Any, + _retry: TurnRetryState, + *, + thinking_spinner: Any, + messages: List[Dict[str, Any]], + api_messages: Any, + api_kwargs: Any, + active_system_prompt: Any, + conversation_history: Any, + api_call_count: int, + effective_task_id: Any, + turn_id: Any, + api_request_id: Any, + api_start_time: float, + retry_count: int, + max_retries: int, +) -> RefusalVerdict: + """HTTP-200 refusal (``finish_reason`` ``content_filter`` / ``guardrail_intervened``). + Deterministic for the unchanged prompt — never retried: one configured-fallback try, + else surface the refusal (explanation may live only in the reasoning channel). The + caller stops its spinner reference; this stops the spinner object.""" + from agent.conversation_loop import ( + _CONTENT_POLICY_RECOVERY_HINT, + _arm_fallback_restart, + _content_policy_blocked_result, + ) + + def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> RefusalVerdict: + return RefusalVerdict(action=action, result=result, active_system_prompt=active_system_prompt) + + _refusal_transport = agent._get_transport() + if agent.api_mode == "anthropic_messages": + _refusal_result = _refusal_transport.normalize_response( + response, strip_tool_prefix=agent._is_anthropic_oauth + ) + else: + _refusal_result = _refusal_transport.normalize_response(response) + _refusal_text = (getattr(_refusal_result, "content", None) or "").strip() + # Some refusals carry the explanation only in the reasoning + # channel; fall back to it so the user sees *something*. + if not _refusal_text: + _refusal_text = (agent._extract_reasoning(_refusal_result) or "").strip() + + agent._invoke_api_request_error_hook( + task_id=effective_task_id, + turn_id=turn_id, + api_request_id=api_request_id, + api_call_count=api_call_count, + api_start_time=api_start_time, + api_kwargs=api_kwargs, + error_type="ContentPolicyBlocked", + error_message=_refusal_text or "model declined to respond (content_filter)", + status_code=None, + retry_count=retry_count, + max_retries=max_retries, + retryable=False, + reason=FailoverReason.content_policy_blocked.value, + ) + + if thinking_spinner: + thinking_spinner.stop("") + if agent.thinking_callback: + agent.thinking_callback("") + + # Deterministic for the unchanged prompt — never retry. Try a + # configured fallback once; otherwise surface the refusal. + if agent._has_pending_fallback(): + agent._buffer_status( + "⚠️ Model declined to respond (safety refusal) — trying fallback..." + ) + if agent._try_activate_fallback(): + active_system_prompt = _arm_fallback_restart( + agent, api_messages, active_system_prompt, _retry) + return _verdict("break") + + agent._flush_status_buffer() + _refusal_log = ( + _refusal_text[:500] + "..." + if len(_refusal_text) > 500 + else _refusal_text + ) + logger.warning( + "%sModel declined to respond (finish_reason=content_filter). " + "model=%s provider=%s refusal=%s", + agent.log_prefix, agent.model, agent.provider, + _refusal_log or "(no text)", + ) + agent._emit_status( + "⚠️ The model declined to respond to this request (safety refusal)." + ) + + _refusal_detail = ( + f"Model's explanation: {_refusal_text}" + if _refusal_text + else "The model returned no explanation." + ) + _refusal_response = ( + "⚠️ The model declined to respond to this request " + "(safety refusal — not a Hermes/gateway failure).\n\n" + f"{_refusal_detail}\n\n" + f"{_CONTENT_POLICY_RECOVERY_HINT}" + ) + + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + return _verdict("return", _content_policy_blocked_result( + messages, + api_call_count, + final_response=_refusal_response, + error_detail=_refusal_text or "model declined (content_filter)", + )) diff --git a/agent/turn_usage.py b/agent/turn_usage.py new file mode 100644 index 0000000000..f2970e4c41 --- /dev/null +++ b/agent/turn_usage.py @@ -0,0 +1,312 @@ +"""Per-response usage accounting for the conversation turn loop. + +After every successful model API call, ``run_conversation`` folds the provider's +``response.usage`` into: the context compressor (``update_from_response`` + the +compression-budget rearm latch), the usage anchor for display/compression math, +per-session token/cost counters, the state.db token-delta queue, and the +observability log line. MoA sessions additionally fold advisor fan-out usage into +the reported counts and price the aggregator at its REAL model/provider. + +``record_response_usage`` owns that block. It mutates ``agent`` exactly as the inline +code did and returns the loop-visible verdict (compression budget counter, and +whether a provider-confirmed recovery rearmed it) as ``ResponseUsageOutcome``. + +Logger name stays ``agent.conversation_loop`` for caplog/log-routing parity. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any, Dict, List + +from agent.model_metadata import capture_usage_anchor +from agent.usage_pricing import estimate_usage_cost, normalize_usage + +logger = logging.getLogger("agent.conversation_loop") + + +@dataclass +class ResponseUsageOutcome: + """What the loop reads back after usage accounting. + + ``compression_attempts`` is the (possibly rearmed-to-zero) budget counter; + ``rearmed`` tells the loop to also clear its preflight-block latch.""" + + compression_attempts: int + rearmed: bool = False + + +def _loop_mod(): + """Lazy ``agent.conversation_loop`` so tests patching + ``agent.conversation_loop.save_context_length`` still intercept, and so this + module never imports the loop at load time (cycle).""" + import agent.conversation_loop as _cl + + return _cl + + +def record_response_usage( + agent: Any, + response: Any, + *, + messages: List[Dict[str, Any]], + api_call_count: int, + api_duration: float, + compression_attempts: int, + max_compression_attempts: int, +) -> ResponseUsageOutcome: + """Fold ``response.usage`` into compressor, anchors, session counters, state.db + and the API-call log line (see module docstring). No-usage responses only + consume a pending compaction verdict. Returns the loop-visible outcome.""" + rearmed = False + # Track actual token usage from response for context management + if hasattr(response, 'usage') and response.usage: + canonical_usage = normalize_usage( + response.usage, + provider=agent.provider, + api_mode=agent.api_mode, + ) + # Aggregator-only usage kept for pricing: advisor tokens are priced + # at each advisor's OWN model rate and added as dollars below. + aggregator_usage = canonical_usage + # MoA: fold advisor fan-out usage into REPORTED token counts — only + # aggregator usage is returned, so advisor spend would be invisible. + _moa_ref_cost = None + _moa_client = getattr(agent, "client", None) + if _moa_client is not None and hasattr(_moa_client, "consume_reference_usage"): + try: + _ref_usage, _moa_ref_cost = _moa_client.consume_reference_usage() + if _ref_usage is not None: + canonical_usage = canonical_usage + _ref_usage + except Exception as _moa_acct_exc: # pragma: no cover - defensive + logger.debug("MoA reference usage accounting failed: %s", _moa_acct_exc) + # Flush the full-turn MoA trace when moa.save_traces is on; on the + # streaming path pass the streamed acting text so the trace is self- + # contained. + if _moa_client is not None and hasattr(_moa_client, "consume_and_save_trace"): + try: + _agg_streamed_text = ( + getattr(agent, "_current_streamed_assistant_text", "") or "" + ) + _moa_client.consume_and_save_trace( + agent.session_id, + aggregator_output_fallback=_agg_streamed_text or None, + ) + except Exception as _moa_trace_exc: # pragma: no cover - defensive + logger.debug("MoA trace flush failed: %s", _moa_trace_exc) + prompt_tokens = canonical_usage.prompt_tokens + completion_tokens = canonical_usage.output_tokens + total_tokens = canonical_usage.total_tokens + # Forward canonical token + cache buckets for context engines; + # legacy keys stay for back-compat. + usage_dict = { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": total_tokens, + "input_tokens": canonical_usage.input_tokens, + "output_tokens": canonical_usage.output_tokens, + "cache_read_tokens": canonical_usage.cache_read_tokens, + "cache_write_tokens": canonical_usage.cache_write_tokens, + "reasoning_tokens": canonical_usage.reasoning_tokens, + } + # Capture the boundary latch before update_from_response() consumes + # it: only the real prompt count right after a compaction rearms the + # budget. + _completed_compaction_pending = bool( + getattr( + agent.context_compressor, + "_verify_compaction_cleared_threshold", + False, + ) + ) + agent.context_compressor.update_from_response(usage_dict) + # Usage-anchored accounting: snapshot exact provider usage against + # the durable transcript; main-loop ONLY. MoA uses pre-fold + # aggregator usage. + _new_anchor = capture_usage_anchor( + aggregator_usage.prompt_tokens, + aggregator_usage.output_tokens, + messages, + ) + if _new_anchor is not None: + agent._usage_anchor = _new_anchor + # Anchor the display meter on the turn's FIRST response: + # later same-turn responses inflate prompt_tokens with replayed + # thinking. Display-only; compression math uses real usage. + if api_call_count == 1: + agent._turn_base_usage_anchor = _new_anchor + _compression_threshold = int( + getattr(agent.context_compressor, "threshold_tokens", 0) + or 0 + ) + if _loop_mod()._should_rearm_compression_budget( + compression_attempts, + completed_compaction_pending=_completed_compaction_pending, + prompt_tokens=prompt_tokens, + threshold_tokens=_compression_threshold, + ): + logger.info( + "Compression budget rearmed after provider-confirmed " + "recovery: prompt=%s < threshold=%s (attempts were %s/%s)", + f"{prompt_tokens:,}", + f"{_compression_threshold:,}", + compression_attempts, + max_compression_attempts, + ) + compression_attempts = 0 + # Confirmed recovery also clears the loop's stale insufficient-progress + # verdict (``_preflight_compression_blocked``), else it stays armed all + # turn and a later pressure spike grows unchecked. + rearmed = True + + # Stash canonical usage for on_turn_complete() (same shape as + # update_from_response); keep the latest call's — last request. + agent._last_turn_usage = dict(usage_dict) + elif getattr( + agent.context_compressor, + "awaiting_real_usage_after_compression", + False, + ): + # No usage -> cannot adjudicate the prior compaction; consume the + # pending verdict so later readings aren't charged to it and + # preflight deferral isn't latched indefinitely. + agent.context_compressor.update_from_response({}) + + if hasattr(response, 'usage') and response.usage: + # Persist only provider-confirmed context lengths, not probe tiers. + if getattr(agent.context_compressor, "_context_probed", False): + ctx = agent.context_compressor.context_length + if getattr(agent.context_compressor, "_context_probe_persistable", False): + _loop_mod().save_context_length(agent.model, agent.base_url, ctx) + agent._safe_print(f"{agent.log_prefix}💾 Cached context length: {ctx:,} tokens for {agent.model}") + agent.context_compressor._context_probed = False + agent.context_compressor._context_probe_persistable = False + + agent.session_prompt_tokens += prompt_tokens + agent.session_completion_tokens += completion_tokens + agent.session_total_tokens += total_tokens + agent.session_api_calls += 1 + agent.session_input_tokens += canonical_usage.input_tokens + agent.session_output_tokens += canonical_usage.output_tokens + agent.session_cache_read_tokens += canonical_usage.cache_read_tokens + agent.session_cache_write_tokens += canonical_usage.cache_write_tokens + agent.session_reasoning_tokens += canonical_usage.reasoning_tokens + # Rolling history for status-bar averages (last 10). + try: + hist = getattr(agent, "_api_latency_history", None) + if hist is not None: + hist.append(float(api_duration)) + ohist = getattr(agent, "_api_output_history", None) + if ohist is not None: + ohist.append(int(canonical_usage.output_tokens or 0)) + except Exception: + pass + + # Log API call details for debugging/observability + _cache_pct = "" + if canonical_usage.cache_read_tokens and prompt_tokens: + _cache_pct = f" cache={canonical_usage.cache_read_tokens}/{prompt_tokens} ({100*canonical_usage.cache_read_tokens/prompt_tokens:.0f}%)" + logger.info( + "API call #%d: model=%s provider=%s in=%d out=%d total=%d latency=%.1fs%s", + agent.session_api_calls, agent.model, agent.provider or "unknown", + prompt_tokens, completion_tokens, total_tokens, + api_duration, _cache_pct, + ) + + # MoA: agent.model/provider are the virtual preset/"moa" with no + # pricing entry, silently dropping aggregator spend. Price at the + # REAL model/provider from the MoA client's aggregator slot. + _agg_cost_model = agent.model + _agg_cost_provider = agent.provider + _agg_cost_base_url = agent.base_url + _agg_slot = getattr(_moa_client, "last_aggregator_slot", None) if _moa_client is not None else None + if _agg_slot and _agg_slot.get("model"): + _agg_cost_model = _agg_slot["model"] + _agg_cost_provider = _agg_slot.get("provider") or agent.provider + _agg_cost_base_url = _agg_slot.get("base_url") or agent.base_url + cost_result = estimate_usage_cost( + _agg_cost_model, + aggregator_usage, + provider=_agg_cost_provider, + base_url=_agg_cost_base_url, + api_key=getattr(agent, "api_key", ""), + ) + if cost_result.amount_usd is not None: + agent.session_estimated_cost_usd += float(cost_result.amount_usd) + # Add MoA advisor cost (already priced per-advisor at each + # advisor's own model rate) on top of the aggregator cost. + if _moa_ref_cost is not None: + try: + agent.session_estimated_cost_usd += float(_moa_ref_cost) + except (TypeError, ValueError): # pragma: no cover - defensive + pass + agent.session_cost_status = cost_result.status + agent.session_cost_source = cost_result.source + + # Persist per-call token deltas for any session_id so non-CLI runs + # can't lose accounting; gateway/session-store writes use absolute + # totals and safely overwrite these deltas. + if agent._session_db and agent.session_id: + try: + # Ensure the row exists: under concurrent SQLite load the + # initial _ensure_db_session() may fail, and UPDATE on a + # missing row silently affects 0 rows. + if not agent._session_db_created: + agent._ensure_db_session() + # Cost delta = aggregator + MoA advisor cost so state.db's + # estimated_cost_usd matches the folded token counts. + _cost_delta = None + if cost_result.amount_usd is not None: + _cost_delta = float(cost_result.amount_usd) + if _moa_ref_cost is not None: + try: + _cost_delta = (_cost_delta or 0.0) + float(_moa_ref_cost) + except (TypeError, ValueError): # pragma: no cover + pass + # Enqueued, not written: a cold state.db UPDATE here stalled + # the tool loop. Drained at finalize via _persist_session. + agent._session_db.queue_token_counts( + agent.session_id, + input_tokens=canonical_usage.input_tokens, + output_tokens=canonical_usage.output_tokens, + cache_read_tokens=canonical_usage.cache_read_tokens, + cache_write_tokens=canonical_usage.cache_write_tokens, + reasoning_tokens=canonical_usage.reasoning_tokens, + estimated_cost_usd=_cost_delta, + cost_status=cost_result.status, + cost_source=cost_result.source, + billing_provider=agent.provider, + billing_base_url=agent.base_url, + billing_mode="subscription_included" + if cost_result.status == "included" else None, + model=agent.model, + api_call_count=1, + ) + except Exception as e: + # Log failures — silent loss here undercounts analytics. + logger.debug( + "Token persistence failed (session=%s, tokens=%d): %s", + agent.session_id, total_tokens, e, + ) + + if agent.verbose_logging: + logging.debug(f"Token usage: prompt={usage_dict['prompt_tokens']:,}, completion={usage_dict['completion_tokens']:,}, total={usage_dict['total_tokens']:,}") + + # Report cache stats for any provider that returns + # ``prompt_tokens_details.cached_tokens``, not only when we inject + # cache_control markers. ``canonical_usage`` is already normalised. + cached = canonical_usage.cache_read_tokens + written = canonical_usage.cache_write_tokens + prompt = usage_dict["prompt_tokens"] + if (cached or written) and not agent.quiet_mode: + hit_pct = (cached / prompt * 100) if prompt > 0 else 0 + agent._vprint( + f"{agent.log_prefix} 💾 Cache: " + f"{cached:,}/{prompt:,} tokens " + f"({hit_pct:.0f}% hit, {written:,} written)" + ) + return ResponseUsageOutcome( + compression_attempts=compression_attempts, + rearmed=rearmed, + ) diff --git a/tests/agent/test_413_image_payload_recovery.py b/tests/agent/test_413_image_payload_recovery.py index f1f4daa98f..c319f3c4b3 100644 --- a/tests/agent/test_413_image_payload_recovery.py +++ b/tests/agent/test_413_image_payload_recovery.py @@ -197,7 +197,7 @@ class TestConversationLoopWiring: def test_413_branch_scores_bytes_not_tokens(self): import inspect - import agent.conversation_loop as loop + import agent.turn_overflow as loop # 413 handler lives in recover_from_overflow src = inspect.getsource(loop) # The byte measurement is taken before and after the 413 compression diff --git a/tests/agent/test_nous_oauth_401_guidance.py b/tests/agent/test_nous_oauth_401_guidance.py index 2569f812fb..3f654ca960 100644 --- a/tests/agent/test_nous_oauth_401_guidance.py +++ b/tests/agent/test_nous_oauth_401_guidance.py @@ -1,5 +1,5 @@ """Tests for the Nous OAuth 401 actionable-guidance branch in -``agent.conversation_loop.run_conversation``. +``agent.turn_recovery.nonretryable_client_error_result`` (the run_conversation terminal branch). Source-inspection style (matches ``test_gemini_fast_fallback.py``): we assert that the guidance strings exist in the function body so that the user-facing @@ -16,14 +16,14 @@ from __future__ import annotations import inspect -from agent import conversation_loop +from agent import turn_recovery def test_nous_provider_is_in_oauth_401_set(): """The provider-set gate that selects OAuth-specific guidance must include ``nous`` alongside ``openai-codex`` and ``xai-oauth``. """ - source = inspect.getsource(conversation_loop.run_conversation) + source = inspect.getsource(turn_recovery.nonretryable_client_error_result) # Be flexible about set element ordering — assert all three are listed # near each other in the gating expression. @@ -33,16 +33,16 @@ def test_nous_provider_is_in_oauth_401_set(): # And the gate string itself must mention all three so future refactors # that split nous off into its own gate still get caught. - needle = "_provider in {\"openai-codex\", \"xai-oauth\", \"nous\"}" + needle = "provider in {\"openai-codex\", \"xai-oauth\", \"nous\"}" assert needle in source, ( "Expected nous to be co-gated with the other OAuth providers in the " - "actionable-401-guidance branch of run_conversation." + "actionable-401-guidance branch of nonretryable_client_error_result." ) def test_nous_401_guidance_strings_present(): """User-facing remediation strings for Nous OAuth 401s must exist.""" - source = inspect.getsource(conversation_loop.run_conversation) + source = inspect.getsource(turn_recovery.nonretryable_client_error_result) # Must tell the user it's an OAuth token problem, NOT an API key problem # (Nous Portal has no API key path — auth_type=oauth_device_code only). diff --git a/tests/agent/test_send_path_history_isolation.py b/tests/agent/test_send_path_history_isolation.py index 051a41472f..1f5f6727a6 100644 --- a/tests/agent/test_send_path_history_isolation.py +++ b/tests/agent/test_send_path_history_isolation.py @@ -172,12 +172,19 @@ class TestSendPathBuildIsWiredToTheClone: import ast import inspect - source = inspect.getsource(cl) - tree = ast.parse(source) + import agent.turn_context as tc + + # The history build lives in turn_context.build_api_messages; the + # prefill insert stays in conversation_loop. Scan both homes. + nodes = [ + node + for mod in (cl, tc) + for node in ast.walk(ast.parse(inspect.getsource(mod))) + ] clone_calls = [] shallow_copies = [] - for node in ast.walk(tree): + for node in nodes: if isinstance(node, ast.Call): fn = node.func if isinstance(fn, ast.Name) and fn.id == "_clone_message_for_send": diff --git a/tests/run_agent/test_69078_image_corrupt_recovery.py b/tests/run_agent/test_69078_image_corrupt_recovery.py index ddc29ddb79..c3b76ae8fe 100644 --- a/tests/run_agent/test_69078_image_corrupt_recovery.py +++ b/tests/run_agent/test_69078_image_corrupt_recovery.py @@ -447,12 +447,12 @@ class TestCanonicalHistoryIsolation: import inspect import re as _re - import agent.conversation_loop as loop_mod + import agent.turn_recovery as loop_mod - src = inspect.getsource(loop_mod) + src = inspect.getsource(loop_mod.recover_after_classification) # Locate the image_corrupt recovery block and inspect its calls. block = _re.search( - r"image_corrupt:\n(.*?)\n\s*(?:continue|else)", src, _re.S + r"image_corrupt:\n(.*?)\n\s*(?:continue|return|else)", src, _re.S ) assert block is not None, "image_corrupt recovery branch not found" body = block.group(1) diff --git a/tests/run_agent/test_image_shrink_recovery.py b/tests/run_agent/test_image_shrink_recovery.py index 0c2984b4e2..de7c68e920 100644 --- a/tests/run_agent/test_image_shrink_recovery.py +++ b/tests/run_agent/test_image_shrink_recovery.py @@ -22,7 +22,7 @@ import sys from types import SimpleNamespace -from agent.conversation_loop import _image_error_max_dimension +from agent.turn_recovery import _image_error_max_dimension from agent.error_classifier import FailoverReason, classify_api_error