Files
hermes-agent/agent/conversation_loop.py
T

7400 lines
388 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""The agent conversation loop — extracted from ``run_agent.AIAgent``.
``run_conversation(agent, ...)`` drives one user turn (model call, tool dispatch,
retries, fallbacks, compression, post-turn hooks). Symbols that callers patch on
``run_agent`` (``handle_function_call``, ``_set_interrupt``, ``OpenAI``) resolve via
``_ra`` so those patches keep working."""
from __future__ import annotations
import json
import logging
import os
import random
import re
import ssl
import sys
import time
from typing import Any, Dict, List, Optional
from agent.codex_responses_adapter import _summarize_user_message_for_log
from agent.conversation_compression import (
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE,
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE,
COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE,
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE,
PRE_API_COMPRESSION_STATUS_TEMPLATE,
compression_blocked_transiently,
compression_skipped_due_to_lock,
context_compression_timed_out,
conversation_history_after_compression,
)
from agent.context_engine import automatic_compaction_status_message
from agent.display import KawaiiSpinner
from agent.error_classifier import FailoverReason, classify_api_error
from agent.fast_mode import begin_turn as begin_fast_mode_turn
from agent.message_metadata import append_message
from agent.turn_context import (
PreflightCompressionTimedOut,
_compression_warrants_another_preflight_pass,
_review_fork_first_request_pending,
build_turn_context,
compose_user_api_content,
reanchor_current_turn_user_idx,
)
from agent.turn_retry_state import TurnRetryState
from agent.turn_recovery import recover_before_classification
from agent.runtime_cwd import resolve_agent_cwd
from agent.message_sanitization import (
close_interrupted_tool_sequence,
_repair_tool_call_arguments,
coalesce_tool_call_id,
_sanitize_messages_surrogates,
_sanitize_structure_non_ascii,
_sanitize_structure_surrogates,
_sanitize_surrogates,
_strip_images_from_messages,
serialized_messages_bytes,
)
# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py; kept local so importing
# hermes_state (module-level DEFAULT_DB_PATH) is not forced at load time.
_STALE_MARKER_RE = re.compile(r"^\[[A-Za-z_][A-Za-z0-9_.-]*\]$")
from agent.model_metadata import (
MINIMUM_CONTEXT_LENGTH,
_estimate_tools_tokens_rough,
anchored_context_tokens,
capture_usage_anchor,
estimate_messages_tokens_rough,
estimate_request_tokens_rough,
get_context_length_from_provider_error,
is_output_cap_error,
parse_available_output_tokens_from_error,
save_context_length,
)
from agent.process_bootstrap import _install_safe_stdio
from agent.prompt_caching import (
build_prompt_cache_plan,
effective_cache_ttl,
strip_anthropic_cache_control,
strip_anthropic_tool_cache_control,
)
from agent.provider_projection import splice_provider_projection
from agent.retry_utils import (
adaptive_rate_limit_backoff,
is_zai_coding_overload_error,
jittered_backoff,
zai_coding_overload_retry_ceiling,
)
from agent.repetition_guard import is_repetition_dominated
from agent.trajectory import has_incomplete_scratchpad
# Bind before the turn starts so a source-tree swap cannot load a skewed
# finalizer at turn end.
from agent.turn_finalizer import finalize_turn
from agent.usage_pricing import estimate_usage_cost, normalize_usage
from agent import empty_response_guard as _empty_guard
from hermes_constants import PARTIAL_STREAM_STUB_ID
from hermes_logging import set_session_context
from tools.skill_provenance import set_current_write_origin
from utils import base_url_host_matches, env_var_enabled
logger = logging.getLogger(__name__)
# Scaffold marker used by _apply_active_turn_redirect and the ghost-row filter
# in the api_messages loop. Module-level so both sites can never drift.
_INTERRUPT_SCAFFOLD_MARKER = "[This response was interrupted by a user correction.]"
# One-time wrap-up notice appended when a wall-clock run budget crosses 80%
# (agent.run_budget_seconds / --run-budget): stop new work, deliver current state.
RUN_BUDGET_WRAPUP_NOTICE = (
"[SYSTEM NOTICE — run time budget nearly exhausted] "
"Run time budget nearly exhausted. Stop new discovery/verification work "
"now. Produce the required final deliverable (answer/JSON/summary) from "
"the state you already have, completing only mandatory writes."
)
def _midturn_request_pressure_tokens(
agent: Any,
api_messages: List[Dict[str, Any]],
effective_system: str,
approx_tokens: int,
) -> int:
"""Token figure the mid-turn pre-API compression guard compares.
Returns the pruned native-Responses estimate when native compaction eligibility is
proven (the generic estimate overstates the wire on compacted sessions, #96995),
else the generic message+tools figure. System prompt is counted exactly once."""
try:
from agent.codex_responses_adapter import (
estimate_native_responses_preflight_tokens,
)
native = estimate_native_responses_preflight_tokens(
agent,
api_messages,
system_prompt=effective_system or "",
tools=getattr(agent, "tools", None) or None,
)
if isinstance(native, int) and not isinstance(native, bool) and native >= 0:
return native
except Exception:
logger.debug(
"native Responses mid-turn estimate unavailable; "
"using generic transcript estimate",
exc_info=True,
)
return approx_tokens + (
_estimate_tools_tokens_rough(agent.tools) if agent.tools else 0
)
def _review_input_budget_exhausted(agent: Any) -> bool:
"""True when a detached review fork has replayed its aggregate input budget.
Only forks with an explicit ``_review_input_token_budget`` are gated (#93057). Fires
at the top of the NEXT iteration, so the budget-crossing request completes first."""
budget = getattr(agent, "_review_input_token_budget", None)
if not isinstance(budget, int) or isinstance(budget, bool) or budget <= 0:
return False
used = getattr(agent, "session_input_tokens", 0)
return isinstance(used, int) and not isinstance(used, bool) and used >= budget
def _maybe_inject_run_budget_wrapup(agent: Any, messages: List[Dict[str, Any]]) -> bool:
"""Inject the one-time wall-clock wrap-up notice when past 80% of budget.
Appends to the NEWEST ``role:"tool"`` message (cache-safe, like /steer); latches
``_run_budget_wrapup_injected`` only on a successful append. Returns True when
injected. Dormant unless ``run_budget_seconds`` + ``_run_budget_started_at`` set."""
budget = getattr(agent, "run_budget_seconds", None)
if not budget:
return False
if getattr(agent, "_run_budget_wrapup_injected", False):
return False
started = getattr(agent, "_run_budget_started_at", None)
if not started:
return False
if (time.time() - started) < 0.8 * float(budget):
return False
for i in range(len(messages) - 1, -1, -1):
msg = messages[i]
if isinstance(msg, dict) and msg.get("role") == "tool":
existing = msg.get("content", "")
if isinstance(existing, str):
msg["content"] = existing + f"\n\n{RUN_BUDGET_WRAPUP_NOTICE}"
else:
# Multimodal content blocks — append a text block.
try:
blocks = list(existing) if existing else []
blocks.append({"type": "text", "text": RUN_BUDGET_WRAPUP_NOTICE})
msg["content"] = blocks
except Exception:
return False
agent._run_budget_wrapup_injected = True
logger.info(
"Run budget wrap-up notice injected (budget=%.0fs, elapsed=%.0fs)",
float(budget),
time.time() - started,
)
return True
return False
def _restore_user_after_reference_handoff(
messages: List[Dict[str, Any]], user_message: Any
) -> bool:
"""Re-append this turn's real user ask when compaction left only a handoff.
Returns True when a restore append happened; only decides whether a restorable
ask exists (#80622)."""
if user_message is None:
return False
if isinstance(user_message, str):
if not user_message.strip():
return False
content: Any = user_message
elif isinstance(user_message, list):
if not user_message:
return False
content = user_message
else:
return False
if (
messages
and isinstance(messages[-1], dict)
and messages[-1].get("role") == "user"
and messages[-1].get("content") == content
):
return False
append_message(messages, {"role": "user", "content": content})
return True
def _should_skip_model_call_for_reference_handoff(
messages: List[Dict[str, Any]], user_message: Any
) -> bool:
"""Guard post-compaction continues against sole-handoff active turns (#80622)."""
from agent.context_compressor import reference_handoff_would_drive_next_model_call
if not reference_handoff_would_drive_next_model_call(messages):
return False
if _restore_user_after_reference_handoff(messages, user_message):
# The restored ask is an actionable non-synthetic user row appended
# after the handoff — by construction the handoff no longer drives.
return False
return True
# Fallback final_response for the sole-handoff skip (#80622). Not a replay of the
# last assistant text: finalize_turn appends final_response as a fresh assistant row.
_HANDOFF_SKIP_FINAL_RESPONSE = (
"Context was compacted. The previous response is complete — "
"awaiting your next message."
)
# Terminal final_response when compression hit its host timeout while the request
# was still oversized; resending would only bounce off the overflow error (#98722).
_COMPRESSION_TIMEOUT_FINAL_RESPONSE = (
"Context compression timed out without reducing this conversation. "
"No messages were dropped. Start a fresh session with /new, or check "
"auxiliary.compression before retrying /compress."
)
# Stable prefix of the local interrupt status string; surfaces (ACP, TUI) match on
# it to treat the text as cancellation metadata rather than assistant prose.
INTERRUPT_WAITING_FOR_MODEL_PREFIX = "Operation interrupted: waiting for model response ("
def _should_rearm_compression_budget(
compression_attempts: int,
*,
completed_compaction_pending: bool,
prompt_tokens: int,
threshold_tokens: int,
) -> bool:
"""Return True after a provider proves a completed compaction worked.
Rough estimates cannot rearm the anti-thrash budget; require the completed-
compaction latch and a positive normalized prompt count below the threshold."""
return bool(
compression_attempts
and completed_compaction_pending
and threshold_tokens > 0
and 0 < prompt_tokens < threshold_tokens
)
# Modules whose presence in a traceback (without any API-call module) marks a
# deterministic local bug not worth retrying. NEVER add "conversation_loop" or
# "run_agent": every exception passes through them; _hit_local would be True (#66267)
_LOCAL_PROCESSING_MODULES = frozenset({
"agent_runtime_helpers",
"message_content",
"message_sanitization",
"chat_completion_helpers", # only local when NOT also an API-call module
})
_API_CALL_MODULES = frozenset({
"chat_completion_helpers",
})
# Max outer-loop exceptions per user turn before giving up; only exceptions that
# ESCAPE the inner retry/fallback machinery count, so this can be small (#92450).
_MAX_OUTER_LOOP_ERRORS = 8
def _is_interpreter_shutdown_error(exc: Exception) -> bool:
"""Check if *exc* is a fatal interpreter-shutdown failure.
Delegates to ``tools.interpreter_shutdown`` (one text-matching site for the
shutdown-race bug class) but keeps the RuntimeError type gate: a ValueError
carrying similar text must not match (#93269)."""
if isinstance(exc, RuntimeError):
from tools.interpreter_shutdown import interpreter_shutting_down
return interpreter_shutting_down(exc)
return False
def _moa_client_consumes_prepared_request(client: Any) -> bool:
"""True when ``client`` is the in-process MoA facade.
Only ``MoAChatCompletions`` exposes ``prepare()``; other clients raise TypeError on
``_moa_prepared_request`` even while ``agent.provider`` stays ``"moa"``."""
completions = getattr(getattr(client, "chat", None), "completions", None)
return callable(getattr(completions, "prepare", None))
def _join_truncated_parts(parts: List[str]) -> str:
"""Join continuation fragments, adding a newline where two would glue together (#78577)."""
joined = ""
for part in parts:
if joined and not joined[-1].isspace() and part and not part[0].isspace():
joined += "\n"
joined += part
return joined
def _moa_reference_metrics_for_hook(agent: Any) -> Any:
"""Per-advisor metrics for post_api_request, or None off the MoA path.
MoA returns only the aggregator response, so a plugin sees one generation for
the whole fan-out; this carries the per-slot advisor spend across the hook boundary."""
client = getattr(agent, "client", None)
getter = getattr(client, "last_reference_metrics", None)
if not callable(getter):
return None
try:
return getter()
except Exception:
return None
def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text: str) -> None:
"""Append a provider-safe checkpoint and correction to the live turn.
Keeps only the *visible* text (demoted to plain text) then adds the correction as a
real user message, so role alternation holds and cached messages stay byte-identical.
INVARIANT: raw chain-of-thought never enters replayable content — inlined CoT reads
as a prefill jailbreak and bricks the session with empty-response storms.
INVARIANT: the interruption scaffold is replay text, carried only in the user
correction's ``api_content``; an on-screen-empty placeholder is ``display_kind=hidden``."""
visible = agent._strip_think_blocks(
getattr(agent, "_current_streamed_assistant_text", "") or ""
).strip()
checkpoint_parts = [_INTERRUPT_SCAFFOLD_MARKER]
if visible:
checkpoint_parts.extend(
["Visible response before the interruption:", visible]
)
checkpoint = "\n\n".join(checkpoint_parts)
correction = (
"[Context from the interrupted assistant response]\n"
f"{checkpoint}\n\n"
f"{text}"
)
# The live tail is normally user or tool, so an assistant placeholder + correction
# keeps strict alternation; if the tail is already assistant, fold the checkpoint
# into the user correction instead of creating assistant→assistant.
if messages and messages[-1].get("role") == "assistant":
# Transcript shows the user's own words; the provider replays the
# scaffolded form so it still sees the interrupted context.
append_message(
messages,
{"role": "user", "content": text, "api_content": correction},
)
else:
# Placeholder preserves role alternation only. Scaffold bytes must never land
# here: api_content is substituted back into content on replay (#81841).
placeholder: Dict[str, Any] = {
"role": "assistant",
"content": visible or "",
}
if not visible:
placeholder["display_kind"] = "hidden"
# Hidden row, but a non-empty neutral api_content so the pre-call
# sanitizer does not re-heal it every call (#88955). Never
# _INTERRUPT_SCAFFOLD_MARKER: as assistant text the model echoes it (#81841)
from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER
placeholder["api_content"] = _INTERRUPTED_PLACEHOLDER
append_message(messages, placeholder)
append_message(
messages,
{"role": "user", "content": text, "api_content": correction},
)
agent._current_streamed_assistant_text = ""
agent._stream_needs_break = True
def _is_copilot_provider(agent: Any) -> bool:
"""Delegate to ``AIAgent._is_copilot_provider`` (single owner of the check).
``agent.provider`` may hold the aliases ``github-copilot`` / ``github``; a bare
``provider == "copilot"`` gate would skip credential recovery for them."""
try:
return bool(agent._is_copilot_provider())
except Exception:
return (getattr(agent, "provider", "") or "").strip().lower() in {
"copilot",
"github-copilot",
"github",
}
def _is_stale_copilot_credential_error(status_code: Optional[int], error_message: str) -> bool:
"""Detect a Copilot 400 that is really a STALE / DEGRADED credential.
Matches status 400 AND ``model_not_available_for_integrator`` or
``model_not_supported`` / "the requested model is not supported", so a wrong model
name never triggers the single-shot re-exchange. Caller enforces scoping/guard."""
lowered = (error_message or "").lower()
is_400 = status_code == 400 or "error code: 400" in lowered
if not is_400:
return False
return (
"model_not_available_for_integrator" in lowered
or "not available for integrator" in lowered
or "model_not_supported" in lowered
or "the requested model is not supported" in lowered
)
def _image_error_max_dimension(error: Exception) -> Optional[int]:
"""Extract a provider-reported image dimension ceiling, if present."""
parts = []
for value in (
error,
getattr(error, "message", None),
getattr(error, "body", None),
):
if value:
try:
parts.append(str(value))
except Exception:
pass
text = " ".join(parts).lower()
if "image" not in text or "dimension" not in text or "max allowed size" not in text:
return None
match = re.search(r"max allowed size(?:\s+for [^:]+)?:\s*(\d{3,5})\s*pixels?", text)
if not match:
return None
try:
max_dimension = int(match.group(1))
except ValueError:
return None
if 512 <= max_dimension <= 8000:
return max_dimension
return None
def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str]:
"""Return a user-facing error when Ollama is loaded with too little context."""
if not getattr(agent, "tools", None):
return None
runtime_ctx = getattr(agent, "_ollama_num_ctx", None)
if not isinstance(runtime_ctx, int) or runtime_ctx <= 0:
return None
if runtime_ctx >= MINIMUM_CONTEXT_LENGTH:
return None
model = getattr(agent, "model", "") or "the selected model"
base_url = getattr(agent, "base_url", "") or "unknown base URL"
provider = getattr(agent, "provider", "") or "unknown"
tool_count = len(getattr(agent, "tools", None) or [])
logger.warning(
"Ollama runtime context too small for Hermes tool use: "
"model=%s provider=%s base_url=%s runtime_context=%d "
"minimum_context=%d estimated_request_tokens=%d tool_count=%d "
"session=%s",
model,
provider,
base_url,
runtime_ctx,
MINIMUM_CONTEXT_LENGTH,
request_tokens,
tool_count,
getattr(agent, "session_id", None) or "none",
)
return (
f"Ollama loaded `{model}` with only {runtime_ctx:,} tokens of runtime "
f"context, but Hermes needs at least {MINIMUM_CONTEXT_LENGTH:,} tokens "
"for reliable tool use.\n\n"
"Increase the Ollama context for this model and restart/reload the "
"model before trying again. A known-good starting point is 65,536 "
"tokens. In Hermes config, set `model.ollama_num_ctx: 65536` "
"(and `model.context_length: 65536` if you also override the displayed "
"model context). If you manage the model through an Ollama Modelfile, "
"set `PARAMETER num_ctx 65536` there instead."
)
def _maybe_grow_local_window(agent: Any, compressor: Any,
request_tokens: int) -> Optional[int]:
"""Try growing the managed local model's context window before compressing.
Returns the new window when the ladder granted one, else None (hold / at native /
not a managed local session). Cheap for non-local providers: one compare."""
provider = (getattr(agent, "provider", "") or "").strip().lower()
if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"):
return None
base_url = getattr(agent, "base_url", "") or ""
if "127.0.0.1" not in base_url and "localhost" not in base_url:
return None
try:
from hermes_cli.local_runtime.growth import maybe_grow_window
current_window = int(getattr(compressor, "context_length", 0) or 0)
if current_window <= 0:
return None
return maybe_grow_window(
getattr(agent, "model", "") or "",
base_url=base_url,
session_tokens=int(request_tokens),
current_window=current_window,
)
except Exception as exc: # noqa: BLE001 — growth must never break a turn
logger.debug("local window growth check failed: %s", exc)
return None
def _ra():
"""Lazy ``run_agent`` reference so patches on ``run_agent.handle_function_call`` /
``run_agent._set_interrupt`` / ``run_agent.OpenAI`` reach this code path."""
import run_agent
return run_agent
def _nous_entitlement_message(capability: str) -> str:
try:
from hermes_cli.nous_account import (
format_nous_portal_entitlement_message,
get_nous_portal_account_info,
)
account_info = get_nous_portal_account_info(force_fresh=True)
message = format_nous_portal_entitlement_message(
account_info,
capability=capability,
)
return message or ""
except Exception:
return ""
def _print_nous_entitlement_guidance(agent, capability: str) -> bool:
message = _nous_entitlement_message(capability)
if not message:
return False
for line in message.splitlines():
agent._vprint(f"{agent.log_prefix} 💡 {line}", force=True)
return True
def _system_prompt_for_hooks(api_kwargs: Any, request_messages: Any) -> Any:
"""System prompt as actually sent to the provider, for observability hooks.
Checks ``system`` (Anthropic), ``instructions`` (Responses/Codex), then
``messages[0]``. Returns None when the request carries no system prompt."""
system_prompt = api_kwargs.get("system")
if system_prompt is None:
system_prompt = api_kwargs.get("instructions")
if system_prompt is None and isinstance(request_messages, list) and request_messages:
first = request_messages[0]
if isinstance(first, dict) and first.get("role") == "system":
system_prompt = first.get("content")
return system_prompt
def _is_nous_inference_route(provider: str, base_url: str) -> bool:
provider = (provider or "").strip().lower()
if provider == "nous":
return True
base = str(base_url or "")
return (
base_url_host_matches(base, "inference-api.nousresearch.com")
)
def _billing_or_entitlement_message(
*,
capability: str,
provider: str,
base_url: str,
model: str,
unverified: bool = False,
) -> str:
if _is_nous_inference_route(provider, base_url):
return _nous_entitlement_message(capability)
provider_label = (provider or "").strip() or "the selected provider"
model_label = (model or "").strip() or "the selected model"
# Anthropic Pro/Max OAuth surfaces exhaustion of the "extra usage" bucket as a hard
# 400; point at the settings page and cycle reset — "add credits" does not apply.
if (provider or "").strip().lower() == "anthropic":
# ``unverified`` (#82154): the "out of extra usage" 400 is also returned for a
# server-side content-filter rejection, so hedge and name the other cause.
if unverified:
lines = [
(
f"{provider_label} reported that your Claude subscription usage may be "
f"exhausted for {model_label} (included quota + extra-usage credits) — "
"but this specific error is not proof of a billing problem."
),
"If https://claude.ai/settings/usage still shows quota remaining, this is "
"probably NOT a billing problem: on a Claude subscription (OAuth) token "
"Anthropic returns this same message when its content filter rejects part "
"of the request — typically a phrase in the system prompt.",
"If usage really is exhausted: wait for the billing cycle to reset, or add "
"extra usage at https://claude.ai/settings/usage",
"You can also switch to an Anthropic API key or another provider with "
"/model <model> --provider <provider>.",
# The exhaustion latch replays the stored error without issuing
# a request, so a real fix looks like it didn't work.
"Retry with a fresh credential state: `hermes auth reset anthropic`. Until "
"that cooldown clears, this error can be replayed from cache without "
"contacting the API.",
]
else:
lines = [
(
f"{provider_label} reported that your Claude subscription usage is "
f"exhausted for {model_label} (included quota + extra-usage credits)."
),
"Options: wait for the billing cycle to reset, or add extra usage at "
"https://claude.ai/settings/usage",
"You can also switch to an Anthropic API key or another provider with "
"/model <model> --provider <provider>.",
]
return "\n".join(lines)
# Provider-agnostic billing URL so every text surface (CLI, gateway, TUI) shows the
# same actionable link, not just OpenRouter.
try:
from agent.billing_links import build_billing_block
_link = build_billing_block(provider=provider, base_url=base_url, model=model)
if _link.provider_label:
provider_label = _link.provider_label
billing_url = _link.billing_url
except Exception:
billing_url = None
lines = [
(
f"{provider_label} reported that billing, credits, or account "
f"entitlement is exhausted for {model_label}."
),
"Add credits or update billing with that provider, then retry.",
]
if billing_url:
lines.append(f"{provider_label} billing: {billing_url}")
lines.append("You can switch providers temporarily with /model <model> --provider <provider>.")
return "\n".join(lines)
def _billing_block_dict(
provider, base_url, model, message="", *, unverified: bool = False
) -> Optional[dict]:
"""Best-effort structured billing descriptor (None if billing_links is unavailable)."""
try:
from agent.billing_links import build_billing_block
block = build_billing_block(
provider=provider, base_url=str(base_url), model=model, message=message
).to_dict()
except Exception:
return None
if block is not None and unverified:
# Carry the classifier's ambiguity into the structured descriptor so
# every surface rendering the block can hedge too (#82154).
block["unverified"] = True
return block
def _billing_terminal_label(summary: str, unverified: bool) -> str:
"""Terminal-failure prefix for a billing-classified error.
``unverified`` (#82154): the Anthropic "out of extra usage" 400 can be a
content-filter rejection, so the line must not assert exhaustion as fact."""
if unverified:
return (
"Provider reported usage/credit exhaustion (unverified — the same "
f"error can be a content-filter rejection, not billing): {summary}"
)
return f"Billing or credits exhausted: {summary}"
def _billing_failure_result(
*,
classified,
summary: str,
messages,
api_call_count: int,
provider: str,
base_url,
model: str,
guidance: Optional[str] = None,
) -> dict:
"""Structured terminal result for a billing-classified failure.
Single construction point so label, guidance, structured block and ambiguity flag
stay consistent across the non-retryable abort and max-retries paths (#82154)."""
unverified = bool(getattr(classified, "billing_unverified", False))
if guidance is None:
guidance = _billing_or_entitlement_message(
capability="model access",
provider=provider,
base_url=str(base_url),
model=model,
unverified=unverified,
)
final = _billing_terminal_label(summary, unverified)
if guidance:
final += f"\n\n{guidance}"
return {
"final_response": final,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": summary,
"failure_reason": classified.reason.value,
# Classifier's own retry verdict so UI (agent/error_surface.py) shows Retry
# only when a re-run can differ, not re-derived from a second taxonomy.
"failure_retryable": bool(classified.retryable),
# The billing verdict may rest on an ambiguous body (#82154) — carry
# that through the structured result, not just the prose.
"billing_unverified": unverified,
"billing_block": _billing_block_dict(
provider, base_url, model, guidance, unverified=unverified
),
}
def _print_billing_or_entitlement_guidance(
agent,
*,
capability: str,
provider: str,
base_url: str,
model: str,
unverified: bool = False,
) -> bool:
message = _billing_or_entitlement_message(
capability=capability,
provider=provider,
base_url=base_url,
model=model,
unverified=unverified,
)
if not message:
return False
for line in message.splitlines():
agent._vprint(f"{agent.log_prefix} 💡 {line}", force=True)
return True
def _try_refresh_nous_paid_entitlement_credentials(agent) -> bool:
"""Refresh Nous runtime credentials after a fresh paid-entitlement check."""
try:
from hermes_cli.nous_account import get_nous_portal_account_info
account_info = get_nous_portal_account_info(force_fresh=True)
if account_info.paid_service_access is not True:
return False
return agent._try_refresh_nous_client_credentials(
force=True,
)
except Exception:
return False
def _restore_or_build_system_prompt(agent, system_message, conversation_history):
"""Restore the cached system prompt from the session DB or build it fresh.
Mutates ``agent._cached_system_prompt`` and persists a freshly-built prompt on first
build. Row states ``missing``/``null``/``empty``/``present`` are logged and DB
failures log at WARNING so silent prefix-cache misses show in ``agent.log``."""
stored_prompt = None
stored_state = "missing"
session_row = None
if conversation_history and agent._session_db:
try:
session_row = agent._session_db.get_session(agent.session_id)
if session_row is not None:
raw_prompt = session_row.get("system_prompt")
if raw_prompt is None:
stored_state = "null"
elif raw_prompt == "":
stored_state = "empty"
else:
stored_prompt = raw_prompt
stored_state = "present"
except Exception as exc:
logger.warning(
"Session DB get_session failed for system-prompt restore "
"(session=%s): %s. Falling back to fresh build — prefix "
"cache will miss for this turn.",
agent.session_id, exc,
)
if stored_prompt and _stored_prompt_matches_runtime(agent, stored_prompt):
# Bot Chat capability epoch: the stored prompt embeds a capability fingerprint;
# a mismatch is a deliberate once-per-change rebuild. Unstamped prompts never
# take this branch; probe failures fail closed to "reuse" so cache is kept.
_bot_stale = False
try:
from tools.bot_mode_probe import (
BOT_CHAT_TITLE,
stored_bot_chat_prompt_needs_upgrade,
stored_prompt_capability_stale,
)
_home_for_epoch = None
try:
from agent.system_prompt import _agent_home
_home_for_epoch = _agent_home(agent)
except Exception:
pass
_bot_stale = stored_prompt_capability_stale(stored_prompt, _home_for_epoch)
if not _bot_stale and getattr(agent, "_bot_mode_protocol", True):
# Legacy upgrade: a Bot Chat prompt predating the epoch mechanism gets
# ONE title-gated migration rebuild; the stamped result cannot re-fire.
_t = str(getattr(agent, "_session_title_hint", "") or "").strip()
if not _t and agent._session_db and agent.session_id:
try:
_t = str(agent._session_db.get_session_title(agent.session_id) or "").strip()
except Exception:
_t = ""
if _t == BOT_CHAT_TITLE:
_bot_stale = stored_bot_chat_prompt_needs_upgrade(stored_prompt, _home_for_epoch)
except Exception:
_bot_stale = False
if _bot_stale:
logger.info(
"Bot Chat capability epoch changed for session %s; rebuilding "
"system prompt to adopt the new capability surface (one-time "
"prefix-cache break).",
agent.session_id,
)
agent._session_title_hint = "Bot Chat"
# The skills index cache (LRU + disk snapshot) does not watch the skills
# dir; a capability refresh must rebuild THROUGH it or new skills are lost.
try:
from agent.prompt_builder import clear_skills_system_prompt_cache
clear_skills_system_prompt_cache(clear_snapshot=True)
except Exception:
pass
agent._cached_system_prompt = agent._build_system_prompt(system_message)
agent._bot_capability_refreshed = True
# Persist so the NEXT turn restores the new bytes verbatim (cache break is
# once per capability change). on_session_start not re-fired: continuation.
if agent._session_db:
try:
agent._session_db.update_system_prompt(
agent.session_id, agent._cached_system_prompt
)
except Exception as exc:
logger.warning(
"Session DB update_system_prompt failed after Bot Chat "
"capability refresh (session=%s): %s. The refresh will "
"re-fire next turn.",
agent.session_id, exc,
)
return
# Continuing session — reuse the exact system prompt from the
# previous turn so the Anthropic cache prefix matches.
agent._cached_system_prompt = stored_prompt
# Same contract for tools[]: pin the array to the order this session already
# sent (tools freeze) instead of re-probing every check_fn on a fresh AIAgent.
try:
saved_tools = session_row.get("tool_names") if session_row else None
if saved_tools:
from tools.mcp_tool import restore_agent_tool_prefix
restore_agent_tool_prefix(agent, json.loads(saved_tools))
except Exception:
logger.debug("tool prefix restore skipped", exc_info=True)
# Prompt-section callbacks are new-session-only; recover their frozen bytes
# from the persisted prompt so a compression rebuild keeps them.
from agent.system_prompt import restore_plugin_prompt_sections
restore_plugin_prompt_sections(agent, stored_prompt)
# The static prefix is not persisted; rebuild it for the early cache breakpoint
# or fresh-per-turn gateway agents fall back to the single-breakpoint layout.
# reconstruct_static_prefix gates on _use_prompt_caching, fails open to legacy.
from agent.system_prompt import reconstruct_static_prefix
reconstruct_static_prefix(agent, system_message=system_message)
return
if stored_prompt:
stored_state = "stale_runtime"
logger.info(
"Stored system prompt for session %s has stale runtime identity; "
"rebuilding for model=%s provider=%s.",
agent.session_id,
getattr(agent, "model", "") or "",
getattr(agent, "provider", "") or "",
)
if conversation_history and stored_state in ("null", "empty"):
# Continuing session with an unusable stored prompt: every turn now rebuilds
# and the prefix cache misses every time.
logger.warning(
"Stored system prompt for session %s is %s; rebuilding "
"from scratch this turn. Prefix cache will miss until "
"the rebuild persists. Investigate the previous turn's "
"update_system_prompt write path.",
agent.session_id, stored_state,
)
# First turn of a new session (or recovering from a broken stored
# prompt) — build from scratch.
agent._cached_system_prompt = agent._build_system_prompt(system_message)
# Plugin hook: on_session_start — fired once for a brand-new session, not on
# continuation.
try:
from hermes_cli.lifecycle import invoke_hook as _invoke_hook
_invoke_hook(
"on_session_start",
session_id=agent.session_id,
model=agent.model,
platform=getattr(agent, "platform", None) or "",
)
except Exception as exc:
logger.warning("on_session_start hook failed: %s", exc)
# Cold-start credits seed (L3) fallback for the first-turn path; TUI/desktop seed at
# session open, so this is idempotent (skips when _credits_state exists). Fail-open.
try:
from agent.credits_tracker import seed_credits_at_session_start
seed_credits_at_session_start(agent)
except Exception:
logger.debug("cold-start credits seed failed (fail-open)", exc_info=True)
# Persist the system prompt snapshot; the gateway path (fresh AIAgent per turn)
# reads this row every turn, so a failure here breaks prefix-cache reuse.
if agent._session_db:
try:
agent._session_db.update_system_prompt(agent.session_id, agent._cached_system_prompt)
from tools.mcp_tool import persist_agent_tool_names
persist_agent_tool_names(agent)
except Exception as exc:
logger.warning(
"Session DB update_system_prompt failed for session %s: "
"%s. Subsequent turns will rebuild the system prompt and "
"miss the prefix cache.",
agent.session_id, exc,
)
def _stored_prompt_matches_runtime(agent, prompt: str) -> bool:
"""Return False when the persisted runtime-identity lines are stale."""
def line_value(label: str) -> str:
"""Last matching line wins.
Safe ONLY for fields in the volatile tier at the END of the prompt; embedded
project context could shadow earlier fields — see ``host_info_value``."""
prefix = f"{label}:"
value = ""
for line in prompt.splitlines():
if line.startswith(prefix):
value = line[len(prefix):].strip()
return value
def host_info_value(label: str) -> str:
"""Read a field from the prompt's own host-info block.
Anchors on the FIRST ``User home directory:`` line so a user's ``AGENTS.md`` row
cannot match; a false mismatch would rebuild the prompt every turn."""
prefix = f"{label}:"
lines = prompt.splitlines()
for idx, line in enumerate(lines):
if not line.startswith("User home directory:"):
continue
for candidate in lines[idx + 1: idx + 4]:
if candidate.startswith(prefix):
return candidate[len(prefix):].strip()
return ""
stored_model = line_value("Model")
current_model = str(getattr(agent, "model", "") or "").strip()
if stored_model and current_model and stored_model != current_model:
return False
stored_provider = line_value("Provider")
current_provider = str(getattr(agent, "provider", "") or "").strip()
if stored_provider and current_provider and stored_provider != current_provider:
return False
# cwd drift check. Compare against resolve_agent_cwd() — the SAME resolver used to
# build the prompt — so TERMINAL_CWD sessions are not falsely rejected.
stored_cwd = host_info_value("Current working directory")
if stored_cwd:
if stored_cwd != str(resolve_agent_cwd()):
return False
# Runtime-surface drift: reusing a desktop-built prompt on a terminal session (or
# vice versa) would inject the wrong runtime hints.
stored_platform = line_value("Platform")
current_platform = str(getattr(agent, "platform", "") or "").strip()
if stored_platform and current_platform and stored_platform != current_platform:
return False
return True
# Named constants for the _get_continuation_prompt variants so
# _is_synthetic_compression_user_turn can recognize them by content after a crash
# persists one; SessionDB projection strips the _length_continuation_nudge tag.
_LENGTH_CONTINUATION_NETWORK_STUB = (
"[System: The previous response was cut off by a "
"network error mid-stream. Continue exactly where "
"you left off. Do not restart or repeat prior text. "
"Finish the answer directly.]"
)
_LENGTH_CONTINUATION_OUTPUT_LIMIT = (
"[System: Your previous response was truncated by the output "
"length limit. Continue exactly where you left off. Do not "
"restart or repeat prior text. Finish the answer directly.]"
)
# The dropped-tools variant interpolates tool names, so
# _is_synthetic_compression_user_turn matches this prefix with str.startswith.
_LENGTH_CONTINUATION_DROPPED_TOOLS_PREFIX = "[System: Your previous tool call "
def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List[str]] = None) -> str:
if is_partial_stub and dropped_tools:
tool_list = ", ".join(dropped_tools[:3])
return (
f"{_LENGTH_CONTINUATION_DROPPED_TOOLS_PREFIX}"
f"({tool_list}) was too large and "
"the stream timed out before it "
"could be delivered. Do NOT retry "
"the same tool call with the same "
"large content. Instead, break the "
"content into multiple smaller tool "
"calls (e.g. use multiple patch calls "
"or write smaller files). Each tool "
"call's arguments must be under ~8K "
"tokens to avoid stream timeouts.]"
)
elif is_partial_stub:
return _LENGTH_CONTINUATION_NETWORK_STUB
else:
return _LENGTH_CONTINUATION_OUTPUT_LIMIT
# Nudge for Codex/Responses turns that returned only internal reasoning: a bare retry
# would be byte-identical (nothing replayable emitted), so the model repeats it.
_CODEX_INCOMPLETE_NUDGE = (
"[System: Your previous response contained only internal reasoning and "
"never produced a visible answer or tool call. Do not keep thinking. "
"Produce your final answer as plain text now (or make the tool call "
"you were planning).]"
)
# Re-prompt after an acknowledgment-only Codex/Responses reply; named so
# _is_synthetic_compression_user_turn can recognize it like _CODEX_INCOMPLETE_NUDGE.
_CODEX_ACK_CONTINUATION_NUDGE = (
"[System: Continue now. Execute the required tool calls and only "
"send your final answer after completing the task.]"
)
# Re-prompt for finish_reason="tool_calls" with empty tool_calls. Named like
# _CODEX_ACK_CONTINUATION_NUDGE: an interrupt mid-retry can persist it.
_DROPPED_TOOLCALL_NUDGE_CONTENT = (
"Your previous turn indicated a tool call but none was "
"included. Do not narrate a plan or restate intent — issue "
"the actual tool call now to continue the task."
)
# Re-prompt for an empty response after tool calls (#9400). Named because its
# _empty_recovery_synthetic metadata flag does not survive SessionDB projection.
_EMPTY_TOOL_RESPONSE_NUDGE = (
"You just executed tool calls but returned an "
"empty response. Please process the tool "
"results above and continue with the task."
)
# Shared recovery trailer for both content-policy refusal paths (HTTP-200
# content_filter and the content_policy_blocked exception) so guidance cannot drift.
_CONTENT_POLICY_RECOVERY_HINT = (
"Try rephrasing the request, narrowing the context, or "
"adding a fallback provider with `hermes fallback add`."
)
# Memo for send-path tool-call argument canonicalization, which re-runs on every
# historical call each iteration. Sound: canonicalization is pure and deterministic;
# malformed strings raise before being stored, so the repair fallback is never memoized.
_CANON_ARGS_CACHE: Dict[str, str] = {}
_CANON_ARGS_CACHE_MAX = 4096
# Count bound alone does not bound MEMORY: argument strings can run 100KB+, so a byte
# budget bounds the worst case while keeping the memo effective for ~0.5-2KB args.
_CANON_ARGS_CACHE_MAX_BYTES = 32 * 1024 * 1024
_canon_args_cache_bytes = 0
def _canonicalize_tool_call_arguments(arg_str: str) -> str:
"""Return the canonical wire form of a tool-call arguments JSON string.
Raises whatever ``json.loads`` raises on malformed input; the caller falls back to
``_repair_tool_call_arguments``."""
global _canon_args_cache_bytes
cached = _CANON_ARGS_CACHE.get(arg_str)
if cached is not None:
return cached
canonical = json.dumps(
json.loads(arg_str), separators=(",", ":"), sort_keys=True,
)
_CANON_ARGS_CACHE[arg_str] = canonical
_canon_args_cache_bytes += len(arg_str) + len(canonical)
while len(_CANON_ARGS_CACHE) > _CANON_ARGS_CACHE_MAX or (
_canon_args_cache_bytes > _CANON_ARGS_CACHE_MAX_BYTES
and len(_CANON_ARGS_CACHE) > 1
):
try:
evicted_key = next(iter(_CANON_ARGS_CACHE))
evicted_val = _CANON_ARGS_CACHE.pop(evicted_key)
_canon_args_cache_bytes -= len(evicted_key) + len(evicted_val)
except (StopIteration, KeyError, RuntimeError):
break
return canonical
def _clone_message_for_send(msg):
"""Structural clone of a history message for the per-call API copy.
Clones every dict/list recursively while sharing immutable leaves, so in-place
send-path rewrites can never reach the persisted transcript (#80498). Cheaper than
copy.deepcopy; messages are JSON-shaped and acyclic, tuples are shared as leaves."""
if isinstance(msg, dict):
return {
k: _clone_message_for_send(v) if isinstance(v, (dict, list)) else v
for k, v in msg.items()
}
if isinstance(msg, list):
return [
_clone_message_for_send(v) if isinstance(v, (dict, list)) else v
for v in msg
]
return msg
def _canonicalize_api_tool_calls(api_messages) -> None:
"""Canonicalize tool-call argument JSON on the send-path message copy.
Rewrites ``tool_calls`` in place (copy-on-write for the dicts it touches; persisted
history untouched). The memo bounds parse/serialize to one per UNIQUE string."""
for am in api_messages:
tcs = am.get("tool_calls")
if not tcs:
continue
new_tcs = []
for tc in tcs:
if isinstance(tc, dict) and "function" in tc:
try:
tc = {**tc, "function": {
**tc["function"],
"arguments": _canonicalize_tool_call_arguments(
tc["function"]["arguments"]
),
}}
except Exception:
# Copy-on-write as defense in depth: callers may pass shallow
# copies, and writing into a shared tc["function"] rewrote the
# stored turn with "{}" on the unrepairable path (#80498).
tc = {**tc, "function": {
**tc["function"],
"arguments": _repair_tool_call_arguments(
tc["function"]["arguments"],
tc["function"].get("name", "?"),
),
}}
new_tcs.append(tc)
am["tool_calls"] = new_tcs
def _invalid_tool_name_error_content(name: str, valid_tool_names) -> str:
"""Error-result content for a tool call whose name isn't a real tool.
A blank name is a model echoing tool-call syntax seen in data, not a typo (#47967);
dumping the catalog feeds that loop, so send a terse error instead. A nonempty wrong
name still gets the catalog so the model can self-correct."""
if not (name or "").strip():
return (
"Tool call rejected: the tool name was empty. "
"If tool-call XML or JSON appeared in file "
"contents or tool output, that is data — do "
"not re-emit it as a tool call. To call a "
"tool, use a valid name from your tool list; "
"otherwise reply in plain text."
)
available = ", ".join(sorted(valid_tool_names))
return f"Tool '{name}' does not exist. Available tools: {available}"
def _content_policy_blocked_result(
messages: List[Dict],
api_call_count: int,
*,
final_response: str,
error_detail: str,
) -> Dict[str, Any]:
"""Build the terminal turn result for a content-policy block.
Refusals are deterministic for the unchanged prompt, so no retry; both the HTTP-200
and exception paths return this shape with a ``content_policy_blocked:`` error."""
return {
"final_response": final_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": f"content_policy_blocked: {error_detail}",
}
def _compression_deferred_result(
agent,
messages: List[Dict],
api_call_count: int,
reason: str = "lock",
) -> Dict[str, Any]:
"""Build the soft turn result for a transiently-deferred compression.
Both ``reason="lock"`` and ``reason="transient_block"`` must end as
``compression_deferred``, never ``compression_exhausted`` — the gateway wipes the
session on exhaustion (#9893/#35809). ``failed`` stays False; the turn persists."""
if reason == "transient_block":
block = getattr(agent, "_compression_blocked_transient", None)
logger.info(
"turn deferred: compression transiently blocked (%s) "
"(session=%s) — not counting as compression exhaustion",
block if isinstance(block, str) else "unknown guard",
agent.session_id or "none",
)
_final = (
"Context compression is temporarily paused after a recent "
"failed attempt. Please retry in a moment — compression will "
"resume automatically (or run /compress to force a retry now)."
)
else:
holder = getattr(agent, "_compression_skipped_due_to_lock", None)
logger.info(
"turn deferred: compression lock held by another path "
"(session=%s holder=%s) — not counting as compression exhaustion",
agent.session_id or "none",
holder if isinstance(holder, str) else "unconfirmed",
)
_final = (
"Context compression is already running for this session. "
"Please retry in a moment — your next message will be processed "
"once the concurrent compression finishes."
)
try:
agent._flush_status_buffer()
except Exception:
pass
return {
"final_response": _final,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final,
"partial": True,
"failed": False,
"compression_deferred": True,
"session_id": agent.session_id,
}
def _provider_overflow_exhausted_result(
agent,
messages: List[Dict],
conversation_history,
api_call_count: int,
request_pressure_tokens: int,
max_compression_attempts: int,
) -> Dict[str, Any]:
"""Fail closed when a rebuilt request is still too large after recovery."""
agent._flush_status_buffer()
logger.error(
"%sContext compression failed after %d attempts; rebuilt request "
"remains over threshold at ~%s tokens.",
agent.log_prefix,
max_compression_attempts,
f"{request_pressure_tokens:,}",
)
agent._persist_session(messages, conversation_history)
final_response = (
"Context length exceeded: compression could not reduce the rebuilt "
"request below the safe threshold."
)
return {
"final_response": final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_exhausted",
}
def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool:
"""Rewrite a cache-decorated system message in place, keeping its blocks.
Assigning a bare string over the ``[static prefix, volatile tail]`` block list drops
both cache_control breakpoints. Only the LAST ``Model:``/``Provider:`` lines change.
Returns False when the shape cannot be safely patched."""
content = system_message.get("content")
if not isinstance(content, list) or not content:
return False
if not all(
isinstance(part, dict) and part.get("type") == "text" for part in content
):
return False
if len(content) == 1:
content[0]["text"] = effective
return True
if len(content) == 2:
head = content[0].get("text") or ""
if head and effective.startswith(head):
tail = effective[len(head):]
if tail:
content[1]["text"] = tail
return True
return False
def _sync_failover_system_message(agent, api_messages, active_system_prompt):
"""Refresh the in-flight system message after a provider failover.
``try_activate_fallback`` rewrites the identity lines on ``_cached_system_prompt``,
but this call block's ``api_messages`` were built pre-failover and are reused each
retry. Mutates ``api_messages[0]`` in place; returns the new ``active_system_prompt``."""
sp = getattr(agent, "_cached_system_prompt", None)
if not isinstance(sp, str) or not sp:
return active_system_prompt
if api_messages and api_messages[0].get("role") == "system":
effective = sp
if agent.ephemeral_system_prompt:
effective = (effective + "\n\n" + agent.ephemeral_system_prompt).strip()
if not _rewrite_system_content_blocks(api_messages[0], effective):
api_messages[0]["content"] = effective
return sp
def _ensure_cached_system_prompt_static(agent, system_message=None) -> None:
"""Rebuild ``_cached_system_prompt_static`` when caching becomes active (#72626).
Sessions restored under a cache-off primary skip the static-prefix rebuild; a later
failover to a cache-on provider would otherwise silently fall back to the legacy
system-plus-3 layout. Wraps ``reconstruct_static_prefix`` (memoizes failures)."""
from agent.system_prompt import reconstruct_static_prefix
reconstruct_static_prefix(
agent, system_message=system_message, log_label="failover redecoration"
)
def _peel_moa_guidance(
messages: List[Dict[str, Any]],
guidance: Any,
) -> List[Dict[str, Any]]:
"""Remove MoA reference guidance attached by ``_attach_reference_guidance``.
Kept adjacent to the attach so the forward/inverse shapes evolve together."""
from agent.moa_loop import peel_reference_guidance
return peel_reference_guidance(messages, guidance)
def _redecorate_prompt_cache_for_provider(
agent,
api_messages: List[Dict[str, Any]],
*,
system_message=None,
moa_prepared: Optional[Dict[str, Any]] = None,
tools_for_api: Optional[List[Dict[str, Any]]] = None,
) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]] | tuple[List[Dict[str, Any]], Optional[Dict[str, Any]], List[Dict[str, Any]]]:
"""Strip and re-apply cache_control for the *current* provider policy.
Decoration runs once per call block for the primary provider, but failover
``continue`` paths reuse ``api_messages`` (#72626), so reshape at the top of each
retry from the mutated in-flight request. MoA guidance is peeled and rebased."""
messages: List[Dict[str, Any]] = [
dict(m) if isinstance(m, dict) else m for m in (api_messages or [])
]
prepared = moa_prepared
guidance = prepared.get("guidance") if isinstance(prepared, dict) else None
if guidance:
messages = _peel_moa_guidance(messages, guidance)
strip_anthropic_cache_control(messages)
planned_tools = strip_anthropic_tool_cache_control(
tools_for_api if tools_for_api is not None else getattr(agent, "tools", [])
)
if prepared is not None and getattr(agent, "provider", None) == "moa":
# Prepared MoA state is canonical: the synchronous acting-aggregator
# sender owns its destination-local cache plan after it resolves the slot.
completions = getattr(getattr(agent.client, "chat", None), "completions", None)
rebase = getattr(completions, "rebase_prepared_request", None)
if callable(rebase):
prepared = rebase(prepared, messages)
messages = prepared["messages"]
if tools_for_api is None:
return messages, prepared
return messages, prepared, planned_tools
# Direct attribute access, not getattr: the flags are always initialized on
# AIAgent, and a default would mask a real init bug as silent cache-off.
if agent._use_prompt_caching:
_ensure_cached_system_prompt_static(agent, system_message=system_message)
static = getattr(agent, "_cached_system_prompt_static", None)
direct_tool_cache = getattr(
agent,
"_direct_native_anthropic_tool_cache_capability",
lambda: False,
)()
from agent.prompt_caching import envelope_tool_part_cache_markers_supported
plan = build_prompt_cache_plan(
messages,
planned_tools,
# Clamp per-destination: a configured 1h regresses to 5m on
# Qwen/Alibaba routes, whose context cache is 5m-only (#84733).
cache_ttl=effective_cache_ttl(
agent._cache_ttl,
provider=agent.provider,
model=agent.model,
),
native_anthropic=agent._use_native_cache_layout,
static_system_prefix=static if isinstance(static, str) else None,
direct_native_tool_cache=direct_tool_cache,
# LiteLLM-style envelope routes forward part-level markers into
# tool_result.content[] → non-retryable 400 (#89886).
tool_part_markers=envelope_tool_part_cache_markers_supported(
getattr(agent, "provider", ""), getattr(agent, "base_url", "")
),
)
messages = plan.messages
planned_tools = plan.tools
if tools_for_api is None:
return messages, prepared
return messages, prepared, planned_tools
def _apply_context_engine_selection(
agent: Any,
api_messages: List[Dict[str, Any]],
conversation_messages: List[Dict[str, Any]],
incoming_message: Optional[Dict[str, Any]],
*,
logger: Any,
) -> List[Dict[str, Any]]:
"""Run the optional per-turn ``ContextEngine.select_context()`` hook.
Returns the (possibly replaced) request list. Fail-open: a missing hook, exception,
or invalid return yields ``api_messages`` unchanged; history is never mutated."""
engine = getattr(agent, "context_compressor", None)
if engine is None or not hasattr(engine, "select_context"):
return api_messages
# Skip the no-op base ``select_context`` so non-implementing engines pay nothing;
# ``hasattr`` is not enough: the ABC defines a default. Lazy import avoids a cycle.
try:
from agent.context_engine import ContextEngine as _CE
if getattr(engine.select_context, "__func__", None) is _CE.select_context:
return api_messages
except Exception:
pass
session_label = getattr(agent, "session_id", None) or "-"
# Structural clones: the engine must not be able to write through nested
# containers into persisted history; only the request list is acted on (#80498).
_conv_copy = [_clone_message_for_send(m) for m in conversation_messages] \
if conversation_messages is not None else None
_incoming_copy = _clone_message_for_send(incoming_message) if isinstance(incoming_message, dict) else incoming_message
try:
selected = engine.select_context(
api_messages,
conversation_messages=_conv_copy,
incoming_message=_incoming_copy,
budget_tokens=getattr(engine, "context_length", 0) or 0,
)
except Exception:
logger.warning(
"Context engine select_context hook failed; using unmodified "
"request messages (session=%s)",
session_label,
exc_info=True,
)
return api_messages
if selected is None:
return api_messages
# Require a NON-EMPTY list of dicts: ``all([])`` is ``True``, so a ``[]`` from a
# buggy engine would otherwise replace the request instead of failing open.
if isinstance(selected, list) and selected and all(isinstance(m, dict) for m in selected):
return selected
logger.warning(
"Context engine select_context returned an invalid value "
"(not a non-empty list of dicts); ignoring (session=%s)",
session_label,
)
return api_messages
def _notify_context_engine_turn_complete(
agent: Any,
messages: List[Dict[str, Any]],
*,
usage: Optional[Dict[str, Any]] = None,
logger: Any,
**meta: Any,
) -> None:
"""Notify the active context engine that a user turn has finished.
Fail-open: a missing/no-op hook or any exception is swallowed. ``messages`` is
passed as a copy so the engine cannot mutate the persisted transcript."""
engine = getattr(agent, "context_compressor", None)
hook = getattr(engine, "on_turn_complete", None)
if engine is None or not callable(hook):
return
# Skip the no-op base ``on_turn_complete`` so non-implementing engines pay nothing
# per turn. Lazy import avoids an import cycle with agent.context_engine.
try:
from agent.context_engine import ContextEngine as _CE
if getattr(hook, "__func__", None) is _CE.on_turn_complete:
return
except Exception:
pass
try:
hook(
# Structural clones: dict(m) would let a hook write into nested containers
# of the persisted transcript (#80498).
[_clone_message_for_send(m) for m in messages],
usage=usage,
**meta,
)
except Exception:
logger.warning(
"Context engine on_turn_complete hook failed (session=%s)",
getattr(agent, "session_id", None) or "-",
exc_info=True,
)
def run_conversation(
agent,
user_message: Any,
system_message: str = None,
conversation_history: List[Dict[str, Any]] = None,
task_id: str = None,
stream_callback: Optional[callable] = None,
persist_user_message: Optional[Any] = None,
persist_user_timestamp: Optional[float] = None,
persist_user_display_kind: Optional[str] = None,
persist_user_display_metadata: Optional[Dict[str, Any]] = None,
persist_user_platform_id: Optional[str] = None,
moa_config: Optional[dict[str, Any]] = None,
) -> Dict[str, Any]:
"""Run a complete conversation with tool calling until completion.
Args:
stream_callback: per-text-delta callback (TTS); None uses the non-streaming path.
persist_user_message: clean text to store when ``user_message`` carries API-only
synthetic prefixes; ``persist_user_timestamp`` / ``persist_user_platform_id``
are stored as metadata (platform id lets restart drain recovery dedup).
persist_user_display_kind/metadata: display-only event rendering (``auto_continue``,
``model_switch``); the model still receives the message unchanged.
Returns: dict with the final response and message history."""
if moa_config is None:
try:
from hermes_cli.moa_config import decode_moa_turn
_decoded_message, _decoded_moa_config = decode_moa_turn(user_message)
if _decoded_moa_config is not None:
user_message = _decoded_message
moa_config = _decoded_moa_config
if persist_user_message is None:
persist_user_message = _decoded_message
except Exception:
pass
# The gateway caches agents across turns; compression state is per-turn, or a stale
# in-place boundary would make a later uncompressed result look compacted.
agent._last_compaction_in_place = False
agent._last_compression_attempt_recorded = False
agent._last_compression_attempt_in_place = None
begin_fast_mode_turn(agent, conversation_history)
# Adopt ~/.hermes/.env credential/base-url edits made since the last turn — a
# Settings save updates .env, not this worker's client (#67821). No-op if unchanged.
try:
agent._try_refresh_env_client_credentials()
except Exception:
logger.debug("per-turn env credential refresh failed", exc_info=True)
# ── Per-turn setup (the prologue) ──
# All once-per-turn setup lives in ``build_turn_context`` (agent/turn_context.py);
# it mutates ``agent`` as the inline code did and returns the locals the loop reads.
try:
_ctx = build_turn_context(
agent,
user_message,
system_message,
conversation_history,
task_id,
stream_callback,
persist_user_message,
persist_user_timestamp,
persist_user_display_kind=persist_user_display_kind,
persist_user_display_metadata=persist_user_display_metadata,
persist_user_platform_id=persist_user_platform_id,
restore_or_build_system_prompt=_restore_or_build_system_prompt,
install_safe_stdio=_install_safe_stdio,
sanitize_surrogates=_sanitize_surrogates,
summarize_user_message_for_log=_summarize_user_message_for_log,
set_session_context=set_session_context,
set_current_write_origin=set_current_write_origin,
ra=_ra,
# MoA turns append per-call aggregated context to the API copy of the
# user message, so no byte-stable api_content sidecar can be stamped.
moa_active=bool(moa_config),
)
except PreflightCompressionTimedOut as _preflight_timeout_exc:
# Preflight compression timed out; no provider call sent (#98424). Return the
# typed recovery result: surfaces hide raw exception text, which would bury the
# actionable guidance and skip the compression_exhausted recovery contract.
logger.warning(
"Turn-start preflight compression timed out — ending turn with "
"typed recovery result: %s",
_preflight_timeout_exc,
)
# Clear the tripwire slot note_turn_start registered; the early return skips the
# persist funnel that clears it. The user row is deliberately NOT persisted:
# the gateway skips persistence for compression_exhausted results (#7100).
from agent.agent_runtime_helpers import note_turn_persisted
note_turn_persisted(agent)
# Not _COMPRESSION_TIMEOUT_FINAL_RESPONSE — that describes a different state
# (compression ran, could not reduce); the exception text carries the guidance.
_final_response = str(_preflight_timeout_exc)
return {
"final_response": _final_response,
"messages": list(conversation_history or []),
"completed": False,
"api_calls": 0,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_timeout",
}
user_message = _ctx.user_message
original_user_message = _ctx.original_user_message
messages = _ctx.messages
conversation_history = _ctx.conversation_history
active_system_prompt = _ctx.active_system_prompt
effective_task_id = _ctx.effective_task_id
turn_id = _ctx.turn_id
current_turn_user_idx = _ctx.current_turn_user_idx
_should_review_memory = _ctx.should_review_memory
_plugin_user_context = _ctx.plugin_user_context
_ext_prefetch_cache = _ctx.ext_prefetch_cache
# Commentary deduplication spans all provider continuations and tool calls
# within one user turn, but must not suppress the same phrase next turn.
agent._delivered_interim_texts = set()
# A configured SessionDB append failure halts only the affected turn. A
# cached gateway agent must recover on the next message if storage did.
agent._incremental_persistence_failed = False
# Cause of the last persistence failure this turn ('locked'/'disk'/'unknown', see
# hermes_state.classify_persistence_error). Reset so a prior diagnosis cannot leak.
agent._last_persistence_error_cause = None
# Per-turn diagnostic: a failed compression-tip adoption in a previous
# turn's flush must not be reported against this turn.
agent._compression_adoption_failed = False
# Main conversation loop counters (pure locals consumed by the loop below).
api_call_count = 0
final_response = None
interrupted = False
failed = False
codex_ack_continuations = 0
length_continue_retries = 0
# Turn-scoped one-shot: armed by a thinking-only truncation, consumed by
# build_api_kwargs; must not survive an interrupted turn into the next one.
agent._ephemeral_reasoning_off = False
# Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS.
_outer_error_count = 0
truncated_tool_call_retries = 0
truncated_response_parts: List[str] = []
compression_attempts = 0
# Per-turn compression attempt cap shared by the pre-API gate, 413 handlers and
# post-tool compaction; a consecutive-ineffective-attempt backstop, rearmed only
# after a provider response reports a prompt below threshold. Default 3 if unset.
max_compression_attempts = getattr(agent, "max_compression_attempts", 3)
_last_preflight_pressure: Optional[int] = None
_preflight_compression_blocked = _ctx.preflight_compression_blocked
# A provider overflow outweighs the rough-estimate calibration that defers preflight
# after compaction: stay armed until the rebuilt request is below the threshold.
_provider_overflow_recovery_pending = False
# Armed when a compression host-timeout ends the turn; finalize reuses the gateway
# context-recovery contract (error/partial/compression_exhausted) (#98722).
_compression_timeout_exhausted = False
_turn_exit_reason = "unknown" # Diagnostic: why the loop ended
# Last answer held back by a verification gate: if the continuation exhausts the
# budget this is the best user-facing result, distinct from error/recovery text.
_pending_verification_response = None
# Whether the pending verification candidate was already streamed as interim.
# ``_response_was_previewed`` is set ONLY if it becomes the final response (#65919).
_pending_verification_response_previewed = False
# If pre-API compression fires after MoA advisors ran, retain their guidance and
# rebase it onto the compacted transcript next iteration — no second fan-out.
pending_moa_prepared_request = None
# Per-turn tally of credential-pool refreshes by (provider, pool-entry-id): caps
# same-entry refreshes on a persistent 401 so fallback takes over (#26080).
agent._auth_pool_refresh_counts = {}
# Per-turn usage forwarded to the context engine's on_turn_complete() hook; left
# None on turns that never reach a response so the hook never sees stale usage.
agent._last_turn_usage = None
# Opt-in runtime: api_mode == codex_app_server hands the whole turn to the codex
# app-server subprocess (see agent/transports/codex_app_server_session.py).
if agent.api_mode == "codex_app_server":
return agent._run_codex_app_server_turn(
user_message=user_message,
original_user_message=original_user_message,
messages=messages,
effective_task_id=effective_task_id,
should_review_memory=_should_review_memory,
)
while (api_call_count < agent.max_iterations and agent.iteration_budget.remaining > 0) or agent._budget_grace_call:
_redirect_text = agent._drain_pending_redirect()
if _redirect_text:
_apply_active_turn_redirect(agent, messages, _redirect_text)
if isinstance(original_user_message, str):
original_user_message = (
f"{original_user_message}\n\n"
f"User correction during the turn: {_redirect_text}"
)
agent._persist_session(messages, conversation_history)
# Reset per-turn checkpoint dedup so each iteration can take one snapshot
agent._checkpoint_mgr.new_turn()
# Check for interrupt request (e.g., user sent new message)
if agent._interrupt_requested:
interrupted = True
_turn_exit_reason = "interrupted_by_user"
if not agent.quiet_mode:
agent._safe_print("\n⚡ Breaking out of tool loop due to interrupt...")
break
# Aggregate input budget for detached auxiliary forks: bounds the whole review,
# not each request. Checked between iterations so the crossing request's writes
# have landed, mirroring the iteration-budget exit (#93057).
if _review_input_budget_exhausted(agent):
_turn_exit_reason = "review_input_budget_exhausted"
if not agent.quiet_mode:
agent._safe_print(
f"\n⏹️ Review input budget exhausted "
f"({int(agent.session_input_tokens):,} tokens) — stopping "
f"the review tool loop before the next provider call."
)
break
api_call_count += 1
agent._api_call_count = api_call_count
agent._touch_activity(f"starting API call #{api_call_count}")
# Grace call: budget exhausted but the model gets one more call. Consume the
# flag so the loop exits after this iteration regardless of outcome.
if agent._budget_grace_call:
agent._budget_grace_call = False
elif not agent.iteration_budget.consume():
_turn_exit_reason = "budget_exhausted"
if not agent.quiet_mode:
agent._safe_print(f"\n⚠️ Iteration budget exhausted ({agent.iteration_budget.used}/{agent.iteration_budget.max_total} iterations used)")
break
# Fire step_callback for gateway hooks (agent:step event)
if agent.step_callback is not None:
try:
prev_tools = []
for _idx, _m in enumerate(reversed(messages)):
if _m.get("role") == "assistant" and _m.get("tool_calls"):
_fwd_start = len(messages) - _idx
_results_by_id = {}
for _tm in messages[_fwd_start:]:
if _tm.get("role") != "tool":
break
_tcid = _tm.get("tool_call_id")
if _tcid:
_results_by_id[_tcid] = _tm.get("content", "")
prev_tools = [
{
"name": tc["function"]["name"],
"result": _results_by_id.get(tc.get("id")),
"arguments": tc["function"].get("arguments"),
}
for tc in _m["tool_calls"]
if isinstance(tc, dict)
]
break
agent.step_callback(api_call_count, prev_tools)
except Exception as _step_err:
logger.debug("step_callback error (iteration %s): %s", api_call_count, _step_err)
# Track tool-calling iterations for skill nudge.
# Counter resets whenever skill_manage is actually used.
if (agent._skill_nudge_interval > 0
and "skill_manage" in agent.valid_tool_names):
agent._iters_since_skill += 1
# ── Pre-API-call /steer drain ──────────────────────────────────
# Drain a /steer sent during the last API call into the newest tool message so
# it lands THIS iteration. Never put in a user message (breaks alternation).
_pre_api_steer = agent._drain_pending_steer()
if _pre_api_steer:
_injected = False
for _si in range(len(messages) - 1, -1, -1):
_sm = messages[_si]
if isinstance(_sm, dict) and _sm.get("role") == "tool":
from agent.prompt_builder import format_steer_marker
marker = format_steer_marker(_pre_api_steer)
existing = _sm.get("content", "")
if isinstance(existing, str):
_sm["content"] = existing + marker
else:
# Multimodal content blocks — append text block
try:
blocks = list(existing) if existing else []
blocks.append({"type": "text", "text": marker})
_sm["content"] = blocks
except Exception:
pass
_injected = True
logger.debug(
"Pre-API-call steer drain: injected into tool msg at index %d",
_si,
)
break
if not _injected:
# No tool message to inject into — put it back so
# the post-tool-execution drain picks it up later.
_lock = getattr(agent, "_pending_steer_lock", None)
if _lock is not None:
with _lock:
if agent._pending_steer:
agent._pending_steer = agent._pending_steer + "\n" + _pre_api_steer
else:
agent._pending_steer = _pre_api_steer
else:
existing = getattr(agent, "_pending_steer", None)
agent._pending_steer = (existing + "\n" + _pre_api_steer) if existing else _pre_api_steer
# ── Wall-clock run-budget wrap-up notice ───────────────────────
# One-shot at 80% of agent.run_budget_seconds: ask the model to wrap up via the
# same cache-safe channel as /steer (newest tool result); off with no budget.
if getattr(agent, "run_budget_seconds", None):
_maybe_inject_run_budget_wrapup(agent, messages)
# Reasoning lives in content via <think> tags for trajectory storage, but some
# providers (Moonshot) also need a 'reasoning_content' field; handle both here.
request_logger = getattr(agent, "logger", None) or logging.getLogger(__name__)
# Per-agent validation cursor skips re-parsing tool_call args already validated.
# Identity-keyed; a rewritten list breaks the prefix match and forces a re-scan.
_sanitize_cursor = getattr(agent, "_sanitize_args_cursor", None)
if _sanitize_cursor is None:
_sanitize_cursor = {}
try:
agent._sanitize_args_cursor = _sanitize_cursor
except Exception:
pass
repaired_tool_calls = agent._sanitize_tool_call_arguments(
messages,
logger=request_logger,
session_id=agent.session_id,
cursor=_sanitize_cursor,
)
if repaired_tool_calls > 0:
request_logger.info(
"Sanitized %s corrupted tool_call arguments before request (session=%s)",
repaired_tool_calls,
agent.session_id or "-",
)
# Drop legacy hidden assistant placeholders carrying the raw interrupt scaffold
# before repair: replayed, the model echoes/self-replicates (#81841).
messages = [
msg for msg in messages
if not (
msg.get("display_kind") == "hidden"
and msg.get("role") == "assistant"
and (
(
isinstance(msg.get("content"), str)
and msg["content"].strip() == _INTERRUPT_SCAFFOLD_MARKER
)
or (
isinstance(msg.get("api_content"), str)
and msg["api_content"].strip() == _INTERRUPT_SCAFFOLD_MARKER
)
)
)
]
# Repair malformed role alternation (tool→user / user→user tails): providers
# return empty content on them and the empty-retry loop spins. The _with_cursor
# variant also recomputes the SessionDB flush cursor after compaction (#44837).
from agent.agent_runtime_helpers import (
fill_empty_non_final_wire_payload,
repair_message_sequence_with_cursor,
)
repaired_seq = repair_message_sequence_with_cursor(agent, messages)
if repaired_seq > 0:
request_logger.info(
"Repaired %s message-alternation violations before request (session=%s)",
repaired_seq,
agent.session_id or "-",
)
api_messages = []
for idx, msg in enumerate(messages):
# Structural clone, NOT msg.copy(): in-place transforms below must not reach
# persisted history via nested containers; see _clone_message_for_send.
api_msg = _clone_message_for_send(msg)
# api_content is the persistence sidecar of the exact bytes sent to the API;
# bookkeeping, never a provider field — pop it from EVERY outgoing copy.
_api_content = api_msg.pop("api_content", None)
# Display-only timeline metadata, never a provider field: strict OpenAI
# backends reject unknown keys once a typed event row enters live history.
api_msg.pop("display_kind", None)
api_msg.pop("display_metadata", None)
# Durable row id from _rows_to_conversation (desktop reactions); only the
# chat-completions transport strips underscore keys, so drop it centrally.
api_msg.pop("_row_id", None)
# Inject ephemeral context (memory prefetch + pre_llm_call user hooks)
# at API time only; `messages` is untouched beyond the api_content stamp.
if idx == current_turn_user_idx and msg.get("role") == "user":
if isinstance(_api_content, str) and _api_content:
# Reuse the prologue's stamp so sidecar and wire cannot drift
# and every pass this turn sends identical bytes.
api_msg["content"] = _api_content
else:
# Callers that bypass the prologue stamping: compose live.
_composed = compose_user_api_content(
api_msg.get("content", ""),
_ext_prefetch_cache,
_plugin_user_context,
)
if _composed is not None:
api_msg["content"] = _composed
elif (
isinstance(_api_content, str)
and _api_content
and msg.get("role") in ("user", "assistant")
):
# Historical row: replay the exact bytes sent live so the prompt-cache
# prefix stays byte-stable. User rows carry the injection sidecar; user
# and assistant rows may carry a sanitize-divergence sidecar.
api_msg["content"] = _api_content
# For ALL assistant messages, pass reasoning back to the API
# This ensures multi-turn reasoning context is preserved
agent._copy_reasoning_content_for_api(msg, api_msg)
# Remove 'reasoning' field - it's for trajectory storage only
# We've copied it to 'reasoning_content' for the API above
if "reasoning" in api_msg:
api_msg.pop("reasoning")
# Remove finish_reason - not accepted by strict APIs (e.g. Mistral)
if "finish_reason" in api_msg:
api_msg.pop("finish_reason")
# Fill empty non-final user/assistant wire copies so the pre-call sanitizer
# stops re-healing and flooding errors.log; durable history is untouched.
# After the reasoning copy so thinking-only turns keep payload (#96870).
fill_empty_non_final_wire_payload(
api_msg, is_final=(idx == len(messages) - 1)
)
# _thinking_prefill survives intentionally: the drop pass below needs it.
# Strip length-continuation marks; some transports keep underscore keys.
api_msg.pop("_length_continuation_fragment", None)
api_msg.pop("_length_continuation_nudge", None)
# Strip Codex Responses fields (call_id, response_item_id): strict providers
# reject unknown fields. New dicts keep the internal list intact for Codex.
if agent._should_sanitize_tool_calls():
# In MoA mode agent.model is the virtual preset name; use the resolved
# aggregator so Gemini keeps thought_signature (extra_content).
_sanitize_model = agent.model
if agent.provider == "moa":
if moa_config:
_agg = moa_config.get("aggregator") or {}
if _agg.get("model"):
_sanitize_model = _agg["model"]
if _sanitize_model == agent.model:
# Virtual-provider mode: no moa_config is threaded through; ask
# the facade for the aggregator slot from the previous create().
_moa_client = getattr(agent, "client", None)
_agg_slot = getattr(_moa_client, "last_aggregator_slot", None)
if _agg_slot and _agg_slot.get("model"):
_sanitize_model = _agg_slot["model"]
agent._sanitize_tool_calls_for_strict_api(api_msg, model=_sanitize_model)
# Keep 'reasoning_details' - OpenRouter uses this for multi-turn reasoning context
# The signature field helps maintain reasoning continuity
api_messages.append(api_msg)
# Final system message = cached prompt + ephemeral additions (API-time only).
# Plugin/recall context goes into the user message, never the system prompt: the
# prompt is built ONCE per session and replayed verbatim (stable cache prefix).
effective_system = active_system_prompt or ""
if agent.ephemeral_system_prompt:
effective_system = (effective_system + "\n\n" + agent.ephemeral_system_prompt).strip()
if effective_system:
api_messages = [{"role": "system", "content": effective_system}] + api_messages
if moa_config:
try:
from agent.message_content import flatten_message_text as _flatten_mt
from agent.moa_loop import _preset_temperature, aggregate_moa_context
_moa_context = aggregate_moa_context(
user_prompt=(
original_user_message
if isinstance(original_user_message, str)
# Multimodal content list: extract visible text rather than
# str()-ing parts, which would leak base64 image payloads.
else _flatten_mt(original_user_message)
),
api_messages=api_messages,
reference_models=moa_config.get("reference_models") or [],
aggregator=moa_config.get("aggregator") or {},
temperature=_preset_temperature(moa_config, "reference_temperature"),
aggregator_temperature=_preset_temperature(moa_config, "aggregator_temperature"),
reference_max_tokens=moa_config.get("reference_max_tokens"),
# None = no per-preset override; inherit
# auxiliary.moa_reference.timeout via call_llm.
reference_timeout=(
float(moa_config["reference_timeout"])
if moa_config.get("reference_timeout")
else None
),
degraded_reference_policy=str(
moa_config.get("degraded_reference_policy") or "loud"
),
agent=agent,
)
if _moa_context:
for _msg in reversed(api_messages):
if _msg.get("role") == "user":
_base = _msg.get("content", "")
if isinstance(_base, str):
_msg["content"] = _base + "\n\n" + _moa_context
elif isinstance(_base, list):
# Multimodal turn: append MoA context as a trailing text
# part instead of silently dropping it.
_msg["content"] = [
*_base,
{"type": "text", "text": "\n\n" + _moa_context},
]
break
except Exception as _moa_exc:
logger.warning("MoA context aggregation failed: %s", _moa_exc)
# Inject ephemeral prefill messages right after the system prompt
# but before conversation history. Same API-call-time-only pattern.
if agent.prefill_messages:
sys_offset = 1 if (api_messages and api_messages[0].get("role") == "system") else 0
for idx, pfm in enumerate(agent.prefill_messages):
# Structural clone: the in-place sanitizers below must not write
# through into agent.prefill_messages' nested containers.
api_messages.insert(sys_offset + idx, _clone_message_for_send(pfm))
# Per-turn context selection hook: an engine may select/replace context for THIS
# call only — request-only, fail-open, and independent of should_compress().
_sel_incoming = (
messages[current_turn_user_idx]
if 0 <= current_turn_user_idx < len(messages)
else None
)
api_messages = _apply_context_engine_selection(
agent,
api_messages,
messages,
_sel_incoming,
logger=request_logger,
)
# Runs unconditionally (not gated on context_compressor) so orphaned tool
# results from session loading or manual message edits are always caught.
api_messages = agent._sanitize_api_messages(api_messages)
# One-time repeated-heal notice goes out via the status/warning callback, NEVER
# appended to messages: the cached prompt prefix stays byte-identical (#96870).
try:
from agent.agent_runtime_helpers import (
consume_pending_sanitizer_heal_notice,
)
_heal_notice = consume_pending_sanitizer_heal_notice()
if _heal_notice:
agent._emit_warning(_heal_notice)
except Exception:
# A notice hiccup must never break the send path.
logger.debug("sanitizer heal notice delivery failed", exc_info=True)
# Drop thinking-only assistant turns + merge adjacent users, API copy only:
# Anthropic-style backends 400 on a trailing `thinking` block; history keeps it.
api_messages = agent._drop_thinking_only_and_merge_users(
api_messages,
drop_codex_reasoning_items=agent.api_mode != "codex_responses",
)
# Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns
# (KV-cache reuse on local servers, better cloud cache hits); API copy only.
for am in api_messages:
if isinstance(am.get("content"), str):
am["content"] = am["content"].strip()
_canonicalize_api_tool_calls(api_messages)
# Strip lone surrogates (U+D800-U+DFFF) that some Ollama-served models emit;
# they crash json.dumps() inside the OpenAI SDK and trigger the 3-retry cycle.
_sanitize_messages_surrogates(api_messages)
# No send-time pad loop here: ``repair_empty_non_final_messages`` (inside
# ``_sanitize_api_messages``) is the single owner of empty-turn repair, and its
# non-whitespace placeholder survives normalization regardless of ordering.
# Build the request-local cache sections LAST, after every transcript mutation;
# the canonical tool registry stays undecorated. Marked ``content`` becomes text
# blocks the whitespace pass skips, so the same row's bytes vary across turns.
tools_for_api = agent.tools
if agent._use_prompt_caching and agent.provider != "moa":
from agent.prompt_caching import (
envelope_tool_part_cache_markers_supported,
)
_static_system_prefix = getattr(agent, "_cached_system_prompt_static", None)
_initial_cache_plan = build_prompt_cache_plan(
api_messages,
tools_for_api,
# Clamp per-destination: a configured 1h regresses to 5m on
# Qwen/Alibaba routes, whose context cache is 5m-only (#84733).
cache_ttl=effective_cache_ttl(
agent._cache_ttl,
provider=agent.provider,
model=agent.model,
),
native_anthropic=agent._use_native_cache_layout,
static_system_prefix=(
_static_system_prefix
if isinstance(_static_system_prefix, str)
else None
),
direct_native_tool_cache=agent._direct_native_anthropic_tool_cache_capability(),
# LiteLLM-style envelope routes forward part-level markers into
# tool_result.content[] → non-retryable 400 (#89886).
tool_part_markers=envelope_tool_part_cache_markers_supported(
getattr(agent, "provider", ""), getattr(agent, "base_url", "")
),
)
api_messages = _initial_cache_plan.messages
tools_for_api = _initial_cache_plan.tools
# Prepare the persistent-MoA request before measuring compression pressure: the
# ephemeral advisor output is absent from ``messages``; ``create()`` reuses the
# prepared request instead of running the advisors again.
_moa_prepared_request = None
if agent.provider == "moa":
_moa_completions = getattr(getattr(agent.client, "chat", None), "completions", None)
if pending_moa_prepared_request is not None:
_rebase_moa_request = getattr(_moa_completions, "rebase_prepared_request", None)
if callable(_rebase_moa_request):
_moa_prepared_request = _rebase_moa_request(
pending_moa_prepared_request, api_messages
)
pending_moa_prepared_request = None
if _moa_prepared_request is None:
_prepare_moa_request = getattr(_moa_completions, "prepare", None)
if callable(_prepare_moa_request):
_moa_prepared_request = _prepare_moa_request(api_messages)
if _moa_prepared_request is not None:
api_messages = _moa_prepared_request["messages"]
# One image-stripped estimate feeds both figures; tools counted separately (50+
# tools ≈ 20-30K tokens); total_chars is a rough proxy for logs/hooks only.
# Charge stale thinking only when the active route replays it (#84371).
from agent.turn_context import _agent_stale_thinking_on_wire
if _agent_stale_thinking_on_wire(agent):
approx_tokens = estimate_messages_tokens_rough(api_messages)
else:
approx_tokens = estimate_messages_tokens_rough(
api_messages, charge_stale_thinking=False
)
# Route-aware: native Responses compaction prunes the wire payload, so the raw
# history figure overstates it and fires needless local compression (#96995).
request_pressure_tokens = _midturn_request_pressure_tokens(
agent, api_messages, effective_system or "", approx_tokens
)
# Usage-anchored override: real prompt_tokens (incl. system + tool schemas) +
# delta estimate replaces the whole-history heuristic when the anchor is fresh.
_anchored_pressure = anchored_context_tokens(
messages, getattr(agent, "_usage_anchor", None)
)
if _anchored_pressure is not None:
request_pressure_tokens = _anchored_pressure
total_chars = approx_tokens * 4
# Stash the rough estimate so update_from_response() can pair it with the real
# count (should_defer_preflight_to_real_usage). getattr: test doubles lack it.
_note_rough = getattr(
agent.context_compressor, "note_request_rough_estimate", None
)
if callable(_note_rough):
_note_rough(request_pressure_tokens)
_runtime_context_error = _ollama_context_limit_error(
agent, request_pressure_tokens
)
if _runtime_context_error:
final_response = _runtime_context_error
failed = True
_turn_exit_reason = "ollama_runtime_context_too_small"
append_message(messages, {"role": "assistant", "content": final_response})
agent._emit_status("❌ Ollama runtime context is too small for Hermes tool use")
api_call_count -= 1
agent._api_call_count = api_call_count
try:
agent.iteration_budget.refund()
except Exception:
pass
break
# Pre-API pressure check: tool results grow a turn and last_prompt_tokens lags
# them. Mirror the turn-prologue guard chain: defer on noisy estimate, skip in
# failure cooldown, then should_compress() (#11529).
_compressor = agent.context_compressor
_preflight_threshold = int(
getattr(_compressor, "threshold_tokens", 0) or 0
)
_provider_overflow_preflight = (
_provider_overflow_recovery_pending
and (
_preflight_threshold <= 0
or request_pressure_tokens >= _preflight_threshold
)
)
if (
_provider_overflow_recovery_pending
and not _provider_overflow_preflight
):
# The outer-loop rebuild includes system prompt, request-only injections and
# tool schemas; only that full request with output runway may be sent.
_provider_overflow_recovery_pending = False
# Compare fully assembled requests, not raw ``messages`` (which omit
# api_content, plugin injections, prefills, MoA context, ephemeral system text).
_previous_preflight_pressure = _last_preflight_pressure
_last_preflight_pressure = None
if (
_previous_preflight_pressure is not None
and request_pressure_tokens >= _preflight_threshold
and not _compression_warrants_another_preflight_pass(
_previous_preflight_pressure,
request_pressure_tokens,
_preflight_threshold,
)
):
# Stop proactive retries this turn without consuming the shared overflow-
# recovery budget; the provider's error handler may still compact.
_preflight_compression_blocked = True
logger.warning(
"Pre-API compression made insufficient progress: ~%s -> "
"~%s request tokens; skipping additional preflight passes",
f"{_previous_preflight_pressure:,}",
f"{request_pressure_tokens:,}",
)
_defer_preflight = getattr(
_compressor, "should_defer_preflight_to_real_usage", lambda _t: False
)
_compression_cooldown = getattr(
_compressor, "get_active_compression_failure_cooldown", lambda: None
)()
if (
agent.compression_enabled
and not _review_fork_first_request_pending(agent)
and len(messages) > 1
and compression_attempts < max_compression_attempts
and (
not _preflight_compression_blocked
or _provider_overflow_preflight
)
and (
not _defer_preflight(request_pressure_tokens)
or _provider_overflow_preflight
)
and not _compression_cooldown
and _compressor.should_compress(request_pressure_tokens)
):
# Managed local runtime: grow the context window before compressing (last
# resort). Only for a llamacpp provider at the supervised base_url.
_grown_window = _maybe_grow_local_window(
agent, _compressor, request_pressure_tokens
)
if _grown_window:
# Bigger window granted: recalibrate the compressor and skip compression
# this pass.
_compressor.update_model(
agent.model,
_grown_window,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown_window // 1024}K "
f"(local model; conversation continues uncompressed)"
)
# Never reached the provider — refund the call/budget like the
# compression path does before its continue.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
continue
if _moa_prepared_request is not None:
pending_moa_prepared_request = _moa_prepared_request
compression_attempts += 1
# Compression is running: reset the blocked-overflow warning dedup so a
# later blocked turn warns again (#62625). getattr: test doubles lack it.
_clear_warn = getattr(agent, "_clear_context_overflow_warn", None)
if callable(_clear_warn):
_clear_warn()
logger.info(
"Pre-API compression: ~%s request tokens >= %s threshold "
"(context=%s, attempt=%s/%s)",
f"{request_pressure_tokens:,}",
f"{int(getattr(_compressor, 'threshold_tokens', 0) or 0):,}",
f"{int(getattr(_compressor, 'context_length', 0) or 0):,}"
if getattr(_compressor, "context_length", 0) else "unknown",
compression_attempts,
max_compression_attempts,
)
_pre_api_status = automatic_compaction_status_message(
_compressor,
phase="pre_api",
default_message=PRE_API_COMPRESSION_STATUS_TEMPLATE.format(
tokens=request_pressure_tokens
),
approx_tokens=request_pressure_tokens,
threshold_tokens=int(
getattr(_compressor, "threshold_tokens", 0) or 0
),
context_length=int(
getattr(_compressor, "context_length", 0) or 0
),
model=agent.model,
attempt=compression_attempts,
max_attempts=max_compression_attempts,
)
if _pre_api_status:
agent._emit_status(_pre_api_status)
_last_preflight_pressure = request_pressure_tokens
_pre_api_input = messages
messages, active_system_prompt = agent._compress_context(
messages,
system_message,
approx_tokens=request_pressure_tokens,
task_id=effective_task_id,
)
if context_compression_timed_out(agent):
# Progress-aware timeout (#98722): never reached the provider — refund
# the call/budget and stop; an overflow retry would only re-compress.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
failed = True
_compression_timeout_exhausted = True
_turn_exit_reason = "context_compression_timeout"
break
if messages is _pre_api_input and (
compression_skipped_due_to_lock(agent)
or compression_blocked_transiently(agent)
):
# Temporary DEFER (lock held / cooldown), not evidence about
# compressibility: refund the attempt, leave the progress blocker
# unarmed and proceed (#69870, #97488).
compression_attempts -= 1
_last_preflight_pressure = None
if pending_moa_prepared_request is _moa_prepared_request:
pending_moa_prepared_request = None
else:
# Reset retry/empty-response state so the compacted request gets a fresh
# chance.
agent._empty_content_retries = 0
agent._thinking_prefill_retries = 0
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
agent._mute_post_response = False
# Re-baseline the flush cursor: rotation returns None (child flushes
# whole); in-place returns list(messages) — None would re-append
# persisted rows. See conversation_history_after_compression().
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
# Never reaches the provider on skip or re-run — refund the call/budget
# in BOTH cases, else budget leaks and api_call_count over-reports.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
if _should_skip_model_call_for_reference_handoff(
messages, user_message
):
# Reference-only handoff must not become the active turn
# after a completed assistant response (#80622).
logger.info(
"Skipping post-compaction model call: reference-only "
"handoff would be the sole active user turn (#80622)"
)
if not final_response:
final_response = _HANDOFF_SKIP_FINAL_RESPONSE
_turn_exit_reason = "compaction_handoff_not_actionable"
break
continue
elif _provider_overflow_preflight and _compression_cooldown:
# Provider proved the request cannot fit and the compressor is unavailable:
# don't resend; let the next user turn retry after cooldown.
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent,
messages,
api_call_count,
reason="transient_block",
)
elif (
_provider_overflow_preflight
and compression_attempts >= max_compression_attempts
):
# All recovery passes consumed and still over threshold: fail closed —
# llama.cpp may silently truncate an oversized retry.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
elif (
agent.compression_enabled
and len(messages) > 1
and compression_attempts < max_compression_attempts
and not _defer_preflight(request_pressure_tokens)
and _compression_cooldown
):
# Summary-LLM cooldown blocks compression: deduped warning only when over
# threshold (should_compress_info reason is None below it) (#62625).
_block_reason = None
try:
_block_reason = _compressor.should_compress_info(
request_pressure_tokens
)[1]
except Exception:
_block_reason = None
if _block_reason:
agent._warn_context_overflow_blocked(
_block_reason,
request_pressure_tokens,
int(getattr(_compressor, "threshold_tokens", 0) or 0),
)
elif not agent.compression_enabled and len(messages) > 1:
# Uncompressed session guard (#89297): compression is disabled, so warn
# (deduped) when the request exceeds the context window; the turn-context
# preflight re-arms the dedup.
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if (
isinstance(_ctx_len, int)
and _ctx_len > 0
and request_pressure_tokens > _ctx_len
):
_warn_fn = getattr(
agent, "_warn_uncompressed_context_overflow", None
)
if callable(_warn_fn):
_warn_fn(request_pressure_tokens, _ctx_len)
if _provider_overflow_preflight:
# Any other gate blocking the forced preflight (e.g. uncompressible one-
# message request) must fail closed: the request is proven not to fit.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
# Thinking spinner for quiet mode (animated during API call)
thinking_spinner = None
if not agent.quiet_mode:
agent._vprint(f"\n{agent.log_prefix}🔄 Making API call #{api_call_count}/{agent.max_iterations}...")
agent._vprint(f"{agent.log_prefix} 📊 Request size: {len(api_messages)} messages, ~{approx_tokens:,} tokens (~{total_chars:,} chars)")
agent._vprint(f"{agent.log_prefix} 🔧 Available tools: {len(agent.tools) if agent.tools else 0}")
else:
# Animated thinking spinner in quiet mode
face = random.choice(KawaiiSpinner.get_thinking_faces())
verb = random.choice(KawaiiSpinner.get_thinking_verbs())
if agent.thinking_callback:
# CLI TUI mode: use prompt_toolkit widget instead of raw spinner
# (works in both streaming and non-streaming modes)
agent.thinking_callback(f"{face} {verb}...")
elif not agent._has_stream_consumers() and agent._should_start_quiet_spinner():
# Raw KawaiiSpinner only when no streaming consumers and the
# spinner output has a safe sink.
spinner_type = random.choice(['brain', 'sparkle', 'pulse', 'moon', 'star'])
thinking_spinner = KawaiiSpinner(f"{face} {verb}...", spinner_type=spinner_type, print_fn=agent._print_fn)
thinking_spinner.start()
# Log request details if verbose
if agent.verbose_logging:
logging.debug(f"API Request - Model: {agent.model}, Messages: {len(messages)}, Tools: {len(agent.tools) if agent.tools else 0}")
logging.debug(f"Last message role: {messages[-1]['role'] if messages else 'none'}")
logging.debug(f"Total message size: ~{approx_tokens:,} tokens")
api_start_time = time.time()
retry_count = 0
max_retries = agent._api_max_retries
_retry = TurnRetryState()
finish_reason = "stop"
response = None # Guard against UnboundLocalError if all retries fail
api_kwargs = None # Guard against UnboundLocalError in except handler
api_request_id = f"{turn_id}:api:{api_call_count}"
agent._current_api_request_id = api_request_id
while retry_count < max_retries:
# ── Nous Portal rate limit guard ──────────────────────
# Skip the call if another session recorded a rate limit: every attempt
# (incl. SDK retries) counts against RPH.
if agent.provider == "nous":
try:
from agent.nous_rate_guard import (
nous_rate_limit_remaining,
format_remaining as _fmt_nous_remaining,
)
_nous_remaining = nous_rate_limit_remaining()
if _nous_remaining is not None and _nous_remaining > 0:
_nous_msg = (
f"Nous Portal rate limit active — "
f"resets in {_fmt_nous_remaining(_nous_remaining)}."
)
agent._buffer_vprint(
f"⏳ {_nous_msg} Trying fallback..."
)
agent._buffer_status(f"⏳ {_nous_msg}")
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# No fallback available — surface buffered context
# so user sees the rate-limit message that led here.
agent._flush_status_buffer()
agent._persist_session(messages, conversation_history)
return {
"final_response": (
f"⏳ {_nous_msg}\n\n"
"No fallback provider available. "
"Try again after the reset, or add a "
"fallback provider in config.yaml."
),
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": _nous_msg,
}
except ImportError:
pass
except Exception:
pass # Never let rate guard break the agent loop
try:
agent._reset_stream_delivery_tracking()
# Per-attempt first-chunk timestamp so a stale value never leaks into
# post_api_request.
agent._last_api_first_chunk_at = None
# api_messages was built for the primary; a fallback (DeepSeek / Kimi /
# MiMo) may require reasoning_content. Re-apply the echo-back pad
# (idempotent).
agent._reapply_reasoning_echo_for_provider(api_messages)
# Same for prompt-cache decoration (#72626): strip the primary's
# breakpoints and re-render for the current provider.
api_messages, _moa_prepared_request, tools_for_api = (
_redecorate_prompt_cache_for_provider(
agent,
api_messages,
system_message=system_message,
moa_prepared=_moa_prepared_request,
tools_for_api=tools_for_api,
)
)
if tools_for_api == agent.tools:
api_kwargs = agent._build_api_kwargs(api_messages)
else:
api_kwargs = agent._build_api_kwargs(
api_messages,
tools_for_api=tools_for_api,
)
# Surrogate chokepoint (#50959): tool descriptions, extra_body and
# kwargs strings can carry invalid code points (HTTP 400). One walk
# makes the payload json.dumps()-safe.
_sanitize_structure_surrogates(api_kwargs)
if agent._force_ascii_payload:
_sanitize_structure_non_ascii(api_kwargs)
if agent.api_mode == "codex_responses":
api_kwargs = agent._get_transport().preflight_kwargs(
api_kwargs,
allow_stream=False,
is_github_responses=agent._is_copilot_url(),
sanitize_harmony_tokens=agent._is_codex_backend(),
)
# OpenRouter caching replays identical responses, even empty ones; an
# empty-response retry must bypass the cache.
if agent._empty_content_retries > 0 and agent._is_openrouter_url():
_xh = dict(api_kwargs.get("extra_headers") or {})
_xh["X-OpenRouter-Cache"] = "false"
api_kwargs["extra_headers"] = _xh
# Copilot x-initiator: first call of a user turn is "user" (billed
# premium); tool-loop follow-ups keep the default "agent" (#3040).
if getattr(agent, "_is_user_initiated_turn", False) and agent._is_copilot_url():
_xh = dict(api_kwargs.get("extra_headers") or {})
_xh["x-initiator"] = "user"
api_kwargs["extra_headers"] = _xh
agent._is_user_initiated_turn = False
try:
from hermes_cli.middleware import apply_llm_request_middleware
_llm_request_mw = apply_llm_request_middleware(
api_kwargs,
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
session_id=agent.session_id or "",
platform=agent.platform or "",
model=agent.model,
provider=agent.provider,
base_url=agent.base_url,
api_mode=agent.api_mode,
api_call_count=api_call_count,
)
api_kwargs = _llm_request_mw.payload
_original_api_kwargs = _llm_request_mw.original_payload
_llm_middleware_trace = _llm_request_mw.trace
except Exception:
_original_api_kwargs = dict(api_kwargs)
_llm_middleware_trace = []
try:
from hermes_cli.lifecycle import (
has_hook,
invoke_hook as _invoke_hook,
)
if has_hook("pre_api_request"):
request_messages = api_kwargs.get("messages")
if not isinstance(request_messages, list):
request_messages = api_kwargs.get("input")
if not isinstance(request_messages, list):
request_messages = api_messages
# Shallow copy: plugins may retain the list; deepcopy is costly.
# ``request_messages``/``conversation_history`` are raw langfuse
# passthroughs.
_request_payload = agent._api_request_payload_for_hook(api_kwargs)
# Anthropic (``system``) and Responses/Codex (``instructions``)
# move the system prompt out of messages; pass it for
# observability.
system_prompt_for_hooks = _system_prompt_for_hooks(
api_kwargs, request_messages
)
_invoke_hook(
"pre_api_request",
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
session_id=agent.session_id or "",
user_message=original_user_message,
conversation_history=list(messages),
platform=agent.platform or "",
model=agent.model,
provider=agent.provider,
base_url=agent.base_url,
api_mode=agent.api_mode,
api_call_count=api_call_count,
retry_count=retry_count,
request_messages=list(request_messages)
if isinstance(request_messages, list)
else [],
system_prompt=system_prompt_for_hooks,
message_count=len(api_messages),
tool_count=len(agent.tools or []),
approx_input_tokens=approx_tokens,
request_char_count=total_chars,
max_tokens=agent.max_tokens,
started_at=api_start_time,
middleware_trace=list(_llm_middleware_trace),
request=_request_payload,
)
except Exception:
pass
if env_var_enabled("HERMES_DUMP_REQUESTS"):
agent._dump_api_request_debug(api_kwargs, reason="preflight")
# Private to the in-process MoA facade; add after middleware/hooks/debug
# dumps so none serializes it into the provider payload.
if _moa_prepared_request is not None and agent.provider == "moa":
# Re-read the live client: rotation/fallback/cleanup rebuild
# agent.client between attempts; a native OpenAI client rejects this
# key (TypeError).
if _moa_client_consumes_prepared_request(agent.client):
api_kwargs["_moa_prepared_request"] = _moa_prepared_request
else:
logger.warning(
"MoA client replaced mid-turn (client=%s); sending the "
"prepared prompt without the MoA handshake",
type(agent.client).__name__,
)
# Always prefer streaming even without consumers: it gives stale-
# stream/read-timeout health checks that quiet callers otherwise lack.
# Falls back if unsupported.
def _stop_spinner():
nonlocal thinking_spinner
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
_use_streaming = True
# Provider signaled "stream not supported": stay non-streaming for the
# session.
if getattr(agent, "_disable_streaming", False):
_use_streaming = False
# ACP clients (`acp://` scheme, any vendor) return a plain
# SimpleNamespace, not a stream; mirrors the Responses API exclusion.
elif (
agent.provider in {"copilot-acp"}
or str(agent.base_url or "").lower().startswith("acp://")
or str(agent.base_url or "").lower().startswith("acp+tcp://")
):
_use_streaming = False
# MoA streams only with a display/TTS consumer
# (MoAChatCompletions.create() honors stream=True); else complete-
# response path.
elif agent.provider == "moa" and not agent._has_stream_consumers():
_use_streaming = False
elif not agent._has_stream_consumers():
# No consumer: still stream for health checking, except Mock clients
# in tests (SimpleNamespace, not stream iterators).
from unittest.mock import Mock
if isinstance(getattr(agent, "client", None), Mock):
_use_streaming = False
def _perform_api_call(next_api_kwargs):
if agent.api_mode == "codex_responses":
next_api_kwargs = agent._get_transport().preflight_kwargs(
next_api_kwargs,
allow_stream=False,
is_github_responses=agent._is_copilot_url(),
sanitize_harmony_tokens=agent._is_codex_backend(),
)
if _use_streaming:
return agent._interruptible_streaming_api_call(
next_api_kwargs, on_first_delta=_stop_spinner
)
from agent import relay_llm
return relay_llm.execute(
next_api_kwargs,
agent._interruptible_api_call,
session_id=str(agent.session_id or ""),
name=str(agent.provider or "provider"),
model_name=str(agent.model or ""),
metadata={
"api_mode": agent.api_mode,
"api_request_id": api_request_id,
"call_role": (
"delegated"
if getattr(agent, "is_subagent", False)
else "fallback"
if int(getattr(agent, "_fallback_index", 0) or 0) > 0
else "primary"
),
"retry_count": retry_count,
},
defer_logical_completion=True,
)
from hermes_cli.middleware import run_llm_execution_middleware
_model_request_active = getattr(agent, "_model_request_active", None)
_redirect_lock = getattr(agent, "_pending_redirect_lock", None)
if _redirect_lock is not None:
with _redirect_lock:
if _model_request_active is not None:
_model_request_active.set()
elif _model_request_active is not None:
_model_request_active.set()
_redirect_crossed_response = False
try:
response = run_llm_execution_middleware(
api_kwargs,
_perform_api_call,
original_request=_original_api_kwargs,
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
session_id=agent.session_id or "",
platform=agent.platform or "",
model=agent.model,
provider=agent.provider,
base_url=agent.base_url,
api_mode=agent.api_mode,
api_call_count=api_call_count,
middleware_trace=list(_llm_middleware_trace),
)
finally:
if _redirect_lock is not None:
with _redirect_lock:
if _model_request_active is not None:
_model_request_active.clear()
_redirect_crossed_response = bool(
agent._pending_redirect
)
else:
if _model_request_active is not None:
_model_request_active.clear()
_redirect_crossed_response = agent._has_pending_redirect()
if _redirect_crossed_response:
# Response and redirect can cross threads: discard the now-stale
# response and rebuild from the correction rather than lose it.
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
if agent.clear_interrupt(preserve_redirect=True):
_retry.restart_with_redirected_messages = True
else:
interrupted = True
break
api_duration = time.time() - api_start_time
# Stop thinking spinner silently -- the response box or tool
# execution messages that follow are more informative.
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
if not agent.quiet_mode:
agent._vprint(f"{agent.log_prefix}⏱️ API call completed in {api_duration:.2f}s")
if agent.verbose_logging:
# Log response with provider info if available
resp_model = getattr(response, 'model', 'N/A') if response else 'N/A'
logging.debug(f"API Response received - Model: {resp_model}, Usage: {response.usage if hasattr(response, 'usage') else 'N/A'}")
# Validate response shape before proceeding
response_invalid = False
error_details = []
if agent.api_mode == "codex_responses":
_ct_v = agent._get_transport()
if not _ct_v.validate_response(response):
if response is None:
response_invalid = True
error_details.append("response is None")
else:
# Terminal provider failure (e.g. quota exhaustion): treat
# as invalid so the fallback chain triggers.
_codex_resp_status = str(getattr(response, "status", "") or "").strip().lower()
if _codex_resp_status in {"failed", "cancelled"}:
_codex_error_obj = getattr(response, "error", None)
_codex_error_msg = (
_codex_error_obj.get("message") if isinstance(_codex_error_obj, dict)
else str(_codex_error_obj) if _codex_error_obj
else f"Responses API returned status '{_codex_resp_status}'"
)
logger.warning(
"Codex response status='%s' (error=%s). Routing to fallback. %s",
_codex_resp_status, _codex_error_msg,
agent._client_log_context(),
)
response_invalid = True
error_details.append(f"response.status={_codex_resp_status}: {_codex_error_msg}")
else:
# output_text fallback: stream backfill may have failed
# but normalize can still recover from output_text
_out_text = getattr(response, "output_text", None)
_out_text_stripped = _out_text.strip() if isinstance(_out_text, str) else ""
if _out_text_stripped:
logger.debug(
"Codex response.output is empty but output_text is present "
"(%d chars); deferring to normalization.",
len(_out_text_stripped),
)
else:
_resp_status = getattr(response, "status", None)
_resp_incomplete = getattr(response, "incomplete_details", None)
logger.warning(
"Codex response.output is empty after stream backfill "
"(status=%s, incomplete_details=%s, model=%s). %s",
_resp_status, _resp_incomplete,
getattr(response, "model", None),
f"api_mode={agent.api_mode} provider={agent.provider}",
)
response_invalid = True
error_details.append("response.output is empty")
elif agent.api_mode == "anthropic_messages":
_tv = agent._get_transport()
if not _tv.validate_response(response):
response_invalid = True
if response is None:
error_details.append("response is None")
else:
error_details.append("response.content invalid (not a non-empty list)")
elif agent.api_mode == "bedrock_converse":
_btv = agent._get_transport()
if not _btv.validate_response(response):
response_invalid = True
if response is None:
error_details.append("response is None")
else:
error_details.append("Bedrock response invalid (no output or choices)")
else:
_ctv = agent._get_transport()
if not _ctv.validate_response(response):
response_invalid = True
if response is None:
error_details.append("response is None")
elif not hasattr(response, 'choices'):
error_details.append("response has no 'choices' attribute")
elif response.choices is None:
error_details.append("response.choices is None")
else:
error_details.append("response.choices is empty")
if response_invalid:
agent._invoke_api_request_error_hook(
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
api_call_count=api_call_count,
api_start_time=api_start_time,
api_kwargs=api_kwargs,
error_type="InvalidAPIResponse",
error_message=", ".join(error_details) or "Invalid API response",
status_code=getattr(getattr(response, "error", None), "code", None),
retry_count=retry_count,
max_retries=max_retries,
retryable=True,
reason="invalid_response",
)
# Stop spinner silently — retry status is now buffered
# and only surfaced if every retry+fallback exhausts.
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
# Invalid response — could be rate limiting, provider timeout,
# upstream server error, or malformed response.
retry_count += 1
# Eager fallback: empty/malformed responses often mean rate limiting
# — switch now instead of extended backoff.
if agent._fallback_index < len(agent._fallback_chain):
agent._buffer_status("⚠️ Empty/malformed response — switching to fallback...")
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# Check for error field in response (some providers include this)
error_msg = "Unknown"
provider_name = "Unknown"
if response and hasattr(response, 'error') and response.error:
error_msg = str(response.error)
# Try to extract provider from error metadata
if hasattr(response.error, 'metadata') and response.error.metadata:
provider_name = response.error.metadata.get('provider_name', 'Unknown')
elif response and hasattr(response, 'message') and response.message:
error_msg = str(response.message)
# Try to get provider from model field (OpenRouter often returns actual model used)
if provider_name == "Unknown" and response and hasattr(response, 'model') and response.model:
provider_name = f"model={response.model}"
# Check for x-openrouter-provider or similar metadata
if provider_name == "Unknown" and response:
# Log all response attributes for debugging
resp_attrs = {k: str(v)[:100] for k, v in vars(response).items() if not k.startswith('_')}
if agent.verbose_logging:
logging.debug(f"Response attributes for invalid response: {resp_attrs}")
# Extract error code from response for contextual diagnostics
_resp_error_code = None
if response and hasattr(response, 'error') and response.error:
_code_raw = getattr(response.error, 'code', None)
if _code_raw is None and isinstance(response.error, dict):
_code_raw = response.error.get('code')
if _code_raw is not None:
try:
_resp_error_code = int(_code_raw)
except (TypeError, ValueError):
pass
# Build a human-readable failure hint from the error code
# and response time, instead of always assuming rate limiting.
if _resp_error_code == 524:
_failure_hint = f"upstream provider timed out (Cloudflare 524, {api_duration:.0f}s)"
elif _resp_error_code == 504:
_failure_hint = f"upstream gateway timeout (504, {api_duration:.0f}s)"
elif _resp_error_code == 429:
_failure_hint = "rate limited by upstream provider (429)"
elif _resp_error_code in {500, 502}:
_failure_hint = f"upstream server error ({_resp_error_code}, {api_duration:.0f}s)"
elif _resp_error_code in {503, 529}:
_failure_hint = f"upstream provider overloaded ({_resp_error_code})"
elif _resp_error_code is not None:
_failure_hint = f"upstream error (code {_resp_error_code}, {api_duration:.0f}s)"
elif api_duration < 10:
_failure_hint = f"fast response ({api_duration:.1f}s) — likely rate limited"
elif api_duration > 60:
_failure_hint = f"slow response ({api_duration:.0f}s) — likely upstream timeout"
else:
_failure_hint = f"response time {api_duration:.1f}s"
agent._buffer_vprint(f"⚠️ Invalid API response (attempt {retry_count}/{max_retries}): {', '.join(error_details)}")
agent._buffer_vprint(f" 🏢 Provider: {provider_name}")
cleaned_provider_error = agent._clean_error_message(error_msg)
agent._buffer_vprint(f" 📝 Provider message: {cleaned_provider_error}")
agent._buffer_vprint(f" ⏱️ {_failure_hint}")
if retry_count >= max_retries:
# Try fallback before giving up
if agent._has_pending_fallback():
agent._buffer_status(f"⚠️ Max retries ({max_retries}) for invalid responses — trying fallback...")
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# Terminal — flush buffered retry trace so user sees what happened.
agent._flush_status_buffer()
agent._emit_status(f"❌ Max retries ({max_retries}) exceeded for invalid responses. Giving up.")
logger.error("%sInvalid API response after %d retries.", agent.log_prefix, max_retries)
agent._persist_session(messages, conversation_history)
_final_response = f"Invalid API response after {max_retries} retries: {_failure_hint}"
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"failed": True # Mark as failure for filtering
}
# Backoff before retry — jittered exponential: 5s base, 120s cap
wait_time = jittered_backoff(retry_count, base_delay=5.0, max_delay=120.0)
agent._buffer_vprint(f"⏳ Retrying in {wait_time:.1f}s ({_failure_hint})...")
logger.warning("Invalid API response (retry %d/%d): %s | Provider: %s", retry_count, max_retries, ', '.join(error_details), provider_name)
# Sleep in small increments to stay responsive to interrupts
sleep_end = time.time() + wait_time
_backoff_touch_counter = 0
while time.time() < sleep_end:
if agent._interrupt_requested:
# A redirect cancels only the live request;
# clear_interrupt() would DESTROY the pending correction.
# Rebuild from it.
if agent.clear_interrupt(preserve_redirect=True):
_retry.restart_with_redirected_messages = True
break
agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during retry wait, aborting.", force=True)
_interrupt_text = f"Operation interrupted during retry ({_failure_hint}, attempt {retry_count}/{max_retries})."
close_interrupted_tool_sequence(messages, _interrupt_text)
agent._persist_session(messages, conversation_history)
agent.clear_interrupt()
return {
"final_response": _interrupt_text,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"interrupted": True,
}
time.sleep(0.2)
# Touch activity every ~30s so the gateway's inactivity
# monitor knows we're alive during backoff waits.
_backoff_touch_counter += 1
if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s
agent._touch_activity(
f"retry backoff ({retry_count}/{max_retries}), "
f"{int(sleep_end - time.time())}s remaining"
)
if _retry.restart_with_redirected_messages:
break # rebuild this iteration from the correction
continue # Retry the API call
agent._turn_received_provider_response = True
# Check finish_reason before proceeding
if agent.api_mode == "codex_responses":
status = getattr(response, "status", None)
if isinstance(status, str):
status = status.strip().lower()
incomplete_details = getattr(response, "incomplete_details", None)
incomplete_reason = None
if isinstance(incomplete_details, dict):
incomplete_reason = incomplete_details.get("reason")
else:
incomplete_reason = getattr(incomplete_details, "reason", None)
if incomplete_reason is not None:
incomplete_reason = str(incomplete_reason).strip().lower()
if status == "incomplete" and incomplete_reason in {"max_output_tokens", "length"}:
# Responses API max-output exhaustion is a normal Codex
# incomplete turn: use the Codex continuation path, not the
# length rollback.
finish_reason = "incomplete"
elif status == "incomplete" and incomplete_reason == "content_filter":
finish_reason = "content_filter"
else:
finish_reason = "stop"
elif agent.api_mode == "anthropic_messages":
_tfr = agent._get_transport()
finish_reason = _tfr.map_finish_reason(response.stop_reason)
elif agent.api_mode == "bedrock_converse":
# Bedrock response already normalized at dispatch — use transport
_bt_fr = agent._get_transport()
_bedrock_result = _bt_fr.normalize_response(response)
finish_reason = _bedrock_result.finish_reason
else:
_cc_fr = agent._get_transport()
_finish_result = _cc_fr.normalize_response(response)
finish_reason = _finish_result.finish_reason
assistant_message = _finish_result
if agent._should_treat_stop_as_truncated(
finish_reason,
assistant_message,
messages,
):
agent._vprint(
f"{agent.log_prefix}⚠️ Treating suspicious Ollama/GLM stop response as truncated",
force=True,
)
finish_reason = "length"
# ── Content-policy refusal (HTTP 200) ──────────────────
# Refusal finish reasons (``content_filter``, ``guardrail_intervened``)
# are deterministic: one fallback try, else return the refusal.
if finish_reason == "content_filter":
_refusal_transport = agent._get_transport()
if agent.api_mode == "anthropic_messages":
_refusal_result = _refusal_transport.normalize_response(
response, strip_tool_prefix=agent._is_anthropic_oauth
)
else:
_refusal_result = _refusal_transport.normalize_response(response)
_refusal_text = (getattr(_refusal_result, "content", None) or "").strip()
# Some refusals carry the explanation only in the reasoning
# channel; fall back to it so the user sees *something*.
if not _refusal_text:
_refusal_text = (agent._extract_reasoning(_refusal_result) or "").strip()
agent._invoke_api_request_error_hook(
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
api_call_count=api_call_count,
api_start_time=api_start_time,
api_kwargs=api_kwargs,
error_type="ContentPolicyBlocked",
error_message=_refusal_text or "model declined to respond (content_filter)",
status_code=None,
retry_count=retry_count,
max_retries=max_retries,
retryable=False,
reason=FailoverReason.content_policy_blocked.value,
)
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
# Deterministic for the unchanged prompt — never retry. Try a
# configured fallback once; otherwise surface the refusal.
if agent._has_pending_fallback():
agent._buffer_status(
"⚠️ Model declined to respond (safety refusal) — trying fallback..."
)
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
agent._flush_status_buffer()
_refusal_log = (
_refusal_text[:500] + "..."
if len(_refusal_text) > 500
else _refusal_text
)
logger.warning(
"%sModel declined to respond (finish_reason=content_filter). "
"model=%s provider=%s refusal=%s",
agent.log_prefix, agent.model, agent.provider,
_refusal_log or "(no text)",
)
agent._emit_status(
"⚠️ The model declined to respond to this request (safety refusal)."
)
_refusal_detail = (
f"Model's explanation: {_refusal_text}"
if _refusal_text
else "The model returned no explanation."
)
_refusal_response = (
"⚠️ The model declined to respond to this request "
"(safety refusal — not a Hermes/gateway failure).\n\n"
f"{_refusal_detail}\n\n"
f"{_CONTENT_POLICY_RECOVERY_HINT}"
)
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return _content_policy_blocked_result(
messages,
api_call_count,
final_response=_refusal_response,
error_detail=_refusal_text or "model declined (content_filter)",
)
if finish_reason == "length":
if getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID:
agent._vprint(
f"{agent.log_prefix}⚠️ Response truncated — stream "
f"ended before completion",
force=True,
)
else:
agent._vprint(
f"{agent.log_prefix}⚠️ Response truncated "
f"(finish_reason='length') - model hit max output tokens",
force=True,
)
# Normalize to one OpenAI-style message so continuation and tool-
# call retry work across transports (Anthropic reuses the loop's
# adapter).
_trunc_msg = None
_trunc_transport = agent._get_transport()
if agent.api_mode == "anthropic_messages":
_trunc_result = _trunc_transport.normalize_response(
response, strip_tool_prefix=agent._is_anthropic_oauth
)
else:
_trunc_result = _trunc_transport.normalize_response(response)
_trunc_msg = _trunc_result
_trunc_content = getattr(_trunc_msg, "content", None) if _trunc_msg else None
_trunc_has_tool_calls = bool(getattr(_trunc_msg, "tool_calls", None)) if _trunc_msg else False
# ── Detect thinking-budget exhaustion ──────────────
# Only when reasoning blocks exist with no visible text after them;
# content=None from non-<think> models is normal truncation.
_has_think_tags = bool(
_trunc_content and re.search(
r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>',
_trunc_content,
re.IGNORECASE,
)
)
_thinking_exhausted = (
not _trunc_has_tool_calls
and _has_think_tags
and (
(_trunc_content is not None and not agent._has_content_after_think_block(_trunc_content))
or _trunc_content is None
)
)
if _thinking_exhausted:
_exhaust_error = (
"Model used all output tokens on reasoning with none left "
"for the response. Try lowering reasoning effort or "
"increasing max_tokens."
)
agent._vprint(
f"{agent.log_prefix}💭 Reasoning exhausted the output token budget — "
f"no visible response was produced.",
force=True,
)
# Return a user-friendly message as the response so CLI and
# gateway display it.
_exhaust_response = (
"⚠️ **Thinking Budget Exhausted**\n\n"
"The model used all its output tokens on reasoning "
"and had none left for the actual response.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n"
"→ Or switch to a larger/non-reasoning model with `/model`"
)
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": _exhaust_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": _exhaust_error,
}
# ── Detect repetition-dominated truncation (#86581) ──
# A repetition loop can burn the whole budget on one fragment; abort
# like _thinking_exhausted (reasoning stripped first).
_visible_trunc = (
agent._strip_think_blocks(_trunc_content)
if isinstance(_trunc_content, str)
else _trunc_content
)
_repetition_dominated = (
not _trunc_has_tool_calls
and bool(_visible_trunc)
and is_repetition_dominated(_visible_trunc)
)
if _repetition_dominated:
_rep_error = (
"Model output entered a repetition loop and was "
"truncated mid-loop; refusing to continue a "
"degenerate response."
)
agent._vprint(
f"{agent.log_prefix}🔁 Response dominated by "
f"repeated text — stopping instead of "
f"continuing a degenerate response.",
force=True,
)
_rep_response = (
"⚠️ **Response Stopped — Repetition Detected**\n\n"
"The model fell into a repetition loop while "
"writing this response, so continuing would only "
"produce more repeated text. The partial response "
"was discarded.\n\n"
"→ Switch to a different model with `/model`\n"
"→ Or resend your message (your conversation "
"history is preserved)"
)
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": _rep_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": _rep_error,
}
if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}:
assistant_message = _trunc_msg
# ── Content-filter stream stall → fallback (#32421) ──
# ``_content_filter_terminated`` is content-deterministic;
# escalate to the fallback before retrying the primary.
_cf_terminated = getattr(
response, "_content_filter_terminated", False
)
if (
_cf_terminated
and agent._fallback_index < len(agent._fallback_chain)
):
agent._vprint(
f"{agent.log_prefix}🛡️ Content filter terminated "
f"stream — activating fallback provider...",
force=True,
)
agent._emit_status(
"Content filter terminated stream; switching to fallback..."
)
if agent._try_activate_fallback():
# Roll partial content back to the last clean turn so
# the fallback gets a coherent continuation point.
if truncated_response_parts:
messages = agent._get_messages_up_to_last_assistant(messages)
# Unmark survivors: their text left the stitched partial.
for _frag in messages:
if isinstance(_frag, dict):
_frag.pop("_length_continuation_fragment", None)
_frag.pop("_length_continuation_nudge", None)
agent._session_messages = messages
length_continue_retries = 0
truncated_response_parts = []
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# No fallback available — fall through to normal
# continuation (best-effort, may loop).
agent._vprint(
f"{agent.log_prefix}⚠️ No fallback provider "
f"configured — retrying with same provider "
f"(may re-hit filter)...",
force=True,
)
if assistant_message is not None and not _trunc_has_tool_calls:
length_continue_retries += 1
# Never append an interim assistant message with NO visible
# content: strict providers reject it (HTTP 400), poisoning
# history. Append only the nudge.
_interim_content = getattr(assistant_message, "content", None)
_is_empty_partial_stub = (
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
and not _interim_content
)
if not _interim_content and not _is_empty_partial_stub:
# Thinking-only truncation: continuing with thinking ON
# re-burns the budget, so drop thinking for one request.
agent._ephemeral_reasoning_off = True
if _interim_content:
interim_msg = agent._build_assistant_message(assistant_message, finish_reason)
# Marked so the ceiling exit can drop the fragment trail.
interim_msg["_length_continuation_fragment"] = True
append_message(messages, interim_msg)
truncated_response_parts.append(_interim_content)
if length_continue_retries < 4:
_is_partial_stream_stub = (
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
)
_dropped_tools = getattr(
response, "_dropped_tool_names", None
)
if _is_partial_stream_stub and _dropped_tools:
_tool_list = ", ".join(_dropped_tools[:3])
agent._vprint(
f"{agent.log_prefix}↻ Stream interrupted mid "
f"tool-call ({_tool_list}) — requesting "
f"chunked retry "
f"({length_continue_retries}/4)..."
)
elif _is_partial_stream_stub:
agent._vprint(
f"{agent.log_prefix}↻ Stream interrupted — "
f"requesting continuation "
f"({length_continue_retries}/4)..."
)
else:
agent._vprint(
f"{agent.log_prefix}↻ Requesting continuation "
f"({length_continue_retries}/4)..."
)
_continue_content = _get_continuation_prompt(
_is_partial_stream_stub, _dropped_tools
)
continue_msg = {
"role": "user",
"content": _continue_content,
"_length_continuation_nudge": True,
}
append_message(messages, continue_msg)
agent._session_messages = messages
_retry.restart_with_length_continuation = True
break
partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip()
# The one-shot reasoning-off override must not leak into the
# next turn when the ceiling exit skips the consuming call.
agent._ephemeral_reasoning_off = False
if partial_response:
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after {length_continue_retries} continuation attempts — keeping the "
f"partial response received so far.",
force=True,
)
_ceiling_final = partial_response
else:
# Every fragment was empty (e.g. reasoning-only model):
# return an actionable message, not a bare None.
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after {length_continue_retries} continuation attempts — no visible "
f"text was produced.",
force=True,
)
_ceiling_final = (
"⚠️ **No visible answer was produced.** The "
"model hit its output-token limit on every "
"continuation attempt — its reasoning "
"consumed the entire budget each time.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/reasoning low` "
"or `/reasoning none`\n"
"→ Or raise max_tokens for this model"
)
# Unanswered continue nudges made every later turn re-truncate.
_turn_start = (
current_turn_user_idx + 1
if isinstance(current_turn_user_idx, int)
and current_turn_user_idx >= 0
else 0
)
messages[_turn_start:] = [
m for m in messages[_turn_start:]
if not (
isinstance(m, dict)
and (
m.get("_length_continuation_fragment")
or m.get("_length_continuation_nudge")
)
)
]
if partial_response:
append_message(messages, {
"role": "assistant",
"content": partial_response,
"finish_reason": "length",
})
agent._session_messages = messages
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": _ceiling_final,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": "Response remained truncated after 4 continuation attempts",
}
if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}:
assistant_message = _trunc_msg
if assistant_message is not None and _trunc_has_tool_calls:
_is_stub_stall = (
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
)
if truncated_tool_call_retries < 4:
truncated_tool_call_retries += 1
if _is_stub_stall:
# Stream broke mid tool-call (network), not a real
# output cap — say so.
agent._buffer_vprint(
f"⚠️ Stream interrupted mid tool-call — "
f"retrying ({truncated_tool_call_retries}/4)..."
)
else:
agent._buffer_vprint(
f"⚠️ Truncated tool call detected — "
f"retrying API call "
f"({truncated_tool_call_retries}/4)..."
)
# Boost max_tokens per retry: a real output-cap
# truncation needs it; harmless for a stall.
_tc_boost_base = agent.max_tokens if agent.max_tokens else 4096
_tc_boost = _tc_boost_base * (2 ** truncated_tool_call_retries)
_tc_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs)
if _tc_requested_cap is not None:
_tc_boost = max(_tc_boost, _tc_requested_cap)
_tc_boost_cap = max(32768, _tc_requested_cap or 0)
agent._ephemeral_max_output_tokens = min(_tc_boost, _tc_boost_cap)
# Don't append the broken response; re-run the same call
# from current state.
continue
agent._flush_status_buffer()
if _is_stub_stall:
agent._vprint(
f"{agent.log_prefix}⚠️ Stream kept dropping mid tool-call after 4 retries — the action was not executed.",
force=True,
)
else:
agent._vprint(
f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.",
force=True,
)
agent._cleanup_task_resources(effective_task_id)
_final_response = (
"Stream repeatedly dropped mid tool-call (network); "
"the tool was not executed"
if _is_stub_stall
else "Response truncated due to output length limit"
)
# Prior tool batches can leave a tool-result tail; this path
# never reaches finalize_turn (#48879).
close_interrupted_tool_sequence(messages, _final_response)
agent._persist_session(messages, conversation_history)
return {
"final_response": _final_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": _final_response,
}
# If we have prior messages, roll back to last complete state
if len(messages) > 1:
agent._vprint(f"{agent.log_prefix} ⏪ Rolling back to last complete assistant turn")
rolled_back_messages = agent._get_messages_up_to_last_assistant(messages)
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": "Response truncated due to output length limit",
"messages": rolled_back_messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": "Response truncated due to output length limit"
}
else:
# First message was truncated - mark as failed
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ First response truncated - cannot recover", force=True)
agent._persist_session(messages, conversation_history)
return {
"final_response": "First response truncated due to output length limit",
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": "First response truncated due to output length limit"
}
# Track actual token usage from response for context management
if hasattr(response, 'usage') and response.usage:
canonical_usage = normalize_usage(
response.usage,
provider=agent.provider,
api_mode=agent.api_mode,
)
# Aggregator-only usage kept for pricing: advisor tokens are priced
# at each advisor's OWN model rate and added as dollars below.
aggregator_usage = canonical_usage
# MoA: fold advisor fan-out usage into REPORTED token counts — only
# aggregator usage is returned, so advisor spend would be invisible.
_moa_ref_cost = None
_moa_client = getattr(agent, "client", None)
if _moa_client is not None and hasattr(_moa_client, "consume_reference_usage"):
try:
_ref_usage, _moa_ref_cost = _moa_client.consume_reference_usage()
if _ref_usage is not None:
canonical_usage = canonical_usage + _ref_usage
except Exception as _moa_acct_exc: # pragma: no cover - defensive
logger.debug("MoA reference usage accounting failed: %s", _moa_acct_exc)
# Flush the full-turn MoA trace when moa.save_traces is on; on the
# streaming path pass the streamed acting text so the trace is self-
# contained.
if _moa_client is not None and hasattr(_moa_client, "consume_and_save_trace"):
try:
_agg_streamed_text = (
getattr(agent, "_current_streamed_assistant_text", "") or ""
)
_moa_client.consume_and_save_trace(
agent.session_id,
aggregator_output_fallback=_agg_streamed_text or None,
)
except Exception as _moa_trace_exc: # pragma: no cover - defensive
logger.debug("MoA trace flush failed: %s", _moa_trace_exc)
prompt_tokens = canonical_usage.prompt_tokens
completion_tokens = canonical_usage.output_tokens
total_tokens = canonical_usage.total_tokens
# Forward canonical token + cache buckets for context engines;
# legacy keys stay for back-compat.
usage_dict = {
"prompt_tokens": prompt_tokens,
"completion_tokens": completion_tokens,
"total_tokens": total_tokens,
"input_tokens": canonical_usage.input_tokens,
"output_tokens": canonical_usage.output_tokens,
"cache_read_tokens": canonical_usage.cache_read_tokens,
"cache_write_tokens": canonical_usage.cache_write_tokens,
"reasoning_tokens": canonical_usage.reasoning_tokens,
}
# Capture the boundary latch before update_from_response() consumes
# it: only the real prompt count right after a compaction rearms the
# budget.
_completed_compaction_pending = bool(
getattr(
agent.context_compressor,
"_verify_compaction_cleared_threshold",
False,
)
)
agent.context_compressor.update_from_response(usage_dict)
# Usage-anchored accounting: snapshot exact provider usage against
# the durable transcript; main-loop ONLY. MoA uses pre-fold
# aggregator usage.
_new_anchor = capture_usage_anchor(
aggregator_usage.prompt_tokens,
aggregator_usage.output_tokens,
messages,
)
if _new_anchor is not None:
agent._usage_anchor = _new_anchor
# Anchor the display meter on the turn's FIRST response:
# later same-turn responses inflate prompt_tokens with replayed
# thinking. Display-only; compression math uses real usage.
if api_call_count == 1:
agent._turn_base_usage_anchor = _new_anchor
_compression_threshold = int(
getattr(agent.context_compressor, "threshold_tokens", 0)
or 0
)
if _should_rearm_compression_budget(
compression_attempts,
completed_compaction_pending=_completed_compaction_pending,
prompt_tokens=prompt_tokens,
threshold_tokens=_compression_threshold,
):
logger.info(
"Compression budget rearmed after provider-confirmed "
"recovery: prompt=%s < threshold=%s (attempts were %s/%s)",
f"{prompt_tokens:,}",
f"{_compression_threshold:,}",
compression_attempts,
max_compression_attempts,
)
compression_attempts = 0
# Confirmed recovery also clears the stale insufficient-progress
# verdict, else _preflight_compression_blocked stays armed all
# turn and a later pressure spike grows unchecked.
_preflight_compression_blocked = False
_last_preflight_pressure = None
# Stash canonical usage for on_turn_complete() (same shape as
# update_from_response); keep the latest call's — last request.
agent._last_turn_usage = dict(usage_dict)
elif getattr(
agent.context_compressor,
"awaiting_real_usage_after_compression",
False,
):
# No usage -> cannot adjudicate the prior compaction; consume the
# pending verdict so later readings aren't charged to it and
# preflight deferral isn't latched indefinitely.
agent.context_compressor.update_from_response({})
if hasattr(response, 'usage') and response.usage:
# Persist only provider-confirmed context lengths, not probe tiers.
if getattr(agent.context_compressor, "_context_probed", False):
ctx = agent.context_compressor.context_length
if getattr(agent.context_compressor, "_context_probe_persistable", False):
save_context_length(agent.model, agent.base_url, ctx)
agent._safe_print(f"{agent.log_prefix}💾 Cached context length: {ctx:,} tokens for {agent.model}")
agent.context_compressor._context_probed = False
agent.context_compressor._context_probe_persistable = False
agent.session_prompt_tokens += prompt_tokens
agent.session_completion_tokens += completion_tokens
agent.session_total_tokens += total_tokens
agent.session_api_calls += 1
agent.session_input_tokens += canonical_usage.input_tokens
agent.session_output_tokens += canonical_usage.output_tokens
agent.session_cache_read_tokens += canonical_usage.cache_read_tokens
agent.session_cache_write_tokens += canonical_usage.cache_write_tokens
agent.session_reasoning_tokens += canonical_usage.reasoning_tokens
# Rolling history for status-bar averages (last 10).
try:
hist = getattr(agent, "_api_latency_history", None)
if hist is not None:
hist.append(float(api_duration))
ohist = getattr(agent, "_api_output_history", None)
if ohist is not None:
ohist.append(int(canonical_usage.output_tokens or 0))
except Exception:
pass
# Log API call details for debugging/observability
_cache_pct = ""
if canonical_usage.cache_read_tokens and prompt_tokens:
_cache_pct = f" cache={canonical_usage.cache_read_tokens}/{prompt_tokens} ({100*canonical_usage.cache_read_tokens/prompt_tokens:.0f}%)"
logger.info(
"API call #%d: model=%s provider=%s in=%d out=%d total=%d latency=%.1fs%s",
agent.session_api_calls, agent.model, agent.provider or "unknown",
prompt_tokens, completion_tokens, total_tokens,
api_duration, _cache_pct,
)
# MoA: agent.model/provider are the virtual preset/"moa" with no
# pricing entry, silently dropping aggregator spend. Price at the
# REAL model/provider from the MoA client's aggregator slot.
_agg_cost_model = agent.model
_agg_cost_provider = agent.provider
_agg_cost_base_url = agent.base_url
_agg_slot = getattr(_moa_client, "last_aggregator_slot", None) if _moa_client is not None else None
if _agg_slot and _agg_slot.get("model"):
_agg_cost_model = _agg_slot["model"]
_agg_cost_provider = _agg_slot.get("provider") or agent.provider
_agg_cost_base_url = _agg_slot.get("base_url") or agent.base_url
cost_result = estimate_usage_cost(
_agg_cost_model,
aggregator_usage,
provider=_agg_cost_provider,
base_url=_agg_cost_base_url,
api_key=getattr(agent, "api_key", ""),
)
if cost_result.amount_usd is not None:
agent.session_estimated_cost_usd += float(cost_result.amount_usd)
# Add MoA advisor cost (already priced per-advisor at each
# advisor's own model rate) on top of the aggregator cost.
if _moa_ref_cost is not None:
try:
agent.session_estimated_cost_usd += float(_moa_ref_cost)
except (TypeError, ValueError): # pragma: no cover - defensive
pass
agent.session_cost_status = cost_result.status
agent.session_cost_source = cost_result.source
# Persist per-call token deltas for any session_id so non-CLI runs
# can't lose accounting; gateway/session-store writes use absolute
# totals and safely overwrite these deltas.
if agent._session_db and agent.session_id:
try:
# Ensure the row exists: under concurrent SQLite load the
# initial _ensure_db_session() may fail, and UPDATE on a
# missing row silently affects 0 rows.
if not agent._session_db_created:
agent._ensure_db_session()
# Cost delta = aggregator + MoA advisor cost so state.db's
# estimated_cost_usd matches the folded token counts.
_cost_delta = None
if cost_result.amount_usd is not None:
_cost_delta = float(cost_result.amount_usd)
if _moa_ref_cost is not None:
try:
_cost_delta = (_cost_delta or 0.0) + float(_moa_ref_cost)
except (TypeError, ValueError): # pragma: no cover
pass
# Enqueued, not written: a cold state.db UPDATE here stalled
# the tool loop. Drained at finalize via _persist_session.
agent._session_db.queue_token_counts(
agent.session_id,
input_tokens=canonical_usage.input_tokens,
output_tokens=canonical_usage.output_tokens,
cache_read_tokens=canonical_usage.cache_read_tokens,
cache_write_tokens=canonical_usage.cache_write_tokens,
reasoning_tokens=canonical_usage.reasoning_tokens,
estimated_cost_usd=_cost_delta,
cost_status=cost_result.status,
cost_source=cost_result.source,
billing_provider=agent.provider,
billing_base_url=agent.base_url,
billing_mode="subscription_included"
if cost_result.status == "included" else None,
model=agent.model,
api_call_count=1,
)
except Exception as e:
# Log failures — silent loss here undercounts analytics.
logger.debug(
"Token persistence failed (session=%s, tokens=%d): %s",
agent.session_id, total_tokens, e,
)
if agent.verbose_logging:
logging.debug(f"Token usage: prompt={usage_dict['prompt_tokens']:,}, completion={usage_dict['completion_tokens']:,}, total={usage_dict['total_tokens']:,}")
# Report cache stats for any provider that returns
# ``prompt_tokens_details.cached_tokens``, not only when we inject
# cache_control markers. ``canonical_usage`` is already normalised.
cached = canonical_usage.cache_read_tokens
written = canonical_usage.cache_write_tokens
prompt = usage_dict["prompt_tokens"]
if (cached or written) and not agent.quiet_mode:
hit_pct = (cached / prompt * 100) if prompt > 0 else 0
agent._vprint(
f"{agent.log_prefix} 💾 Cache: "
f"{cached:,}/{prompt:,} tokens "
f"({hit_pct:.0f}% hit, {written:,} written)"
)
_retry.has_retried_429 = False # Reset on success
# Don't clear the retry buffer: bytes back != usable content; it is
# cleared once genuine content lands. Clearing Nous rate-limit state
# proves the limit reset so other sessions may resume.
if agent.provider == "nous":
try:
from agent.nous_rate_guard import clear_nous_rate_limit
clear_nous_rate_limit()
except Exception:
pass
from agent import relay_llm
relay_llm.complete_logical_call(
api_request_id,
outcome="success",
)
agent._touch_activity(f"API call #{api_call_count} completed")
break # Success, exit retry loop
except InterruptedError:
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
if agent._has_pending_redirect():
# redirect() cancelled only this request: keep the correction
# queued, clear the cancellation bit, let the outer loop rebuild.
# Never materialize incomplete signed/encrypted reasoning items.
if agent.clear_interrupt(preserve_redirect=True):
_retry.restart_with_redirected_messages = True
break
api_elapsed = time.time() - api_start_time
agent._vprint(f"{agent.log_prefix}⚡ Interrupted during API call.", force=True)
interrupted = True
# Keep assistant text already streamed before the stop, else the next
# turn has no record of the half-finished reply.
_partial = agent._strip_think_blocks(
getattr(agent, "_current_streamed_assistant_text", "") or ""
).strip()
if _partial:
append_message(messages, {"role": "assistant", "content": _partial})
final_response = _partial
else:
final_response = f"{INTERRUPT_WAITING_FOR_MODEL_PREFIX}{api_elapsed:.1f}s elapsed)."
agent._persist_session(messages, conversation_history)
break
except Exception as api_error:
# Stop spinner silently — retry status is buffered and
# only flushed when every retry+fallback is exhausted.
if thinking_spinner:
thinking_spinner.stop("")
thinking_spinner = None
if agent.thinking_callback:
agent.thinking_callback("")
# Pre-classification recovery (encoding sanitization, image rejection,
# Bedrock SDK streaming fallback) — see agent/turn_recovery.py.
_recovered, active_system_prompt = recover_before_classification(
agent,
api_error,
messages=messages,
api_messages=api_messages,
api_kwargs=api_kwargs,
active_system_prompt=active_system_prompt,
)
if _recovered:
continue
status_code = getattr(api_error, "status_code", None)
error_context = agent._extract_api_error_context(api_error)
# ── Interpreter finalization: abandon immediately ──
# Process is exiting mid-flight: retries/rotation/fallbacks are futile
# and the retry trace spams the shell. One log line; shared predicate.
from tools.interpreter_shutdown import interpreter_shutting_down
if interpreter_shutting_down(api_error):
logger.warning(
"%sInterpreter is shutting down — abandoning turn "
"during API call #%d (%s)",
agent.log_prefix, api_call_count, api_error,
)
_shutdown_summary = (
"Turn abandoned: the process was shutting down "
"before the model call could complete."
)
return {
"final_response": _shutdown_summary,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": _shutdown_summary,
"failure_reason": "interpreter_shutdown",
"failure_retryable": False,
}
# ── Classify the error for structured recovery decisions ──
_compressor = getattr(agent, "context_compressor", None)
_ctx_len = getattr(_compressor, "context_length", 200000) if _compressor else 200000
classified = classify_api_error(
api_error,
provider=getattr(agent, "provider", "") or "",
model=getattr(agent, "model", "") or "",
approx_tokens=approx_tokens,
context_length=_ctx_len,
num_messages=len(api_messages) if api_messages else 0,
)
logger.debug(
"Error classified: reason=%s status=%s retryable=%s compress=%s rotate=%s fallback=%s",
classified.reason.value, classified.status_code,
classified.retryable, classified.should_compress,
classified.should_rotate_credential, classified.should_fallback,
)
agent._invoke_api_request_error_hook(
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
api_call_count=api_call_count,
api_start_time=api_start_time,
api_kwargs=api_kwargs,
error_type=type(api_error).__name__,
error_message=str(api_error),
status_code=status_code,
retry_count=retry_count,
max_retries=max_retries,
retryable=classified.retryable,
reason=classified.reason.value,
)
if (
classified.reason == FailoverReason.billing
and _is_nous_inference_route(
getattr(agent, "provider", "") or "",
getattr(agent, "base_url", "") or "",
)
and not _retry.nous_paid_entitlement_refresh_attempted
):
_retry.nous_paid_entitlement_refresh_attempted = True
if _try_refresh_nous_paid_entitlement_credentials(agent):
agent._vprint(
f"{agent.log_prefix}🔐 Nous paid access verified — "
"refreshed runtime credentials and retrying request...",
force=True,
)
continue
recovered_with_pool, _retry.has_retried_429 = agent._recover_with_credential_pool(
status_code=status_code,
has_retried_429=_retry.has_retried_429,
classified_reason=classified.reason,
error_context=error_context,
billing_unverified=classified.billing_unverified,
)
if recovered_with_pool:
continue
# Image-too-large recovery: shrink oversized native image parts
# in-place and retry once; otherwise fall through to normal handling.
if (
classified.reason == FailoverReason.image_too_large
and not _retry.image_shrink_retry_attempted
):
_retry.image_shrink_retry_attempted = True
image_max_dimension = _image_error_max_dimension(api_error) or 8000
if agent._try_shrink_image_parts_in_messages(
api_messages,
max_dimension=image_max_dimension,
):
agent._vprint(
f"{agent.log_prefix}📐 Image(s) exceeded provider size limit — "
f"shrank and retrying...",
force=True,
)
continue
else:
logger.info(
"image-shrink recovery: no data-URL image parts found "
"or shrink didn't reduce size; surfacing original error."
)
# Multimodal-tool-content recovery: strict OpenAI-spec providers 400
# on list-type tool content. Strip images, mark (provider, model)
# no-list-tool-content for the session, retry once (#27344).
if (
classified.reason == FailoverReason.multimodal_tool_content_unsupported
and not _retry.multimodal_tool_content_retry_attempted
):
_retry.multimodal_tool_content_retry_attempted = True
if agent._try_strip_image_parts_from_tool_messages(api_messages):
agent._vprint(
f"{agent.log_prefix}📐 Provider rejected list-type tool content — "
f"downgraded screenshots to text and retrying...",
force=True,
)
continue
else:
logger.info(
"multimodal-tool-content recovery: no list-type tool "
"messages with image parts found; surfacing original error."
)
# Image-corrupt recovery: provider rejected the image bytes; shrinking
# can't help, so strip image parts and retry once (#69078).
if classified.reason == FailoverReason.image_corrupt:
# Strip ONLY the per-call copy: replacing msg["content"] on the
# shallow api_messages rows keeps canonical history's images
# (copy-on-write; transient rejection must not erase history).
_imgs_removed = False
if isinstance(api_messages, list):
_imgs_removed = _strip_images_from_messages(api_messages)
if _imgs_removed:
agent._vprint(
f"{agent.log_prefix}⚠️ Provider rejected a corrupted image — "
f"stripped images from the retry payload and retrying...",
force=True,
)
continue
else:
logger.info(
"image-corrupt recovery: no image parts found to "
"strip; surfacing original error."
)
# Anthropic OAuth subscription rejected the 1M-context beta: disable it
# for this session, rebuild the client, retry once. Reactive so capable
# subscriptions keep full 1M context (#17680).
if (
classified.reason == FailoverReason.oauth_long_context_beta_forbidden
and agent.api_mode == "anthropic_messages"
and agent._is_anthropic_oauth
and not _retry.oauth_1m_beta_retry_attempted
):
_retry.oauth_1m_beta_retry_attempted = True
if not getattr(agent, "_oauth_1m_beta_disabled", False):
agent._oauth_1m_beta_disabled = True
try:
agent._anthropic_client.close()
except Exception:
pass
agent._rebuild_anthropic_client()
agent._vprint(
f"{agent.log_prefix}🔕 OAuth subscription doesn't support "
f"the 1M-context beta — disabled for this session and retrying...",
force=True,
)
continue
if (
agent.api_mode == "codex_responses"
and agent.provider in {"openai-codex", "xai-oauth"}
and status_code == 401
and not _retry.codex_auth_retry_attempted
):
_retry.codex_auth_retry_attempted = True
if agent._try_refresh_codex_client_credentials(force=True):
_label = "xAI OAuth" if agent.provider == "xai-oauth" else "Codex"
agent._buffer_vprint(f"🔐 {_label} auth refreshed after 401. Retrying request...")
continue
if (
agent.api_mode == "chat_completions"
and agent.provider == "vertex"
and status_code == 401
and not _retry.vertex_auth_retry_attempted
):
_retry.vertex_auth_retry_attempted = True
if agent._try_refresh_vertex_client_credentials():
agent._buffer_vprint("🔐 Vertex AI token refreshed after 401. Retrying request...")
continue
if (
agent.api_mode in ("chat_completions", "anthropic_messages")
and agent.provider == "nous"
and status_code == 401
and not _retry.nous_auth_retry_attempted
):
_retry.nous_auth_retry_attempted = True
if agent._try_refresh_nous_client_credentials(force=True):
agent._buffer_vprint(f"🔐 Nous agent key refreshed after 401. Retrying request...")
continue
# Refresh didn't help: likely Portal OAuth expired/revoked,
# no credits, or agent key blocked.
from hermes_constants import display_hermes_home as _dhh_fn
_dhh = _dhh_fn()
_body_text = ""
try:
_body = getattr(api_error, "body", None) or getattr(api_error, "response", None)
if _body is not None:
_body_text = str(_body)[:200]
except Exception:
pass
print(f"{agent.log_prefix}🔐 Nous 401 — Portal authentication failed.")
if _body_text:
print(f"{agent.log_prefix} Response: {_body_text}")
if not _print_nous_entitlement_guidance(agent, "Nous model access"):
print(f"{agent.log_prefix} Most likely: Portal OAuth expired, account out of credits, or agent key revoked.")
print(f"{agent.log_prefix} Troubleshooting:")
print(f"{agent.log_prefix} • Re-authenticate: hermes auth add nous")
print(f"{agent.log_prefix} • Check credits / billing: https://portal.nousresearch.com")
print(f"{agent.log_prefix} • Verify stored credentials: {_dhh}/auth.json")
print(f"{agent.log_prefix} • Switch providers temporarily: /model <model> --provider openrouter")
if (
_is_copilot_provider(agent)
and status_code == 401
and not _retry.copilot_auth_retry_attempted
):
_retry.copilot_auth_retry_attempted = True
if agent._try_refresh_copilot_client_credentials():
agent._buffer_vprint("🔐 Copilot credentials refreshed after 401. Retrying request...")
continue
if (
agent.api_mode == "anthropic_messages"
and status_code == 401
and hasattr(agent, '_anthropic_api_key')
and not _retry.anthropic_auth_retry_attempted
):
_retry.anthropic_auth_retry_attempted = True
from agent.anthropic_adapter import _is_oauth_token
from agent.azure_identity_adapter import is_token_provider
if agent._try_refresh_anthropic_client_credentials():
print(f"{agent.log_prefix}🔐 Anthropic credentials refreshed after 401. Retrying request...")
continue
# Credential refresh didn't help — show diagnostic info
key = agent._anthropic_api_key
print(f"{agent.log_prefix}🔐 Anthropic 401 — authentication failed.")
if is_token_provider(key):
# Azure Foundry Entra ID: JWT minted per-request by an httpx
# hook; 401 = Azure rejected it (RBAC, az login, IMDS).
print(f"{agent.log_prefix} Auth method: Microsoft Entra ID (httpx event hook)")
print(f"{agent.log_prefix} Run `hermes doctor` for credential-chain diagnostics, or")
print(f"{agent.log_prefix} `az login` if your developer session expired.")
else:
auth_method = "Bearer (OAuth/setup-token)" if _is_oauth_token(key) else "x-api-key (API key)"
print(f"{agent.log_prefix} Auth method: {auth_method}")
print(f"{agent.log_prefix} Token prefix: {key[:12]}..." if isinstance(key, str) and len(key) > 12 else f"{agent.log_prefix} Token: (empty or short)")
print(f"{agent.log_prefix} Troubleshooting:")
from hermes_constants import display_hermes_home as _dhh_fn
_dhh = _dhh_fn()
print(f"{agent.log_prefix} • Check ANTHROPIC_TOKEN in {_dhh}/.env for Hermes-managed OAuth/setup tokens")
print(f"{agent.log_prefix} • Check ANTHROPIC_API_KEY in {_dhh}/.env for API keys or legacy token values")
print(f"{agent.log_prefix} • For API keys: verify at https://platform.claude.com/settings/keys")
print(f"{agent.log_prefix} • For Claude Code: run 'claude /login' to refresh, then retry")
print(f"{agent.log_prefix} • Legacy cleanup: hermes config set ANTHROPIC_TOKEN \"\"")
print(f"{agent.log_prefix} • Clear stale keys: hermes config set ANTHROPIC_API_KEY \"\"")
# Thinking block signature recovery: upstream mutation invalidates
# Anthropic's signature (400). Strip ``reasoning_details`` from
# ``api_messages`` only, never ``messages`` (state.db). One-shot.
if (
classified.reason == FailoverReason.thinking_signature
and not _retry.thinking_sig_retry_attempted
):
_retry.thinking_sig_retry_attempted = True
_api_stripped = 0
for _m in api_messages:
if isinstance(_m, dict) and "reasoning_details" in _m:
_m.pop("reasoning_details", None)
_api_stripped += 1
agent._vprint(
f"{agent.log_prefix}⚠️ Thinking block signature invalid, "
f"stripped reasoning_details from api_messages for retry...",
force=True,
)
logger.warning(
"%sThinking block signature recovery: stripped "
"reasoning_details from %d api_messages "
"(canonical messages unchanged)",
agent.log_prefix, _api_stripped,
)
continue
# ── Invalid encrypted reasoning replay recovery ───────
# 400 ``invalid_encrypted_content`` on a stale ``codex_reasoning_items``
# blob: disable replay for the session, strip cached items, retry once.
if (
classified.reason == FailoverReason.invalid_encrypted_content
and not _retry.invalid_encrypted_content_retry_attempted
and agent.api_mode == "codex_responses"
and bool(getattr(agent, "_codex_reasoning_replay_enabled", True))
and any(
isinstance(_m, dict)
and _m.get("role") == "assistant"
and isinstance(_m.get("codex_reasoning_items"), list)
and _m.get("codex_reasoning_items")
for _m in messages
)
):
_retry.invalid_encrypted_content_retry_attempted = True
replay_stats = agent._disable_codex_reasoning_replay(messages)
agent._vprint(
f"{agent.log_prefix}⚠️ Encrypted reasoning replay was rejected by the provider — "
f"disabled replay and stripped {replay_stats['items']} item(s) from "
f"{replay_stats['messages']} message(s), retrying...",
force=True,
)
logger.warning(
"%sInvalid encrypted reasoning recovery: disabled replay and stripped %d items from %d messages",
agent.log_prefix,
replay_stats["items"],
replay_stats["messages"],
)
continue
# ── Native compaction rejection recovery ──────────────
# Structured 400 naming ``context_management``: disable native
# compaction for the session, retry once; local compression takes over.
if (
agent.api_mode == "codex_responses"
and not _retry.native_compaction_reject_retry_attempted
and bool(getattr(agent, "codex_responses_native_compaction", False))
):
from agent.native_compaction import is_native_compaction_rejection
if is_native_compaction_rejection(
api_error, getattr(api_error, "status_code", None)
):
_retry.native_compaction_reject_retry_attempted = True
agent.codex_responses_native_compaction = False
agent._vprint(
f"{agent.log_prefix}⚠️ Provider rejected native compaction "
f"(context_management) — disabled for this session, "
f"local compression stays active. Retrying...",
force=True,
)
logger.warning(
"%sNative compaction rejection recovery: disabled "
"codex_responses_native for this session and retrying",
agent.log_prefix,
)
continue
# ── llama.cpp grammar-parse recovery ──────────────────
# ``json-schema-to-grammar`` rejects regex escapes and most ``format``
# values: strip ``pattern``/``format`` from ``agent.tools``, retry once.
if (
classified.reason == FailoverReason.llama_cpp_grammar_pattern
and not _retry.llama_cpp_grammar_retry_attempted
):
_retry.llama_cpp_grammar_retry_attempted = True
try:
from tools.schema_sanitizer import strip_pattern_and_format
_, _stripped = strip_pattern_and_format(agent.tools)
except Exception as _strip_exc: # pragma: no cover — defensive
logger.warning(
"%sllama.cpp grammar recovery: strip helper failed: %s",
agent.log_prefix, _strip_exc,
)
_stripped = 0
if _stripped:
agent._vprint(
f"{agent.log_prefix}⚠️ llama.cpp rejected tool schema grammar — "
f"stripped {_stripped} pattern/format keyword(s), retrying...",
force=True,
)
logger.warning(
"%sllama.cpp grammar recovery: stripped %d "
"pattern/format keyword(s) from tool schemas",
agent.log_prefix, _stripped,
)
continue
# No keywords found to strip — fall through to normal
# retry path rather than loop forever on the same error.
logger.warning(
"%sllama.cpp grammar error but no pattern/format "
"keywords to strip — falling through to normal retry",
agent.log_prefix,
)
retry_count += 1
elapsed_time = time.time() - api_start_time
agent._touch_activity(
f"API error recovery (attempt {retry_count}/{max_retries})"
)
error_type = type(api_error).__name__
error_msg = str(api_error).lower()
_error_summary = agent._summarize_api_error(api_error)
logger.warning(
"API call failed (attempt %s/%s) error_type=%s %s summary=%s",
retry_count,
max_retries,
error_type,
agent._client_log_context(),
_error_summary,
)
_provider = getattr(agent, "provider", "unknown")
_base = getattr(agent, "base_url", "unknown")
_model = getattr(agent, "model", "unknown")
_status_code_str = f" [HTTP {status_code}]" if status_code else ""
agent._buffer_vprint(f"⚠️ API call failed (attempt {retry_count}/{max_retries}): {error_type}{_status_code_str}")
agent._buffer_vprint(f" 🔌 Provider: {_provider} Model: {_model}")
agent._buffer_vprint(f" 🌐 Endpoint: {_base}")
agent._buffer_vprint(f" 📝 Error: {_error_summary}")
if status_code and status_code < 500:
_err_body = getattr(api_error, "body", None)
_err_body_str = str(_err_body)[:300] if _err_body else None
if _err_body_str:
agent._buffer_vprint(f" 📋 Details: {_err_body_str}")
agent._buffer_vprint(f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens")
# OpenRouter "no tool endpoints" hint, buffered with the retry trace
# so it only surfaces if every retry+fallback exhausts.
if (
agent._is_openrouter_url()
and "support tool use" in error_msg
):
agent._buffer_vprint(
f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings."
)
if agent.providers_allowed:
agent._buffer_vprint(
" Your provider_routing.only restriction is filtering out tool-capable providers."
)
agent._buffer_vprint(
" Try removing the restriction or adding providers that support tools for this model."
)
agent._buffer_vprint(
f" Check which providers support tools: https://openrouter.ai/models/{_model}"
)
# Bare 404 on a ``vendor/model`` catalogue usually means the id lost its
# prefix; the provider never names the model, so we do (#78796).
if getattr(api_error, "status_code", None) == 404:
try:
from hermes_cli.model_normalize import suggest_prefixed_model_id
_suggestion = suggest_prefixed_model_id(_provider, _model)
except Exception:
_suggestion = None
if _suggestion:
agent._buffer_vprint(
f" 💡 Model '{_model}' is not a valid id for provider {_provider} — "
f"it is missing its vendor prefix."
)
agent._buffer_vprint(
f" Did you mean '{_suggestion}'? Re-pick it with `hermes model`."
)
# Check for interrupt before deciding to retry
if agent._interrupt_requested:
# Preserve a pending redirect: the user is steering, not stopping
# — rebuild the turn from the correction instead of aborting.
if agent.clear_interrupt(preserve_redirect=True):
_retry.restart_with_redirected_messages = True
break
agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during error handling, aborting retries.", force=True)
_interrupt_text = f"Operation interrupted: handling API error ({error_type}: {agent._clean_error_message(str(api_error))})."
close_interrupted_tool_sequence(messages, _interrupt_text)
agent._persist_session(messages, conversation_history)
agent.clear_interrupt()
return {
"final_response": _interrupt_text,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"interrupted": True,
}
# Check 413 BEFORE the generic 4xx handler: compress + retry, not abort.
status_code = getattr(api_error, "status_code", None)
# ── Respect disabled auto-compaction on overflow ──────
# ``compression.enabled: false`` forbids every automatic trigger, incl.
# these overflow recovery paths; error out. Output-cap errors exempt.
_overflow_reasons = {
FailoverReason.long_context_tier,
FailoverReason.payload_too_large,
FailoverReason.context_overflow,
}
_is_output_cap_error = (
is_output_cap_error(error_msg)
or parse_available_output_tokens_from_error(error_msg) is not None
)
if (
classified.reason in _overflow_reasons
and not getattr(agent, "compression_enabled", True)
and not _is_output_cap_error
):
agent._flush_status_buffer()
agent._vprint(
f"{agent.log_prefix}❌ Context overflow, but auto-compaction is disabled "
f"(compression.enabled: false).",
force=True,
)
agent._vprint(
f"{agent.log_prefix} 💡 Run /compress to compact manually, /new to start fresh, "
f"switch to a larger-context model, or reduce attachments.",
force=True,
)
logger.error(
f"{agent.log_prefix}Context overflow ({classified.reason.value}) with "
f"auto-compaction disabled — not compressing."
)
agent._persist_session(messages, conversation_history)
_final_response = (
"Context overflow and auto-compaction is disabled "
"(compression.enabled: false). Run /compress to compact manually, "
"/new to start fresh, or switch to a larger-context model."
)
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compaction_disabled": True,
}
# ── Anthropic Sonnet long-context tier gate ───────────
# 429 "Extra usage is required for long context requests" is a
# subscription-tier limit, not transient: cap at 200k and compress.
if classified.reason == FailoverReason.long_context_tier:
_reduced_ctx = 200000
compressor = agent.context_compressor
old_ctx = compressor.context_length
if old_ctx > _reduced_ctx:
compressor.update_model(
model=agent.model,
context_length=_reduced_ctx,
base_url=agent.base_url,
api_key=getattr(agent, "api_key", ""),
provider=agent.provider,
api_mode=agent.api_mode,
)
# Context probing flags — only set on built-in
# compressor (plugin engines manage their own).
if hasattr(compressor, "_context_probed"):
compressor._context_probed = True
# Don't persist — subscription-tier limit, not a model
# capability; 1M should return if extra usage is enabled.
compressor._context_probe_persistable = False
agent._buffer_vprint(
f"⚠️ Anthropic long-context tier "
f"requires extra usage — reducing context: "
f"{old_ctx:,} → {_reduced_ctx:,} tokens"
)
compression_attempts += 1
if compression_attempts <= max_compression_attempts:
original_len = len(messages)
# Option A (LCM issue 441): overhead-aware request size so recovery arms on
# the true request (msgs + tools + system), not the tool-blind message count.
messages, active_system_prompt = agent._compress_context(
messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id,
)
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
if len(messages) < original_len or old_ctx > _reduced_ctx:
agent._buffer_status(
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE.format(
new_ctx=_reduced_ctx, old_ctx=old_ctx
)
)
time.sleep(2)
# Provider proved the request doesn't fit the reduced
# window; row count isn't proof the rebuilt one does.
# Recheck the complete request before the next call.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
# Fall through to normal error handling if compression
# is exhausted or didn't help.
# Eager fallback: rate-limit/billing switch immediately (primary won't
# recover in the retry window); transport errors get 1 retry first.
is_rate_limited = classified.reason in {
FailoverReason.rate_limit,
FailoverReason.billing,
FailoverReason.upstream_rate_limit,
}
# Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only
# the max_tokens clamp fixes it (#72281). Parsed once; gates the
# eager-fallback exemption and the overflow entry below.
_wrapped_output_cap_budget = (
parse_available_output_tokens_from_error(error_msg)
if classified.reason == FailoverReason.rate_limit
else None
)
_is_transport_failure = classified.reason in {
FailoverReason.timeout,
FailoverReason.overloaded,
}
# Z.AI overload 429s classify `overloaded`, which `is_rate_limited`
# excludes. Detect directly so the long backoff runs, and raise the
# ceiling to reach it (see zai_coding_overload_retry_ceiling()).
_is_zai_coding_overload = is_zai_coding_overload_error(
base_url=str(_base), model=_model, error=api_error
)
if _is_zai_coding_overload:
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
_should_fallback = (
(is_rate_limited and _wrapped_output_cap_budget is None)
or (_is_transport_failure and retry_count >= 2)
)
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
# No eager fallback while credential pool rotation may recover
# (_pool_may_recover_from_rate_limit, #11314). Exception: an
# upstream-aggregator 429 — the pool can't help, always fall back.
_is_upstream = classified.reason == FailoverReason.upstream_rate_limit
pool_may_recover = (
False if _is_upstream
else _ra()._pool_may_recover_from_rate_limit(
agent._credential_pool,
)
)
if not pool_may_recover:
if _is_upstream:
_upstream_name = (classified.error_context or {}).get(
"upstream_provider", "aggregator"
)
agent._buffer_status(
f"⚠️ Upstream {_upstream_name} rate-limited — "
"switching to fallback model..."
)
elif classified.reason == FailoverReason.billing:
if classified.billing_unverified:
# Ambiguous body (#82154) — don't assert billing.
agent._buffer_status(
"⚠️ Provider reported usage/credit exhaustion "
"(unverified — may be a content-filter rejection) "
"— switching to fallback provider..."
)
else:
agent._buffer_status(
"⚠️ Billing or credits exhausted — switching to fallback provider..."
)
elif _is_transport_failure:
agent._buffer_status(
"⚠️ Provider unreachable — switching to fallback provider..."
)
else:
agent._buffer_status("⚠️ Rate limited — switching to fallback provider...")
if agent._try_activate_fallback(reason=classified.reason):
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# ── Auth-failure provider failover ───────────────────────
# A 401/403 surviving credential refresh means a broken credential or
# endpoint: escalate to the fallback chain; False -> terminal handling.
if (
classified.is_auth
and not _retry.auth_failover_attempted
and agent._fallback_index < len(agent._fallback_chain)
):
_retry.auth_failover_attempted = True
agent._buffer_status(
"🔐 Authentication failed and could not be refreshed — "
"switching to fallback provider..."
)
if agent._try_activate_fallback(reason=classified.reason):
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# ── Nous Portal: record rate limit & skip retries ─────
# A genuine account-level 429 is recorded to a shared file so ALL
# sessions back off; is_genuine_nous_rate_limit excludes upstream 429s.
if (
is_rate_limited
and agent.provider == "nous"
and classified.reason == FailoverReason.rate_limit
and not recovered_with_pool
):
_genuine_nous_rate_limit = False
try:
from agent.nous_rate_guard import (
is_genuine_nous_rate_limit,
record_nous_rate_limit,
)
_err_resp = getattr(api_error, "response", None)
_err_hdrs = (
getattr(_err_resp, "headers", None)
if _err_resp else None
)
_genuine_nous_rate_limit = is_genuine_nous_rate_limit(
headers=_err_hdrs,
last_known_state=agent._rate_limit_state,
)
if _genuine_nous_rate_limit:
record_nous_rate_limit(
headers=_err_hdrs,
error_context=error_context,
)
else:
logger.info(
"Nous 429 looks like upstream capacity "
"(no exhausted bucket in headers or "
"last-known state) -- not tripping "
"cross-session breaker."
)
except Exception:
pass
if _genuine_nous_rate_limit:
# Re-enter the loop exactly once so the top-of-loop Nous guard
# runs (retry_count = max_retries would skip it entirely).
retry_count = max(0, max_retries - 1)
continue
# Upstream capacity 429: normal retry logic will typically succeed.
is_payload_too_large = (
classified.reason == FailoverReason.payload_too_large
)
# GitHub Models free tier caps requests at 8K tokens, under the system
# prompt + tool schema floor; compression can't help, so say so.
if (
status_code == 413
and isinstance(agent.base_url, str)
and base_url_host_matches(agent.base_url, "models.inference.ai.azure.com")
):
agent._vprint(
f"{agent.log_prefix} 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every",
force=True,
)
agent._vprint(
f"{agent.log_prefix} request at ~8K tokens. Hermes' system prompt + tool schemas baseline",
force=True,
)
agent._vprint(
f"{agent.log_prefix} exceeds that floor, so this endpoint cannot run an agentic loop.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} Use the `copilot` provider with a Copilot subscription token (`hermes",
force=True,
)
agent._vprint(
f"{agent.log_prefix} setup` → GitHub Copilot), or pick any other provider.",
force=True,
)
if is_payload_too_large:
compression_attempts += 1
if compression_attempts > max_compression_attempts:
# Terminal — surface the buffered retry trace.
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached for payload-too-large error.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error("%s413 compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Request payload too large: max compression attempts ({max_compression_attempts}) reached."
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
}
agent._buffer_status(f"⚠️ Request payload too large (413) — compression attempt {compression_attempts}/{max_compression_attempts}...")
original_len = len(messages)
# A 413 is a BYTE-size error: score progress in payload bytes,
# never the token estimate, which is deliberately byte-blind to
# images and wedged sessions on "no progress" (#88960 / #47339).
original_bytes = serialized_messages_bytes(messages)
_overflow_input = messages
# Option A (LCM issue 441): overhead-aware request size so recovery arms on the
# true request (msgs + tools + system), not the tool-blind message count.
messages, active_system_prompt = agent._compress_context(
messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id,
# Provider proved the request doesn't fit: ignore the
# summary-failure cooldown for this ONE attempt (#100661).
bypass_cooldown=True,
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
# Lock-skip: another path holds the compression lock. A
# temporary defer, not exhaustion — refund the attempt and
# end softly so the gateway does NOT auto-reset (#69870).
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count
)
if messages is _overflow_input and compression_blocked_transiently(agent):
# Transient-block: a timed guard no-oped compression. A
# defer, never compression_exhausted (auto-reset) (#97488).
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count,
reason="transient_block",
)
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
# Re-measure: same-count compression and media aging can shrink
# the request without shrinking the array. Bytes are the yardstick
# for a 413; tokens only for status display.
new_tokens = estimate_messages_tokens_rough(messages)
approx_tokens = new_tokens # update for downstream logging
new_bytes = serialized_messages_bytes(messages)
made_progress = (
len(messages) < original_len
or (new_bytes > 0 and new_bytes < original_bytes * 0.95)
)
if made_progress:
if len(messages) < original_len:
agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
else:
agent._buffer_status(
f"🗜️ Compressed {original_bytes:,} → {new_bytes:,} "
f"payload bytes, retrying..."
)
time.sleep(2) # Brief pause between compression retries
_retry.restart_with_compressed_messages = True
break
else:
if agent._try_strip_image_parts_from_tool_messages(
api_messages,
remember_model=False,
):
agent._buffer_status(
"📐 Compression could not reduce the request further — "
"removed retained vision payloads and retrying..."
)
continue
# Terminal — surface buffered context so the user
# sees what compression attempts were made.
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Payload too large and cannot compress further.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error("%s413 payload too large. Cannot compress further.", agent.log_prefix)
agent._persist_session(messages, conversation_history)
_final_response = "Request payload too large (413). Cannot compress further."
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
}
# Check context-length errors BEFORE the generic 4xx handler; the
# classifier also covers 400/disconnect + large-session heuristics.
is_context_length_error = (
classified.reason == FailoverReason.context_overflow
# Relay-wrapped output-cap 429s (parsed above) go to the clamp
# below, not failover or generic retries (#72281).
or _wrapped_output_cap_budget is not None
)
if is_context_length_error:
compressor = agent.context_compressor
old_ctx = compressor.context_length
# Two errors: "prompt too long" = INPUT overflows the window (shrink
# context_length + compress); "max_tokens too large" = input fits
# but input + max_tokens > window (shrink OUTPUT cap only).
available_out = parse_available_output_tokens_from_error(error_msg)
if available_out is not None:
# Output-cap error: provider available_tokens is the
# authoritative bound; also estimate the real request shape
# (API-only content), use the smaller minus a margin.
request_input_estimate = estimate_request_tokens_rough(
api_messages, tools=agent.tools or None,
)
local_available_out = old_ctx - request_input_estimate
if local_available_out > 0:
safe_out = max(1, min(available_out, local_available_out) - 64)
else:
# Local estimate can overshoot; fall back to the
# authoritative provider-reported budget.
safe_out = max(1, available_out - 64)
agent._ephemeral_max_output_tokens = safe_out
agent._buffer_vprint(
f"⚠️ Output cap too large for current prompt — "
f"retrying with max_tokens={safe_out:,} "
f"(provider_available={available_out:,}, "
f"estimated_request_tokens={request_input_estimate:,}; "
f"context_length unchanged at {old_ctx:,})"
)
# Still count against compression_attempts so we don't
# loop forever if the error keeps recurring.
compression_attempts += 1
if compression_attempts > max_compression_attempts:
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached."
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
}
# Also compress history so the output-cap retry doesn't spin on
# max_tokens alone; dropping the middle window makes the total
# fit. (#55546)
try:
original_len = len(messages)
original_tokens = estimate_messages_tokens_rough(messages)
_overflow_input = messages
messages, active_system_prompt = agent._compress_context(
messages, system_message,
approx_tokens=request_input_estimate,
task_id=effective_task_id,
bypass_cooldown=True, # #100661 provider-proven overflow
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count
)
if messages is _overflow_input and compression_blocked_transiently(agent):
# #97488: timed transient guard — defer, never
# exhaustion (gateway auto-reset).
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count,
reason="transient_block",
)
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
new_tokens = estimate_messages_tokens_rough(messages)
if len(messages) < original_len:
agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
elif new_tokens > 0 and new_tokens < original_tokens * 0.95:
agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
except Exception:
# Compression must never turn an output-cap error
# fatal — fall through and retry on max_tokens alone.
logger.warning(
"%sOutput-cap compression hit an error; retrying on max_tokens only.",
agent.log_prefix,
)
_retry.restart_with_compressed_messages = True
break
# Output-cap error with unparseable budget: compression can't help
# (input already fits) and would death-loop on the same 400. Fail
# fast. (#55546)
if is_output_cap_error(error_msg):
agent._flush_status_buffer()
agent._vprint(
f"{agent.log_prefix}❌ The provider rejected the request because "
f"max_tokens exceeds its output cap for this model.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} 💡 Lower model.max_tokens in your config.yaml to "
f"at or below the model's max-output limit. "
f"(This is an output-cap error, not a context overflow — "
f"compression cannot fix it.)",
force=True,
)
logger.error(
f"{agent.log_prefix}Output-cap error not routed into compression "
f"(max_tokens over provider cap): {error_msg[:200]}"
)
agent._persist_session(messages, conversation_history)
_final_response = (
"max_tokens exceeds the provider's output cap for this model. "
"Lower model.max_tokens in config.yaml."
)
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
}
# Input too large: shrink context_length only when the provider
# reports the real limit; else keep the window and compress. Guessed
# probe tiers can turn a configured 1M window into 256K/128K/64K.
new_ctx = get_context_length_from_provider_error(error_msg, old_ctx)
_provider_lower = (getattr(agent, "provider", "") or "").lower()
_base_lower = (getattr(agent, "base_url", "") or "").rstrip("/").lower()
is_minimax_provider = (
_provider_lower in {"minimax", "minimax-cn"}
or _base_lower.startswith((
"https://api.minimax.io/anthropic",
"https://api.minimaxi.com/anthropic",
))
)
minimax_delta_only_overflow = (
is_minimax_provider
and new_ctx is None
and "context window exceeds limit (" in error_msg
)
if new_ctx is not None:
agent._buffer_vprint(f"Context limit detected from API: {new_ctx:,} tokens (was {old_ctx:,})")
compressor.update_model(
model=agent.model,
context_length=new_ctx,
base_url=agent.base_url,
api_key=getattr(agent, "api_key", ""),
provider=agent.provider,
api_mode=agent.api_mode,
)
# Persist the provider-reported limit before compression/retry:
# rate limit, missing usage, or restart must not lose confirmed
# metadata. Probe flags remain a fallback if this write fails.
save_context_length(agent.model, agent.base_url, new_ctx)
# Probe flags only on the built-in compressor (plugin engines
# manage their own); provider-sourced value, so safe to cache.
if hasattr(compressor, "_context_probed"):
compressor._context_probed = True
compressor._context_probe_persistable = True
agent._buffer_vprint(f"⚠️ Context length exceeded — using provider limit: {old_ctx:,} → {new_ctx:,} tokens")
elif minimax_delta_only_overflow:
agent._buffer_vprint(
f"Provider reported overflow amount only; "
f"keeping context_length at {old_ctx:,} tokens and compressing."
)
else:
agent._buffer_vprint(
f"⚠️ Context length exceeded, but provider did not report a max context length; "
f"keeping context_length at {old_ctx:,} tokens and compressing."
)
compression_attempts += 1
if compression_attempts > max_compression_attempts:
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached."
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
}
agent._buffer_status(COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE.format(tokens=approx_tokens, attempt=compression_attempts, cap=max_compression_attempts))
original_len = len(messages)
original_tokens = estimate_messages_tokens_rough(messages)
_overflow_input = messages
# Pass the OVERHEAD-AWARE size (msgs + tool schemas + system) so LCM
# forced-overflow recovery arms on the TRUE request; approx_tokens
# stays for status. See hermes-lcm _should_force_overflow_recovery.
messages, active_system_prompt = agent._compress_context(
messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id,
# Provider proved the request doesn't fit: ignore the
# summary-failure cooldown for this ONE attempt (bounded by
# max_compression_attempts). (#100661)
bypass_cooldown=True,
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
# Lock-skip: another path holds the compression lock, so this
# pass no-oped. Temporary defer, not exhaustion — refund the
# attempt, end the turn softly, no auto-reset. (#69870)
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count
)
if messages is _overflow_input and compression_blocked_transiently(agent):
# Transient block: a timed guard (host-timeout cooldown /
# structural backoff) no-oped this pass — defer softly, never
# compression_exhausted (auto-reset). (#97488)
compression_attempts -= 1
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent, messages, api_call_count,
reason="transient_block",
)
if context_compression_timed_out(agent):
# Host timeout: recovery spent its wait budget with no committed
# summary. Re-sending would hit the same overflow; end the turn
# via the typed recovery contract. (#98722)
agent._persist_session(messages, conversation_history)
_final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_timeout",
}
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
# Re-estimate after compression: same-message-count compression
# (tool-result pruning, in-place summarization) can shrink the
# request. (#39550)
new_tokens = estimate_messages_tokens_rough(messages)
approx_tokens = new_tokens # update for downstream logging
if len(messages) < original_len or (new_tokens > 0 and new_tokens < original_tokens * 0.95) or (new_ctx and new_ctx < old_ctx):
if len(messages) < original_len:
agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
elif new_tokens > 0 and new_tokens < original_tokens * 0.95:
agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
time.sleep(2) # Brief pause between compression retries
# Rebuild the full request and force normal preflight to honor
# it; message count alone doesn't prove system/tool-inclusive
# pressure fell.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
else:
# Can't compress further and already at minimum tier
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Context length exceeded and cannot compress further.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.", force=True)
logger.error("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}")
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further."
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
}
# Non-retryable: ValueError/TypeError are local bugs, except
# UnicodeEncodeError (surrogate path above) and json.JSONDecodeError, a
# transient provider/network failure that must be retried (#14782).
is_local_validation_error = (
isinstance(api_error, (ValueError, TypeError))
and not isinstance(
api_error, (UnicodeEncodeError, json.JSONDecodeError)
)
# ssl.SSLError inherits from OSError *and* ValueError, so the
# ValueError check would misclassify a TLS failure as a local bug;
# keep it retryable.
and not isinstance(api_error, ssl.SSLError)
# "NoneType is not iterable" TypeErrors are upstream shape
# mismatches (e.g. Codex response.completed.output=null), reachable
# via shims/mocks — retryable so the fallback path runs.
and not (
isinstance(api_error, TypeError)
and "nonetype" in str(api_error).lower()
and "not iterable" in str(api_error).lower()
)
)
# ``FailoverReason.billing`` (402) is deliberately NOT excluded: pool
# rotation and eager fallback already gave up, so retrying only burns
# paid requests on a depleted balance. Mirrors 401/403. (#31273)
is_client_error = (
is_local_validation_error
or (
not classified.retryable
and not classified.should_compress
and classified.reason not in {
FailoverReason.rate_limit,
FailoverReason.overloaded,
FailoverReason.context_overflow,
FailoverReason.payload_too_large,
FailoverReason.long_context_tier,
FailoverReason.thinking_signature,
}
)
) and not is_context_length_error
if is_client_error:
# Copilot self-heal BEFORE fallback: a stale credential yields a 400
# ``model_not_available_for_integrator`` / ``model_not_supported``,
# not a 401. Fresh token + client rebuild, one retry, SAME provider.
if (
_is_copilot_provider(agent)
and not _retry.copilot_stale_cred_retry_attempted
and _is_stale_copilot_credential_error(
status_code, str(getattr(api_error, "message", "") or api_error)
)
):
_retry.copilot_stale_cred_retry_attempted = True
if agent._try_recover_stale_copilot_credential():
agent._buffer_vprint(
"🔐 Copilot credential re-exchanged after "
"model_not_available 400. Retrying request..."
)
retry_count = 0
continue
# Try fallback before aborting; announce it only when a fallback
# chain exists, else "trying fallback..." lies before a silent abort
# (#35314).
if agent._has_pending_fallback():
if classified.reason == FailoverReason.content_policy_blocked:
agent._buffer_status("⚠️ Provider safety filter blocked this request — trying fallback...")
elif classified.reason == FailoverReason.ssl_cert_verification:
agent._buffer_status("⚠️ TLS certificate verification failed — trying fallback...")
else:
agent._buffer_status(f"⚠️ Non-retryable error (HTTP {status_code}) — trying fallback...")
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
if api_kwargs is not None:
agent._dump_api_request_debug(
api_kwargs, reason="non_retryable_client_error", error=api_error,
)
# Terminal — flush buffered context so the user sees
# what was tried before the abort.
agent._flush_status_buffer()
# Summarize once: Cloudflare/proxy HTML pages and raw provider
# bodies must be collapsed here or they leak verbatim via the
# ``error`` field.
_nonretryable_summary = agent._summarize_api_error(api_error)
if classified.reason == FailoverReason.content_policy_blocked:
agent._emit_status(
f"❌ Provider safety filter blocked this request: "
f"{_nonretryable_summary}"
)
elif classified.reason == FailoverReason.ssl_cert_verification:
agent._emit_status(
f"❌ TLS certificate verification failed: "
f"{_nonretryable_summary}"
)
else:
agent._emit_status(
f"❌ Non-retryable error (HTTP {status_code}): "
f"{_nonretryable_summary}"
)
agent._vprint(f"{agent.log_prefix}❌ Non-retryable client error (HTTP {status_code}). Aborting.", force=True)
agent._vprint(f"{agent.log_prefix} 🔌 Provider: {_provider} Model: {_model}", force=True)
agent._vprint(f"{agent.log_prefix} 🌐 Endpoint: {_base}", force=True)
# Actionable guidance for common auth errors
if classified.is_auth or classified.reason == FailoverReason.billing:
if classified.reason == FailoverReason.billing and _print_billing_or_entitlement_guidance(
agent,
capability="model access",
provider=_provider,
base_url=str(_base),
model=_model,
unverified=classified.billing_unverified,
):
pass
elif _provider == "nous" and _print_nous_entitlement_guidance(
agent,
"Nous model access",
):
pass
elif _provider in {"openai-codex", "xai-oauth", "nous"} and status_code == 401:
if _provider == "openai-codex":
agent._vprint(f"{agent.log_prefix} 💡 Codex OAuth token was rejected (HTTP 401). Your token may have been", force=True)
agent._vprint(f"{agent.log_prefix} refreshed by another client (Codex CLI, VS Code). To fix:", force=True)
agent._vprint(f"{agent.log_prefix} 1. Run `codex` in your terminal to generate fresh tokens.", force=True)
agent._vprint(f"{agent.log_prefix} 2. Then run `hermes auth` to re-authenticate.", force=True)
elif _provider == "xai-oauth":
agent._vprint(f"{agent.log_prefix} 💡 xAI OAuth token was rejected (HTTP 401). To fix:", force=True)
agent._vprint(f"{agent.log_prefix} re-authenticate with xAI Grok OAuth (SuperGrok / Premium+) from `hermes model`.", force=True)
else: # nous
agent._vprint(f"{agent.log_prefix} 💡 Nous Portal OAuth token was rejected (HTTP 401). Your token may be", force=True)
agent._vprint(f"{agent.log_prefix} expired, revoked, or your account may be out of credits. To fix:", force=True)
agent._vprint(f"{agent.log_prefix} 1. Re-authenticate: hermes portal", force=True)
agent._vprint(f"{agent.log_prefix} 2. Check your portal account: https://portal.nousresearch.com", force=True)
# ``:free`` is OpenRouter slug syntax; Nous Portal will reject
# the model name even after a successful re-auth.
if isinstance(_model, str) and _model.endswith(":free"):
agent._vprint(f"{agent.log_prefix} ⚠️ Note: `{_model}` looks like an OpenRouter slug (`:free` suffix).", force=True)
agent._vprint(f"{agent.log_prefix} Nous Portal won't recognize that model name. Either switch to a", force=True)
agent._vprint(f"{agent.log_prefix} Nous catalog model, or run `/model openrouter:{_model}` to use OpenRouter.", force=True)
else:
agent._vprint(f"{agent.log_prefix} 💡 Your API key was rejected by the provider. Check:", force=True)
agent._vprint(f"{agent.log_prefix} • Is the key valid? Run: hermes setup", force=True)
agent._vprint(f"{agent.log_prefix} • Does your account have access to {_model}?", force=True)
if base_url_host_matches(str(_base), "openrouter.ai"):
agent._vprint(f"{agent.log_prefix} • Check credits: https://openrouter.ai/settings/credits", force=True)
else:
agent._vprint(f"{agent.log_prefix} 💡 This type of error won't be fixed by retrying.", force=True)
# Content-policy blocks get their own guidance: the provider refused
# this prompt, so recovery is a rephrase or another model, not
# key/retry advice.
if classified.reason == FailoverReason.content_policy_blocked:
agent._vprint(
f"{agent.log_prefix} 💡 The provider's safety filter rejected this specific prompt.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} • Try rephrasing the request, narrowing the context, or splitting into smaller steps.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} • Configure a fallback provider so future blocks route automatically:",
force=True,
)
agent._vprint(
f"{agent.log_prefix} hermes fallback add (interactive picker — same as `hermes model`)",
force=True,
)
# TLS certificate failures are environment problems — name the knobs
# that fix each common cause.
if classified.reason == FailoverReason.ssl_cert_verification:
agent._vprint(
f"{agent.log_prefix} 💡 The TLS certificate chain could not be verified. This fails the same",
force=True,
)
agent._vprint(
f"{agent.log_prefix} way on every retry — fix the environment, then try again:",
force=True,
)
agent._vprint(
f"{agent.log_prefix} • Corporate TLS-inspecting proxy? Point Python at its CA bundle:",
force=True,
)
agent._vprint(
f"{agent.log_prefix} export SSL_CERT_FILE=/path/to/corp-ca.pem (also REQUESTS_CA_BUNDLE)",
force=True,
)
agent._vprint(
f"{agent.log_prefix} • Missing/stale system CA store? Install/refresh it:",
force=True,
)
agent._vprint(
f"{agent.log_prefix} pip install --upgrade certifi (macOS: run 'Install Certificates.command')",
force=True,
)
agent._vprint(
f"{agent.log_prefix} • Self-signed local endpoint (llama.cpp, LM Studio, vLLM)? Use http://",
force=True,
)
agent._vprint(
f"{agent.log_prefix} for localhost, or add the server's cert to your trust store.",
force=True,
)
logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error)
# Skip persistence on likely context-overflow (400 + large session):
# persisting the failed message grows the session and repeats the
# failure. (#1630)
if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80):
agent._vprint(
f"{agent.log_prefix}⚠️ Skipping session persistence "
f"for large failed session to prevent growth loop.",
force=True,
)
else:
agent._persist_session(messages, conversation_history)
if classified.reason == FailoverReason.content_policy_blocked:
_policy_response = (
"⚠️ The model provider's safety filter blocked this request "
"(not a Hermes/gateway failure).\n\n"
f"Provider message: {_nonretryable_summary}\n\n"
f"{_CONTENT_POLICY_RECOVERY_HINT}"
)
return _content_policy_blocked_result(
messages,
api_call_count,
final_response=_policy_response,
error_detail=_nonretryable_summary,
)
# Billing walls get the same structured recovery descriptor as the
# max-retries path so every surface renders one consistent signal.
if classified.reason == FailoverReason.billing:
return _billing_failure_result(
classified=classified,
summary=_nonretryable_summary,
messages=messages,
api_call_count=api_call_count,
provider=_provider,
base_url=_base,
model=_model,
)
return {
"final_response": _nonretryable_summary,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": _nonretryable_summary,
}
if retry_count >= max_retries:
# Before fallback, rebuild the primary client once for transient
# transport errors (stale pool, TCP reset). Once per API call block.
if not _retry.primary_recovery_attempted and agent._try_recover_primary_transport(
api_error, retry_count=retry_count, max_retries=max_retries,
):
_retry.primary_recovery_attempted = True
retry_count = 0
# Transport recovery starts a fresh attempt cycle: re-open
# fallback state so a follow-on 429 can still activate
# fallback_providers.
_retry.has_retried_429 = False
agent._fallback_index = 0
agent._fallback_activated = False
continue
# Try fallback before giving up entirely
if agent._has_pending_fallback():
agent._buffer_status(f"⚠️ Max retries ({max_retries}) exhausted — trying fallback...")
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
_retry.restart_with_rebuilt_messages = True
break
# Terminal — flush buffered retry/fallback trace.
agent._flush_status_buffer()
_final_summary = agent._summarize_api_error(api_error)
_billing_guidance = ""
if classified.reason == FailoverReason.billing:
if classified.billing_unverified:
# Ambiguous body (#82154) — hedge the terminal line.
agent._emit_status(
"❌ Provider reported usage/credit exhaustion "
f"(unverified — may be a content-filter rejection) — {_final_summary}"
)
else:
agent._emit_status(f"❌ Billing or credits exhausted — {_final_summary}")
_billing_guidance = _billing_or_entitlement_message(
capability="model access",
provider=_provider,
base_url=str(_base),
model=_model,
unverified=classified.billing_unverified,
)
_print_billing_or_entitlement_guidance(
agent,
capability="model access",
provider=_provider,
base_url=str(_base),
model=_model,
unverified=classified.billing_unverified,
)
elif is_rate_limited:
agent._emit_status(f"❌ Rate limited after {max_retries} retries — {_final_summary}")
else:
agent._emit_status(f"❌ API failed after {max_retries} retries — {_final_summary}")
agent._vprint(f"{agent.log_prefix} 💀 Final error: {_final_summary}", force=True)
# SSE stream-drop (e.g. "Network connection lost"): usually a
# proxy/CDN cutting a very large tool call mid-response; give
# actionable guidance.
_is_stream_drop = (
not getattr(api_error, "status_code", None)
and any(p in error_msg for p in (
"connection lost", "connection reset",
"connection closed", "network connection",
"network error", "terminated",
))
)
if _is_stream_drop:
agent._vprint(
f"{agent.log_prefix} 💡 The provider's stream "
f"connection keeps dropping. This often happens "
f"when the model tries to write a very large "
f"file in a single tool call.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} Try asking the model "
f"to use execute_code with Python's open() for "
f"large files, or to write the file in smaller "
f"sections.",
force=True,
)
# Thinking-timeout: a known reasoning model hit a transport error
# before the first content token. Distinct from _is_stream_drop;
# detection lives in agent.thinking_timeout_guidance. (#52310)
from agent.thinking_timeout_guidance import (
is_thinking_timeout,
)
_is_thinking_timeout = is_thinking_timeout(
classified,
_model,
error_msg,
)
if _is_thinking_timeout:
agent._vprint(
f"{agent.log_prefix} 💡 The model's thinking "
f"phase exceeded the upstream proxy's idle "
f"timeout before the first content token "
f"arrived. This is a known issue with "
f"reasoning models behind cloud gateways "
f"(NVIDIA NIM, OpenAI, Anthropic, DeepSeek).",
force=True,
)
agent._vprint(
f"{agent.log_prefix} Workarounds in priority order:",
force=True,
)
agent._vprint(
f"{agent.log_prefix} 1. Set "
f"`providers.{_provider}.models.{_model}.stale_timeout_seconds: 900` "
f"in `~/.hermes/config.yaml` to extend the per-call "
f"timeout. (Hermes's built-in floor is 600s for "
f"known reasoning models — if you still see this "
f"after raising, the upstream cap is even shorter.)",
force=True,
)
agent._vprint(
f"{agent.log_prefix} 2. Lower `reasoning_budget` or set "
f"`reasoning_effort: medium` on this model if the provider supports it.",
force=True,
)
agent._vprint(
f"{agent.log_prefix} 3. Use a smaller / faster reasoning "
f"model if the task doesn't require deep thinking.",
force=True,
)
logger.error(
"%sAPI call failed after %s retries. %s | provider=%s model=%s msgs=%s tokens=~%s",
agent.log_prefix, max_retries, _final_summary,
_provider, _model, len(api_messages), f"{approx_tokens:,}",
)
if api_kwargs is not None:
agent._dump_api_request_debug(
api_kwargs, reason="max_retries_exhausted", error=api_error,
)
agent._persist_session(messages, conversation_history)
_billing_block = None
_billing_unverified = False
if classified.reason == FailoverReason.billing:
_billing_unverified = classified.billing_unverified
_final_response = _billing_terminal_label(
_final_summary, _billing_unverified
)
if _billing_guidance:
_final_response += f"\n\n{_billing_guidance}"
# Structured recovery descriptor so every surface renders
# the same link + label from one signal (see helper).
_billing_block = _billing_block_dict(
_provider, _base, _model, _billing_guidance,
unverified=_billing_unverified,
)
else:
_final_response = f"API call failed after {max_retries} retries: {_final_summary}"
if _is_thinking_timeout:
# Thinking-timeout guidance overrides stream-drop guidance,
# which would wrongly suggest splitting large file writes.
from agent.thinking_timeout_guidance import (
build_thinking_timeout_guidance,
)
_final_response += build_thinking_timeout_guidance(
provider=_provider,
model=_model,
)
elif _is_stream_drop:
_final_response += (
"\n\nThe provider's stream connection keeps "
"dropping — this often happens when generating "
"very large tool call responses (e.g. write_file "
"with long content). Try asking me to use "
"execute_code with Python's open() for large "
"files, or to write in smaller sections."
)
return {
"final_response": _final_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": _final_summary,
# Expose the classified reason so callers (kanban worker in
# cli.py) can tell a quota wall (``rate_limit`` / ``billing``)
# from a task failure.
"failure_reason": classified.reason.value,
# The classifier's own retry verdict — UI surfaces use
# this instead of re-deriving from the reason string.
"failure_retryable": bool(classified.retryable),
# True when the billing verdict rests on an ambiguous
# body (#82154) — may be a content-filter rejection.
"billing_unverified": _billing_unverified,
# Present only for billing walls: structured recovery
# descriptor (provider, billing_url, is_nous, message).
"billing_block": _billing_block,
}
# For rate limits, respect the Retry-After header if present
_retry_after = None
if is_rate_limited:
_resp_headers = getattr(getattr(api_error, "response", None), "headers", None)
if _resp_headers and hasattr(_resp_headers, "get"):
_ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After")
if _ra_raw:
try:
# Cap at 600s: Anthropic Tier 1 buckets reset in ~171s,
# so a 120s cap retried early and re-tripped the limit.
# (#26293)
_retry_after = min(float(_ra_raw), 600)
except (TypeError, ValueError):
pass
wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0)
_backoff_policy = None
if (is_rate_limited or _is_zai_coding_overload) and not _retry_after:
wait_time, _backoff_policy = adaptive_rate_limit_backoff(
retry_count,
base_url=str(_base),
model=_model,
error=api_error,
default_wait=wait_time,
)
if is_rate_limited or _is_zai_coding_overload:
_policy_note = ""
if _backoff_policy == "zai_coding_overload_long":
_policy_note = " (Z.AI Coding overload adaptive long backoff)"
elif _backoff_policy == "zai_coding_overload_short":
_policy_note = " (Z.AI Coding overload short retry)"
_wait_reason = "Provider overloaded" if _is_zai_coding_overload and not is_rate_limited else "Rate limited"
_rate_limit_status = f"⏱️ {_wait_reason}. Waiting {wait_time:.1f}s (attempt {retry_count + 1}/{max_retries}){_policy_note}..."
# Normal retries are buffered to avoid chatter; long Z.AI Coding
# waits can last minutes, so surface progress immediately.
if _backoff_policy == "zai_coding_overload_long":
agent._emit_status(_rate_limit_status)
else:
agent._buffer_status(_rate_limit_status)
else:
agent._buffer_status(f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})...")
logger.warning(
"Retrying API call in %ss (attempt %s/%s) %s policy=%s error=%s",
wait_time,
retry_count,
max_retries,
agent._client_log_context(),
_backoff_policy or "default",
api_error,
)
# Sleep in small increments so we can respond to interrupts quickly
# instead of blocking the entire wait_time in one sleep() call
sleep_end = time.time() + wait_time
_backoff_touch_counter = 0
while time.time() < sleep_end:
if agent._interrupt_requested:
# Same preserve-redirect rule as the retry-wait above: a
# steering correction must survive backoff, not die as
# "Operation interrupted".
if agent.clear_interrupt(preserve_redirect=True):
_retry.restart_with_redirected_messages = True
break
agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during retry wait, aborting.", force=True)
_interrupt_text = f"Operation interrupted: retrying API call after error (retry {retry_count}/{max_retries})."
close_interrupted_tool_sequence(messages, _interrupt_text)
agent._persist_session(messages, conversation_history)
agent.clear_interrupt()
return {
"final_response": _interrupt_text,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"interrupted": True,
}
time.sleep(0.2) # Check interrupt every 200ms
# Touch activity every ~30s so the gateway's inactivity
# monitor knows we're alive during backoff waits.
_backoff_touch_counter += 1
if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s
agent._touch_activity(
f"error retry backoff ({retry_count}/{max_retries}), "
f"{int(sleep_end - time.time())}s remaining"
)
if _retry.restart_with_redirected_messages:
# Leave the retry loop — the check below rebuilds this iteration
# from the correction instead of re-firing the stale request.
break
if _retry.restart_with_redirected_messages:
# Cancelled request produced no valid assistant item: reuse the same logical
# iteration after the outer loop appends partial context + correction.
api_call_count -= 1
agent.iteration_budget.refund()
_retry.restart_with_redirected_messages = False
continue
# If the API call was interrupted, skip response processing
if interrupted:
_turn_exit_reason = "interrupted_during_api_call"
break
if _retry.restart_with_compressed_messages:
api_call_count -= 1
agent.iteration_budget.refund()
# Compression restarts count toward the retry limit so a compression that
# shrinks messages but not enough can't loop forever.
retry_count += 1
_retry.restart_with_compressed_messages = False
if _should_skip_model_call_for_reference_handoff(
messages, user_message
):
logger.info(
"Skipping compressed-restart model call: reference-only "
"handoff would be the sole active user turn (#80622)"
)
if not final_response:
final_response = _HANDOFF_SKIP_FINAL_RESPONSE
_turn_exit_reason = "compaction_handoff_not_actionable"
break
# In-loop compression rebuilt `messages`; re-anchor the current-turn index
# like the prologue, AFTER the handoff guard (it may re-append this turn's
# ask). A stale anchor injects prefetch into a historical row.
current_turn_user_idx = reanchor_current_turn_user_idx(
messages, user_message
)
agent._persist_user_message_idx = current_turn_user_idx
continue
if _retry.restart_with_rebuilt_messages:
# A stall/failure escalated to the fallback chain: re-issue against the
# active fallback provider, refunding budget/count for the stalled attempt.
api_call_count -= 1
agent.iteration_budget.refund()
_retry.restart_with_rebuilt_messages = False
# Failover shrank the compressor window: clear the preflight block so
# preflight re-runs before the first fallback call. Hoisted to the single
# consumer. (#84733)
_preflight_compression_blocked = False
continue
if _retry.restart_with_length_continuation:
# Boost output budget per retry: 2×, 4×, 8×, 16× base, capped at 32 768, via
# _ephemeral_max_output_tokens. Keep a larger original provider/model
# default as the floor so retries never downshift.
_boost_base = agent.max_tokens if agent.max_tokens else 4096
_boost = _boost_base * (2 ** length_continue_retries)
_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs)
if _requested_cap is not None:
_boost = max(_boost, _requested_cap)
_boost_cap = max(32768, _requested_cap or 0)
agent._ephemeral_max_output_tokens = min(_boost, _boost_cap)
continue
# All retries may exhaust with `response` still None; break out cleanly.
if response is None:
_turn_exit_reason = "all_retries_exhausted_no_response"
print(f"{agent.log_prefix}❌ All API retries exhausted with no successful response.")
agent._persist_session(messages, conversation_history)
break
try:
_transport = agent._get_transport()
_normalize_kwargs = {}
if agent.api_mode == "anthropic_messages":
_normalize_kwargs["strip_tool_prefix"] = agent._is_anthropic_oauth
normalized = _transport.normalize_response(response, **_normalize_kwargs)
assistant_message = normalized
finish_reason = normalized.finish_reason
# Some OpenAI-compatible servers (llama-server) return content as dict/list,
# which crashes downstream .strip(); normalize to str.
if assistant_message.content is not None and not isinstance(assistant_message.content, str):
raw = assistant_message.content
if isinstance(raw, dict):
assistant_message.content = raw.get("text", "") or raw.get("content", "") or json.dumps(raw)
elif isinstance(raw, list):
# Multimodal content list — extract text parts
parts = []
for part in raw:
if isinstance(part, str):
parts.append(part)
elif isinstance(part, dict) and part.get("type") == "text":
parts.append(part.get("text", ""))
elif isinstance(part, dict) and "text" in part:
parts.append(str(part["text"]))
assistant_message.content = "\n".join(parts)
else:
assistant_message.content = str(raw)
# ── Agent-as-provider projection ──────────────────────────────
# Splice the provider-agent's own tool work in as call/result rows before
# this turn's assistant message; no-op for ordinary providers.
splice_provider_projection(agent, response, messages)
try:
from hermes_cli.lifecycle import (
has_hook,
invoke_hook as _invoke_hook,
)
if has_hook("post_api_request"):
_assistant_tool_calls = (
getattr(assistant_message, "tool_calls", None) or []
)
_assistant_text = assistant_message.content or ""
_api_ended_at = api_start_time + api_duration
_invoke_hook(
"post_api_request",
task_id=effective_task_id,
turn_id=turn_id,
api_request_id=api_request_id,
session_id=agent.session_id or "",
platform=agent.platform or "",
model=agent.model,
provider=agent.provider,
base_url=agent.base_url,
api_mode=agent.api_mode,
api_call_count=api_call_count,
api_duration=api_duration,
started_at=api_start_time,
ended_at=_api_ended_at,
# First stream chunk time (epoch s) from
# interruptible_streaming_api_call; None if not streamed / no
# chunk. TTFB = first_chunk_at - started_at.
first_chunk_at=getattr(
agent, "_last_api_first_chunk_at", None
),
finish_reason=finish_reason,
message_count=len(api_messages),
response_model=getattr(response, "model", None),
response=agent._api_response_payload_for_hook(
response,
assistant_message,
finish_reason=finish_reason,
),
usage=agent._usage_summary_for_api_request_hook(response),
assistant_message=assistant_message,
assistant_content_chars=len(_assistant_text),
assistant_tool_call_count=len(_assistant_tool_calls),
moa_references=_moa_reference_metrics_for_hook(agent),
)
except Exception:
pass
# Handle assistant response
if assistant_message.content and not agent.quiet_mode:
if agent.verbose_logging:
agent._vprint(f"{agent.log_prefix}🤖 Assistant: {assistant_message.content}")
else:
agent._vprint(f"{agent.log_prefix}🤖 Assistant: {assistant_message.content[:100]}{'...' if len(assistant_message.content) > 100 else ''}")
# Notify progress callback of model's thinking (used by subagent
# delegation to relay the child's reasoning to the parent display).
if (assistant_message.content and agent.tool_progress_callback):
_think_text = assistant_message.content.strip()
# Strip reasoning XML tags that shouldn't leak to parent display
_think_text = re.sub(
r'</?(?:REASONING_SCRATCHPAD|think|reasoning)>', '', _think_text
).strip()
# For subagents: relay first line to parent display (existing behaviour).
# For all agents with a structured callback: emit reasoning.available event.
first_line = _think_text.split('\n')[0][:80] if _think_text else ""
if first_line and getattr(agent, '_delegate_depth', 0) > 0:
try:
agent.tool_progress_callback("_thinking", first_line)
except Exception:
pass
elif _think_text:
try:
agent.tool_progress_callback("reasoning.available", "_thinking", _think_text[:500], None)
except Exception:
pass
# Check for incomplete <REASONING_SCRATCHPAD> (opened but never closed)
# This means the model ran out of output tokens mid-reasoning — retry up to 2 times
if has_incomplete_scratchpad(assistant_message.content or ""):
agent._incomplete_scratchpad_retries += 1
agent._buffer_vprint("⚠️ Incomplete <REASONING_SCRATCHPAD> detected (opened but never closed)")
if agent._incomplete_scratchpad_retries <= 2:
agent._buffer_vprint(f"🔄 Retrying API call ({agent._incomplete_scratchpad_retries}/2)...")
# Don't add the broken message, just retry
continue
else:
# Max retries - discard this turn and save as partial
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max retries (2) for incomplete scratchpad. Saving as partial.", force=True)
agent._incomplete_scratchpad_retries = 0
rolled_back_messages = agent._get_messages_up_to_last_assistant(messages)
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": "Incomplete REASONING_SCRATCHPAD after 2 retries",
"messages": rolled_back_messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": "Incomplete REASONING_SCRATCHPAD after 2 retries"
}
# Reset incomplete scratchpad counter on clean response
agent._incomplete_scratchpad_retries = 0
if agent.api_mode == "codex_responses" and finish_reason == "incomplete":
agent._codex_incomplete_retries += 1
interim_msg = agent._build_assistant_message(assistant_message, finish_reason)
interim_has_content = bool((interim_msg.get("content") or "").strip())
interim_has_reasoning = bool(interim_msg.get("reasoning", "").strip()) if isinstance(interim_msg.get("reasoning"), str) else False
interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items"))
interim_has_codex_message_items = bool(interim_msg.get("codex_message_items"))
if (
interim_has_content
or interim_has_reasoning
or interim_has_codex_reasoning
or interim_has_codex_message_items
):
last_msg = messages[-1] if messages else None
# Dedup on visible content only (content + reasoning): opaque
# provider state drifts per continuation and would defeat dedup
# (#52711).
last_interim_visible = (
agent._interim_assistant_visible_text(last_msg)
if isinstance(last_msg, dict)
else ""
)
current_interim_visible = agent._interim_assistant_visible_text(interim_msg)
if last_interim_visible or current_interim_visible:
same_visible_output = last_interim_visible == current_interim_visible
else:
# Preserve the existing reasoning-only behavior when
# neither response has text eligible for interim delivery.
same_visible_output = (
(last_msg.get("content") or "") == (interim_msg.get("content") or "")
and (last_msg.get("reasoning") or "") == (interim_msg.get("reasoning") or "")
) if isinstance(last_msg, dict) else False
visible_duplicate = (
isinstance(last_msg, dict)
and last_msg.get("role") == "assistant"
and last_msg.get("finish_reason") == "incomplete"
and same_visible_output
)
if visible_duplicate:
# Update replay state in-place: keep the latest provider payload
# without re-emitting identical user-visible commentary.
for _key in (
"content",
"reasoning",
"reasoning_content",
"reasoning_details",
"codex_reasoning_items",
"codex_message_items",
):
if _key in interim_msg:
if _key == "codex_reasoning_items":
# Merge, don't overwrite: the earlier response's
# native compaction checkpoint is the only copy. See
# merge_interim_reasoning_items.
from agent.native_compaction import (
merge_interim_reasoning_items,
)
last_msg[_key] = merge_interim_reasoning_items(
last_msg.get(_key), interim_msg[_key]
)
else:
last_msg[_key] = interim_msg[_key]
else:
append_message(messages, interim_msg)
agent._emit_interim_assistant_message(interim_msg)
if agent._codex_incomplete_retries < 3:
# If the interim has nothing the Responses converter will replay, a
# bare retry is byte-identical and fails identically; append a
# user-role nudge so the retry differs and asks for the answer.
interim_replayable = (
interim_has_content
or interim_has_codex_reasoning
or interim_has_codex_message_items
)
# Replayable ≠ different: an interim holding only a ``compaction``
# checkpoint in ``codex_reasoning_items`` is replayable yet re-sends
# identically. One bare retry, then always nudge.
if not interim_replayable or agent._codex_incomplete_retries >= 2:
_last_msg = messages[-1] if messages else None
_already_nudged = (
isinstance(_last_msg, dict)
and _last_msg.get("role") == "user"
and _last_msg.get("content") == _CODEX_INCOMPLETE_NUDGE
)
# Alternation guard: the user-role nudge may only follow an
# assistant message; after a too-empty interim it would create
# user→user / tool→user.
_last_is_assistant = (
isinstance(_last_msg, dict)
and _last_msg.get("role") == "assistant"
)
if not _already_nudged and _last_is_assistant:
append_message(messages, {
"role": "user",
"content": _CODEX_INCOMPLETE_NUDGE,
})
if not agent.quiet_mode:
agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({agent._codex_incomplete_retries}/3)")
# Show the continuation on the spinner/status line and gateway
# heartbeat; these retries can take minutes and otherwise look like
# infinite thinking (#64434).
agent._emit_wait_notice(
f"↻ model returned reasoning with no final answer — "
f"asking it to continue "
f"({agent._codex_incomplete_retries}/3)"
)
agent._session_messages = messages
continue
agent._codex_incomplete_retries = 0
agent._persist_session(messages, conversation_history)
return {
"final_response": "Codex response remained incomplete after 3 continuation attempts",
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": "Codex response remained incomplete after 3 continuation attempts",
}
elif hasattr(agent, "_codex_incomplete_retries"):
agent._codex_incomplete_retries = 0
# Check for tool calls
if assistant_message.tool_calls:
if not agent.quiet_mode:
agent._vprint(f"{agent.log_prefix}🔧 Processing {len(assistant_message.tool_calls)} tool call(s)...")
if agent.verbose_logging:
for tc in assistant_message.tool_calls:
raw_args = tc.function.arguments
args_preview = raw_args[:200] if isinstance(raw_args, str) else repr(raw_args)[:200]
logging.debug("Tool call: %s with args: %s...", tc.function.name, args_preview)
# Uniquify duplicate tool-call ids BEFORE any downstream consumer: the
# pre-API sanitizer keeps only the first call/result per id. See
# _uniquify_tool_call_ids.
agent._uniquify_tool_call_ids(assistant_message.tool_calls)
# Validate tool call names - detect model hallucinations
# Repair mismatched tool names before validating
for tc in assistant_message.tool_calls:
if tc.function.name not in agent.valid_tool_names:
repaired = agent._repair_tool_call(tc.function.name)
if repaired:
print(f"{agent.log_prefix}🔧 Auto-repaired tool name: '{tc.function.name}' -> '{repaired}'")
tc.function.name = repaired
invalid_tool_calls = [
tc.function.name for tc in assistant_message.tool_calls
if tc.function.name not in agent.valid_tool_names
]
# Mixed batch: error-result ONLY the invalid calls and run the valid
# ones; voiding the turn discards real work. Strikes advance only when a
# turn has NO valid call, so a degenerate model still halts at 3.
_mixed_invalid_batch = bool(invalid_tool_calls) and any(
tc.function.name in agent.valid_tool_names
for tc in assistant_message.tool_calls
)
if _mixed_invalid_batch:
agent._invalid_tool_retries = 0
invalid_name = invalid_tool_calls[0]
invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name
_n_valid = sum(
1 for tc in assistant_message.tool_calls
if tc.function.name in agent.valid_tool_names
)
agent._buffer_vprint(
f"⚠️ Unknown tool '{invalid_preview}' in batch — erroring that call, "
f"executing {_n_valid} valid call(s)"
)
elif invalid_tool_calls:
# Track retries for invalid tool calls
agent._invalid_tool_retries += 1
# Return helpful error to model — model can agent-correct next turn
invalid_name = invalid_tool_calls[0]
invalid_preview = invalid_name[:80] + "..." if len(invalid_name) > 80 else invalid_name
agent._buffer_vprint(f"⚠️ Unknown tool '{invalid_preview}' — sending error to model for agent-correction ({agent._invalid_tool_retries}/3)")
if agent._invalid_tool_retries >= 3:
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max retries (3) for invalid tool calls exceeded. Stopping as partial.", force=True)
agent._invalid_tool_retries = 0
_final_response = f"Model generated invalid tool call: {invalid_preview}"
# Prior retries or an earlier tool batch leave a tool-result
# tail; close it as interrupt aborts do so the next turn is not
# tool→user. (#48879)
close_interrupted_tool_sequence(messages, _final_response)
agent._persist_session(messages, conversation_history)
return {
"final_response": _final_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": _final_response
}
assistant_msg = agent._build_assistant_message(assistant_message, finish_reason)
append_message(messages, assistant_msg)
for tc in assistant_message.tool_calls:
_tc_name = tc.function.name
if _tc_name not in agent.valid_tool_names:
# See _invalid_tool_name_error_content for the
# blank-name anti-priming rationale (#47967).
content = _invalid_tool_name_error_content(
_tc_name, agent.valid_tool_names
)
else:
content = "Skipped: another tool call in this turn used an invalid name. Please retry this tool call."
append_message(messages, {
"role": "tool",
"name": tc.function.name,
"tool_call_id": coalesce_tool_call_id(tc),
"content": content,
})
continue
# Reset retry counter on successful tool call validation
agent._invalid_tool_retries = 0
# Validate tool call arguments are valid JSON
# Handle empty strings as empty objects (common model quirk)
invalid_json_args = []
for tc in assistant_message.tool_calls:
args = tc.function.arguments
if isinstance(args, (dict, list)):
tc.function.arguments = json.dumps(args)
continue
if args is not None and not isinstance(args, str):
tc.function.arguments = str(args)
args = tc.function.arguments
# Treat empty/whitespace strings as empty object
if not args or not args.strip():
tc.function.arguments = "{}"
continue
try:
json.loads(args)
except json.JSONDecodeError as e:
if (
_mixed_invalid_batch
and tc.function.name not in agent.valid_tool_names
):
# This call never executes (invalid-name error result
# below); don't let its broken args trigger the whole-turn
# JSON retry.
continue
invalid_json_args.append((tc.function.name, str(e)))
if invalid_json_args:
# Routers may rewrite finish_reason "length" → "tool_calls", hiding
# truncation; args not ending in } or ] (stripped) were cut off
# mid-stream.
_truncated = any(
not (tc.function.arguments or "").rstrip().endswith(("}", "]"))
for tc in assistant_message.tool_calls
if tc.function.name in {n for n, _ in invalid_json_args}
)
if _truncated:
agent._vprint(
f"{agent.log_prefix}⚠️ Truncated tool call arguments detected "
f"(finish_reason={finish_reason!r}) — refusing to execute.",
force=True,
)
agent._invalid_json_retries = 0
agent._cleanup_task_resources(effective_task_id)
_final_response = "Response truncated due to output length limit"
# Same tool-tail close as interrupt / invalid-tool
# exhaustion — this path never reaches finalize_turn.
close_interrupted_tool_sequence(messages, _final_response)
agent._persist_session(messages, conversation_history)
return {
"final_response": _final_response,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"partial": True,
"error": _final_response,
}
# Track retries for invalid JSON arguments
agent._invalid_json_retries += 1
tool_name, error_msg = invalid_json_args[0]
agent._buffer_vprint(f"⚠️ Invalid JSON in tool call arguments for '{tool_name}': {error_msg}")
if agent._invalid_json_retries < 3:
agent._buffer_vprint(f"🔄 Retrying API call ({agent._invalid_json_retries}/3)...")
# Don't add anything to messages, just retry the API call
continue
else:
# Instead of returning partial, inject tool error results so the model can recover.
# Using tool results (not user messages) preserves role alternation.
agent._buffer_vprint("⚠️ Injecting recovery tool results for invalid JSON...")
agent._invalid_json_retries = 0 # Reset for next attempt
# Append the assistant message with its (broken) tool_calls
recovery_assistant = agent._build_assistant_message(assistant_message, finish_reason)
append_message(messages, recovery_assistant)
# Respond with tool error results for each tool call
invalid_names = {name for name, _ in invalid_json_args}
for tc in assistant_message.tool_calls:
if tc.function.name in invalid_names:
err = next(e for n, e in invalid_json_args if n == tc.function.name)
tool_result = (
f"Error: Invalid JSON arguments. {err}. "
f"For tools with no required parameters, use an empty object: {{}}. "
f"Please retry with valid JSON."
)
else:
tool_result = "Skipped: other tool call in this response had invalid JSON."
append_message(messages, {
"role": "tool",
"name": tc.function.name,
"tool_call_id": coalesce_tool_call_id(tc),
"content": tool_result,
})
continue
# Reset retry counter on successful JSON validation
agent._invalid_json_retries = 0
# ── Post-call guardrails ──────────────────────────
assistant_message.tool_calls = agent._cap_delegate_task_calls(
assistant_message.tool_calls
)
assistant_message.tool_calls = agent._deduplicate_tool_calls(
assistant_message.tool_calls
)
# Collect invalid calls so the assistant message keeps EVERY emitted
# call (each tool_call needs a matching result) while only valid ones
# dispatch.
_invalid_batch_calls = []
if _mixed_invalid_batch:
_invalid_batch_calls = [
tc for tc in assistant_message.tool_calls
if tc.function.name not in agent.valid_tool_names
]
assistant_msg = agent._build_assistant_message(assistant_message, finish_reason)
turn_content = assistant_message.content or ""
# A bare bracketed token (e.g. ``[memory]``) beside a function call is
# protocol scaffolding; persisting it lets the post-tool fallback replay
# it forever (#78148).
if (
assistant_message.tool_calls
and _STALE_MARKER_RE.fullmatch(turn_content.strip())
):
logger.warning(
"Discarding bare tool-call marker from assistant content: %s",
turn_content,
)
turn_content = ""
assistant_msg["content"] = ""
# Classify tools regardless of visible content: a substantive tool-only
# turn must invalidate any older housekeeping fallback.
_HOUSEKEEPING_TOOLS = frozenset({
"memory", "todo_list", "skill_manage", "session_search",
})
_all_housekeeping = all(
tc.function.name in _HOUSEKEEPING_TOOLS
for tc in assistant_message.tool_calls
)
# Substantive tools clear any older fallback so a two-turn-old
# housekeeping narration isn't attributed to the preceding tool turn.
if assistant_message.tool_calls and not _all_housekeeping:
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
# Also clear the mute flag a prior housekeeping turn may have set,
# else _vprint suppresses this turn's tool progress until the
# no-tool-call branch clears it.
agent._mute_post_response = False
# Content + tool_calls in one turn: keep the content as a fallback final
# response in case the follow-up turn after tools is empty.
if turn_content and agent._has_content_after_think_block(turn_content):
agent._last_content_with_tools = turn_content
# Mute only when EVERY tool call is post-response housekeeping
# (memory, todo, skill_manage); substantive tools keep output on.
agent._last_content_tools_all_housekeeping = _all_housekeeping
if _all_housekeeping and agent._has_stream_consumers():
agent._mute_post_response = True
elif agent._should_emit_quiet_tool_messages():
clean = agent._strip_think_blocks(turn_content).strip()
if clean:
agent._vprint(f" ┊ 💬 {clean}")
# Pop thinking-only prefill message(s) before appending
# (tool-call path — same rationale as the final-response path).
_had_prefill = False
while (
messages
and isinstance(messages[-1], dict)
and messages[-1].get("_thinking_prefill")
):
messages.pop()
_had_prefill = True
# Tool calls after a prefill recovery reset the prefill counter, so
# each tool-call success is a fresh start, not a cumulative burn.
if _had_prefill:
agent._thinking_prefill_retries = 0
agent._empty_content_retries = 0
# Re-arm the post-tool nudge so it can fire on a LATER tool round.
agent._post_tool_empty_retried = False
# A landed tool call recovers any dropped-tool-call stall; refresh that
# budget so it guards each stall independently, not the whole run.
agent._dropped_toolcall_retries = 0
previous_msg = messages[-1] if messages else None
current_interim_visible = agent._interim_assistant_visible_text(assistant_msg)
previous_interim_visible = (
agent._interim_assistant_visible_text(previous_msg)
if isinstance(previous_msg, dict)
else ""
)
duplicate_previous_interim = (
bool(current_interim_visible)
and isinstance(previous_msg, dict)
and previous_msg.get("role") == "assistant"
and previous_msg.get("finish_reason") == "incomplete"
and previous_interim_visible == current_interim_visible
)
append_message(messages, assistant_msg)
# Mixed batch: error-result invalid calls and drop them from execution.
# The assistant message keeps all calls so tool_call/result pairs hold.
if _invalid_batch_calls:
for tc in _invalid_batch_calls:
append_message(messages, {
"role": "tool",
"name": tc.function.name,
"tool_call_id": coalesce_tool_call_id(tc),
"content": _invalid_tool_name_error_content(
tc.function.name, agent.valid_tool_names
),
})
assistant_message.tool_calls = [
tc for tc in assistant_message.tool_calls
if tc.function.name in agent.valid_tool_names
]
_tool_turn_persisted = None
try:
# Persist the tool-call turn before any tool side effects so resume
# sees the executed block if a destructive tool restarts Hermes.
_tool_turn_persisted = agent._flush_messages_to_session_db(
messages, conversation_history
)
except Exception as exc:
_tool_turn_persisted = False
from hermes_state import classify_persistence_error
agent._last_persistence_error_cause = (
classify_persistence_error(exc)
)
logger.warning(
"Incremental tool-call persistence failed before execution "
"(session=%s): %s",
agent.session_id or "none",
exc,
)
if _tool_turn_persisted is False:
# Canonical append failed: never project the row or run tools from
# process-only state; break rather than retry the unpersisted turn.
# If the flush recorded no cause, the cause is genuinely unknown.
if getattr(agent, "_last_persistence_error_cause", None) is None:
agent._last_persistence_error_cause = "unknown"
_turn_exit_reason = "session_persistence_failed"
final_response = ""
failed = True
break
# A UI must never observe an assistant/tool-call row that is only an
# in-memory projection: emit interim commentary after the DB append.
if not duplicate_previous_interim:
agent._emit_interim_assistant_message(assistant_msg)
# Flush open streaming boxes before tools so early content doesn't wrap
# tool feed lines. Display callback only — TTS (_stream_callback) must
# NOT receive None (its end-of-stream marker).
if agent.stream_delta_callback:
try:
agent.stream_delta_callback(None)
except Exception:
pass
agent._execute_tool_calls(assistant_message, messages, effective_task_id, api_call_count)
if getattr(agent, "_incremental_persistence_failed", False):
# Tool result could not be made canonical: never send the in-memory
# result to the model or project later events from this turn.
_turn_exit_reason = "session_persistence_failed"
final_response = ""
failed = True
break
if agent._tool_guardrail_halt_decision is not None:
decision = agent._tool_guardrail_halt_decision
_turn_exit_reason = "guardrail_halt"
final_response = agent._toolguard_controlled_halt_response(decision)
agent._emit_status(
f"⚠️ Tool guardrail halted {decision.tool_name}: {decision.code}"
)
append_message(messages, {"role": "assistant", "content": final_response})
# Emit the halt message so it isn't mistaken for a crash; the stream
# callback is still alive, so SSE/TUI clients see the explanation.
if final_response:
agent._safe_print(f"\n{final_response}\n")
if agent.stream_delta_callback:
try:
agent.stream_delta_callback(final_response)
agent.stream_delta_callback(None)
except Exception:
pass
break
# Reset per-turn retry counters so one truncation can't poison the turn.
truncated_tool_call_retries = 0
# Defer the paragraph break: _fire_stream_delta() prepends one "\n\n"
# when real text arrives, so tool iterations don't stack blank lines.
agent._stream_needs_break = True
# Refund the iteration when the ONLY tool was execute_code (programmatic
# tool calling) — cheap RPC-style calls shouldn't eat the budget.
_tc_names = {tc.function.name for tc in assistant_message.tool_calls}
if _tc_names == {"execute_code"}:
agent.iteration_budget.refund()
# Decide compression from API-reported prompt tokens (tight lower bound;
# tool results get counted on the next call). If last_prompt_tokens is 0
# (disconnect / no usage data) fall back to a rough estimate. (#2153)
_compressor = agent.context_compressor
if _compressor.last_prompt_tokens > 0:
# Only prompt_tokens: thinking models inflate completion_tokens with
# reasoning that uses no context → premature compression. (#12026)
_real_tokens = _compressor.last_prompt_tokens
elif _compressor.last_prompt_tokens == -1:
# Compression just ran, no API prompt count yet: don't treat a rough
# schema-heavy post-compression estimate as real context pressure.
_real_tokens = 0
else:
# Include tool schemas (20-30K tokens the messages-only estimate
# misses) and stay route-aware: on a compacted native-Codex session
# the generic durable-history figure would false-trigger. (#14695)
_real_tokens = _midturn_request_pressure_tokens(
agent,
messages,
active_system_prompt or "",
estimate_request_tokens_rough(
messages, tools=agent.tools or None
),
)
if (
agent.compression_enabled
and compression_attempts < max_compression_attempts
and _compressor.should_compress(_real_tokens)
):
compression_attempts += 1
# Compression is running: reset blocked-overflow warning dedup so a
# future blocked turn can warn again. getattr: test doubles lack it.
_clear_warn = getattr(agent, "_clear_context_overflow_warn", None)
if callable(_clear_warn):
_clear_warn()
agent._safe_print(" ⟳ compacting context…")
_post_tool_input = messages
# Pass overhead-aware _real_tokens, not last_prompt_tokens (0 in
# the no-usage fallback), so the overflow guard sees the true size.
messages, active_system_prompt = agent._compress_context(
messages, system_message,
approx_tokens=_real_tokens,
task_id=effective_task_id,
)
if (
messages is _post_tool_input
and compression_skipped_due_to_lock(agent)
):
# Lock-skip no-op is a temporary defer, not evidence about
# compressibility: refund so a lock-loser loop doesn't burn the
# budget toward compression_exhausted. (#69870)
compression_attempts -= 1
else:
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
if _should_skip_model_call_for_reference_handoff(
messages, user_message
):
logger.info(
"Skipping post-tool compaction model call: "
"reference-only handoff would be the sole "
"active user turn (#80622)"
)
if not final_response:
final_response = _HANDOFF_SKIP_FINAL_RESPONSE
_turn_exit_reason = "compaction_handoff_not_actionable"
break
elif agent.compression_enabled:
# Over threshold but compression blocked (cooldown/anti-thrash):
# deduped warning so context can't silently overflow. (#62625)
_block_reason = None
_info = getattr(_compressor, "should_compress_info", None)
if _info is not None:
try:
_block_reason = _info(_real_tokens)[1]
except Exception:
_block_reason = None
if _block_reason:
agent._warn_context_overflow_blocked(
_block_reason,
_real_tokens,
int(getattr(_compressor, "threshold_tokens", 0) or 0),
)
# Proactive tool-result prune (deterministic, no LLM, keeps tail):
# no-op unless proactive_prune_tokens is exceeded; commits only past
# proactive_prune_min_reclaim_tokens so cache breaks stay episodic.
_prune = getattr(_compressor, "prune_tool_results_only", None)
if callable(_prune):
try:
_pruned_msgs, _pruned_n = _prune(
messages, current_tokens=_real_tokens
)
except Exception:
logger.debug(
"proactive tool-result prune failed; skipping",
exc_info=True,
)
_pruned_msgs, _pruned_n = messages, 0
# Standard no-op caller contract: only commit when the
# engine returned a NEW list object with a non-zero count.
if _pruned_n and _pruned_msgs is not messages:
# Do NOT rebuild conversation_history: rows already carry
# _DB_PERSISTED_MARKER, and on a stale in-place flag the
# helper could seed unpersisted rows into history_ids.
messages = _pruned_msgs
# Save session log incrementally (so progress is visible even if interrupted)
agent._session_messages = messages
# Touch activity so slow post-tool work plus a slow follow-up API call
# can't exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT).
agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}")
# Continue loop for next response
continue
else:
# No tool calls — final response. (Dropped tool-call recovery lives at
# the finalization chokepoint below so it catches every path.)
final_response = assistant_message.content or ""
# Unmute: _mute_post_response from a housekeeping tool turn must not
# silence empty-response warnings on the final response path.
agent._mute_post_response = False
# Check if response only has think block with no actual content after it
if not agent._has_content_after_think_block(final_response):
# Partial stream recovery: content streamed before the connection
# died becomes the final response instead of fallback or retries.
_partial_streamed = (
getattr(agent, "_current_streamed_assistant_text", "") or ""
)
if agent._has_content_after_think_block(_partial_streamed):
_turn_exit_reason = "partial_stream_recovery"
_recovered = agent._strip_think_blocks(_partial_streamed).strip()
logger.info(
"Partial stream content delivered (%d chars) "
"— using as final response",
len(_recovered),
)
agent._emit_status(
"↻ Stream interrupted — using delivered content "
"as final response"
)
final_response = _recovered
# A streamed fragment isn't a confirmed preview: keep
# response_previewed false so gateway fallback delivery can
# send the text plus the abnormal-turn explanation.
agent._response_was_previewed = False
break
# Prior turn had real content + ONLY housekeeping tools: model is
# done, reuse it. With substantive tools it was mid-task narration
# and the empty reply is a choke; let the post-tool nudge handle it.
fallback = getattr(agent, '_last_content_with_tools', None)
if fallback and getattr(agent, '_last_content_tools_all_housekeeping', False):
_turn_exit_reason = "fallback_prior_turn_content"
logger.info("Empty follow-up after tool calls — using prior turn content as final response")
agent._emit_status("↻ Empty response after tool calls — using earlier content as final answer")
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
agent._empty_content_retries = 0
# Do NOT modify the assistant message content (injected text
# poisoned history); use the fallback as the response and break.
final_response = agent._strip_think_blocks(fallback).strip()
agent._response_was_previewed = True
break
# ── Post-tool-call empty response nudge ───────────
# Empty after tool results (no prior content, or only mid-task
# narration): nudge once via a user-level hint. (#9400)
_prior_was_tool = any(
m.get("role") == "tool"
for m in messages[-5:] # check recent messages
)
# Ollama puts <think> in content, not reasoning_content, so
# _has_structured misses it; detect here to route to prefill.
_has_inline_thinking = bool(
re.search(
r'<think>|<thinking>|<reasoning>',
final_response or "",
re.IGNORECASE,
)
)
if (
_prior_was_tool
and not getattr(agent, "_post_tool_empty_retried", False)
and not _has_inline_thinking # thinking model still working — let prefill handle
):
agent._post_tool_empty_retried = True
# Clear stale narration so it doesn't resurface
# on a later empty response after the nudge.
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
logger.info(
"Empty response after tool calls — nudging model "
"to continue processing"
)
agent._buffer_status(
"⚠️ Model returned empty after tool calls — "
"nudging to continue"
)
# Append the empty assistant first so the sequence stays valid:
# tool → assistant("(empty)") → user (APIs reject tool→user).
_nudge_msg = agent._build_assistant_message(assistant_message, finish_reason)
_nudge_msg["content"] = "(empty)"
_nudge_msg["_empty_recovery_synthetic"] = True
append_message(messages, _nudge_msg)
append_message(messages, {
"role": "user",
"content": _EMPTY_TOOL_RESPONSE_NUDGE,
"_empty_recovery_synthetic": True,
})
continue
# ── Thinking-only prefill continuation ──────────
# Reasoning but no text: append as-is and continue so the model sees
# its own reasoning and writes text. Covers _has_inline_thinking.
_has_structured = bool(
getattr(assistant_message, "reasoning", None)
or getattr(assistant_message, "reasoning_content", None)
or getattr(assistant_message, "reasoning_details", None)
or _has_inline_thinking
)
if _has_structured and agent._thinking_prefill_retries < 2:
agent._thinking_prefill_retries += 1
logger.info(
"Thinking-only response (no visible content) — "
"prefilling to continue (%d/2)",
agent._thinking_prefill_retries,
)
agent._buffer_status(
f"↻ Thinking-only response — prefilling to continue "
f"({agent._thinking_prefill_retries}/2)"
)
interim_msg = agent._build_assistant_message(
assistant_message, "incomplete"
)
interim_msg["_thinking_prefill"] = True
append_message(messages, interim_msg)
agent._session_messages = messages
continue
# ── Empty response retry ──────────────────────
# Retry up to 3 times before fallback; covers truly empty replies
# AND reasoning-only replies after prefill exhaustion.
_truly_empty = not agent._strip_think_blocks(
final_response
).strip()
_prefill_exhausted = (
_has_structured
and agent._thinking_prefill_retries >= 2
)
_empty_candidate = _truly_empty and (
not _has_structured or _prefill_exhausted
)
if _empty_candidate:
# Each empty attempt re-bills the full input; record its
# signature so deterministic empties stop burning paid retries.
# Fails open: missing usage or any output keeps the budget.
_empty_guard.record_empty_attempt(
agent,
finish_reason=finish_reason,
response=response,
)
_empty_retry_budget = (
_empty_guard.empty_retry_budget(agent, response)
if _empty_candidate
else _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET
)
_deterministic_empty = _empty_candidate and (
_empty_guard.deterministic_empty(agent)
)
if (
_empty_candidate
and agent._empty_content_retries < _empty_retry_budget
and not _deterministic_empty
):
agent._empty_content_retries += 1
wait_time = jittered_backoff(
agent._empty_content_retries,
base_delay=5.0,
max_delay=60.0,
)
logger.warning(
"Empty response (no content or reasoning) — "
"retry %d/%d in %.1fs (model=%s)",
agent._empty_content_retries,
_empty_retry_budget, wait_time, agent.model,
)
_budget_note = (
" — high-cost request, reduced retry budget"
if _empty_retry_budget < _empty_guard.DEFAULT_EMPTY_RETRY_BUDGET
else ""
)
agent._buffer_status(
f"⚠️ Empty response from model — retrying "
f"({agent._empty_content_retries}/{_empty_retry_budget}) "
f"in {wait_time:.0f}s{_budget_note}"
)
# Sleep in small increments to stay responsive to interrupts
sleep_end = time.time() + wait_time
_backoff_touch_counter = 0
while time.time() < sleep_end:
if agent._interrupt_requested:
agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during empty-response retry wait, aborting.", force=True)
_interrupt_text = (
f"Operation interrupted: retrying empty response from model "
f"(retry {agent._empty_content_retries}/{_empty_retry_budget})."
)
close_interrupted_tool_sequence(messages, _interrupt_text)
agent._persist_session(messages, conversation_history)
agent.clear_interrupt()
return {
"final_response": _interrupt_text,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"interrupted": True,
}
time.sleep(0.2)
_backoff_touch_counter += 1
if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s
agent._touch_activity(
f"empty response retry backoff ({agent._empty_content_retries}/{_empty_retry_budget}), "
f"{int(sleep_end - time.time())}s remaining"
)
continue
if _truly_empty and _deterministic_empty:
logger.warning(
"Deterministic empty response detected "
"(consecutive zero-output completions, "
"model=%s provider=%s finish_reason=%s) — "
"skipping remaining retries",
agent.model, agent.provider, finish_reason,
)
agent._buffer_status(
"⚠️ Model is deterministically returning empty "
"(zero output tokens) — skipping further retries "
"to avoid repeat charges"
)
# ── Exhausted retries — try fallback provider ──
# Before "(empty)", switch to the next provider in the chain.
if _truly_empty and agent._fallback_chain:
logger.warning(
"Empty response after %d retries — "
"attempting fallback (model=%s, provider=%s)",
agent._empty_content_retries, agent.model,
agent.provider,
)
agent._buffer_status(
"⚠️ Model returning empty responses — "
"switching to fallback provider..."
)
if agent._try_activate_fallback():
active_system_prompt = _sync_failover_system_message(
agent, api_messages, active_system_prompt)
agent._empty_content_retries = 0
agent._buffer_status(
f"↻ Switched to fallback: {agent.model} "
f"({agent.provider})"
)
logger.info(
"Fallback activated after empty responses: "
"now using %s on %s",
agent.model, agent.provider,
)
# OUTER loop: `continue` re-runs preflight against the
# fallback's window; `break` would end the turn without
# calling the fallback. Clear the preflight block. (#84733)
_preflight_compression_blocked = False
continue
# Retries and fallback exhausted — fall through to "(empty)".
# Surface the buffered retry trace and, if known, what the empty
# streak cost (each attempt re-billed the full input).
_streak_cost = _empty_guard.streak_cost_usd(agent)
if _streak_cost is not None:
agent._buffer_status(
f"ℹ️ Estimated cost of these empty attempts: "
f"~${_streak_cost:.2f} (input tokens are billed "
f"per attempt even when no answer is produced)"
)
agent._flush_status_buffer()
_turn_exit_reason = "empty_response_exhausted"
reasoning_text = agent._extract_reasoning(assistant_message)
agent._drop_trailing_empty_response_scaffolding(messages)
assistant_msg = agent._build_assistant_message(assistant_message, finish_reason)
assistant_msg["content"] = "(empty)"
# Gateway failure sentinel, not content: persisting it lets later
# "continue" turns replay assistant("(empty)") and loop on empties.
assistant_msg["_empty_terminal_sentinel"] = True
append_message(messages, assistant_msg)
if reasoning_text:
reasoning_preview = reasoning_text[:500] + "..." if len(reasoning_text) > 500 else reasoning_text
logger.warning(
"Reasoning-only response (no visible content) "
"after exhausting retries and fallback. "
"Reasoning: %s", reasoning_preview,
)
agent._emit_status(
"⚠️ Model produced reasoning but no visible "
"response after all retries. Returning empty."
)
else:
logger.warning(
"Empty response (no content or reasoning) "
"after %d retries. No fallback available. "
"model=%s provider=%s",
agent._empty_content_retries, agent.model,
agent.provider,
)
agent._emit_status(
"❌ Model returned no content after all retries"
+ (" and fallback attempts." if agent._fallback_chain else
". No fallback providers configured.")
)
# Delivery-only: show labeled reasoning instead of bare "(empty)"
# when the model thought but wrote no text. The persisted row keeps
# the sentinel; reasoning is never promoted earlier in the ladder.
if reasoning_text:
final_response = (
"⚠️ The model produced only internal reasoning and "
"no final answer, despite retries"
+ (" and fallback" if agent._fallback_chain else "")
+ ". Its last reasoning, which may contain the "
"answer:\n\n" + reasoning_preview
)
else:
final_response = "(empty)"
break
# Reset retry counter/signature on successful content
agent._empty_content_retries = 0
agent._thinking_prefill_retries = 0
# Surface the one-shot fallback switch notice before dropping the retry
# buffer so a provider/model switch stays visible on success.
agent._emit_pending_fallback_notice()
agent._clear_status_buffer()
from agent.agent_runtime_helpers import (
intent_ack_continuation_mode,
trailing_continue_intent,
)
_ack_mode = intent_ack_continuation_mode(agent)
# Said-continue-but-stopped guard: no tool calls but the short reply
# TAILS with an announced next action. Fires mid-task too; reuses the
# SAME bounded continuation path and counter (max 2 per turn).
_stall_continue_intent = (
bool(getattr(agent, "_stall_guards", True))
and agent.valid_tool_names
and codex_ack_continuations < 2
and trailing_continue_intent(
agent._strip_think_blocks(final_response or "")
)
)
if _stall_continue_intent or (
_ack_mode != "off"
and agent.valid_tool_names
and codex_ack_continuations < 2
and agent._looks_like_codex_intermediate_ack(
user_message=user_message,
assistant_content=final_response,
messages=messages,
require_workspace=(_ack_mode == "codex_only"),
)
):
if _stall_continue_intent:
logger.info(
"Stall guard: turn ending on trailing continue-"
"intent with no tool calls — re-prompting to act "
"(%d/2)", codex_ack_continuations + 1,
)
codex_ack_continuations += 1
interim_msg = agent._build_assistant_message(assistant_message, "incomplete")
append_message(messages, interim_msg)
agent._emit_interim_assistant_message(interim_msg)
continue_msg = {
"role": "user",
"content": _CODEX_ACK_CONTINUATION_NUDGE,
}
append_message(messages, continue_msg)
agent._session_messages = messages
# An acknowledgment is non-final: its text must not suppress
# iteration-limit summarization if the continuation exhausts budget.
final_response = None
continue
codex_ack_continuations = 0
if truncated_response_parts:
final_response = _join_truncated_parts([*truncated_response_parts, final_response])
truncated_response_parts = []
length_continue_retries = 0
# The continuation recovered, so the fragments stay in the transcript.
for _frag in messages:
if isinstance(_frag, dict):
_frag.pop("_length_continuation_fragment", None)
_frag.pop("_length_continuation_nudge", None)
final_response = agent._strip_think_blocks(final_response).strip()
final_msg = agent._build_assistant_message(assistant_message, finish_reason)
# ── Dropped tool-call recovery (copilot/Claude) ────────
# finish_reason="tool_calls" with empty tool_calls would end the turn
# unstarted; re-prompt (max 3 CONSECUTIVE stalls, reset per tool round).
if (
finish_reason == "tool_calls"
and not assistant_message.tool_calls
and getattr(agent, "_dropped_toolcall_retries", 0) < 3
):
agent._dropped_toolcall_retries = getattr(agent, "_dropped_toolcall_retries", 0) + 1
logger.warning(
"finish_reason=tool_calls with empty tool_calls array "
"(narration only) — re-prompting to emit the call "
"(retry %d/3, model=%s provider=%s)",
agent._dropped_toolcall_retries, agent.model, agent.provider,
)
agent._emit_status(
"↻ Model signaled a tool call but sent none — "
f"re-prompting ({agent._dropped_toolcall_retries}/3)"
)
# Both halves of the re-prompt pair are ephemeral scaffolding; flag
# them so the flush never persists them and the finalization pop
# can strip an unanswered tail pair.
final_msg["_dropped_toolcall_nudge"] = True
append_message(messages, final_msg)
append_message(messages, {
"role": "user",
"content": _DROPPED_TOOLCALL_NUDGE_CONTENT,
"_dropped_toolcall_nudge": True,
})
agent._session_messages = messages
final_response = None
continue
# Genuine turn end (no dropped-tool-call mismatch): clear stall budget.
agent._dropped_toolcall_retries = 0
# Pop prefill / empty-retry scaffolding before the final response or
# verification follow-up; it must not become durable transcript.
while (
messages
and isinstance(messages[-1], dict)
and (
messages[-1].get("_thinking_prefill")
or messages[-1].get("_empty_recovery_synthetic")
or messages[-1].get("_empty_terminal_sentinel")
or messages[-1].get("_dropped_toolcall_nudge")
)
):
messages.pop()
try:
from agent.verification_stop import (
build_verify_on_stop_nudge,
verify_on_stop_enabled,
)
if verify_on_stop_enabled():
_verify_nudge = build_verify_on_stop_nudge(
session_id=getattr(agent, "session_id", None),
changed_paths=getattr(agent, "_turn_file_mutation_paths", set()),
attempts=getattr(agent, "_verification_stop_nudges", 0),
)
else:
_verify_nudge = None
except Exception:
logger.debug("verification stop-loop check failed", exc_info=True)
_verify_nudge = None
if _verify_nudge:
agent._verification_stop_nudges = (
getattr(agent, "_verification_stop_nudges", 0) + 1
)
final_msg["finish_reason"] = "verification_required"
# Real content: persist and emit as interim so the user sees the
# attempted answer; only the nudge is flagged synthetic. (#65919)
agent._emit_interim_assistant_message(final_msg)
append_message(messages, final_msg)
try:
agent._flush_messages_to_session_db(messages, conversation_history)
except Exception:
logger.debug("verify-on-stop interim flush failed", exc_info=True)
append_message(messages, {
"role": "user",
"content": _verify_nudge,
"_verification_stop_synthetic": True,
})
agent._session_messages = messages
# Internal nudge: stay silent on the terminal, debug-log only.
logger.debug("verification stop-loop nudge issued (attempt %d)",
agent._verification_stop_nudges)
# Keep the answer only as a budget-exhaustion fallback; clear
# ``final_response`` so the finalizer can tell this gate from error
# exits. Mark previewed only if the candidate is reused. (#61631)
_pending_verification_response = final_response
_pending_verification_response_previewed = (
agent._interim_content_was_streamed(final_response or "")
)
final_response = None
continue
# pre_verify hook gate: after code edits a registered hook may keep the
# agent going one more turn; no default continuation cost.
_verify_nudge2 = None
_edited = sorted(getattr(agent, "_turn_file_mutation_paths", set()) or [])
_attempt = getattr(agent, "_pre_verify_nudges", 0)
try:
from agent.verify_hooks import max_verify_nudges
from hermes_cli.lifecycle import has_hook
from hermes_cli.plugins import get_pre_verify_continue_message
if _edited and has_hook("pre_verify") and _attempt < max_verify_nudges():
# Posture is fixed for the session — resolve once + cache.
coding = getattr(agent, "_resolved_is_coding", None)
if coding is None:
from agent.coding_context import is_coding_context
coding = bool(is_coding_context(platform=getattr(agent, "platform", "") or ""))
agent._resolved_is_coding = coding
_verify_nudge2 = get_pre_verify_continue_message(
session_id=getattr(agent, "session_id", None) or "",
platform=getattr(agent, "platform", "") or "",
model=getattr(agent, "model", "") or "",
coding=coding,
attempt=_attempt,
final_response=final_response,
changed_paths=_edited,
)
except Exception:
logger.debug("pre_verify hook check failed", exc_info=True)
_verify_nudge2 = None
if _verify_nudge2:
agent._pre_verify_nudges = _attempt + 1
final_msg["finish_reason"] = "verify_hook_continue"
# Real content: persist and emit as interim so the user sees the
# attempted answer; only the nudge is flagged synthetic. (#65919)
agent._emit_interim_assistant_message(final_msg)
append_message(messages, final_msg)
try:
agent._flush_messages_to_session_db(messages, conversation_history)
except Exception:
logger.debug("pre_verify interim flush failed", exc_info=True)
append_message(messages, {
"role": "user",
"content": _verify_nudge2,
"_pre_verify_synthetic": True,
})
agent._session_messages = messages
logger.debug("pre_verify nudge issued (attempt %d)",
agent._pre_verify_nudges)
_pending_verification_response = final_response
_pending_verification_response_previewed = (
agent._interim_content_was_streamed(final_response or "")
)
final_response = None
continue
# ── Kanban worker terminal-tool stop guard ─────────────
# Workers must end with kanban_complete / kanban_block; a narrated stop
# is recorded as protocol_violation, so nudge once or twice first.
try:
from agent.kanban_stop import build_kanban_stop_nudge
_kanban_nudge = build_kanban_stop_nudge(
messages=messages,
attempts=getattr(agent, "_kanban_stop_nudges", 0),
)
except Exception:
logger.debug("kanban stop-loop check failed", exc_info=True)
_kanban_nudge = None
if _kanban_nudge:
agent._kanban_stop_nudges = (
getattr(agent, "_kanban_stop_nudges", 0) + 1
)
final_msg["finish_reason"] = "kanban_terminal_required"
final_msg["_kanban_stop_synthetic"] = True
append_message(messages, final_msg)
append_message(messages, {
"role": "user",
"content": _kanban_nudge,
"_kanban_stop_synthetic": True,
})
agent._session_messages = messages
logger.info(
"kanban stop-loop nudge issued (attempt %d) task=%s",
agent._kanban_stop_nudges,
os.environ.get("HERMES_KANBAN_TASK", ""),
)
agent._emit_status(
"⚠️ Kanban worker tried to exit without "
"kanban_complete/kanban_block — nudging to finish"
)
# Same finalizer contract as verify-on-stop: clear final_response so
# budget exhaustion doesn't treat the narrated stop as an answer.
_pending_verification_response = final_response
_pending_verification_response_previewed = (
agent._interim_content_was_streamed(final_response or "")
)
final_response = None
continue
append_message(messages, final_msg)
# Make the answer durable before leaving the loop; _DB_PERSISTED_MARKER
# keeps _persist_session idempotent. Failure must NOT abort the turn:
# _persist_session retries the write. (#81641)
try:
agent._flush_messages_to_session_db(messages, conversation_history)
except Exception:
logger.warning(
"final text-turn flush failed (session=%s) — reply is "
"not yet durable; relying on finalize_turn retry",
getattr(agent, "session_id", None) or "none",
exc_info=True,
)
_turn_exit_reason = f"text_response(finish_reason={finish_reason})"
if not agent.quiet_mode:
agent._safe_print(f"🎉 Conversation completed after {api_call_count} OpenAI-compatible API call(s)")
break
except Exception as e:
# Count every escaped exception before classification so permanent
# failures terminate even with an unlimited turn budget. (#92450)
_outer_error_count += 1
# Phase-aware classification: deterministic local post-processing bugs
# (traceback via local helpers, never API helpers) aren't retried (#66267).
# Interpreter shutdown makes every executor op raise: break. (#93217)
if sys.is_finalizing() or _is_interpreter_shutdown_error(e):
error_msg = (
f"Interpreter is shutting down — cannot continue "
f"(API call #{api_call_count}): {e}"
)
try:
agent._safe_print(f"❌ {error_msg}")
except (OSError, ValueError):
pass
logger.warning(error_msg)
# Best-effort persist — the dying executor may raise the same error;
# don't let it mask the shutdown exit. finalize_turn retries.
try:
agent._persist_session(messages, conversation_history)
except Exception:
pass
_turn_exit_reason = "interpreter_shutdown"
final_response = (
"Session is shutting down. Your conversation can be "
"resumed with: hermes --resume <session-id>"
)
# Don't append: a prefill/interim assistant may already be the tail
# (assistant→assistant). finalize_turn appends only when safe.
break
tb_module_names: set[str] = set()
_tb = e.__traceback__
while _tb is not None:
_fname = os.path.splitext(os.path.basename(_tb.tb_frame.f_code.co_filename))[0]
tb_module_names.add(_fname)
_tb = _tb.tb_next
_hit_local = bool(tb_module_names & _LOCAL_PROCESSING_MODULES)
_hit_api = bool(tb_module_names & _API_CALL_MODULES)
_is_local_processing_error = _hit_local and not _hit_api
if _is_local_processing_error:
error_msg = (
f"Error during local message processing after "
f"OpenAI-compatible API call #{api_call_count}: {str(e)}"
)
else:
error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}"
# Honor the _vprint contract: suppress_status_output silences hard
# failures; quiet_mode -q still shows them. Traceback is logged below.
if getattr(agent, "suppress_status_output", False):
logger.error(error_msg)
else:
try:
print(f"❌ {error_msg}")
except (OSError, ValueError):
logger.error(error_msg)
# ERROR level with traceback so outer-loop failures land in agent.log
# AND errors.log and stay reproducible.
logger.exception("Outer loop error in API call #%d", api_call_count)
# An appended assistant tool_calls message needs a role="tool" result
# per tool_call_id; fill in error results for unanswered ones.
for idx in range(len(messages) - 1, -1, -1):
msg = messages[idx]
if not isinstance(msg, dict):
break
if msg.get("role") == "tool":
continue
if msg.get("role") == "assistant" and msg.get("tool_calls"):
answered_ids = {
m["tool_call_id"]
for m in messages[idx + 1:]
if isinstance(m, dict) and m.get("role") == "tool"
}
for tc in msg["tool_calls"]:
if not tc or not isinstance(tc, dict): continue
if tc["id"] not in answered_ids:
err_msg = {
"role": "tool",
"name": _ra().AIAgent._get_tool_call_name_static(tc),
"tool_call_id": tc["id"],
"content": f"Error executing tool: {error_msg}",
}
append_message(messages, err_msg)
break
# Non-tool errors are already printed; a synthetic message would pollute
# history and risk breaking role alternation.
# Local errors are deterministic: stop early instead of retrying until the
# budget is gone; a small per-turn cap prevents infinite spinning (#92450).
_outer_error_cap = min(_MAX_OUTER_LOOP_ERRORS, max(1, agent.max_iterations))
if (
_is_local_processing_error
or api_call_count >= agent.max_iterations - 1
or _outer_error_count >= _outer_error_cap
):
if _is_local_processing_error:
_turn_exit_reason = f"local_processing_error({error_msg[:80]})"
final_response = f"I apologize, but I encountered an error while processing the model response: {error_msg}"
elif _outer_error_count >= _outer_error_cap:
failed = True
_turn_exit_reason = f"repeated_outer_errors({error_msg[:80]})"
final_response = f"I apologize, but I encountered repeated errors: {error_msg}"
else:
_turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})"
final_response = f"I apologize, but I encountered repeated errors: {error_msg}"
# Don't append the assistant message: a prefill/interim assistant may be
# the tail. finalize_turn appends only when _tail_role != "assistant".
break
# Post-loop finalization lives in agent/turn_finalizer.finalize_turn.
result = finalize_turn(
agent,
final_response=final_response,
api_call_count=api_call_count,
interrupted=interrupted,
failed=failed,
messages=messages,
conversation_history=conversation_history,
effective_task_id=effective_task_id,
turn_id=turn_id,
user_message=user_message,
original_user_message=original_user_message,
_should_review_memory=_should_review_memory,
_turn_exit_reason=_turn_exit_reason,
_pending_verification_response=_pending_verification_response,
_pending_verification_response_previewed=_pending_verification_response_previewed,
)
if _compression_timeout_exhausted:
# Reuse the gateway's context-recovery contract: transcript stays intact while
# future input can move to a clean session (#98722).
result["error"] = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
result["partial"] = True
result["compression_exhausted"] = True
return result
__all__ = ["run_conversation"]