fix(ux): plain-language, actionable user-facing messages (core)
Squashed integration of the user-facing message audit for this surface set. Full per-finding receipts: /tmp/ux-audit/lanes/*-receipt.md (campaign artifacts).
This commit is contained in:
@@ -2107,7 +2107,8 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str:
|
||||
|
||||
except Exception as e:
|
||||
logger.warning("Failed to get summary response: %s", e)
|
||||
final_response = f"I reached the maximum iterations ({agent.max_iterations}) but couldn't summarize. Error: {str(e)}"
|
||||
from agent.turn_failure_copy import site_copy
|
||||
final_response = site_copy("max_iterations_no_summary", limit=agent.max_iterations)
|
||||
finally:
|
||||
from agent import relay_llm
|
||||
relay_llm.complete_logical_call(summary_api_request_id, outcome=summary_call_outcome)
|
||||
|
||||
@@ -1808,15 +1808,14 @@ def check_compression_model_feasibility(agent: Any) -> None:
|
||||
if client is None or not aux_model:
|
||||
if _aux_cfg_provider and _aux_cfg_provider != "auto":
|
||||
msg = (
|
||||
"⚠ Configured auxiliary compression provider "
|
||||
f"'{_aux_cfg_provider}' is unavailable — context "
|
||||
"compression will drop middle turns without a summary. "
|
||||
"Check auxiliary.compression in config.yaml and reauthenticate that provider."
|
||||
f"⚠ Configured auxiliary compression provider '{_aux_cfg_provider}' is unavailable, "
|
||||
"so older messages in long chats will be cut without a summary. Sign in to that "
|
||||
"provider again, or change auxiliary.compression in your config."
|
||||
)
|
||||
else:
|
||||
msg = (
|
||||
"⚠ No auxiliary LLM provider configured — context compression will drop middle turns without a summary. "
|
||||
"Run `hermes setup` or set OPENROUTER_API_KEY."
|
||||
"⚠ No auxiliary LLM provider configured: Hermes has no helper model for summarising "
|
||||
"long chats, so older messages will be cut without a summary. Run `hermes setup` to add one."
|
||||
)
|
||||
agent._compression_warning = msg
|
||||
agent._emit_status(msg)
|
||||
@@ -3033,10 +3032,13 @@ def _warn_summary_or_aux_fallback(agent: Any) -> None:
|
||||
_aux_key = (_aux_fail_model, _aux_fail_err)
|
||||
if _aux_fail_model and getattr(agent, "_last_aux_fallback_warning_key", None) != _aux_key:
|
||||
agent._last_aux_fallback_warning_key = _aux_key
|
||||
logger.warning(
|
||||
"Configured compression model %r failed (%s); recovered using the main model.",
|
||||
_aux_fail_model, _aux_fail_err or "unknown error",
|
||||
)
|
||||
agent._emit_warning(
|
||||
f"ℹ Configured compression model '{_aux_fail_model}' failed "
|
||||
f"({_aux_fail_err or 'unknown error'}). Recovered using main model — "
|
||||
"check auxiliary.compression.model in config.yaml."
|
||||
f"ℹ Configured compression model '{_aux_fail_model}' failed, so Hermes summarised "
|
||||
"with your main model instead. Check auxiliary.compression.model in your config."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -39,6 +39,7 @@ from agent.turn_retry_state import TurnRetryState
|
||||
from agent.turn_api_call import handle_api_interrupt, nous_rate_limit_guard, perform_api_call
|
||||
from agent.turn_api_error import handle_api_error
|
||||
from agent.turn_api_request import build_api_request
|
||||
from agent.turn_failure_copy import site_copy
|
||||
from agent.turn_final_response import finish_text_response
|
||||
from agent.turn_finalizer import finalize_turn
|
||||
from agent.turn_iteration_prep import (
|
||||
@@ -876,11 +877,6 @@ _EMPTY_TOOL_RESPONSE_NUDGE = (
|
||||
)
|
||||
|
||||
|
||||
# Shared trailer for both content-policy refusal paths so guidance cannot drift.
|
||||
_CONTENT_POLICY_RECOVERY_HINT = (
|
||||
"Try rephrasing the request, narrowing the context, or adding a fallback provider with "
|
||||
"`hermes fallback add`."
|
||||
)
|
||||
|
||||
|
||||
# Memo for send-path tool-call argument canonicalization (re-run on every historical call
|
||||
@@ -970,6 +966,7 @@ def _content_policy_blocked_result(
|
||||
return {
|
||||
"final_response": final_response, "messages": messages, "api_calls": api_call_count,
|
||||
"completed": False, "failed": True, "error": f"content_policy_blocked: {error_detail}",
|
||||
"failure_reason": "content_policy_blocked", "failure_retryable": False,
|
||||
}
|
||||
|
||||
|
||||
@@ -1040,9 +1037,10 @@ def _provider_overflow_exhausted_result(
|
||||
# providers.
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return _partial_turn_result(
|
||||
"Context length exceeded: compression could not reduce the rebuilt request below the safe threshold.",
|
||||
site_copy("context_overflow", model=agent.model),
|
||||
messages, api_call_count, failed=True, compression_exhausted=True,
|
||||
turn_exit_reason="context_compression_exhausted",
|
||||
failure_reason="context_overflow", failure_retryable=False,
|
||||
)
|
||||
|
||||
|
||||
@@ -1268,6 +1266,7 @@ def _preflight_timeout_result(agent, exc, conversation_history) -> Dict[str, Any
|
||||
return _partial_turn_result(
|
||||
str(exc), list(conversation_history or []), 0,
|
||||
failed=True, compression_exhausted=True, turn_exit_reason="context_compression_timeout",
|
||||
failure_reason="context_overflow", failure_retryable=False,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -888,6 +888,8 @@ class KawaiiSpinner:
|
||||
# ── Cute tool message (completion line that replaces the spinner) ─────────
|
||||
|
||||
_ERROR_SUFFIX_MAX_LEN = 48
|
||||
# A degraded backend (Docker down, SSH host unreachable) needs the whole reason plus the fix hint.
|
||||
_DEGRADED_SUFFIX_MAX_LEN = 200
|
||||
|
||||
|
||||
def _trim_error(msg: str) -> str:
|
||||
@@ -900,17 +902,32 @@ def _trim_error(msg: str) -> str:
|
||||
return _tail_trunc(msg, _ERROR_SUFFIX_MAX_LEN)
|
||||
|
||||
|
||||
def _degraded_suffix(data: dict) -> str:
|
||||
"""`` [<reason> — <retry_hint>]`` for a ``status: degraded`` terminal result (hint omitted when empty)."""
|
||||
reason = str(data.get("reason") or data.get("error") or "terminal backend unavailable").strip()
|
||||
hint = str(data.get("retry_hint") or "").strip()
|
||||
text = f"{reason} — {hint}" if hint else reason
|
||||
return f" [{_tail_trunc(text, _DEGRADED_SUFFIX_MAX_LEN)}]"
|
||||
|
||||
|
||||
def _detect_tool_failure(tool_name: str, result: str | None) -> tuple[bool, str]:
|
||||
"""Return ``(is_failure, suffix)`` for a tool result, e.g. ``(True, " [exit 1]")``."""
|
||||
if result is None or file_mutation_result_landed(tool_name, result):
|
||||
return False, ""
|
||||
data = safe_json_loads(result)
|
||||
|
||||
# A denied/timed-out approval carries one human sentence; show it instead of the model-facing
|
||||
# "BLOCKED: ... Do NOT retry" text (which stays in the JSON for the model).
|
||||
if isinstance(data, dict) and data.get("user_summary"):
|
||||
return True, f" [{_tail_trunc(str(data['user_summary']), _DEGRADED_SUFFIX_MAX_LEN)}]"
|
||||
|
||||
# Terminal: non-zero exit code is the canonical failure signal.
|
||||
if tool_name == "terminal":
|
||||
exit_code = data.get("exit_code") if isinstance(data, dict) else None
|
||||
if exit_code is None or exit_code == 0:
|
||||
return False, ""
|
||||
if data.get("status") == "degraded":
|
||||
return True, _degraded_suffix(data)
|
||||
err_msg = data.get("error")
|
||||
return True, f" [{_trim_error(str(err_msg))}]" if err_msg else f" [exit {exit_code}]"
|
||||
|
||||
|
||||
+12
-3
@@ -28,9 +28,17 @@ LAYER_GATEWAY = "gateway"
|
||||
LAYER_DISK = "disk"
|
||||
|
||||
# failure_reason → UI layer. Unlisted reasons fall back to LAYER_PROVIDER:
|
||||
# every FailoverReason comes from classifying a provider call.
|
||||
# every FailoverReason comes from classifying a provider call. Loop-site codes
|
||||
# (agent/turn_failure_copy.py::SITE_FAILURE_CODES) are listed explicitly: the
|
||||
# ones that are not provider verdicts map to the gateway layer so the client
|
||||
# does not offer "Switch provider"; the ones the model/provider caused
|
||||
# (cut-off output, empty or broken reply) stay on the provider layer, where the
|
||||
# clients' per-code copy names the real fix (`continue`, smaller steps, /retry).
|
||||
_REASON_TO_LAYER = {
|
||||
"auth": LAYER_AUTH, "auth_permanent": LAYER_AUTH, "billing": LAYER_BILLING, "billing_unverified": LAYER_BILLING,
|
||||
"loop_error": LAYER_GATEWAY, "interpreter_shutdown": LAYER_GATEWAY, "session_busy": LAYER_GATEWAY,
|
||||
"truncated": LAYER_PROVIDER, "empty_response": LAYER_PROVIDER, "invalid_response": LAYER_PROVIDER,
|
||||
"context_overflow": LAYER_PROVIDER, # a bigger-window model IS the fix, so Switch provider applies
|
||||
}
|
||||
|
||||
# Failures between us and the base_url (not a provider verdict); on a
|
||||
@@ -43,6 +51,7 @@ _TRANSPORT_REASONS = {"timeout", "ssl_cert_verification"}
|
||||
_NON_RETRYABLE_REASONS = {
|
||||
"auth", "auth_permanent", "billing", "billing_unverified", "content_policy_blocked",
|
||||
"provider_policy_blocked", "model_not_found", "format_error", "ssl_cert_verification",
|
||||
"context_overflow", "interpreter_shutdown",
|
||||
}
|
||||
|
||||
# Providers whose base_url is user-supplied rather than a known vendor.
|
||||
@@ -82,7 +91,7 @@ def _surface(layer: str, code: str, retryable: bool, provider: str = "", model:
|
||||
# OAuth providers are fixed by signing in again; API-key providers by
|
||||
# replacing the key. The client's one-click recovery needs to know which
|
||||
# and how to name the account it re-opens.
|
||||
surface["auth_kind"] = _auth_kind(provider)
|
||||
surface["auth_kind"] = auth_kind(provider)
|
||||
surface["provider_label"] = _provider_label(provider)
|
||||
return surface
|
||||
|
||||
@@ -96,7 +105,7 @@ def _provider_label(provider: str) -> str:
|
||||
return provider
|
||||
|
||||
|
||||
def _auth_kind(provider: Optional[str]) -> str:
|
||||
def auth_kind(provider: Optional[str]) -> str:
|
||||
"""``"oauth"`` for providers whose credential is an OAuth/subscription grant
|
||||
(desktop Accounts tab), ``"api_key"`` for everything else."""
|
||||
try:
|
||||
|
||||
+25
-55
@@ -23,14 +23,6 @@ from typing import Dict, Mapping, Optional
|
||||
# at gateway startup when gateway.multiplex_profiles is true.
|
||||
_MULTIPLEX_ACTIVE: bool = False
|
||||
|
||||
# Context-local counterpart: a task serving a profile OTHER than the process's own, inside a
|
||||
# process that is not a multiplexer as a whole — the desktop backend's cron ticker firing a
|
||||
# sibling profile's job. Every isolation keyed on ``is_multiplex_active()`` (the routed-dotenv
|
||||
# guard, ``get_secret``'s fail-closed miss, subprocess scrubbing, passthrough) applies inside
|
||||
# it while the process's own turns keep single-profile semantics. A contextvar, so it reaches
|
||||
# the pool worker together with the home override via ``copy_context()``.
|
||||
_MULTIPLEX_CONTEXT: ContextVar[bool] = ContextVar("_MULTIPLEX_CONTEXT", default=False)
|
||||
|
||||
|
||||
def set_multiplex_active(active: bool) -> None:
|
||||
"""Mark whether the process is a profile multiplexer (get_secret fails closed)."""
|
||||
@@ -38,19 +30,8 @@ def set_multiplex_active(active: bool) -> None:
|
||||
_MULTIPLEX_ACTIVE = bool(active)
|
||||
|
||||
|
||||
def set_multiplex_context(active: bool) -> Token:
|
||||
"""Run the current task under multiplex semantics regardless of the process flag.
|
||||
Returns a reset token; pair with :func:`reset_multiplex_context` in a ``finally``."""
|
||||
return _MULTIPLEX_CONTEXT.set(bool(active))
|
||||
|
||||
|
||||
def reset_multiplex_context(token: Token) -> None:
|
||||
_MULTIPLEX_CONTEXT.reset(token)
|
||||
|
||||
|
||||
def is_multiplex_active() -> bool:
|
||||
"""True in a multiplexing process, or for a task running under multiplex semantics."""
|
||||
return _MULTIPLEX_ACTIVE or _MULTIPLEX_CONTEXT.get()
|
||||
return _MULTIPLEX_ACTIVE
|
||||
|
||||
|
||||
_SECRET_SCOPE: ContextVar[Optional[Mapping[str, str]]] = ContextVar("_SECRET_SCOPE", default=None)
|
||||
@@ -61,8 +42,28 @@ class UnscopedSecretError(RuntimeError):
|
||||
|
||||
The fix is to wrap the call path in ``set_secret_scope(...)`` (the per-turn
|
||||
/ per-adapter profile scope), not to widen the global allowlist.
|
||||
|
||||
``str(exc)`` is the ONE sentence an end user can act on; the developer diagnosis
|
||||
(which secret, which doc) rides ``__notes__`` so tracebacks and logs keep it.
|
||||
"""
|
||||
|
||||
def __init__(self, secret_name: str = "", developer_detail: str = ""):
|
||||
# Older callers passed the whole developer sentence positionally
|
||||
# (``UnscopedSecretError("get_secret('X') called with no scope ...")``); a secret
|
||||
# name never contains whitespace, so treat such a string as the detail.
|
||||
if secret_name and not developer_detail and any(ch.isspace() for ch in secret_name):
|
||||
secret_name, developer_detail = "", secret_name
|
||||
what = f"this profile's {secret_name}" if secret_name else "this profile's API key"
|
||||
super().__init__(
|
||||
f"Hermes could not read {what} (an internal profile-scoping bug on the multiplexed "
|
||||
"gateway, not your configuration). Run `hermes gateway restart`; if it keeps happening, "
|
||||
"report it with `hermes debug share`."
|
||||
)
|
||||
self.secret_name = secret_name
|
||||
self.developer_detail = developer_detail
|
||||
if developer_detail:
|
||||
self.add_note(developer_detail)
|
||||
|
||||
|
||||
def set_secret_scope(secrets: Optional[Mapping[str, str]]) -> Token:
|
||||
"""Install the active profile's secret mapping; ``None`` clears. Returns a reset token."""
|
||||
@@ -144,15 +145,16 @@ def get_secret(name: str, default: Optional[str] = None) -> Optional[str]:
|
||||
val = scope.get(name)
|
||||
if val is not None:
|
||||
return val
|
||||
return default if is_multiplex_active() else _environ_or(name, default)
|
||||
if is_multiplex_active():
|
||||
return default if _MULTIPLEX_ACTIVE else _environ_or(name, default)
|
||||
if _MULTIPLEX_ACTIVE:
|
||||
raise UnscopedSecretError(
|
||||
name,
|
||||
f"get_secret({name!r}) called with no profile secret scope active "
|
||||
f"while multiplexing is on. This credential read must run inside a "
|
||||
f"set_secret_scope(...) block (the per-turn / per-adapter profile "
|
||||
f"scope). Reading os.environ here would risk leaking another "
|
||||
f"profile's value. See website/docs/developer-guide/multiplexing-gateway.md "
|
||||
f"(Workstream A)."
|
||||
f"(Workstream A).",
|
||||
)
|
||||
return _environ_or(name, default)
|
||||
|
||||
@@ -248,36 +250,4 @@ def build_profile_secret_scope(hermes_home: Path) -> Dict[str, str]:
|
||||
except Exception:
|
||||
external_secrets = {}
|
||||
secrets.update((k, v) for k, v in external_secrets.items() if not _is_global_env(k))
|
||||
# Administrator-managed ``.env`` LAST, with override: the launch process applies it that way
|
||||
# (``env_loader._apply_managed_env``) so policy beats a user's own value. Under multiplex
|
||||
# semantics ``get_secret`` never reads ``os.environ`` on a scope miss, so a scope built from
|
||||
# the profile files alone would drop a managed-only credential and let the user's value win a
|
||||
# managed-vs-user collision (#111187 review). Every multiplex-authoritative scope — gateway
|
||||
# turn, routed cron fire, external worker — is built here, so managed authority is composed
|
||||
# once, not restored by each consumer.
|
||||
from hermes_cli.managed_scope import load_managed_env # fail-open: {} when no managed scope
|
||||
|
||||
secrets.update((k, v) for k, v in load_managed_env().items() if not _is_global_env(k))
|
||||
return secrets
|
||||
|
||||
|
||||
def refresh_installed_secret_scope(hermes_home: Path) -> bool:
|
||||
"""Fold a fresh build of *hermes_home*'s secrets into the INSTALLED scope, in place.
|
||||
|
||||
A scope is frozen when installed, but a fire can learn of new values afterwards: a routed cron
|
||||
fire's first agent build discovers plugin secret sources, and under multiplex semantics the
|
||||
reload that follows is hydrate-only (never ``os.environ``), so nothing else would carry those
|
||||
values into the scope this fire already holds. The caller names the home the installed scope
|
||||
was built for. True when a scope was updated; False when none is installed."""
|
||||
scope = _SECRET_SCOPE.get()
|
||||
if not isinstance(scope, dict):
|
||||
return False
|
||||
# REPLACE, don't merge: the rebuild is the profile's current truth, so a name a source has
|
||||
# stopped supplying (rotated, revoked, source removed) must disappear from the fire's scope
|
||||
# rather than survive as the stale value dict.update() would keep.
|
||||
rebuilt = build_profile_secret_scope(hermes_home)
|
||||
# Update first, then drop what is gone: a concurrent reader never sees an emptied scope.
|
||||
scope.update(rebuilt)
|
||||
for name in [n for n in scope if n not in rebuilt]:
|
||||
scope.pop(name, None)
|
||||
return True
|
||||
|
||||
@@ -36,17 +36,16 @@ def is_thinking_timeout(classified: object, model: str, error_msg: str) -> bool:
|
||||
|
||||
|
||||
def build_thinking_timeout_guidance(provider: str, model: str, model_label: Optional[str] = None) -> str:
|
||||
"""User-facing guidance appended to the final response. ``model`` is used verbatim in
|
||||
the config snippet so it is copy-pasteable; ``model_label`` is the optional prose name."""
|
||||
"""User-facing guidance appended to the final response: easiest fix first (``/reasoning
|
||||
low``), the config knob last. ``model`` is used verbatim in the config path so it is
|
||||
copy-pasteable; ``model_label`` is the optional prose name."""
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
label = model_label or model
|
||||
return (
|
||||
"\n\nThe model's thinking phase exceeded the upstream proxy's idle timeout before the first content token "
|
||||
"arrived. This is a "
|
||||
f"known issue with reasoning models (like {label}) behind cloud "
|
||||
"gateways (NVIDIA NIM, OpenAI, Anthropic, DeepSeek). Workarounds in priority order:\n"
|
||||
f"1. Set `providers.{provider}.models.{model}.stale_timeout_seconds: 900` "
|
||||
"in `~/.hermes/config.yaml` to extend the per-call timeout. (Hermes's built-in floor is 600s for known "
|
||||
"reasoning models — if you still see this after raising, the upstream cap is even shorter.)\n2. Lower "
|
||||
"`reasoning_budget` or set `reasoning_effort: medium` on this model if the provider supports it.\n3. Use a "
|
||||
"smaller / faster reasoning model if the task doesn't require deep thinking."
|
||||
f"{label} was thinking for so long that the connection timed out before it wrote anything "
|
||||
"(common for reasoning models behind cloud gateways such as NVIDIA NIM, OpenAI, Anthropic, "
|
||||
"DeepSeek). Easiest fixes: `/reasoning low`, or switch to a faster model with /model. "
|
||||
f"Advanced: set `providers.{provider}.models.{model}.stale_timeout_seconds: 900` in "
|
||||
f"`{display_hermes_home()}/config.yaml` to allow a longer wait."
|
||||
)
|
||||
|
||||
@@ -14,7 +14,9 @@ import logging
|
||||
import time
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from agent.error_classifier import FailoverReason
|
||||
from agent.message_metadata import append_message
|
||||
from agent.turn_failure_copy import site_copy, stamp_failure
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
@@ -237,7 +239,7 @@ def nous_rate_limit_guard(
|
||||
if anon_auth.route_is_welcome_host(getattr(agent, "base_url", "")):
|
||||
_nous_msg = anon_auth.FREE_TIER_RATE_LIMIT_CHAT.format(reset=reset)
|
||||
else:
|
||||
_nous_msg = f"Nous Portal rate limit active — resets in {reset}."
|
||||
_nous_msg = f"Your Nous account has hit its rate limit; it resets in {reset}."
|
||||
agent._buffer_vprint(f"⏳ {_nous_msg} Trying fallback...")
|
||||
agent._buffer_status(f"⏳ {_nous_msg}")
|
||||
if agent._try_activate_fallback():
|
||||
@@ -249,18 +251,14 @@ def nous_rate_limit_guard(
|
||||
# No fallback — surface the buffered rate-limit context that led here.
|
||||
agent._flush_status_buffer()
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return _verdict("return", {
|
||||
"final_response": (
|
||||
f"⏳ {_nous_msg}\n\n"
|
||||
"No fallback provider available. Try again after the reset, or add a "
|
||||
"fallback provider in config.yaml."
|
||||
),
|
||||
return _verdict("return", stamp_failure({
|
||||
"final_response": f"⏳ {_nous_msg}\n\n{site_copy('nous_rate_limit')}",
|
||||
"messages": messages,
|
||||
"api_calls": api_call_count,
|
||||
"completed": False,
|
||||
"failed": True,
|
||||
"error": _nous_msg,
|
||||
})
|
||||
}, FailoverReason.rate_limit.value, True))
|
||||
except Exception:
|
||||
pass # Never let rate guard break the agent loop
|
||||
return _verdict("fallthrough")
|
||||
|
||||
@@ -16,6 +16,7 @@ from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent import empty_response_guard as _empty_guard
|
||||
from agent.message_metadata import append_message
|
||||
from agent.turn_failure_copy import site_copy
|
||||
from agent.turn_recovery import interruptible_backoff_sleep
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
@@ -131,11 +132,7 @@ def _terminal_empty(agent: Any, assistant_message: Any, finish_reason: str, mess
|
||||
agent._emit_status(
|
||||
"⚠️ Model produced reasoning but no visible response after all retries. Returning empty."
|
||||
)
|
||||
return (
|
||||
"⚠️ The model produced only internal reasoning and no final answer, despite retries"
|
||||
+ (" and fallback" if agent._fallback_chain else "")
|
||||
+ ". Its last reasoning, which may contain the answer:\n\n" + reasoning_preview
|
||||
)
|
||||
return site_copy("reasoning_only", model=agent.model, preview=reasoning_preview)
|
||||
|
||||
|
||||
def recover_empty_response(
|
||||
|
||||
+34
-39
@@ -17,17 +17,19 @@ from agent.tool_result_classification import (
|
||||
|
||||
_NO_REPLY = "⚠️ No reply: "
|
||||
|
||||
# One text for "the model produced nothing after retries" on every surface (CLI explainer,
|
||||
# gateway ``(empty)`` rewrite, desktop); the model name is filled in by the explainer.
|
||||
EMPTY_RESPONSE_EXPLANATION = (
|
||||
"{model} didn't produce a reply this time, even after retries. "
|
||||
"Send `continue` to try again, or switch models with /model."
|
||||
)
|
||||
|
||||
# Exact ``turn_exit_reason`` → explanation body (prefixed with ``_NO_REPLY``).
|
||||
_EXIT_REASON_EXPLANATIONS: Dict[str, str] = {
|
||||
"empty_response_exhausted": (
|
||||
"the model returned empty content after retries and any "
|
||||
"fallback providers. Try `continue`, switch model/provider, "
|
||||
"or inspect the tool output above."
|
||||
),
|
||||
"empty_response_exhausted": EMPTY_RESPONSE_EXPLANATION,
|
||||
"all_retries_exhausted_no_response": (
|
||||
"all API retries were exhausted before a response was "
|
||||
"produced (provider errors / rate limits). Try `continue` "
|
||||
"or switch provider."
|
||||
"the model provider didn't answer after all retries. "
|
||||
"Send /retry, or switch models with /model."
|
||||
),
|
||||
"partial_stream_recovery": (
|
||||
"streaming stopped early and only a partial response was "
|
||||
@@ -110,29 +112,20 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = {
|
||||
"database). Your message should already be saved — "
|
||||
"please send it again in a moment."
|
||||
),
|
||||
# The forensic runbook for both (WAL generations, manifest.json, sidecars) lives in the
|
||||
# logger.error at hermes_state.py::_raise_if_db_replaced — never in the chat reply.
|
||||
"replaced": (
|
||||
"the turn was stopped because the state database file "
|
||||
"was replaced underneath this process. Do not run "
|
||||
"`hermes doctor --fix` or in-place FTS repair — stop "
|
||||
"the process, restore the intended state.db, then "
|
||||
"restart. Unwritten messages were diverted to "
|
||||
"sessions/<session_id>.jsonl and, on the gateway, "
|
||||
"pending_messages/pending-*.json."
|
||||
"the session database file was replaced while Hermes was running, so this "
|
||||
"message was not saved (a copy is kept in {home}/sessions/). Stop Hermes "
|
||||
"(`hermes gateway stop`), run `hermes doctor` — not `hermes doctor --fix`, which "
|
||||
"would repair the wrong file in place — then start it again and send your message "
|
||||
"once more. Advanced recovery steps are in the log."
|
||||
),
|
||||
"deleted_wal": (
|
||||
"the turn was stopped because a live Hermes process held a retired "
|
||||
"state.db-wal generation after its pathname was deleted or "
|
||||
"replaced. Stop the gateway, dashboard, and cron writers; "
|
||||
"do not overwrite the current state.db or delete its sidecars. "
|
||||
"Check the logs for whether Hermes captured the retired generation, "
|
||||
"then read the adjacent state.db.retired-wal-*/manifest.json. If "
|
||||
"manifest.main.mode is `copied`, inspect that artifact with `hermes "
|
||||
"sessions recover --source <state.db.retired-wal-*/state.db> "
|
||||
"--inspect-only` before deciding whether its committed frames belong "
|
||||
"on the current database. A `header_only` artifact is forensic and "
|
||||
"does not contain a copied state.db to inspect. Unwritten messages "
|
||||
"were diverted to sessions/<session_id>.jsonl and, on the gateway, "
|
||||
"pending_messages/pending-*.json."
|
||||
"the session database was changed or replaced while Hermes was running, so this "
|
||||
"message was not saved (a copy is kept in {home}/sessions/). Stop Hermes "
|
||||
"(`hermes gateway stop`), run `hermes doctor`, then start it again and send your "
|
||||
"message once more. Advanced recovery steps are in the log."
|
||||
),
|
||||
"corrupt": (
|
||||
"the turn was stopped because the state database "
|
||||
@@ -162,18 +155,16 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = {
|
||||
"send your message again."
|
||||
),
|
||||
"disk": (
|
||||
"the turn was stopped because session storage could not "
|
||||
"be written (the transcript would have been lost on "
|
||||
"restart). This is often a full disk — free some space "
|
||||
"(or fix state.db permissions), then send your message "
|
||||
"again."
|
||||
"Hermes couldn't save this conversation to disk, so it stopped rather than lose "
|
||||
"your messages. The disk is probably full: free some space (or fix the permissions "
|
||||
"on {home}/state.db), then send your message again."
|
||||
),
|
||||
}
|
||||
_PERSISTENCE_DEFAULT_EXPLANATION = (
|
||||
"the turn was stopped because session storage could not be "
|
||||
"written (the transcript would have been lost on restart). "
|
||||
"Check the state database health (`hermes doctor`), then "
|
||||
"send your message again."
|
||||
"Hermes couldn't save this conversation, so it stopped rather than lose your messages. "
|
||||
"Possible causes: the drive is out of room, or another Hermes process is holding the "
|
||||
"database. Close other Hermes windows, run `hermes doctor` to check storage, then send "
|
||||
"your message again."
|
||||
)
|
||||
|
||||
|
||||
@@ -306,7 +297,7 @@ class TurnExplainersMixin:
|
||||
|
||||
@staticmethod
|
||||
def _format_turn_completion_explanation(
|
||||
turn_exit_reason: str, persistence_cause: Optional[str] = None, db_path=None
|
||||
turn_exit_reason: str, persistence_cause: Optional[str] = None, db_path=None, model: str = "",
|
||||
) -> str:
|
||||
"""User-facing explanation for an abnormal turn ending, or "" for normal / unknown reasons.
|
||||
|
||||
@@ -324,10 +315,14 @@ class TurnExplainersMixin:
|
||||
if reason.startswith(prefix):
|
||||
body = text
|
||||
break
|
||||
if body is not None and "{model}" in body:
|
||||
body = body.format(model=model or "The model")
|
||||
if body is None and reason == "session_persistence_failed":
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
body = _PERSISTENCE_CAUSE_EXPLANATIONS.get(
|
||||
persistence_cause or "unknown", _PERSISTENCE_DEFAULT_EXPLANATION
|
||||
)
|
||||
).replace("{home}", display_hermes_home())
|
||||
if persistence_cause in ("corrupt", "fts_index"):
|
||||
# Copy-pasteable, so name the store that actually failed and pin the profile:
|
||||
# a multi-profile backend (Desktop serve) hosts sessions whose state.db is NOT
|
||||
|
||||
@@ -334,9 +334,12 @@ def _lease_not_acquired_result(agent, session_id: str, conversation_history) ->
|
||||
agent._emit_warning(timeout_msg)
|
||||
except Exception:
|
||||
logger.debug("Failed to emit session turn lease timeout warning", exc_info=True)
|
||||
# Stamped so Desktop/TUI show "session busy, send again" instead of code="unknown".
|
||||
return {
|
||||
"final_response": timeout_msg,
|
||||
**base,
|
||||
"failed": True,
|
||||
"error": f"session_turn_lease_timeout:{session_id}",
|
||||
"failure_reason": "session_busy",
|
||||
"failure_retryable": True,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
"""User-facing copy and ``failure_reason`` stamping for terminal failed-turn results.
|
||||
|
||||
Every terminal result dict the turn loop returns must carry ``failure_reason`` (a
|
||||
``FailoverReason`` value or one of :data:`SITE_FAILURE_CODES`) and ``failure_retryable`` so
|
||||
``agent/error_surface.py`` yields a specific descriptor instead of ``unknown``. The copy
|
||||
tables here say WHAT happened and WHAT TO DO in plain words; raw provider detail rides a
|
||||
trailing "Provider said:" / "Details:" line.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, NamedTuple, Optional, Tuple
|
||||
|
||||
from agent.error_classifier import FailoverReason
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
# Failure codes minted by loop sites that are not provider verdicts (see module docstring).
|
||||
SITE_FAILURE_CODES = frozenset({
|
||||
"context_overflow", "truncated", "invalid_response", "empty_response", "loop_error",
|
||||
"interpreter_shutdown", "session_busy",
|
||||
})
|
||||
|
||||
|
||||
def stamp_failure(result: Dict[str, Any], reason: str, retryable: bool) -> Dict[str, Any]:
|
||||
"""Stamp the UI verdict fields on a terminal result (in place; returns it)."""
|
||||
result["failure_reason"] = reason
|
||||
result["failure_retryable"] = bool(retryable)
|
||||
return result
|
||||
|
||||
|
||||
def provider_label_for(provider: Any) -> str:
|
||||
"""Human-friendly provider name for chat copy (``"OpenRouter"``, ``"Nous Portal"``…)."""
|
||||
from hermes_cli.models import provider_label
|
||||
|
||||
return provider_label(str(provider or ""))
|
||||
|
||||
|
||||
# ---- turn_exit_reason → failure verdict (finalize_turn stamps these) --------------------------
|
||||
|
||||
class ExitFailure(NamedTuple):
|
||||
"""Verdict for a loop exit. ``fails_turn`` False = advisory: the descriptor fields are
|
||||
stamped so Desktop/TUI show a specific code, but ``failed``/``completed`` keep the values
|
||||
the loop chose — cron silence, the kanban dispatcher breaker and gateway transcript
|
||||
persistence all key on ``failed`` and must not change because a code was added."""
|
||||
|
||||
reason: str
|
||||
retryable: bool
|
||||
fails_turn: bool = True
|
||||
|
||||
|
||||
# (exit-reason prefix, failure_reason, retryable, fails_turn). Prefix match: several reasons
|
||||
# carry a parenthesised detail (``local_processing_error(...)``).
|
||||
_EXIT_REASON_FAILURES: Tuple[Tuple[str, str, bool, bool], ...] = (
|
||||
# Advisory: the reasoning-only text may literally be the answer, and cron stays silent.
|
||||
("empty_response_exhausted", "empty_response", True, False),
|
||||
("all_retries_exhausted_no_response", FailoverReason.server_error.value, True, True),
|
||||
("interpreter_shutdown", "interpreter_shutdown", False, True),
|
||||
# Advisory: a deterministic local bug is not a task failure for the kanban breaker.
|
||||
("local_processing_error", "loop_error", False, False),
|
||||
("repeated_outer_errors", "loop_error", True, True),
|
||||
("error_near_max_iterations", "loop_error", True, True),
|
||||
("context_compression_timeout", "context_overflow", False, True),
|
||||
("context_compression_exhausted", "context_overflow", False, True),
|
||||
("ollama_runtime_context_too_small", "context_overflow", False, True),
|
||||
# Advisory: the loop ends these as an incomplete (not failed) turn with an explainer.
|
||||
("redirect_restart_limit_exceeded", "loop_error", True, False),
|
||||
("rebuilt_restart_limit_exceeded", "loop_error", True, False),
|
||||
)
|
||||
|
||||
|
||||
# Provider error code carried inside an HTTP-200 body → classifier reason.
|
||||
_INVALID_RESPONSE_CODES: Dict[int, str] = {
|
||||
429: FailoverReason.rate_limit.value,
|
||||
500: FailoverReason.server_error.value, 502: FailoverReason.server_error.value,
|
||||
503: FailoverReason.overloaded.value, 529: FailoverReason.overloaded.value,
|
||||
504: FailoverReason.timeout.value, 524: FailoverReason.timeout.value,
|
||||
}
|
||||
|
||||
|
||||
def invalid_response_failure_reason(response: Any) -> str:
|
||||
"""``failure_reason`` for an empty/malformed HTTP-200 body: the embedded provider error
|
||||
code when there is one (so the desktop shows Retry + Switch provider consistently), else
|
||||
the ``invalid_response`` site code."""
|
||||
err = getattr(response, "error", None) if response is not None else None
|
||||
code = getattr(err, "code", None) if err is not None else None
|
||||
if code is None and isinstance(err, dict):
|
||||
code = err.get("code")
|
||||
try:
|
||||
return _INVALID_RESPONSE_CODES.get(int(code), "invalid_response") if code is not None else "invalid_response"
|
||||
except (TypeError, ValueError):
|
||||
return "invalid_response"
|
||||
|
||||
|
||||
def exit_reason_failure(turn_exit_reason: Any) -> Optional[ExitFailure]:
|
||||
""":class:`ExitFailure` for a loop exit that carries a failure verdict, else None."""
|
||||
reason = str(turn_exit_reason or "")
|
||||
for prefix, code, retryable, fails_turn in _EXIT_REASON_FAILURES:
|
||||
if reason.startswith(prefix):
|
||||
return ExitFailure(code, retryable, fails_turn)
|
||||
return None
|
||||
|
||||
|
||||
# ---- chat copy tables -----------------------------------------------------------------------
|
||||
|
||||
_NEXT_STEPS_RETRY = "Wait a minute and send /retry, or switch models with /model."
|
||||
_NEXT_STEPS_LOOP = (
|
||||
"Your message is saved. Send `continue` to try again, or start a new session with /new. "
|
||||
"If it happens again, run `hermes doctor` and share the error details."
|
||||
)
|
||||
|
||||
# Lead sentence per classifier reason once retries and fallback are exhausted.
|
||||
_EXHAUSTED_LEADS: Dict[str, str] = {
|
||||
FailoverReason.rate_limit.value: "{label} rate-limited every one of {attempts} attempts",
|
||||
FailoverReason.upstream_rate_limit.value: "{label} rate-limited every one of {attempts} attempts",
|
||||
FailoverReason.overloaded.value: "{label} reported it was overloaded on all {attempts} attempts",
|
||||
FailoverReason.server_error.value: "{label} returned a server error on all {attempts} attempts",
|
||||
FailoverReason.timeout.value: "{label} didn't respond in time on any of {attempts} attempts",
|
||||
}
|
||||
_EXHAUSTED_DEFAULT_LEAD = "{label} didn't answer after {attempts} attempts"
|
||||
|
||||
# Terminal copy for a non-retryable provider rejection, keyed by classifier reason.
|
||||
_NONRETRYABLE_COPY: Dict[str, str] = {
|
||||
FailoverReason.model_not_found.value: (
|
||||
"Model '{model}' isn't available on {label}. Pick a different model with /model "
|
||||
"(or `hermes model` in a terminal).{prefix_hint}"
|
||||
),
|
||||
FailoverReason.format_error.value: (
|
||||
"{label} rejected this request as malformed, so the model didn't answer. Start a clean "
|
||||
"session with /new or switch models with /model; if it keeps happening, run `hermes doctor`."
|
||||
),
|
||||
FailoverReason.ssl_cert_verification.value: (
|
||||
"Hermes couldn't verify {label}'s security certificate, so the connection was refused. "
|
||||
"This is usually a corporate proxy or an outdated certificate store on this computer — "
|
||||
"see the terminal or `{home}/logs/agent.log` for the exact fix, or try another provider "
|
||||
"with /model."
|
||||
),
|
||||
FailoverReason.provider_policy_blocked.value: (
|
||||
"{label}'s account settings don't allow this model for your request, so it didn't "
|
||||
"answer. Check the provider's data/privacy settings, or switch models with /model."
|
||||
),
|
||||
}
|
||||
_NONRETRYABLE_DEFAULT_COPY = (
|
||||
"{label} rejected the request and retrying won't help. Pick another model with /model, "
|
||||
"or check the details in `{home}/logs/agent.log`."
|
||||
)
|
||||
_AUTH_COPY: Dict[str, str] = {
|
||||
"oauth": (
|
||||
"{label} rejected your sign-in, so the model can't be reached. Sign in again: "
|
||||
"`hermes portal` for Nous, `hermes auth add <provider> --type oauth` for other accounts."
|
||||
),
|
||||
"api_key": (
|
||||
"{label} rejected your API key, so the model can't be reached. Update it in "
|
||||
"Settings → Providers, or run `hermes setup` in a terminal."
|
||||
),
|
||||
}
|
||||
|
||||
CONTENT_POLICY_NEXT_STEPS = (
|
||||
"Try rewording your message or removing sensitive attachments, or switch to another "
|
||||
"model with /model."
|
||||
)
|
||||
|
||||
# ---- one reason → "what happened" gloss, shared by cron, subagent and chat notices ------------
|
||||
|
||||
# FailoverReason / site code → one clause (no HTTP codes, no "provider" jargon). ``{subject}``
|
||||
# is who was asking ("the job", "it"), ``{possessive}`` its possessive ("the job's", "its").
|
||||
# Reasons absent here are NOT provider-shaped; callers fall back to the raw error text.
|
||||
FAILURE_CAUSE_GLOSS: Dict[str, str] = {
|
||||
FailoverReason.timeout.value: "the AI model service did not respond in time",
|
||||
FailoverReason.rate_limit.value: "the AI model service was rate-limited (too many requests)",
|
||||
FailoverReason.upstream_rate_limit.value: "the AI model service was rate-limited (too many requests)",
|
||||
FailoverReason.overloaded.value: "the AI model service is overloaded right now",
|
||||
FailoverReason.server_error.value: "the AI model service returned an internal error",
|
||||
FailoverReason.billing.value: "the AI model service says the account's usage or credit limit is reached",
|
||||
# Wire-level billing code (not a FailoverReason) that error_surface routes to the billing layer.
|
||||
"billing_unverified": "the AI model service says the account's usage or credit limit is reached",
|
||||
FailoverReason.auth.value: "the AI model service rejected the sign-in",
|
||||
FailoverReason.auth_permanent.value: "the AI model service rejected the sign-in",
|
||||
FailoverReason.model_not_found.value: "the model {subject} uses was not found at the AI model service",
|
||||
FailoverReason.content_policy_blocked.value: "the AI model service's safety filter rejected the request",
|
||||
"context_overflow": "{possessive} request grew too large for the model",
|
||||
"payload_too_large": "{possessive} request grew too large for the model",
|
||||
}
|
||||
|
||||
|
||||
def failure_cause_gloss(reason: Any, *, subject: str = "it", possessive: str = "its") -> Optional[str]:
|
||||
"""Plain clause for a classified ``failure_reason``; None when the reason has no gloss."""
|
||||
template = FAILURE_CAUSE_GLOSS.get(str(reason or ""))
|
||||
return template.format(subject=subject, possessive=possessive) if template else None
|
||||
|
||||
|
||||
# ---- site-code copy -------------------------------------------------------------------------
|
||||
|
||||
# Chat copy for the codes in SITE_FAILURE_CODES that a loop site renders itself
|
||||
# (``empty_response`` is worded by agent/turn_explainers.py, ``session_busy`` by the lease).
|
||||
_FAILURE_CODE_COPY: Dict[str, str] = {
|
||||
"context_overflow": (
|
||||
"This conversation has grown too long for {model} to read, and Hermes couldn't shrink "
|
||||
"it enough automatically. Start a new session with /new (your history is kept), or try "
|
||||
"/compress once more. Switching to a model with a bigger context window also works."
|
||||
),
|
||||
"truncated": (
|
||||
"The model's reply was cut off before it finished (it hit its output length limit), so "
|
||||
"Hermes didn't run the incomplete action. Nothing was changed. Send `continue`, ask for "
|
||||
"the work in smaller steps, or raise max_tokens for this model."
|
||||
),
|
||||
"invalid_response": (
|
||||
"{label} sent back an empty or broken reply {attempts} times — it is probably overloaded "
|
||||
"or rate-limiting you. " + _NEXT_STEPS_RETRY + "\n\nDetails: {detail}"
|
||||
),
|
||||
"loop_error": (
|
||||
"Hermes hit repeated errors and stopped this turn so it wouldn't keep retrying. "
|
||||
+ _NEXT_STEPS_LOOP + "\n\nDetails: {detail}"
|
||||
),
|
||||
"interpreter_shutdown": (
|
||||
"Hermes was shutting down and stopped this turn. Your conversation is saved — reopen "
|
||||
"it{resume} and send your message again."
|
||||
),
|
||||
}
|
||||
|
||||
# One-off outcome strings: deterministic loop exits that are NOT failure codes (the result
|
||||
# they ride carries a code from the table above, or none at all).
|
||||
_ONE_OFF_COPY: Dict[str, str] = {
|
||||
"payload_too_large": (
|
||||
"This conversation (including attachments) has grown too large to send to {model}, and "
|
||||
"Hermes couldn't shrink it enough automatically. Start a new session with /new (your "
|
||||
"history is kept), or try /compress once more."
|
||||
),
|
||||
"compression_disabled": (
|
||||
"This conversation is too long for {model} and automatic shrinking is turned off in "
|
||||
"your settings (compression.enabled). Run /compress to shrink it now, /new to start "
|
||||
"fresh, or pick a model with a bigger context window."
|
||||
),
|
||||
"stream_dropped_tool_call": (
|
||||
"The connection to {label} kept dropping while the model was writing a large action, "
|
||||
"so nothing was run. Check your network and send /retry; asking for the file in smaller "
|
||||
"pieces also helps."
|
||||
),
|
||||
# Rides failure_reason="loop_error" (advisory; the turn is incomplete, not failed).
|
||||
"local_processing_error": (
|
||||
"Hermes hit an internal error while handling the model's reply and stopped this turn. "
|
||||
+ _NEXT_STEPS_LOOP + "\n\nDetails: {detail}"
|
||||
),
|
||||
"reasoning_only": (
|
||||
"⚠️ {model} spent all of its output budget thinking and never wrote an answer. Lower "
|
||||
"its reasoning effort with `/reasoning low`, or switch to a different model with /model. "
|
||||
"Its last thoughts, which may contain the answer:\n\n{preview}"
|
||||
),
|
||||
"max_iterations_no_summary": (
|
||||
"I ran out of steps for this turn ({limit} tool calls) before finishing, and couldn't "
|
||||
"produce a summary. Send `continue` to keep going, or raise `max_iterations` in your config."
|
||||
),
|
||||
"nous_rate_limit": (
|
||||
"Wait for the reset and send /retry, or switch models with /model. To avoid waits, add "
|
||||
"a backup provider with `hermes fallback add`."
|
||||
),
|
||||
}
|
||||
_SITE_COPY: Dict[str, str] = {**_FAILURE_CODE_COPY, **_ONE_OFF_COPY}
|
||||
|
||||
|
||||
def site_copy(code: str, **fields: Any) -> str:
|
||||
"""Chat copy for a failure code or one-off loop outcome; unknown fields default to empty strings."""
|
||||
fields.setdefault("home", display_hermes_home())
|
||||
return _SITE_COPY[code].format_map(_Defaults(fields))
|
||||
|
||||
|
||||
def exhausted_copy(reason: str, *, label: str, attempts: int, summary: str) -> str:
|
||||
"""Chat copy once retries + fallback are exhausted (``max_retries_exhausted_result``)."""
|
||||
lead = _EXHAUSTED_LEADS.get(reason, _EXHAUSTED_DEFAULT_LEAD).format(label=label, attempts=attempts)
|
||||
return (
|
||||
f"{lead} — it looks temporarily unavailable. {_NEXT_STEPS_RETRY} To avoid this in future, "
|
||||
f"add a backup provider with `hermes fallback add`.\n\nProvider said: {summary}"
|
||||
)
|
||||
|
||||
|
||||
def nonretryable_copy(
|
||||
classified: Any, *, provider: Any, model: Any, summary: str, prefix_suggestion: Optional[str] = None,
|
||||
) -> str:
|
||||
"""Chat copy for a terminal non-retryable rejection (auth, model missing, TLS, generic 4xx)."""
|
||||
label = provider_label_for(provider)
|
||||
if getattr(classified, "is_auth", False):
|
||||
from agent.error_surface import auth_kind
|
||||
|
||||
template = _AUTH_COPY[auth_kind(str(provider or ""))]
|
||||
else:
|
||||
template = _NONRETRYABLE_COPY.get(classified.reason.value, _NONRETRYABLE_DEFAULT_COPY)
|
||||
prefix_hint = (
|
||||
f" If you typed the name yourself it may be missing its vendor prefix — did you mean "
|
||||
f"'{prefix_suggestion}'?"
|
||||
if prefix_suggestion else ""
|
||||
)
|
||||
body = template.format(label=label, model=model, home=display_hermes_home(), prefix_hint=prefix_hint)
|
||||
return f"{body}\n\nProvider said: {summary}"
|
||||
|
||||
|
||||
def content_policy_copy(*, label: str, summary: str) -> str:
|
||||
return (
|
||||
f"{label}'s safety filter refused this request, so the model didn't answer. "
|
||||
f"{CONTENT_POLICY_NEXT_STEPS}\n\nProvider said: {summary}"
|
||||
)
|
||||
|
||||
|
||||
def short_detail(exc: Any, limit: int = 200) -> str:
|
||||
"""First line of an exception's text, capped, for a trailing ``Details:`` line."""
|
||||
text = (str(exc) or type(exc).__name__).strip().splitlines()
|
||||
first = text[0] if text else type(exc).__name__
|
||||
return first if len(first) <= limit else first[: limit - 1] + "…"
|
||||
|
||||
|
||||
class _Defaults(dict):
|
||||
def __missing__(self, key: str) -> str:
|
||||
return ""
|
||||
@@ -12,6 +12,7 @@ from contextlib import suppress
|
||||
from typing import Any, Callable, List, Optional, Tuple
|
||||
|
||||
from agent.codex_responses_adapter import _summarize_user_message_for_log
|
||||
from agent.turn_failure_copy import exit_reason_failure, stamp_failure
|
||||
from agent.context_compressor import _DB_PERSISTED_MARKER
|
||||
from agent.message_content import flatten_message_text
|
||||
from agent.message_metadata import append_message, stamp_message_timestamp
|
||||
@@ -377,6 +378,7 @@ def _explain_abnormal_exit(agent, final_response, _turn_exit_reason, preserved_v
|
||||
_explanation = agent._format_turn_completion_explanation(
|
||||
_turn_exit_reason, getattr(agent, "_last_persistence_error_cause", None),
|
||||
db_path=getattr(getattr(agent, "_session_db", None), "db_path", None),
|
||||
model=str(getattr(agent, "model", "") or ""),
|
||||
)
|
||||
if _explanation:
|
||||
# Replace the bare sentinel; keep a partial fragment and append why.
|
||||
@@ -451,6 +453,15 @@ def finalize_turn(
|
||||
logger=logger,
|
||||
)
|
||||
|
||||
# Loop exits that are failures in their own right (outer-loop error cap, shutdown, context
|
||||
# that could not be shrunk) carry the verdict the UI descriptor needs; a bare
|
||||
# ``turn_exit_reason`` collapsed to code="unknown", retryable=True on every surface.
|
||||
# Advisory verdicts (``fails_turn=False``) only add the code: ``failed``/``completed`` keep
|
||||
# the loop's values so cron, kanban and transcript persistence behave as before.
|
||||
_exit_failure = None if interrupted else exit_reason_failure(_turn_exit_reason)
|
||||
if _exit_failure is not None and _exit_failure.fails_turn:
|
||||
failed = True
|
||||
|
||||
completed = (
|
||||
final_response is not None
|
||||
and not failed
|
||||
@@ -578,6 +589,10 @@ def finalize_turn(
|
||||
)
|
||||
_cause = getattr(agent, "_last_persistence_error_cause", None)
|
||||
result["failure_reason"] = "session_persistence_failed:" + (_cause or "unknown")
|
||||
elif _exit_failure is not None:
|
||||
if failed:
|
||||
result["error"] = final_response or str(_turn_exit_reason)
|
||||
stamp_failure(result, _exit_failure.reason, _exit_failure.retryable)
|
||||
# Cleanup failures are surfaced, but the response is returned either way (#8049).
|
||||
if _cleanup_errors:
|
||||
result["cleanup_errors"] = _cleanup_errors
|
||||
|
||||
@@ -507,7 +507,7 @@ def apply_retry_restarts(
|
||||
# All retries may exhaust with `response` still None; break out cleanly.
|
||||
if response is None:
|
||||
_turn_exit_reason = "all_retries_exhausted_no_response"
|
||||
print(f"{agent.log_prefix}❌ All API retries exhausted with no successful response.")
|
||||
agent._emit_status("❌ The model provider didn't answer after all retries. Send /retry, or switch models with /model.")
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return _verdict("break")
|
||||
return _verdict("fallthrough")
|
||||
|
||||
@@ -14,6 +14,7 @@ import sys
|
||||
from typing import Any
|
||||
|
||||
from agent.message_metadata import append_message
|
||||
from agent.turn_failure_copy import short_detail, site_copy
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
@@ -81,7 +82,11 @@ def handle_outer_loop_error(
|
||||
except Exception:
|
||||
pass
|
||||
_turn_exit_reason = "interpreter_shutdown"
|
||||
final_response = "Session is shutting down. Your conversation can be resumed with: hermes --resume <session-id>"
|
||||
failed = True
|
||||
_sid = getattr(agent, "session_id", None)
|
||||
final_response = site_copy(
|
||||
"interpreter_shutdown", resume=f" (CLI: `hermes --resume {_sid}`)" if _sid else "",
|
||||
)
|
||||
return _verdict("break")
|
||||
|
||||
# Deterministic local post-processing bugs (traceback via local helpers, never API
|
||||
@@ -96,9 +101,9 @@ def handle_outer_loop_error(
|
||||
)
|
||||
|
||||
if _is_local_processing_error:
|
||||
error_msg = f"Error during local message processing after OpenAI-compatible API call #{api_call_count}: {str(e)}"
|
||||
error_msg = f"Error during local message processing after API call #{api_call_count}: {str(e)}"
|
||||
else:
|
||||
error_msg = f"Error during OpenAI-compatible API call #{api_call_count}: {str(e)}"
|
||||
error_msg = f"Error during API call #{api_call_count}: {str(e)}"
|
||||
# Honor the _vprint contract: suppress_status_output silences hard failures;
|
||||
# quiet_mode -q still shows them. Traceback is logged below.
|
||||
if getattr(agent, "suppress_status_output", False):
|
||||
@@ -147,15 +152,20 @@ def handle_outer_loop_error(
|
||||
or api_call_count >= agent.max_iterations - 1
|
||||
or _outer_error_count >= _outer_error_cap
|
||||
):
|
||||
# finalize_turn stamps failure_reason=loop_error from the exit reason so the desktop
|
||||
# card stops reading these as "unknown"/retryable. The deterministic local bug keeps
|
||||
# ``failed`` as it was (an incomplete, not failed, turn): flipping it made every such
|
||||
# child run a strike against the kanban dispatcher breaker.
|
||||
detail = short_detail(e)
|
||||
if _is_local_processing_error:
|
||||
_turn_exit_reason = f"local_processing_error({error_msg[:80]})"
|
||||
final_response = f"I apologize, but I encountered an error while processing the model response: {error_msg}"
|
||||
final_response = site_copy("local_processing_error", detail=detail)
|
||||
elif _outer_error_count >= _outer_error_cap:
|
||||
failed = True
|
||||
_turn_exit_reason = f"repeated_outer_errors({error_msg[:80]})"
|
||||
final_response = f"I apologize, but I encountered repeated errors: {error_msg}"
|
||||
final_response = site_copy("loop_error", detail=detail)
|
||||
else:
|
||||
_turn_exit_reason = f"error_near_max_iterations({error_msg[:80]})"
|
||||
final_response = f"I apologize, but I encountered repeated errors: {error_msg}"
|
||||
final_response = site_copy("loop_error", detail=detail)
|
||||
return _verdict("break")
|
||||
return _verdict("fallthrough")
|
||||
|
||||
@@ -25,6 +25,7 @@ from agent.model_metadata import (
|
||||
get_context_length_from_provider_error, is_output_cap_error,
|
||||
parse_available_output_tokens_from_error,
|
||||
)
|
||||
from agent.turn_failure_copy import site_copy, stamp_failure
|
||||
from agent.turn_retry_state import TurnRetryState
|
||||
from utils import base_url_host_matches
|
||||
|
||||
@@ -99,7 +100,7 @@ class _Recovery(OverflowVerdict):
|
||||
if log:
|
||||
logger.error(*log)
|
||||
agent._persist_session(self.messages, self.conversation_history)
|
||||
result = {
|
||||
result = stamp_failure({
|
||||
"final_response": final_response,
|
||||
"messages": self.messages,
|
||||
"completed": False,
|
||||
@@ -107,7 +108,7 @@ class _Recovery(OverflowVerdict):
|
||||
"error": final_response,
|
||||
"partial": True,
|
||||
"failed": True,
|
||||
}
|
||||
}, "context_overflow", False)
|
||||
if compression_exhausted:
|
||||
# Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The
|
||||
# bloated transcript remains intact while future input can move to a clean session instead of
|
||||
@@ -124,7 +125,7 @@ class _Recovery(OverflowVerdict):
|
||||
return None
|
||||
if payload_too_large:
|
||||
return self.fail_turn(
|
||||
f"Request payload too large: max compression attempts ({cap}) reached.",
|
||||
site_copy("payload_too_large", model=self.agent.model),
|
||||
notices=(
|
||||
f"❌ Max compression attempts ({cap}) reached for payload-too-large error.",
|
||||
_RETRY_HINT,
|
||||
@@ -132,7 +133,7 @@ class _Recovery(OverflowVerdict):
|
||||
log=("%s413 compression failed after %d attempts.", self.agent.log_prefix, cap),
|
||||
)
|
||||
return self.fail_turn(
|
||||
f"Context length exceeded: max compression attempts ({cap}) reached.",
|
||||
site_copy("context_overflow", model=self.agent.model),
|
||||
notices=(f"❌ Max compression attempts ({cap}) reached.", _RETRY_HINT),
|
||||
log=("%sContext compression failed after %d attempts.", self.agent.log_prefix, cap),
|
||||
)
|
||||
@@ -268,7 +269,7 @@ def _recover_payload_too_large(st: _Recovery, _retry: TurnRetryState) -> Overflo
|
||||
return st.done("continue")
|
||||
|
||||
return st.fail_turn(
|
||||
"Request payload too large (413). Cannot compress further.",
|
||||
site_copy("payload_too_large", model=agent.model),
|
||||
notices=("❌ Payload too large and cannot compress further.", _RETRY_HINT),
|
||||
log=("%s413 payload too large. Cannot compress further.", agent.log_prefix),
|
||||
)
|
||||
@@ -407,10 +408,10 @@ def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: st
|
||||
|
||||
# Can't compress further and already at minimum tier.
|
||||
return st.fail_turn(
|
||||
f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further.",
|
||||
site_copy("context_overflow", model=agent.model),
|
||||
notices=(
|
||||
"❌ Context length exceeded and cannot compress further.",
|
||||
" 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.",
|
||||
f"❌ The conversation is too long for the model ({new_tokens:,} tokens) and cannot be shrunk further.",
|
||||
_RETRY_HINT,
|
||||
),
|
||||
log=("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}"),
|
||||
)
|
||||
|
||||
+90
-87
@@ -27,7 +27,12 @@ from agent.message_sanitization import (
|
||||
close_interrupted_tool_sequence,
|
||||
)
|
||||
from agent.thinking_timeout_guidance import build_thinking_timeout_guidance, is_thinking_timeout
|
||||
from agent.turn_failure_copy import (
|
||||
CONTENT_POLICY_NEXT_STEPS, content_policy_copy, exhausted_copy, nonretryable_copy, provider_label_for,
|
||||
site_copy, stamp_failure,
|
||||
)
|
||||
from agent.turn_retry_state import TurnRetryState
|
||||
from hermes_constants import display_hermes_home
|
||||
from utils import base_url_host_matches
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
@@ -663,11 +668,23 @@ def _welcome_tier_guidance(classified: Any, *, model: Any, in_chat: bool) -> str
|
||||
|
||||
# Terminal status label per non-retryable reason (default names the HTTP status).
|
||||
_NONRETRYABLE_LABELS = {
|
||||
FailoverReason.content_policy_blocked: "Provider safety filter blocked this request",
|
||||
FailoverReason.ssl_cert_verification: "TLS certificate verification failed",
|
||||
FailoverReason.content_policy_blocked: "The provider's safety filter refused this request",
|
||||
FailoverReason.ssl_cert_verification: "The provider's security certificate could not be verified",
|
||||
}
|
||||
|
||||
|
||||
def _missing_vendor_prefix_suggestion(api_error: Exception, provider: Any, model: Any) -> Optional[str]:
|
||||
"""Prefixed catalogue id when a bare 404 most likely means ``vendor/model`` lost its prefix."""
|
||||
if getattr(api_error, "status_code", None) != 404:
|
||||
return None
|
||||
try:
|
||||
from hermes_cli.model_normalize import suggest_prefixed_model_id
|
||||
|
||||
return suggest_prefixed_model_id(str(provider or ""), str(model or ""))
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def nonretryable_client_error_result(
|
||||
agent: Any, api_error: Exception, classified: Any, *, status_code: Optional[int],
|
||||
api_kwargs: Any, api_messages: Any, messages: List[Dict[str, Any]], conversation_history: Any,
|
||||
@@ -677,9 +694,7 @@ def nonretryable_client_error_result(
|
||||
the retry trace, print auth / billing / content-policy / TLS guidance, persist (skipped
|
||||
for likely context-overflow 400s so the failure does not grow the session), build result."""
|
||||
# Result/guidance helpers stay in the loop module (tests import + patch them there).
|
||||
from agent.conversation_loop import (
|
||||
_CONTENT_POLICY_RECOVERY_HINT, _billing_failure_result, _content_policy_blocked_result,
|
||||
)
|
||||
from agent.conversation_loop import _billing_failure_result, _content_policy_blocked_result
|
||||
|
||||
if api_kwargs is not None:
|
||||
agent._dump_api_request_debug(api_kwargs, reason="non_retryable_client_error", error=api_error)
|
||||
@@ -688,15 +703,18 @@ def nonretryable_client_error_result(
|
||||
# Summarize once: Cloudflare/proxy HTML pages and raw provider bodies must be
|
||||
# collapsed here or they leak verbatim via the ``error`` field.
|
||||
_nonretryable_summary = agent._summarize_api_error(api_error)
|
||||
_label = _NONRETRYABLE_LABELS.get(classified.reason, f"Non-retryable error (HTTP {status_code})")
|
||||
_plabel = provider_label_for(provider)
|
||||
_label = _NONRETRYABLE_LABELS.get(classified.reason, f"{_plabel} rejected the request and retrying won't help")
|
||||
agent._emit_status(f"❌ {_label}: {_nonretryable_summary}")
|
||||
_vlines(
|
||||
agent,
|
||||
f"❌ Non-retryable client error (HTTP {status_code}). Aborting.",
|
||||
f" 🔌 Provider: {provider} Model: {model}",
|
||||
f" 🌐 Endpoint: {base_url}",
|
||||
)
|
||||
# The endpoint/status trace is developer detail: verbose only (the log has it always).
|
||||
if getattr(agent, "verbose_logging", False):
|
||||
_vlines(
|
||||
agent,
|
||||
f" 🔌 Provider: {provider} Model: {model} (HTTP {status_code})",
|
||||
f" 🌐 Endpoint: {base_url}",
|
||||
)
|
||||
_welcome_hint = _welcome_tier_guidance(classified, model=model, in_chat=False)
|
||||
_prefix_suggestion = _missing_vendor_prefix_suggestion(api_error, provider, model)
|
||||
if _welcome_hint:
|
||||
# A free-tier gate or a wrong-host refusal: the way forward is a sign-in or another
|
||||
# provider, never the key/credits advice below.
|
||||
@@ -705,28 +723,30 @@ def nonretryable_client_error_result(
|
||||
_print_nonretryable_auth_guidance(
|
||||
agent, classified, status_code=status_code, provider=provider, base_url=base_url, model=model
|
||||
)
|
||||
else:
|
||||
_vlines(agent, " 💡 This type of error won't be fixed by retrying.")
|
||||
elif classified.reason == FailoverReason.model_not_found:
|
||||
_vlines(agent, f" 💡 Model '{model}' isn't available on {_plabel}. Pick another with /model.")
|
||||
if _prefix_suggestion:
|
||||
_vlines(agent, f" Did you mean '{_prefix_suggestion}'? It looks like the vendor prefix is missing.")
|
||||
elif classified.reason not in _NONRETRYABLE_LABELS:
|
||||
_vlines(agent, f" 💡 Fix: pick another model (/model), or check `{display_hermes_home()}/logs/agent.log`.")
|
||||
# Content-policy blocks: the provider refused this prompt, so recovery is a rephrase
|
||||
# or another model, not key/retry advice.
|
||||
if classified.reason == FailoverReason.content_policy_blocked:
|
||||
_vlines(
|
||||
agent,
|
||||
" 💡 The provider's safety filter rejected this specific prompt.",
|
||||
" • Try rephrasing the request, narrowing the context, or splitting into smaller steps.",
|
||||
" • Configure a fallback provider so future blocks route automatically:",
|
||||
" hermes fallback add (interactive picker — same as `hermes model`)",
|
||||
f" 💡 {CONTENT_POLICY_NEXT_STEPS}",
|
||||
" To route future blocks to another provider automatically: hermes fallback add",
|
||||
)
|
||||
# TLS certificate failures are environment problems — name the knobs for each cause.
|
||||
if classified.reason == FailoverReason.ssl_cert_verification:
|
||||
_vlines(
|
||||
agent,
|
||||
" 💡 The TLS certificate chain could not be verified. This fails the same",
|
||||
" 💡 Hermes couldn't verify the provider's security certificate. This fails the same",
|
||||
" way on every retry — fix the environment, then try again:",
|
||||
" • Corporate TLS-inspecting proxy? Point Python at its CA bundle:",
|
||||
" export SSL_CERT_FILE=/path/to/corp-ca.pem (also REQUESTS_CA_BUNDLE)",
|
||||
" • Missing/stale system CA store? Install/refresh it:",
|
||||
" pip install --upgrade certifi (macOS: run 'Install Certificates.command')",
|
||||
" • Missing/stale system CA store? Refresh it (in Hermes's venv: `uv pip install",
|
||||
" --upgrade certifi`; macOS: run 'Install Certificates.command').",
|
||||
" • Self-signed local endpoint (llama.cpp, LM Studio, vLLM)? Use http://",
|
||||
" for localhost, or add the server's cert to your trust store.",
|
||||
)
|
||||
@@ -740,14 +760,10 @@ def nonretryable_client_error_result(
|
||||
else:
|
||||
agent._persist_session(messages, conversation_history)
|
||||
if classified.reason == FailoverReason.content_policy_blocked:
|
||||
_policy_response = (
|
||||
"⚠️ The model provider's safety filter blocked this request "
|
||||
"(not a Hermes/gateway failure).\n\n"
|
||||
f"Provider message: {_nonretryable_summary}\n\n"
|
||||
f"{_CONTENT_POLICY_RECOVERY_HINT}"
|
||||
)
|
||||
return _content_policy_blocked_result(
|
||||
messages, api_call_count, final_response=_policy_response, error_detail=_nonretryable_summary,
|
||||
messages, api_call_count,
|
||||
final_response="⚠️ " + content_policy_copy(label=_plabel, summary=_nonretryable_summary),
|
||||
error_detail=_nonretryable_summary,
|
||||
)
|
||||
# Billing walls get the same structured recovery descriptor as the max-retries path
|
||||
# so every surface renders one consistent signal.
|
||||
@@ -756,9 +772,14 @@ def nonretryable_client_error_result(
|
||||
classified=classified, summary=_nonretryable_summary, messages=messages,
|
||||
api_call_count=api_call_count, provider=provider, base_url=base_url, model=model,
|
||||
)
|
||||
_final_response = _nonretryable_summary
|
||||
if _welcome_hint:
|
||||
_final_response += f"\n\n{_welcome_tier_guidance(classified, model=model, in_chat=True)}"
|
||||
_final_response = f"{_nonretryable_summary}\n\n{_welcome_tier_guidance(classified, model=model, in_chat=True)}"
|
||||
else:
|
||||
# Every surface reads final_response; the CLI hint lines above never reach chat.
|
||||
_final_response = nonretryable_copy(
|
||||
classified, provider=provider, model=model, summary=_nonretryable_summary,
|
||||
prefix_suggestion=_prefix_suggestion,
|
||||
)
|
||||
result = _failed_turn_result(_final_response, messages, api_call_count, _nonretryable_summary)
|
||||
# Same verdict fields as the max-retries path: without them the UI descriptor
|
||||
# (agent/error_surface.py) reads a rejected OAuth token as a retryable
|
||||
@@ -839,18 +860,7 @@ def max_retries_exhausted_result(
|
||||
# Distinct from _is_stream_drop; detection lives in agent.thinking_timeout_guidance.
|
||||
_is_thinking_timeout = is_thinking_timeout(classified, model, error_msg)
|
||||
if _is_thinking_timeout:
|
||||
_vlines(
|
||||
agent,
|
||||
" 💡 The model's thinking phase exceeded the upstream proxy's idle "
|
||||
"timeout before the first content token arrived. This is a known issue with "
|
||||
"reasoning models behind cloud gateways (NVIDIA NIM, OpenAI, Anthropic, DeepSeek).",
|
||||
" Workarounds in priority order:",
|
||||
f" 1. Set `providers.{provider}.models.{model}.stale_timeout_seconds: 900` "
|
||||
"in `~/.hermes/config.yaml` to extend the per-call timeout. (Hermes's built-in floor is 600s for "
|
||||
"known reasoning models — if you still see this after raising, the upstream cap is even shorter.)",
|
||||
" 2. Lower `reasoning_budget` or set `reasoning_effort: medium` on this model if the provider supports it.",
|
||||
" 3. Use a smaller / faster reasoning model if the task doesn't require deep thinking.",
|
||||
)
|
||||
_vlines(agent, f" 💡 {build_thinking_timeout_guidance(provider=provider, model=model).strip()}")
|
||||
|
||||
logger.error(
|
||||
"%sAPI call failed after %s retries. %s | provider=%s model=%s msgs=%s tokens=~%s",
|
||||
@@ -872,21 +882,23 @@ def max_retries_exhausted_result(
|
||||
provider, base_url, model, _billing_guidance, unverified=_billing_unverified
|
||||
)
|
||||
else:
|
||||
_final_response = f"API call failed after {max_retries} retries: {_final_summary}"
|
||||
# Every surface reads final_response (the 💡 lines above are CLI-only), so the chat
|
||||
# text carries the plain what-happened + next step itself.
|
||||
_final_response = exhausted_copy(
|
||||
classified.reason.value, label=provider_label_for(provider), attempts=max_retries,
|
||||
summary=_final_summary,
|
||||
)
|
||||
if _welcome_hint:
|
||||
_final_response += f"\n\n{_welcome_tier_guidance(classified, model=model, in_chat=True)}"
|
||||
if _is_thinking_timeout:
|
||||
# Thinking-timeout guidance overrides stream-drop guidance, which would wrongly
|
||||
# suggest splitting large file writes.
|
||||
_final_response += build_thinking_timeout_guidance(provider=provider, model=model)
|
||||
_final_response += "\n\n" + build_thinking_timeout_guidance(provider=provider, model=model)
|
||||
elif _is_stream_drop:
|
||||
_final_response += (
|
||||
"\n\nThe provider's stream connection keeps "
|
||||
"dropping — this often happens when generating "
|
||||
"very large tool call responses (e.g. write_file "
|
||||
"with long content). Try asking me to use "
|
||||
"execute_code with Python's open() for large "
|
||||
"files, or to write in smaller sections."
|
||||
"\n\nThe connection kept dropping while the model was writing — this often "
|
||||
"happens when it writes a very large file in one go. Ask me to write the file in "
|
||||
"smaller sections (or via execute_code with Python's open())."
|
||||
)
|
||||
result = _failed_turn_result(_final_response, messages, api_call_count, _final_summary)
|
||||
result.update({
|
||||
@@ -921,20 +933,21 @@ def log_api_error_attempt(
|
||||
_provider = getattr(agent, "provider", "unknown")
|
||||
_base = getattr(agent, "base_url", "unknown")
|
||||
_model = getattr(agent, "model", "unknown")
|
||||
_status_code_str = f" [HTTP {status_code}]" if status_code else ""
|
||||
_blines(
|
||||
agent,
|
||||
f"⚠️ API call failed (attempt {retry_count}/{max_retries}): {error_type}{_status_code_str}",
|
||||
f" 🔌 Provider: {_provider} Model: {_model}",
|
||||
f" 🌐 Endpoint: {_base}",
|
||||
f" 📝 Error: {_error_summary}",
|
||||
)
|
||||
if status_code and status_code < 500:
|
||||
_err_body = getattr(api_error, "body", None)
|
||||
_err_body_str = str(_err_body)[:300] if _err_body else None
|
||||
if _err_body_str:
|
||||
_blines(agent, f" 📋 Details: {_err_body_str}")
|
||||
_blines(agent, f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens")
|
||||
_blines(agent, f"⚠️ Attempt {retry_count}/{max_retries} failed: {_error_summary}")
|
||||
# Exception class, endpoint, raw body and token counts are developer detail: verbose only.
|
||||
if getattr(agent, "verbose_logging", False):
|
||||
_status_code_str = f" [HTTP {status_code}]" if status_code else ""
|
||||
_blines(
|
||||
agent,
|
||||
f" 🔌 {error_type}{_status_code_str} Provider: {_provider} Model: {_model}",
|
||||
f" 🌐 Endpoint: {_base}",
|
||||
)
|
||||
if status_code and status_code < 500:
|
||||
_err_body = getattr(api_error, "body", None)
|
||||
_err_body_str = str(_err_body)[:300] if _err_body else None
|
||||
if _err_body_str:
|
||||
_blines(agent, f" 📋 Details: {_err_body_str}")
|
||||
_blines(agent, f" ⏱️ Elapsed: {elapsed_time:.2f}s Context: {len(api_messages)} msgs, ~{approx_tokens:,} tokens")
|
||||
|
||||
if agent._is_openrouter_url() and "support tool use" in error_msg:
|
||||
_blines(agent, f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings.")
|
||||
@@ -949,19 +962,13 @@ def log_api_error_attempt(
|
||||
|
||||
# Bare 404 on a ``vendor/model`` catalogue usually means the id lost its prefix; the
|
||||
# provider never names the model, so we do.
|
||||
if getattr(api_error, "status_code", None) == 404:
|
||||
try:
|
||||
from hermes_cli.model_normalize import suggest_prefixed_model_id
|
||||
|
||||
_suggestion = suggest_prefixed_model_id(_provider, _model)
|
||||
except Exception:
|
||||
_suggestion = None
|
||||
if _suggestion:
|
||||
_blines(
|
||||
agent,
|
||||
f" 💡 Model '{_model}' is not a valid id for provider {_provider} — it is missing its vendor prefix.",
|
||||
f" Did you mean '{_suggestion}'? Re-pick it with `hermes model`.",
|
||||
)
|
||||
_suggestion = _missing_vendor_prefix_suggestion(api_error, _provider, _model)
|
||||
if _suggestion:
|
||||
_blines(
|
||||
agent,
|
||||
f" 💡 Model '{_model}' is not a valid id for provider {_provider} — it is missing its vendor prefix.",
|
||||
f" Did you mean '{_suggestion}'? Re-pick it with /model.",
|
||||
)
|
||||
return error_type, error_msg, _provider, _base, _model
|
||||
|
||||
|
||||
@@ -1363,25 +1370,21 @@ def route_classified_error(
|
||||
agent._flush_status_buffer()
|
||||
_vlines(
|
||||
agent,
|
||||
"❌ Context overflow, but auto-compaction is disabled (compression.enabled: false).",
|
||||
" 💡 Run /compress to compact manually, /new to start fresh, "
|
||||
"switch to a larger-context model, or reduce attachments.",
|
||||
"❌ The conversation is too long for the model and automatic shrinking is off (compression.enabled: false).",
|
||||
" 💡 Run /compress to shrink it now, /new to start fresh, "
|
||||
"pick a model with a bigger context window, or remove attachments.",
|
||||
)
|
||||
logger.error(
|
||||
f"{agent.log_prefix}Context overflow ({classified.reason.value}) with "
|
||||
f"auto-compaction disabled — not compressing."
|
||||
)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
_final_response = (
|
||||
"Context overflow and auto-compaction is disabled "
|
||||
"(compression.enabled: false). Run /compress to compact manually, "
|
||||
"/new to start fresh, or switch to a larger-context model."
|
||||
)
|
||||
return _verdict("return", {
|
||||
_final_response = site_copy("compression_disabled", model=agent.model)
|
||||
return _verdict("return", stamp_failure({
|
||||
"final_response": _final_response, "messages": messages, "completed": False,
|
||||
"api_calls": api_call_count, "error": _final_response, "partial": True, "failed": True,
|
||||
"compaction_disabled": True,
|
||||
})
|
||||
}, "context_overflow", False))
|
||||
|
||||
# Anthropic 429 "Extra usage is required for long context requests" is a
|
||||
# subscription-tier limit, not transient: cap at 200k and compress.
|
||||
|
||||
@@ -14,6 +14,7 @@ import time
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from agent.turn_api_call import stop_thinking_spinner
|
||||
from agent.turn_failure_copy import invalid_response_failure_reason, provider_label_for, site_copy, stamp_failure
|
||||
from agent.turn_truncation import handle_content_policy_refusal, recover_from_truncation
|
||||
from agent.turn_usage import record_response_usage
|
||||
|
||||
@@ -287,15 +288,23 @@ def retry_invalid_response(
|
||||
agent._emit_status(f"❌ Max retries ({max_retries}) exceeded for invalid responses. Giving up.")
|
||||
logger.error("%sInvalid API response after %d retries.", agent.log_prefix, max_retries)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
_final_response = f"Invalid API response after {max_retries} retries: {_failure_hint}"
|
||||
return _verdict("return", {
|
||||
# "model=<id>" is describe_invalid_response's OpenRouter fallback, not a provider name.
|
||||
_label = (
|
||||
provider_label_for(agent.provider)
|
||||
if provider_name in ("Unknown", "") or provider_name.startswith("model=")
|
||||
else provider_name
|
||||
)
|
||||
_final_response = site_copy(
|
||||
"invalid_response", label=_label, attempts=max_retries, detail=_failure_hint,
|
||||
)
|
||||
return _verdict("return", stamp_failure({
|
||||
"final_response": _final_response,
|
||||
"messages": messages,
|
||||
"completed": False,
|
||||
"api_calls": api_call_count,
|
||||
"error": _final_response,
|
||||
"error": f"Invalid API response after {max_retries} retries: {_failure_hint}",
|
||||
"failed": True,
|
||||
})
|
||||
}, invalid_response_failure_reason(response), True))
|
||||
|
||||
wait_time = jittered_backoff(retry_count, base_delay=5.0, max_delay=120.0)
|
||||
agent._buffer_vprint(f"⏳ Retrying in {wait_time:.1f}s ({_failure_hint})...")
|
||||
|
||||
@@ -16,6 +16,7 @@ from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.message_metadata import append_message
|
||||
from agent.message_sanitization import close_interrupted_tool_sequence, coalesce_tool_call_id
|
||||
from agent.turn_failure_copy import site_copy, stamp_failure
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
@@ -56,14 +57,14 @@ def _partial_exit(agent, messages, conversation_history, api_call_count, final_r
|
||||
This path never reaches finalize_turn, so persist here."""
|
||||
close_interrupted_tool_sequence(messages, final_response)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return {
|
||||
return stamp_failure({
|
||||
"final_response": final_response,
|
||||
"messages": messages,
|
||||
"api_calls": api_call_count,
|
||||
"completed": False,
|
||||
"partial": True,
|
||||
"error": final_response,
|
||||
}
|
||||
}, "truncated", True)
|
||||
|
||||
|
||||
def validate_tool_calls(
|
||||
@@ -176,8 +177,7 @@ def validate_tool_calls(
|
||||
agent._invalid_json_retries = 0
|
||||
agent._cleanup_task_resources(effective_task_id)
|
||||
return _verdict("return", _partial_exit(
|
||||
agent, messages, conversation_history, api_call_count,
|
||||
"Response truncated due to output length limit",
|
||||
agent, messages, conversation_history, api_call_count, site_copy("truncated"),
|
||||
))
|
||||
|
||||
agent._invalid_json_retries += 1
|
||||
|
||||
+19
-18
@@ -12,13 +12,14 @@ from __future__ import annotations
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from agent.error_classifier import FailoverReason
|
||||
from agent.message_metadata import append_message
|
||||
from agent.message_sanitization import close_interrupted_tool_sequence
|
||||
from agent.repetition_guard import is_repetition_dominated
|
||||
from agent.turn_api_call import stop_thinking_spinner
|
||||
from agent.turn_failure_copy import content_policy_copy, provider_label_for, site_copy, stamp_failure
|
||||
from agent.turn_retry_state import TurnRetryState
|
||||
from agent.usage_pricing import normalize_usage
|
||||
from hermes_constants import PARTIAL_STREAM_STUB_ID
|
||||
@@ -27,8 +28,8 @@ logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
_CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_messages"}
|
||||
_THINK_TAG_RE = re.compile(r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', re.IGNORECASE)
|
||||
_TRUNCATED_FINAL = "Response truncated due to output length limit"
|
||||
_FIRST_TRUNCATED_FINAL = "First response truncated due to output length limit"
|
||||
_TRUNCATED_FINAL = site_copy("truncated")
|
||||
_FIRST_TRUNCATED_FINAL = _TRUNCATED_FINAL
|
||||
# #106260: a stream that died on a context-overflow error after partial delivery must not seed a
|
||||
# continuation — the transcript already cannot fit, and appending the partial stub grows every
|
||||
# later request into the same overflow. End the turn via the recovery contract instead.
|
||||
@@ -157,20 +158,22 @@ class _Trunc(TruncationVerdict):
|
||||
self, final_response: str, error: Optional[str] = None, *,
|
||||
result_messages: Optional[List[Dict[str, Any]]] = None, cleanup: bool = True,
|
||||
failed: bool = False, compression_exhausted: bool = False,
|
||||
failure: Tuple[str, bool] = ("truncated", True),
|
||||
) -> TruncationVerdict:
|
||||
"""Persist and end the turn as partial (or ``failed``).
|
||||
|
||||
``compression_exhausted`` forwards the #98722 typed bit so the gateway can
|
||||
move future input off a bloated session (run_turn.py consumes it).
|
||||
move future input off a bloated session (run_turn.py consumes it). ``failure`` is
|
||||
the ``(failure_reason, retryable)`` verdict for the UI descriptor.
|
||||
"""
|
||||
agent = self.agent
|
||||
if cleanup:
|
||||
agent._cleanup_task_resources(self.effective_task_id)
|
||||
agent._persist_session(self.messages, self.conversation_history)
|
||||
return self.done("return", partial_result(
|
||||
return self.done("return", stamp_failure(partial_result(
|
||||
self.messages if result_messages is None else result_messages, self.api_call_count,
|
||||
final_response, error, failed=failed, compression_exhausted=compression_exhausted,
|
||||
))
|
||||
), *failure))
|
||||
|
||||
@property
|
||||
def is_stub(self) -> bool:
|
||||
@@ -334,7 +337,7 @@ def _retry_truncated_tool_call(st: _Trunc, api_kwargs: Any) -> TruncationVerdict
|
||||
f"{agent.log_prefix}⚠️ Stream kept dropping mid tool-call after 4 retries — the action was not executed.",
|
||||
force=True,
|
||||
)
|
||||
_final_response = "Stream repeatedly dropped mid tool-call (network); the tool was not executed"
|
||||
_final_response = site_copy("stream_dropped_tool_call", label=provider_label_for(agent.provider))
|
||||
else:
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.",
|
||||
@@ -344,7 +347,10 @@ def _retry_truncated_tool_call(st: _Trunc, api_kwargs: Any) -> TruncationVerdict
|
||||
agent._cleanup_task_resources(st.effective_task_id)
|
||||
# Prior tool batches can leave a tool-result tail; this path never reaches finalize_turn.
|
||||
close_interrupted_tool_sequence(st.messages, _final_response)
|
||||
return st.end_turn(_final_response, cleanup=False)
|
||||
return st.end_turn(
|
||||
_final_response, cleanup=False,
|
||||
failure=(FailoverReason.timeout.value if st.is_stub else "truncated", True),
|
||||
)
|
||||
|
||||
|
||||
def recover_from_truncation(
|
||||
@@ -403,6 +409,7 @@ def recover_from_truncation(
|
||||
error=_CONTEXT_OVERFLOW_PARTIAL_FINAL,
|
||||
failed=True,
|
||||
compression_exhausted=True,
|
||||
failure=("context_overflow", False),
|
||||
)
|
||||
|
||||
_trunc_msg = normalize_response_for_agent(agent, response)
|
||||
@@ -557,9 +564,7 @@ def handle_content_policy_refusal(
|
||||
"""HTTP-200 refusal (``finish_reason`` ``content_filter`` / ``guardrail_intervened``).
|
||||
Deterministic for the unchanged prompt — never retried: one configured-fallback try,
|
||||
else surface the refusal (explanation may live only in the reasoning channel)."""
|
||||
from agent.conversation_loop import (
|
||||
_CONTENT_POLICY_RECOVERY_HINT, _arm_fallback_restart, _content_policy_blocked_result
|
||||
)
|
||||
from agent.conversation_loop import _arm_fallback_restart, _content_policy_blocked_result
|
||||
|
||||
_refusal_result = normalize_response_for_agent(agent, response)
|
||||
_refusal_text = (getattr(_refusal_result, "content", None) or "").strip()
|
||||
@@ -590,13 +595,9 @@ def handle_content_policy_refusal(
|
||||
_refusal_log or "(no text)",
|
||||
)
|
||||
agent._emit_status("⚠️ The model declined to respond to this request (safety refusal).")
|
||||
_refusal_detail = (
|
||||
f"Model's explanation: {_refusal_text}" if _refusal_text else "The model returned no explanation."
|
||||
)
|
||||
_refusal_response = (
|
||||
"⚠️ The model declined to respond to this request (safety refusal — not a Hermes/gateway failure).\n\n"
|
||||
f"{_refusal_detail}\n\n"
|
||||
f"{_CONTENT_POLICY_RECOVERY_HINT}"
|
||||
_refusal_response = "⚠️ " + content_policy_copy(
|
||||
label=provider_label_for(agent.provider),
|
||||
summary=_refusal_text or "the model returned no explanation",
|
||||
)
|
||||
agent._cleanup_task_resources(effective_task_id)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
|
||||
@@ -2860,18 +2860,21 @@ class HermesCLI(CLIProcessNotificationsMixin, CLIAgentSetupMixin, CLICommandsMix
|
||||
# the store before relying on resume.
|
||||
self._session_db_unavailable = True
|
||||
logger.warning("Failed to initialize SessionDB — session will NOT be indexed for search: %s", e)
|
||||
from hermes_state_user_copy import describe_storage_failure, storage_failure_details
|
||||
failure = describe_storage_failure(e)
|
||||
try:
|
||||
Console(stderr=True).print(
|
||||
"[bold yellow]⚠ Session store unavailable[/bold yellow] — "
|
||||
"this conversation will [bold]NOT be saved[/bold] to disk and "
|
||||
"cannot be resumed later. Searching past sessions is also disabled.\n"
|
||||
f" Reason: {e}\n"
|
||||
" Fix the state.db store (e.g. `hermes update` to rebuild the venv) to restore persistence."
|
||||
"this conversation will [bold]NOT be saved[/bold] and cannot be resumed later. "
|
||||
"Searching past sessions is also disabled.\n"
|
||||
f" Reason: {failure.gloss}.\n"
|
||||
f" {failure.action}\n"
|
||||
f" [dim]Details: {storage_failure_details(e)}[/dim]"
|
||||
)
|
||||
except Exception:
|
||||
print(
|
||||
"WARNING: Session store unavailable — this conversation will NOT be "
|
||||
f"saved to disk and cannot be resumed later. Reason: {e}"
|
||||
f"saved and cannot be resumed later. Reason: {failure.gloss}. {failure.action}"
|
||||
)
|
||||
_run_state_db_auto_maintenance(self._session_db)
|
||||
_run_checkpoint_auto_maintenance()
|
||||
@@ -3110,19 +3113,30 @@ class HermesCLI(CLIProcessNotificationsMixin, CLIAgentSetupMixin, CLICommandsMix
|
||||
self.preloaded_skills += [name for name in loaded_skills if name not in self.preloaded_skills]
|
||||
|
||||
def _show_tool_availability_warnings(self):
|
||||
"""Warn about tools disabled by missing API keys (not system deps)."""
|
||||
"""Warn about toolsets switched off at startup (missing API keys, unusable terminal backend)."""
|
||||
try:
|
||||
# Runs on a daemon thread on the snapshot fast path: keep the imports to modules the
|
||||
# registry walk already loaded plus the pure notices module (a heavy import here races
|
||||
# importlib's module locks against the main thread).
|
||||
from model_tools import check_tool_availability
|
||||
from hermes_cli.tool_availability_notices import (
|
||||
current_terminal_backend, filter_to_enabled_toolsets, tool_availability_warning_lines,
|
||||
)
|
||||
from tools.terminal_tool import terminal_backend_unavailable_reason
|
||||
from toolsets import resolve_toolset
|
||||
|
||||
available, unavailable = check_tool_availability()
|
||||
api_key_missing = [u for u in unavailable if u["missing_vars"]]
|
||||
|
||||
if api_key_missing:
|
||||
_, unavailable = check_tool_availability()
|
||||
# Only toolsets this CLI session actually has. The selection is usually a composite bundle
|
||||
# (``hermes-cli``), so expand it to tool names before matching — a raw name comparison
|
||||
# matched nothing on a default install and silently dropped the terminal notice.
|
||||
unavailable = filter_to_enabled_toolsets(unavailable, self.enabled_toolsets or [], resolve_toolset)
|
||||
lines = tool_availability_warning_lines(
|
||||
unavailable, terminal_reason=terminal_backend_unavailable_reason(),
|
||||
terminal_backend=current_terminal_backend())
|
||||
if lines:
|
||||
self._console_print()
|
||||
self._console_print("[yellow]⚠️ Some tools disabled (missing API keys):[/]")
|
||||
for item in api_key_missing:
|
||||
self._console_print(f" [dim]• {item['name']}[/] [dim italic]({', '.join(item['missing_vars'])})[/]")
|
||||
self._console_print("[dim] Run 'hermes setup' to configure[/]")
|
||||
for line in lines:
|
||||
self._console_print(line)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@@ -3398,8 +3412,10 @@ class HermesCLI(CLIProcessNotificationsMixin, CLIAgentSetupMixin, CLICommandsMix
|
||||
_cprint(f"{_DIM}Did you mean: {', '.join(sorted(matches))}?{_RST}")
|
||||
else:
|
||||
# Exact token with no handler (never re-dispatch the same token: recursion), or no match.
|
||||
_cprint(f"\033[1;31mUnknown command: {cmd_lower}{_RST}")
|
||||
_cprint(f"{_DIM}{_ACCENT}Type /help for available commands{_RST}")
|
||||
from hermes_cli.cli_unknown_command import unknown_command_lines
|
||||
lead, pointer = unknown_command_lines(cmd_lower, all_known)
|
||||
_cprint(f"\033[1;31m{lead}{_RST}")
|
||||
_cprint(f"{_DIM}{_ACCENT}{pointer}{_RST}")
|
||||
return True
|
||||
|
||||
def _drain_interrupt_queue_to_pending_input(self) -> None:
|
||||
|
||||
+78
-164
@@ -99,19 +99,20 @@ def _set_cron_session_title(session_db, session_id, base_title):
|
||||
|
||||
|
||||
def _fallback_chain_phrase() -> str:
|
||||
"""Fallback-chain clause for a provider-failure message: "exhausted" vs "none configured" (most
|
||||
installs). Fails open to the ambiguous wording if config can't be read — never crash delivery.
|
||||
"""Backup-provider clause for a provider-failure notice: "the backups failed too" vs "none
|
||||
configured" (most installs). Fails open to the former if config can't be read — never crash
|
||||
delivery.
|
||||
"""
|
||||
try:
|
||||
cfg = load_config() or {}
|
||||
chain = get_fallback_chain(cfg)
|
||||
except Exception:
|
||||
return "Fallback chain was exhausted or unavailable."
|
||||
return "No backup provider succeeded either."
|
||||
if chain:
|
||||
return "Fallback chain was exhausted or unavailable."
|
||||
return "No backup provider succeeded either."
|
||||
return (
|
||||
"No fallback chain configured — add one with `hermes fallback add`, "
|
||||
"or set a cron fleet default via `cron.model` + `cron.model_provider` in config.yaml."
|
||||
"No backup provider is configured — add one with `hermes fallback add`, "
|
||||
"or set a cron-wide default via `cron.model` + `cron.model_provider` in config.yaml."
|
||||
)
|
||||
|
||||
|
||||
@@ -216,100 +217,57 @@ def _log_tick_yield_once(reason: str) -> None:
|
||||
|
||||
|
||||
def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
|
||||
"""Compact one-line failure message for chat delivery (full details stay in cron output)."""
|
||||
"""One-line failure notice for chat delivery (full details stay in the run output).
|
||||
|
||||
Deterministic scheduler/script shapes are matched first (their text can contain "timed out"
|
||||
and would otherwise be blamed on the model service); everything else goes through the shared
|
||||
``classify_api_error`` verdict and the copy table in ``scheduler_failure_copy``."""
|
||||
from cron.scheduler_failure_copy import (
|
||||
classify_cron_failure_reason, generic_failure_notice, inactivity_notice,
|
||||
provider_failure_notice, script_timeout_notice)
|
||||
|
||||
job_name = job.get("name") or job.get("id") or "cron job"
|
||||
job_id = job.get("id") or job_name
|
||||
text = (error or "unknown error").strip()
|
||||
lower = text.lower()
|
||||
|
||||
# no_agent jobs never reach a model, so provider errors are structurally impossible for them.
|
||||
# Gate on job MODE before substring matching, or a script's own wording ("timed out", "429")
|
||||
# would blame the wrong subsystem; the generic cleaner below reports what actually happened.
|
||||
provider_reachable = not job.get("no_agent")
|
||||
|
||||
# Script runner contract ("Script timed out after {n}s: {path}") — also for agent jobs with a
|
||||
# context script. Must precede generic timeout matching so it never claims a provider fallback.
|
||||
# context script. Must precede provider classification so it never claims a model failure.
|
||||
# See #78503, #82460.
|
||||
if lower.startswith("script timed out"):
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: script timed out. "
|
||||
"No model was invoked. Full details saved in cron output."
|
||||
)
|
||||
return script_timeout_notice(job_name, job_id)
|
||||
|
||||
# Whole-token 429: substrings in job ids/ports/hashes tripped false rate-limit alerts.
|
||||
if provider_reachable and (
|
||||
# Provider/API failures are the common noisy path. Keep these short. Match 429 as a whole token
|
||||
# (#83188 @cation98): bare substring matching let identifiers containing those digits (job ids,
|
||||
# ports, hashes) trip a false "provider rate limit" alert.
|
||||
re.search(r"\b429\b", text) or "rate limit" in lower or "usage limit" in lower
|
||||
):
|
||||
reason = "rate limit"
|
||||
if "weekly usage limit" in lower:
|
||||
reason = "weekly usage limit"
|
||||
elif "quota" in lower:
|
||||
reason = "quota limit"
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: provider {reason}. "
|
||||
f"{_fallback_chain_phrase()} "
|
||||
"Full details saved in cron output."
|
||||
)
|
||||
|
||||
# Scheduler inactivity watchdog shape ("idle for {n}s (limit {m}s)"). Must precede the generic
|
||||
# provider-timeout branch: the job's own tool going quiet involves no provider/fallback chain.
|
||||
# The scheduler's own inactivity watchdog (see the TimeoutError raised above at "Cron job '{job_name}'
|
||||
# idle for {secs}s (limit {limit}s) — last activity: {desc}") produces a message that contains the
|
||||
# substring "timed out"/"timeout" nowhere, but DOES contain "idle for ... (limit ...)" — however
|
||||
# older/other call sites can still phrase an inactivity abort using "timed out" wording, so match on the
|
||||
# "idle for Ns (limit" shape specifically (case-insensitive) BEFORE the generic provider- timeout branch
|
||||
# below. Without this, an inactivity timeout — the job's OWN tool call/turn going quiet, no provider or
|
||||
# fallback chain ever involved — gets rewritten into a misleading "provider timeout / fallback chain
|
||||
# exhausted" message, sending the operator to debug the wrong system entirely (field-reported: a stuck
|
||||
# `terminal` tool call tripped the 600s inactivity limit and was reported as a provider/fallback
|
||||
# failure). Mirrors the same reordering fix upstream issue #59549 applied for script timeouts vs
|
||||
# provider timeouts — check the more specific, deterministic signature first.
|
||||
# Scheduler inactivity watchdog ("idle for {n}s (limit {m}s)"): the job's OWN tool call went
|
||||
# quiet, no model service involved. Its text may still contain "timed out", so it must be
|
||||
# recognised before the classifier (field-reported: a stuck `terminal` call was blamed on the
|
||||
# provider and the operator debugged the wrong system).
|
||||
if re.search(r"idle for \d+s\s*\(limit \d+s\)", lower):
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: the job itself stalled — no tool/API "
|
||||
"activity for the configured inactivity window. Not a provider or "
|
||||
"fallback-chain issue; check what the job was doing when it went "
|
||||
"quiet. Full details saved in cron output."
|
||||
)
|
||||
return inactivity_notice(job_name, job_id)
|
||||
|
||||
if provider_reachable and (
|
||||
"readtimeout" in lower or "timed out" in lower or "timeout" in lower
|
||||
):
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: provider timeout. "
|
||||
f"{_fallback_chain_phrase()} "
|
||||
"Full details saved in cron output."
|
||||
)
|
||||
|
||||
# Whole-token 401/403 and auth wording so "oauth", "4015" etc. don't trip a false auth message.
|
||||
if provider_reachable and (
|
||||
re.search(r"authenticat|authoriz", lower) or re.search(r"\b(401|403)\b", text)
|
||||
):
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: provider authentication error. "
|
||||
"Full details saved in cron output."
|
||||
)
|
||||
# no_agent jobs never reach a model, so provider errors are structurally impossible for them:
|
||||
# gate on job MODE before classifying, or a script's own wording ("429", "timed out") would
|
||||
# blame the wrong subsystem.
|
||||
if not job.get("no_agent"):
|
||||
notice = provider_failure_notice(
|
||||
job_name, job_id, classify_cron_failure_reason(text),
|
||||
backup_provider_phrase=_fallback_chain_phrase())
|
||||
if notice is not None:
|
||||
return notice
|
||||
|
||||
# Strip exception wrappers; bound input first so a multi-KB blob can't slow the regexes.
|
||||
cleaned = re.sub(r"^(RuntimeError|Exception|ValueError|HTTPStatusError):\s*", "", text[:2000])
|
||||
cleaned = re.sub(r"\s+", " ", cleaned).strip()
|
||||
cleaned = re.sub(r"\s+", " ", cleaned).strip().rstrip(".")
|
||||
if len(cleaned) > 180:
|
||||
cleaned = cleaned[:177].rstrip() + "..."
|
||||
message = f"⚠️ Cron '{job_name}' failed: {cleaned}"
|
||||
message = generic_failure_notice(job_name, job_id, cleaned)
|
||||
|
||||
# Import-class failures in a gateway whose checkout changed underneath it (mixed sys.modules)
|
||||
# read like code bugs. When boot SHA ≠ disk HEAD, APPEND cause + fix — never replace the raw
|
||||
# error, which carries the failing symbol. Fail-safe: skew is None on non-git/no-fingerprint
|
||||
# (message unchanged); no_agent jobs excluded via the same mode gate (a fresh subprocess
|
||||
# resolves imports against disk, so its ImportError is the script's own problem).
|
||||
# Import-class failures (#95294 part 3): a long-lived gateway whose checkout was updated underneath it
|
||||
# (interrupted `hermes update`, manual git pull) serves MIXED modules — old entries frozen in
|
||||
# sys.modules, new files loaded by lazy imports — and every agent cron job then dies with `cannot import
|
||||
# name X` / ModuleNotFoundError. The error itself reads like a code bug, so operators debug the wrong
|
||||
# thing (2 days on the reporting incident, 15 missed jobs).
|
||||
if provider_reachable and re.search(
|
||||
# Import-class failures (#95294 part 3): a long-lived gateway whose checkout was updated
|
||||
# underneath it (interrupted `hermes update`, manual git pull) serves MIXED modules and every
|
||||
# agent cron job dies with `cannot import name X`. The error reads like a code bug, so APPEND
|
||||
# cause + fix — never replace the raw error, which carries the failing symbol. Fail-safe: skew
|
||||
# is None on non-git/no-fingerprint; no_agent jobs excluded (a fresh subprocess resolves
|
||||
# imports against disk, so its ImportError is the script's own problem).
|
||||
if not job.get("no_agent") and re.search(
|
||||
r"cannot import name|modulenotfounderror|importerror", lower
|
||||
):
|
||||
try:
|
||||
@@ -1505,12 +1463,12 @@ def _blocked_config_result(job_id: str, job_name: str, _pf_reason: str) -> tuple
|
||||
f"**Job ID:** {job_id}\n"
|
||||
f"**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}\n"
|
||||
f"**Status:** BLOCKED (configuration)\n\n"
|
||||
"Pre-dispatch validation found a configuration problem and "
|
||||
"the agent was NOT run (no tokens spent).\n\n"
|
||||
"The pre-run configuration check found a problem, so the agent did not run "
|
||||
"(nothing was charged).\n\n"
|
||||
f"**Reason:** {_pf_reason}\n\n"
|
||||
"The job will stay blocked (without re-alerting) until the "
|
||||
"configuration is fixed; the next healthy run clears this "
|
||||
"state. Set `cron.preflight: false` in config.yaml to disable this validation."
|
||||
"Hermes tries again at the next scheduled time and clears this state on the first healthy "
|
||||
"run; this alert is not repeated. Check with `hermes cron doctor`. Set `cron.preflight: "
|
||||
"false` in config.yaml to disable this check."
|
||||
)
|
||||
return False, blocked_doc, "", f"{marker} {_pf_reason}"
|
||||
|
||||
@@ -1837,18 +1795,22 @@ def _final_response_from_result(result: dict, job_id: str, job_name: str, AIAgen
|
||||
from hermes_state_errors import PERSISTENCE_ERROR_CAUSES as _causes
|
||||
except Exception:
|
||||
_causes = ("locked", "disk", "unknown")
|
||||
# The finalizer fills the model name into the explainer; render with the same name (and
|
||||
# the bare form) or the comparison below misses and the warning is delivered.
|
||||
_model = str(result.get("model") or "")
|
||||
for _cause in (None, *_causes):
|
||||
try:
|
||||
_variant = AIAgent._format_turn_completion_explanation(turn_exit_reason, _cause)
|
||||
except TypeError:
|
||||
for _kwargs in ({"model": _model}, {}):
|
||||
try:
|
||||
_variant = AIAgent._format_turn_completion_explanation(turn_exit_reason)
|
||||
_variant = AIAgent._format_turn_completion_explanation(turn_exit_reason, _cause, **_kwargs)
|
||||
except TypeError:
|
||||
try:
|
||||
_variant = AIAgent._format_turn_completion_explanation(turn_exit_reason)
|
||||
except Exception:
|
||||
_variant = ""
|
||||
except Exception:
|
||||
_variant = ""
|
||||
except Exception:
|
||||
_variant = ""
|
||||
if _variant:
|
||||
_explainer_variants.append(_variant.strip())
|
||||
if _variant:
|
||||
_explainer_variants.append(_variant.strip())
|
||||
if final_response.strip() in _explainer_variants:
|
||||
logger.info(
|
||||
"Job '%s': abnormal empty turn (%s) — suppressing explainer for cron delivery",
|
||||
@@ -2598,12 +2560,8 @@ def _compose_run_delivery(
|
||||
if blocked_config and not success:
|
||||
# Bypass the generic failure summarizer (its auth/timeout heuristics would mislabel this).
|
||||
_pf_text = re.sub(r"\[blocked_config[^\]]*\]\s*", "", err).strip()
|
||||
deliver_content = (
|
||||
f"⛔ Cron '{job.get('name') or job['id']}' blocked by "
|
||||
f"configuration validation (no LLM call was made): "
|
||||
f"{_pf_text} "
|
||||
"This alert is sent once; the job stays blocked until the configuration is fixed."
|
||||
)
|
||||
from cron.scheduler_failure_copy import blocked_config_notice
|
||||
deliver_content = blocked_config_notice(job.get("name") or job["id"], _pf_text)
|
||||
elif success:
|
||||
deliver_content = final_response
|
||||
_resolve_incidents_for_recovered_job(job)
|
||||
@@ -2861,44 +2819,6 @@ def _deliver_crash_failure(
|
||||
|
||||
|
||||
|
||||
def _install_fire_secret_scope() -> "tuple[contextvars.Token, Optional[contextvars.Token]]":
|
||||
"""Install the firing profile's secret scope for the span ``_run_one_job_body`` runs, delivery
|
||||
included, and return the tokens ``_reset_fire_secret_scope`` needs.
|
||||
|
||||
Hydrate the profile's external secret sources BEFORE freezing the scope — the order
|
||||
gateway/run.py and the external cron worker already use: ``build_profile_secret_scope`` only
|
||||
READS the per-home source map, so a scope frozen first would carry no vault-backed value.
|
||||
|
||||
For a fire routed to a profile other than the process's own (marked by
|
||||
``cron.scheduler_provider._profile_cron_scope``) also run under multiplex semantics — for
|
||||
exactly this span and no wider. The desktop backend ticks every local profile from a process
|
||||
that is not a multiplexer, so nothing else isolates that fire; and switching the context on
|
||||
here rather than at the tick means no read is ever fail-closed without a scope to read — the
|
||||
restart-safe handoff in ``run_one_job`` runs before this and keeps today's semantics (#107692).
|
||||
"""
|
||||
from agent.secret_scope import (
|
||||
build_profile_secret_scope, set_multiplex_context, set_secret_scope)
|
||||
from cron.scheduler_provider import routed_profile_fire
|
||||
from hermes_cli.env_loader import hydrate_profile_secret_sources
|
||||
|
||||
home = Path(_get_hermes_home())
|
||||
hydrate_profile_secret_sources(home)
|
||||
scope_token = set_secret_scope(build_profile_secret_scope(home))
|
||||
context_token = set_multiplex_context(True) if routed_profile_fire() else None
|
||||
return scope_token, context_token
|
||||
|
||||
|
||||
def _reset_fire_secret_scope(tokens: "tuple[contextvars.Token, Optional[contextvars.Token]]") -> None:
|
||||
"""Undo ``_install_fire_secret_scope`` — the context first, so multiplex semantics never
|
||||
outlive the scope they depend on."""
|
||||
from agent.secret_scope import reset_multiplex_context, reset_secret_scope
|
||||
|
||||
scope_token, context_token = tokens
|
||||
if context_token is not None:
|
||||
reset_multiplex_context(context_token)
|
||||
reset_secret_scope(scope_token)
|
||||
|
||||
|
||||
def _run_one_job_body(
|
||||
job: dict, *, adapters=None, loop=None, verbose: bool = False,
|
||||
extra_prompt: Optional[str] = None, fire_claim_lost: Optional[_CancelEventLike] = None,
|
||||
@@ -2915,8 +2835,10 @@ def _run_one_job_body(
|
||||
job["id"], source="direct", scheduled_instant=job.get("_scheduled_instant"))["id"]
|
||||
delivery_attempted = False
|
||||
delivery_error = None
|
||||
from agent.secret_scope import (
|
||||
build_profile_secret_scope, reset_secret_scope, set_secret_scope)
|
||||
|
||||
_fire_scope_tokens = None
|
||||
_scope_token = None
|
||||
_terminal_scope_token = None
|
||||
try:
|
||||
# Commit a finite one-shot's dispatch BEFORE its side effect so a tick dying mid-run cannot
|
||||
@@ -2942,7 +2864,7 @@ def _run_one_job_body(
|
||||
|
||||
# get_secret() fails closed outside a scope; the ticker thread has none. Delivery adapters
|
||||
# resolve credentials, so the scope must span delivery too (reset in the outer finally).
|
||||
_fire_scope_tokens = _install_fire_secret_scope()
|
||||
_scope_token = set_secret_scope(build_profile_secret_scope(_get_hermes_home()))
|
||||
# Same for terminal policy (gateway/run.py _profile_runtime_scope): else the ticker reads
|
||||
# process-global TERMINAL_* env a concurrent profile pinned. Resolution failure installs a
|
||||
# refusal scope — terminal execution raises instead of using the launch process's policy.
|
||||
@@ -3084,8 +3006,8 @@ def _run_one_job_body(
|
||||
finally:
|
||||
# Function-level on purpose: must scope delivery, deferred teardown, claim-loss handling and
|
||||
# bookkeeping — not just run_job. Do not move into the run block's finally.
|
||||
if _fire_scope_tokens is not None:
|
||||
_reset_fire_secret_scope(_fire_scope_tokens)
|
||||
if _scope_token is not None:
|
||||
reset_secret_scope(_scope_token)
|
||||
if _terminal_scope_token is not None:
|
||||
from tools.terminal_scope import reset_terminal_scope
|
||||
|
||||
@@ -3170,12 +3092,6 @@ def _launch_external_cron_worker(job: dict) -> bool:
|
||||
ownership handoff: in a transient user scope, or — when no user D-Bus
|
||||
session exists and ``cron.require_restart_safe_scope`` is false — as a
|
||||
direct subprocess (process separation kept, cgroup isolation lost).
|
||||
|
||||
A fire routed to a profile other than the process's own is multiplexed at THIS boundary too.
|
||||
``run_one_job`` switches the context on in ``_install_fire_secret_scope``, which runs AFTER
|
||||
this handoff, so a routed desktop fire on the managed path serialized ``multiplex_active=False``
|
||||
and built the worker environment with the launch profile's residue and no scrub (review on
|
||||
f5f88d5058). Enable it for exactly this span; the worker then re-establishes it from the payload.
|
||||
"""
|
||||
execution_id = str(job["execution_id"])
|
||||
job_id = str(job["id"])
|
||||
@@ -3192,25 +3108,26 @@ def _launch_external_cron_worker(job: dict) -> bool:
|
||||
str(ack_path),
|
||||
]
|
||||
|
||||
from agent.secret_scope import is_multiplex_active
|
||||
from cron.scheduler_provider import routed_profile_fire
|
||||
from agent.secret_scope import (
|
||||
build_profile_secret_scope,
|
||||
is_multiplex_active,
|
||||
reset_secret_scope,
|
||||
set_secret_scope,
|
||||
)
|
||||
from hermes_cli.env_loader import hydrate_profile_secret_sources
|
||||
from tools.environments.local import build_subprocess_env, strip_launch_profile_env
|
||||
from tools.process_registry import (
|
||||
restart_safe_gateway_child_argv,
|
||||
systemd_user_bus_env,
|
||||
)
|
||||
|
||||
# A fire routed to another profile is multiplexed at this handoff even when the process flag is
|
||||
# off (the desktop ticker is not a multiplexer): the payload says so, and the worker env is built
|
||||
# under that context so the scrub and the launch-residue strip both apply. The context is set
|
||||
# for exactly the env build; nothing else here reads it.
|
||||
multiplex_active = is_multiplex_active() or routed_profile_fire()
|
||||
try:
|
||||
require_restart_safe_scope = bool(
|
||||
(load_config_readonly().get("cron") or {}).get("require_restart_safe_scope", False)
|
||||
)
|
||||
except Exception:
|
||||
require_restart_safe_scope = False
|
||||
multiplex_active = is_multiplex_active()
|
||||
dispatch = restart_safe_gateway_child_argv(
|
||||
command,
|
||||
unit_suffix=f"cron-{job_id}-exec-{execution_id}",
|
||||
@@ -3247,19 +3164,16 @@ def _launch_external_cron_worker(job: dict) -> bool:
|
||||
raise
|
||||
|
||||
profile_home = _get_hermes_home().resolve()
|
||||
# Same hydrate -> scope -> (routed) multiplex-context install the in-process fire uses, for exactly
|
||||
# the env build; the helper's reset order keeps the context from outliving its scope.
|
||||
fire_scope_tokens = _install_fire_secret_scope()
|
||||
hydrate_profile_secret_sources(profile_home)
|
||||
secret_token = set_secret_scope(build_profile_secret_scope(profile_home))
|
||||
try:
|
||||
# No restore_managed_env here: the worker re-runs load_hermes_dotenv -> _apply_managed_env at
|
||||
# import, and strip_launch_profile_env leaves managed keys in place.
|
||||
worker_env = strip_launch_profile_env(build_subprocess_env(
|
||||
scrub_secrets=multiplex_active,
|
||||
inherit_profile_home=True,
|
||||
extra={"HERMES_HOME": str(profile_home)},
|
||||
))
|
||||
finally:
|
||||
_reset_fire_secret_scope(fire_scope_tokens)
|
||||
reset_secret_scope(secret_token)
|
||||
worker_env = systemd_user_bus_env(worker_env)
|
||||
# Unattended worker: the gateway sets HERMES_EXEC_ASK at startup (interactive launches set
|
||||
# the other two), and an inherited presence var makes every env-fallback consumer in the
|
||||
|
||||
@@ -746,7 +746,8 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str, *, deferred: Opt
|
||||
except Exception:
|
||||
found = False
|
||||
if not found:
|
||||
return "bot-chat delivery failed: hermes CLI not resolvable"
|
||||
return ("Hermes could not deliver this result to Bot Chat: the `hermes` command was not found. "
|
||||
"The result is saved; run `hermes cron runs` to see it, or `hermes doctor` if this keeps happening")
|
||||
argv = [sys.executable, "-m", "hermes_cli.main"]
|
||||
|
||||
def _fail(msg: str, **log_kwargs) -> str:
|
||||
@@ -781,9 +782,13 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str, *, deferred: Opt
|
||||
creationflags=windows_hide_flags())
|
||||
if result.returncode != 0:
|
||||
tail = (result.stderr or result.stdout or "").strip()[-500:]
|
||||
return _fail(
|
||||
f"bot-chat delivery to profile '{profile_label}' failed (exit {result.returncode}) at {home}"
|
||||
+ (f": {tail}" if tail else ""))
|
||||
logger.warning(
|
||||
"Job '%s': bot-chat delivery to profile '%s' failed (exit %s) at %s%s",
|
||||
job_id, profile_label, result.returncode, home, f": {tail}" if tail else "")
|
||||
return (
|
||||
f"Hermes could not deliver this result to Bot Chat (profile '{profile_label}'). "
|
||||
"The result is saved; run `hermes cron runs` to see it, or `hermes doctor` if this keeps happening"
|
||||
+ (f". Details: {tail[-200:]}" if tail else ""))
|
||||
logger.info("Job '%s': delivered to Bot Chat of profile '%s'", job_id, profile_label)
|
||||
return None
|
||||
except subprocess.TimeoutExpired:
|
||||
@@ -793,7 +798,12 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str, *, deferred: Opt
|
||||
"still complete; raise cron.bot_chat_delivery_timeout_seconds if "
|
||||
"this recurs)")
|
||||
except Exception as e:
|
||||
return _fail(f"bot-chat delivery failed: {str(e) or type(e).__name__}", exc_info=True)
|
||||
logger.warning(
|
||||
"Job '%s': bot-chat delivery to profile '%s' failed: %s", job_id, profile_label,
|
||||
str(e) or type(e).__name__, exc_info=True)
|
||||
return (
|
||||
f"Hermes could not deliver this result to Bot Chat (profile '{profile_label}'). "
|
||||
"The result is saved; run `hermes cron runs` to see it, or `hermes doctor` if this keeps happening")
|
||||
finally:
|
||||
if query_file:
|
||||
with contextlib.suppress(OSError):
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
"""Plain-language copy for the one-line cron failure notice delivered to a job's chat.
|
||||
|
||||
The scheduler classifies the failure text through ``agent.error_classifier.classify_api_error``
|
||||
(one classifier for the whole app, no cron-local regex ladder) and looks the verdict up here.
|
||||
Every notice says WHAT happened and WHAT TO DO, and names the exact ``hermes cron`` command plus
|
||||
the real output directory — "cron output" alone sent operators hunting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
|
||||
def cron_output_dir_display(job_id: str) -> str:
|
||||
"""User-facing path of a job's saved run output (profile-aware)."""
|
||||
return f"{display_hermes_home()}/cron/output/{job_id}/"
|
||||
|
||||
|
||||
_HTTP_STATUS_IN_TEXT = re.compile(r"(?:\bHTTP\b|\bError code\b|\bstatus(?: code)?\b)\W{0,3}(\b[45]\d\d\b)", re.I)
|
||||
_LEADING_EXC_TYPE = re.compile(r"^(?:[\w.]+\.)?([A-Z]\w*(?:Error|Timeout|Exception))\s*:")
|
||||
|
||||
|
||||
def classify_cron_failure_reason(text: str) -> str:
|
||||
"""``FailoverReason`` value for a cron failure string (``"unknown"`` when unclassifiable).
|
||||
|
||||
The scheduler only has ``str(exc)``, so the status code and exception type the classifier
|
||||
keys on are rebuilt from the text: a whole-token HTTP code after ``HTTP`` / ``Error code`` /
|
||||
``status`` (a bare ``429`` inside a job id or hash never counts, #83188) and a leading
|
||||
``ReadTimeout:``-style type prefix."""
|
||||
from agent.error_classifier import classify_api_error
|
||||
|
||||
status = _HTTP_STATUS_IN_TEXT.search(text)
|
||||
type_name = _LEADING_EXC_TYPE.match(text)
|
||||
exc_cls = type(type_name.group(1), (Exception,), {}) if type_name else Exception
|
||||
exc = exc_cls(text)
|
||||
if status:
|
||||
exc.status_code = int(status.group(1))
|
||||
return classify_api_error(exc).reason.value
|
||||
|
||||
|
||||
# What happened, per reason: the one gloss table shared with subagent notices lives in
|
||||
# agent/turn_failure_copy.py so the two never drift; the job is the subject here.
|
||||
def _provider_failure_cause(reason: str) -> Optional[str]:
|
||||
from agent.turn_failure_copy import failure_cause_gloss
|
||||
|
||||
return failure_cause_gloss(reason, subject="this job", possessive="the job's")
|
||||
|
||||
|
||||
_TRANSIENT_REASONS = frozenset({"timeout", "rate_limit", "upstream_rate_limit", "overloaded", "server_error"})
|
||||
|
||||
# Reason -> what to do. Transient reasons get the backup-provider clause from the scheduler
|
||||
# (it knows whether a fallback chain is configured) instead of a fixed sentence.
|
||||
_PROVIDER_FAILURE_ACTION: dict[str, str] = {
|
||||
"billing": (
|
||||
"Top up or wait for the limit to reset, or pin another provider with "
|
||||
"`hermes cron edit {job_id} --provider <name>`."
|
||||
),
|
||||
"auth": (
|
||||
"Sign in again with /login (or `hermes auth add <provider>` in a terminal), or pin a "
|
||||
"working provider with `hermes cron edit {job_id} --provider <name>`, then "
|
||||
"`hermes cron run {job_id}` to retry."
|
||||
),
|
||||
"model_not_found": "Pick another model with `hermes cron edit {job_id} --model <name>`.",
|
||||
"context_overflow": "Shorten the job's prompt with `hermes cron edit {job_id} --prompt <text>`.",
|
||||
}
|
||||
_PROVIDER_FAILURE_ACTION["auth_permanent"] = _PROVIDER_FAILURE_ACTION["auth"]
|
||||
_PROVIDER_FAILURE_ACTION["billing_unverified"] = _PROVIDER_FAILURE_ACTION["billing"]
|
||||
_PROVIDER_FAILURE_ACTION["payload_too_large"] = _PROVIDER_FAILURE_ACTION["context_overflow"]
|
||||
_PROVIDER_FAILURE_ACTION["content_policy_blocked"] = (
|
||||
"Reword the job's prompt with `hermes cron edit {job_id} --prompt <text>`, or pick another "
|
||||
"model with `hermes cron edit {job_id} --model <name>`."
|
||||
)
|
||||
_DEFAULT_FAILURE_ACTION = "Run it again with `hermes cron run {job_id}`, or edit it with `hermes cron edit {job_id}`."
|
||||
|
||||
|
||||
def provider_failure_notice(
|
||||
job_name: str, job_id: str, reason: str, *, backup_provider_phrase: str,
|
||||
) -> Optional[str]:
|
||||
"""The notice for a provider-shaped ``reason``, or None when the reason is not one."""
|
||||
cause = _provider_failure_cause(reason)
|
||||
if cause is None:
|
||||
return None
|
||||
if reason in _TRANSIENT_REASONS:
|
||||
action = (
|
||||
f"{backup_provider_phrase} It will run again at its next scheduled time; "
|
||||
f"`hermes cron run {job_id}` tries now."
|
||||
)
|
||||
else:
|
||||
action = _PROVIDER_FAILURE_ACTION.get(reason, _DEFAULT_FAILURE_ACTION).format(job_id=job_id)
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: {cause}. {action} "
|
||||
f"Run log: `hermes cron runs {job_id}`."
|
||||
)
|
||||
|
||||
|
||||
def generic_failure_notice(job_name: str, job_id: str, cleaned_error: str) -> str:
|
||||
"""Unclassified failure: the cleaned error text plus where to look and what to do."""
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: {cleaned_error}. "
|
||||
f"See the full run with `hermes cron runs {job_id}` (output saved under "
|
||||
f"{cron_output_dir_display(job_id)}); run it again with `hermes cron run {job_id}`, "
|
||||
f"edit it with `hermes cron edit {job_id}`, or pause it with `hermes cron pause {job_id}`."
|
||||
)
|
||||
|
||||
|
||||
def script_timeout_notice(job_name: str, job_id: str) -> str:
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: its script timed out. No model was invoked. "
|
||||
f"Check the script's output under {cron_output_dir_display(job_id)} or `hermes cron runs {job_id}`, "
|
||||
f"then run it again with `hermes cron run {job_id}`."
|
||||
)
|
||||
|
||||
|
||||
def inactivity_notice(job_name: str, job_id: str) -> str:
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' failed: the job stalled — it stopped doing anything for too long "
|
||||
f"and was cut off. Check what it was doing in the saved output under "
|
||||
f"{cron_output_dir_display(job_id)} (`hermes cron runs {job_id}`), then run it again with "
|
||||
f"`hermes cron run {job_id}`."
|
||||
)
|
||||
|
||||
|
||||
def blocked_config_notice(job_name: str, reason: str) -> str:
|
||||
"""One-time notice when the pre-run configuration check refused to start the job."""
|
||||
reason = reason.rstrip()
|
||||
if reason and reason[-1] not in ".!?":
|
||||
reason += "."
|
||||
return (
|
||||
f"⛔ Cron '{job_name}' did not run: {reason} Nothing was charged. Hermes will try again at "
|
||||
"the next scheduled time once this is fixed and will not repeat this alert; check with "
|
||||
"`hermes cron doctor`."
|
||||
)
|
||||
@@ -6,7 +6,6 @@ execution + delivery stay in cron.scheduler.run_job / _deliver_result; never rei
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
from contextvars import ContextVar
|
||||
import inspect
|
||||
import logging
|
||||
import threading
|
||||
@@ -92,38 +91,12 @@ def _existing_profile_homes(profile_homes: list) -> list:
|
||||
return [entry for entry in profile_homes if Path(_profile_entry(entry)[1]).is_dir()]
|
||||
|
||||
|
||||
# Set by _profile_cron_scope: this task fires a profile OTHER than the process's own. A marker only.
|
||||
# Multiplex semantics are switched on where the profile's secret scope is installed
|
||||
# (cron.scheduler._install_fire_secret_scope) and off with it — never at the tick — so no read can
|
||||
# be fail-closed without a scope to read: run_one_job's restart-safe handoff runs before that
|
||||
# scope and keeps today's semantics (its own scope is #107413 / #106050's seam).
|
||||
_ROUTED_PROFILE_FIRE: ContextVar[bool] = ContextVar("_ROUTED_PROFILE_FIRE", default=False)
|
||||
|
||||
|
||||
def routed_profile_fire() -> bool:
|
||||
"""True inside a tick for a profile other than the process's own (marker, see above)."""
|
||||
return _ROUTED_PROFILE_FIRE.get()
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def _profile_cron_scope(home):
|
||||
"""Scope the calling thread to one profile's home + cron store for the block.
|
||||
|
||||
A profile OTHER than the process's own is MARKED as a routed fire (``routed_profile_fire``).
|
||||
The desktop backend ticks every local profile from one process "like a multiplex gateway"
|
||||
without setting the process-global multiplex flag, so every isolation keyed on
|
||||
``is_multiplex_active()`` was inert for those fires: a sibling profile's ``.env`` landed in
|
||||
the shared ``os.environ`` with ``override=True`` and a scope miss read the launch profile's
|
||||
credentials (#107692). ``cron.scheduler._install_fire_secret_scope`` turns the marker into
|
||||
multiplex semantics for exactly the span the profile's secret scope covers. The process's own
|
||||
profile keeps single-profile semantics. The override and the marker both reach the pool
|
||||
worker via ``copy_context()``. Under a real multiplexer the process flag is already on."""
|
||||
"""Scope the calling thread to one profile's home + cron store for the block."""
|
||||
from cron.jobs import use_cron_store
|
||||
from hermes_constants import (
|
||||
get_process_hermes_home, reset_hermes_home_override, set_hermes_home_override)
|
||||
from hermes_constants import set_hermes_home_override, reset_hermes_home_override
|
||||
|
||||
routed = Path(home).resolve() != get_process_hermes_home().resolve()
|
||||
routed_token = _ROUTED_PROFILE_FIRE.set(routed)
|
||||
# Record per-profile heartbeat after each tick cycle. Distinguish a COMPLETED cycle (``_tick_error``
|
||||
# unset) — where each profile's beat reflects its own outcome, so a yielding profile does not darken
|
||||
# healthy siblings — from an aborted one (exception), where no profile completed and all beats are
|
||||
@@ -134,7 +107,6 @@ def _profile_cron_scope(home):
|
||||
yield
|
||||
finally:
|
||||
reset_hermes_home_override(home_token)
|
||||
_ROUTED_PROFILE_FIRE.reset(routed_token)
|
||||
|
||||
|
||||
class CronScheduler(ABC):
|
||||
|
||||
@@ -349,24 +349,7 @@ def _run_job_script(
|
||||
# reader threads on non-UTF-8 Windows (#45099).
|
||||
"encoding": "utf-8",
|
||||
"errors": "replace"}
|
||||
# A routed profile's script (desktop multi-profile ticker, multiplex gateway) must see ITS
|
||||
# profile's .env + vault values — the process env holds the launch profile's. Drop the
|
||||
# launch profile's dotenv-owned residue first (a name only the launch .env defines must
|
||||
# come through UNSET, not with the launch value — the scrub only knows classified secrets),
|
||||
# then overlay the installed scope, then sanitize, so routed values pass the same scrub /
|
||||
# passthrough rules as any other. No-op outside multiplex or for the launch profile's own
|
||||
# fires; the parent process is never mutated.
|
||||
from agent.secret_scope import current_secret_scope, is_multiplex_active
|
||||
from tools.environments.local import restore_managed_env, strip_launch_profile_env
|
||||
base = strip_launch_profile_env(dict(os.environ))
|
||||
if is_multiplex_active():
|
||||
# Single-profile: the scope IS os.environ, so overlaying it would only re-sanitize
|
||||
# values the child already inherits byte-identical.
|
||||
base.update(current_secret_scope() or {})
|
||||
# Administrator-managed values keep their precedence over the routed profile's own .env,
|
||||
# exactly as they do in the launch process (``_apply_managed_env`` applies them last).
|
||||
restore_managed_env(base)
|
||||
env = build_subprocess_env(base=base)
|
||||
env = build_subprocess_env()
|
||||
env.update(env_overlay)
|
||||
# Subprocess cwd only (default: scripts-dir parent). NEVER os.chdir() the process.
|
||||
# Use the job's workdir as the subprocess cwd when configured, otherwise default to the scripts-dir
|
||||
|
||||
@@ -387,19 +387,37 @@ def _fmt_block_loop_detected(ev, n) -> tuple:
|
||||
return msg, None, None
|
||||
|
||||
|
||||
def _fmt_gave_up(ev, n) -> tuple:
|
||||
# The dispatcher auto-blocked the task after ``failures`` consecutive non-success attempts
|
||||
# (spawn failure, crash, or timeout alike): it is now Blocked and waiting for a human.
|
||||
failures = _payload(ev, "failures")
|
||||
count = f"it failed {int(failures)} times in a row" if failures else "it kept failing"
|
||||
last = _clip(ev, "error", " (last: {})", 160)
|
||||
return (
|
||||
f"⛔ {n.head} is now blocked: {count}{last}. Fix the cause, then `hermes kanban unblock "
|
||||
f"{n.task_id}` (or `hermes kanban reassign {n.task_id}`). Logs: `hermes kanban log {n.task_id}`.",
|
||||
None, None,
|
||||
)
|
||||
|
||||
|
||||
def _fmt_timed_out(ev, n) -> tuple:
|
||||
limit = int(_payload(ev, "limit_seconds") or 0)
|
||||
minutes = max(1, round(limit / 60)) if limit else 0
|
||||
span = f"its {minutes}-minute limit" if minutes else "its time limit"
|
||||
return f"⏱ {n.head} ran past {span} and was stopped; it will be retried automatically.", None, None
|
||||
|
||||
|
||||
# archived / unblocked are claimed (so the cursor advances past them) but
|
||||
# intentionally silent (no formatter), and excluded from _WAKE_KINDS so they
|
||||
# never wake the creator.
|
||||
_EVENT_FORMATTERS: dict[str, Callable[[Any, "_KanbanNotification"], tuple]] = {
|
||||
"completed": _fmt_completed,
|
||||
"blocked": lambda ev, n: (f"⏸ {n.head} blocked{_clip(ev, 'reason', ': {}', 160)}", None, None),
|
||||
"gave_up": lambda ev, n: (
|
||||
f"✖ {n.head} gave up after repeated spawn failures{_clip(ev, 'error', _NL, 200)}", None, None,
|
||||
),
|
||||
"crashed": lambda ev, n: (f"✖ {n.head} worker crashed (pid gone); dispatcher will retry", None, None),
|
||||
"timed_out": lambda ev, n: (
|
||||
f"⏱ {n.head} timed out (max_runtime={int(_payload(ev, 'limit_seconds') or 0)}s); will retry", None, None,
|
||||
"gave_up": _fmt_gave_up,
|
||||
"crashed": lambda ev, n: (
|
||||
f"✖ {n.head} — its worker stopped unexpectedly; it will be retried automatically.", None, None,
|
||||
),
|
||||
"timed_out": _fmt_timed_out,
|
||||
"status": lambda ev, n: (f"🔄 {n.head} → {_payload(ev, 'status') or ''}", None, None),
|
||||
"review_requested": _fmt_review_requested,
|
||||
"changes_requested": _fmt_changes_requested,
|
||||
|
||||
@@ -384,6 +384,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
|
||||
|
||||
from gateway.config import Platform, PlatformConfig
|
||||
from gateway.platforms.helpers import fence_state_after
|
||||
from gateway.platforms.base_exec_approval import (
|
||||
EA_HEADER_TEXT, EA_REASON_LABEL_TEXT, approval_timeout_seconds, format_approval_deadline_line)
|
||||
from gateway.platforms.event import MessageEvent, MessageType, ProcessingOutcome
|
||||
from gateway.session import SessionSource, build_session_key
|
||||
from gateway.session_transcript import TranscriptReadError
|
||||
@@ -437,6 +439,20 @@ GATEWAY_SECRET_CAPTURE_UNSUPPORTED_MESSAGE = (
|
||||
"Secure secret entry is not supported over messaging. "
|
||||
"Load this skill in the local CLI to be prompted, or add the key to ~/.hermes/.env manually.")
|
||||
|
||||
# One sentence for every "you may not press/run this" refusal on every platform (slash commands,
|
||||
# approval buttons, pickers, prompts). ``{platform}`` is the ``Platform.value`` for the
|
||||
# ``hermes pairing approve`` command (hermes_cli/subcommands/pairing.py) that lets the owner fix it.
|
||||
# Kept under 200 chars: Telegram's answerCallbackQuery truncates longer text.
|
||||
UNAUTHORIZED_ACTION_NOTICE = (
|
||||
"This bot is private and you're not on its allowed list. If you own it, run "
|
||||
"`hermes pairing approve {platform} <request-id>` on the host (`hermes pairing list` shows the id).")
|
||||
|
||||
|
||||
def unauthorized_action_notice(platform: Any) -> str:
|
||||
"""``UNAUTHORIZED_ACTION_NOTICE`` for a ``Platform`` member or its string name."""
|
||||
name = getattr(platform, "value", platform)
|
||||
return UNAUTHORIZED_ACTION_NOTICE.format(platform=str(name or "<platform>"))
|
||||
|
||||
|
||||
def safe_url_for_log(url: str, max_len: int = 80) -> str:
|
||||
"""Return a URL string safe for logs (no query/fragment/userinfo)."""
|
||||
@@ -2514,11 +2530,14 @@ class BasePlatformAdapter(ABC):
|
||||
# No running loop (unit tests): close the coroutine to avoid a never-awaited warning.
|
||||
coro.close()
|
||||
|
||||
# ── ``_format_exec_approval`` templates; adapters override to keep historical wording.
|
||||
_EA_HEADER: str = "⚠️ Command Approval Required\n\n"
|
||||
# ── ``_format_exec_approval`` templates; adapters override only the MARKUP (bold, HTML,
|
||||
# fences) — the words come from ``gateway.platforms.base_exec_approval`` so every surface
|
||||
# says the same thing.
|
||||
_EA_HEADER: str = f"⚠️ {EA_HEADER_TEXT}\n\n"
|
||||
_EA_CODE_OPEN: str = "```\n"
|
||||
_EA_CODE_CLOSE: str = "\n```\n"
|
||||
_EA_REASON_LABEL: str = "Reason: "
|
||||
_EA_REASON_LABEL: str = f"{EA_REASON_LABEL_TEXT}: "
|
||||
_EA_DEADLINE_PREFIX: str = "\n\n" # separates the deadline line from the reason line
|
||||
_EA_SMART_DENY_LINE: str = "\n\nSmart DENY: owner override applies to this one operation only."
|
||||
_EA_CMD_BUDGET: int = 3000
|
||||
_EA_REASON_BUDGET: int = 0 # 0 = the reason is never truncated
|
||||
@@ -2537,17 +2556,23 @@ class BasePlatformAdapter(ABC):
|
||||
"""Chars of command preview that fit; platforms with a hard message cap compute it."""
|
||||
return self._EA_CMD_BUDGET
|
||||
|
||||
def _ea_deadline_line(self) -> str:
|
||||
"""The "doing nothing means it will NOT run" line, with the configured approvals.timeout."""
|
||||
return self._EA_DEADLINE_PREFIX + self._ea_escape(format_approval_deadline_line(approval_timeout_seconds()))
|
||||
|
||||
def _format_exec_approval(
|
||||
self, command: str, description: str = "dangerous command", smart_denied: bool = False) -> str:
|
||||
"""Shared exec-approval prompt text: header + fenced (truncated) command + reason,
|
||||
plus the smart-deny line. Buttons/trailing instructions stay platform-local."""
|
||||
"""Shared exec-approval prompt text: header + fenced (truncated) command + why it was
|
||||
flagged + the deadline line, plus the smart-deny line. Buttons/trailing instructions stay
|
||||
platform-local."""
|
||||
if self._EA_REASON_BUDGET:
|
||||
description = self._truncate_preview(str(description or ""), self._EA_REASON_BUDGET)
|
||||
cmd_preview = self._truncate_preview(
|
||||
str(command or ""), self._exec_approval_cmd_budget(description, smart_denied))
|
||||
text = (f"{self._EA_HEADER}"
|
||||
f"{self._EA_CODE_OPEN}{self._ea_escape(cmd_preview)}{self._EA_CODE_CLOSE}"
|
||||
f"{self._EA_REASON_LABEL}{self._ea_escape(description)}")
|
||||
f"{self._EA_REASON_LABEL}{self._ea_escape(description)}"
|
||||
f"{self._ea_deadline_line()}")
|
||||
return text + self._EA_SMART_DENY_LINE if smart_denied else text
|
||||
|
||||
# ── Exec-approval prompt (template method). The choice set is one rule for every button
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Shared wording for the exec-approval prompt every messaging surface renders.
|
||||
|
||||
The button card (``BasePlatformAdapter._format_exec_approval``) and the plain-text
|
||||
``/approve`` fallback (``gateway.run._format_exec_approval_fallback``) must tell the user
|
||||
the same three things: what Hermes wants to run, why it was flagged, and that silence means
|
||||
the command does NOT run once ``approvals.timeout`` elapses. Keeping the text here means one
|
||||
edit changes every platform; adapters only wrap these strings in their own markup.
|
||||
|
||||
No imports from ``gateway.platforms.base`` or ``gateway.run`` — both import this module.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
# Bare strings; adapters add their own bold/HTML around them.
|
||||
EA_HEADER_TEXT = "Hermes wants to run a command that needs your OK"
|
||||
EA_REASON_LABEL_TEXT = "Why it was flagged"
|
||||
|
||||
# Timeout notice posted when nobody answered the prompt (``{window}`` = "5 minutes").
|
||||
APPROVAL_TIMED_OUT_NOTICE = (
|
||||
"⌛ Approval timed out after {window} — the command was NOT run. "
|
||||
"Ask me to try again if you still want it, or raise approvals.timeout in config.yaml.")
|
||||
|
||||
|
||||
def approval_timeout_seconds() -> int:
|
||||
"""The configured ``approvals.timeout`` (default 300s); module attribute so tests can pin it."""
|
||||
from tools.approval_context import _get_approval_timeout
|
||||
return _get_approval_timeout()
|
||||
|
||||
|
||||
def format_approval_window(seconds: int) -> str:
|
||||
"""Human wording for a timeout (300 → "5 minutes"); one formatter shared with the CLI notice and
|
||||
the tool result's ``user_summary`` — see ``tools.approval_context.format_approval_window``."""
|
||||
from tools.approval_context import format_approval_window as _shared
|
||||
return _shared(seconds)
|
||||
|
||||
|
||||
def format_approval_deadline_line(timeout_s: int) -> str:
|
||||
"""The last line of every approval prompt: doing nothing is a safe no."""
|
||||
return f"If you don't answer within {format_approval_window(timeout_s)} it will NOT run."
|
||||
|
||||
|
||||
def format_approval_timed_out_notice(timeout_s: int) -> str:
|
||||
return APPROVAL_TIMED_OUT_NOTICE.format(window=format_approval_window(timeout_s))
|
||||
@@ -40,6 +40,7 @@ except ImportError:
|
||||
|
||||
from gateway.config import Platform, PlatformConfig
|
||||
from gateway.platforms.base import BasePlatformAdapter, ExecApprovalPrompt, SendResult, transcode_to_ogg_opus
|
||||
from gateway.platforms.base_exec_approval import EA_HEADER_TEXT
|
||||
from gateway.platforms.helpers import bounded_put
|
||||
from gateway.platforms.event import MessageEvent, MessageType
|
||||
from gateway.platforms.whatsapp_common import WhatsAppBehaviorMixin, _get_wsecret
|
||||
@@ -467,7 +468,7 @@ class WhatsAppCloudAdapter(WhatsAppBehaviorMixin, BasePlatformAdapter):
|
||||
interactive = {"type": "list", "body": {"text": body_text}, "action": {"button": "Choose", "sections": [{"title": "Options", "rows": rows}]}}
|
||||
return await self._send_interactive(chat_id, interactive, metadata, self._clarify_state, clarify_id, session_key)
|
||||
|
||||
_EA_HEADER = "⚠️ *Command Approval Required*\n\n"
|
||||
_EA_HEADER = f"⚠️ *{EA_HEADER_TEXT}*\n\n"
|
||||
_EA_CODE_CLOSE = "\n```\n\n"
|
||||
_EA_CMD_BUDGET = 800 # body caps at 1024; leave room for the framing prose
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ from gateway.config import Platform, PlatformConfig
|
||||
from gateway.platforms.base import (
|
||||
BasePlatformAdapter, ExecApprovalPrompt, SendResult,
|
||||
)
|
||||
from gateway.platforms.base_exec_approval import EA_HEADER_TEXT
|
||||
from gateway.platforms.event import MessageEvent, MessageType, ProcessingOutcome
|
||||
from gateway.relay.descriptor import CapabilityDescriptor
|
||||
from gateway.relay.egress import (
|
||||
@@ -1991,7 +1992,7 @@ class RelayAdapter(BasePlatformAdapter):
|
||||
|
||||
_PROMPT_UNAVAILABLE = SendResult(success=False, error="relay prompt op unavailable")
|
||||
|
||||
_EA_HEADER = "⚠️ **Command Approval Required**\n\n"
|
||||
_EA_HEADER = f"⚠️ **{EA_HEADER_TEXT}**\n\n"
|
||||
_EA_SMART_DENY_LINE = "\n\n**Smart DENY:** owner override applies to this one operation only."
|
||||
_EA_CMD_BUDGET = 1500
|
||||
|
||||
|
||||
+75
-42
@@ -189,18 +189,15 @@ def hygiene_compaction_recovered(
|
||||
|
||||
def _hygiene_compression_timeout_message(
|
||||
*, total_exhausted: bool, elapsed: float, idle_timeout: float, progress_observed: bool) -> str:
|
||||
"""Describe the host timeout that actually ended hygiene compression."""
|
||||
"""Describe the host timeout that actually ended hygiene compression. Chat users cannot edit
|
||||
model config, so the copy names /compress, /new and `hermes doctor`, never a config key or the
|
||||
raw second counts (those stay in the gateway log)."""
|
||||
lead = (
|
||||
"⚠️ Shortening the conversation history took too long, so I skipped it and kept "
|
||||
"everything as-is. Run /compress to try again or /new to start fresh.")
|
||||
if total_exhausted:
|
||||
progress = " after summary output was observed" if progress_observed else ""
|
||||
return (
|
||||
"⚠️ Context compression reached its total ceiling after "
|
||||
f"{elapsed:.1f}s{progress}. No messages were dropped — continuing "
|
||||
"without compression. Run /compress to retry or /reset for a clean session.")
|
||||
return (
|
||||
f"⚠️ Context compression timed out after {idle_timeout:.1f}s with no "
|
||||
"output from the summary model. No messages were dropped — continuing "
|
||||
"without compression. Run /compress to retry, /reset for a clean "
|
||||
"session, or check your auxiliary.compression model configuration.")
|
||||
return lead
|
||||
return lead + " If this keeps happening, run `hermes doctor` on the host."
|
||||
|
||||
|
||||
def _cached_agent_for_hygiene(gateway, session_key: str):
|
||||
@@ -562,30 +559,44 @@ def _redact_approval_command(cmd: "str | None") -> str:
|
||||
def _format_exec_approval_fallback(
|
||||
command: str, description: str, command_prefix: str, *, allow_permanent: bool = True,
|
||||
allow_session: bool = True, smart_denied: bool = False) -> str:
|
||||
"""Render the text fallback from approval capabilities, not platform names."""
|
||||
"""Render the text fallback from approval capabilities, not platform names. Same words as
|
||||
the button card (``BasePlatformAdapter._format_exec_approval``), plus the typed ``/approve``
|
||||
steps a surface without buttons needs."""
|
||||
from gateway.platforms.base_exec_approval import (
|
||||
EA_HEADER_TEXT, EA_REASON_LABEL_TEXT, approval_timeout_seconds, format_approval_deadline_line)
|
||||
cmd_preview = command[:200] + "..." if len(command) > 200 else command
|
||||
heading = ("⚠️ **Smart DENY — owner override for one operation:**" if smart_denied
|
||||
else "⚠️ **Dangerous command requires approval:**")
|
||||
else f"⚠️ **{EA_HEADER_TEXT}**")
|
||||
|
||||
choices = [f"Reply `{command_prefix}approve` to execute this one operation"]
|
||||
choices = [f"Reply `{command_prefix}approve` to run it once"]
|
||||
if not smart_denied and allow_session:
|
||||
choices.append(f"`{command_prefix}approve session` to approve this pattern for the session")
|
||||
choices.append(f"`{command_prefix}approve session` to allow this pattern for the rest of this session")
|
||||
if allow_permanent:
|
||||
choices.append(f"`{command_prefix}approve always` to approve permanently")
|
||||
choices.append(f"`{command_prefix}approve always` to allow it permanently")
|
||||
choices.append(f"`{command_prefix}deny` to cancel")
|
||||
return (
|
||||
f"{heading}\n```\n{cmd_preview}\n```\nReason: {description}\n\n"
|
||||
+ ", ".join(choices[:-1]) + f", or {choices[-1]}.")
|
||||
f"{heading}\n```\n{cmd_preview}\n```\n{EA_REASON_LABEL_TEXT}: {description}\n\n"
|
||||
+ ", ".join(choices[:-1]) + f", or {choices[-1]}.\n"
|
||||
+ format_approval_deadline_line(approval_timeout_seconds()))
|
||||
|
||||
# Ordered: auth beats policy beats rate-limit beats connection; first match wins.
|
||||
# Ordered: auth beats policy beats rate-limit beats connection; first match wins. Copy names the
|
||||
# slash command the chat user can run; raw provider text stays in the gateway log (`hermes logs`).
|
||||
_PROVIDER_ERROR_REPLIES = (
|
||||
(_GATEWAY_AUTH_ERROR_RE, "⚠️ Provider authentication failed. Check the configured credentials; "
|
||||
"raw provider details are in the gateway logs."),
|
||||
(_GATEWAY_PROVIDER_POLICY_RE, "⚠️ The model provider rejected the request. I kept the raw provider "
|
||||
"error out of chat; check gateway logs for details or try rephrasing."),
|
||||
(_GATEWAY_RATE_LIMIT_RE, "⏱️ The model provider is rate-limiting requests. Please wait a moment and try again."),
|
||||
(_GATEWAY_CONNECTION_ERROR_RE, "⚠️ The model server is not responding — it looks like the configured "
|
||||
"model endpoint is not running or is unreachable."))
|
||||
(_GATEWAY_AUTH_ERROR_RE, "⚠️ Sign-in to the AI model service failed. Use /login to sign in again, "
|
||||
"or ask whoever runs this bot to run `hermes doctor` on the host."),
|
||||
(_GATEWAY_PROVIDER_POLICY_RE, "⚠️ The AI model service rejected this request. Try rephrasing your "
|
||||
"message, or use /model to switch models."),
|
||||
(_GATEWAY_RATE_LIMIT_RE, "⏱️ The AI model service is rate-limiting requests. Wait a moment, then use /retry."),
|
||||
(_GATEWAY_CONNECTION_ERROR_RE, "⚠️ The AI model service isn't reachable right now — the configured model "
|
||||
"endpoint is not running or is unreachable. Wait a moment and use /retry; "
|
||||
"if it persists, run `hermes doctor` on the host."))
|
||||
|
||||
|
||||
# Shared by the failed-turn normalizer and ``run_turn._hmwa_agent_error_reply``; canonical
|
||||
# commands (/compress, /new) — the /compact and /reset aliases are absent from /help.
|
||||
_CONTEXT_OVERFLOW_REPLY = (
|
||||
"⚠️ This conversation has grown too long for me to read all at once. "
|
||||
"Use /compress to shorten the history, or /new to start a fresh conversation.")
|
||||
|
||||
|
||||
def _gateway_provider_error_reply(text: str) -> str:
|
||||
@@ -594,8 +605,8 @@ def _gateway_provider_error_reply(text: str) -> str:
|
||||
if pattern.search(text):
|
||||
return reply
|
||||
return (
|
||||
"⚠️ The model provider failed after retries. I kept raw provider details "
|
||||
"out of chat; check gateway logs for diagnostics.")
|
||||
"⚠️ The AI model service kept failing. Use /retry to try again, or /model to switch "
|
||||
"models. Details are in the gateway log (`hermes logs`).")
|
||||
|
||||
|
||||
# Provider/API failure envelope preambles (not ordinary assistant prose), anchored at line start.
|
||||
@@ -2908,20 +2919,23 @@ def _format_concise_process_notification(
|
||||
"""One-line completion message for ``concise`` display mode; failure appends a short output tail."""
|
||||
ok = exit_code in {0, None}
|
||||
icon = "✅" if ok else "❌"
|
||||
verb = "finished" if ok else f"failed (exit {exit_code})"
|
||||
parts = [f"{icon} Background task {verb}"]
|
||||
parts = [f"{icon} Background task {'finished' if ok else 'failed'}"]
|
||||
short_cmd = _shorten_command_for_display(command)
|
||||
if short_cmd:
|
||||
parts.append(f"— `{short_cmd}`")
|
||||
details = []
|
||||
if isinstance(duration_seconds, (int, float)) and duration_seconds >= 0:
|
||||
secs = int(duration_seconds)
|
||||
if secs >= 3600:
|
||||
dur = f"{secs // 3600}h {(secs % 3600) // 60}m"
|
||||
details.append(f"{secs // 3600}h {(secs % 3600) // 60}m")
|
||||
elif secs >= 60:
|
||||
dur = f"{secs // 60}m {secs % 60}s"
|
||||
details.append(f"{secs // 60}m {secs % 60}s")
|
||||
else:
|
||||
dur = f"{secs}s"
|
||||
parts.append(f"({dur})")
|
||||
details.append(f"{secs}s")
|
||||
if not ok:
|
||||
details.append(f"exit {exit_code}")
|
||||
if details:
|
||||
parts.append(f"({', '.join(details)})")
|
||||
text = " ".join(parts)
|
||||
if not ok and output:
|
||||
tail_lines = [ln for ln in output.strip().splitlines() if ln.strip()][-5:]
|
||||
@@ -2929,7 +2943,9 @@ def _format_concise_process_notification(
|
||||
if len(tail) > 500:
|
||||
tail = tail[-500:]
|
||||
if tail:
|
||||
text += f"\n```\n{tail}\n```"
|
||||
text += f". Last output:\n```\n{tail}\n```"
|
||||
if not ok:
|
||||
text += "\nAsk me to rerun it or show the full log."
|
||||
return text
|
||||
|
||||
|
||||
@@ -3027,12 +3043,13 @@ def _normalize_empty_agent_response(
|
||||
"turn was stopped to protect your conversation history. "
|
||||
"Your message should already be saved — please send it again in a moment.")
|
||||
if is_overflow:
|
||||
return (
|
||||
"⚠️ Session too large for the model's context window.\n"
|
||||
"Use /compact to compress the conversation, or /reset to start fresh.")
|
||||
return _CONTEXT_OVERFLOW_REPLY
|
||||
# Raw exception text (class names, JSON bodies, URLs) stays in the gateway log.
|
||||
logger.warning("Agent turn failed; reply sanitized for chat. Detail: %s", str(error_detail)[:500])
|
||||
return (
|
||||
f"The request failed: {str(error_detail)[:300]}\n"
|
||||
"Try again or use /reset to start a fresh session.")
|
||||
"⚠️ Something went wrong and I couldn't finish this reply. Use /retry to try again, "
|
||||
"or /new to start a fresh conversation. Technical details are in the gateway log "
|
||||
"(`hermes logs`).")
|
||||
|
||||
api_calls = int(agent_result.get("api_calls", 0) or 0)
|
||||
if agent_result.get("interrupted"):
|
||||
@@ -3061,8 +3078,24 @@ def _normalize_empty_agent_response(
|
||||
if _is_gateway_hidden_reasoning_incomplete_turn(agent_result):
|
||||
return ""
|
||||
if agent_result.get("partial"):
|
||||
err = agent_result.get("error", "processing incomplete")
|
||||
return f"⚠️ Processing stopped: {str(err)[:200]}. Try again."
|
||||
# ``error`` mirrors the loop's own final text (curated, e.g. "Response truncated due to
|
||||
# output length limit") and is kept; a raw provider envelope goes to the log instead.
|
||||
err = str(agent_result.get("error") or "processing incomplete")
|
||||
# A loop site code (truncated, context_overflow, ...) already wrote the full
|
||||
# what-happened / what-to-do sentence: deliver it verbatim. Wrapping it would cut it
|
||||
# mid-sentence at 200 chars and append a second, conflicting set of instructions.
|
||||
from agent.turn_failure_copy import SITE_FAILURE_CODES
|
||||
if (str(agent_result.get("failure_reason") or "") in SITE_FAILURE_CODES
|
||||
and err.strip() and not _looks_like_gateway_provider_error(err)):
|
||||
return err if err.startswith("⚠️") else f"⚠️ {err}"
|
||||
if _looks_like_gateway_provider_error(err):
|
||||
logger.warning("Agent turn ended partially; reply sanitized for chat. Detail: %s", err[:500])
|
||||
reason = ""
|
||||
else:
|
||||
reason = f": {err[:200]}"
|
||||
return (
|
||||
f"⚠️ I had to stop before finishing{reason}. Use /retry to try again, or /compress "
|
||||
"if this conversation has grown very long.")
|
||||
return (
|
||||
"⚠️ Processing completed but no response was generated. "
|
||||
"This may be a transient error — try sending your message again.")
|
||||
|
||||
+33
-21
@@ -21,6 +21,10 @@ from gateway.config import Platform
|
||||
from gateway.platforms.base import EphemeralReply
|
||||
from gateway.platforms.event import MessageEvent, MessageType
|
||||
from gateway.run_common import _UNSET
|
||||
from gateway.run_inbound_unauthorized import (
|
||||
PAIRING_RATE_LIMITED_REPLY, UnauthorizedOwnerNotifier, pairing_code_reply, pairing_profile_arg,
|
||||
unauthorized_owner_hint,
|
||||
)
|
||||
from gateway.session import (
|
||||
SessionSource, is_shared_multi_user_session, neutralize_untrusted_inline_text
|
||||
)
|
||||
@@ -88,21 +92,9 @@ class GatewayInboundMixin:
|
||||
code = pairing_store.generate_code(platform_name, source.user_id, source.user_name or "")
|
||||
adapter = self._adapter_for_source(source)
|
||||
if code:
|
||||
store_profile = getattr(pairing_store, "profile", None)
|
||||
profile_arg = (
|
||||
f"-p {store_profile} "
|
||||
if isinstance(store_profile, str) and store_profile and store_profile != "default"
|
||||
else ""
|
||||
)
|
||||
reply = (
|
||||
f"Hi~ I don't recognize you yet!\n\n"
|
||||
f"Here's your pairing code: `{code}`\n\n"
|
||||
f"Ask the bot owner to run:\n"
|
||||
f"`hermes {profile_arg}pairing approve "
|
||||
f"{platform_name} {code}`"
|
||||
)
|
||||
reply = pairing_code_reply(platform_name, code, pairing_profile_arg(pairing_store))
|
||||
else:
|
||||
reply = "Too many pairing requests right now~ Please try again later!"
|
||||
reply = PAIRING_RATE_LIMITED_REPLY
|
||||
if adapter:
|
||||
await adapter.send(source.chat_id, reply)
|
||||
if not code:
|
||||
@@ -129,6 +121,21 @@ class GatewayInboundMixin:
|
||||
except Exception:
|
||||
logger.warning("Failed to deliver unauthorized-DM decline on %s", platform_name, exc_info=True)
|
||||
|
||||
async def _hm_report_ignored_dm(self, source: SessionSource) -> None:
|
||||
"""Unauthorized DM under behaviour ``ignore``: nothing goes to the sender. The owner gets the
|
||||
sender's ID and the allowlist fix in the WARNING log and, once per sender, in the home channel."""
|
||||
from hermes_constants import display_hermes_home
|
||||
platform_name = source.platform.value if source.platform else "unknown"
|
||||
hint = unauthorized_owner_hint(
|
||||
platform_name, source.user_id, source.user_name or "", hermes_home=display_hermes_home(),
|
||||
)
|
||||
logger.warning("Unauthorized user (ignored): %s", hint)
|
||||
notifier = getattr(self, "_unauthorized_owner_notifier", None)
|
||||
if notifier is None:
|
||||
notifier = self._unauthorized_owner_notifier = UnauthorizedOwnerNotifier()
|
||||
if notifier.first_time(platform_name, source.user_id) and getattr(self, "config", None) is not None:
|
||||
await notifier.notify(self, source, hint)
|
||||
|
||||
async def _hm_admit_event(
|
||||
self, event: "MessageEvent"
|
||||
) -> Optional[Tuple["MessageEvent", SessionSource, bool]]:
|
||||
@@ -214,15 +221,20 @@ class GatewayInboundMixin:
|
||||
# posts, sender_chat): can't be paired but may be authorized via a chat allowlist.
|
||||
logger.debug("Ignoring message with no user_id from %s", source.platform.value)
|
||||
return None
|
||||
logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value)
|
||||
# DMs get a pairing code or a one-time decline, groups are ignored. A bot cannot pair, and
|
||||
# answering one mid-cooldown is outbound traffic.
|
||||
if source.chat_type == "dm" and not getattr(source, "is_bot", False):
|
||||
behavior = self._get_unauthorized_dm_behavior(source.platform, profile=source.profile)
|
||||
if behavior == "pair":
|
||||
await self._hm_offer_pairing_code(source)
|
||||
elif behavior == "decline":
|
||||
await self._hm_send_unauthorized_decline(source)
|
||||
pairable_dm = source.chat_type == "dm" and not getattr(source, "is_bot", False)
|
||||
behavior = self._get_unauthorized_dm_behavior(source.platform, profile=source.profile) if pairable_dm else None
|
||||
if behavior == "pair":
|
||||
logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value)
|
||||
await self._hm_offer_pairing_code(source)
|
||||
elif behavior == "decline":
|
||||
logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value)
|
||||
await self._hm_send_unauthorized_decline(source)
|
||||
elif pairable_dm:
|
||||
await self._hm_report_ignored_dm(source)
|
||||
else:
|
||||
logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value)
|
||||
return None
|
||||
# The busy path charged this event on arrival; a drained follow-up must not pay twice.
|
||||
if not getattr(event, "_bot_loop_admitted", False) and not self._admit_bot_message_for_source(source):
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Copy and owner-side signalling for unauthorized inbound senders.
|
||||
|
||||
Two audiences, two rules:
|
||||
|
||||
* The **stranger** gets either a pairing code (behaviour ``pair``) or nothing at all (behaviour
|
||||
``ignore``: an allowlist is configured, so any reply would leak that the bot exists).
|
||||
* The **owner** is the one who can fix a mis-typed allowlist, so an ignored DM is logged at
|
||||
WARNING with the sender's ID and the allowlist / pairing-mode fix, and surfaced once per
|
||||
(platform, user) per gateway process in the platform's home channel when one is configured.
|
||||
No pairing request is minted for an ignored sender: strangers must not create server state
|
||||
when the owner has restricted access.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections import OrderedDict
|
||||
|
||||
from gateway.pairing import CODE_TTL_SECONDS, _allowlist_env_for_platform
|
||||
|
||||
# Display names come from the stranger. Bound them and keep the mention/markdown surface small in
|
||||
# the owner's channel: a name is never a reason to render a link, mention or new section.
|
||||
_MAX_SENDER_NAME_CHARS = 64
|
||||
# Distinct (platform, user_id) senders remembered per process; a bot rotating IDs evicts the oldest
|
||||
# instead of growing the set without limit.
|
||||
_MAX_SEEN_SENDERS = 2048
|
||||
|
||||
logger = logging.getLogger("gateway.run")
|
||||
|
||||
|
||||
def pairing_profile_arg(pairing_store) -> str:
|
||||
"""``-p <profile> `` when the store belongs to a non-default profile, else ``""``."""
|
||||
store_profile = getattr(pairing_store, "profile", None)
|
||||
if isinstance(store_profile, str) and store_profile and store_profile != "default":
|
||||
return f"-p {store_profile} "
|
||||
return ""
|
||||
|
||||
|
||||
def pairing_code_reply(platform_name: str, code: str, profile_arg: str = "") -> str:
|
||||
"""The DM a first-time sender receives: what happened, how long the code lives, what to do
|
||||
whether they are the owner or a guest, and that they must message again after approval."""
|
||||
hours = max(1, CODE_TTL_SECONDS // 3600)
|
||||
validity = f"{hours} hour" if hours == 1 else f"{hours} hours"
|
||||
approve_cmd = f"hermes {profile_arg}pairing approve {platform_name} {code}"
|
||||
return (
|
||||
"Hi! I don't recognize you yet, so I can't reply until the person running this bot "
|
||||
"approves you.\n\n"
|
||||
f"Your pairing code: `{code}` (valid for {validity})\n\n"
|
||||
f"If you run this bot, open a terminal and run: `{approve_cmd}`. "
|
||||
"Otherwise send that command to the bot owner. After approval, send your message again."
|
||||
)
|
||||
|
||||
|
||||
PAIRING_RATE_LIMITED_REPLY = (
|
||||
"Too many pairing requests right now. Wait a few minutes, then send your message again.")
|
||||
|
||||
|
||||
def unauthorized_owner_hint(
|
||||
platform_name: str, user_id: str, user_name: str = "", *, hermes_home: str,
|
||||
) -> str:
|
||||
"""One-line hint for the owner (log + home channel): who was dropped and how to let them in.
|
||||
No pairing request is minted for an ignored sender (a configured allowlist means the owner chose
|
||||
to restrict access), so the ways in are the allowlist itself or switching the platform to
|
||||
pairing mode."""
|
||||
from gateway.session import neutralize_untrusted_inline_text
|
||||
|
||||
safe_name = neutralize_untrusted_inline_text(user_name or "", max_chars=_MAX_SENDER_NAME_CHARS)
|
||||
# Mention and link sigils are stripped so a hostile display name cannot ping the channel or
|
||||
# smuggle a link; the ID next to it is the value the owner actually acts on.
|
||||
safe_name = "".join(ch for ch in safe_name if ch not in "@<>[]()`*_~#").strip()
|
||||
who = f"{safe_name} ({user_id})" if safe_name else str(user_id)
|
||||
env_var = _allowlist_env_for_platform(platform_name)
|
||||
allowlist = (
|
||||
f"add the ID to {env_var} in {hermes_home}/.env and restart the gateway"
|
||||
if env_var else "add the ID to this platform's allowed-users list and restart the gateway"
|
||||
)
|
||||
return (
|
||||
f"Dropped a message from unrecognized {platform_name} user {who}. If that is you or someone "
|
||||
f"you trust, {allowlist}; or set `unauthorized_dm_behavior: pair` for {platform_name} in "
|
||||
f"{hermes_home}/config.yaml so unknown senders receive a pairing code you can approve with "
|
||||
f"`hermes pairing approve {platform_name} <code>`."
|
||||
)
|
||||
|
||||
|
||||
class UnauthorizedOwnerNotifier:
|
||||
"""Tells the owner's home channel about the first drop of each unrecognized DM sender.
|
||||
|
||||
One notice per (platform, user_id) per gateway process: the first drop is the useful signal (an
|
||||
owner who typo'd their own ID); repeats would only let a stranger spam the home channel.
|
||||
"""
|
||||
|
||||
def __init__(self, max_seen: int = _MAX_SEEN_SENDERS) -> None:
|
||||
self._seen: OrderedDict[tuple[str, str], None] = OrderedDict()
|
||||
self._max_seen = max(1, int(max_seen))
|
||||
|
||||
def first_time(self, platform_name: str, user_id: str) -> bool:
|
||||
key = (platform_name, str(user_id))
|
||||
if key in self._seen:
|
||||
self._seen.move_to_end(key)
|
||||
return False
|
||||
self._seen[key] = None
|
||||
while len(self._seen) > self._max_seen:
|
||||
self._seen.popitem(last=False)
|
||||
return True
|
||||
|
||||
async def notify(self, runner, source, hint: str) -> None:
|
||||
"""Best-effort post to the source platform's home channel; silent when none is configured."""
|
||||
for platform, _cfg, home, transport in runner._home_channel_transports():
|
||||
if platform != source.platform:
|
||||
continue
|
||||
if str(home.chat_id) == str(source.chat_id):
|
||||
# The stranger's DM *is* the home channel (misconfiguration); posting there would
|
||||
# answer the unauthorized user, which the ignore behaviour exists to prevent.
|
||||
return
|
||||
await runner._send_home_channel_message(
|
||||
platform, home, transport, f"⚠️ {hint}", "unauthorized-sender notice failed for %s:%s: %s",
|
||||
)
|
||||
return
|
||||
@@ -26,6 +26,18 @@ from gateway.run_shutdown import _log_suppressed, _notice_target_key, _send_erro
|
||||
# Log-record parity with the origin module.
|
||||
logger = logging.getLogger("gateway.run")
|
||||
|
||||
# A failed /update leaves the previous version running; the full pip/git log stays on the host
|
||||
# (`hermes update` re-runs it in the terminal) and only a short tail is quoted in chat.
|
||||
_UPDATE_FAILED_NOTICE = (
|
||||
"❌ Hermes update failed; the previous version is still running. Run `hermes update` on the "
|
||||
"host to see the full error, or try /update again later.")
|
||||
|
||||
|
||||
def _update_output_tail(output: str, limit: int) -> str:
|
||||
"""Last ``limit`` chars of an update log, prefixed with an ellipsis when cut."""
|
||||
return output if len(output) <= limit else "…" + output[-limit:]
|
||||
|
||||
|
||||
_VIDEO_EXTS = {'.mp4', '.mov', '.avi', '.mkv', '.webm', '.3gp'}
|
||||
# Routing fields copied verbatim from a process watcher onto its synthetic completion event.
|
||||
_WATCHER_ROUTE_FIELDS = ("session_key", "platform", "chat_type", "chat_id", "thread_id", "user_id", "user_name")
|
||||
@@ -584,8 +596,7 @@ class GatewayNotificationsMixin:
|
||||
with _log_suppressed(logging.WARNING, "Update final notification failed: %s"):
|
||||
exit_code = self._update_exit_code(paths)
|
||||
await target.send(
|
||||
"✅ Hermes update finished." if exit_code == 0
|
||||
else "❌ Hermes update failed (exit code {}).".format(exit_code)
|
||||
"✅ Hermes update finished." if exit_code == 0 else _UPDATE_FAILED_NOTICE
|
||||
)
|
||||
logger.info("Update finished (exit=%s), notified %s", exit_code, session_key)
|
||||
self._clear_update_markers(paths, session_key)
|
||||
@@ -661,16 +672,14 @@ class GatewayNotificationsMixin:
|
||||
metadata = self._pending_marker_metadata(platform, chat_id, pending, adapter)
|
||||
from tools.ansi_strip import strip_ansi
|
||||
output = strip_ansi(output).strip()
|
||||
if output:
|
||||
if len(output) > 3500:
|
||||
output = "…" + output[-3500:]
|
||||
status = "✅ Hermes update finished." if exit_code == 0 else "❌ Hermes update failed."
|
||||
msg = f"{status}\n\n```\n{output}\n```"
|
||||
if exit_code == 0:
|
||||
msg = "✅ Hermes update finished successfully."
|
||||
if output:
|
||||
msg = f"{msg}\n\n```\n{_update_output_tail(output, 3500)}\n```"
|
||||
else:
|
||||
msg = (
|
||||
"✅ Hermes update finished successfully." if exit_code == 0 else
|
||||
"❌ Hermes update failed. Check the gateway logs or run `hermes update` manually for details."
|
||||
)
|
||||
msg = _UPDATE_FAILED_NOTICE
|
||||
if output:
|
||||
msg = f"{msg}\n\nLast lines:\n```\n{_update_output_tail(output, 800)}\n```"
|
||||
await adapter.send(chat_id, msg, metadata=_non_conversational_metadata(metadata, platform=platform))
|
||||
logger.info("Sent post-update notification to %s:%s (exit=%s)", platform_str, chat_id, exit_code)
|
||||
except Exception as e:
|
||||
@@ -844,7 +853,7 @@ class GatewayNotificationsMixin:
|
||||
logger.info("state.db recovered before the home-channel warning went out; not broadcasting")
|
||||
return
|
||||
from hermes_constants import get_default_hermes_root, profile_cli_selector
|
||||
from hermes_state import _default_db_path, classify_persistence_error, format_session_db_unavailable
|
||||
from hermes_state import _default_db_path, classify_persistence_error
|
||||
cause = classify_persistence_error(error)
|
||||
# Copy-pasteable, so name the real store and pin the profile: a bare `hermes` follows
|
||||
# active_profile, which may be a different database (#105887).
|
||||
@@ -877,9 +886,12 @@ class GatewayNotificationsMixin:
|
||||
"recovery tools or restore a backup unless `hermes doctor` confirms damage."
|
||||
)
|
||||
else:
|
||||
from hermes_state_user_copy import describe_storage_failure
|
||||
failure = describe_storage_failure(error)
|
||||
message = (
|
||||
f"⚠️ Session database unavailable — messages may not be persisted. "
|
||||
f"{format_session_db_unavailable()}\nRun `hermes doctor` for diagnostics."
|
||||
"⚠️ Session database unavailable — messages may not be saved and /resume will be "
|
||||
f"empty. Cause: {failure.gloss}. Run `hermes {profile_arg}doctor --fix` on the "
|
||||
"gateway machine, then `hermes gateway restart`."
|
||||
)
|
||||
logger.warning("Broadcasting state.db failure warning to home channels: %s", error)
|
||||
for platform, _platform_cfg, home, transport in self._home_channel_transports():
|
||||
|
||||
+10
-6
@@ -847,10 +847,11 @@ class GatewayShutdownMixin:
|
||||
except Exception as e:
|
||||
logger.debug("Cron interrupt targets unresolved for %s: %s", job_id, e)
|
||||
continue
|
||||
job_name = job.get("name") or job_id
|
||||
msg = (
|
||||
f"⚠️ Cron job '{job.get('name') or job_id}' was interrupted — "
|
||||
f"the gateway is {action} and killed the run before it "
|
||||
"finished. No result was produced for this run."
|
||||
f"⚠️ Scheduled job '{job_name}' was cut short because Hermes is {action}; "
|
||||
"no result this run. It will run again on schedule, or run it now with "
|
||||
f"`hermes cron run {job_name}` once Hermes is back."
|
||||
)
|
||||
for target in targets or ():
|
||||
try:
|
||||
@@ -931,11 +932,14 @@ class GatewayShutdownMixin:
|
||||
Called at the start of stop() while adapters are connected; send failures never block shutdown.
|
||||
"""
|
||||
restart_source = self._restart_command_source if self._restart_requested else None
|
||||
msg = "⚠️ Gateway shutting down — Your current task will be interrupted."
|
||||
msg = (
|
||||
"⚠️ Hermes is shutting down — your current task will be interrupted. "
|
||||
"When it is back online, send any message and I'll try to pick up where we left off."
|
||||
)
|
||||
if self._restart_requested:
|
||||
msg = (
|
||||
"⚠️ Gateway restarting — Your current task will be interrupted. "
|
||||
"Send any message after restart and I'll try to resume where you left off."
|
||||
"⚠️ Hermes is restarting — your current task will be interrupted. "
|
||||
"Send any message after the restart and I'll try to resume where you left off."
|
||||
)
|
||||
restart_key = None
|
||||
if restart_source is not None:
|
||||
|
||||
+39
-25
@@ -58,6 +58,12 @@ _UNEXPECTED_SILENCE_REPLY = (
|
||||
)
|
||||
|
||||
|
||||
def _bg_prompt_preview(prompt: str, limit: int = 60) -> str:
|
||||
"""Short single-line quote of a /bg prompt for its failure notice (the task id means nothing to the user)."""
|
||||
text = " ".join(str(prompt or "").split())
|
||||
return text if len(text) <= limit else text[: limit - 1].rstrip() + "…"
|
||||
|
||||
|
||||
def is_context_overflow_failure_result(agent_result: dict, history_len: int) -> bool:
|
||||
"""One verdict for "this failed turn is a context overflow", shared by transcript persistence
|
||||
(#1630 skip) and the user-facing reply so the two can never disagree.
|
||||
@@ -1085,11 +1091,12 @@ class GatewayTurnMixin:
|
||||
# Force-redact: provider exception text may contain credentials; this reaches users.
|
||||
from agent.redact import redact_sensitive_text
|
||||
_err = redact_sensitive_text(getattr(_comp, "_last_summary_error", None) or "unknown error", force=True)
|
||||
logger.warning("Session hygiene compression aborted: %s", _err)
|
||||
await self._hmwa_hygiene_notify(
|
||||
source, attempt.meta, "⚠️ Context compression aborted "
|
||||
f"({_err}). No messages were dropped — "
|
||||
"conversation is unchanged. Run /compress to retry, /reset for a clean "
|
||||
"session, or check your auxiliary.compression model configuration.",
|
||||
source, attempt.meta,
|
||||
"⚠️ Shortening the conversation history failed, so I kept everything as-is. "
|
||||
"Run /compress to try again or /new to start fresh. If this keeps happening, "
|
||||
"run `hermes doctor` on the host.",
|
||||
"compression-failure warning",
|
||||
)
|
||||
# Configured aux model failed, recovered on the main model: only the user can fix that config.
|
||||
@@ -1395,12 +1402,14 @@ class GatewayTurnMixin:
|
||||
_intentional_silence = False
|
||||
response = _UNEXPECTED_SILENCE_REPLY
|
||||
|
||||
# "(empty)" = the model produced no visible content after exhausting all retries.
|
||||
# "(empty)" = the model produced no visible content after exhausting all retries. One
|
||||
# text with the CLI explainer and the desktop (agent/turn_explainers.py) so the user
|
||||
# reads the same words on every surface.
|
||||
if response == "(empty)" and not _intentional_silence:
|
||||
response = (
|
||||
"⚠️ The model returned no response after processing tool results. This can happen "
|
||||
"with some models — try again or rephrase your question."
|
||||
)
|
||||
from agent.turn_explainers import EMPTY_RESPONSE_EXPLANATION
|
||||
|
||||
_model = str(agent_result.get("model") or "").strip() or "The model"
|
||||
response = "⚠️ " + EMPTY_RESPONSE_EXPLANATION.format(model=_model)
|
||||
agent_messages = agent_result.get("messages", [])
|
||||
logger.info(
|
||||
"response ready: platform=%s chat=%s time=%.1fs api_calls=%d response=%d chars",
|
||||
@@ -1814,10 +1823,13 @@ class GatewayTurnMixin:
|
||||
|
||||
return response
|
||||
|
||||
# Chat-side next steps keyed by HTTP status; Hermes commands only (/login is the gateway's own
|
||||
# sign-in, `hermes login` / `hermes auth` the host equivalents).
|
||||
_STATUS_HINTS = {
|
||||
401: " Check your API key or run `claude /login` to refresh OAuth credentials.",
|
||||
402: " Your API balance or quota is exhausted. Check your provider dashboard.",
|
||||
529: " The API is temporarily overloaded. Please try again shortly.",
|
||||
401: (" Your sign-in to the AI model service has expired or the API key is wrong. "
|
||||
"Use /login here, or run `hermes login` / `hermes auth` on the host."),
|
||||
402: " Your AI model service balance or quota is used up. Top it up on the service's website, or use /model to switch models.",
|
||||
529: " The AI model service is temporarily overloaded. Wait a moment, then use /retry.",
|
||||
}
|
||||
|
||||
async def _hmwa_agent_error_reply(self, e, event, source, session_entry, session_key, prepared):
|
||||
@@ -1830,10 +1842,8 @@ class GatewayTurnMixin:
|
||||
if status_code in {400, 500} and len(prepared.history) > 50:
|
||||
# Context overflow / payload too large: a deterministic rejection (#107567), and the same
|
||||
# no-grow rule as the persist path (#1630) — nothing is written into an oversized session.
|
||||
return (
|
||||
"⚠️ Session too large for the model's context window.\nUse /compact to "
|
||||
"compress the conversation, or /reset to start fresh."
|
||||
)
|
||||
from gateway.run import _CONTEXT_OVERFLOW_REPLY
|
||||
return _CONTEXT_OVERFLOW_REPLY
|
||||
# Replay can coalesce inputs; only this input's durable marker establishes ownership.
|
||||
try:
|
||||
if prepared.message_text is not None and session_entry is not None:
|
||||
@@ -1866,10 +1876,11 @@ class GatewayTurnMixin:
|
||||
else:
|
||||
status_hint = " Your plan's usage limit has been reached. Please wait until it resets."
|
||||
elif status_code == 400:
|
||||
status_hint = " The request was rejected by the API."
|
||||
status_hint = " The AI model service rejected the request."
|
||||
return self._hmwa_add_failed_turn_notice(
|
||||
f"Sorry, I encountered an unexpected error.{status_hint}\n"
|
||||
"Try again or use /reset to start a fresh session.",
|
||||
f"⚠️ Something went wrong and I couldn't finish this reply.{status_hint}\n"
|
||||
"Use /retry to try again, or /new to start a fresh conversation. "
|
||||
"Technical details are in the gateway log (`hermes logs`).",
|
||||
self._PARTIAL_FAILED_TURN_NOTICE,
|
||||
)
|
||||
|
||||
@@ -2214,7 +2225,8 @@ class GatewayTurnMixin:
|
||||
if not runtime_kwargs.get("api_key"):
|
||||
await adapter.send(
|
||||
source.chat_id,
|
||||
f"❌ Background task {task_id} failed: no provider credentials configured.",
|
||||
"❌ The background task couldn't start because no AI model sign-in is "
|
||||
"configured. Use /login, or run `hermes setup` on the host.",
|
||||
metadata=_thread_metadata,
|
||||
)
|
||||
return
|
||||
@@ -2325,7 +2337,9 @@ class GatewayTurnMixin:
|
||||
logger.exception("Background task %s failed", task_id)
|
||||
with suppress(Exception):
|
||||
await adapter.send(
|
||||
chat_id=source.chat_id, content=f"❌ Background task {task_id} failed: {e}",
|
||||
chat_id=source.chat_id,
|
||||
content=(f"❌ Your background task \"{_bg_prompt_preview(prompt)}\" failed before finishing. "
|
||||
"Send /bg again to retry, or /agents to see what is still running."),
|
||||
metadata=_thread_metadata,
|
||||
)
|
||||
|
||||
@@ -3286,10 +3300,10 @@ class GatewayTurnMixin:
|
||||
return
|
||||
try:
|
||||
await _warn_adapter.send(
|
||||
source.chat_id, f"⚠️ No activity for {int(worker.agent_warning // 60) or 1} min. "
|
||||
"If the agent does not respond soon, it will be timed out in "
|
||||
f"{int((worker.agent_timeout - worker.agent_warning) // 60) or 1} min. "
|
||||
"You can continue waiting or use /reset.",
|
||||
source.chat_id, f"⚠️ I seem to be stuck (no activity for {int(worker.agent_warning // 60) or 1} min). "
|
||||
"If nothing happens in the next "
|
||||
f"{int((worker.agent_timeout - worker.agent_warning) // 60) or 1} min I'll give up on this task. "
|
||||
"You can keep waiting, send /stop to cancel it, or /new to start a fresh conversation.",
|
||||
metadata=_interim_metadata(_status_thread_metadata),
|
||||
)
|
||||
except Exception as _warn_err:
|
||||
|
||||
@@ -176,7 +176,7 @@ class TurnRunner:
|
||||
if status in SUBAGENT_FAILURE_STATUSES and ctx._run_still_current():
|
||||
line = format_subagent_failure_line(
|
||||
kwargs.get("goal"), status, error=kwargs.get("summary") or preview,
|
||||
duration_seconds=kwargs.get("duration_seconds"),
|
||||
duration_seconds=kwargs.get("duration_seconds"), failure_reason=kwargs.get("failure_reason"),
|
||||
)
|
||||
self._schedule(self._runner._deliver_platform_notice(ctx.source, line), "subagent failure notice scheduling error")
|
||||
except Exception:
|
||||
@@ -1351,6 +1351,7 @@ class TurnRunner:
|
||||
"""Send the approval request from the agent thread: the adapter's interactive button
|
||||
approvals (``send_exec_approval``) when available, else plain text with ``/approve`` steps."""
|
||||
from gateway.run import _approval_send_outcome, _format_exec_approval_fallback, _interim_metadata, _redact_approval_command
|
||||
from gateway.run_turn_runner_approval_settle import register_timeout_notice
|
||||
ctx = self._ctx
|
||||
adapter = ctx._status_adapter
|
||||
# Slack's assistant_threads_setStatus disables the compose box, so the user can't type
|
||||
@@ -1377,6 +1378,11 @@ class TurnRunner:
|
||||
raise RuntimeError("send_exec_approval: loop unavailable")
|
||||
outcome = _approval_send_outcome(fut, timeout=15)
|
||||
if outcome == "sent":
|
||||
# Without this, a card whose timer runs out keeps live buttons and nobody
|
||||
# learns the command did NOT run (only the TUI registered a settle hook).
|
||||
register_timeout_notice(
|
||||
self, approval_data, command=cmd,
|
||||
card_message_id=getattr(fut.result(timeout=0), "message_id", None))
|
||||
return
|
||||
if outcome == "ambiguous":
|
||||
# Timeout ≠ failure: the card may have posted with a late ack. The prompt
|
||||
@@ -1432,6 +1438,9 @@ class TurnRunner:
|
||||
)
|
||||
if fut is not None:
|
||||
fut.result(timeout=15)
|
||||
# No card to edit on the text path: the prompt has no buttons to drop and carries
|
||||
# the /approve instructions, so the timeout notice is posted as a new message.
|
||||
register_timeout_notice(self, approval_data, command=cmd, card_message_id=None)
|
||||
except Exception as e:
|
||||
logger.error("Failed to send approval request: %s", e)
|
||||
|
||||
@@ -1785,7 +1794,16 @@ class TurnRunner:
|
||||
model, runtime_kwargs.get("provider"), ctx.session_key or "",
|
||||
)
|
||||
except Exception as exc:
|
||||
return {"final_response": f"⚠️ Provider authentication failed: {exc}", "messages": [], "api_calls": 0, "tools": []}
|
||||
# Model/credential resolution failed before the turn began; the raw text (URLs, status
|
||||
# codes) belongs in the log, and the chat gets the commands that fix it.
|
||||
logger.warning("Model resolution failed for session %s: %s", ctx.session_key or "", exc)
|
||||
return {
|
||||
"final_response": (
|
||||
"⚠️ I couldn't connect to the AI model service, so this message wasn't processed. "
|
||||
"Use /login to sign in again, or /model to pick a different model. If it keeps "
|
||||
"failing, run `hermes doctor` on the host."),
|
||||
"messages": [], "api_calls": 0, "tools": [],
|
||||
}
|
||||
pr = runner._provider_routing
|
||||
reasoning_config = runner._resolve_session_reasoning_config(source=ctx.source, session_key=ctx.session_key, model=model)
|
||||
runner._reasoning_config = reasoning_config
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Tell the chat when an exec-approval prompt times out (messaging platforms).
|
||||
|
||||
``tools.approval_gateway_wait._await_gateway_decision`` calls ``entry.settle(reason)`` once the
|
||||
wait ends. Only the TUI registered such a hook, so on Telegram / Slack / WhatsApp a card whose
|
||||
timer ran out kept live buttons and the user never learned the command did NOT run. The turn
|
||||
runner registers the hook here right after the prompt was delivered.
|
||||
|
||||
Best-effort by design: a failed notice is logged at debug — the approval already resolved as
|
||||
"no", and nothing here may block the agent thread.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from gateway.platforms.base_exec_approval import approval_timeout_seconds, format_approval_timed_out_notice
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def register_timeout_notice(
|
||||
runner, approval_data: dict, *, command: str, card_message_id: Optional[str]) -> None:
|
||||
"""Arm a settle hook that posts the timed-out notice for ``approval_data['request_id']``.
|
||||
|
||||
``runner`` is the ``TurnRunner`` (for ``_ctx`` and ``_schedule``); ``card_message_id`` is the
|
||||
delivered BUTTON card's id when the adapter returned one, so the card itself is edited in place
|
||||
(which also drops its buttons). The plain-text prompt passes ``None``: it has no buttons to
|
||||
drop and rewriting it would erase the record of what was asked. ``command`` is the
|
||||
already-redacted command shown to the user. The notice is skipped when the run is no longer
|
||||
current (``ctx._run_still_current``).
|
||||
"""
|
||||
from tools.approval import register_gateway_settle
|
||||
|
||||
request_id = approval_data.get("request_id")
|
||||
session_key = runner._ctx.session_key or ""
|
||||
if not request_id or not session_key:
|
||||
return
|
||||
timeout_s = approval_timeout_seconds()
|
||||
|
||||
def settle(reason: str) -> None:
|
||||
if reason != "timeout":
|
||||
return # answered / interrupted / notify_failed already produced their own feedback
|
||||
# Same guard as every other late notice in TurnRunner: after /stop, /new or a restart the
|
||||
# turn is over and this chat belongs to a newer run — do not edit or post into it.
|
||||
still_current = getattr(runner._ctx, "_run_still_current", None)
|
||||
if callable(still_current) and not still_current():
|
||||
return
|
||||
runner._schedule(
|
||||
_post_timeout_notice(runner._ctx, command, card_message_id, timeout_s),
|
||||
"Approval timeout notice scheduling error")
|
||||
|
||||
register_gateway_settle(session_key, request_id, settle)
|
||||
|
||||
|
||||
async def _post_timeout_notice(ctx, command: str, card_message_id: Optional[str], timeout_s: int) -> None:
|
||||
from gateway.run import _interim_metadata
|
||||
|
||||
adapter = ctx._status_adapter
|
||||
notice = format_approval_timed_out_notice(timeout_s)
|
||||
metadata = _interim_metadata(ctx._status_thread_metadata)
|
||||
try:
|
||||
# Plain markdown, not the card's platform markup: ``edit_message`` re-formats it itself.
|
||||
if card_message_id and await _edit_card(adapter, ctx._status_chat_id, card_message_id, f"{notice}\n```\n{command}\n```"):
|
||||
return
|
||||
await adapter.send(ctx._status_chat_id, notice, metadata=metadata)
|
||||
except Exception:
|
||||
logger.debug("Approval timeout notice failed", exc_info=True)
|
||||
|
||||
|
||||
async def _edit_card(adapter, chat_id: str, message_id: str, content: str) -> bool:
|
||||
"""Edit the card in place (drops the buttons on platforms whose edit replaces the markup)."""
|
||||
edit: Optional[Callable[..., Any]] = getattr(adapter, "edit_message", None)
|
||||
if edit is None:
|
||||
return False
|
||||
try:
|
||||
result = await edit(chat_id, message_id, content)
|
||||
except Exception:
|
||||
logger.debug("Approval card edit failed; sending the notice as a new message", exc_info=True)
|
||||
return False
|
||||
return bool(getattr(result, "success", False))
|
||||
@@ -40,7 +40,8 @@ def format_session_stall_notification(idle_seconds: float) -> str:
|
||||
See #72016.
|
||||
"""
|
||||
mins = max(1, int(idle_seconds // 60))
|
||||
return f"⚠️ Agent session appears stalled (last activity {mins} min ago). Try /new to reset."
|
||||
return (f"⚠️ I seem to be stuck (no activity for {mins} min). Send /stop to cancel the current "
|
||||
"task, or /new to start a fresh conversation.")
|
||||
|
||||
|
||||
def _finite_float(value: Any) -> Optional[float]:
|
||||
|
||||
@@ -68,8 +68,9 @@ async def _quiet(call, default=None):
|
||||
return default
|
||||
|
||||
|
||||
HISTORY_UNREADABLE = ("⚠️ Conversation history is unreadable (state.db). "
|
||||
"This is not a new conversation — earlier messages exist but cannot be loaded.")
|
||||
HISTORY_UNREADABLE = ("⚠️ I can't read this conversation's history right now (your earlier messages "
|
||||
"exist but cannot be loaded). Run `hermes doctor --fix` on the host, or use /new "
|
||||
"to start fresh.")
|
||||
|
||||
|
||||
def _quiet_sync(call, default=None):
|
||||
|
||||
+37
-4
@@ -6,6 +6,8 @@ gateway, sessions, …) is built by ``hermes_cli/subcommands/<group>.py`` and wi
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import difflib
|
||||
import re
|
||||
from functools import lru_cache
|
||||
|
||||
# `--profile` / `-p` is consumed by ``main._apply_profile_override`` before argparse runs
|
||||
@@ -24,6 +26,13 @@ _VALUE_FLAGS_FALLBACK: frozenset[str] = frozenset({
|
||||
_OPTIONAL_VALUE_FLAGS_FALLBACK: frozenset[str] = frozenset({"-c", "--continue"})
|
||||
|
||||
|
||||
def _cfg_path() -> str:
|
||||
"""``~/.hermes/config.yaml`` spelled for the active profile, for help text."""
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
return f"{display_hermes_home()}/config.yaml"
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def top_level_value_flag_sets() -> tuple[frozenset[str], frozenset[str]]:
|
||||
"""(required-value, optional-value) top-level flags, derived from the REAL parser.
|
||||
@@ -163,7 +172,7 @@ def _add_top_level_flags(parser: argparse.ArgumentParser) -> None:
|
||||
inherited(parser, "--pass-session-id", action="store_true", default=False,
|
||||
help="Include the session ID in the agent's system prompt")
|
||||
inherited(parser, "--ignore-user-config", action="store_true", default=False,
|
||||
help="Ignore ~/.hermes/config.yaml and fall back to built-in defaults (credentials in .env are still loaded)")
|
||||
help=f"Ignore {_cfg_path()} and fall back to built-in defaults (credentials in .env are still loaded)")
|
||||
inherited(parser, "--ignore-rules", action="store_true", default=False,
|
||||
help="Skip auto-injection of AGENTS.md, SOUL.md, .cursorrules, memory, and preloaded skills")
|
||||
inherited(parser, "--safe-mode", action="store_true", default=False,
|
||||
@@ -264,7 +273,7 @@ def _build_chat_parser(subparsers) -> argparse.ArgumentParser:
|
||||
inherited(chat_parser, "--pass-session-id", action="store_true", default=SUPPRESS,
|
||||
help="Include the session ID in the agent's system prompt")
|
||||
inherited(chat_parser, "--ignore-user-config", action="store_true", default=SUPPRESS,
|
||||
help="Ignore ~/.hermes/config.yaml and fall back to built-in defaults (credentials in .env are still loaded). Useful for isolated CI runs, reproduction, and third-party integrations.")
|
||||
help=f"Ignore {_cfg_path()} and fall back to built-in defaults (credentials in .env are still loaded). Useful for isolated CI runs, reproduction, and third-party integrations.")
|
||||
inherited(chat_parser, "--ignore-rules", action="store_true", default=SUPPRESS,
|
||||
help="Skip auto-injection of AGENTS.md, SOUL.md, .cursorrules, memory, and preloaded skills. Combine with --ignore-user-config for a fully isolated run.")
|
||||
inherited(chat_parser, "--safe-mode", action="store_true", default=SUPPRESS,
|
||||
@@ -280,6 +289,28 @@ def _build_chat_parser(subparsers) -> argparse.ArgumentParser:
|
||||
return chat_parser
|
||||
|
||||
|
||||
class HermesArgumentParser(argparse.ArgumentParser):
|
||||
"""argparse parser whose unknown-subcommand error is three short lines, not a 70-name dump.
|
||||
|
||||
Stock argparse prints the full usage block plus ``(choose from 'chat', 'model', …)`` when
|
||||
the first positional is not a registered subcommand. That buries the only useful fact
|
||||
(the word is not a command) and offers no closest match. Every other error keeps the
|
||||
stock usage + message shape.
|
||||
"""
|
||||
|
||||
def _check_value(self, action, value):
|
||||
if isinstance(action, argparse._SubParsersAction) and value not in action.choices:
|
||||
# ``self.prog`` is "hermes" at the top level and "hermes gateway" for a nested group
|
||||
# (argparse hands add_parser() the parent's class), so the copy stays correct for both.
|
||||
lines = [f"{self.prog}: '{value}' is not a `{self.prog}` command."]
|
||||
close = difflib.get_close_matches(str(value), list(action.choices), n=3, cutoff=0.6)
|
||||
if close:
|
||||
lines.append(f"Did you mean: {', '.join(close)}?")
|
||||
lines.append(f"Run `{self.prog} --help` to see all commands.")
|
||||
self.exit(2, "\n".join(lines) + "\n")
|
||||
super()._check_value(action, value)
|
||||
|
||||
|
||||
def build_top_level_parser():
|
||||
"""Build the top-level parser, the subparsers action, and the ``chat`` subparser.
|
||||
|
||||
@@ -287,9 +318,11 @@ def build_top_level_parser():
|
||||
``chat_parser.set_defaults(func= cmd_chat)`` and registers further subparsers via
|
||||
``subparsers.add_parser(...)``.
|
||||
"""
|
||||
parser = argparse.ArgumentParser(
|
||||
parser = HermesArgumentParser(
|
||||
prog="hermes", description="Hermes Agent - AI assistant with tool-calling capabilities",
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter, epilog=_EPILOGUE)
|
||||
_add_top_level_flags(parser)
|
||||
subparsers = parser.add_subparsers(dest="command", help="Command to run")
|
||||
# metavar keeps the usage line to ``hermes [...] <command>`` instead of the brace list of
|
||||
# every subcommand name; ``hermes --help`` still lists each command with its help row.
|
||||
subparsers = parser.add_subparsers(dest="command", help="Command to run", metavar="<command>")
|
||||
return parser, subparsers, _build_chat_parser(subparsers)
|
||||
|
||||
@@ -149,15 +149,17 @@ def _is_same_writer(entry: dict[str, Any], metadata: Optional[dict[str, Any]]) -
|
||||
|
||||
|
||||
def session_already_owned_message(session_id: str, entry: dict[str, Any]) -> str:
|
||||
"""Refusal text for a session another live process holds.
|
||||
|
||||
Contract shared with the TUI/Desktop surfaces: the FIRST line is the plain user sentence
|
||||
(no lease/pid/owner jargon); the second line is ``Details: ...`` for logs and bug reports.
|
||||
"""
|
||||
surface = str(entry.get("surface") or "another surface")
|
||||
pid = entry.get("pid")
|
||||
started = _optional_float(entry.get("started_at"))
|
||||
age = f", lease age {format_age(time.time() - started)}" if started else ""
|
||||
age = f" {format_age(time.time() - started)} ago" if started else ""
|
||||
return (
|
||||
f"Session {session_id} already has a live owner ({surface}, pid {pid}{age}). "
|
||||
"Its turn activity is unknown; an open lease does not mean a turn is running. "
|
||||
"Attach through a compatible owner, or close the session in its owning surface "
|
||||
"before resuming here. Do not delete a live owner's lease to force a takeover."
|
||||
"This chat is open in another Hermes window/terminal. Use it there, or start a new chat here.\n"
|
||||
f"Details: session {session_id} opened by {surface}{age}."
|
||||
)
|
||||
|
||||
|
||||
|
||||
+5
-3
@@ -1505,10 +1505,12 @@ def resolve_provider(
|
||||
return "bedrock"
|
||||
except ImportError:
|
||||
pass # boto3 not installed
|
||||
from hermes_constants import display_hermes_home
|
||||
raise AuthError(
|
||||
"No inference provider configured. Run 'hermes model' to choose a "
|
||||
"provider and model, or set an API key (OPENROUTER_API_KEY, "
|
||||
"OPENAI_API_KEY, etc.) in ~/.hermes/.env.",
|
||||
"Hermes is not connected to any AI provider yet. Run `hermes model` to pick one (the free "
|
||||
"Nous tier needs no API key), type `/login` in chat, or add a key with "
|
||||
f"`hermes auth add <provider>`. (Advanced: put an API key such as OPENROUTER_API_KEY in "
|
||||
f"{display_hermes_home()}/.env.)",
|
||||
code="no_provider_configured")
|
||||
|
||||
|
||||
|
||||
@@ -131,6 +131,18 @@ def _is_known_provider(provider: str, configured_provider: dict | None) -> bool:
|
||||
or provider.startswith(CUSTOM_POOL_PREFIX) or configured_provider is not None)
|
||||
|
||||
|
||||
def _unknown_provider_exit(provider: str) -> SystemExit:
|
||||
"""Did-you-mean over the known provider ids plus the two commands that list/pick them."""
|
||||
import difflib
|
||||
known = sorted(set(PROVIDER_REGISTRY) | {"openrouter"}
|
||||
| {entry["name"] for entry in _get_custom_provider_entries()})
|
||||
close = difflib.get_close_matches(provider, known, n=3, cutoff=0.5)
|
||||
hint = f" Did you mean {', '.join(close)}?" if close else ""
|
||||
return SystemExit(
|
||||
f"Unknown provider '{provider}'.{hint} Run `hermes auth` to see the provider list, or "
|
||||
"`hermes model` to pick one interactively.")
|
||||
|
||||
|
||||
def _display_source(source: str) -> str:
|
||||
return source.split(":", 1)[1] if source.startswith("manual:") else source
|
||||
|
||||
@@ -348,7 +360,7 @@ def auth_add_command(args) -> None:
|
||||
provider = _normalize_provider(getattr(args, "provider", ""))
|
||||
configured_provider = _configured_provider_entry(provider)
|
||||
if not _is_known_provider(provider, configured_provider):
|
||||
raise SystemExit(f"Unknown provider: {provider}")
|
||||
raise _unknown_provider_exit(provider)
|
||||
if configured_provider is not None:
|
||||
_migrate_legacy_custom_pool_key(provider, configured_provider["pool_key"])
|
||||
|
||||
@@ -573,10 +585,12 @@ def auth_refresh_command(args) -> None:
|
||||
refreshed = pool.try_refresh_matching(credential_id=matched.id)
|
||||
if refreshed is None:
|
||||
after = next((e for e in pool.entries() if e.id == matched.id), None)
|
||||
state = "removed from pool" if after is None else (after.last_status or "unknown")
|
||||
label = PROVIDER_REGISTRY[provider].name if provider in PROVIDER_REGISTRY else provider
|
||||
state = ("it was removed from the pool" if after is None
|
||||
else "the saved session is no longer valid")
|
||||
raise SystemExit(
|
||||
f"Refresh failed for {provider} credential #{index} ({matched.label}); "
|
||||
f"status now: {state}.")
|
||||
f"Could not renew the {label} sign-in for credential #{index} ({matched.label}); {state}. "
|
||||
f"Sign in again with `hermes auth add {provider} --type oauth`.")
|
||||
status = refreshed.last_status or "ok"
|
||||
if status == "ok":
|
||||
print(f"Refreshed {provider} credential #{index} ({refreshed.label}); status: ok")
|
||||
@@ -721,7 +735,7 @@ def _interactive_add() -> None:
|
||||
provider = _pick_provider("Provider to add credential for")
|
||||
configured_provider = _configured_provider_entry(provider)
|
||||
if not _is_known_provider(provider, configured_provider):
|
||||
raise SystemExit(f"Unknown provider: {provider}")
|
||||
raise _unknown_provider_exit(provider)
|
||||
|
||||
auth_type = "api_key"
|
||||
if provider in _OAUTH_CAPABLE_PROVIDERS:
|
||||
|
||||
@@ -358,9 +358,11 @@ def _poll_for_token(
|
||||
raise ValueError("Token response did not include access_token")
|
||||
|
||||
def _error(_response, error_payload) -> Exception:
|
||||
error_code = error_payload.get("error", "")
|
||||
description = error_payload.get("error_description") or "Unknown authentication error"
|
||||
return RuntimeError(f"{error_code}: {description}")
|
||||
# Plain copy per OAuth error code; the raw ``code: description`` stays on a Details line.
|
||||
from hermes_cli.auth_error_copy import device_flow_error
|
||||
return device_flow_error(
|
||||
str(error_payload.get("error", "") or ""),
|
||||
str(error_payload.get("error_description") or "Unknown authentication error"))
|
||||
|
||||
return _poll_device_token_generic(
|
||||
lambda: client.post(
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Plain-language copy for sign-in and provider-setup failures (CLI).
|
||||
|
||||
One table of ``(predicate, lead sentence)`` classifies an exception into a sentence a first-time
|
||||
user can act on; the raw exception is demoted to a ``Details:`` line so nothing is lost for
|
||||
support. Callers print ``sign_in_failure_lines(...)`` / ``provider_setup_failure_lines(...)``
|
||||
line by line.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Sequence, Tuple
|
||||
|
||||
# httpx class names and stdlib bases that mean "the request never got a usable answer".
|
||||
_NETWORK_ERROR_TYPES = frozenset({
|
||||
"ConnectError", "ConnectTimeout", "ReadTimeout", "PoolTimeout", "WriteTimeout", "TimeoutException",
|
||||
"RemoteProtocolError", "ReadError", "ProxyError", "UnsupportedProtocol", "NetworkError",
|
||||
})
|
||||
|
||||
# OAuth device-flow error codes (RFC 8628 §3.5) -> plain copy. ``{retry}`` is the retry command.
|
||||
DEVICE_FLOW_ERROR_COPY = {
|
||||
"expired_token": (
|
||||
"The sign-in code expired before it was approved in the browser. Run `{retry}` to get a new code."),
|
||||
"access_denied": (
|
||||
"Sign-in was declined in the browser. Run `{retry}` to try again, or `hermes model` to pick a "
|
||||
"different provider."),
|
||||
"invalid_grant": (
|
||||
"The sign-in code was not accepted by the server. Run `{retry}` to get a new code."),
|
||||
"invalid_client": (
|
||||
"The server did not recognize this copy of Hermes. Run `hermes update`, then `{retry}` again."),
|
||||
}
|
||||
|
||||
|
||||
class SignInCopyError(RuntimeError):
|
||||
"""Exception whose ``str()`` is already user copy (lead line + ``Details:`` line)."""
|
||||
|
||||
def __init__(self, message: str, *, oauth_error_code: str = "") -> None:
|
||||
super().__init__(message)
|
||||
self.oauth_error_code = oauth_error_code
|
||||
|
||||
|
||||
def is_network_error(exc: BaseException) -> bool:
|
||||
"""True for connection/DNS/timeout failures from httpx, requests or the stdlib."""
|
||||
if isinstance(exc, SignInCopyError):
|
||||
return False
|
||||
names = {cls.__name__ for cls in type(exc).__mro__}
|
||||
return bool(names & _NETWORK_ERROR_TYPES) or isinstance(exc, (ConnectionError, TimeoutError))
|
||||
|
||||
|
||||
def is_cancelled(exc: BaseException) -> bool:
|
||||
return isinstance(exc, (KeyboardInterrupt, EOFError)) or (
|
||||
isinstance(exc, SystemExit) and exc.code in (130, None, 0))
|
||||
|
||||
|
||||
def device_flow_error(code: str, description: str, *, retry_command: str = "hermes portal") -> SignInCopyError:
|
||||
"""Exception for an OAuth device-flow error code whose text is already user-facing.
|
||||
|
||||
Unknown codes keep the server's description as the lead (it is the only information available)
|
||||
but still name the retry command.
|
||||
"""
|
||||
lead = DEVICE_FLOW_ERROR_COPY.get(code, "").format(retry=retry_command)
|
||||
if not lead:
|
||||
lead = (f"Sign-in did not complete: {description or 'the server rejected the request'}. "
|
||||
f"Run `{retry_command}` to try again.")
|
||||
details = f"{code}: {description}" if code else description
|
||||
return SignInCopyError(f"{lead}\n Details: {details}" if details else lead, oauth_error_code=code)
|
||||
|
||||
|
||||
def _details_line(exc: BaseException) -> str:
|
||||
text = str(exc).strip() or type(exc).__name__
|
||||
return f" Details: {text}"
|
||||
|
||||
|
||||
_Rule = Tuple[Callable[[BaseException], bool], str]
|
||||
|
||||
|
||||
def _classify(exc: BaseException, rules: Sequence[_Rule], other: str) -> str:
|
||||
return next((copy for pred, copy in rules if pred(exc)), other)
|
||||
|
||||
|
||||
def sign_in_failure_lines(
|
||||
exc: BaseException, *, service_host: str = "portal.nousresearch.com", retry_command: str = "hermes portal",
|
||||
) -> list:
|
||||
"""Lines to print when a device-code / browser sign-in fails for any non-timeout reason."""
|
||||
if isinstance(exc, SignInCopyError):
|
||||
return str(exc).splitlines()
|
||||
rules: Sequence[_Rule] = (
|
||||
(is_cancelled, "Sign-in was cancelled. Run `{retry}` when you want to try again."),
|
||||
(is_network_error,
|
||||
"Could not sign in: Hermes could not reach {host}. Check your internet connection or proxy, "
|
||||
"then run `{retry}` again."),
|
||||
)
|
||||
lead = _classify(
|
||||
exc, rules,
|
||||
"Could not sign in. Run `{retry}` to try again, or `hermes model` to pick a different provider.")
|
||||
lines = [lead.format(host=service_host, retry=retry_command)]
|
||||
if not is_cancelled(exc):
|
||||
lines.append(_details_line(exc))
|
||||
return lines
|
||||
|
||||
|
||||
def provider_setup_failure_lines(exc: BaseException, *, retry_command: str = "hermes model") -> list:
|
||||
"""Lines to print when the setup wizard's provider step fails: reason, that nothing was saved,
|
||||
and how to retry."""
|
||||
nothing_saved = (
|
||||
"Your provider settings were not changed. Continue the wizard now and run "
|
||||
f"`{retry_command}` afterwards to try again.")
|
||||
if isinstance(exc, SignInCopyError):
|
||||
lead, *details = str(exc).splitlines()
|
||||
return [f"Could not finish connecting a provider: {lead[0].lower()}{lead[1:]}", nothing_saved, *details]
|
||||
rules: Sequence[_Rule] = (
|
||||
(is_cancelled, "sign-in was cancelled"),
|
||||
(is_network_error, "no internet connection, or the provider could not be reached"),
|
||||
)
|
||||
reason = _classify(exc, rules, "something went wrong while talking to the provider")
|
||||
lines = [f"Could not finish connecting a provider ({reason}).", nothing_saved]
|
||||
if not is_cancelled(exc):
|
||||
lines.append(_details_line(exc))
|
||||
return lines
|
||||
@@ -1582,5 +1582,13 @@ def _login_nous(args, pconfig: ProviderConfig) -> None:
|
||||
print("\nLogin cancelled.")
|
||||
raise SystemExit(130)
|
||||
except Exception as exc:
|
||||
print(f"Login failed: {exc}")
|
||||
from hermes_cli.auth_error_copy import sign_in_failure_lines
|
||||
logger.debug("nous login failed: %r", exc)
|
||||
print()
|
||||
for line in sign_in_failure_lines(exc, service_host=_portal_host(getattr(args, "portal_url", None))):
|
||||
print(line)
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
def _portal_host(portal_url: Optional[str]) -> str:
|
||||
return urlparse(portal_url or DEFAULT_NOUS_PORTAL_URL).hostname or "portal.nousresearch.com"
|
||||
|
||||
@@ -35,6 +35,13 @@ logger = logging.getLogger(__name__)
|
||||
# snapshots (see ``create_quick_snapshot``); defined here because the exclusion set needs it.
|
||||
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
|
||||
|
||||
|
||||
def _snapshot_recovery_hint() -> str:
|
||||
"""How to restore a state snapshot. There is no `hermes snapshot` subcommand — only the /snapshot
|
||||
slash command inside a `hermes` session (hermes_cli/commands.py)."""
|
||||
return ("To restore a newer snapshot, start `hermes` in a terminal and run `/snapshot list`, then "
|
||||
"`/snapshot restore <id>` (CLI only).")
|
||||
|
||||
# Directory names to skip (matched against each path component). ``hermes-agent`` only matches at
|
||||
# the root (``_should_exclude``) so skill dirs like ``skills/.../hermes-agent/`` survive. The
|
||||
# dependency/cache entries matter: one plugin venv or pip/uv cache under HERMES_HOME walked
|
||||
@@ -1002,8 +1009,7 @@ def run_import(args) -> None:
|
||||
for rel, before, after in db_shrunk:
|
||||
print(f" {rel}: {before[0]} session(s) / {before[1]} message(s)"
|
||||
f" -> {after[0]} / {after[1]}")
|
||||
print(" Anything recorded after the backup was taken is not in it. "
|
||||
"Recover from a newer backup or snapshot: hermes snapshot list")
|
||||
print(f" Anything recorded after the backup was taken is not in it. {_snapshot_recovery_hint()}")
|
||||
if skipped_runtime:
|
||||
_print_capped(f"\n Preserved {len(skipped_runtime)} runtime state "
|
||||
f"file(s) (kept this machine's, not the backup's):",
|
||||
@@ -1229,7 +1235,7 @@ def _create_quick_snapshot_locked(
|
||||
# Surface on stdout: a log-and-continue made a missing state.db backup look like a
|
||||
# successful pre-update snapshot (#68474).
|
||||
print(f" ⚠ CRITICAL: could not snapshot DB file(s): {', '.join(failed_dbs)}\n"
|
||||
f" ⚠ If sessions disappear after update, check {root} and run: hermes snapshot list")
|
||||
f" ⚠ If sessions disappear after the update, check {root}. {_snapshot_recovery_hint()}")
|
||||
logger.error("Quick snapshot failed to capture DB file(s): %s", ", ".join(failed_dbs))
|
||||
if not manifest:
|
||||
shutil.rmtree(staging_dir, ignore_errors=True)
|
||||
|
||||
+13
-1
@@ -2,6 +2,7 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -756,7 +757,18 @@ def _mcp_server_line(srv: dict, *, dim: str, text: str) -> str:
|
||||
"configured": f"[dim {dim}]— configured[/]"}.get(status)
|
||||
if suffix is not None:
|
||||
return f"[dim {dim}]{name}[/] [dim]({transport})[/] {suffix}"
|
||||
return f"[red]{name}[/] [dim]({transport})[/] [red]— failed[/]"
|
||||
return _mcp_failed_line(name, transport, srv.get("error"))
|
||||
|
||||
|
||||
def _mcp_failed_line(name: str, transport: str, error: Optional[str]) -> str:
|
||||
"""Failed MCP connect: the short reason (already humanised by ``_format_connect_error``) and the
|
||||
exact next command, so 'failed' is never the whole story."""
|
||||
from rich.markup import escape
|
||||
reason = escape(" ".join(str(error or "").split())[:120]) or "no details recorded"
|
||||
next_cmd = (f"hermes mcp login {name}" if re.search(r"\b401\b|unauthori[sz]ed", reason, re.I)
|
||||
else f"hermes mcp test {name}")
|
||||
return (f"[red]{name}[/] [dim]({transport})[/] [red]— could not connect:[/] {reason} "
|
||||
f"[dim]— run `{next_cmd}`[/]")
|
||||
|
||||
|
||||
def _truncate_tool_names(tool_names: List[str]) -> List[Optional[str]]:
|
||||
|
||||
@@ -613,7 +613,8 @@ class CLIAgentSetupMixin:
|
||||
return True
|
||||
except Exception as e:
|
||||
console = ChatConsole()
|
||||
console.print(f"[bold red]Failed to initialize agent: {e}[/]")
|
||||
from hermes_cli.cli_chat_error_copy import agent_init_failure_message
|
||||
console.print(f"[bold red]{_escape(agent_init_failure_message(e))}[/]")
|
||||
from hermes_constants import partial_update_hint
|
||||
for line in partial_update_hint(e):
|
||||
console.print(line)
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Plain-language copy for chat-turn failures shown in the CLI response panel.
|
||||
|
||||
The chat panel used to echo ``Error: HTTP 401: Invalid API key`` as the assistant's answer. These
|
||||
helpers map the classifier verdict (``agent/error_classifier.py``) to WHAT happened + WHAT TO DO,
|
||||
and demote the raw provider text to a ``Details:`` line.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
_SUMMARY_LIMIT = 120
|
||||
|
||||
# FailoverReason.value -> plain copy. ``{provider}`` / ``{model}`` are filled at render time.
|
||||
_REASON_COPY: dict[str, str] = {
|
||||
"auth": "Your {provider} key was rejected. Run `hermes model` to re-enter it.",
|
||||
"auth_permanent": "Your {provider} key was rejected. Run `hermes model` to re-enter it.",
|
||||
"billing": "Your {provider} account is out of credit. Top up at the provider, or run /model to switch.",
|
||||
"model_not_found": "'{model}' isn't available on {provider}. Run /model to pick a valid model.",
|
||||
"rate_limit": "Rate limited by {provider}; wait a minute or /model to switch.",
|
||||
"upstream_rate_limit": "Rate limited by {provider}; wait a minute or /model to switch.",
|
||||
"overloaded": "{provider} is overloaded right now. Send /retry in a moment, or /model to switch.",
|
||||
"server_error": "{provider} had an internal error. Send /retry in a moment, or /model to switch.",
|
||||
"timeout": "{provider} did not answer in time. Send /retry, or /model to switch.",
|
||||
}
|
||||
_UNKNOWN_COPY = "The model request failed. Run /model to switch or `hermes doctor` to check the setup."
|
||||
|
||||
|
||||
def _short(text: str, limit: int = _SUMMARY_LIMIT) -> str:
|
||||
first = (text or "").strip().splitlines()[0] if (text or "").strip() else ""
|
||||
return first if len(first) <= limit else first[: limit - 1].rstrip() + "…"
|
||||
|
||||
|
||||
def chat_error_response(
|
||||
error: Exception | str, *, provider: str = "", model: str = "", failure_reason: str | None = None,
|
||||
) -> str:
|
||||
"""Two-line panel text: plain sentence with the fix command, then ``Details: <raw>``.
|
||||
|
||||
``failure_reason`` is the verdict the turn loop already stamped on its result
|
||||
(``agent/turn_failure_copy.stamp_failure``). When present it is used as-is instead of
|
||||
re-classifying a summarised string (which has no status code and almost always lands on
|
||||
'unknown'); for the loop's own site codes the ``error`` text is already the user-facing copy,
|
||||
so it is returned verbatim rather than wrapped and demoted to a Details line."""
|
||||
reason = str(failure_reason or "").strip()
|
||||
if reason:
|
||||
from agent.turn_failure_copy import SITE_FAILURE_CODES
|
||||
|
||||
text = str(error or "").strip()
|
||||
if reason in SITE_FAILURE_CODES and text:
|
||||
return text
|
||||
else:
|
||||
from agent.error_classifier import classify_api_error
|
||||
|
||||
exc = error if isinstance(error, Exception) else Exception(str(error))
|
||||
reason = classify_api_error(exc, provider=provider or "", model=model or "").reason.value
|
||||
copy = _REASON_COPY.get(reason, _UNKNOWN_COPY)
|
||||
lead = copy.format(provider=provider or "the provider", model=model or "the current model")
|
||||
return f"{lead}\nDetails: {_short(str(error), 300)}"
|
||||
|
||||
|
||||
def agent_init_failure_message(error: BaseException) -> str:
|
||||
"""Copy for a failed AIAgent build on first message: the user's turn was dropped."""
|
||||
return (
|
||||
f"Hermes couldn't start the model connection: {_short(str(error)) or type(error).__name__}. "
|
||||
"Your message was not sent. Run `hermes doctor` to check the setup, "
|
||||
"or /model to pick a different provider."
|
||||
)
|
||||
@@ -331,8 +331,12 @@ class CLIChatTurnMixin:
|
||||
except Exception as exc:
|
||||
logging.error("run_conversation raised: %s", exc, exc_info=True)
|
||||
_summary = getattr(self.agent, '_summarize_api_error', lambda e: str(e)[:300])(exc)
|
||||
from hermes_cli.cli_chat_error_copy import chat_error_response
|
||||
turn.result = {
|
||||
"final_response": f"Error: {_summary}", "messages": [], "api_calls": 0,
|
||||
"final_response": chat_error_response(
|
||||
exc, provider=str(getattr(self.agent, "provider", "") or self.provider or ""),
|
||||
model=str(getattr(self.agent, "model", "") or self.model or "")),
|
||||
"messages": [], "api_calls": 0,
|
||||
"completed": False, "failed": True, "error": _summary,
|
||||
}
|
||||
finally:
|
||||
@@ -476,7 +480,12 @@ class CLIChatTurnMixin:
|
||||
response = turn.result.get("final_response", "") if turn.result else ""
|
||||
# "failed"/"partial" with an empty final_response: no usable answer.
|
||||
if turn.result and (turn.result.get("failed") or turn.result.get("partial")) and not response:
|
||||
response = f"Error: {turn.result.get('error', 'Unknown error')}"
|
||||
from hermes_cli.cli_chat_error_copy import chat_error_response
|
||||
response = chat_error_response(
|
||||
str(turn.result.get("error") or "Unknown error"),
|
||||
provider=str(getattr(self.agent, "provider", "") or self.provider or ""),
|
||||
model=str(getattr(self.agent, "model", "") or self.model or ""),
|
||||
failure_reason=turn.result.get("failure_reason"))
|
||||
# Stop continuous voice on persistent errors (e.g. 429) — else error→record→error loops.
|
||||
if self._voice_continuous:
|
||||
self._voice_continuous = False
|
||||
|
||||
@@ -357,7 +357,7 @@ def _without_session_meta(messages) -> list:
|
||||
|
||||
def _db_unavailable_line() -> str:
|
||||
from hermes_state import format_session_db_unavailable
|
||||
return f" {format_session_db_unavailable()}"
|
||||
return f" {format_session_db_unavailable(details=True)}"
|
||||
|
||||
|
||||
def _print_side_result_panel(cli, *, header_lines, body, title_suffix, empty_note, console=None) -> None:
|
||||
|
||||
@@ -116,7 +116,7 @@ class CLILoopsMixin:
|
||||
if len(parts) == 1:
|
||||
# No argument: show current title and session ID.
|
||||
if not self._session_db:
|
||||
_cprint(f" {format_session_db_unavailable()}")
|
||||
_cprint(f" {format_session_db_unavailable(details=True)}")
|
||||
return
|
||||
_cprint(f" Session ID: {self.session_id}")
|
||||
session = self._session_db.get_session(self.session_id)
|
||||
@@ -132,7 +132,7 @@ class CLILoopsMixin:
|
||||
_cprint(" Usage: /title <your session title>")
|
||||
return
|
||||
if not self._session_db:
|
||||
_cprint(f" {format_session_db_unavailable()}")
|
||||
_cprint(f" {format_session_db_unavailable(details=True)}")
|
||||
return
|
||||
# Sanitize early so feedback matches what gets stored. A rejection (e.g. too
|
||||
# long) prints that one reason and stops — never a second, contradictory
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Copy for an unknown slash command: says nothing was sent and suggests a near-miss."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import difflib
|
||||
from collections.abc import Iterable
|
||||
|
||||
|
||||
def unknown_command_lines(typed: str, known: Iterable[str]) -> tuple[str, str]:
|
||||
"""(lead line, pointer line) for a slash token with no handler.
|
||||
|
||||
``known`` holds names WITH the leading slash (``hermes_cli.commands.COMMANDS`` keys plus skill
|
||||
commands). A close match (typo) becomes ``Did you mean /model?``; prefix expansion already ran
|
||||
before this, so only fuzzy matches are considered here.
|
||||
"""
|
||||
base = typed.split()[0] if typed.split() else typed
|
||||
close = difflib.get_close_matches(base, list(known), n=1, cutoff=0.6)
|
||||
hint = f" Did you mean {close[0]}?" if close else ""
|
||||
return (f"Unknown command {base} — nothing was sent.{hint}", "Type /help for the full list.")
|
||||
+58
-27
@@ -50,18 +50,36 @@ class InvalidUserConfigError(RuntimeError):
|
||||
|
||||
|
||||
_PARSE_FAILURE_FALLBACK_MSG = {
|
||||
"last-known-good": (
|
||||
"Keeping the previously loaded config for this process — "
|
||||
"edits to config.yaml are being IGNORED until the YAML is fixed."),
|
||||
"last-known-good-backup": (
|
||||
"Loading the LAST KNOWN GOOD copy from backups/config/ instead — edits to config.yaml "
|
||||
"since that copy are being IGNORED until the YAML is fixed."),
|
||||
"refuse-write": (
|
||||
"REFUSING to write config.yaml so the existing file is preserved. "
|
||||
"Fix the YAML (hermes config edit) and retry.")}
|
||||
"last-known-good": "Hermes is running on the settings it loaded before the edit until it is fixed, so recent changes are not applied.",
|
||||
"last-known-good-backup": "Hermes is running on your last good settings until it is fixed, so recent changes are not applied.",
|
||||
"refuse-write": "Nothing was written, so the existing file is preserved."}
|
||||
_PARSE_FAILURE_DEFAULTS_MSG = (
|
||||
"Falling back to default config — every user override (auxiliary providers, fallback chain, "
|
||||
"model settings) is being IGNORED. Fix the YAML and restart.")
|
||||
"Hermes is running on default settings until it is fixed, so none of your saved settings are applied.")
|
||||
_PARSE_FAILURE_REPAIR_MSG = "Open it with `hermes config edit`, fix {where}, then run `hermes config check`."
|
||||
|
||||
|
||||
def _yaml_error_location(exc: Exception) -> str:
|
||||
"""``"line 12"`` from a PyYAML problem mark (1-based), else ``""``."""
|
||||
mark = getattr(exc, "problem_mark", None) or getattr(exc, "context_mark", None)
|
||||
line = getattr(mark, "line", None)
|
||||
return f"line {line + 1}" if isinstance(line, int) else ""
|
||||
|
||||
|
||||
def _yaml_error_details(exc: Exception) -> str:
|
||||
"""Single-line ``Details:`` text: the PyYAML problem, or the exception's first line."""
|
||||
problem = getattr(exc, "problem", None)
|
||||
text = f"{problem}" if problem else str(exc).strip()
|
||||
return " ".join(text.split())
|
||||
|
||||
|
||||
def format_config_parse_failure(config_path: Path, exc: Exception, *, fallback: str = "defaults") -> str:
|
||||
"""User copy for an unparseable config.yaml: what happened, what Hermes is doing, how to fix.
|
||||
Only the problem line/column is printed; the raw PyYAML text goes to a ``Details:`` line."""
|
||||
where = _yaml_error_location(exc)
|
||||
at = f" at {where}" if where else ""
|
||||
fallback_msg = _PARSE_FAILURE_FALLBACK_MSG.get(fallback, _PARSE_FAILURE_DEFAULTS_MSG)
|
||||
repair = _PARSE_FAILURE_REPAIR_MSG.format(where=where or "the problem")
|
||||
return f"Your settings file ({config_path}) has a formatting error{at}. {fallback_msg} {repair}"
|
||||
|
||||
|
||||
def _warn_config_parse_failure(
|
||||
@@ -84,13 +102,12 @@ def _warn_config_parse_failure(
|
||||
_CONFIG_PARSE_WARNED.add(key)
|
||||
from hermes_cli.config_backups import backup_config
|
||||
backup_path = backup_config(config_path, "corrupt")
|
||||
msg = f"Failed to parse {config_path}: {exc}. " + _PARSE_FAILURE_FALLBACK_MSG.get(
|
||||
fallback, _PARSE_FAILURE_DEFAULTS_MSG)
|
||||
msg = format_config_parse_failure(config_path, exc, fallback=fallback)
|
||||
if backup_path is not None:
|
||||
msg += f" A copy of the corrupted file was saved to {backup_path}."
|
||||
logger.warning(msg)
|
||||
msg += f" A copy of the broken file was saved to {backup_path}."
|
||||
logger.warning("%s Details: %s", msg, _yaml_error_details(exc))
|
||||
try:
|
||||
sys.stderr.write(f"⚠️ hermes config: {msg}\n")
|
||||
sys.stderr.write(f"⚠️ hermes config: {msg}\n Details: {_yaml_error_details(exc)}\n")
|
||||
sys.stderr.flush()
|
||||
except Exception:
|
||||
pass
|
||||
@@ -495,12 +512,14 @@ def require_parseable_user_config(*, ignore_user_config: bool = False) -> None:
|
||||
|
||||
from hermes_cli.config_backups import backup_config
|
||||
backup_path = backup_config(config_path, "corrupt")
|
||||
where = _yaml_error_location(parse_error)
|
||||
message = (
|
||||
f"Refusing non-interactive startup because {config_path} is invalid: "
|
||||
f"{parse_error}. Repair the file or pass --ignore-user-config to "
|
||||
"intentionally run with built-in defaults.")
|
||||
f"Hermes stopped because your settings file ({config_path}) has a formatting error"
|
||||
f"{f' at {where}' if where else ''}. Fix it with `hermes config edit` and check with "
|
||||
"`hermes config check`, or add --ignore-user-config to run once with default settings.")
|
||||
if backup_path is not None:
|
||||
message += f" A copy was saved to {backup_path}."
|
||||
message += f" A copy of the broken file is at {backup_path}."
|
||||
message += f" Details: {_yaml_error_details(parse_error)}"
|
||||
logger.error(message)
|
||||
raise InvalidUserConfigError(message) from parse_error
|
||||
|
||||
@@ -1935,11 +1954,23 @@ def read_raw_config_readonly() -> Dict[str, Any]:
|
||||
|
||||
|
||||
def _refuse_overwrite(config_path: Path, reason: str, exc: Exception, fix: str) -> RuntimeError:
|
||||
return RuntimeError(f"Refusing to overwrite {config_path}: existing config.yaml {reason} ({exc}). {fix}")
|
||||
"""Error for a write that must not replace an existing config.yaml. Plain lead + ``Details:``."""
|
||||
where = _yaml_error_location(exc)
|
||||
at = f" ({where})" if where else ""
|
||||
return RuntimeError(
|
||||
f"Your settings file ({config_path}) {reason}{at}, so this change was not saved. {fix} "
|
||||
f"Details: {_yaml_error_details(exc)}")
|
||||
|
||||
|
||||
def _backups_dir_display() -> str:
|
||||
from hermes_constants import display_hermes_home
|
||||
return f"{display_hermes_home()}/backups/config/"
|
||||
|
||||
|
||||
_FIX_PERMS = "Fix the file permissions or move it aside first."
|
||||
_FIX_YAML = "Fix the file or restore a copy from backups/config/ first."
|
||||
_FIX_YAML = (
|
||||
"Fix it with `hermes config edit` and check with `hermes config check`, or copy the newest good "
|
||||
"file from {backups} over config.yaml.")
|
||||
|
||||
|
||||
def require_readable_config_before_write(config_path: Optional[Path] = None) -> Dict[str, Any]:
|
||||
@@ -1964,16 +1995,16 @@ def require_readable_config_before_write(config_path: Optional[Path] = None) ->
|
||||
raise _refuse_overwrite(config_path, "cannot be read", exc, _FIX_PERMS) from exc
|
||||
except Exception as exc:
|
||||
_warn_config_parse_failure(config_path, exc, fallback="refuse-write")
|
||||
raise _refuse_overwrite(config_path, "is not valid YAML", exc, _FIX_YAML) from exc
|
||||
raise _refuse_overwrite(
|
||||
config_path, "has a formatting error", exc, _FIX_YAML.format(backups=_backups_dir_display())) from exc
|
||||
if loaded is None:
|
||||
return {}
|
||||
if not isinstance(loaded, dict):
|
||||
exc = TypeError(f"top-level YAML must be a mapping, got {type(loaded).__name__}")
|
||||
_warn_config_parse_failure(config_path, exc, fallback="refuse-write")
|
||||
raise RuntimeError(
|
||||
f"Refusing to overwrite {config_path}: top-level YAML must be a mapping, got "
|
||||
f"{type(loaded).__name__}. Fix the file or restore a copy from backups/config/ first."
|
||||
) from exc
|
||||
raise _refuse_overwrite(
|
||||
config_path, f"must start with settings names, but its top level is a {type(loaded).__name__}",
|
||||
exc, _FIX_YAML.format(backups=_backups_dir_display())) from exc
|
||||
return loaded
|
||||
|
||||
|
||||
|
||||
+32
-8
@@ -168,10 +168,11 @@ def _last_run_display(job: Dict[str, Any]) -> str:
|
||||
if last_status == "ok":
|
||||
return color("ok", Colors.GREEN)
|
||||
if last_status == "delivery_queued":
|
||||
return color("delivery_queued: completion unverified; do not resend", Colors.YELLOW)
|
||||
return color("finished; delivery is still in progress", Colors.YELLOW)
|
||||
if last_status == "delivery_failed":
|
||||
# Agent succeeded but the result never reached the user — not green; last_error is None.
|
||||
return color(f"delivery_failed: {job.get('last_delivery_error') or '?'}", Colors.YELLOW)
|
||||
return color(f"ran, but the result was not delivered ({_short_reason(job.get('last_delivery_error'))}). "
|
||||
f"{_delivery_fix_hint(job)}", Colors.YELLOW)
|
||||
display = color(f"{last_status}: {job.get('last_error', '?')}", Colors.RED)
|
||||
streak = int(job.get("failure_streak") or 0)
|
||||
if streak >= 2:
|
||||
@@ -215,13 +216,36 @@ def _job_rows(job: Dict[str, Any]) -> List[tuple[str, str]]:
|
||||
] + [(label, value) for label, value in optional if value]
|
||||
|
||||
|
||||
def _short_reason(text: Any, limit: int = 120) -> str:
|
||||
"""First line of an adapter/error blob, whitespace-collapsed and capped, or 'no details'."""
|
||||
first = str(text or "").strip().splitlines()
|
||||
reason = " ".join(first[0].split()) if first else ""
|
||||
return (reason[: limit - 1] + "…") if len(reason) > limit else (reason or "no details")
|
||||
|
||||
|
||||
def _delivery_fix_hint(job: Dict[str, Any]) -> str:
|
||||
return (f"Check the target with `hermes cron status` or change it with "
|
||||
f"`hermes cron edit {job.get('id', '<id>')} --deliver <target>`.")
|
||||
|
||||
|
||||
def _missed_fire_line(job: Dict[str, Any], fire_err: Dict[str, Any]) -> str:
|
||||
"""A scheduled fire that never reached the runner: what was skipped, when, and how to run it now.
|
||||
|
||||
The stored ``detail`` is operator text (loopback / api_server adapter); keep it as a dim
|
||||
second sentence and lead with the human cause (the gateway was unreachable)."""
|
||||
return (f"{color('⚠ A scheduled run was skipped', Colors.RED)} at {fire_err.get('at', '?')}: the messaging "
|
||||
f"gateway was unreachable. Run `hermes gateway restart`, then `hermes cron run {job.get('id', '<id>')}` "
|
||||
f"to run it now. {color('Details: ' + _short_reason(fire_err.get('detail')), Colors.DIM)}")
|
||||
|
||||
|
||||
def _job_warnings(job: Dict[str, Any]) -> List[str]:
|
||||
"""Delivery / fire warning lines for one job in ``cron list``."""
|
||||
lines = []
|
||||
if queued := job.get("last_delivery_queued"):
|
||||
lines.append(f"Delivery queued (completion unverified; do not resend): {queued}")
|
||||
lines.append(f"Delivery still in progress (the result was handed off but not confirmed yet): {queued}")
|
||||
if job.get("last_delivery_error"):
|
||||
lines.append(f"{color('⚠ Delivery failed:', Colors.YELLOW)} {job['last_delivery_error']}")
|
||||
lines.append(f"{color('⚠ The result was not delivered:', Colors.YELLOW)} "
|
||||
f"{_short_reason(job['last_delivery_error'])}. {_delivery_fix_hint(job)}")
|
||||
# A live adapter acked the last send but returned no message_id / raw_response
|
||||
# (Slack/Matrix/Mattermost shape): accepted as delivered, but say so here.
|
||||
if unverified := job.get("last_delivery_unverified"):
|
||||
@@ -229,8 +253,7 @@ def _job_warnings(job: Dict[str, Any]) -> List[str]:
|
||||
f"{_unverified_targets(unverified)} without message_id/raw_response")
|
||||
fire_err = job.get("last_fire_error")
|
||||
if isinstance(fire_err, dict) and fire_err.get("detail"):
|
||||
lines.append(f"{color('⚠ Missed scheduled fire:', Colors.RED)} "
|
||||
f"{fire_err.get('at', '?')} {fire_err['detail']}")
|
||||
lines.append(_missed_fire_line(job, fire_err))
|
||||
return lines
|
||||
|
||||
|
||||
@@ -469,7 +492,7 @@ def _script_health_issue(script: str) -> Optional[str]:
|
||||
try:
|
||||
path.relative_to(scripts_dir)
|
||||
except ValueError:
|
||||
return f"script resolves outside HERMES_HOME/scripts: {script!r}"
|
||||
return f"script resolves outside {scripts_dir}: {script!r}"
|
||||
if not path.exists():
|
||||
return f"script not found: {path}"
|
||||
if not path.is_file():
|
||||
@@ -505,7 +528,8 @@ def _cron_doctor_issues_for_job(job: Dict[str, Any]) -> List[str]:
|
||||
if last_status and last_status not in {"ok", "delivery_failed", "delivery_queued"}:
|
||||
issues.append(f"last run failed: {str(job.get('last_error') or 'unknown error').strip()}")
|
||||
if delivery_err := str(job.get("last_delivery_error") or "").strip():
|
||||
issues.append(f"last delivery failed: {delivery_err}")
|
||||
issues.append(f"last run finished but the result was not delivered ({_short_reason(delivery_err)}). "
|
||||
f"{_delivery_fix_hint(job)}")
|
||||
if unverified := job.get("last_delivery_unverified"):
|
||||
issues.append("last delivery unverified (adapter acked without evidence): "
|
||||
+ _unverified_targets(unverified))
|
||||
|
||||
@@ -132,7 +132,9 @@ def _ack_advisory(ack_target: str) -> None:
|
||||
if ack_advisory(ack_target):
|
||||
print(color(f" ✓ Acknowledged advisory {ack_target}. It will no longer trigger startup banners.", Colors.GREEN))
|
||||
else:
|
||||
print(color(f" ✗ Failed to persist ack for {ack_target}. Check ~/.hermes/config.yaml is writable.", Colors.RED))
|
||||
print(color(f" ✗ Could not save the acknowledgement for {ack_target}. Make sure {_DHH}/config.yaml is "
|
||||
f"writable (`hermes config path` prints the exact file), then re-run "
|
||||
f"`hermes doctor --ack {ack_target}`.", Colors.RED))
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
||||
@@ -256,7 +256,7 @@ def _validate_model_config(config_path, issues: list) -> None:
|
||||
with warn_on_error(""):
|
||||
if not _provider_has_credentials(runtime_provider):
|
||||
_fail_and_issue(f"model.provider '{runtime_provider}' is set but no API key is configured",
|
||||
"(check ~/.hermes/.env or run 'hermes setup')",
|
||||
f"(add it to {_DHH}/.env or run 'hermes setup')",
|
||||
f"No credentials found for provider '{runtime_provider}'. Run 'hermes setup' or set the provider's "
|
||||
f"API key in {_DHH}/.env, or switch providers with 'hermes config set model.provider <name>'", issues)
|
||||
|
||||
|
||||
@@ -148,13 +148,15 @@ _BUILTIN_TERMINAL_BACKENDS = {"local", "docker", "singularity", "modal", "manage
|
||||
def _check_docker_backend(terminal_env: str, running_in_container: bool, issues: list[str]) -> None:
|
||||
if terminal_env == "docker":
|
||||
if not _safe_which("docker"):
|
||||
_fail_and_issue("docker not found", "(required for TERMINAL_ENV=docker)", "Install Docker or change TERMINAL_ENV", issues)
|
||||
_fail_and_issue("Docker not installed", "(needed for the 'docker' terminal backend)",
|
||||
"Install Docker, or run `hermes setup terminal` to switch backend.", issues)
|
||||
else:
|
||||
# `docker version` hits /version, which socket proxies (tecnativa) allow by default; `docker info`
|
||||
# needs /info and is commonly blocked, giving a false "daemon not running". The backend itself
|
||||
# probes with `docker version` too (environments/docker.py).
|
||||
_require(_run_ok(["docker", "version"], timeout=10), ("docker", "(daemon running)"), ("docker daemon not running", ""),
|
||||
"Start Docker daemon", issues)
|
||||
_require(_run_ok(["docker", "version"], timeout=10), ("docker", "(daemon running)"),
|
||||
("Docker daemon not running", "(needed for the 'docker' terminal backend)"),
|
||||
"Start Docker, or run `hermes setup terminal` to switch backend.", issues)
|
||||
elif _safe_which("docker"):
|
||||
check_ok("docker", "(optional)")
|
||||
elif _is_termux():
|
||||
@@ -166,7 +168,8 @@ def _check_docker_backend(terminal_env: str, running_in_container: bool, issues:
|
||||
def _check_ssh_backend(issues: list[str]) -> None:
|
||||
ssh_host = os.getenv("TERMINAL_SSH_HOST")
|
||||
if not ssh_host:
|
||||
return _fail_and_issue("TERMINAL_SSH_HOST not set", "(required for TERMINAL_ENV=ssh)", "Set TERMINAL_SSH_HOST in .env", issues)
|
||||
return _fail_and_issue("SSH host not configured", "(needed for the 'ssh' terminal backend)",
|
||||
"run `hermes setup terminal` and enter the SSH host and user.", issues)
|
||||
ssh_user, ssh_port, ssh_key = (os.getenv(f"TERMINAL_SSH_{k}") for k in ("USER", "PORT", "KEY"))
|
||||
cmd = ["ssh", "-o", "ConnectTimeout=5", "-o", "BatchMode=yes"]
|
||||
if ssh_port:
|
||||
@@ -186,7 +189,8 @@ def _require(cond, ok, bad, issue: str, issues: list[str]) -> None:
|
||||
|
||||
def _check_daytona_backend(issues: list[str]) -> None:
|
||||
_require(os.getenv("DAYTONA_API_KEY"), ("Daytona API key", "(configured)"),
|
||||
("DAYTONA_API_KEY not set", "(required for TERMINAL_ENV=daytona)"), "Set DAYTONA_API_KEY environment variable", issues)
|
||||
("Daytona API key missing", "(needed for the 'daytona' terminal backend)"),
|
||||
"run `hermes setup terminal` (Daytona) to enter it.", issues)
|
||||
try:
|
||||
from daytona import Daytona # noqa: F401 — SDK presence check
|
||||
check_ok("daytona SDK", "(installed)")
|
||||
|
||||
@@ -28,20 +28,6 @@ _SCOPED_SKIP_LOGGED: set[str] = set() # routed profile homes whose multiplex d
|
||||
# env-var name → source label ("bitwarden", …) for externally injected credentials; setup / `hermes
|
||||
# model` tell users WHERE a key came from when .env lacks it.
|
||||
_SECRET_SOURCES: dict[str, str] = {}
|
||||
# Every env-var name an external source SUPPLIED for some home, whether it was applied or lost to a
|
||||
# pre-existing process value (``skipped_existing``). ``_SECRET_SOURCES`` is provenance metadata and only
|
||||
# names applied values; the scrub that keeps a launch profile's source-supplied names out of a routed
|
||||
# child must see the skipped ones too, or a name already in the process env leaks with the launch value.
|
||||
_SOURCE_SUPPLIED_NAMES: set[str] = set()
|
||||
# Every KEY name a dotenv file loaded into ``os.environ`` during this process's lifetime. A key removed
|
||||
# or renamed in the launch ``.env`` after boot stays in ``os.environ`` (dotenv never unsets), but a
|
||||
# re-parse of the current file no longer names it — so the launch-residue strip for a routed child must
|
||||
# work from what was LOADED, not from what the file says now. Additive for the process lifetime.
|
||||
_LOADED_DOTENV_KEYS: set[str] = set()
|
||||
# KEY names loaded from the administrator-managed ``.env`` (``_apply_managed_env``). Kept OUT of the launch
|
||||
# residue: those values are policy that beats the user's own ``.env`` for every profile, so a routed child
|
||||
# must keep them — and keep them LAST, over the routed profile's scope (review on f5f88d5058).
|
||||
_MANAGED_DOTENV_KEYS: set[str] = set()
|
||||
# Immutable per-home snapshots: os.environ is shared across profiles and a later home's apply may overwrite it.
|
||||
_SECRET_SOURCE_VALUES_BY_HOME: dict[str, dict[str, str]] = {}
|
||||
# HERMES_HOME paths already pulled external secrets for: load_hermes_dotenv() runs at import time from
|
||||
@@ -89,43 +75,12 @@ def get_secret_source(env_var: str) -> str | None:
|
||||
return _SECRET_SOURCES.get(env_var)
|
||||
|
||||
|
||||
def _record_supplied_names(report) -> set[str]:
|
||||
"""Every name *report*'s sources supplied — applied, or skipped because a value already existed —
|
||||
recorded into ``_SOURCE_SUPPLIED_NAMES`` so the routed-child scrub knows the source owns it."""
|
||||
supplied = set(report.provenance)
|
||||
for src in report.sources:
|
||||
supplied.update(src.skipped_existing)
|
||||
_SOURCE_SUPPLIED_NAMES.update(supplied)
|
||||
return supplied
|
||||
|
||||
|
||||
def secret_source_names() -> tuple[str, ...]:
|
||||
"""Every env-var name some profile's external secret source APPLIED (names only — the map is
|
||||
process-wide, so a value must be resolved through the active profile's secret scope). Consumers that
|
||||
forward source values into a child (MCP stdio env) want exactly these; see ``source_supplied_names``
|
||||
for the wider set the routed-child scrub needs."""
|
||||
"""Every env-var name some profile's external secret source supplied (names only — the map is
|
||||
process-wide, so a value must be resolved through the active profile's secret scope)."""
|
||||
return tuple(_SECRET_SOURCES)
|
||||
|
||||
|
||||
def source_supplied_names() -> tuple[str, ...]:
|
||||
"""Every env-var name an external source supplied for any home — applied, or lost to a pre-existing
|
||||
process value (``skipped_existing``). The launch value in ``os.environ`` is still not a routed
|
||||
profile's to inherit, so the strip must see the skipped names too."""
|
||||
return tuple(sorted(set(_SECRET_SOURCES) | _SOURCE_SUPPLIED_NAMES))
|
||||
|
||||
|
||||
def launch_dotenv_keys() -> frozenset[str]:
|
||||
"""KEY names any NON-managed dotenv file loaded into this process's ``os.environ`` so far (see
|
||||
``_LOADED_DOTENV_KEYS``); the launch profile's residue set for routed children."""
|
||||
return frozenset(_LOADED_DOTENV_KEYS)
|
||||
|
||||
|
||||
def managed_dotenv_keys() -> frozenset[str]:
|
||||
"""KEY names the administrator-managed ``.env`` loaded (see ``_MANAGED_DOTENV_KEYS``). Policy for
|
||||
every profile: never stripped from a routed child, and re-applied over the routed scope."""
|
||||
return frozenset(_MANAGED_DOTENV_KEYS)
|
||||
|
||||
|
||||
def get_secret_source_values(hermes_home: str | os.PathLike) -> dict[str, str]:
|
||||
"""Return the external-secret value snapshot for ``hermes_home``."""
|
||||
return dict(_SECRET_SOURCE_VALUES_BY_HOME.get(str(Path(hermes_home).resolve()), {}))
|
||||
@@ -184,10 +139,6 @@ def _hydrate_profile_secret_sources(home: Path) -> dict[str, str]:
|
||||
# mixed report are still snapshotted below and can be used while the failed source recovers.
|
||||
if all(src.result.ok for src in report.sources):
|
||||
_APPLIED_HOMES.add(home_key)
|
||||
# Same ownership bookkeeping as the process-global path: a name this profile's source supplied — applied,
|
||||
# or skipped because the private mapping already had it — is a source-owned name the routed-child scrub
|
||||
# must know about, or a sibling still inherits the launch value for it (review on f5f88d5058).
|
||||
_record_supplied_names(report)
|
||||
values: dict[str, str] = {}
|
||||
for name, applied in report.provenance.items():
|
||||
value = local_env.get(name)
|
||||
@@ -209,7 +160,6 @@ def reset_secret_source_cache(hermes_home: str | os.PathLike | None = None) -> N
|
||||
if hermes_home is None:
|
||||
_APPLIED_HOMES.clear()
|
||||
_SECRET_SOURCES.clear()
|
||||
_SOURCE_SUPPLIED_NAMES.clear()
|
||||
_SECRET_SOURCE_VALUES_BY_HOME.clear()
|
||||
return
|
||||
home_key = str(Path(hermes_home).resolve())
|
||||
@@ -284,7 +234,7 @@ def _sanitize_loaded_credentials() -> None:
|
||||
)
|
||||
|
||||
|
||||
def _load_dotenv_with_fallback(path: Path, *, override: bool, managed: bool = False) -> None:
|
||||
def _load_dotenv_with_fallback(path: Path, *, override: bool) -> None:
|
||||
try:
|
||||
# utf-8-sig strips a leading BOM (PowerShell 5.1 / Notepad); plain utf-8 would keep U+FEFF on the
|
||||
# first key name and silently drop it from os.environ under its canonical name.
|
||||
@@ -294,9 +244,6 @@ def _load_dotenv_with_fallback(path: Path, *, override: bool, managed: bool = Fa
|
||||
if raw.startswith(codecs.BOM_UTF8):
|
||||
raw = raw[len(codecs.BOM_UTF8) :]
|
||||
load_dotenv(stream=io.StringIO(raw.decode("latin-1")), override=override)
|
||||
# Same scanner both branches: it re-reads the file with the same latin-1 fallback. Managed keys are
|
||||
# recorded separately: they are administrator policy, not launch-profile residue.
|
||||
(_MANAGED_DOTENV_KEYS if managed else _LOADED_DOTENV_KEYS).update(_env_keys_defined_in_dotenv(path))
|
||||
_sanitize_loaded_credentials() # httpx encodes headers as ASCII
|
||||
|
||||
|
||||
@@ -387,8 +334,6 @@ def load_hermes_dotenv(
|
||||
# Multiplex gateway: while a routed profile-home override is active, copying that profile's .env
|
||||
# into os.environ would expose its credentials to sibling turns and every spawned child. Unscoped
|
||||
# startup loads keep the normal path; external sources still refresh against the profile mapping.
|
||||
# (``is_multiplex_active()`` is also true, context-locally, for a routed cron fire in the desktop
|
||||
# backend — see ``cron.scheduler_provider._profile_cron_scope``.)
|
||||
from agent.secret_scope import is_multiplex_active
|
||||
from hermes_constants import get_hermes_home_override
|
||||
|
||||
@@ -490,7 +435,7 @@ def _apply_managed_env() -> None:
|
||||
if not managed_env.exists():
|
||||
return
|
||||
_sanitize_env_file_if_needed(managed_env)
|
||||
_load_dotenv_with_fallback(managed_env, override=True, managed=True)
|
||||
_load_dotenv_with_fallback(managed_env, override=True)
|
||||
|
||||
|
||||
def _apply_external_secret_sources(home_path: Path) -> None:
|
||||
@@ -551,7 +496,9 @@ def _apply_external_secret_sources(home_path: Path) -> None:
|
||||
# ``EnvironmentFile=``. Under multiplex the scope is the only credential source, so an empty
|
||||
# snapshot failed every default-profile turn for the process lifetime (#102041).
|
||||
values: dict[str, str] = {}
|
||||
supplied = _record_supplied_names(report)
|
||||
supplied = set(report.provenance)
|
||||
for src in report.sources:
|
||||
supplied.update(src.skipped_existing)
|
||||
for name in supplied:
|
||||
if name in os.environ:
|
||||
values[name] = os.environ[name]
|
||||
|
||||
+23
-7
@@ -2289,7 +2289,8 @@ def _run_systemctl(args: list[str], *, system: bool = False, **kwargs) -> subpro
|
||||
try:
|
||||
return subprocess.run(_systemctl_cmd(system) + args, **kwargs)
|
||||
except FileNotFoundError:
|
||||
raise RuntimeError("systemctl is not available on this system") from None
|
||||
from hermes_cli.gateway_command_errors import SystemctlUnavailableError
|
||||
raise SystemctlUnavailableError() from None
|
||||
|
||||
|
||||
def _service_scope_label(system: bool = False) -> str:
|
||||
@@ -4586,10 +4587,10 @@ def _guard_existing_gateway_process_conflict(replace: bool = False) -> None:
|
||||
pass
|
||||
return
|
||||
|
||||
print_error(f"Another gateway instance is already running (PID {pid}).")
|
||||
print(" Use 'hermes gateway restart' to replace it,")
|
||||
print(" or 'hermes gateway stop' first.")
|
||||
print(" Or use 'hermes gateway run --replace' to auto-replace.")
|
||||
print_error(f"A gateway is already running (PID {pid}), so your bots are most likely online already.")
|
||||
print(" Check with `hermes gateway status`.")
|
||||
print(" To restart it: `hermes gateway restart`. To stop it: `hermes gateway stop`.")
|
||||
print(" To replace it from here: `hermes gateway run --replace`.")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
@@ -5899,6 +5900,15 @@ def gateway_command(args):
|
||||
# System-scope action typed without sudo; the wizard intercepts this earlier with guidance.
|
||||
print(str(e))
|
||||
sys.exit(1)
|
||||
except (subprocess.CalledProcessError, RuntimeError) as e:
|
||||
# systemctl exited non-zero or is missing entirely: guidance, not a traceback.
|
||||
from hermes_cli.gateway_command_errors import explain_service_failure
|
||||
lines = explain_service_failure(e)
|
||||
if lines is None:
|
||||
raise
|
||||
print_error(lines[0])
|
||||
_print_indented("\n".join(lines[1:]))
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _maybe_redirect_run_to_s6_supervision(args) -> bool:
|
||||
@@ -6079,7 +6089,10 @@ _NO_BACKEND_MESSAGES = {
|
||||
"Service uninstall is not applicable inside a Docker container.",
|
||||
"To stop the gateway, stop or remove the container:", "",
|
||||
" docker stop <container>", " docker rm <container>"),
|
||||
("uninstall", "unsupported"): (1, "Not supported on this platform."),
|
||||
("uninstall", "unsupported"): (1,
|
||||
"Running the gateway as a background service is not available on this platform "
|
||||
"(no systemd, launchd or Scheduled Tasks), so there is nothing to uninstall.",
|
||||
"Stop a manually started gateway with: hermes gateway stop"),
|
||||
("start", "termux"): (1,
|
||||
"Gateway service start is not supported on Termux because there is no system service manager.",
|
||||
"Run manually: hermes gateway"),
|
||||
@@ -6093,7 +6106,10 @@ _NO_BACKEND_MESSAGES = {
|
||||
" docker start <container> # start a stopped container",
|
||||
" docker restart <container> # restart a running container", "",
|
||||
"Or run the gateway directly: hermes gateway run"),
|
||||
("start", "unsupported"): (1, "Not supported on this platform."),
|
||||
("start", "unsupported"): (1,
|
||||
"Running the gateway as a background service is not available on this platform "
|
||||
"(no systemd, launchd or Scheduled Tasks).",
|
||||
"Run it directly with: hermes gateway run"),
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
"""User-facing copy for ``hermes gateway start/stop/restart`` failures on systemd hosts.
|
||||
|
||||
``hermes_cli/gateway.py`` is a facade; this sibling owns the small exception -> guidance table so
|
||||
the most common Linux service failures (``systemctl`` exited non-zero, or there is no ``systemctl``
|
||||
at all) end as a next step instead of a traceback.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
|
||||
|
||||
class SystemctlUnavailableError(RuntimeError):
|
||||
"""``systemctl`` is not installed (Alpine, minimal containers, some WSL setups)."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__("systemctl is not available on this system")
|
||||
|
||||
|
||||
_JOURNAL_HINT = 'journalctl --user -u hermes-gateway --since "5 min ago"'
|
||||
|
||||
_SYSTEMCTL_FAILED_LINES = (
|
||||
"Could not {verb} the gateway service; systemd reported an error.",
|
||||
"See why with `hermes gateway status --deep` or `{journal}`.",
|
||||
"To reinstall the service run `hermes gateway install --force`.",
|
||||
)
|
||||
|
||||
_NO_SYSTEMCTL_LINES = (
|
||||
"This system has no systemd, so Hermes cannot install a background service here.",
|
||||
"Run the gateway directly with `hermes gateway run` (keep it alive with tmux or screen).",
|
||||
)
|
||||
|
||||
|
||||
def _verb_for(exc: subprocess.CalledProcessError) -> str:
|
||||
cmd = exc.cmd if isinstance(exc.cmd, (list, tuple)) else str(exc.cmd).split()
|
||||
for token in cmd:
|
||||
if token in ("start", "stop", "restart"):
|
||||
return token
|
||||
return "start"
|
||||
|
||||
|
||||
def explain_service_failure(exc: BaseException) -> list[str] | None:
|
||||
"""Lines to print for a systemd service failure escaping the gateway command, or None
|
||||
when *exc* is not one this module knows how to explain (callers re-raise)."""
|
||||
if isinstance(exc, SystemctlUnavailableError):
|
||||
return list(_NO_SYSTEMCTL_LINES)
|
||||
if isinstance(exc, subprocess.CalledProcessError):
|
||||
lines = [line.format(verb=_verb_for(exc), journal=_JOURNAL_HINT) for line in _SYSTEMCTL_FAILED_LINES]
|
||||
lines.append(f"Details: {exc}")
|
||||
return lines
|
||||
return None
|
||||
+55
-15
@@ -416,32 +416,68 @@ def _inside_mcp_add_args(argv: list, index: int) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def _looks_like_hermes_invocation() -> bool:
|
||||
"""False when ``sys.argv`` belongs to a test runner rather than a ``hermes`` run.
|
||||
|
||||
pytest's own ``-p no:xdist`` reaches ``_scan_profile_flag`` through ``sys.argv`` at import
|
||||
time; it must stay a silent skip, while a real ``hermes -p 'Work Bot'`` must fail loudly.
|
||||
"""
|
||||
return "pytest" not in (sys.argv[0] or "")
|
||||
|
||||
|
||||
def _exit_invalid_profile_name(value: str) -> None:
|
||||
from hermes_cli.profiles import _invalid_profile_name_error
|
||||
|
||||
print(f"Error: {_invalid_profile_name_error(value)}", file=sys.stderr)
|
||||
print("Run `hermes profile list` to see your profiles.", file=sys.stderr)
|
||||
sys.exit(2)
|
||||
|
||||
|
||||
def _looks_like_option_value(value: str) -> bool:
|
||||
"""A ``-p`` value that clearly belongs to some other tool (pytest's ``-p no:xdist``, a
|
||||
third-party ``-p --flag``), never a mistyped profile name."""
|
||||
return value.startswith("-") or ":" in value
|
||||
|
||||
|
||||
def _scan_profile_flag(argv: list) -> tuple:
|
||||
"""Find -p/--profile/--profile= in argv -> (name, tokens_consumed, index).
|
||||
|
||||
Historically the flag worked even after the subcommand (`hermes chat -p
|
||||
coder`), so scan broadly; stop at ``--`` and at the `mcp add --args`
|
||||
passthrough region. Values that can't be profile names (pytest's
|
||||
``-p no:xdist``) are rejected so resolve_profile_env never sys.exits on them.
|
||||
passthrough region. The value is normalised (strip + casefold, matching
|
||||
``profiles.normalize_profile_name``) before validation so ``-p Work`` selects
|
||||
``work``. A value that cannot be a profile name is rejected so
|
||||
resolve_profile_env never sys.exits on it; the rejection is explained (exit 2)
|
||||
only when the flag comes BEFORE the first subcommand token under a real
|
||||
``hermes`` run — after a subcommand, ``-p`` may belong to that subcommand or a
|
||||
plugin (`hermes kanban ... -p 8080`), and option-looking values (``no:xdist``,
|
||||
``--flag``) are always a silent skip.
|
||||
"""
|
||||
from hermes_cli._parser import top_level_value_flag_sets
|
||||
|
||||
value_flags, optional_value_flags = top_level_value_flag_sets()
|
||||
i = 0
|
||||
saw_subcommand = False
|
||||
while i < len(argv):
|
||||
arg = argv[i]
|
||||
if arg == "--" or (arg == "--args" and _inside_mcp_add_args(argv, i)):
|
||||
break
|
||||
if arg in {"--profile", "-p"} and i + 1 < len(argv):
|
||||
if re.match(_PROFILE_NAME_RE, argv[i + 1]):
|
||||
return argv[i + 1], 2, i
|
||||
raw = argv[i + 1]
|
||||
value = raw.strip().casefold()
|
||||
if re.match(_PROFILE_NAME_RE, value):
|
||||
return value, 2, i
|
||||
if not saw_subcommand and not _looks_like_option_value(raw) and _looks_like_hermes_invocation():
|
||||
_exit_invalid_profile_name(raw)
|
||||
break
|
||||
if arg.startswith("--profile="):
|
||||
return arg.split("=", 1)[1], 1, i
|
||||
return arg.split("=", 1)[1].strip().casefold(), 1, i
|
||||
takes_value = "=" not in arg and i + 1 < len(argv) and (
|
||||
arg in value_flags
|
||||
or (arg in optional_value_flags and not argv[i + 1].startswith("-"))
|
||||
)
|
||||
if not takes_value and not arg.startswith("-"):
|
||||
saw_subcommand = True
|
||||
i += 2 if takes_value else 1
|
||||
return None, 0, None
|
||||
|
||||
@@ -1421,7 +1457,11 @@ def _resolve_continue_arg(args, *, use_tui: bool) -> None:
|
||||
args.resume = last_id
|
||||
else:
|
||||
kind = "TUI" if use_tui else "CLI"
|
||||
print(f"No previous {kind} session found to continue.")
|
||||
print(
|
||||
f"No previous {kind} session to continue. Start a new one with "
|
||||
"`hermes`, or list sessions with `hermes sessions list`.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
@@ -1919,7 +1959,11 @@ def _resolve_active_provider(config, model_cfg, effective_provider, custom_provi
|
||||
try:
|
||||
active = resolve_provider("auto")
|
||||
except AuthError as exc:
|
||||
if effective_provider == "auto":
|
||||
if exc.code == "no_provider_configured":
|
||||
# The picker that is about to open IS the fix; a warning that says
|
||||
# "run `hermes model`" from inside `hermes model` is circular.
|
||||
print("No provider is set up yet — pick one below. (Nous Portal works without an API key.)")
|
||||
elif effective_provider == "auto":
|
||||
print(f"Warning: {format_auth_error(exc)} Falling back to auto provider detection.")
|
||||
active = None # no provider yet; default to first in list
|
||||
|
||||
@@ -2466,14 +2510,10 @@ def _dashboard_prepare_runtime(args, headless_backend) -> bool:
|
||||
import fastapi # noqa: F401
|
||||
import uvicorn # noqa: F401
|
||||
except ImportError as e:
|
||||
print("Web UI dependencies not installed (need fastapi + uvicorn).")
|
||||
print(
|
||||
f"Re-install the package into this interpreter so metadata updates apply:\n"
|
||||
f" cd {PROJECT_ROOT}\n"
|
||||
f" {sys.executable} -m pip install -e .\n"
|
||||
"If `pip` is missing in this venv, use: uv pip install -e ."
|
||||
)
|
||||
print(f"Import error: {e}")
|
||||
from hermes_cli.main_dep_hints import missing_optional_deps_message
|
||||
|
||||
print(missing_optional_deps_message("dashboard", "its web-server packages (fastapi, uvicorn)", "all"))
|
||||
print(f"Details: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
# Seed bundled skills on first dashboard launch so the desktop GUI's
|
||||
|
||||
@@ -81,9 +81,11 @@ def cmd_acp(args):
|
||||
try:
|
||||
from acp_adapter.entry import main as acp_main
|
||||
acp_main([flag for attr, flag in _ACP_FLAGS if getattr(args, attr, False)])
|
||||
except ImportError:
|
||||
print("ACP dependencies not installed.", file=sys.stderr)
|
||||
print("Install them with: pip install -e '.[acp]'", file=sys.stderr)
|
||||
except ImportError as e:
|
||||
from hermes_cli.main_dep_hints import missing_optional_deps_message
|
||||
|
||||
print(missing_optional_deps_message("ACP server", "its protocol packages", "acp"), file=sys.stderr)
|
||||
print(f"Details: {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
"""Plain-language copy for surfaces that cannot start because an optional dependency group is missing.
|
||||
|
||||
``hermes dashboard`` (fastapi + uvicorn) and ``hermes acp`` (agent-client-protocol) are installed by
|
||||
the ``[all]`` / ``[acp]`` extras. A partial install, a pip-less uv venv or an interrupted update can
|
||||
leave them out; the built-in repair is ``hermes update``, which reinstalls the extras into the same
|
||||
interpreter. The manual fallback names the checkout directory and interpreter explicitly because a bare
|
||||
``pip install -e '.[acp]'`` fails on PEP 668 systems and in venvs without pip.
|
||||
"""
|
||||
|
||||
import sys
|
||||
|
||||
|
||||
def missing_optional_deps_message(surface: str, what: str, extra: str) -> str:
|
||||
"""``surface`` = "dashboard"/"ACP server"; ``what`` = "its web-server packages"; ``extra`` = "all"/"acp"."""
|
||||
from hermes_cli.main import PROJECT_ROOT
|
||||
|
||||
return (
|
||||
f"The {surface} can't start: {what} are missing from this install.\n"
|
||||
"Run `hermes update` to reinstall dependencies. If that fails, run manually:\n"
|
||||
f" cd {PROJECT_ROOT} && {sys.executable} -m pip install -e '.[{extra}]'\n"
|
||||
f" (no pip in this venv: uv pip install -e '.[{extra}]')"
|
||||
)
|
||||
@@ -454,7 +454,11 @@ def _tui_node_bin(bin: str) -> str:
|
||||
if ensure_dependency("node"):
|
||||
path = find_node_executable("node")
|
||||
if not path:
|
||||
print(f"{bin} not found — install Node.js to use the TUI.")
|
||||
print(
|
||||
f"Node.js is required for the TUI but `{bin}` was not found. Install it from "
|
||||
"https://nodejs.org (run `hermes doctor` for the install hint for your OS), then "
|
||||
"retry `hermes --tui`. To keep working now, run `hermes --cli`."
|
||||
)
|
||||
sys.exit(1)
|
||||
return path
|
||||
|
||||
|
||||
@@ -644,7 +644,8 @@ def cmd_mcp_add(args):
|
||||
try:
|
||||
tools = _probe_single_server(name, server_config)
|
||||
except Exception as exc:
|
||||
_error(f"Failed to connect: {redact_mcp_probe_text(exc)}")
|
||||
_error(f"Failed to connect: {_probe_failure_reason(exc)}")
|
||||
_info(_probe_failure_next_step(name, exc))
|
||||
if _confirm("Save config anyway (you can test later)?", default=False):
|
||||
server_config["enabled"] = False
|
||||
if _save_mcp_server(name, server_config):
|
||||
@@ -738,6 +739,25 @@ def cmd_mcp_list(args=None):
|
||||
print()
|
||||
|
||||
|
||||
def _probe_failure_reason(exc: BaseException) -> str:
|
||||
"""Plain reason for a failed probe: ``_format_connect_error`` unwraps ExceptionGroups and names a
|
||||
missing executable; the result is redacted like every other probe string."""
|
||||
from tools.mcp_tool_errors import _format_connect_error
|
||||
return redact_mcp_probe_text(_format_connect_error(exc))
|
||||
|
||||
|
||||
def _probe_failure_next_step(name: str, exc: BaseException) -> str:
|
||||
"""The one command that fixes the common probe failures (sign-in, missing command, everything else)."""
|
||||
from tools.mcp_tool_errors import _format_connect_error, _is_auth_error, _unwrap_exception_group
|
||||
root = _unwrap_exception_group(exc)
|
||||
if _is_auth_error(root) or getattr(getattr(root, "response", None), "status_code", None) in (401, 403):
|
||||
return f"The server rejected the sign-in. Run: hermes mcp login {name}"
|
||||
if "missing executable" in _format_connect_error(exc):
|
||||
return (f"Install that command, or set mcp_servers.{name}.command in {display_hermes_home()}/config.yaml "
|
||||
"to its full path.")
|
||||
return f"Check the server is running and the URL/command in its config, then run: hermes mcp test {name}"
|
||||
|
||||
|
||||
def cmd_mcp_test(args):
|
||||
"""Test connection to an MCP server."""
|
||||
name = args.name
|
||||
@@ -767,7 +787,9 @@ def cmd_mcp_test(args):
|
||||
try:
|
||||
tools = _probe_single_server(name, cfg)
|
||||
except Exception as exc:
|
||||
_error(f"Connection failed ({(time.monotonic() - start) * 1000:.0f}ms): {redact_mcp_probe_text(exc)}")
|
||||
elapsed = time.monotonic() - start
|
||||
_error(f"Connection failed ({elapsed:.1f}s): {_probe_failure_reason(exc)}")
|
||||
_info(_probe_failure_next_step(name, exc))
|
||||
return
|
||||
_success(f"Connected ({(time.monotonic() - start) * 1000:.0f}ms)")
|
||||
_success(f"Tools discovered: {len(tools)}")
|
||||
|
||||
@@ -10,6 +10,7 @@ from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import subprocess
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from hermes_cli.cli_output import line_input
|
||||
from hermes_cli.config import clear_model_endpoint_credentials
|
||||
@@ -156,16 +157,45 @@ def _pick_model_or_prompt(model_list, prompt: str, **kwargs):
|
||||
return _ask(prompt, cancel_msg=None)
|
||||
|
||||
|
||||
def _login_retry_context(args) -> tuple[str, str]:
|
||||
"""(retry_command, service_host) for a login helper's failure copy, from the ``ProviderConfig``
|
||||
among its positional args. Nous keeps ``hermes portal``; every other OAuth provider is retried
|
||||
with ``hermes auth add <provider>`` and named by its own portal host (``hermes login`` no longer
|
||||
exists). Falls back to ``hermes model`` when no provider config is in play."""
|
||||
pconfig = next((a for a in args if hasattr(a, "id") and hasattr(a, "portal_base_url")), None)
|
||||
if pconfig is None:
|
||||
return "hermes model", "the sign-in service"
|
||||
provider_id = str(getattr(pconfig, "id", "") or "")
|
||||
host = urlparse(str(getattr(pconfig, "portal_base_url", "") or "")).hostname or "the sign-in service"
|
||||
if provider_id == "nous":
|
||||
return "hermes portal", host
|
||||
return (f"hermes auth add {provider_id}" if provider_id else "hermes model"), host
|
||||
|
||||
|
||||
def _run_login(login_fn, *args, **kwargs) -> bool:
|
||||
"""Run an OAuth login helper; print the standard failure line and return False
|
||||
on SystemExit / any exception."""
|
||||
"""Run an OAuth login helper; print plain failure copy (what happened + retry command) and
|
||||
return False on SystemExit / any exception. The retry command and service host come from the
|
||||
provider config passed to the helper, so a MiniMax failure never says ``hermes portal`` /
|
||||
``portal.nousresearch.com``. Helpers that print their own copy raise ``SystemExit(1)`` with no
|
||||
message, which stays silent; a SystemExit that carries a message (or a non-cancel code from a
|
||||
helper that printed nothing) gets a one-line explanation so the user is never left with no
|
||||
output."""
|
||||
from hermes_cli.auth_error_copy import sign_in_failure_lines
|
||||
retry_command, service_host = _login_retry_context(args)
|
||||
try:
|
||||
login_fn(*args, **kwargs)
|
||||
except SystemExit:
|
||||
print("Login cancelled or failed.")
|
||||
except SystemExit as exc:
|
||||
if exc.code in (130, None, 0):
|
||||
print("Sign-in was cancelled.")
|
||||
elif isinstance(exc.code, str) and exc.code.strip():
|
||||
print(f"Sign-in did not complete: {exc.code.strip()}")
|
||||
print(f"Run `{retry_command}` to try again.")
|
||||
else:
|
||||
print(f"Sign-in did not complete; run `{retry_command}` to try again.")
|
||||
return False
|
||||
except Exception as exc:
|
||||
print(f"Login failed: {exc}")
|
||||
for line in sign_in_failure_lines(exc, service_host=service_host, retry_command=retry_command):
|
||||
print(line)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
@@ -1312,7 +1312,10 @@ def _creds_for_switched_provider(st: _Switch) -> Optional[ModelSwitchResult]:
|
||||
try:
|
||||
st.resolve_runtime(requested=st.target_provider)
|
||||
except Exception as e:
|
||||
return st.fail_on_target(f"Could not resolve credentials for provider '{st.provider_label}': {e}")
|
||||
return st.fail_on_target(
|
||||
f"{st.provider_label} is not connected: no API key or login was found for it. Add one with "
|
||||
f"`hermes auth add {st.target_provider}`, or pick a connected provider in /model.\n"
|
||||
f" Details: {e}")
|
||||
return None
|
||||
|
||||
|
||||
|
||||
+15
-37
@@ -1129,10 +1129,6 @@ class PluginManager(PluginLoaderMixin, PluginDispatchMixin, PluginLedgerMixin):
|
||||
self.home_path = Path(self.scope_key)
|
||||
self._discovery_lock = threading.RLock()
|
||||
self._discovered: bool = False
|
||||
# True once a discovery re-applied plugin secret sources for this home: the per-home snapshot and
|
||||
# the installed scope may then hold plugin-supplied names, and a later discovery that finds NO
|
||||
# enabled plugin source (plugin removed / disabled) must still reconcile once to drop them.
|
||||
self._plugin_secret_sources_reconciled: bool = False
|
||||
self._cli_ref = None # Set by CLI after plugin discovery
|
||||
self._gateway_message_injector: tuple[object, Callable] | None = None
|
||||
self._context_engine = None # Set by a plugin via register_context_engine()
|
||||
@@ -1272,34 +1268,24 @@ class PluginManager(PluginLoaderMixin, PluginDispatchMixin, PluginLedgerMixin):
|
||||
plugin_sources = list_plugin_sources()
|
||||
except Exception:
|
||||
return
|
||||
enabled_names: list[str] = []
|
||||
if plugin_sources:
|
||||
if not plugin_sources:
|
||||
return
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
secrets = (load_config() or {}).get("secrets") or {}
|
||||
except Exception:
|
||||
secrets = {}
|
||||
|
||||
def _enabled(source) -> bool:
|
||||
section = secrets.get(getattr(source, "name", ""))
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
secrets = (load_config() or {}).get("secrets") or {}
|
||||
return bool(source.is_enabled(section if isinstance(section, dict) else {}))
|
||||
except Exception:
|
||||
secrets = {}
|
||||
return False # mirrors the orchestrator: a raising is_enabled() is skipped
|
||||
|
||||
def _enabled(source) -> bool:
|
||||
section = secrets.get(getattr(source, "name", ""))
|
||||
try:
|
||||
return bool(source.is_enabled(section if isinstance(section, dict) else {}))
|
||||
except Exception:
|
||||
return False # mirrors the orchestrator: a raising is_enabled() is skipped
|
||||
|
||||
enabled_names = [getattr(s, "name", "") for s in plugin_sources if _enabled(s)]
|
||||
enabled_names = [getattr(s, "name", "") for s in plugin_sources if _enabled(s)]
|
||||
if not enabled_names:
|
||||
# Nothing enabled now. If an earlier discovery re-applied plugin sources for this home, the
|
||||
# snapshot and installed scope still carry that plugin's names (force-reload unloads the
|
||||
# registration first, so this is exactly the "last plugin source removed" path) — reconcile
|
||||
# once so they drop out. A home that never had one stays a no-op: no re-pull, no re-load.
|
||||
if not self._plugin_secret_sources_reconciled:
|
||||
return
|
||||
# The marker is cleared only AFTER the cleanup below succeeds: reset/reload/refresh are
|
||||
# fallible, and clearing first left the stale credential active with no retry on the next
|
||||
# discovery (review on f5f88d5058).
|
||||
else:
|
||||
self._plugin_secret_sources_reconciled = True
|
||||
return
|
||||
try:
|
||||
# Reset and reload the SAME home the process (or routed turn) resolves to: under multiplex this
|
||||
# runs at gateway boot after sibling profiles may already have hydrated, and a global clear
|
||||
@@ -1308,16 +1294,8 @@ class PluginManager(PluginLoaderMixin, PluginDispatchMixin, PluginLedgerMixin):
|
||||
home = get_hermes_home()
|
||||
reset_secret_source_cache(home)
|
||||
load_hermes_dotenv(hermes_home=home)
|
||||
# A scope installed for this home was frozen BEFORE these sources existed — a routed cron
|
||||
# fire builds its scope in run_one_job and only then, on its first agent build, discovers
|
||||
# plugins; under multiplex semantics the load above is hydrate-only, so fold the values
|
||||
# into the installed scope or THIS fire never sees the plugin credential.
|
||||
from agent.secret_scope import refresh_installed_secret_scope
|
||||
refresh_installed_secret_scope(Path(home))
|
||||
if not enabled_names:
|
||||
self._plugin_secret_sources_reconciled = False # cleanup succeeded; nothing left to drop
|
||||
logger.debug("Re-applied secret sources after plugin discovery for: %s",
|
||||
", ".join(sorted(enabled_names)) or "<none — reconciled removed plugin sources>")
|
||||
", ".join(sorted(enabled_names)))
|
||||
except Exception as exc:
|
||||
logger.debug("secret source re-apply after discovery failed: %s", exc)
|
||||
|
||||
|
||||
+26
-11
@@ -404,6 +404,18 @@ def _display_after_install(plugin_dir: Path, identifier: str) -> None:
|
||||
console.print()
|
||||
|
||||
|
||||
def _clone_failure_message(git_url: str, git_error: str) -> str:
|
||||
"""Plain-words clone failure: what to check (address, network, private repo), raw git text last.
|
||||
|
||||
The text reaches ``_fail`` -> Rich ``console.print``: escape the git output so ``[...]`` in it is
|
||||
not parsed as markup."""
|
||||
from rich.markup import escape
|
||||
return (f"Could not download the plugin from {git_url}. Check the address (browse the catalog "
|
||||
"with `hermes plugins search`), check your internet connection, or, if the repository "
|
||||
"is private, sign in first with `gh auth login` (or set GITHUB_TOKEN in your .env).\n"
|
||||
f"Details: {escape(git_error.strip())}")
|
||||
|
||||
|
||||
def _require_installed_plugin(name: str, plugins_dir: Path, console) -> Path:
|
||||
"""The plugin path if it exists; else exit 1 (invalid name, or a listing of installed plugins)."""
|
||||
try:
|
||||
@@ -411,11 +423,19 @@ def _require_installed_plugin(name: str, plugins_dir: Path, console) -> Path:
|
||||
except ValueError as e:
|
||||
_fail(console, f"[red]Error:[/red] {e}")
|
||||
if not target.exists():
|
||||
installed = ", ".join(d.name for d in plugins_dir.iterdir() if d.is_dir()) or "(none)"
|
||||
_fail(console, f"[red]Error:[/red] Plugin '{name}' not found in {plugins_dir}.\nInstalled plugins: {installed}")
|
||||
_fail(console, _unknown_plugin_message(name, downloaded_only=True))
|
||||
return target
|
||||
|
||||
|
||||
def _unknown_plugin_message(name: str, *, downloaded_only: bool = False) -> str:
|
||||
"""``No plugin named ...`` with the exact-name rule and the two commands that resolve it."""
|
||||
scope = (" This command only works on downloaded plugins; bundled ones can only be enabled or disabled."
|
||||
if downloaded_only else " Bundled plugins can only be enabled or disabled.")
|
||||
return (f"[red]No plugin named '{name}'.[/red] Run `hermes plugins list` to see the exact names "
|
||||
f"(nested plugins use their full key, e.g. web/firecrawl).{scope} "
|
||||
"To add one: `hermes plugins install <owner/repo>`.")
|
||||
|
||||
|
||||
# ── Install metadata + git plumbing ─────────────────────────────────────────────────────────
|
||||
|
||||
_EXACT_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$")
|
||||
@@ -569,12 +589,7 @@ def _clone_plugin_repo(tmp_clone: Path, git_url: str, revision: Optional[str]) -
|
||||
except subprocess.TimeoutExpired as e:
|
||||
raise PluginOperationError("Git clone timed out after 60 seconds.") from e
|
||||
if result.returncode != 0:
|
||||
error = _safe_git_error(result, git_url)
|
||||
if re.search(r"could not read Username|Authentication failed|Repository not found", error):
|
||||
error += (
|
||||
"\n\nIf this repository is private, authenticate first: run `gh auth login`, set GITHUB_TOKEN "
|
||||
"(or GH_TOKEN) in your .env, or store a credential in git's credential helper for this host.")
|
||||
raise PluginOperationError(f"Git clone failed:\n{error}")
|
||||
raise PluginOperationError(_clone_failure_message(git_url, _safe_git_error(result, git_url)))
|
||||
_scrub_cloned_origin(tmp_clone, git_exe, git_url)
|
||||
if revision:
|
||||
_checkout_exact_revision(tmp_clone, git_exe, revision, source_url=git_url)
|
||||
@@ -1006,7 +1021,7 @@ def cmd_enable(name: str, allow_tool_override: Optional[bool] = None) -> None:
|
||||
_refuse_legacy_relay(name)
|
||||
resolved = _resolve_plugin_key_and_source(name)
|
||||
if resolved is None:
|
||||
_fail(console, f"[red]Plugin '{name}' is not installed or bundled.[/red]")
|
||||
_fail(console, _unknown_plugin_message(name))
|
||||
key, source = resolved
|
||||
_refuse_legacy_relay(key)
|
||||
|
||||
@@ -1137,7 +1152,7 @@ def cmd_capabilities(name: Optional[str] = None) -> None:
|
||||
rows.append((key, entry[3], declared, granted, effective))
|
||||
|
||||
if name is not None and not rows:
|
||||
_fail(console, f"[red]Plugin '{name}' is not installed or bundled.[/red]")
|
||||
_fail(console, _unknown_plugin_message(name))
|
||||
if not rows:
|
||||
console.print("[dim]No plugins declare or hold capabilities.[/dim]")
|
||||
return
|
||||
@@ -1188,7 +1203,7 @@ def cmd_disable(name: str) -> None:
|
||||
console = _console()
|
||||
key = _resolve_plugin_key(name)
|
||||
if key is None:
|
||||
_fail(console, f"[red]Plugin '{name}' is not installed or bundled.[/red]")
|
||||
_fail(console, _unknown_plugin_message(name))
|
||||
enabled = _get_enabled_set()
|
||||
disabled = _get_disabled_set()
|
||||
if key not in enabled and key in disabled:
|
||||
|
||||
+41
-9
@@ -186,6 +186,38 @@ def _missing_profile_error(canon: str) -> FileNotFoundError:
|
||||
return FileNotFoundError(f"Profile '{canon}' does not exist. Create it with: hermes profile create {canon}")
|
||||
|
||||
|
||||
def _unknown_profile_error(canon: str) -> FileNotFoundError:
|
||||
"""For delete/rename/export of a name that matches no profile (likely a typo)."""
|
||||
return FileNotFoundError(f"No profile named '{canon}'. See your profiles with: hermes profile list")
|
||||
|
||||
|
||||
def _profile_exists_error(canon: str) -> FileExistsError:
|
||||
return FileExistsError(
|
||||
f"A profile named '{canon}' already exists. Switch to it with `hermes profile use {canon}`, "
|
||||
"see all profiles with `hermes profile list`, or choose a different name."
|
||||
)
|
||||
|
||||
|
||||
_PROFILE_NAME_RULE = (
|
||||
"Use lowercase letters, numbers, '-' or '_', starting with a letter or number, "
|
||||
"up to 64 characters"
|
||||
)
|
||||
|
||||
|
||||
def _suggest_profile_name(name: str) -> str:
|
||||
"""Best-effort valid id derived from *name* (``'My Work'`` -> ``'my-work'``); ``my-work`` if nothing usable."""
|
||||
candidate = re.sub(r"[^a-z0-9_-]+", "-", name.strip().lower()).strip("-_")[:64]
|
||||
return candidate if _PROFILE_ID_RE.match(candidate) else "my-work"
|
||||
|
||||
|
||||
def _invalid_profile_name_error(name: str) -> ValueError:
|
||||
suggestion = _suggest_profile_name(name)
|
||||
return ValueError(
|
||||
f"{name!r} is not a valid profile name. {_PROFILE_NAME_RULE} (for example: {suggestion}). "
|
||||
f"Then run `hermes profile create {suggestion}`."
|
||||
)
|
||||
|
||||
|
||||
# Validation
|
||||
|
||||
def normalize_profile_name(name: str) -> str:
|
||||
@@ -216,7 +248,7 @@ def validate_profile_name(name: str) -> None:
|
||||
if name == "default":
|
||||
return # special alias for ~/.hermes
|
||||
if not _PROFILE_ID_RE.match(name):
|
||||
raise ValueError(f"Invalid profile name {name!r}. Must match [a-z0-9][a-z0-9_-]{{0,63}}")
|
||||
raise _invalid_profile_name_error(name)
|
||||
if name in _RESERVED_NAMES:
|
||||
raise ValueError(
|
||||
f"Profile name {name!r} is reserved — it collides with either "
|
||||
@@ -229,7 +261,7 @@ def validate_alias_name(name: str) -> None:
|
||||
"""Raise ``ValueError`` unless *name* is a safe wrapper filename: it is used verbatim
|
||||
under ``~/.local/bin``, so ``../../.bashrc`` must never escape the wrapper dir."""
|
||||
if not _PROFILE_ID_RE.match(name):
|
||||
raise ValueError(f"Invalid alias name {name!r}. Must match [a-z0-9][a-z0-9_-]{{0,63}}")
|
||||
raise ValueError(f"Invalid alias name {name!r}. {_PROFILE_NAME_RULE}.")
|
||||
|
||||
|
||||
def _canon_valid(name: str) -> str:
|
||||
@@ -244,7 +276,7 @@ def _existing_profile_dir(name: str) -> Tuple[str, Path]:
|
||||
canon = _canon_valid(name)
|
||||
profile_dir = get_profile_dir(canon)
|
||||
if not profile_dir.is_dir():
|
||||
raise FileNotFoundError(f"Profile '{canon}' does not exist.")
|
||||
raise _unknown_profile_error(canon)
|
||||
return canon, profile_dir
|
||||
|
||||
|
||||
@@ -259,7 +291,7 @@ def get_profile_dir(name: str) -> Path:
|
||||
# regex only, not _RESERVED_NAMES: a pre-reserved-list dir like
|
||||
# profiles/hermes may still exist and must keep resolving.
|
||||
if not _PROFILE_ID_RE.match(canon):
|
||||
raise ValueError(f"Invalid profile name {canon!r}. Must match [a-z0-9][a-z0-9_-]{{0,63}}")
|
||||
raise _invalid_profile_name_error(canon)
|
||||
return _get_profiles_root() / canon
|
||||
|
||||
|
||||
@@ -880,10 +912,10 @@ def create_profile(
|
||||
# Empty shells left by post-delete mkdir may be replaced. Identity files mean the
|
||||
# leftover is not a shell — fail closed, no rmtree.
|
||||
if (profile_dir / "config.yaml").exists() or (profile_dir / ".env").exists():
|
||||
raise FileExistsError(f"Profile '{canon}' already exists at {profile_dir}")
|
||||
raise _profile_exists_error(canon)
|
||||
shutil.rmtree(profile_dir)
|
||||
if profile_dir.exists():
|
||||
raise FileExistsError(f"Profile '{canon}' already exists at {profile_dir}")
|
||||
raise _profile_exists_error(canon)
|
||||
source_dir = _resolve_clone_source(clone_from) if cloning else None
|
||||
if source_dir is not None and clone_channels:
|
||||
from hermes_cli.profile_channels import clone_channels_refusal
|
||||
@@ -1666,7 +1698,7 @@ def import_profile(archive_path: str, name: Optional[str] = None) -> Path:
|
||||
)
|
||||
profile_dir = get_profile_dir(canon)
|
||||
if profile_dir.exists():
|
||||
raise FileExistsError(f"Profile '{canon}' already exists at {profile_dir}")
|
||||
raise _profile_exists_error(canon)
|
||||
_get_profiles_root().mkdir(parents=True, exist_ok=True)
|
||||
with tempfile.TemporaryDirectory(prefix="hermes_profile_import_") as tmpdir:
|
||||
staging_root = Path(tmpdir)
|
||||
@@ -1749,9 +1781,9 @@ def rename_profile(old_name: str, new_name: str) -> Path:
|
||||
old_dir = get_profile_dir(old_canon)
|
||||
new_dir = get_profile_dir(new_canon)
|
||||
if not old_dir.is_dir():
|
||||
raise FileNotFoundError(f"Profile '{old_canon}' does not exist.")
|
||||
raise _unknown_profile_error(old_canon)
|
||||
if new_dir.exists():
|
||||
raise FileExistsError(f"Profile '{new_canon}' already exists.")
|
||||
raise _profile_exists_error(new_canon)
|
||||
|
||||
# 1. Stop gateway if running
|
||||
if _check_gateway_running(old_dir):
|
||||
|
||||
@@ -13,6 +13,8 @@ from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Optional
|
||||
|
||||
from hermes_constants import display_hermes_home
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -26,7 +28,9 @@ logger = logging.getLogger(__name__)
|
||||
@dataclass(frozen=True)
|
||||
class Advisory:
|
||||
"""``id`` is lowercase-hyphen, stable and never reused (it is what acks key on). ``remediation``
|
||||
is ordered: uninstall command first, then credential audit/rotation guidance.
|
||||
is ordered: uninstall command first, then credential audit/rotation guidance. Steps may use
|
||||
``{hermes_home}``; ``full_remediation_text`` fills it with ``display_hermes_home()`` at render
|
||||
time so a profile / HERMES_HOME user is sent to their own .env.
|
||||
"""
|
||||
|
||||
id: str
|
||||
@@ -59,7 +63,7 @@ ADVISORIES: tuple[Advisory, ...] = (
|
||||
compromised=(("mistralai", frozenset({"2.4.6"})),),
|
||||
remediation=(
|
||||
"Run: pip uninstall -y mistralai (or: uv pip uninstall mistralai)",
|
||||
"Rotate API keys in ~/.hermes/.env (OpenRouter, Anthropic, OpenAI, "
|
||||
"Rotate API keys in {hermes_home}/.env (OpenRouter, Anthropic, OpenAI, "
|
||||
"Nous, GitHub, AWS, Google, Mistral, etc.).",
|
||||
"Audit ~/.npmrc, ~/.pypirc, ~/.aws/credentials, ~/.config/gh/hosts.yml, "
|
||||
"and any other credential files for tokens that may have been read.",
|
||||
@@ -186,7 +190,7 @@ def full_remediation_text(hit: AdvisoryHit) -> list[str]:
|
||||
a.summary,
|
||||
"",
|
||||
"Remediation:",
|
||||
*(f" {i}. {step}" for i, step in enumerate(a.remediation, 1)),
|
||||
*(f" {i}. {step.format(hermes_home=display_hermes_home())}" for i, step in enumerate(a.remediation, 1)),
|
||||
]
|
||||
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ def _confirm_prompt(prompt: str) -> bool:
|
||||
|
||||
|
||||
def _not_found(session_id) -> int:
|
||||
print(f"Session '{session_id}' not found.")
|
||||
print(f"No session '{session_id}'. Run: hermes sessions list to find the id.")
|
||||
return 1
|
||||
|
||||
|
||||
@@ -988,7 +988,9 @@ def cmd_sessions(args, sessions_parser=None):
|
||||
# mode=ro cannot create the store; a reader on a fresh profile reports empty rather than failing.
|
||||
if observational and not _default_db_path().exists():
|
||||
return _print_empty_store(action, args)
|
||||
print(f"Error: Could not open session database: {e}")
|
||||
print("Could not open your session history database. "
|
||||
"Run: hermes sessions repair to fix it (a backup is made first).")
|
||||
print(f"Details: {e}")
|
||||
return 1
|
||||
try:
|
||||
handler = _DB_HANDLERS.get(action)
|
||||
|
||||
+5
-2
@@ -380,8 +380,11 @@ def setup_model_provider(config: dict, *, quick: bool = False):
|
||||
_info(None, "Provider setup skipped.")
|
||||
except Exception as exc:
|
||||
logger.debug("select_provider_and_model error during setup: %s", exc)
|
||||
print_warning(f"Provider setup encountered an error: {exc}")
|
||||
print_info("You can try again later with: hermes model")
|
||||
from hermes_cli.auth_error_copy import provider_setup_failure_lines
|
||||
lead, *rest = provider_setup_failure_lines(exc, retry_command="hermes model")
|
||||
print_warning(lead)
|
||||
for line in rest:
|
||||
print_info(line)
|
||||
|
||||
# Re-sync from disk in place: cmd_model saved via its own load/save cycle and the wizard's
|
||||
# final save_config(config) must not clobber it with stale values. Rotation, vision and TTS
|
||||
|
||||
@@ -64,9 +64,12 @@ def _run_portal_one_shot(config: dict) -> None:
|
||||
" Sign up: https://portal.nousresearch.com/manage-subscription", None)
|
||||
|
||||
def _on_error(exc: Exception) -> None:
|
||||
from hermes_cli.auth_error_copy import provider_setup_failure_lines
|
||||
print()
|
||||
print_error(f" Nous Portal setup encountered an error: {exc}")
|
||||
print_info(" You can retry later with `hermes portal`.")
|
||||
lead, *rest = provider_setup_failure_lines(exc, retry_command="hermes portal")
|
||||
print_error(f" {lead}")
|
||||
for line in rest:
|
||||
print_info(f" {line}")
|
||||
|
||||
if not _run_nous_flow(config, context="`hermes portal`", cancel_exc=(KeyboardInterrupt, EOFError, SystemExit),
|
||||
cancel_lines=(None, " Setup cancelled.", " You can retry later with `hermes portal`."),
|
||||
@@ -95,8 +98,11 @@ def _run_first_time_quick_setup(config: dict, hermes_home, is_existing: bool):
|
||||
"Sign up: https://portal.nousresearch.com/manage-subscription", None)
|
||||
|
||||
def _on_error(exc: Exception) -> None:
|
||||
print_warning(f"Nous Portal setup encountered an error: {exc}")
|
||||
print_info("You can try again later with: hermes model")
|
||||
from hermes_cli.auth_error_copy import provider_setup_failure_lines
|
||||
lead, *rest = provider_setup_failure_lines(exc, retry_command="hermes model")
|
||||
print_warning(lead)
|
||||
for line in rest:
|
||||
print_info(line)
|
||||
|
||||
_run_nous_flow(config, context="quick setup", cancel_exc=(KeyboardInterrupt, EOFError),
|
||||
cancel_lines=(None, "Nous Portal setup cancelled."), print_error=_on_error)
|
||||
|
||||
@@ -304,7 +304,7 @@ def _prompt_for_category(c: Console, existing: List[str]) -> str:
|
||||
c.print(f"[dim]Existing: {', '.join(existing)}[/]")
|
||||
else:
|
||||
c.print("[bold]Category[/] "
|
||||
"[dim](optional — press Enter to install flat at ~/.hermes/skills/<name>/)[/]")
|
||||
f"[dim](optional — press Enter to install flat at {display_hermes_home()}/skills/<name>/)[/]")
|
||||
answer = _line_input("Category: ")
|
||||
if answer and not _VALID_CATEGORY_RE.match(answer):
|
||||
c.print(f"[dim]Invalid category {answer!r} — installing flat.[/]")
|
||||
@@ -491,15 +491,31 @@ def inspect_skill(identifier: str) -> Optional[dict]:
|
||||
# --- install ---
|
||||
|
||||
def _install_blocked(c: Console, bundle, message: str, verdict: str, detail: str,
|
||||
q_path: Optional[Path] = None, lead: str = "") -> None:
|
||||
q_path: Optional[Path] = None, lead: str = "", label: str = "Installation blocked:") -> None:
|
||||
"""Print the blocked-install line, drop the quarantine copy, append the audit row."""
|
||||
c.print(f"{lead}[bold red]Installation blocked:[/] {message}")
|
||||
c.print(f"{lead}[bold red]{label}[/] {message}")
|
||||
if q_path is not None:
|
||||
shutil.rmtree(q_path, ignore_errors=True)
|
||||
from tools.skills_hub import append_audit_log
|
||||
append_audit_log("BLOCKED", bundle.name, bundle.source, bundle.trust_level, verdict, detail)
|
||||
|
||||
|
||||
def _scan_block_message(result, identifier: str) -> str:
|
||||
"""User-facing sentence for a scan-blocked install (the audit row keeps the scanner's raw reason).
|
||||
|
||||
Says what happened (not installed), why in plain words (high-risk patterns), whether ``--force``
|
||||
can help, and the read-only next step (``hermes skills inspect``). The hard-block rule mirrors
|
||||
``tools.skills_guard.should_allow_install``: a dangerous verdict on a non-official source."""
|
||||
n = len(result.findings)
|
||||
findings = f"{n} high-risk pattern(s)" if n else "high-risk patterns"
|
||||
hard_block = result.verdict == "dangerous" and result.trust_level in ("community", "trusted")
|
||||
policy = ("Hermes never installs unverified skills with high-risk findings, even with --force."
|
||||
if hard_block else "Re-run with --force to install anyway.")
|
||||
return (f"the security scan found {findings} in '{identifier}' (listed above). "
|
||||
f"{policy} Review the findings or ask the author to fix them; to read the skill without "
|
||||
f"installing, run `hermes skills inspect {identifier}`.")
|
||||
|
||||
|
||||
def _invalid_path(c: Console, bundle, exc: ValueError, q_path: Optional[Path] = None) -> None:
|
||||
_install_blocked(c, bundle, f"{exc}\n", "invalid_path", str(exc), q_path=q_path)
|
||||
|
||||
@@ -598,14 +614,15 @@ def _print_fetch_failure(c: Console, sources, identifier: str, meta=None, source
|
||||
c.print("[dim]Stale index entry: the skill was likely renamed or removed by "
|
||||
"its author. Try `hermes skills search` for an alternative.[/]\n")
|
||||
return
|
||||
c.print(f"[bold red]Error:[/] Could not fetch '{identifier}' from any source.")
|
||||
c.print(f"[bold red]Error:[/] Could not download '{identifier}'.")
|
||||
if rate_limited:
|
||||
c.print("[yellow]Hint:[/] GitHub API rate limit exhausted "
|
||||
"(unauthenticated: 60 requests/hour).\n"
|
||||
"Set [bold]GITHUB_TOKEN[/] in your .env or install the [bold]gh[/] CLI and run "
|
||||
"[bold]gh auth login[/] to raise the limit to 5,000/hr.\n")
|
||||
else:
|
||||
c.print()
|
||||
c.print(f"Check the name with [bold]hermes skills search {identifier.rsplit('/', 1)[-1]}[/] "
|
||||
"and check your internet connection. If it keeps failing, run [bold]hermes doctor[/].\n")
|
||||
|
||||
|
||||
def _scan_quarantined(c: Console, q_path: Path, bundle, meta, identifier: str):
|
||||
@@ -703,10 +720,10 @@ def do_install(identifier: str, category: str = "", force: bool = False,
|
||||
c.print(f"[dim]Quarantined to {q_path.relative_to(q_path.parent.parent.parent)}[/]")
|
||||
|
||||
result = _scan_quarantined(c, q_path, bundle, meta, identifier)
|
||||
allowed, reason = should_allow_install(result, force=force)
|
||||
allowed, _reason = should_allow_install(result, force=force)
|
||||
if not allowed:
|
||||
_install_blocked(c, bundle, reason, result.verdict, f"{len(result.findings)}_findings",
|
||||
q_path=q_path, lead="\n")
|
||||
_install_blocked(c, bundle, _scan_block_message(result, identifier), result.verdict,
|
||||
f"{len(result.findings)}_findings", q_path=q_path, lead="\n", label="Not installed:")
|
||||
return
|
||||
# Advisory second opinion — warn-and-continue by design (PII-class findings are
|
||||
# informational); the install confirmation below is where the user decides.
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Startup notice lines for toolsets that are switched off because their requirements are not met.
|
||||
|
||||
Pure rendering: ``cli.py::_show_tool_availability_warnings`` feeds it the ``unavailable`` list from
|
||||
``check_tool_availability()`` (entries carry ``name`` / ``env_vars`` / ``tools``) and prints the lines.
|
||||
Each line says what is off and the exact command that fixes it, in plain words; env-var names only
|
||||
appear as secondary detail for toolsets with a single obvious key.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Iterable, Optional
|
||||
|
||||
# Toolsets whose ``env_vars`` list is a multi-provider dump that means nothing to a user; render one
|
||||
# sentence per toolset instead. Provider names must exist under plugins/web/ (or be the Nous-managed row).
|
||||
_MULTI_PROVIDER_NOTICES: dict[str, str] = {
|
||||
"web": ("[yellow]⚠ Web search is off[/] — no search provider is set up yet (any one of Nous subscription, Exa, "
|
||||
"Tavily, Firecrawl, Brave, or free DuckDuckGo works). Run [bold]hermes setup tools[/] and set one up under "
|
||||
"\"Web Search & Scraping\"."),
|
||||
}
|
||||
|
||||
_GENERIC_FOOTER = "[dim] Run 'hermes setup tools' to configure[/]"
|
||||
|
||||
|
||||
def filter_to_enabled_toolsets(unavailable: list[dict], enabled: Iterable[str],
|
||||
resolve: Callable[[str], Iterable[str]]) -> list[dict]:
|
||||
"""Keep only the *unavailable* entries this session would actually load.
|
||||
|
||||
``enabled`` is the CLI's toolset selection as configured — on a default install that is a
|
||||
composite bundle such as ``["hermes-cli"]``, never the individual names ``check_tool_availability``
|
||||
reports — so each entry is expanded to tool names through ``resolve`` (``toolsets.resolve_toolset``)
|
||||
and an unavailable toolset counts as enabled when its name is listed directly or any of its tools
|
||||
is inside the expansion. An empty selection means "everything", so nothing is filtered."""
|
||||
names = [str(t) for t in (enabled or []) if str(t)]
|
||||
if not names:
|
||||
return list(unavailable)
|
||||
enabled_tools: set[str] = set()
|
||||
for name in names:
|
||||
try:
|
||||
enabled_tools.update(str(t) for t in (resolve(name) or ()))
|
||||
except Exception:
|
||||
continue
|
||||
name_set = set(names)
|
||||
|
||||
def _kept(item: dict) -> bool:
|
||||
if str(item.get("name") or "") in name_set:
|
||||
return True
|
||||
return any(str(t) in enabled_tools for t in (item.get("tools") or ()))
|
||||
|
||||
return [item for item in unavailable if _kept(item)]
|
||||
|
||||
|
||||
def current_terminal_backend() -> str:
|
||||
"""The terminal backend selected for this process (``terminal.backend`` / TERMINAL_ENV), e.g. 'docker'."""
|
||||
from tools.terminal_tool import _get_env_config
|
||||
return str(_get_env_config().get("env_type") or "local")
|
||||
|
||||
|
||||
def _terminal_line(backend: str, reason: Optional[str]) -> str:
|
||||
detail = f" ({reason})" if reason else ""
|
||||
return (f"[yellow]⚠ Terminal tool disabled:[/] the '{backend}' backend is not usable{detail}. "
|
||||
"Run [bold]hermes doctor[/] for details, or [bold]hermes setup terminal[/] to pick another backend.")
|
||||
|
||||
|
||||
def tool_availability_warning_lines(unavailable: list[dict], *, terminal_reason: Optional[str],
|
||||
terminal_backend: str = "local") -> list[str]:
|
||||
"""Rich-markup lines to print at CLI startup for *unavailable* toolsets; ``[]`` when there is nothing
|
||||
worth saying. ``terminal_reason`` is ``terminal_backend_unavailable_reason()`` (None when unknown)."""
|
||||
lines: list[str] = []
|
||||
generic: list[dict] = []
|
||||
for item in unavailable:
|
||||
name = str(item.get("name") or "")
|
||||
if name == "terminal":
|
||||
lines.append(_terminal_line(terminal_backend, terminal_reason))
|
||||
elif name in _MULTI_PROVIDER_NOTICES:
|
||||
lines.append(_MULTI_PROVIDER_NOTICES[name])
|
||||
elif item.get("env_vars"):
|
||||
generic.append(item)
|
||||
if generic:
|
||||
lines.append("[yellow]⚠️ Some tools disabled (missing API keys):[/]")
|
||||
lines.extend(f" [dim]• {item['name']}[/] [dim italic]({', '.join(item['env_vars'])})[/]" for item in generic)
|
||||
lines.append(_GENERIC_FOOTER)
|
||||
return lines
|
||||
@@ -1151,17 +1151,21 @@ def _handle_update_called_process_error(
|
||||
if gateway_mode:
|
||||
_write_gateway_update_exit_code(desktop_build_ok)
|
||||
else:
|
||||
print(f"✗ {stage}: {e}")
|
||||
_print_called_process_error_tail(e)
|
||||
if _called_process_error_is_python_dep_install(e):
|
||||
print(
|
||||
" The git update already finished. Re-downloading the source "
|
||||
"ZIP cannot fix a dependency install error and would overwrite local files.")
|
||||
print(f"✗ {stage} (the code update itself succeeded).")
|
||||
_print_called_process_error_tail(e)
|
||||
print()
|
||||
print(" Hermes may not start until the dependencies are installed. Fix the error above")
|
||||
print(" (usually network or disk space), then run `hermes update` again.")
|
||||
if _m()._is_windows():
|
||||
print(" Retry through the venv interpreter:")
|
||||
print(" If `hermes update` itself will not start, retry through the venv interpreter:")
|
||||
print(
|
||||
' venv\\Scripts\\python.exe -c '
|
||||
'"from hermes_cli.main import main; main()" update --yes')
|
||||
else:
|
||||
print(f"✗ {stage}.")
|
||||
print(f" Details: {e}")
|
||||
_print_called_process_error_tail(e)
|
||||
_finalize_receipt("failed", 'Update receipt finalize failed: %s')
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
@@ -46,7 +46,25 @@ _PRE_UPDATE_SNAPSHOT_KEEP = 1
|
||||
# small hard-to-regenerate state, not a multi-GB state.db (24 GB cost ~60s + 24 GB/update).
|
||||
_PRE_UPDATE_SNAPSHOT_MAX_FILE_SIZE = 1 << 30 # 1 GiB
|
||||
|
||||
_SQLITE_WAL_BUG_DETAIL = "SQLite {} still has the WAL-reset corruption bug"
|
||||
#: Reinstalling through the official installer swaps in a Python whose SQLite is safe; the
|
||||
#: one-liner differs per OS (mirrors ``uninstall._REINSTALL_HINT``). windows -> command
|
||||
_REINSTALL_ONE_LINER = {
|
||||
True: "iex (irm https://hermes-agent.nousresearch.com/install.ps1)",
|
||||
False: "curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash",
|
||||
}
|
||||
|
||||
|
||||
def _sqlite_partial_completion_lines(sqlite_version: str) -> list[str]:
|
||||
"""Shared ``⚠ Update partially complete`` wording for a vulnerable post-update SQLite, so the
|
||||
two completion banners cannot drift. The lead names the consequence, the second line the
|
||||
exact fix command."""
|
||||
from hermes_cli.update_cmd import _m
|
||||
return [
|
||||
f"⚠ Update partially complete — your Python's SQLite ({sqlite_version}) has a known "
|
||||
"corruption bug. Hermes works, but sessions could be damaged.",
|
||||
f" Fix: run the installer again ({_REINSTALL_ONE_LINER[bool(_m()._is_windows())]}) "
|
||||
"which installs a safe Python, then run `hermes doctor` to confirm.",
|
||||
]
|
||||
|
||||
|
||||
def _load_updates_cfg() -> dict:
|
||||
@@ -405,8 +423,8 @@ def _print_verified_update_completion(message: str) -> bool:
|
||||
_print_update_completion(message)
|
||||
return True
|
||||
print()
|
||||
print(f"⚠ Update partially complete — {_SQLITE_WAL_BUG_DETAIL.format(sqlite_info.sqlite_version_string)}.")
|
||||
print(" Rebuild the Hermes venv with a uv-managed Python, restart Hermes, then verify with `hermes doctor`.")
|
||||
for line in _sqlite_partial_completion_lines(sqlite_info.sqlite_version_string):
|
||||
print(line)
|
||||
return False
|
||||
|
||||
|
||||
@@ -440,21 +458,16 @@ def _print_update_summary(*, node_failures: list, desktop_build_ok: bool, pre_up
|
||||
parts.append(f"Node.js dependencies for {', '.join(node_failures)} did not refresh")
|
||||
if not desktop_build_ok:
|
||||
parts.append("the desktop app was not rebuilt and is still on the previous build")
|
||||
if not sqlite_runtime_ok and sqlite_info is not None:
|
||||
parts.append(_SQLITE_WAL_BUG_DETAIL.format(sqlite_info.sqlite_version_string))
|
||||
print("⚠ Update partially complete — " + "; ".join(parts) + ".")
|
||||
if parts:
|
||||
print("⚠ Update partially complete — " + "; ".join(parts) + ".")
|
||||
if node_failures:
|
||||
print(" Code and Python deps are updated, but the dashboard/TUI may")
|
||||
print(" be in a mixed state until the Node deps are rebuilt.")
|
||||
if not desktop_build_ok:
|
||||
print(" Run `hermes desktop` to retry the desktop rebuild.")
|
||||
if not sqlite_runtime_ok:
|
||||
print(
|
||||
" The Python runtime remediation did not complete. Run `hermes "
|
||||
"update` again; if SQLite is unchanged, rebuild the Hermes venv "
|
||||
"with a uv-managed Python, restart Hermes, then verify with "
|
||||
"`hermes doctor`."
|
||||
)
|
||||
for line in _sqlite_partial_completion_lines(sqlite_info.sqlite_version_string):
|
||||
print(line)
|
||||
else:
|
||||
_print_update_completion(_update_complete_message(pre_update_version))
|
||||
return desktop_build_ok and sqlite_runtime_ok
|
||||
|
||||
@@ -134,12 +134,12 @@ def describe_holder(holder: UpdateHolder) -> str:
|
||||
minutes, seconds = divmod(int(max(holder.age_seconds, 0)), 60)
|
||||
elapsed = f"{minutes}m {seconds}s" if minutes else f"{seconds}s"
|
||||
return (
|
||||
f"✗ Another Hermes update is already running (PID {holder.pid}, "
|
||||
f"started {elapsed} ago).\n"
|
||||
f"✗ Another Hermes update is already running (started {elapsed} ago, "
|
||||
f"process {holder.pid}).\n"
|
||||
"\n"
|
||||
" Two updates mutating the same checkout corrupt it: one rewrites\n"
|
||||
" source while the other is mid-install. Wait for it to finish, or\n"
|
||||
" close the window/dashboard tab that started it, then retry."
|
||||
" Running two at once would corrupt the install. Wait for it to finish\n"
|
||||
" (watch `hermes logs`), or close the Desktop/dashboard window that\n"
|
||||
" started it, then run `hermes update` again."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -379,13 +379,19 @@ def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
|
||||
print(_FLEET_ROW_LINES.get(entry.get("state"), _FLEET_ROW_UNKNOWN).format(
|
||||
profile=entry.get("profile"), pid=entry.get("pid"), short=sha[:8] if isinstance(sha, str) and sha else "?",
|
||||
))
|
||||
any_stale, any_down = "stale" in states, "down" in states
|
||||
if any_stale or any_down:
|
||||
stale_or_down = sum(1 for entry in fleet if entry.get("state") in ("stale", "down"))
|
||||
if stale_or_down:
|
||||
print()
|
||||
if any_stale:
|
||||
print(" ⚠ Stale gateways keep serving pre-update code until restarted:")
|
||||
if any_down:
|
||||
print(" ⚠ Down gateways stopped serving messaging entirely — restart them:")
|
||||
print(" hermes gateway restart # active profile")
|
||||
print(" hermes -p <profile> gateway restart # named profile")
|
||||
return any_stale or any_down
|
||||
if "stale" in states:
|
||||
print(" ⚠ Stale gateways keep serving pre-update code until restarted.")
|
||||
if "down" in states:
|
||||
print(" ⚠ Down gateways stopped serving messaging entirely.")
|
||||
# ``✓ Update complete!`` was already printed before the restart phase (the exit code
|
||||
# must land before a systemd restart can kill this process), so this verdict has to
|
||||
# supersede it explicitly — otherwise the output says success while the exit code is 1.
|
||||
print()
|
||||
print(
|
||||
f"✗ Update not complete: {stale_or_down} gateway(s) still running the old code (or stopped).")
|
||||
print(" Run `hermes gateway restart` (or `hermes -p <profile> gateway restart` for a named")
|
||||
print(" profile), then `hermes gateway status` to confirm.")
|
||||
return stale_or_down > 0
|
||||
|
||||
+44
-14
@@ -103,10 +103,11 @@ class SessionResumeTooLargeError(ValueError):
|
||||
self, message_count: int, limit: int = _MAX_SAFE_MESSAGES, scope: str = "across its lineage",
|
||||
):
|
||||
self.message_count, self.limit = message_count, limit
|
||||
self.scope = scope
|
||||
super().__init__(
|
||||
f"session has at least {message_count} active messages {scope}; "
|
||||
f"safe resume limit is {limit}. Export the session instead, or set "
|
||||
"sessions.max_resume_messages: 0 in config.yaml to disable the guard."
|
||||
f"This session is too long to reload safely ({message_count} messages; limit {limit}). "
|
||||
"Start a fresh chat and use `hermes sessions export` to keep a copy, or raise the limit "
|
||||
"with `hermes config set sessions.max_resume_messages 0`."
|
||||
)
|
||||
|
||||
|
||||
@@ -336,13 +337,43 @@ def _strip_stale_tool_call_markers(messages: List[Dict[str, Any]]) -> List[Dict[
|
||||
return messages
|
||||
|
||||
|
||||
def format_session_db_unavailable(prefix: str = "Session database not available") -> str:
|
||||
"""User-facing message with the captured init cause (+ WAL-docs hint for NFS/SMB locking failures)."""
|
||||
_SESSION_DB_CONSEQUENCE = "Sessions will not be saved until this is fixed."
|
||||
_NETWORK_DRIVE_HINT = " If the database lives on a network drive, move it to a local disk."
|
||||
_NETWORK_DRIVE_GLOSS = "the session database could not be opened; it may be on a network or unsupported drive"
|
||||
_NETWORK_DRIVE_ACTION = (
|
||||
"Move it to a local disk (`hermes doctor` shows where it is), then start Hermes again."
|
||||
)
|
||||
|
||||
|
||||
def format_session_db_unavailable(
|
||||
prefix: str = "Hermes can't open its session history right now",
|
||||
*,
|
||||
details: bool = False,
|
||||
) -> str:
|
||||
"""User-facing one-liner: ``<prefix>: <gloss>. <consequence> <action>[ network hint]``.
|
||||
|
||||
The cause table lives in ``hermes_state_user_copy`` so CLI, gateway and TUI agree. Chat
|
||||
surfaces (gateway, TUI) get the one-liner; ``details=True`` (the CLI banner) appends a
|
||||
``Details: <raw cause>`` line for the raw SQLite text. Network filesystems (NFS/SMB/FUSE/ZFS)
|
||||
cannot host SQLite's write-ahead log: when the raw cause carries one of those markers the
|
||||
message names the network-drive suspicion, because ``hermes doctor --fix`` cannot repair a
|
||||
mount — only moving the file can."""
|
||||
cause = get_last_init_error()
|
||||
if not cause:
|
||||
return f"{prefix}."
|
||||
hint = " (state.db may be on NFS/SMB/FUSE/ZFS — see https://www.sqlite.org/wal.html)"
|
||||
return f"{prefix}: {cause}{hint if any(m in cause.lower() for m in _WAL_INCOMPAT_MARKERS) else ''}."
|
||||
return f"{prefix}. {_SESSION_DB_CONSEQUENCE} Run `hermes doctor` to check the storage location."
|
||||
from hermes_state_user_copy import describe_storage_failure
|
||||
failure = describe_storage_failure(cause)
|
||||
gloss, action, hint = failure.gloss, failure.action, ""
|
||||
if any(m in cause.lower() for m in _WAL_INCOMPAT_MARKERS):
|
||||
if failure.cause == "unknown":
|
||||
gloss, action = _NETWORK_DRIVE_GLOSS, _NETWORK_DRIVE_ACTION
|
||||
else:
|
||||
hint = _NETWORK_DRIVE_HINT
|
||||
text = f"{prefix}: {gloss}. {_SESSION_DB_CONSEQUENCE} {action}{hint}"
|
||||
if details:
|
||||
from hermes_state_user_copy import storage_failure_details
|
||||
text += f"\nDetails: {storage_failure_details(cause)}"
|
||||
return text
|
||||
|
||||
|
||||
# Auto-repair at most once per DB path per process (no repair loops; serialises concurrent
|
||||
@@ -664,13 +695,12 @@ class SessionDB(
|
||||
except OSError:
|
||||
zsize = -1
|
||||
qpath = quarantine_invalid_state_db(self.db_path, already_locked=already_locked)
|
||||
where = f"moved aside to {qpath}" if qpath else "left in place (it could not be moved aside)"
|
||||
msg = (
|
||||
f"state.db has no SQLite header ({zsize} bytes). "
|
||||
f"Preserved at {qpath or '(quarantine failed — file left in place)'}. "
|
||||
f"Restore from {self.db_path.parent / 'state-snapshots'} via `hermes snapshot list` / "
|
||||
f"`hermes snapshot restore <id>` if available, or salvage the preserved bytes with "
|
||||
f"`hermes sessions recover --source {qpath or self.db_path}`. "
|
||||
"Opening a fresh empty database so the agent can start."
|
||||
f"state.db was empty or damaged ({zsize} bytes) and has been {where}; Hermes started with a "
|
||||
"fresh, empty session database. To bring old sessions back, run "
|
||||
f"`hermes sessions recover --source {qpath or self.db_path} --inspect-only`, or restore a "
|
||||
"snapshot with `/snapshot list` then `/snapshot restore <id>` (terminal `hermes` chat only)."
|
||||
)
|
||||
logger.error(msg)
|
||||
_set_last_init_error(msg)
|
||||
|
||||
@@ -690,8 +690,9 @@ def quarantine_invalid_state_db(path: Path, *, already_locked: bool = False) ->
|
||||
if not acquired:
|
||||
logger.error("quarantine lock for %s not acquired within 5s — refusing to "
|
||||
"quarantine without the cross-process lock. The invalid file "
|
||||
"is left in place. If sessions fail to load, restore from "
|
||||
"state-snapshots via `hermes snapshot list` / `hermes snapshot restore <id>`.",
|
||||
"is left in place. If sessions fail to load, run `hermes sessions recover "
|
||||
"--source <state.db> --inspect-only`, or restore a snapshot with "
|
||||
"`/snapshot list` / `/snapshot restore <id>` (terminal `hermes` chat only).",
|
||||
path)
|
||||
return None
|
||||
return _do_quarantine()
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Plain-language copy for "session storage is unavailable / could not be written" notices.
|
||||
|
||||
One table keyed by ``classify_persistence_error``'s cause bucket feeds every surface (CLI banner,
|
||||
gateway home-channel warning, TUI/Desktop RPC errors) so they agree on what happened, what to do,
|
||||
and a machine-readable ``code`` a GUI can attach a "Run doctor" button to.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from hermes_state_errors import classify_persistence_error, is_disk_full_error
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StorageFailure:
|
||||
cause: str # classify_persistence_error bucket
|
||||
code: str # machine-readable, stable: storage_locked | storage_readonly | storage_corrupt | disk_full | ...
|
||||
gloss: str # what happened, one clause, lowercase start
|
||||
action: str # what to do, one sentence naming the exact command
|
||||
|
||||
|
||||
_DOCTOR = "Run `hermes doctor --fix` to diagnose and repair."
|
||||
|
||||
# cause -> (code, gloss, action). "disk" is split by is_disk_full_error at lookup time.
|
||||
_STORAGE_FAILURES: dict[str, tuple[str, str, str]] = {
|
||||
"locked": (
|
||||
"storage_locked",
|
||||
"the session database is locked by another Hermes process",
|
||||
"Wait a moment and try again; if it persists, stop the other Hermes process (`hermes gateway stop`).",
|
||||
),
|
||||
"disk_full": (
|
||||
"disk_full",
|
||||
"the disk is full",
|
||||
"Free some disk space, then try again.",
|
||||
),
|
||||
"disk": (
|
||||
"storage_readonly",
|
||||
"the session database file is read-only or not writable",
|
||||
_DOCTOR,
|
||||
),
|
||||
"corrupt": (
|
||||
"storage_corrupt",
|
||||
"the session database file is damaged",
|
||||
f"{_DOCTOR} Recovery: `hermes sessions recover --source <state.db> --inspect-only`.",
|
||||
),
|
||||
"fts_index": (
|
||||
"storage_index_corrupt",
|
||||
"the session search index is damaged (the messages themselves are intact)",
|
||||
"Run `hermes doctor --fix` (or `hermes sessions repair`) to rebuild it.",
|
||||
),
|
||||
"replaced": (
|
||||
"storage_replaced",
|
||||
"the session database file was replaced while Hermes was running",
|
||||
"Stop Hermes (`hermes gateway stop`), run `hermes doctor`, then start it again.",
|
||||
),
|
||||
"deleted_wal": (
|
||||
"storage_replaced",
|
||||
"the session database file was changed or replaced while Hermes was running",
|
||||
"Stop Hermes (`hermes gateway stop`), run `hermes doctor`, then start it again.",
|
||||
),
|
||||
"compression": (
|
||||
"storage_busy",
|
||||
"another process is compressing this session",
|
||||
"Send your message again once compression finishes.",
|
||||
),
|
||||
"compression_closed": (
|
||||
"storage_session_rotated",
|
||||
"this session was rotated by context compression",
|
||||
"Refresh the client (or start a new turn) and send your message again.",
|
||||
),
|
||||
"turn_lease": (
|
||||
"storage_busy",
|
||||
"another Hermes process took over this session",
|
||||
"Wait for it to finish, then send your message again.",
|
||||
),
|
||||
"unknown": (
|
||||
"storage_unavailable",
|
||||
"the session database could not be opened",
|
||||
_DOCTOR,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def describe_storage_failure(exc_or_str) -> StorageFailure:
|
||||
"""Plain-language description of a persistence failure (never raises)."""
|
||||
cause = classify_persistence_error(exc_or_str)
|
||||
key = "disk_full" if cause == "disk" and is_disk_full_error(exc_or_str) else cause
|
||||
code, gloss, action = _STORAGE_FAILURES.get(key, _STORAGE_FAILURES["unknown"])
|
||||
return StorageFailure(cause=cause, code=code, gloss=gloss, action=action)
|
||||
|
||||
|
||||
def storage_failure_details(exc_or_str, limit: int = 200) -> str:
|
||||
"""Raw cause for a trailing, secondary "Details:" line (never the lead sentence)."""
|
||||
text = " ".join(str(exc_or_str or "").split())
|
||||
return text if len(text) <= limit else text[: limit - 3].rstrip() + "..."
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Keuse [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Tyd verstreke - opdrag word geweier"
|
||||
timeout: " ⏱ Geen antwoord binne {waited} — die opdrag is nie uitgevoer nie. Vra weer om te herprobeer, of verhoog die limiet: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Eenmalig toegelaat"
|
||||
allowed_session: " ✓ Vir hierdie sessie toegelaat"
|
||||
allowed_always: " ✓ By permanente toelaatlys gevoeg"
|
||||
|
||||
+1
-1
@@ -16,7 +16,7 @@ approval:
|
||||
prompt_smart_deny: " الاختيار [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ انتهت المهلة - يتم رفض الأمر"
|
||||
timeout: " ⏱ لا توجد إجابة خلال {waited} — لم يتم تنفيذ الأمر. اطلب مرة أخرى لإعادة المحاولة، أو ارفع الحد: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ مسموح مرة واحدة"
|
||||
allowed_session: " ✓ مسموح لهذه الجلسة"
|
||||
allowed_always: " ✓ أُضيف إلى قائمة السماح الدائمة"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Auswahl [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Zeitüberschreitung – Befehl wird abgelehnt"
|
||||
timeout: " ⏱ Keine Antwort innerhalb von {waited} — der Befehl wurde nicht ausgeführt. Erneut anfragen, um es noch einmal zu versuchen, oder das Limit erhöhen: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Einmalig erlaubt"
|
||||
allowed_session: " ✓ Für diese Sitzung erlaubt"
|
||||
allowed_always: " ✓ Zur dauerhaften Erlaubnisliste hinzugefügt"
|
||||
|
||||
+1
-1
@@ -21,7 +21,7 @@ approval:
|
||||
prompt_smart_deny: " Choice [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Timeout - denying command"
|
||||
timeout: " ⏱ No answer in {waited} — the command was not run. Ask again to retry, or raise the limit: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Allowed once"
|
||||
allowed_session: " ✓ Allowed for this session"
|
||||
allowed_always: " ✓ Added to permanent allowlist"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Opción [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Tiempo agotado — comando denegado"
|
||||
timeout: " ⏱ Sin respuesta en {waited} — el comando no se ejecutó. Pide de nuevo para reintentar, o sube el límite: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Permitido una vez"
|
||||
allowed_session: " ✓ Permitido en esta sesión"
|
||||
allowed_always: " ✓ Añadido a la lista de permitidos permanente"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Choix [o/R] : "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "r,refuser"
|
||||
timeout: " ⏱ Délai dépassé — commande refusée"
|
||||
timeout: " ⏱ Aucune réponse en {waited} — la commande n'a pas été exécutée. Redemandez pour réessayer, ou augmentez la limite : hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Autorisé une fois"
|
||||
allowed_session: " ✓ Autorisé pour cette session"
|
||||
allowed_always: " ✓ Ajouté à la liste d'autorisation permanente"
|
||||
|
||||
+1
-1
@@ -15,7 +15,7 @@ approval:
|
||||
prompt_smart_deny: " Rogha [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Am istigh — ag diúltú don ordú"
|
||||
timeout: " ⏱ Gan freagra laistigh de {waited} — níor ritheadh an t-ordú. Iarr arís le triail eile a bhaint as, nó ardaigh an teorainn: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Ceadaithe uair amháin"
|
||||
allowed_session: " ✓ Ceadaithe don seisiún seo"
|
||||
allowed_always: " ✓ Curtha leis an liosta ceadaithe buan"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Választás [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Időtúllépés - parancs elutasítva"
|
||||
timeout: " ⏱ Nem érkezett válasz {waited} alatt — a parancs nem futott le. Kérd újra a megismétléshez, vagy növeld a határt: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Egyszer engedélyezve"
|
||||
allowed_session: " ✓ Engedélyezve ehhez a munkamenethez"
|
||||
allowed_always: " ✓ Hozzáadva az állandó engedélylistához"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Scelta [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Tempo scaduto — comando negato"
|
||||
timeout: " ⏱ Nessuna risposta entro {waited} — il comando non è stato eseguito. Richiedi di nuovo per riprovare, oppure aumenta il limite: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Consentito una volta"
|
||||
allowed_session: " ✓ Consentito per questa sessione"
|
||||
allowed_always: " ✓ Aggiunto alla lista permessi permanente"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " 選択 [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ タイムアウト — コマンドを拒否しました"
|
||||
timeout: " ⏱ {waited} 以内に応答がなかったため、コマンドは実行されませんでした。もう一度依頼して再試行するか、制限を引き上げてください: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ 今回のみ許可"
|
||||
allowed_session: " ✓ このセッション中は許可"
|
||||
allowed_always: " ✓ 永続的な許可リストに追加"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " 선택 [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ 시간 초과 - 명령을 거부합니다"
|
||||
timeout: " ⏱ {waited} 동안 응답이 없어 명령을 실행하지 않았습니다. 다시 요청해 재시도하거나 제한을 늘리세요: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ 한 번 허용됨"
|
||||
allowed_session: " ✓ 이 세션에서 허용됨"
|
||||
allowed_always: " ✓ 영구 허용 목록에 추가됨"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Escolha [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Tempo esgotado — comando negado"
|
||||
timeout: " ⏱ Sem resposta em {waited} — o comando não foi executado. Peça novamente para tentar de novo, ou aumente o limite: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Permitido uma vez"
|
||||
allowed_session: " ✓ Permitido nesta sessão"
|
||||
allowed_always: " ✓ Adicionado à lista de permissões permanente"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ approval:
|
||||
prompt_smart_deny: " Выбор [o/D]: "
|
||||
smart_deny_once_inputs: "o,once"
|
||||
smart_deny_deny_inputs: "d,deny"
|
||||
timeout: " ⏱ Время ожидания истекло — команда отклонена"
|
||||
timeout: " ⏱ Нет ответа в течение {waited} — команда не была выполнена. Попросите снова, чтобы повторить, или увеличьте лимит: hermes config set approvals.timeout {suggested}"
|
||||
allowed_once: " ✓ Разрешено один раз"
|
||||
allowed_session: " ✓ Разрешено для этого сеанса"
|
||||
allowed_always: " ✓ Добавлено в постоянный список разрешённых"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user