From 083f8a60711d17eb713b7d3189dea249322f914c Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:33:22 +0530 Subject: [PATCH 001/748] =?UTF-8?q?feat(agent):=20unified=20deadline=20lay?= =?UTF-8?q?er=20=E2=80=94=20bounded=20execution=20primitive=20+=20timeout?= =?UTF-8?q?=20resolver=20(#85125=20Phase=201)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One shared foundation for the timeout/hang backlog instead of per-incident site-local fixes: - agent/deadline.py: run_bounded_async (thread-timer deadline that survives a blocked event loop, generalizing the telegram adapter primitive), run_bounded_sync, clamp_timeout (kills the #83220 time_t OverflowError class at the boundary), resolve_timeout (config.yaml timeouts: section > legacy env bridge > default), kill_process_tree (whole-tree termination for the #71148 orphan class), DeadlineExpired (our deadline, mechanically distinct from provider timeouts). - tool_executor._resolve_concurrent_tool_timeout migrates onto the resolver; exact legacy env-var contract preserved (default 420, 0 disables). - timeouts: accepted as a known config root; documented in cli-config.yaml.example. Pure addition otherwise — no behavior change, no new env vars, no cache impact. Later phases (#85125) migrate tool-execution, MCP, and subprocess call sites onto these primitives. --- agent/deadline.py | 471 +++++++++++++++++++++++++++++++++++ agent/tool_executor.py | 28 +-- cli-config.yaml.example | 16 ++ hermes_cli/config.py | 1 + tests/agent/test_deadline.py | 401 +++++++++++++++++++++++++++++ 5 files changed, 902 insertions(+), 15 deletions(-) create mode 100644 agent/deadline.py create mode 100644 tests/agent/test_deadline.py diff --git a/agent/deadline.py b/agent/deadline.py new file mode 100644 index 0000000000..c7aa5d28ca --- /dev/null +++ b/agent/deadline.py @@ -0,0 +1,471 @@ +"""Unified deadline layer — one bounded-execution primitive, one timeout resolver. + +Phase 1 of the architectural fix for the timeout/hang backlog +(https://github.com/NousResearch/hermes-agent/issues/85125). + +The tree currently carries at least six site-local deadline mechanisms, each +built for one incident, none shared (tool_executor batch deadline, telegram +``_await_with_thread_deadline``, gateway turn lease, reasoning stale floors, +``human_wait_ceiling``, per-MCP-handler timeouts). Every new stall report +grows that list by one. This module is the shared foundation the call sites +migrate onto in later phases: + +* :func:`resolve_timeout` — one config-first resolution path for timeout + values (``timeouts:`` section in config.yaml > legacy env var > default), + so new surfaces stop inventing ``HERMES_*_TIMEOUT`` env vars (".env is for + secrets only") and hardcoded literals stop ignoring user config + (#63302, #53161, #43272 class). + +* :func:`clamp_timeout` — platform-safe clamping. Large user-supplied + timeouts overflow ``time_t`` inside ``threading.Lock.acquire(timeout=...)`` + / ``Thread.join(timeout=...)`` on macOS and kill whole tool batches + (#83220). Clamping at the shared boundary fixes that class once, for + every consumer. + +* :func:`run_bounded_async` — a wall-clock deadline for awaitables that does + NOT depend on event-loop timers. ``asyncio.wait_for`` schedules its expiry + on the loop; when the loop thread itself is blocked in a synchronous call + (family A of the #84047 stall triage), every asyncio-based timeout in the + process is silently disabled. This helper drives the deadline from a + daemon ``threading.Timer`` (generalizing the proven telegram-adapter + primitive) and abandons cancellation-shielded tasks instead of waiting for + cancellation to complete. + +* :func:`run_bounded_sync` — the same contract for synchronous callables + bounded from a synchronous context (daemon worker thread, abandoned on + expiry). + +* :func:`kill_process_tree` — portable whole-tree termination so + kill-on-timeout stops orphaning descendants (#71148, #59549, #84967, + #68139 class). + +Design invariants: + +* Exceptions raised by the bounded operation propagate unchanged — callers + keep their existing error handling. Only the *timeout* outcome is + reified (as :class:`BoundedResult`), because that is the outcome the + call sites keep getting wrong. +* A timeout produced by this layer is OUR deadline, not the provider's. + Callers that feed errors into ``agent/error_classifier.py`` should + classify :class:`DeadlineExpired` distinctly from transport timeouts + (the #59549 / #80323 misattribution class). +* ``None`` timeout means unbounded, and non-positive resolved values are + normalized to ``None`` (matching the existing + ``HERMES_CONCURRENT_TOOL_TIMEOUT_S`` convention). +""" + +from __future__ import annotations + +import asyncio +import faulthandler +import logging +import os +import subprocess +import sys +import threading +import time +from dataclasses import dataclass +from typing import Any, Awaitable, Callable, Optional + +logger = logging.getLogger(__name__) + +__all__ = [ + "MAX_SAFE_TIMEOUT_S", + "BoundedResult", + "DeadlineExpired", + "clamp_timeout", + "resolve_timeout", + "run_bounded_async", + "run_bounded_sync", + "kill_process_tree", +] + +# Upper bound for any timeout handed to platform wait primitives. +# +# CPython converts ``threading.Lock.acquire(timeout=...)`` / +# ``Thread.join(timeout=...)`` deadlines to an absolute timestamp; very large +# relative timeouts overflow ``time_t`` on macOS and raise +# ``OverflowError: timestamp out of range for platform time_t`` (#83220). +# One year is semantically "unbounded" for every wait in this codebase while +# staying far below any platform conversion limit. +MAX_SAFE_TIMEOUT_S = 31_536_000.0 # 365 days + +# Grace period after a deadline fires before concluding the event loop thread +# is blocked in a synchronous call and dumping stacks (family A diagnostics). +_LOOP_BLOCKED_DUMP_GRACE_S = 5.0 + + +class DeadlineExpired(TimeoutError): + """A deadline enforced by this layer expired. + + Distinct from transport/provider timeout types on purpose: when this is + raised (or a :class:`BoundedResult` reports ``timed_out``), the timeout + was Hermes's own bound — error classification must not attribute it to + the provider (#59549 / #80323 misattribution class). + """ + + def __init__(self, label: str, timeout_s: float): + super().__init__(f"deadline expired after {timeout_s:.1f}s: {label}") + self.label = label + self.timeout_s = timeout_s + + +@dataclass(frozen=True) +class BoundedResult: + """Outcome of a bounded operation. + + ``timed_out`` is the reified outcome; on completion ``value`` holds the + operation's return value. Operation exceptions are never captured here — + they propagate to the caller unchanged. + """ + + timed_out: bool + value: Any + elapsed_s: float + timeout_s: Optional[float] + label: str + + def raise_if_timed_out(self) -> Any: + """Return ``value``, raising :class:`DeadlineExpired` on timeout.""" + if self.timed_out: + raise DeadlineExpired(self.label, float(self.timeout_s or 0.0)) + return self.value + + +def clamp_timeout(timeout: Optional[float]) -> Optional[float]: + """Normalize a timeout value for platform wait primitives. + + * ``None`` stays ``None`` (unbounded). + * Non-positive values become ``None`` (unbounded) — matching the existing + ``HERMES_CONCURRENT_TOOL_TIMEOUT_S`` "0 disables the bound" convention. + * Values above :data:`MAX_SAFE_TIMEOUT_S` are capped so they can never + overflow ``time_t`` inside ``Lock.acquire`` / ``Thread.join`` on macOS + (#83220). + * Non-numeric values are treated as unset (``None``) with a warning + rather than crashing the call path they were meant to protect. + """ + if timeout is None: + return None + try: + value = float(timeout) + except (TypeError, ValueError): + logger.warning("clamp_timeout: non-numeric timeout %r; treating as unbounded", timeout) + return None + if value != value: # NaN + logger.warning("clamp_timeout: NaN timeout; treating as unbounded") + return None + if value <= 0: + return None + return min(value, MAX_SAFE_TIMEOUT_S) + + +# --------------------------------------------------------------------------- +# Timeout resolution: config.yaml ``timeouts:`` section > legacy env var > +# registered default. +# --------------------------------------------------------------------------- + +def _timeouts_section() -> dict: + """Read the ``timeouts:`` root section from config.yaml (read-only). + + Isolated for testability and so a broken config read can never take down + the call path the timeout was protecting. + """ + try: + from hermes_cli.config import load_config_readonly + + section = load_config_readonly().get("timeouts") + return section if isinstance(section, dict) else {} + except Exception: + logger.debug("timeouts: config read failed; using defaults", exc_info=True) + return {} + + +def _lookup_dotted(section: dict, key: str) -> Any: + """Walk ``a.b.c`` through nested dicts; return None when absent.""" + node: Any = section + for part in key.split("."): + if not isinstance(node, dict) or part not in node: + return None + node = node[part] + return node + + +def resolve_timeout( + key: str, + *, + default: Optional[float], + env_var: Optional[str] = None, +) -> Optional[float]: + """Resolve a timeout in seconds for a dotted config key. + + Precedence (established by the ``providers.*.request_timeout_seconds`` + pattern — config wins over the legacy env var): + + 1. ``timeouts.`` in config.yaml (dotted key walks nested maps, e.g. + ``tools.concurrent_batch`` reads ``timeouts: {tools: {concurrent_batch: ...}}``) + 2. ``env_var`` when set and non-empty (legacy bridge — internal mechanism + and back-compat only; new surfaces must not grow new user-facing + ``HERMES_*`` timeout env vars) + 3. ``default`` + + The winning value is passed through :func:`clamp_timeout`, so ``0`` or a + negative value means "unbounded" and oversized values are made + platform-safe. Invalid (non-numeric) config/env values fall through to + the next source with a warning instead of breaking the protected path. + """ + raw = _lookup_dotted(_timeouts_section(), key) + if raw is not None: + try: + return clamp_timeout(float(raw)) + except (TypeError, ValueError): + logger.warning("timeouts.%s: invalid value %r in config.yaml; ignoring", key, raw) + + if env_var: + env_raw = os.getenv(env_var, "").strip() + if env_raw: + try: + return clamp_timeout(float(env_raw)) + except ValueError: + logger.warning("invalid %s=%r; ignoring", env_var, env_raw) + + return clamp_timeout(default) + + +# --------------------------------------------------------------------------- +# Bounded execution — async flavor. +# +# Generalizes plugins/platforms/telegram/adapter.py:_await_with_thread_deadline +# (the #63309 fix): the deadline is driven by a daemon threading.Timer so a +# blocked event loop cannot disable it, and a second timer dumps all thread +# stacks when the loop provably failed to process the expiry — the one piece +# of information loop-blocked hangs otherwise never surface. +# --------------------------------------------------------------------------- + +def _consume_abandoned(task: "asyncio.Future[Any]") -> None: + """Observe an abandoned task's outcome so it never logs 'never retrieved'.""" + try: + if not task.cancelled(): + task.exception() + except Exception: + pass + + +async def _run_abandon_cleanup(on_abandon: Callable[[], Awaitable[Any]]) -> None: + """Run abandonment cleanup fully fire-and-forget (its failures swallowed).""" + try: + await on_abandon() + except Exception: + logger.debug("deadline abandon-cleanup failed", exc_info=True) + + +def _dump_blocked_loop_diagnostics(label: str, timeout_s: float) -> None: + logger.warning( + "[deadline] %r deadline (%.0fs) expired but the event loop has not " + "processed the expiry after a further %.0fs — the loop thread appears " + "BLOCKED in a synchronous call, which is why no asyncio timeout can " + "fire. Dumping all thread stacks to stderr to identify the blocking " + "frame.", + label, + timeout_s, + _LOOP_BLOCKED_DUMP_GRACE_S, + ) + try: + faulthandler.dump_traceback(all_threads=True) + except Exception: + logger.debug("faulthandler traceback dump failed", exc_info=True) + + +async def run_bounded_async( + awaitable: Awaitable[Any], + timeout: Optional[float], + *, + label: str = "operation", + on_abandon: Optional[Callable[[], Awaitable[Any]]] = None, + dump_on_blocked_loop: bool = True, +) -> BoundedResult: + """Await ``awaitable`` under a wall-clock deadline independent of loop timers. + + On completion returns ``BoundedResult(timed_out=False, value=...)``; + exceptions from the operation (including ``asyncio.CancelledError`` from a + caller cancelling *us*) propagate unchanged. + + On timeout the underlying task is cancelled and **abandoned** — we do not + await cancellation completion, because cancellation-shielded scopes (anyio, + httpcore init, MCP SDK teardown) are exactly the paths that wedge forever. + ``on_abandon`` (zero-arg callable returning an awaitable) is scheduled as + detached best-effort cleanup for the half-built state the abandoned task + may leave behind. Returns ``BoundedResult(timed_out=True, value=None)``. + + ``timeout=None`` (or a non-positive resolved value) awaits unbounded. + """ + timeout_s = clamp_timeout(timeout) + start = time.monotonic() + if timeout_s is None: + value = await awaitable + return BoundedResult(False, value, time.monotonic() - start, None, label) + + task = asyncio.ensure_future(awaitable) + loop = asyncio.get_running_loop() + deadline: "asyncio.Future[None]" = loop.create_future() + loop_processed_expiry = threading.Event() + + def _mark_expired() -> None: + loop_processed_expiry.set() + if not deadline.done(): + deadline.set_result(None) + + def _expire_from_thread() -> None: + loop.call_soon_threadsafe(_mark_expired) + + def _watchdog_check() -> None: + if not loop_processed_expiry.is_set(): + _dump_blocked_loop_diagnostics(label, timeout_s) + + timer = threading.Timer(timeout_s, _expire_from_thread) + timer.daemon = True + timer.start() + watchdog: Optional[threading.Timer] = None + if dump_on_blocked_loop: + watchdog = threading.Timer( + timeout_s + _LOOP_BLOCKED_DUMP_GRACE_S, _watchdog_check + ) + watchdog.daemon = True + watchdog.start() + try: + done, _ = await asyncio.wait( + {task, deadline}, return_when=asyncio.FIRST_COMPLETED + ) + if task in done: + if not deadline.done(): + deadline.cancel() + value = await task + return BoundedResult(False, value, time.monotonic() - start, timeout_s, label) + + task.cancel() + task.add_done_callback(_consume_abandoned) + if on_abandon is not None: + cleanup = asyncio.ensure_future(_run_abandon_cleanup(on_abandon)) + cleanup.add_done_callback(_consume_abandoned) + logger.warning("[deadline] %r timed out after %.1fs; task abandoned", label, timeout_s) + return BoundedResult(True, None, time.monotonic() - start, timeout_s, label) + finally: + timer.cancel() + if watchdog is not None: + watchdog.cancel() + # cancel() cannot stop a Timer whose callback is already running; + # setting the event closes that race so a completed await can never + # be misreported as a blocked loop. + loop_processed_expiry.set() + + +# --------------------------------------------------------------------------- +# Bounded execution — sync flavor. +# --------------------------------------------------------------------------- + +def run_bounded_sync( + fn: Callable[[], Any], + timeout: Optional[float], + *, + label: str = "operation", + on_timeout: Optional[Callable[[], None]] = None, +) -> BoundedResult: + """Run ``fn`` in a daemon worker thread under a wall-clock deadline. + + On completion returns its value (exceptions re-raised in the caller). + On expiry the worker thread is **abandoned** (daemon, so it cannot block + interpreter exit), ``on_timeout`` (if given) runs best-effort in the + caller's thread — e.g. to mark a backend suspect or kill a subprocess — + and ``BoundedResult(timed_out=True)`` is returned. + + ``timeout=None`` (or non-positive) blocks until ``fn`` returns. + """ + timeout_s = clamp_timeout(timeout) + start = time.monotonic() + if timeout_s is None: + return BoundedResult(False, fn(), time.monotonic() - start, None, label) + + box: dict[str, Any] = {} + done = threading.Event() + + def _worker() -> None: + try: + box["value"] = fn() + except BaseException as exc: # re-raised in caller; must not vanish + box["exc"] = exc + finally: + done.set() + + thread = threading.Thread( + target=_worker, name=f"deadline-{label}", daemon=True + ) + thread.start() + if not done.wait(timeout_s): + logger.warning("[deadline] %r timed out after %.1fs; worker abandoned", label, timeout_s) + if on_timeout is not None: + try: + on_timeout() + except Exception: + logger.debug("deadline on_timeout callback failed", exc_info=True) + return BoundedResult(True, None, time.monotonic() - start, timeout_s, label) + + if "exc" in box: + raise box["exc"] + return BoundedResult(False, box.get("value"), time.monotonic() - start, timeout_s, label) + + +# --------------------------------------------------------------------------- +# Whole-tree process termination. +# --------------------------------------------------------------------------- + +def kill_process_tree(pid: int, *, sig: Optional[int] = None) -> bool: + """Terminate ``pid`` and all its descendants, portably. + + Kill-on-timeout that signals only the direct child orphans process trees + (cron scripts, in-container shells, browser daemons — #71148 class). + + * POSIX: signals the process group when ``pid`` leads one (callers that + spawn with ``start_new_session=True`` / ``preexec_fn=os.setsid`` get + full-tree kill), falling back to the single process otherwise. + ``sig`` defaults to ``SIGKILL``. + * Windows: ``taskkill /F /T`` terminates the tree without requiring + psutil. ``sig`` is ignored (Windows has no equivalent). + + Returns True when a termination call was issued without error, False when + the process was already gone or the call failed (callers treat both as + "nothing more we can do"). + """ + if sys.platform == "win32": + try: + subprocess.run( + ["taskkill", "/F", "/T", "/PID", str(pid)], + capture_output=True, + timeout=15, + check=False, + ) + return True + except Exception: + logger.debug("kill_process_tree: taskkill failed for pid %s", pid, exc_info=True) + return False + + import signal as _signal + + if sig is None: + sig = _signal.SIGKILL + try: + pgid = os.getpgid(pid) + except (ProcessLookupError, PermissionError, OSError): + pgid = None + try: + if pgid is not None and pgid == pid: + # pid leads its own group: kill the whole tree in one syscall. + os.killpg(pgid, sig) + else: + # Not a group leader (killing its group would hit our own group + # or an unrelated one) — signal the single process. + os.kill(pid, sig) + return True + except ProcessLookupError: + return False + except (PermissionError, OSError): + logger.debug("kill_process_tree: signal failed for pid %s", pid, exc_info=True) + return False diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 1438cd08fc..87ee404a5b 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -158,21 +158,19 @@ def _parse_tool_arguments(raw_arguments: Any) -> tuple[dict, Optional[str]]: def _resolve_concurrent_tool_timeout() -> float | None: - raw = os.getenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", "").strip() - if not raw: - return _DEFAULT_CONCURRENT_TOOL_TIMEOUT_S - try: - value = float(raw) - except ValueError: - logger.warning( - "invalid HERMES_CONCURRENT_TOOL_TIMEOUT_S=%r; using %.0fs", - raw, - _DEFAULT_CONCURRENT_TOOL_TIMEOUT_S, - ) - return _DEFAULT_CONCURRENT_TOOL_TIMEOUT_S - if value <= 0: - return None - return value + """Resolve the per-batch concurrent tool deadline. + + Delegates to the unified resolver (#85125): ``timeouts.tools.concurrent_batch`` + in config.yaml wins, the legacy ``HERMES_CONCURRENT_TOOL_TIMEOUT_S`` env var + remains the back-compat bridge, and ``0``/negative still disables the bound. + """ + from agent.deadline import resolve_timeout + + return resolve_timeout( + "tools.concurrent_batch", + default=_DEFAULT_CONCURRENT_TOOL_TIMEOUT_S, + env_var="HERMES_CONCURRENT_TOOL_TIMEOUT_S", + ) def _flush_session_db_after_tool_progress( diff --git a/cli-config.yaml.example b/cli-config.yaml.example index fb36868ce8..344648e49f 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -166,6 +166,22 @@ model: # gpt-5.4: # stale_timeout_seconds: 1800 # Longer non-stream stale timeout for slow large-context turns +# ============================================================================= +# Unified Timeouts (operation deadlines) +# ============================================================================= +# One place to override Hermes's internal operation deadlines (seconds). +# Keys are dotted paths resolved by agent/deadline.py:resolve_timeout(). +# Precedence: this section > legacy HERMES_* env var (back-compat) > built-in +# default. 0 or a negative value disables the bound (unbounded); very large +# values are clamped to a platform-safe maximum automatically. +# +# Currently resolved keys (more paths migrate here over time — see issue #85125): +# +# timeouts: +# tools: +# concurrent_batch: 420 # Deadline for a parallel tool-call batch +# # (legacy env: HERMES_CONCURRENT_TOOL_TIMEOUT_S) + # ============================================================================= # OpenRouter Provider Routing (only applies when using OpenRouter) # ============================================================================= diff --git a/hermes_cli/config.py b/hermes_cli/config.py index c863e214f2..1cc807492f 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -1878,6 +1878,7 @@ _EXTRA_KNOWN_ROOT_KEYS = { "require_mention", # top-level convenience form honored by the gateway (#3979) "unauthorized_dm_behavior", # top-level form read by gateway/config.py "signal", # Signal settings bridged to env vars by gateway/config.py + "timeouts", # unified timeout resolution section (agent/deadline.py, #85125) } _KNOWN_ROOT_KEYS = frozenset(DEFAULT_CONFIG.keys()) | _EXTRA_KNOWN_ROOT_KEYS diff --git a/tests/agent/test_deadline.py b/tests/agent/test_deadline.py new file mode 100644 index 0000000000..c31dfb59dc --- /dev/null +++ b/tests/agent/test_deadline.py @@ -0,0 +1,401 @@ +"""Tests for agent/deadline.py — the unified deadline layer (#85125). + +Covers: +* clamp_timeout normalization (None / non-positive / oversized / NaN / junk) +* resolve_timeout precedence: config.yaml ``timeouts:`` > legacy env var > default +* run_bounded_sync: completion, exception propagation, timeout + on_timeout +* run_bounded_async: completion, exception propagation, timeout + abandonment + of cancellation-shielded tasks, on_abandon cleanup +* kill_process_tree: descendants of a session-leader child die with it (POSIX) +* backward-compat contract of tool_executor._resolve_concurrent_tool_timeout + after its migration onto resolve_timeout +""" + +from __future__ import annotations + +import asyncio +import os +import signal +import subprocess +import sys +import threading +import time + +import pytest + +from agent.deadline import ( + MAX_SAFE_TIMEOUT_S, + BoundedResult, + DeadlineExpired, + clamp_timeout, + kill_process_tree, + resolve_timeout, + run_bounded_async, + run_bounded_sync, +) + + +# --------------------------------------------------------------------------- +# clamp_timeout +# --------------------------------------------------------------------------- + +class TestClampTimeout: + def test_none_stays_none(self): + assert clamp_timeout(None) is None + + def test_zero_and_negative_mean_unbounded(self): + assert clamp_timeout(0) is None + assert clamp_timeout(-5) is None + + def test_normal_value_passes_through(self): + assert clamp_timeout(420.0) == 420.0 + + def test_oversized_value_clamped_to_platform_safe_max(self): + # The #83220 class: >time_t deadlines crash Lock.acquire on macOS. + assert clamp_timeout(10**18) == MAX_SAFE_TIMEOUT_S + + def test_clamped_value_safe_for_threading_primitives(self): + # Regression proof for #83220: the clamped value must be accepted by + # the exact primitives that used to overflow. + big = clamp_timeout(float(10**15)) + assert big is not None + lock = threading.Lock() + assert lock.acquire(timeout=min(big, 0.001)) + lock.release() + + def test_nan_and_junk_treated_as_unbounded(self): + assert clamp_timeout(float("nan")) is None + assert clamp_timeout("not-a-number") is None # type: ignore[arg-type] + + +# --------------------------------------------------------------------------- +# resolve_timeout +# --------------------------------------------------------------------------- + +class TestResolveTimeout: + def test_default_wins_when_nothing_configured(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.delenv("HERMES_TEST_DEADLINE_X", raising=False) + assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 42.0 + + def test_env_var_beats_default(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.setenv("HERMES_TEST_DEADLINE_X", "17.5") + assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 17.5 + + def test_config_beats_env_var(self, monkeypatch): + monkeypatch.setattr( + "agent.deadline._timeouts_section", lambda: {"a": {"b": 99}} + ) + monkeypatch.setenv("HERMES_TEST_DEADLINE_X", "17.5") + assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 99.0 + + def test_dotted_key_walks_nested_maps(self, monkeypatch): + monkeypatch.setattr( + "agent.deadline._timeouts_section", + lambda: {"tools": {"concurrent_batch": 300}}, + ) + assert resolve_timeout("tools.concurrent_batch", default=420.0) == 300.0 + + def test_zero_config_value_means_unbounded(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {"a": {"b": 0}}) + assert resolve_timeout("a.b", default=42.0) is None + + def test_invalid_config_value_falls_through_to_env(self, monkeypatch): + monkeypatch.setattr( + "agent.deadline._timeouts_section", lambda: {"a": {"b": "soon"}} + ) + monkeypatch.setenv("HERMES_TEST_DEADLINE_X", "17.5") + assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 17.5 + + def test_invalid_env_value_falls_through_to_default(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.setenv("HERMES_TEST_DEADLINE_X", "banana") + assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 42.0 + + def test_broken_config_read_never_breaks_the_protected_path(self, monkeypatch): + # _timeouts_section swallows config-load failures internally; prove + # the public contract by making the underlying loader raise. + import agent.deadline as dl + + def _boom(): + raise RuntimeError("config unreadable") + + monkeypatch.setattr("hermes_cli.config.load_config_readonly", _boom) + assert dl._timeouts_section() == {} + assert resolve_timeout("a.b", default=5.0) == 5.0 + + +# --------------------------------------------------------------------------- +# run_bounded_sync +# --------------------------------------------------------------------------- + +class TestRunBoundedSync: + def test_completion_returns_value(self): + result = run_bounded_sync(lambda: "ok", 5.0, label="t") + assert result.timed_out is False + assert result.value == "ok" + assert result.raise_if_timed_out() == "ok" + + def test_unbounded_when_timeout_none(self): + result = run_bounded_sync(lambda: 7, None, label="t") + assert result.timed_out is False and result.value == 7 + + def test_exception_propagates_unchanged(self): + class Boom(RuntimeError): + pass + + with pytest.raises(Boom): + run_bounded_sync(lambda: (_ for _ in ()).throw(Boom("x")), 5.0, label="t") + + def test_timeout_abandons_worker_and_reports(self): + release = threading.Event() + + def _wedged(): + release.wait(30) + return "late" + + start = time.monotonic() + result = run_bounded_sync(_wedged, 0.2, label="wedged") + elapsed = time.monotonic() - start + assert result.timed_out is True + assert result.value is None + assert elapsed < 5.0 # returned near the deadline, not after 30s + with pytest.raises(DeadlineExpired) as exc_info: + result.raise_if_timed_out() + assert "wedged" in str(exc_info.value) + release.set() + + def test_on_timeout_callback_runs(self): + release = threading.Event() + fired = [] + result = run_bounded_sync( + lambda: release.wait(30), + 0.1, + label="t", + on_timeout=lambda: fired.append(True), + ) + assert result.timed_out and fired == [True] + release.set() + + def test_on_timeout_callback_failure_is_swallowed(self): + release = threading.Event() + result = run_bounded_sync( + lambda: release.wait(30), + 0.1, + label="t", + on_timeout=lambda: (_ for _ in ()).throw(RuntimeError("cleanup boom")), + ) + assert result.timed_out is True + release.set() + + def test_deadline_expired_is_a_timeout_error(self): + # Error-classification contract: our deadline must be catchable as + # TimeoutError but distinguishable by type from transport timeouts. + assert issubclass(DeadlineExpired, TimeoutError) + + +# --------------------------------------------------------------------------- +# run_bounded_async +# --------------------------------------------------------------------------- + +class TestRunBoundedAsync: + def test_completion_returns_value(self): + async def scenario(): + async def op(): + return "ok" + + return await run_bounded_async(op(), 5.0, label="t") + + result = asyncio.run(scenario()) + assert result.timed_out is False and result.value == "ok" + + def test_unbounded_when_timeout_none(self): + async def scenario(): + async def op(): + return 7 + + return await run_bounded_async(op(), None, label="t") + + result = asyncio.run(scenario()) + assert result.timed_out is False and result.value == 7 + + def test_exception_propagates_unchanged(self): + class Boom(RuntimeError): + pass + + async def scenario(): + async def op(): + raise Boom("x") + + await run_bounded_async(op(), 5.0, label="t") + + with pytest.raises(Boom): + asyncio.run(scenario()) + + def test_timeout_returns_promptly(self): + async def scenario(): + async def op(): + await asyncio.sleep(30) + + start = time.monotonic() + result = await run_bounded_async(op(), 0.2, label="slow") + return result, time.monotonic() - start + + result, elapsed = asyncio.run(scenario()) + assert result.timed_out is True + assert elapsed < 5.0 + + def test_timeout_abandons_cancellation_shielded_task(self): + """The family-A killer case: asyncio.wait_for cannot expire a shielded + scope; the thread-timer deadline must return anyway.""" + + async def scenario(): + hung = asyncio.Event() + + async def inner(): + await hung.wait() + + async def shielded(): + # Shield swallows the cancellation run_bounded_async issues. + await asyncio.shield(asyncio.ensure_future(inner())) + + start = time.monotonic() + result = await run_bounded_async(shielded(), 0.2, label="shielded") + elapsed = time.monotonic() - start + hung.set() # release the orphan so the loop can drain + await asyncio.sleep(0) + return result, elapsed + + result, elapsed = asyncio.run(scenario()) + assert result.timed_out is True + assert elapsed < 5.0 + + def test_on_abandon_cleanup_runs_detached(self): + async def scenario(): + cleaned = asyncio.Event() + + async def _cleanup(): + cleaned.set() + + async def op(): + await asyncio.sleep(30) + + result = await run_bounded_async( + op(), 0.1, label="t", on_abandon=_cleanup + ) + await asyncio.wait_for(cleaned.wait(), timeout=5.0) + return result + + result = asyncio.run(scenario()) + assert result.timed_out is True + + def test_completed_op_never_reports_timeout(self): + # Race guard: completion just under the deadline must report success. + async def scenario(): + async def op(): + await asyncio.sleep(0.01) + return "made it" + + return await run_bounded_async(op(), 5.0, label="t") + + result = asyncio.run(scenario()) + assert result.timed_out is False and result.value == "made it" + + +# --------------------------------------------------------------------------- +# kill_process_tree +# --------------------------------------------------------------------------- + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX process-group semantics") +class TestKillProcessTree: + def test_kills_descendants_of_session_leader(self, tmp_path): + """A child spawned with start_new_session must die with its own child. + + This is the orphan-tree class (#71148): killing only the direct child + leaves grandchildren running. + """ + started = tmp_path / "grandchild_started" + marker = tmp_path / "grandchild_alive" + grandchild_py = tmp_path / "grandchild.py" + grandchild_py.write_text( + "import pathlib, time\n" + f"pathlib.Path({str(started)!r}).write_text('x')\n" + "time.sleep(10)\n" + f"pathlib.Path({str(marker)!r}).write_text('x')\n" + ) + parent_py = tmp_path / "parent.py" + parent_py.write_text( + "import subprocess, sys, time\n" + f"subprocess.Popen([sys.executable, {str(grandchild_py)!r}])\n" + "time.sleep(10)\n" + ) + proc = subprocess.Popen( + [sys.executable, str(parent_py)], start_new_session=True + ) + deadline = time.monotonic() + 10 + while not started.exists() and time.monotonic() < deadline: + time.sleep(0.05) + assert started.exists(), "grandchild never spawned — test harness broken" + assert kill_process_tree(proc.pid) is True + proc.wait(timeout=5) + # Grandchild must be dead too: marker never appears. + time.sleep(1.5) + assert not marker.exists() + + def test_already_dead_pid_returns_false(self): + proc = subprocess.Popen([sys.executable, "-c", "pass"]) + proc.wait(timeout=10) + assert kill_process_tree(proc.pid) in (False, True) # reaped or zombie-signalable + + def test_non_group_leader_falls_back_to_single_kill(self): + # Child in OUR process group: killpg would signal the test runner. + proc = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(30)"]) + try: + assert os.getpgid(proc.pid) != proc.pid # not a leader + assert kill_process_tree(proc.pid, sig=signal.SIGTERM) is True + proc.wait(timeout=5) + finally: + if proc.poll() is None: + proc.kill() + + +# --------------------------------------------------------------------------- +# tool_executor migration contract +# --------------------------------------------------------------------------- + +class TestConcurrentToolTimeoutMigration: + """_resolve_concurrent_tool_timeout keeps its exact legacy contract.""" + + def _resolver(self): + from agent import tool_executor + + return tool_executor._resolve_concurrent_tool_timeout + + def test_default_unchanged(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.delenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", raising=False) + assert self._resolver()() == 420.0 + + def test_env_var_still_works(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.setenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", "60") + assert self._resolver()() == 60.0 + + def test_env_zero_still_disables(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.setenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", "0") + assert self._resolver()() is None + + def test_env_invalid_still_falls_back_to_default(self, monkeypatch): + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {}) + monkeypatch.setenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", "junk") + assert self._resolver()() == 420.0 + + def test_new_config_key_wins(self, monkeypatch): + monkeypatch.setattr( + "agent.deadline._timeouts_section", + lambda: {"tools": {"concurrent_batch": 300}}, + ) + monkeypatch.setenv("HERMES_CONCURRENT_TOOL_TIMEOUT_S", "60") + assert self._resolver()() == 300.0 From 8ac9ff18aecddab97a6ae9722086b2fa8fe99211 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:46:52 +0530 Subject: [PATCH 002/748] fix(agent): harden deadline layer per self-review MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - run_bounded_async: cancel + abandon the inner task when the CALLER is cancelled (leak the telegram original also had) - kill_process_tree: check taskkill exit code (Windows contract parity), suppress console flash via windows_hide_flags, and sweep a psutil descendant snapshot taken before signalling — reaches grandchildren in their own setsid sessions and the non-group-leader case (#71148 class) - resolve_timeout: reject bool (YAML true would become a 1s deadline) and NaN config values with fall-through instead of resolving unbounded - BoundedResult: kw_only to prevent positional transposition - tests: real clamped-value time_t regression proof, own-session descendant kill, external-cancellation task cleanup, bool/NaN config fall-through; pin already-dead-pid contract --- agent/deadline.py | 139 ++++++++++++++++++++++++++--------- tests/agent/test_deadline.py | 89 ++++++++++++++++++++-- 2 files changed, 189 insertions(+), 39 deletions(-) diff --git a/agent/deadline.py b/agent/deadline.py index c7aa5d28ca..5df4e869f0 100644 --- a/agent/deadline.py +++ b/agent/deadline.py @@ -29,7 +29,10 @@ migrate onto in later phases: process is silently disabled. This helper drives the deadline from a daemon ``threading.Timer`` (generalizing the proven telegram-adapter primitive) and abandons cancellation-shielded tasks instead of waiting for - cancellation to complete. + cancellation to complete. The telegram adapter's private copy + (``plugins/platforms/telegram/adapter.py:_await_with_thread_deadline``) + migrates onto this in Phase 2 of #85125 — do not let the two drift in the + meantime; fix bugs here first. * :func:`run_bounded_sync` — the same contract for synchronous callables bounded from a synchronous context (daemon worker thread, abandoned on @@ -37,7 +40,10 @@ migrate onto in later phases: * :func:`kill_process_tree` — portable whole-tree termination so kill-on-timeout stops orphaning descendants (#71148, #59549, #84967, - #68139 class). + #68139 class). Existing site-local tree-kills that migrate onto this in + Phase 4 of #85125: ``gateway/status.py`` (taskkill wrapper + psutil + snapshot/reap pair) and ``tools/code_execution_tool.py`` (psutil + recursive children kill). Design invariants: @@ -110,7 +116,7 @@ class DeadlineExpired(TimeoutError): self.timeout_s = timeout_s -@dataclass(frozen=True) +@dataclass(frozen=True, kw_only=True) class BoundedResult: """Outcome of a bounded operation. @@ -215,10 +221,19 @@ def resolve_timeout( """ raw = _lookup_dotted(_timeouts_section(), key) if raw is not None: - try: - return clamp_timeout(float(raw)) - except (TypeError, ValueError): - logger.warning("timeouts.%s: invalid value %r in config.yaml; ignoring", key, raw) + # Explicit float() (clamp_timeout would also convert) so that invalid + # config values FALL THROUGH to the env var / default instead of + # resolving as unbounded — do not "simplify" this away. bool is + # rejected because YAML `true` would silently become a 1-second + # deadline; NaN is rejected for the same fall-through reason. + if not isinstance(raw, bool): + try: + value = float(raw) + if value == value: # not NaN + return clamp_timeout(value) + except (TypeError, ValueError): + pass + logger.warning("timeouts.%s: invalid value %r in config.yaml; ignoring", key, raw) if env_var: env_raw = os.getenv(env_var, "").strip() @@ -302,7 +317,7 @@ async def run_bounded_async( start = time.monotonic() if timeout_s is None: value = await awaitable - return BoundedResult(False, value, time.monotonic() - start, None, label) + return BoundedResult(timed_out=False, value=value, elapsed_s=time.monotonic() - start, timeout_s=None, label=label) task = asyncio.ensure_future(awaitable) loop = asyncio.get_running_loop() @@ -332,14 +347,23 @@ async def run_bounded_async( watchdog.daemon = True watchdog.start() try: - done, _ = await asyncio.wait( - {task, deadline}, return_when=asyncio.FIRST_COMPLETED - ) + try: + done, _ = await asyncio.wait( + {task, deadline}, return_when=asyncio.FIRST_COMPLETED + ) + except asyncio.CancelledError: + # The CALLER cancelled us. Without this, `task` would keep running + # unobserved (and later log "exception was never retrieved") — + # a leak the telegram original also had. Cancel + abandon it, then + # let the cancellation propagate. + task.cancel() + task.add_done_callback(_consume_abandoned) + raise if task in done: if not deadline.done(): deadline.cancel() value = await task - return BoundedResult(False, value, time.monotonic() - start, timeout_s, label) + return BoundedResult(timed_out=False, value=value, elapsed_s=time.monotonic() - start, timeout_s=timeout_s, label=label) task.cancel() task.add_done_callback(_consume_abandoned) @@ -347,7 +371,7 @@ async def run_bounded_async( cleanup = asyncio.ensure_future(_run_abandon_cleanup(on_abandon)) cleanup.add_done_callback(_consume_abandoned) logger.warning("[deadline] %r timed out after %.1fs; task abandoned", label, timeout_s) - return BoundedResult(True, None, time.monotonic() - start, timeout_s, label) + return BoundedResult(timed_out=True, value=None, elapsed_s=time.monotonic() - start, timeout_s=timeout_s, label=label) finally: timer.cancel() if watchdog is not None: @@ -377,12 +401,17 @@ def run_bounded_sync( caller's thread — e.g. to mark a backend suspect or kill a subprocess — and ``BoundedResult(timed_out=True)`` is returned. + Intended for infrequent, seconds-scale blocking backend calls. Do NOT + use per-item in hot loops: each call spawns a thread, and every timeout + permanently leaks an abandoned daemon thread — a wedged backend called + in a retry loop would accumulate them. + ``timeout=None`` (or non-positive) blocks until ``fn`` returns. """ timeout_s = clamp_timeout(timeout) start = time.monotonic() if timeout_s is None: - return BoundedResult(False, fn(), time.monotonic() - start, None, label) + return BoundedResult(timed_out=False, value=fn(), elapsed_s=time.monotonic() - start, timeout_s=None, label=label) box: dict[str, Any] = {} done = threading.Event() @@ -406,11 +435,11 @@ def run_bounded_sync( on_timeout() except Exception: logger.debug("deadline on_timeout callback failed", exc_info=True) - return BoundedResult(True, None, time.monotonic() - start, timeout_s, label) + return BoundedResult(timed_out=True, value=None, elapsed_s=time.monotonic() - start, timeout_s=timeout_s, label=label) if "exc" in box: raise box["exc"] - return BoundedResult(False, box.get("value"), time.monotonic() - start, timeout_s, label) + return BoundedResult(timed_out=False, value=box.get("value"), elapsed_s=time.monotonic() - start, timeout_s=timeout_s, label=label) # --------------------------------------------------------------------------- @@ -423,26 +452,43 @@ def kill_process_tree(pid: int, *, sig: Optional[int] = None) -> bool: Kill-on-timeout that signals only the direct child orphans process trees (cron scripts, in-container shells, browser daemons — #71148 class). - * POSIX: signals the process group when ``pid`` leads one (callers that - spawn with ``start_new_session=True`` / ``preexec_fn=os.setsid`` get - full-tree kill), falling back to the single process otherwise. - ``sig`` defaults to ``SIGKILL``. - * Windows: ``taskkill /F /T`` terminates the tree without requiring - psutil. ``sig`` is ignored (Windows has no equivalent). + * Windows: ``taskkill /F /T`` terminates the tree (``sig`` ignored; + Windows has no equivalent). Console-window flash is suppressed via + ``windows_hide_flags`` and the exit code is checked, so a dead or + inaccessible PID reports ``False`` like the POSIX path. + * POSIX: the descendant set is snapshotted via psutil (a hard + dependency) BEFORE any signal — once the parent dies its children are + reparented and can no longer be found by a parent walk. Then the + process group is signalled when ``pid`` leads one (covers + grandchildren in the same session in one syscall), and every + snapshotted descendant is signalled individually — which also reaches + descendants that created their OWN sessions (a child that called + ``setsid``, exactly what user shell commands do; see + tools/environments/base.py). ``sig`` defaults to ``SIGKILL``. + psutil's identity-aware ``Process`` (PID + create time) means a + recycled PID is never signalled. - Returns True when a termination call was issued without error, False when - the process was already gone or the call failed (callers treat both as - "nothing more we can do"). + Returns True when the target (or any of its tree) was signalled, False + when the process was already gone or every termination call failed. """ if sys.platform == "win32": try: - subprocess.run( + from hermes_cli._subprocess_compat import windows_hide_flags + + creationflags = windows_hide_flags() + except Exception: + creationflags = 0 + try: + proc = subprocess.run( ["taskkill", "/F", "/T", "/PID", str(pid)], capture_output=True, timeout=15, check=False, + creationflags=creationflags, ) - return True + # taskkill exits non-zero for not-found / access-denied; keep the + # cross-platform contract (False = nothing was terminated). + return proc.returncode == 0 except Exception: logger.debug("kill_process_tree: taskkill failed for pid %s", pid, exc_info=True) return False @@ -451,21 +497,48 @@ def kill_process_tree(pid: int, *, sig: Optional[int] = None) -> bool: if sig is None: sig = _signal.SIGKILL + + # Snapshot descendants while the parent is still alive — after it dies + # they reparent to init/subreaper and a parent walk finds nothing. + descendants: list = [] try: + import psutil + + descendants = psutil.Process(int(pid)).children(recursive=True) + except Exception: + # Already gone, or psutil unavailable in a stripped env — the + # group-signal below still covers same-session descendants. + descendants = [] + + signalled = False + try: + # NOTE: getpgid→killpg has an inherent TOCTOU (pid could be reaped and + # recycled between the calls). All existing killpg sites share it; the + # psutil sweep below is identity-aware and does not. pgid = os.getpgid(pid) except (ProcessLookupError, PermissionError, OSError): pgid = None try: if pgid is not None and pgid == pid: - # pid leads its own group: kill the whole tree in one syscall. + # pid leads its own group: one syscall covers the whole group. + # (The == check guards against signalling the caller's own group + # when pid is not a leader.) os.killpg(pgid, sig) else: - # Not a group leader (killing its group would hit our own group - # or an unrelated one) — signal the single process. os.kill(pid, sig) - return True + signalled = True except ProcessLookupError: - return False + pass except (PermissionError, OSError): logger.debug("kill_process_tree: signal failed for pid %s", pid, exc_info=True) - return False + + # Sweep the snapshot: reaches descendants outside the parent's group + # (their own setsid sessions) and the non-group-leader case. + for child in descendants: + try: + if child.is_running(): # identity-aware: recycled PIDs skipped + child.send_signal(sig) + signalled = True + except Exception: + continue + return signalled diff --git a/tests/agent/test_deadline.py b/tests/agent/test_deadline.py index c31dfb59dc..9e5238b9af 100644 --- a/tests/agent/test_deadline.py +++ b/tests/agent/test_deadline.py @@ -55,12 +55,15 @@ class TestClampTimeout: assert clamp_timeout(10**18) == MAX_SAFE_TIMEOUT_S def test_clamped_value_safe_for_threading_primitives(self): - # Regression proof for #83220: the clamped value must be accepted by - # the exact primitives that used to overflow. + # Regression proof for #83220: the clamped value itself must be + # accepted by the exact primitive that used to overflow. Acquiring an + # uncontended lock returns immediately regardless of timeout, so + # passing the full clamped value is safe and actually exercises the + # time_t conversion. big = clamp_timeout(float(10**15)) assert big is not None lock = threading.Lock() - assert lock.acquire(timeout=min(big, 0.001)) + assert lock.acquire(timeout=big) lock.release() def test_nan_and_junk_treated_as_unbounded(self): @@ -113,6 +116,18 @@ class TestResolveTimeout: monkeypatch.setenv("HERMES_TEST_DEADLINE_X", "banana") assert resolve_timeout("a.b", default=42.0, env_var="HERMES_TEST_DEADLINE_X") == 42.0 + def test_bool_config_value_rejected(self, monkeypatch): + # YAML `true` must not silently become a 1-second deadline. + monkeypatch.setattr("agent.deadline._timeouts_section", lambda: {"a": {"b": True}}) + assert resolve_timeout("a.b", default=42.0) == 42.0 + + def test_nan_config_value_falls_through(self, monkeypatch): + # NaN must fall through to the next source, not resolve as unbounded. + monkeypatch.setattr( + "agent.deadline._timeouts_section", lambda: {"a": {"b": float("nan")}} + ) + assert resolve_timeout("a.b", default=42.0) == 42.0 + def test_broken_config_read_never_breaks_the_protected_path(self, monkeypatch): # _timeouts_section swallows config-load failures internally; prove # the public contract by making the underlying loader raise. @@ -302,6 +317,33 @@ class TestRunBoundedAsync: result = asyncio.run(scenario()) assert result.timed_out is False and result.value == "made it" + def test_external_cancellation_cancels_inner_task(self): + # If the CALLER cancels run_bounded_async, the inner task must not be + # leaked running unobserved. + async def scenario(): + started = asyncio.Event() + inner_cancelled = asyncio.Event() + + async def op(): + started.set() + try: + await asyncio.sleep(30) + except asyncio.CancelledError: + inner_cancelled.set() + raise + + outer = asyncio.ensure_future( + run_bounded_async(op(), 25.0, label="t") + ) + await started.wait() + outer.cancel() + with pytest.raises(asyncio.CancelledError): + await outer + await asyncio.wait_for(inner_cancelled.wait(), timeout=5.0) + return True + + assert asyncio.run(scenario()) is True + # --------------------------------------------------------------------------- # kill_process_tree @@ -343,10 +385,45 @@ class TestKillProcessTree: time.sleep(1.5) assert not marker.exists() + def test_kills_descendant_in_its_own_session(self, tmp_path): + """A descendant that setsid'd out of the parent's group must die too. + + killpg on the parent's group cannot reach it; the psutil descendant + sweep must (tools/environments/base.py documents user commands doing + exactly this). + """ + started = tmp_path / "setsid_grandchild_started" + marker = tmp_path / "setsid_grandchild_alive" + grandchild_py = tmp_path / "grandchild.py" + grandchild_py.write_text( + "import pathlib, time\n" + f"pathlib.Path({str(started)!r}).write_text('x')\n" + "time.sleep(10)\n" + f"pathlib.Path({str(marker)!r}).write_text('x')\n" + ) + parent_py = tmp_path / "parent.py" + parent_py.write_text( + "import subprocess, sys, time\n" + # grandchild leaves the parent's session/group entirely + f"subprocess.Popen([sys.executable, {str(grandchild_py)!r}], start_new_session=True)\n" + "time.sleep(10)\n" + ) + proc = subprocess.Popen( + [sys.executable, str(parent_py)], start_new_session=True + ) + deadline = time.monotonic() + 10 + while not started.exists() and time.monotonic() < deadline: + time.sleep(0.05) + assert started.exists(), "grandchild never spawned — test harness broken" + assert kill_process_tree(proc.pid) is True + proc.wait(timeout=5) + time.sleep(1.5) + assert not marker.exists() + def test_already_dead_pid_returns_false(self): - proc = subprocess.Popen([sys.executable, "-c", "pass"]) - proc.wait(timeout=10) - assert kill_process_tree(proc.pid) in (False, True) # reaped or zombie-signalable + proc = subprocess.Popen([sys.executable, "-c", "pass"], start_new_session=True) + proc.wait(timeout=10) # reaped: PID is gone from the process table + assert kill_process_tree(proc.pid) is False def test_non_group_leader_falls_back_to_single_kill(self): # Child in OUR process group: killpg would signal the test runner. From 8b387962ac3230ee1cf0a8040dae239eb0c0a247 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:53:39 +0530 Subject: [PATCH 003/748] fix(agent): suppress windows-footgun lint on POSIX-only killpg branch The os.killpg call sits below an early 'if sys.platform == win32: return' so it can never execute on Windows; the scanner is line-based and needs the inline marker. --- agent/deadline.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/agent/deadline.py b/agent/deadline.py index 5df4e869f0..5aa58e6c06 100644 --- a/agent/deadline.py +++ b/agent/deadline.py @@ -523,7 +523,7 @@ def kill_process_tree(pid: int, *, sig: Optional[int] = None) -> bool: # pid leads its own group: one syscall covers the whole group. # (The == check guards against signalling the caller's own group # when pid is not a leader.) - os.killpg(pgid, sig) + os.killpg(pgid, sig) # windows-footgun: ok — POSIX-only branch (win32 returns above) else: os.kill(pid, sig) signalled = True From e505ff9777ca579f0ba2dee64f56a01aa483cd87 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 13:54:15 -0500 Subject: [PATCH 004/748] feat(desktop): add reset-to-defaults to the statusbar context menu MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Once you have toggled a few items on and off there is no way to get back to the shipped layout short of remembering which ids are in STATUSBAR_HIDDEN_BY_DEFAULT. Add a row to the bar's right-click menu that restores that set. The row is disabled rather than hidden when nothing is customized, so it also advertises that a shipped layout exists. Reset touches item layout only — whole-bar visibility is a separate preference, and resetting from the bar's own menu should not make the bar you are right-clicking disappear. --- .../src/app/shell/statusbar-controls.tsx | 20 +++++- .../app/shell/statusbar-visibility.test.tsx | 62 +++++++++++++++++++ apps/desktop/src/i18n/en.ts | 1 + apps/desktop/src/i18n/types.ts | 1 + apps/desktop/src/i18n/zh.ts | 1 + apps/desktop/src/store/statusbar-prefs.ts | 17 +++++ 6 files changed, 101 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/shell/statusbar-controls.tsx b/apps/desktop/src/app/shell/statusbar-controls.tsx index 7718682c74..a245d1dcd2 100644 --- a/apps/desktop/src/app/shell/statusbar-controls.tsx +++ b/apps/desktop/src/app/shell/statusbar-controls.tsx @@ -17,7 +17,13 @@ import { ContribRender } from '@/contrib/react/boundary' import { useI18n } from '@/i18n' import { useKeybindHint } from '@/lib/keybinds/use-keybind-hint' import { cn } from '@/lib/utils' -import { $statusbarHiddenIds, setStatusbarItemVisible, toggleStatusbarVisible } from '@/store/statusbar-prefs' +import { + $statusbarHiddenIds, + isStatusbarLayoutDefault, + resetStatusbarLayout, + setStatusbarItemVisible, + toggleStatusbarVisible +} from '@/store/statusbar-prefs' // Shared chrome styling for interactive statusbar items (button / link / menu // trigger). The 'text' variant intentionally omits hover/transition/disabled. @@ -176,6 +182,18 @@ function StatusbarVisibilityMenu({ ))} + {/* Disabled rather than hidden when nothing is customized: the row is + also how you find out there IS a shipped layout to get back to. + Groups with the hide row below — both act on the bar, not an item. */} + { + event.preventDefault() + resetStatusbarLayout() + }} + > + {copy.resetStatusbar} + )} diff --git a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx index b4bce684bb..e65f272e05 100644 --- a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx +++ b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx @@ -125,6 +125,68 @@ describe('statusbar item visibility', () => { }) }) +describe('reset to defaults', () => { + it('puts a customized bar back to the shipped show/hide set', async () => { + $statusbarHiddenIds.set(['gateway-health']) + + const statusbar = bar([item('cron', 'Cron'), item('gateway-health', 'Gateway')]) + + expect(screen.queryByText('Gateway')).toBeNull() + expect(within(statusbar).getByText('Cron')).toBeTruthy() + + openContextMenu(statusbar) + fireEvent.click(await screen.findByRole('menuitem', { name: /reset to defaults/i })) + + expect($statusbarHiddenIds.get()).toEqual([...STATUSBAR_HIDDEN_BY_DEFAULT]) + expect(within(statusbar).getByText('Gateway')).toBeTruthy() + // Scoped to the bar: the menu stays open after a reset, so an unscoped query + // matches its still-listed 'Cron' checkbox row rather than a bar item. + expect(within(statusbar).queryByText('Cron')).toBeNull() + }) + + it('disables the row when the layout is already default', async () => { + const statusbar = bar([item('cron', 'Cron'), item('gateway-health', 'Gateway')]) + + openContextMenu(statusbar) + + const row = await screen.findByRole('menuitem', { name: /reset to defaults/i }) + expect(row.getAttribute('data-disabled')).not.toBeNull() + }) + + it('enables the row as soon as one item differs, in either direction', async () => { + const statusbar = bar([item('cron', 'Cron'), item('gateway-health', 'Gateway')]) + + // Showing a default-hidden item counts… + $statusbarHiddenIds.set(STATUSBAR_HIDDEN_BY_DEFAULT.filter(id => id !== 'cron')) + openContextMenu(statusbar) + expect( + (await screen.findByRole('menuitem', { name: /reset to defaults/i })).getAttribute('data-disabled') + ).toBeNull() + + // …and so does hiding a default-shown one. + $statusbarHiddenIds.set([...STATUSBAR_HIDDEN_BY_DEFAULT, 'gateway-health']) + expect( + (await screen.findByRole('menuitem', { name: /reset to defaults/i })).getAttribute('data-disabled') + ).toBeNull() + }) + + it('leaves whole-bar visibility alone — reset is about items, not the bar', async () => { + // Set to the NON-default so a reset that wrongly restored bar visibility too + // would flip this back to true and fail. StatusbarControls doesn't read the + // atom (the controller gates the mount), so the menu is still reachable here. + $statusbarVisible.set(false) + $statusbarHiddenIds.set([]) + + const statusbar = bar([item('gateway-health', 'Gateway')]) + + openContextMenu(statusbar) + fireEvent.click(await screen.findByRole('menuitem', { name: /reset to defaults/i })) + + expect($statusbarHiddenIds.get()).toEqual([...STATUSBAR_HIDDEN_BY_DEFAULT]) + expect($statusbarVisible.get()).toBe(false) + }) +}) + describe('whole-bar visibility', () => { it('hides the bar from the context menu, leaving the keybind as the way back', async () => { const statusbar = bar([item('gateway-health', 'Gateway')]) diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 35916ae65c..b0b3d46fbe 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -2561,6 +2561,7 @@ export const en: Translations = { gatewayTitle: 'Gateway', customizeTitle: 'Show in status bar', hideStatusbar: 'Hide status bar', + resetStatusbar: 'Reset to defaults', toggleApprovalMode: 'Approvals', toggleBackendVersion: 'Backend version', toggleCommandCenter: 'Command Center', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 4b34a851b0..4772ef6561 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2153,6 +2153,7 @@ export interface Translations { gatewayTitle: string customizeTitle: string hideStatusbar: string + resetStatusbar: string toggleApprovalMode: string toggleBackendVersion: string toggleCommandCenter: string diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 471fa8767a..d33f5e1b59 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -2739,6 +2739,7 @@ export const zh: Translations = { gatewayTitle: '网关', customizeTitle: '在状态栏中显示', hideStatusbar: '隐藏状态栏', + resetStatusbar: '恢复默认设置', toggleApprovalMode: '审批', toggleBackendVersion: '后端版本', toggleCommandCenter: '命令中心', diff --git a/apps/desktop/src/store/statusbar-prefs.ts b/apps/desktop/src/store/statusbar-prefs.ts index 612c9be18e..7318e18d7e 100644 --- a/apps/desktop/src/store/statusbar-prefs.ts +++ b/apps/desktop/src/store/statusbar-prefs.ts @@ -51,3 +51,20 @@ export function setStatusbarItemVisible(id: string, visible: boolean) { $statusbarHiddenIds.set(visible ? hidden.filter(entry => entry !== id) : [...hidden, id]) } + +/** Pure so the menu can derive its reset row's disabled state from the hidden + * list it already subscribes to, rather than reading the atom out of band. + * Set-compared: order is incidental (items are appended as they're hidden) and + * a duplicated id shouldn't read as a customization. */ +export function isStatusbarLayoutDefault(hidden: readonly string[]) { + const ids = new Set(hidden) + + return ids.size === STATUSBAR_HIDDEN_BY_DEFAULT.length && STATUSBAR_HIDDEN_BY_DEFAULT.every(id => ids.has(id)) +} + +/** Put the show/hide set back to what ships. Only touches item layout — whole-bar + * visibility is a separate preference, and resetting from the bar's own menu + * shouldn't make the bar the user is right-clicking disappear. */ +export function resetStatusbarLayout() { + $statusbarHiddenIds.set([...STATUSBAR_HIDDEN_BY_DEFAULT]) +} From 316f31c9d5450c73896f6a3a727112dca13c6c04 Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Fri, 14 Aug 2026 01:55:51 +0800 Subject: [PATCH 005/748] fix(agent): honor prompt caching capabilities for aliases --- agent/agent_runtime_helpers.py | 22 ++++++++ hermes_cli/config.py | 45 ++++++++++++++++ .../test_custom_provider_context_length.py | 52 ++++++++++++++++++- .../test_anthropic_prompt_cache_policy.py | 52 +++++++++++++++++++ website/docs/user-guide/configuring-models.md | 20 +++++++ 5 files changed, 190 insertions(+), 1 deletion(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index a2248e9b63..6a26178354 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2231,6 +2231,28 @@ def anthropic_prompt_cache_policy( and (eff_provider == "anthropic" or base_url_hostname(eff_base_url) == "api.anthropic.com") ) + # A custom Anthropic-compatible route may use a bare model alias that is + # canonicalized only after Hermes sends the request. In that case model + # spelling cannot prove cache support. Honor an exact route+model + # capability declaration instead; explicit false is authoritative too. + # This preserves the runtime model id (and therefore request/cache keys) + # while avoiding unsafe alias-name guesses. + custom_prompt_caching = None + if is_anthropic_wire: + try: + from hermes_cli.config import get_custom_provider_model_capability + + custom_prompt_caching = get_custom_provider_model_capability( + model=eff_model, + base_url=eff_base_url, + capability="prompt_caching", + custom_providers=getattr(agent, "_custom_providers", None), + ) + except Exception: + pass + if custom_prompt_caching is not None: + return custom_prompt_caching, custom_prompt_caching + # MiniMax-M3 rides MiniMax's server-side automatic prefix cache on the # Anthropic wire (content-keyed, no marker needed); explicit cache_control # is documented for M2.7/M2.5/M2.1/M2 only, so markers on M3 are dead diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 3fb11c7f30..42eef07067 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -1782,6 +1782,51 @@ def get_custom_provider_context_length( return None +def get_custom_provider_model_capability( + model: str, + base_url: str, + capability: str, + custom_providers: Optional[List[Dict[str, Any]]] = None, + config: Optional[Dict[str, Any]] = None, +) -> Optional[bool]: + """Return an explicit boolean capability for one custom-provider model. + + Matching is scoped to the normalized route and exact runtime model id so + aliases can declare capabilities without changing the id sent upstream. + Missing or non-boolean declarations return ``None``. + """ + if not model or not base_url or not capability: + return None + if custom_providers is None: + try: + custom_providers = get_compatible_custom_providers(config) + except Exception: + return None + if not isinstance(custom_providers, list): + return None + + target_url = normalize_route_base_url(base_url) + if not target_url: + return None + + for entry in custom_providers: + if not isinstance(entry, dict): + continue + entry_url = normalize_route_base_url(entry.get("base_url")) + if not entry_url or entry_url != target_url: + continue + models = entry.get("models") + if not isinstance(models, dict): + continue + model_cfg = models.get(model) + if not isinstance(model_cfg, dict): + continue + value = model_cfg.get(capability) + if isinstance(value, bool): + return value + return None + + def _coerce_config_version(value: Any) -> int: """Return a safe integer config version, treating invalid values as legacy.""" if isinstance(value, bool): diff --git a/tests/hermes_cli/test_custom_provider_context_length.py b/tests/hermes_cli/test_custom_provider_context_length.py index c783604b0e..04818f2274 100644 --- a/tests/hermes_cli/test_custom_provider_context_length.py +++ b/tests/hermes_cli/test_custom_provider_context_length.py @@ -8,7 +8,10 @@ from __future__ import annotations from unittest.mock import patch -from hermes_cli.config import get_custom_provider_context_length +from hermes_cli.config import ( + get_custom_provider_context_length, + get_custom_provider_model_capability, +) class TestGetCustomProviderContextLength: @@ -49,6 +52,53 @@ class TestGetCustomProviderContextLength: assert get_custom_provider_context_length("m", "http://x", []) is None +class TestGetCustomProviderModelCapability: + def test_matches_exact_model_on_normalized_route(self): + custom = [ + { + "base_url": "https://example.invalid/anthropic/", + "models": {"fable": {"prompt_caching": True}}, + } + ] + + assert get_custom_provider_model_capability( + "fable", + "https://example.invalid/anthropic", + "prompt_caching", + custom, + ) is True + assert get_custom_provider_model_capability( + "opus", + "https://example.invalid/anthropic", + "prompt_caching", + custom, + ) is None + + def test_false_is_preserved_and_non_boolean_is_ignored(self): + custom = [ + { + "base_url": "https://example.invalid/anthropic", + "models": { + "disabled": {"prompt_caching": False}, + "invalid": {"prompt_caching": "true"}, + }, + } + ] + + assert get_custom_provider_model_capability( + "disabled", + "https://example.invalid/anthropic", + "prompt_caching", + custom, + ) is False + assert get_custom_provider_model_capability( + "invalid", + "https://example.invalid/anthropic", + "prompt_caching", + custom, + ) is None + + class TestGetModelContextLengthHonorsOverride: """agent.model_metadata.get_model_context_length must honor the diff --git a/tests/run_agent/test_anthropic_prompt_cache_policy.py b/tests/run_agent/test_anthropic_prompt_cache_policy.py index e0a4ae00fa..ea2ed73a4e 100644 --- a/tests/run_agent/test_anthropic_prompt_cache_policy.py +++ b/tests/run_agent/test_anthropic_prompt_cache_policy.py @@ -161,6 +161,58 @@ class TestThirdPartyAnthropicGateway: assert agent._anthropic_prompt_cache_policy() == (False, False) + def test_bare_alias_with_explicit_prompt_caching_capability_caches(self): + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="fable", + ) + agent._custom_providers = [ + { + "name": "anthropic-proxy", + "base_url": "https://gateway.example.com/anthropic", + "models": {"fable": {"prompt_caching": True}}, + } + ] + + assert agent._anthropic_prompt_cache_policy() == (True, True) + + def test_explicit_prompt_caching_false_is_authoritative(self): + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="claude-fable-5", + ) + agent._custom_providers = [ + { + "name": "anthropic-proxy", + "base_url": "https://gateway.example.com/anthropic", + "models": {"claude-fable-5": {"prompt_caching": False}}, + } + ] + + assert agent._anthropic_prompt_cache_policy() == (False, False) + + def test_bare_alias_without_capability_stays_conservative(self): + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="fable", + ) + agent._custom_providers = [ + { + "name": "anthropic-proxy", + "base_url": "https://gateway.example.com/anthropic", + "models": {"fable": {"context_length": 1_000_000}}, + } + ] + + assert agent._anthropic_prompt_cache_policy() == (False, False) + + class TestMiniMaxAnthropicWire: """MiniMax's own model family on its Anthropic-compatible endpoint. diff --git a/website/docs/user-guide/configuring-models.md b/website/docs/user-guide/configuring-models.md index 487ae9529c..887c0486ad 100644 --- a/website/docs/user-guide/configuring-models.md +++ b/website/docs/user-guide/configuring-models.md @@ -185,6 +185,26 @@ providers: With discovery off, the model picker (`hermes model`, `/model`) shows the configured list instead of a live probe. +For an Anthropic-compatible gateway that resolves a bare model alias only +after receiving the request, opt the alias into native prompt-cache markers +with the per-model `prompt_caching` capability: + +```yaml +providers: + anthropic-proxy: + api: https://gateway.example.com/anthropic + transport: anthropic_messages + models: + fable: + context_length: 1000000 + prompt_caching: true +``` + +Hermes matches this declaration to the exact provider route and runtime model +id, without rewriting the alias. Set `prompt_caching: false` to explicitly +disable cache markers for a model; when omitted, Hermes keeps its normal +provider and model capability detection. + :::note Legacy format Older configs used a top-level `custom_providers:` list (with `base_url` instead of `api`). It still works and is auto-migrated to the `providers:` dict on `hermes update` (config v12). ::: From 4fa728b6bed3d0f4e1313ea0c6d183963fc5446d Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:23:39 +0530 Subject: [PATCH 006/748] fix: follow-up polish for salvaged PR #85512 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Log (debug) instead of silently swallowing capability-lookup failures in anthropic_prompt_cache_policy — a swallowed failure would otherwise downgrade an explicit prompt_caching: true to (False, False) with zero trace. Matches the sibling MoA branch's logger.debug style. - Use load_config_readonly() for the None-fallback in get_custom_provider_model_capability: the helper only reads, and the fallback fires on the blank-stub paths (agent init before _custom_providers is assigned, MoA/auxiliary destination planning), so skip the ~135us defensive deepcopy per call. - Add route-isolation regression tests at both levels (config helper + agent policy): a prompt_caching declaration for one provider route must never apply to another route with the same model name. Mutation-checked: both tests fail when the URL match is disabled. --- agent/agent_runtime_helpers.py | 7 +++++-- hermes_cli/config.py | 6 ++++++ .../test_custom_provider_context_length.py | 21 +++++++++++++++++++ .../test_anthropic_prompt_cache_policy.py | 19 +++++++++++++++++ 4 files changed, 51 insertions(+), 2 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 6a26178354..35d9848496 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2248,8 +2248,11 @@ def anthropic_prompt_cache_policy( capability="prompt_caching", custom_providers=getattr(agent, "_custom_providers", None), ) - except Exception: - pass + except Exception as _cap_exc: + logger.debug( + "custom-provider prompt_caching capability lookup failed: %s", + _cap_exc, + ) if custom_prompt_caching is not None: return custom_prompt_caching, custom_prompt_caching diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 42eef07067..29c7a4551e 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -1799,6 +1799,12 @@ def get_custom_provider_model_capability( return None if custom_providers is None: try: + if config is None: + # Read-only path: this helper never mutates the entries it + # scans, and get_compatible_custom_providers shallow-copies + # each entry before normalizing, so the no-deepcopy cache is + # safe here (~135us saved per call on the blank-stub paths). + config = load_config_readonly() custom_providers = get_compatible_custom_providers(config) except Exception: return None diff --git a/tests/hermes_cli/test_custom_provider_context_length.py b/tests/hermes_cli/test_custom_provider_context_length.py index 04818f2274..2220262a17 100644 --- a/tests/hermes_cli/test_custom_provider_context_length.py +++ b/tests/hermes_cli/test_custom_provider_context_length.py @@ -98,6 +98,27 @@ class TestGetCustomProviderModelCapability: custom, ) is None + def test_capability_is_route_isolated(self): + """A declaration for one route must not apply to another route. + + Guards normalize_route_base_url matching: if the URL comparison ever + regresses to a model-only (or hostname-only) shortcut, this pins the + failure. + """ + custom = [ + { + "base_url": "https://other.example.invalid/anthropic", + "models": {"fable": {"prompt_caching": True}}, + } + ] + + assert get_custom_provider_model_capability( + "fable", + "https://example.invalid/anthropic", + "prompt_caching", + custom, + ) is None + class TestGetModelContextLengthHonorsOverride: diff --git a/tests/run_agent/test_anthropic_prompt_cache_policy.py b/tests/run_agent/test_anthropic_prompt_cache_policy.py index ea2ed73a4e..9fe66cd4c3 100644 --- a/tests/run_agent/test_anthropic_prompt_cache_policy.py +++ b/tests/run_agent/test_anthropic_prompt_cache_policy.py @@ -212,6 +212,25 @@ class TestThirdPartyAnthropicGateway: assert agent._anthropic_prompt_cache_policy() == (False, False) + def test_capability_on_other_route_does_not_apply(self): + """prompt_caching declared for a DIFFERENT base_url must not enable + caching for this agent's route — route isolation at the policy level.""" + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="fable", + ) + agent._custom_providers = [ + { + "name": "other-proxy", + "base_url": "https://other.example.com/anthropic", + "models": {"fable": {"prompt_caching": True}}, + } + ] + + assert agent._anthropic_prompt_cache_policy() == (False, False) + class TestMiniMaxAnthropicWire: """MiniMax's own model family on its Anthropic-compatible endpoint. From a0939901df6534ace6bb8044c59b491148a1b27c Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:46:09 +0530 Subject: [PATCH 007/748] fix: address review feedback from #85512 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - /model switch now refreshes agent._custom_providers from the config loaded during the switch before re-evaluating cache policy — a prompt_caching flag added to config.yaml after session start was invisible to a mid-session switch (policy read the stale init-time snapshot while context_length resolution used the live list). - Production-path test: real config.yaml in the modern providers: dict shape through the real loader chain, exercising the init-order fallback (no _custom_providers attr) for both the fable opt-in and the opus explicit opt-out. - Pin operator kill-switch precedence: _cache_disabled (prompt_caching. cache_ttl falsy) beats an explicit per-model prompt_caching: true. --- agent/agent_runtime_helpers.py | 7 +++ .../test_anthropic_prompt_cache_policy.py | 62 +++++++++++++++++++ 2 files changed, 69 insertions(+) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 35d9848496..a9bf1a1d38 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2756,6 +2756,13 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo ) # ── Re-evaluate prompt caching ── + # Refresh the custom-provider snapshot from the config just loaded above + # so the per-model ``prompt_caching`` capability lookup sees the same + # live list the context-length resolution used — without this, a flag + # added to config.yaml after session start is invisible to a /model + # switch (the policy would read the stale init-time snapshot). + if _sm_custom_providers is not None: + agent._custom_providers = _sm_custom_providers agent._use_prompt_caching, agent._use_native_cache_layout = ( agent._anthropic_prompt_cache_policy( provider=new_provider, diff --git a/tests/run_agent/test_anthropic_prompt_cache_policy.py b/tests/run_agent/test_anthropic_prompt_cache_policy.py index 9fe66cd4c3..018c6edc79 100644 --- a/tests/run_agent/test_anthropic_prompt_cache_policy.py +++ b/tests/run_agent/test_anthropic_prompt_cache_policy.py @@ -231,6 +231,68 @@ class TestThirdPartyAnthropicGateway: assert agent._anthropic_prompt_cache_policy() == (False, False) + def test_operator_cache_disable_beats_explicit_capability_true(self): + """prompt_caching.cache_ttl disable (agent._cache_disabled) is a + global operator kill-switch — it must win over a per-model + prompt_caching: true declaration (#33555 semantics).""" + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="fable", + ) + agent._custom_providers = [ + { + "name": "anthropic-proxy", + "base_url": "https://gateway.example.com/anthropic", + "models": {"fable": {"prompt_caching": True}}, + } + ] + agent._cache_disabled = True + + assert agent._anthropic_prompt_cache_policy() == (False, False) + + def test_modern_providers_yaml_through_real_loader(self, tmp_path, monkeypatch): + """Production path: a real config.yaml in the modern ``providers:`` + dict shape, loaded through the real normalizer chain — including the + init-order fallback where ``_custom_providers`` is NOT yet set on the + agent and the policy loads config itself.""" + import textwrap + + hermes_home = tmp_path / ".hermes" + hermes_home.mkdir() + (hermes_home / "config.yaml").write_text( + textwrap.dedent( + """ + providers: + anthropic-proxy: + api: https://gateway.example.com/anthropic + transport: anthropic_messages + models: + fable: + context_length: 1000000 + prompt_caching: true + opus: + prompt_caching: false + """ + ) + ) + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + # load_config's cache is keyed by resolved config path, so pointing + # HERMES_HOME at a fresh tempdir needs no cache invalidation. + agent = _make_agent( + provider="custom:anthropic-proxy", + base_url="https://gateway.example.com/anthropic", + api_mode="anthropic_messages", + model="fable", + ) + # No agent._custom_providers — exercises the config fallback the + # init-time call (agent_init before the snapshot assignment) hits. + assert agent._anthropic_prompt_cache_policy() == (True, True) + + agent.model = "opus" + assert agent._anthropic_prompt_cache_policy() == (False, False) + class TestMiniMaxAnthropicWire: """MiniMax's own model family on its Anthropic-compatible endpoint. From ddbef9cd79a4348612f0b9bf0bc5f03c5bc9bd17 Mon Sep 17 00:00:00 2001 From: Xie <38648863+EvenXieWF@users.noreply.github.com> Date: Mon, 3 Aug 2026 09:40:43 +0800 Subject: [PATCH 008/748] fix(auxiliary): keep ZAI Coding Plan routing --- agent/auxiliary_client.py | 15 +++++++++++---- tests/agent/test_minimax_auxiliary_url.py | 18 ++++++++++++++++-- 2 files changed, 27 insertions(+), 6 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 307d0be442..f5e827b7c5 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1271,13 +1271,20 @@ def _to_openai_base_url(base_url: str) -> str: Anthropic-**only** custom gateways (path ends in ``/anthropic`` but has no sibling ``/v1``) must keep their path; rewriting them to ``/v1`` yields 404 on compression/vision/title_generation (#83642). + + ZAI exposes its general API and Coding Plan on separate endpoints. Its + Anthropic-compatible Coding Plan endpoint maps to ``/api/coding/paas/v4`` + on the OpenAI wire, not the general ``/api/paas/v4`` endpoint. Rewriting + to the general endpoint changes the billing pool and can return a false + insufficient-balance error for a valid Coding Plan key. """ url = str(base_url or "").strip().rstrip("/") if url.endswith("/anthropic"): - # ZAI (open.bigmodel.cn) uses /api/anthropic for Anthropic wire - # but /api/paas/v4 for OpenAI wire — the generic /v1 rewrite is wrong. - if "open.bigmodel.cn" in url or "bigmodel" in url: - rewritten = url[: -len("/anthropic")] + "/paas/v4" + # ZAI uses /api/anthropic for the Coding Plan's Anthropic wire. The + # matching OpenAI-wire endpoint is /api/coding/paas/v4; /api/paas/v4 + # is the independently billed general API. + if "open.bigmodel.cn" in url or "bigmodel" in url or "api.z.ai" in url: + rewritten = url[: -len("/anthropic")] + "/coding/paas/v4" logger.debug("Auxiliary client: rewrote ZAI base URL %s → %s", url, rewritten) return rewritten if _is_dual_surface_anthropic_host(url): diff --git a/tests/agent/test_minimax_auxiliary_url.py b/tests/agent/test_minimax_auxiliary_url.py index 581310802c..0f07670c7c 100644 --- a/tests/agent/test_minimax_auxiliary_url.py +++ b/tests/agent/test_minimax_auxiliary_url.py @@ -1,7 +1,9 @@ -"""Tests for MiniMax auxiliary client URL normalization. +"""Tests for Anthropic-to-OpenAI auxiliary URL normalization. MiniMax and MiniMax-CN set inference_base_url to the /anthropic path. -The auxiliary client uses the OpenAI SDK, which needs /v1 instead. +The auxiliary client uses the OpenAI SDK, which needs /v1 instead. ZAI's +Anthropic endpoint belongs to Coding Plan, whose OpenAI-compatible peer is the +separately billed /api/coding/paas/v4 endpoint. """ import sys @@ -37,5 +39,17 @@ class TestToOpenaiBaseUrl: url = "https://minimax.io.evil.example.com/anthropic" assert _to_openai_base_url(url) == url + def test_zai_anthropic_routes_to_coding_plan_openai_endpoint(self): + assert ( + _to_openai_base_url("https://api.z.ai/api/anthropic") + == "https://api.z.ai/api/coding/paas/v4" + ) + + def test_bigmodel_anthropic_routes_to_coding_plan_openai_endpoint(self): + assert ( + _to_openai_base_url("https://open.bigmodel.cn/api/anthropic") + == "https://open.bigmodel.cn/api/coding/paas/v4" + ) + def test_none(self): assert _to_openai_base_url(None) == "" From 17a675574c2dca5d659738bb8775ac068ccc12b1 Mon Sep 17 00:00:00 2001 From: sjungwon03 Date: Thu, 13 Aug 2026 12:19:32 -0700 Subject: [PATCH 009/748] fix: preserve anthropic_messages api_mode during fallback activation Fallback activation determined api_mode from the POST-rewrite client base_url, losing the Anthropic wire signal for /anthropic endpoints routed through provider 'custom', and never honored an explicit fb.api_mode config field. Pre-compute fb_api_mode from the ORIGINAL fallback base_url hint (before _to_openai_base_url rewriting), honor the explicit api_mode config field, check provider name before the base_url gate, and pass api_mode into resolve_provider_client at the fallback call site. Salvaged from PR #79787 (chat_completion_helpers.py hunks; the auxiliary_client.py hunk is redundant with #85466's wrap_base fix). --- agent/chat_completion_helpers.py | 98 ++++++++++++++++++-------------- 1 file changed, 56 insertions(+), 42 deletions(-) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 658919a5fd..44fd782fa6 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -2034,6 +2034,26 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool fb_base_url_hint = (fb.get("base_url") or "").strip() or None fb_api_key_hint = resolve_entry_api_key(fb) + # Determine api_mode from the ORIGINAL base_url (before URL transformation). + # resolve_provider_client() calls _to_openai_base_url() which rewrites + # /anthropic to /v1, losing the anthropic wire signal. Pre-compute here. + fb_api_mode = "chat_completions" + if fb.get("api_mode"): + fb_api_mode = str(fb.get("api_mode")).strip() + elif fb_provider == "anthropic": + # Provider-name check must not be gated on fb_base_url_hint: + # an entry that names provider: anthropic without an explicit + # base_url uses the provider's default endpoint and must still + # resolve to anthropic_messages, not chat_completions. + fb_api_mode = "anthropic_messages" + elif fb_base_url_hint: + _orig_url = fb_base_url_hint.rstrip("/").lower() + if ( + _orig_url.endswith("/anthropic") + or base_url_hostname(fb_base_url_hint) == "api.anthropic.com" + ): + fb_api_mode = "anthropic_messages" + # For Ollama Cloud endpoints, pull OLLAMA_API_KEY from env # when no explicit key is in the fallback config. Host match # (not substring) — see GHSA-76xc-57q6-vm5m. @@ -2044,7 +2064,8 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool fb_client, _resolved_fb_model = resolve_provider_client( fb_provider, model=fb_model, raw_codex=True, explicit_base_url=fb_base_url_hint, - explicit_api_key=fb_api_key_hint) + explicit_api_key=fb_api_key_hint, + api_mode=fb_api_mode) if fb_client is None: logger.warning( "Fallback to %s failed: provider not configured", @@ -2062,50 +2083,43 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool ) # Determine api_mode from provider / base URL / model - fb_api_mode = "chat_completions" + # If fb_api_mode was already pre-computed from original URL, preserve it + if not locals().get('fb_api_mode'): + fb_api_mode = "chat_completions" fb_base_url = str(fb_client.base_url) _fb_is_azure = agent._is_azure_openai_url(fb_base_url) - if fb_provider == "openai-codex": - fb_api_mode = "codex_responses" - elif fb_provider in {"nous", "nous-portal", "nousresearch"}: - # Portal is dual-wire: anthropic/* must land on /v1/messages. - # resolve_provider_client still returns an OpenAI client for - # Nous; the anthropic_messages branch below rebuilds the native - # client from that credential + base_url. - from hermes_cli.providers import nous_api_mode + + # Only re-determine if not already set from original URL + if fb_api_mode == "chat_completions": + if fb_provider == "openai-codex": + fb_api_mode = "codex_responses" + elif fb_provider in {"nous", "nous-portal", "nousresearch"}: + # Portal is dual-wire: anthropic/* must land on /v1/messages. + # resolve_provider_client still returns an OpenAI client for + # Nous; the anthropic_messages branch below rebuilds the native + # client from that credential + base_url. + from hermes_cli.providers import nous_api_mode - fb_api_mode = nous_api_mode(fb_model) - elif ( - fb_provider == "anthropic" - or fb_base_url.rstrip("/").lower().endswith("/anthropic") - or base_url_hostname(fb_base_url) == "api.anthropic.com" - ): - # Custom providers (e.g. cron-anthropic) point at the native - # api.anthropic.com host with no "/anthropic" path suffix, so the - # name/suffix checks above miss them and they default to - # chat_completions → POST /v1/chat/completions → 404. Match the - # host the same way determine_api_mode() and _detect_api_mode_for_url() - # do on the primary path. (#32243, #49247) - fb_api_mode = "anthropic_messages" - elif _fb_is_azure: - # Azure OpenAI serves gpt-5.x on /chat/completions — does NOT - # support the Responses API. Stay on chat_completions. - fb_api_mode = "chat_completions" - elif agent._is_direct_openai_url(fb_base_url): - fb_api_mode = "codex_responses" - elif agent._provider_model_requires_responses_api( - fb_model, - provider=fb_provider, - ): - # GPT-5.x models usually need Responses API, but keep - # provider-specific exceptions like Copilot gpt-5-mini on - # chat completions. - fb_api_mode = "codex_responses" - elif fb_provider == "bedrock" or ( - base_url_hostname(fb_base_url).startswith("bedrock-runtime.") - and base_url_host_matches(fb_base_url, "amazonaws.com") - ): - fb_api_mode = "bedrock_converse" + fb_api_mode = nous_api_mode(fb_model) + elif _fb_is_azure: + # Azure OpenAI serves gpt-5.x on /chat/completions — does NOT + # support the Responses API. Stay on chat_completions. + fb_api_mode = "chat_completions" + elif agent._is_direct_openai_url(fb_base_url): + fb_api_mode = "codex_responses" + elif agent._provider_model_requires_responses_api( + fb_model, + provider=fb_provider, + ): + # GPT-5.x models usually need Responses API, but keep + # provider-specific exceptions like Copilot gpt-5-mini on + # chat completions. + fb_api_mode = "codex_responses" + elif fb_provider == "bedrock" or ( + base_url_hostname(fb_base_url).startswith("bedrock-runtime.") + and base_url_host_matches(fb_base_url, "amazonaws.com") + ): + fb_api_mode = "bedrock_converse" old_model = agent.model old_provider = agent.provider From d0be93bd9a6fa34bb64091859d1869e3516ac159 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 12:22:04 -0700 Subject: [PATCH 010/748] fix: explicit fallback api_mode always wins; clean up dead-code guard Maintainer fixup on the #79787 salvage: - An explicit fb.api_mode of "chat_completions" was silently overridden by the codex_responses / bedrock re-detection pass (which only skipped re-detection when the pre-computed mode was non-default). Track explicitness in fb_api_mode_explicit and gate the whole re-detection block on it. - Replace the locals().get('fb_api_mode') dead-code hack with clean code (fb_api_mode is always bound at that point). - Restore the post-resolve /anthropic + api.anthropic.com host check for named custom providers whose base_url comes from config rather than the fallback entry (#32243, #49247), which the PR's restructure dropped. - Add regression tests: explicit api_mode honored (incl. explicit chat_completions not overridden), /anthropic-hint fallback detected pre-rewrite, api_mode forwarded to resolve_provider_client, plain fallback unchanged. --- agent/chat_completion_helpers.py | 35 +++- .../test_fallback_api_mode_preservation.py | 182 ++++++++++++++++++ 2 files changed, 207 insertions(+), 10 deletions(-) create mode 100644 tests/run_agent/test_fallback_api_mode_preservation.py diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 44fd782fa6..e618488e76 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -2035,10 +2035,16 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool fb_base_url_hint = (fb.get("base_url") or "").strip() or None fb_api_key_hint = resolve_entry_api_key(fb) # Determine api_mode from the ORIGINAL base_url (before URL transformation). - # resolve_provider_client() calls _to_openai_base_url() which rewrites - # /anthropic to /v1, losing the anthropic wire signal. Pre-compute here. + # resolve_provider_client() calls _to_openai_base_url() which can rewrite + # a dual-surface /anthropic base to /v1, losing the Anthropic wire signal + # from the client's post-rewrite base_url. Pre-compute here so detection + # sees the URL the user actually configured. (#79787) + # + # An explicit ``api_mode`` on the fallback entry always wins — including + # an explicit "chat_completions" — and suppresses all re-detection below. + fb_api_mode_explicit = bool(str(fb.get("api_mode") or "").strip()) fb_api_mode = "chat_completions" - if fb.get("api_mode"): + if fb_api_mode_explicit: fb_api_mode = str(fb.get("api_mode")).strip() elif fb_provider == "anthropic": # Provider-name check must not be gated on fb_base_url_hint: @@ -2082,15 +2088,14 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool fb_model, fb_provider, _norm_err, ) - # Determine api_mode from provider / base URL / model - # If fb_api_mode was already pre-computed from original URL, preserve it - if not locals().get('fb_api_mode'): - fb_api_mode = "chat_completions" + # Re-determine api_mode from provider / resolved base URL / model when + # the pre-computed pass above landed on the default and the user did + # not pin api_mode explicitly. An explicit fb.api_mode (even + # "chat_completions") must never be overridden here. fb_base_url = str(fb_client.base_url) _fb_is_azure = agent._is_azure_openai_url(fb_base_url) - - # Only re-determine if not already set from original URL - if fb_api_mode == "chat_completions": + + if not fb_api_mode_explicit and fb_api_mode == "chat_completions": if fb_provider == "openai-codex": fb_api_mode = "codex_responses" elif fb_provider in {"nous", "nous-portal", "nousresearch"}: @@ -2101,6 +2106,16 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool from hermes_cli.providers import nous_api_mode fb_api_mode = nous_api_mode(fb_model) + elif ( + fb_base_url.rstrip("/").lower().endswith("/anthropic") + or base_url_hostname(fb_base_url) == "api.anthropic.com" + ): + # Named custom providers (e.g. cron-anthropic) resolve their + # base_url from config rather than the fallback entry, so the + # pre-resolve hint check above never sees it. Match the host + # the same way determine_api_mode() and _detect_api_mode_for_url() + # do on the primary path. (#32243, #49247) + fb_api_mode = "anthropic_messages" elif _fb_is_azure: # Azure OpenAI serves gpt-5.x on /chat/completions — does NOT # support the Responses API. Stay on chat_completions. diff --git a/tests/run_agent/test_fallback_api_mode_preservation.py b/tests/run_agent/test_fallback_api_mode_preservation.py new file mode 100644 index 0000000000..ccda8fac3a --- /dev/null +++ b/tests/run_agent/test_fallback_api_mode_preservation.py @@ -0,0 +1,182 @@ +"""Fallback activation must preserve the Anthropic wire signal (PR #79787). + +Three behaviors salvaged from PR #79787: + +1. An explicit ``api_mode`` on the fallback entry is honored — and always + wins, including an explicit ``chat_completions`` that would otherwise be + overridden by codex_responses / bedrock re-detection. +2. ``fb_api_mode`` is detected from the ORIGINAL ``base_url`` hint before + ``resolve_provider_client`` / ``_to_openai_base_url`` can rewrite a + dual-surface ``/anthropic`` base to ``/v1``. +3. ``api_mode`` is passed into ``resolve_provider_client`` at the fallback + call site so the resolver keeps the Anthropic wire for custom bases. +""" + +from unittest.mock import MagicMock, patch + +from run_agent import AIAgent + + +def _make_agent(fallback_model=None): + with ( + patch("run_agent.get_tool_definitions", return_value=[]), + patch("run_agent.check_toolset_requirements", return_value={}), + patch("run_agent.OpenAI"), + ): + agent = AIAgent( + api_key="test-key", + base_url="https://openrouter.ai/api/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + fallback_model=fallback_model, + ) + agent.client = MagicMock() + return agent + + +def _mock_client(base_url="https://openrouter.ai/api/v1", api_key="fb-key"): + mock = MagicMock() + mock.base_url = base_url + mock.api_key = api_key + return mock + + +def _activate(agent, resolved_base_url, resolved_model, build_anthropic=None): + """Run _try_activate_fallback with the standard mock stack. + + Returns the mock for resolve_provider_client so callers can assert on + the api_mode kwarg passed at the fallback call site. + """ + patches = [ + patch( + "agent.chat_completion_helpers._fallback_entry_unavailable_without_network", + return_value=None, + ), + patch( + "agent.auxiliary_client.resolve_provider_client", + return_value=( + _mock_client(base_url=resolved_base_url), + resolved_model, + ), + ), + patch( + "hermes_cli.model_normalize.normalize_model_for_provider", + side_effect=lambda m, p: m, + ), + patch( + "agent.anthropic_adapter.build_anthropic_client", + side_effect=build_anthropic + or (lambda api_key, base_url, timeout=None, **kw: MagicMock()), + ), + ] + with patches[0], patches[1] as mock_rpc, patches[2], patches[3]: + assert agent._try_activate_fallback() is True + return mock_rpc + + +class TestExplicitApiModeHonored: + def test_explicit_anthropic_messages_honored(self): + fbs = [{ + "provider": "custom", + "model": "claude-opus-4-6", + "base_url": "https://gateway.example.com/v1", + "api_key": "k", + "api_mode": "anthropic_messages", + }] + agent = _make_agent(fallback_model=fbs) + mock_rpc = _activate(agent, "https://gateway.example.com/v1", "claude-opus-4-6") + assert agent.api_mode == "anthropic_messages" + # Behavior (3): api_mode forwarded to the resolver. + assert mock_rpc.call_args.kwargs["api_mode"] == "anthropic_messages" + + def test_explicit_chat_completions_not_overridden_by_redetection(self): + """An explicit chat_completions must survive re-detection. + + api.openai.com would normally re-detect to codex_responses via + _is_direct_openai_url; the explicit config field must win. + """ + fbs = [{ + "provider": "custom", + "model": "gpt-5.2", + "base_url": "https://api.openai.com/v1", + "api_key": "k", + "api_mode": "chat_completions", + }] + agent = _make_agent(fallback_model=fbs) + _activate(agent, "https://api.openai.com/v1", "gpt-5.2") + assert agent.api_mode == "chat_completions" + + def test_explicit_chat_completions_not_overridden_by_bedrock_redetection(self): + fbs = [{ + "provider": "bedrock", + "model": "anthropic.claude-3-5-sonnet", + "api_key": "k", + "api_mode": "chat_completions", + }] + agent = _make_agent(fallback_model=fbs) + _activate( + agent, + "https://bedrock-runtime.us-east-1.amazonaws.com", + "anthropic.claude-3-5-sonnet", + ) + assert agent.api_mode == "chat_completions" + + +class TestOriginalUrlDetection: + def test_anthropic_suffix_hint_survives_rewrite(self): + """Dual-surface /anthropic base rewritten to /v1 by the resolver: + detection must run on the ORIGINAL hint, not the rewritten client URL. + """ + fbs = [{ + "provider": "custom", + "model": "MiniMax-M2.5", + "base_url": "https://api.minimax.io/anthropic", + "api_key": "k", + }] + agent = _make_agent(fallback_model=fbs) + mock_rpc = _activate(agent, "https://api.minimax.io/v1", "MiniMax-M2.5") + assert agent.api_mode == "anthropic_messages" + assert mock_rpc.call_args.kwargs["api_mode"] == "anthropic_messages" + + def test_anthropic_host_hint_detected(self): + fbs = [{ + "provider": "custom", + "model": "claude-opus-4-6", + "base_url": "https://api.anthropic.com", + "api_key": "k", + }] + agent = _make_agent(fallback_model=fbs) + _activate(agent, "https://api.anthropic.com", "claude-opus-4-6") + assert agent.api_mode == "anthropic_messages" + + def test_provider_anthropic_without_base_url(self): + """provider: anthropic with no explicit base_url must still resolve + to anthropic_messages (follow-up commit 38303343 in PR #79787).""" + fbs = [{"provider": "anthropic", "model": "claude-opus-4-6", "api_key": "k"}] + agent = _make_agent(fallback_model=fbs) + mock_rpc = _activate(agent, "https://api.anthropic.com", "claude-opus-4-6") + assert agent.api_mode == "anthropic_messages" + assert mock_rpc.call_args.kwargs["api_mode"] == "anthropic_messages" + + +class TestPlainFallbackUnchanged: + def test_plain_openrouter_fallback_stays_chat_completions(self): + fbs = [{ + "provider": "openrouter", + "model": "z-ai/glm-5", + "api_key": "k", + }] + agent = _make_agent(fallback_model=fbs) + mock_rpc = _activate(agent, "https://openrouter.ai/api/v1", "z-ai/glm-5") + assert agent.api_mode == "chat_completions" + assert mock_rpc.call_args.kwargs["api_mode"] == "chat_completions" + + def test_post_resolve_anthropic_host_still_detected(self): + """Named custom providers resolve api.anthropic.com from config, not + the fallback entry — post-resolve detection must still catch it + (#32243, #49247).""" + fbs = [{"provider": "cron-anthropic", "model": "claude-opus-4-6", "api_key": "k"}] + agent = _make_agent(fallback_model=fbs) + _activate(agent, "https://api.anthropic.com/v1", "claude-opus-4-6") + assert agent.api_mode == "anthropic_messages" From bfff32ae8c6a9c585431997a6cc3d791b6ec9af5 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 12:22:22 -0700 Subject: [PATCH 011/748] chore: map contributor email for @sjungwon03 (PR #79787 salvage) --- contributors/emails/sjungwon03@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/sjungwon03@gmail.com diff --git a/contributors/emails/sjungwon03@gmail.com b/contributors/emails/sjungwon03@gmail.com new file mode 100644 index 0000000000..f71489d565 --- /dev/null +++ b/contributors/emails/sjungwon03@gmail.com @@ -0,0 +1 @@ +sjungwon03 From a364390dab3bb5050d639e94c4ed1b4c4ed77fe3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:14:57 -0700 Subject: [PATCH 012/748] feat: forward repeat through cron.manage add (#85602) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit cronjob(action=create) has supported a repeat cap since the tool existed, but the ws handler dropped the param — UIs building on cron.manage could not create run-N-times jobs. Forward it (digit strings accepted, None keeps schedule-kind defaults). One-shot relative schedules (bare 30m/2h) already flow through the schedule string untouched. --- tui_gateway/methods_tools.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index aae1f295ad..22d3d3f1de 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -1653,6 +1653,14 @@ def _(rid, params: dict) -> dict: name=jid, schedule=params.get("schedule", ""), prompt=params.get("prompt", ""), + # Optional repeat cap ("run N times"); None keeps the + # schedule-kind default (once for one-shot, forever + # for recurring). + repeat=( + int(params["repeat"]) + if str(params.get("repeat", "")).strip().isdigit() + else None + ), ) ), ) From 1c87772186c35115d6735d48a2b156b3c7cdf7b9 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 3 Jul 2026 15:55:14 +0800 Subject: [PATCH 013/748] fix(tui_gateway): restore openrouter provider on session resume MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit BARE_BILLING_PROVIDERS incorrectly included "openrouter" alongside "auto" and "custom". OpenRouter is a fully routable provider with its own API key and base_url — sessions that used OpenRouter store billing_provider="openrouter", and dropping it forces resume to the current global model (e.g. a custom endpoint), which is the wrong provider for the stored model. Remove "openrouter" from the bare-bucket set so OpenRouter sessions correctly restore their provider identity on resume. Fixes #57588 --- tests/test_tui_gateway_server.py | 36 ++++++++++++++++++++++++++++---- tui_gateway/server.py | 7 ++++++- 2 files changed, 38 insertions(+), 5 deletions(-) diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index 1407c9586f..8799be8afe 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -3125,10 +3125,11 @@ def test_session_cwd_set_profile_session_updates_profile_db(monkeypatch, tmp_pat def test_stored_session_runtime_overrides_skips_bare_billing_provider(): - """A bare billing bucket ("custom"/"auto"/"openrouter") must not be restored as the - provider identity on resume. A custom endpoint that never used `/model` persists only + """A bare billing bucket ("custom"/"auto") must not be restored as the provider + identity on resume. A custom endpoint that never used `/model` persists only `billing_provider="custom"`; restoring that broke `session.resume` with "No LLM provider - configured" (agent_init treats it as non-routable). A real provider, or an explicit + configured" (agent_init treats it as non-routable). ``"openrouter"`` is NOT a bare bucket + — it is a fully routable provider; see #57588. A real provider, or an explicit `model_config.provider`, is still restored. """ # Bare "custom" bucket, no explicit model_config.provider: no provider override restored. @@ -3136,7 +3137,7 @@ def test_stored_session_runtime_overrides_skips_bare_billing_provider(): assert "provider_override" not in ov assert ov["model_override"]["provider"] is None - for bare in ("auto", "openrouter", "custom"): + for bare in ("auto", "custom"): ov = server._stored_session_runtime_overrides({"model": "m", "billing_provider": bare}) assert "provider_override" not in ov @@ -3165,6 +3166,33 @@ def test_stored_session_runtime_overrides_restores_explicit_normal_tier(): assert overrides["service_tier_override"] == "" +def test_openrouter_session_resume_restores_provider(): + """OpenRouter is a fully routable provider — sessions that used OpenRouter must + restore the "openrouter" provider override on resume, not fall through to whatever + the current global model is. (#57588) + """ + # OpenRouter session with no explicit model_config.provider (the common case + # for sessions that never used /model): billing_provider="openrouter" should + # be restored as the provider override. + ov = server._stored_session_runtime_overrides( + {"model": "anthropic/claude-opus-4.8", "billing_provider": "openrouter"} + ) + assert ov["provider_override"] == "openrouter" + assert ov["model_override"]["provider"] == "openrouter" + assert ov["model_override"]["model"] == "anthropic/claude-opus-4.8" + + # When an explicit model_config.provider exists, it takes precedence over + # billing_provider (this path was already correct). + ov = server._stored_session_runtime_overrides( + { + "model": "anthropic/claude-opus-4.8", + "billing_provider": "openrouter", + "model_config": {"provider": "openrouter", "base_url": "https://openrouter.ai/api/v1"}, + } + ) + assert ov["provider_override"] == "openrouter" + + def test_persist_live_session_runtime_preserves_resume_metadata(monkeypatch): updates = {} diff --git a/tui_gateway/server.py b/tui_gateway/server.py index fc0c3f02f5..33aaabd285 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3768,7 +3768,12 @@ def _resolve_startup_runtime() -> tuple[str, str | None]: # Bare billing buckets are not routable provider identities (kept in parity with the # provider gate in agent_init). Restoring one as a session provider override breaks resume. -_BARE_BILLING_PROVIDERS = {"auto", "openrouter", "custom"} +# ``openrouter`` is deliberately excluded — it is a fully routable provider with its own +# API key and base_url. Sessions that used OpenRouter store +# ``billing_provider="openrouter"``; dropping it forces resume to the current global +# model (e.g. a custom endpoint), which is the wrong provider for the stored model. +# See #57588. +_BARE_BILLING_PROVIDERS = {"auto", "custom"} def _stored_session_runtime_overrides(row: dict | None) -> dict: From cdea8214cbbda51c64290d44d07558d5e4f589be Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:26:32 +0530 Subject: [PATCH 014/748] docs: fix stale parity claim in _BARE_BILLING_PROVIDERS comment The set is no longer in parity with agent_init's fail-fast gate (which still skips openrouter for a different reason: default route, not unroutable). Say so instead of claiming parity. --- tui_gateway/server.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 33aaabd285..d5719a7d6a 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3766,13 +3766,15 @@ def _resolve_startup_runtime() -> tuple[str, str | None]: return model, None -# Bare billing buckets are not routable provider identities (kept in parity with the -# provider gate in agent_init). Restoring one as a session provider override breaks resume. -# ``openrouter`` is deliberately excluded — it is a fully routable provider with its own -# API key and base_url. Sessions that used OpenRouter store -# ``billing_provider="openrouter"``; dropping it forces resume to the current global -# model (e.g. a custom endpoint), which is the wrong provider for the stored model. -# See #57588. +# Bare billing buckets are not routable provider identities; restoring one as a +# session provider override breaks resume. (agent_init's fail-fast gate is a +# DIFFERENT set that also skips "openrouter" — there it means "default route, +# don't fail fast", not "unroutable".) +# ``openrouter`` is deliberately excluded here — it is a fully routable provider +# with its own API key and base_url. Sessions that used OpenRouter store +# ``billing_provider="openrouter"``; dropping it forces resume to the current +# global model (e.g. a custom endpoint), which is the wrong provider for the +# stored model. See #57588. _BARE_BILLING_PROVIDERS = {"auto", "custom"} From dafdba324affcce21842b5c7f3da3ca66a7bebf4 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:30:51 +0530 Subject: [PATCH 015/748] feat(models): per-model metadata overrides via model_overrides config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a unified model_overrides config section that lets users manually declare context_window, max_output_tokens, capabilities, cost, and family for any provider+model — winning over models.dev, OpenRouter, and hardcoded defaults. Resolution order (first hit wins): 1. model_overrides.. (per-provider+model) 2. model_overrides.._default (per-provider default) 3. model_overrides._default (global default) 4. Normal catalog resolution Key subtlety: an unknown model id (not in the catalog) derives base metadata from sensible defaults before patching, so overriding a model the catalog doesn't know yet is the supported self-unblock path. This is exactly the #84482 scenario (Upstage solar-pro4/syn-pro wrong context) and the #8731 scenario (custom/local models with manual capability declaration). Wired into: - get_model_capabilities() — patches capability fields; unknown models get safe defaults (tools on, vision/reasoning off) before patching - lookup_models_dev_context() — context_window override, checked before catalog lookup so it works even for providers not in PROVIDER_TO_MODELS_DEV - get_model_info() — merges override dict onto catalog entry (shallow merge); for unknown models, the override is the sole source of metadata - get_model_context_length() — step 0b in the resolution pipeline, before custom_providers (0c) and before any network probe Config example: model_overrides: upstage: solar-pro4: context_window: 524288 syn-pro: context_window: 65536 custom:my-local-vllm: my-llava-model: context_window: 8192 supports_vision: true supports_reasoning: false supports_tools: true _default: context_window: 128000 Fixes #8731 Fixes #84482 Refs #47247 --- agent/model_metadata.py | 17 ++- agent/models_dev.py | 239 +++++++++++++++++++++++++++---- hermes_cli/config_defaults.py | 31 +++++ tests/agent/test_models_dev.py | 248 +++++++++++++++++++++++++++++++++ 4 files changed, 504 insertions(+), 31 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 6006c28b13..c137bdd6a9 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2577,6 +2577,7 @@ def get_model_context_length( Resolution order: 0. Explicit config override (model.context_length or custom_providers per-model) + 0b. model_overrides config (per-provider+model context_window override) 0c. Endpoint-scoped metadata for models validated on one multiplexed endpoint 1. Persistent cache (previously discovered via probing). Nous URLs, LM Studio, and Codex OAuth bypass the cache here so their provider @@ -2638,7 +2639,21 @@ def get_model_context_length( logger.debug("MoA aggregator context-length resolution failed", exc_info=True) # Fall through to the generic default if aggregator resolution failed. - # 0b. custom_providers per-model override — check before any probe. + # 0b. model_overrides config — per-provider+model context_window override. + # This is the supported self-unblock path for models with wrong or missing + # context in models.dev (#84482) and for custom/local models not in the + # catalog (#8731). Checked before custom_providers (step 0c) and before any + # network probe so it never blocks. + if provider and model: + try: + from agent.models_dev import _override_context_window + mo_ctx = _override_context_window(provider, model) + if mo_ctx is not None and mo_ctx > 0: + return mo_ctx + except Exception: + pass # fall through to other resolution paths + + # 0c. custom_providers per-model override — check before any probe. # This closes the gap where /model switch and display paths used to fall # back to 128K despite the user having a per-model context_length set. # See #15779. diff --git a/agent/models_dev.py b/agent/models_dev.py index 52c4100741..bcab594546 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -499,7 +499,17 @@ def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: Returns the context window in tokens, or None if not found. Handles case-insensitive matching and filters out context=0 entries. + + A ``model_overrides`` config entry for this provider+model (or its + ``_default`` fallback) wins over the catalog value — this is the + supported self-unblock path for models with wrong or missing context + in models.dev (#84482). """ + # Config override — checked before catalog so it always wins. + override_ctx = _override_context_window(provider, model) + if override_ctx is not None: + return override_ctx + mdev_provider_id = PROVIDER_TO_MODELS_DEV.get(provider) if not mdev_provider_id: return None @@ -586,6 +596,103 @@ class ModelCapabilities: model_family: str = "" +# --------------------------------------------------------------------------- # +# Per-model metadata overrides (config.yaml → model_overrides) # +# --------------------------------------------------------------------------- # +# +# Resolution order for every query function below: +# 1. ``model_overrides..`` — explicit per-provider+model +# 2. ``model_overrides.._default`` — per-provider default +# 3. ``model_overrides._default`` — global default +# 4. models.dev / OpenRouter / hardcoded — normal catalog resolution +# +# An override may set any subset of fields; unspecified fields fall through to +# the catalog value. For a model id NOT in the catalog, the override is the +# only source of metadata — this is the supported self-unblock path for new +# or custom models (#84482, #8731). + +_OVERRIDE_CACHE: Optional[Dict[str, Any]] = None +_OVERRIDE_CACHE_CFG_HASH: int = 0 + + +def _load_model_overrides() -> Dict[str, Any]: + """Load and cache the ``model_overrides`` config section. + + Caches by ``id(cfg)`` so a config reload (new dict identity) invalidates + automatically. Returns empty dict on any failure. + """ + global _OVERRIDE_CACHE, _OVERRIDE_CACHE_CFG_HASH + try: + from hermes_cli.config import cfg_get, load_config_readonly + cfg = load_config_readonly() + cfg_id = id(cfg) + if cfg_id == _OVERRIDE_CACHE_CFG_HASH and _OVERRIDE_CACHE is not None: + return _OVERRIDE_CACHE + raw = cfg_get(cfg, "model_overrides", default={}) + overrides = raw if isinstance(raw, dict) else {} + _OVERRIDE_CACHE = overrides + _OVERRIDE_CACHE_CFG_HASH = cfg_id + return overrides + except Exception: + return {} + + +def _resolve_model_override( + provider: str, model: str +) -> Optional[Dict[str, Any]]: + """Resolve the override dict for a provider+model, or None. + + Checks per-provider+model, then per-provider ``_default``, then global + ``_default``. Returns the first match (which may be partially populated — + callers only read the keys they care about). + """ + overrides = _load_model_overrides() + if not overrides: + return None + + provider_key = (provider or "").strip() + model_key = (model or "").strip() + if not provider_key and not model_key: + return None + + # 1. Per-provider+model + provider_section = overrides.get(provider_key) + if isinstance(provider_section, dict) and model_key: + model_section = provider_section.get(model_key) + if isinstance(model_section, dict): + return model_section + + # 2. Per-provider _default + if isinstance(provider_section, dict): + default = provider_section.get("_default") + if isinstance(default, dict): + return default + + # 3. Global _default + global_default = overrides.get("_default") + if isinstance(global_default, dict): + return global_default + + return None + + +def _override_context_window( + provider: str, model: str +) -> Optional[int]: + """Return the overridden context_window, or None.""" + ov = _resolve_model_override(provider, model) + if ov is None: + return None + raw = ov.get("context_window") + if raw is None: + return None + try: + ctx = int(raw) + return ctx if ctx > 0 else None + except (TypeError, ValueError): + return None + + def _get_provider_models(provider: str) -> Optional[Dict[str, Any]]: """Resolve a Hermes provider ID to its models dict from models.dev. @@ -629,6 +736,15 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit Uses the existing fetch_models_dev() and PROVIDER_TO_MODELS_DEV mapping. Returns None if model not found. + ``model_overrides`` config entries (per-provider+model, per-provider + ``_default``, or global ``_default``) win over catalog values. For a + model id NOT in the catalog, the override is the only source of + metadata — this is the supported self-unblock path for custom/local + models (#8731) and for models with wrong context in models.dev + (#84482). An override may set any subset of fields; unspecified fields + fall through to the catalog value (or sensible defaults when the model + is absent from the catalog entirely). + Extracts from model entry fields: - reasoning (bool) → supports_reasoning - tool_call (bool) → supports_tools @@ -637,42 +753,81 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit - limit.output (int) → max_output_tokens - family (str) → model_family """ + # Check config override first — it may fully replace the catalog entry + # or patch specific fields. For unknown models (not in catalog), the + # override is the sole source of metadata. + override = _resolve_model_override(provider, model) + models = _get_provider_models(provider) - if models is None: + entry = _find_model_entry(models, model) if models is not None else None + + # If no catalog entry and no override, we can't resolve capabilities. + if entry is None and override is None: return None - entry = _find_model_entry(models, model) - if entry is None: - return None + # Start from catalog entry (if found), else use defaults. + if entry is not None: + supports_tools = bool(entry.get("tool_call", False)) + # Vision: prefer explicit `modalities.input` when models.dev provides it. + # The older `attachment` flag can be stale or too broad for image routing; + # fall back to it only when the input modalities are absent/invalid. + input_mods = entry.get("modalities", {}) + if isinstance(input_mods, dict): + input_mods = input_mods.get("input") + else: + input_mods = None + if isinstance(input_mods, list): + supports_vision = "image" in input_mods + else: + supports_vision = bool(entry.get("attachment", False)) + supports_reasoning = bool(entry.get("reasoning", False)) - # Extract capability flags (default to False if missing) - supports_tools = bool(entry.get("tool_call", False)) - # Vision: prefer explicit `modalities.input` when models.dev provides it. - # The older `attachment` flag can be stale or too broad for image routing; - # fall back to it only when the input modalities are absent/invalid. - input_mods = entry.get("modalities", {}) - if isinstance(input_mods, dict): - input_mods = input_mods.get("input") + limit = entry.get("limit", {}) + if not isinstance(limit, dict): + limit = {} + + ctx = limit.get("context") + context_window = int(ctx) if isinstance(ctx, (int, float)) and ctx > 0 else 200000 + + out = limit.get("output") + max_output_tokens = int(out) if isinstance(out, (int, float)) and out > 0 else 8192 + + model_family = entry.get("family", "") or "" else: - input_mods = None - if isinstance(input_mods, list): - supports_vision = "image" in input_mods - else: - supports_vision = bool(entry.get("attachment", False)) - supports_reasoning = bool(entry.get("reasoning", False)) + # Unknown model — derive sensible defaults. The override will + # patch whichever fields it specifies; the rest stay at defaults + # that are safe for agentic use (tools on, vision/reasoning off). + supports_tools = True + supports_vision = False + supports_reasoning = False + context_window = 200000 + max_output_tokens = 8192 + model_family = "" - # Extract limits - limit = entry.get("limit", {}) - if not isinstance(limit, dict): - limit = {} - - ctx = limit.get("context") - context_window = int(ctx) if isinstance(ctx, (int, float)) and ctx > 0 else 200000 - - out = limit.get("output") - max_output_tokens = int(out) if isinstance(out, (int, float)) and out > 0 else 8192 - - model_family = entry.get("family", "") or "" + # Apply override patches (each field is optional in the override dict). + if override is not None: + if "supports_tools" in override: + supports_tools = bool(override["supports_tools"]) + if "supports_vision" in override: + supports_vision = bool(override["supports_vision"]) + if "supports_reasoning" in override: + supports_reasoning = bool(override["supports_reasoning"]) + if "context_window" in override: + try: + ctx_ov = int(override["context_window"]) + if ctx_ov > 0: + context_window = ctx_ov + except (TypeError, ValueError): + pass + if "max_output_tokens" in override: + try: + out_ov = int(override["max_output_tokens"]) + if out_ov > 0: + max_output_tokens = out_ov + except (TypeError, ValueError): + pass + if "model_family" in override: + model_family = str(override["model_family"] or "") return ModelCapabilities( supports_tools=supports_tools, @@ -884,27 +1039,51 @@ def get_model_info( Accepts Hermes or models.dev provider ID. Tries exact match then case-insensitive fallback. Returns None if not found. + + ``model_overrides`` config entries (per-provider+model, per-provider + ``_default``, or global ``_default``) patch the catalog entry's fields + when present. For a model id NOT in the catalog, the override is the + sole source of metadata — this is the supported self-unblock path + for custom/local models (#8731) and for models with wrong context + in models.dev (#84482). """ + override = _resolve_model_override(provider_id, model_id) + mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) data = fetch_models_dev() pdata = data.get(mdev_id) if not isinstance(pdata, dict): + # No catalog data — return from override alone if we have one. + if override is not None: + return _parse_model_info(model_id, override, mdev_id) return None models = pdata.get("models", {}) if not isinstance(models, dict): + if override is not None: + return _parse_model_info(model_id, override, mdev_id) return None # Exact match raw = models.get(model_id) if isinstance(raw, dict): + if override is not None: + merged = {**raw, **override} + return _parse_model_info(model_id, merged, mdev_id) return _parse_model_info(model_id, raw, mdev_id) # Case-insensitive fallback model_lower = model_id.lower() for mid, mdata in models.items(): if mid.lower() == model_lower and isinstance(mdata, dict): + if override is not None: + merged = {**mdata, **override} + return _parse_model_info(mid, merged, mdev_id) return _parse_model_info(mid, mdata, mdev_id) + # Model not in catalog — return from override alone if we have one. + if override is not None: + return _parse_model_info(model_id, override, mdev_id) + return None diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 584b10b36a..4094873737 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2539,6 +2539,37 @@ DEFAULT_CONFIG = { "providers": {}, }, + # Per-model metadata overrides — manually declare context_window, + # max_output_tokens, capabilities, or cost for any provider+model. + # Overrides win over models.dev, OpenRouter, and hardcoded defaults. + # + # Two scopes: + # 1. Per-provider+model: model_overrides.. + # 2. Per-provider default: model_overrides.._default + # 3. Global default: model_overrides._default + # + # An unknown model id (not in models.dev) inherits base metadata from + # its family/dated-snapshot entry before patching, so overriding a + # model the catalog doesn't know yet is the supported self-unblock + # path (#84482). + # + # Example: + # model_overrides: + # upstage: + # solar-pro4: + # context_window: 524288 + # syn-pro: + # context_window: 65536 + # custom:my-local-vllm: + # my-llava-model: + # context_window: 8192 + # supports_vision: true + # supports_reasoning: false + # supports_tools: true + # _default: + # context_window: 128000 + "model_overrides": {}, + # Network settings — workarounds for connectivity issues. "network": { # Force IPv4 connections. On servers with broken or unreachable IPv6, diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 2e87c79a62..ee34dee110 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -9,8 +9,11 @@ import pytest from agent.models_dev import ( PROVIDER_TO_MODELS_DEV, _extract_context, + _override_context_window, + _resolve_model_override, fetch_models_dev, get_model_capabilities, + get_model_info, get_provider_info, lookup_models_dev_context, ) @@ -390,3 +393,248 @@ class TestGetModelCapabilities: assert caps is not None assert caps.supports_vision is False + +# --------------------------------------------------------------------------- +# Per-model metadata overrides (model_overrides config) +# --------------------------------------------------------------------------- + + +class TestModelOverrides: + """Tests for the model_overrides config system.""" + + def _setup_overrides(self, overrides_dict): + """Patch _load_model_overrides to return the given dict.""" + import agent.models_dev as md + return patch.object(md, "_load_model_overrides", return_value=overrides_dict) + + # --- _resolve_model_override --- + + def test_per_provider_model_override(self): + """Per-provider+model override is found first.""" + overrides = { + "upstage": { + "solar-pro4": {"context_window": 524288}, + }, + } + with self._setup_overrides(overrides): + result = _resolve_model_override("upstage", "solar-pro4") + assert result is not None + assert result["context_window"] == 524288 + + def test_per_provider_default_fallback(self): + """Per-provider _default is used when model not found.""" + overrides = { + "upstage": { + "_default": {"context_window": 128000}, + }, + } + with self._setup_overrides(overrides): + result = _resolve_model_override("upstage", "unknown-model") + assert result is not None + assert result["context_window"] == 128000 + + def test_global_default_fallback(self): + """Global _default is used when provider not found.""" + overrides = { + "_default": {"context_window": 65536}, + } + with self._setup_overrides(overrides): + result = _resolve_model_override("unknown-provider", "unknown-model") + assert result is not None + assert result["context_window"] == 65536 + + def test_no_override_returns_none(self): + """No override found returns None.""" + with self._setup_overrides({}): + result = _resolve_model_override("anthropic", "claude-sonnet-4") + assert result is None + + def test_per_provider_model_beats_default(self): + """Per-provider+model wins over per-provider _default.""" + overrides = { + "upstage": { + "solar-pro4": {"context_window": 524288}, + "_default": {"context_window": 128000}, + }, + } + with self._setup_overrides(overrides): + result = _resolve_model_override("upstage", "solar-pro4") + assert result is not None + assert result["context_window"] == 524288 + + def test_per_provider_default_beats_global(self): + """Per-provider _default wins over global _default.""" + overrides = { + "upstage": { + "_default": {"context_window": 128000}, + }, + "_default": {"context_window": 65536}, + } + with self._setup_overrides(overrides): + result = _resolve_model_override("upstage", "unknown-model") + assert result is not None + assert result["context_window"] == 128000 + + # --- _override_context_window --- + + def test_override_context_window_returns_value(self): + overrides = { + "upstage": { + "syn-pro": {"context_window": 65536}, + }, + } + with self._setup_overrides(overrides): + ctx = _override_context_window("upstage", "syn-pro") + assert ctx == 65536 + + def test_override_context_window_returns_none_when_missing(self): + with self._setup_overrides({}): + ctx = _override_context_window("upstage", "syn-pro") + assert ctx is None + + def test_override_context_window_rejects_zero(self): + overrides = { + "upstage": { + "bad-model": {"context_window": 0}, + }, + } + with self._setup_overrides(overrides): + ctx = _override_context_window("upstage", "bad-model") + assert ctx is None + + # --- get_model_capabilities with overrides --- + + def test_caps_override_unknown_model(self): + """Override provides capabilities for a model NOT in the catalog (#8731).""" + overrides = { + "custom:my-vllm": { + "my-llava-model": { + "context_window": 8192, + "supports_vision": True, + "supports_reasoning": False, + "supports_tools": True, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + caps = get_model_capabilities("custom:my-vllm", "my-llava-model") + assert caps is not None + assert caps.context_window == 8192 + assert caps.supports_vision is True + assert caps.supports_reasoning is False + assert caps.supports_tools is True + + def test_caps_override_patches_existing_catalog_entry(self): + """Override patches specific fields on a known catalog entry (#84482).""" + overrides = { + "anthropic": { + "claude-sonnet-4": { + "context_window": 500000, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=CAPS_REGISTRY): + caps = get_model_capabilities("anthropic", "claude-sonnet-4") + assert caps is not None + # Override wins + assert caps.context_window == 500000 + # Non-overridden fields preserved from catalog + assert caps.supports_vision is True + assert caps.supports_tools is True + + def test_caps_no_override_no_catalog_returns_none(self): + """No override and no catalog entry → None.""" + with self._setup_overrides({}), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + caps = get_model_capabilities("anthropic", "unknown-model") + assert caps is None + + def test_caps_override_default_for_unknown_model(self): + """Per-provider _default provides capabilities for unknown models.""" + overrides = { + "custom:my-vllm": { + "_default": { + "context_window": 32768, + "supports_tools": True, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + caps = get_model_capabilities("custom:my-vllm", "some-new-model") + assert caps is not None + assert caps.context_window == 32768 + assert caps.supports_tools is True + + # --- lookup_models_dev_context with overrides --- + + def test_context_lookup_override_wins_over_catalog(self): + """Override context_window wins over models.dev catalog value.""" + overrides = { + "anthropic": { + "claude-opus-4-6": {"context_window": 500000}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): + ctx = lookup_models_dev_context("anthropic", "claude-opus-4-6") + assert ctx == 500000 + + def test_context_lookup_override_for_unknown_provider(self): + """Override works for providers not in PROVIDER_TO_MODELS_DEV.""" + overrides = { + "upstage": { + "solar-pro4": {"context_window": 524288}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + ctx = lookup_models_dev_context("upstage", "solar-pro4") + assert ctx == 524288 + + # --- get_model_info with overrides --- + + def test_model_info_override_for_unknown_model(self): + """Override provides full metadata for a model not in the catalog.""" + overrides = { + "custom:my-vllm": { + "my-llava-model": { + "name": "My LLaVA Model", + "family": "llava", + "reasoning": False, + "tool_call": True, + "limit": {"context": 8192, "output": 4096}, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + info = get_model_info("custom:my-vllm", "my-llava-model") + assert info is not None + assert info.name == "My LLaVA Model" + assert info.family == "llava" + assert info.context_window == 8192 + assert info.max_output == 4096 + assert info.tool_call is True + assert info.reasoning is False + + def test_model_info_override_merges_with_catalog(self): + """Override patches specific fields on a known catalog entry.""" + overrides = { + "anthropic": { + "claude-sonnet-4-6": { + "limit": {"context": 500000, "output": 64000}, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): + info = get_model_info("anthropic", "claude-sonnet-4-6") + assert info is not None + # Override wins + assert info.context_window == 500000 + # Non-overridden fields preserved from catalog + assert info.name == "claude-sonnet-4-6" + From de47d19f1f860c22bbb2235511d486c6342d21a9 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:38:11 +0530 Subject: [PATCH 016/748] fix(models): one canonical override schema, fill-gap _default semantics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-ups on the model_overrides feature: - ONE canonical override schema everywhere. get_model_info previously merged the override dict raw into the models.dev catalog shape ({**raw, **override}), so the documented context_window/supports_* keys silently did nothing on that path (cost guard, inventory) while working in capabilities/context paths — same config key, two incompatible schemas. Overrides are now translated into the catalog shape at the get_model_info boundary (_override_to_catalog_shape), and sub-dicts (limit, modalities) are MERGED, not clobbered — an override setting only context_window no longer wipes the catalog's limit.output. - _default is now a FILL-GAP default, not an override: it applies only to models the catalog does not know (the #8731/#84482 self-unblock path) and never displaces catalog data. A _default: {context_window: 128000} can no longer clamp every model of a provider. Explicit per-provider+model entries keep their win-over-catalog semantics. - Early-chain _override_context_window (model_metadata step 0b) is explicit-only, so a _default can never preempt custom_providers per-model settings or live probes; fill-gap defaults apply at the lookup_models_dev_context catalog-miss boundary (step 5f) instead. This fixes the precedence inversion where a provider/global _default silently overrode an explicit per-endpoint per-model context_length. - Provider keys accept BOTH id spaces (Hermes id and models.dev id: copilot/github-copilot both work) and model ids match case-insensitively, mirroring catalog lookup. - Malformed override values (context_window: '512k') log a one-shot warning instead of being silently swallowed. - DEFAULT_CONFIG comment: removed the false family/dated-snapshot inheritance claim, documented the recognized field list, fill-gap semantics, and the id-space rule. - Tests: rewritten for the new contracts (fill-gap invariants, dual-id-space keys, sub-dict merge preservation, one-shot warning); added a real-config-yaml e2e plumbing test (mutation-checked: fails when the config key wiring is broken). --- agent/model_metadata.py | 12 +- agent/models_dev.py | 367 +++++++++++++++++++++++---------- hermes_cli/config_defaults.py | 37 +++- tests/agent/test_models_dev.py | 255 ++++++++++++++++++++--- 4 files changed, 512 insertions(+), 159 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index c137bdd6a9..a638798a97 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2639,11 +2639,13 @@ def get_model_context_length( logger.debug("MoA aggregator context-length resolution failed", exc_info=True) # Fall through to the generic default if aggregator resolution failed. - # 0b. model_overrides config — per-provider+model context_window override. - # This is the supported self-unblock path for models with wrong or missing - # context in models.dev (#84482) and for custom/local models not in the - # catalog (#8731). Checked before custom_providers (step 0c) and before any - # network probe so it never blocks. + # 0b. model_overrides config — EXPLICIT per-provider+model context_window + # override only (fill-gap _default entries are applied later, inside + # lookup_models_dev_context at step 5f, once the catalog has actually + # missed — so a _default can never preempt custom_providers or live + # probes). This is the supported self-unblock path for models with + # wrong context in models.dev (#84482) and for custom/local models + # (#8731). Config-read only; never blocks on the network. if provider and model: try: from agent.models_dev import _override_context_window diff --git a/agent/models_dev.py b/agent/models_dev.py index bcab594546..804694e7d8 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -500,28 +500,28 @@ def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: Returns the context window in tokens, or None if not found. Handles case-insensitive matching and filters out context=0 entries. - A ``model_overrides`` config entry for this provider+model (or its - ``_default`` fallback) wins over the catalog value — this is the - supported self-unblock path for models with wrong or missing context - in models.dev (#84482). + An EXPLICIT ``model_overrides`` config entry for this provider+model + wins over the catalog value; ``_default`` entries fill the gap only + when the catalog has no answer — the supported self-unblock path for + models with wrong or missing context in models.dev (#84482). """ - # Config override — checked before catalog so it always wins. + # Explicit config override — checked before catalog so it always wins. override_ctx = _override_context_window(provider, model) if override_ctx is not None: return override_ctx mdev_provider_id = PROVIDER_TO_MODELS_DEV.get(provider) if not mdev_provider_id: - return None + return _default_override_context(provider) data = fetch_models_dev() provider_data = data.get(mdev_provider_id) if not isinstance(provider_data, dict): - return None + return _default_override_context(provider) models = provider_data.get("models", {}) if not isinstance(models, dict): - return None + return _default_override_context(provider) # Exact match entry = models.get(model) @@ -560,7 +560,16 @@ def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: if ctx: return ctx - return None + # Catalog miss — a _default override may fill the gap (#84482). + return _default_override_context(provider) + + +def _default_override_context(provider: str) -> Optional[int]: + """Fill-gap context from a ``_default`` override, for catalog misses.""" + default = _default_model_override(provider) + if default is None: + return None + return _override_int(default, "context_window") def _extract_context(entry: Dict[str, Any]) -> Optional[int]: @@ -600,19 +609,27 @@ class ModelCapabilities: # Per-model metadata overrides (config.yaml → model_overrides) # # --------------------------------------------------------------------------- # # -# Resolution order for every query function below: -# 1. ``model_overrides..`` — explicit per-provider+model -# 2. ``model_overrides.._default`` — per-provider default -# 3. ``model_overrides._default`` — global default -# 4. models.dev / OpenRouter / hardcoded — normal catalog resolution +# Canonical override schema (the ONLY key space consumers accept): +# context_window, max_output_tokens, supports_tools, supports_vision, +# supports_reasoning, model_family # -# An override may set any subset of fields; unspecified fields fall through to -# the catalog value. For a model id NOT in the catalog, the override is the -# only source of metadata — this is the supported self-unblock path for new -# or custom models (#84482, #8731). +# Resolution semantics: +# 1. ``model_overrides..`` — explicit override. Always +# wins over the catalog for the fields it sets (partial patch). +# 2. ``model_overrides.._default`` / ``model_overrides._default`` +# — FILL-GAP defaults. They apply ONLY to models the catalog does not +# know (the #8731/#84482 self-unblock path for custom/local/new +# models) and never displace catalog data for known models. A +# ``_default: {context_window: 128000}`` therefore cannot clamp every +# catalog-known model of a provider. +# +# Provider keys accept the Hermes provider id (as used elsewhere in +# config.yaml) or the models.dev provider id. Model ids match exactly, +# then case-insensitively (mirroring catalog lookup). _OVERRIDE_CACHE: Optional[Dict[str, Any]] = None -_OVERRIDE_CACHE_CFG_HASH: int = 0 +_OVERRIDE_CACHE_CFG_ID: int = 0 +_OVERRIDE_WARNED_KEYS: set = set() def _load_model_overrides() -> Dict[str, Any]: @@ -621,76 +638,210 @@ def _load_model_overrides() -> Dict[str, Any]: Caches by ``id(cfg)`` so a config reload (new dict identity) invalidates automatically. Returns empty dict on any failure. """ - global _OVERRIDE_CACHE, _OVERRIDE_CACHE_CFG_HASH + global _OVERRIDE_CACHE, _OVERRIDE_CACHE_CFG_ID try: from hermes_cli.config import cfg_get, load_config_readonly cfg = load_config_readonly() cfg_id = id(cfg) - if cfg_id == _OVERRIDE_CACHE_CFG_HASH and _OVERRIDE_CACHE is not None: + if cfg_id == _OVERRIDE_CACHE_CFG_ID and _OVERRIDE_CACHE is not None: return _OVERRIDE_CACHE raw = cfg_get(cfg, "model_overrides", default={}) overrides = raw if isinstance(raw, dict) else {} _OVERRIDE_CACHE = overrides - _OVERRIDE_CACHE_CFG_HASH = cfg_id + _OVERRIDE_CACHE_CFG_ID = cfg_id return overrides except Exception: return {} -def _resolve_model_override( - provider: str, model: str -) -> Optional[Dict[str, Any]]: - """Resolve the override dict for a provider+model, or None. +def _provider_override_section(provider: str) -> Optional[Dict[str, Any]]: + """Return the override section for *provider*, or None. - Checks per-provider+model, then per-provider ``_default``, then global - ``_default``. Returns the first match (which may be partially populated — - callers only read the keys they care about). + Accepts either the Hermes provider id or the models.dev provider id as + the config key, so ``copilot`` and ``github-copilot`` both work + regardless of which id space a caller passes in. """ overrides = _load_model_overrides() if not overrides: return None - provider_key = (provider or "").strip() - model_key = (model or "").strip() - if not provider_key and not model_key: + if not provider_key: return None - # 1. Per-provider+model - provider_section = overrides.get(provider_key) - if isinstance(provider_section, dict) and model_key: - model_section = provider_section.get(model_key) - if isinstance(model_section, dict): - return model_section - - # 2. Per-provider _default - if isinstance(provider_section, dict): - default = provider_section.get("_default") - if isinstance(default, dict): - return default - - # 3. Global _default - global_default = overrides.get("_default") - if isinstance(global_default, dict): - return global_default + candidates = [provider_key] + mapped = PROVIDER_TO_MODELS_DEV.get(provider_key) + if mapped and mapped != provider_key: + candidates.append(mapped) + # Reverse: caller passed a models.dev id, config keyed by Hermes id. + for hermes_id, mdev_id in PROVIDER_TO_MODELS_DEV.items(): + if mdev_id == provider_key and hermes_id != provider_key: + candidates.append(hermes_id) + for key in candidates: + section = overrides.get(key) + if isinstance(section, dict): + return section return None -def _override_context_window( - provider: str, model: str -) -> Optional[int]: - """Return the overridden context_window, or None.""" - ov = _resolve_model_override(provider, model) - if ov is None: +def _explicit_model_override(provider: str, model: str) -> Optional[Dict[str, Any]]: + """Return the explicit per-provider+model override dict, or None. + + Model ids match exactly first, then case-insensitively (skipping the + ``_default`` sentinel), mirroring catalog lookup behavior. + """ + model_key = (model or "").strip() + if not model_key: return None - raw = ov.get("context_window") + section = _provider_override_section(provider) + if section is None: + return None + + entry = section.get(model_key) + if isinstance(entry, dict): + return entry + + model_lower = model_key.lower() + for mid, mdata in section.items(): + if mid == "_default": + continue + if mid.lower() == model_lower and isinstance(mdata, dict): + return mdata + return None + + +def _default_model_override(provider: str) -> Optional[Dict[str, Any]]: + """Return the fill-gap ``_default`` override for *provider*, or None. + + Checks the per-provider ``_default`` first, then the global one. Only + consulted for models the catalog does not know — see the block comment. + """ + section = _provider_override_section(provider) + if section is not None: + default = section.get("_default") + if isinstance(default, dict): + return default + overrides = _load_model_overrides() + global_default = overrides.get("_default") + if isinstance(global_default, dict): + return global_default + return None + + +def _override_for( + provider: str, model: str, *, catalog_hit: bool +) -> Optional[Dict[str, Any]]: + """Select the override dict for a lookup, honoring fill-gap semantics. + + Explicit per-provider+model overrides always apply. ``_default`` + entries apply only when the catalog has no entry for the model. + """ + explicit = _explicit_model_override(provider, model) + if explicit is not None: + return explicit + if catalog_hit: + return None + return _default_model_override(provider) + + +def _override_int(override: Dict[str, Any], key: str) -> Optional[int]: + """Coerce an override field to a positive int, warning once on garbage.""" + raw = override.get(key) if raw is None: return None try: - ctx = int(raw) - return ctx if ctx > 0 else None + value = int(raw) + if value > 0: + return value except (TypeError, ValueError): + pass + warn_key = (key, repr(raw)) + if warn_key not in _OVERRIDE_WARNED_KEYS: + _OVERRIDE_WARNED_KEYS.add(warn_key) + logger.warning( + "model_overrides: ignoring invalid %s value %r " + "(expected a positive integer)", key, raw, + ) + return None + + +def _override_context_window(provider: str, model: str) -> Optional[int]: + """Return the EXPLICITLY overridden context_window, or None. + + Explicit-only on purpose: this runs early in the resolution chain + (agent/model_metadata.py step 0b, before custom_providers and live + probes), where a ``_default`` must not preempt more specific sources. + Fill-gap defaults are applied later by ``lookup_models_dev_context`` + once the catalog has actually missed. + """ + ov = _explicit_model_override(provider, model) + if ov is None: return None + return _override_int(ov, "context_window") + + +def _override_to_catalog_shape(override: Dict[str, Any]) -> Dict[str, Any]: + """Translate canonical override keys into a models.dev-shaped patch. + + ``get_model_info``/``_parse_model_info`` consume the raw catalog shape + (``limit.context``, ``tool_call``, ...). All override consumers accept + ONE canonical schema (the documented ``context_window``/``supports_*`` + keys), so this boundary translates rather than forcing users to know + the internal catalog shape. + """ + patch: Dict[str, Any] = {} + limit: Dict[str, Any] = {} + ctx = _override_int(override, "context_window") + if ctx is not None: + limit["context"] = ctx + out = _override_int(override, "max_output_tokens") + if out is not None: + limit["output"] = out + if limit: + patch["limit"] = limit + if "supports_tools" in override: + patch["tool_call"] = bool(override["supports_tools"]) + if "supports_reasoning" in override: + patch["reasoning"] = bool(override["supports_reasoning"]) + if "supports_vision" in override: + patch["attachment"] = bool(override["supports_vision"]) + patch["_vision_override"] = bool(override["supports_vision"]) + if "model_family" in override: + patch["family"] = str(override["model_family"] or "") + return patch + + +def _merge_catalog_entry_with_override( + raw: Dict[str, Any], override: Dict[str, Any] +) -> Dict[str, Any]: + """Patch a catalog entry with a canonical-schema override. + + Sub-dicts (``limit``, ``modalities``) are merged, not clobbered — an + override setting only ``context_window`` must not wipe the catalog's + ``limit.output``. + """ + shaped = _override_to_catalog_shape(override) + merged = dict(raw) + limit_patch = shaped.pop("limit", None) + if limit_patch: + base_limit = raw.get("limit") + base_limit = dict(base_limit) if isinstance(base_limit, dict) else {} + base_limit.update(limit_patch) + merged["limit"] = base_limit + vision_override = shaped.pop("_vision_override", None) + if vision_override is not None: + base_mods = raw.get("modalities") + base_mods = dict(base_mods) if isinstance(base_mods, dict) else {} + input_mods = base_mods.get("input") + input_mods = list(input_mods) if isinstance(input_mods, list) else [] + if vision_override and "image" not in input_mods: + input_mods.append("image") + elif not vision_override and "image" in input_mods: + input_mods.remove("image") + base_mods["input"] = input_mods + merged["modalities"] = base_mods + merged.update(shaped) + return merged def _get_provider_models(provider: str) -> Optional[Dict[str, Any]]: @@ -736,14 +887,13 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit Uses the existing fetch_models_dev() and PROVIDER_TO_MODELS_DEV mapping. Returns None if model not found. - ``model_overrides`` config entries (per-provider+model, per-provider - ``_default``, or global ``_default``) win over catalog values. For a - model id NOT in the catalog, the override is the only source of - metadata — this is the supported self-unblock path for custom/local - models (#8731) and for models with wrong context in models.dev - (#84482). An override may set any subset of fields; unspecified fields - fall through to the catalog value (or sensible defaults when the model - is absent from the catalog entirely). + EXPLICIT ``model_overrides`` entries (per-provider+model) win over + catalog values for the fields they set. ``_default`` entries fill the + gap only for models the catalog does not know — the supported + self-unblock path for custom/local models (#8731) and for models with + wrong metadata in models.dev (#84482). An override may set any subset + of fields; unspecified fields fall through to the catalog value (or + sensible defaults when the model is absent from the catalog). Extracts from model entry fields: - reasoning (bool) → supports_reasoning @@ -753,14 +903,13 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit - limit.output (int) → max_output_tokens - family (str) → model_family """ - # Check config override first — it may fully replace the catalog entry - # or patch specific fields. For unknown models (not in catalog), the - # override is the sole source of metadata. - override = _resolve_model_override(provider, model) - models = _get_provider_models(provider) entry = _find_model_entry(models, model) if models is not None else None + # Select the override AFTER the catalog lookup: explicit overrides + # always apply; _default entries only fill gaps for catalog misses. + override = _override_for(provider, model, catalog_hit=entry is not None) + # If no catalog entry and no override, we can't resolve capabilities. if entry is None and override is None: return None @@ -812,20 +961,12 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit supports_vision = bool(override["supports_vision"]) if "supports_reasoning" in override: supports_reasoning = bool(override["supports_reasoning"]) - if "context_window" in override: - try: - ctx_ov = int(override["context_window"]) - if ctx_ov > 0: - context_window = ctx_ov - except (TypeError, ValueError): - pass - if "max_output_tokens" in override: - try: - out_ov = int(override["max_output_tokens"]) - if out_ov > 0: - max_output_tokens = out_ov - except (TypeError, ValueError): - pass + ctx_ov = _override_int(override, "context_window") + if ctx_ov is not None: + context_window = ctx_ov + out_ov = _override_int(override, "max_output_tokens") + if out_ov is not None: + max_output_tokens = out_ov if "model_family" in override: model_family = str(override["model_family"] or "") @@ -1040,50 +1181,52 @@ def get_model_info( Accepts Hermes or models.dev provider ID. Tries exact match then case-insensitive fallback. Returns None if not found. - ``model_overrides`` config entries (per-provider+model, per-provider - ``_default``, or global ``_default``) patch the catalog entry's fields - when present. For a model id NOT in the catalog, the override is the - sole source of metadata — this is the supported self-unblock path - for custom/local models (#8731) and for models with wrong context - in models.dev (#84482). + ``model_overrides`` entries use the SAME canonical schema as every + other consumer (``context_window``, ``max_output_tokens``, + ``supports_*``, ``model_family``) — they are translated into the + catalog shape at this boundary, and sub-dicts (``limit``, + ``modalities``) are merged rather than clobbered. EXPLICIT entries + patch known catalog models; ``_default`` entries fill the gap only + for models the catalog does not know (#8731, #84482). """ - override = _resolve_model_override(provider_id, model_id) - mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) + def _from_override_alone() -> Optional[ModelInfo]: + override = _override_for(provider_id, model_id, catalog_hit=False) + if override is None: + return None + shaped = _merge_catalog_entry_with_override({}, override) + shaped.pop("_vision_override", None) + return _parse_model_info(model_id, shaped, mdev_id) + data = fetch_models_dev() pdata = data.get(mdev_id) if not isinstance(pdata, dict): - # No catalog data — return from override alone if we have one. - if override is not None: - return _parse_model_info(model_id, override, mdev_id) - return None + return _from_override_alone() models = pdata.get("models", {}) if not isinstance(models, dict): + return _from_override_alone() + + def _with_override(mid: str, raw: Dict[str, Any]) -> ModelInfo: + override = _override_for(provider_id, model_id, catalog_hit=True) if override is not None: - return _parse_model_info(model_id, override, mdev_id) - return None + merged = _merge_catalog_entry_with_override(raw, override) + merged.pop("_vision_override", None) + return _parse_model_info(mid, merged, mdev_id) + return _parse_model_info(mid, raw, mdev_id) # Exact match raw = models.get(model_id) if isinstance(raw, dict): - if override is not None: - merged = {**raw, **override} - return _parse_model_info(model_id, merged, mdev_id) - return _parse_model_info(model_id, raw, mdev_id) + return _with_override(model_id, raw) # Case-insensitive fallback model_lower = model_id.lower() for mid, mdata in models.items(): if mid.lower() == model_lower and isinstance(mdata, dict): - if override is not None: - merged = {**mdata, **override} - return _parse_model_info(mid, merged, mdev_id) - return _parse_model_info(mid, mdata, mdev_id) + return _with_override(mid, mdata) - # Model not in catalog — return from override alone if we have one. - if override is not None: - return _parse_model_info(model_id, override, mdev_id) - - return None + # Model not in catalog — an override (explicit or _default) may still + # provide the metadata. + return _from_override_alone() diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 4094873737..e52290d05b 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2540,18 +2540,33 @@ DEFAULT_CONFIG = { }, # Per-model metadata overrides — manually declare context_window, - # max_output_tokens, capabilities, or cost for any provider+model. - # Overrides win over models.dev, OpenRouter, and hardcoded defaults. + # max_output_tokens, capabilities, or model family for any + # provider+model. Recognized fields: context_window, + # max_output_tokens, supports_tools, supports_vision, + # supports_reasoning, model_family. # - # Two scopes: - # 1. Per-provider+model: model_overrides.. - # 2. Per-provider default: model_overrides.._default - # 3. Global default: model_overrides._default + # Semantics: + # 1. Explicit (model_overrides..): wins over + # models.dev, OpenRouter, and hardcoded defaults for the fields + # it sets. NOTE: an explicit model.context_length (global) and a + # custom_providers per-model context_length are user settings at + # other layers and are consulted in the resolution chain order + # documented in agent/model_metadata.py. + # 2. Fill-gap defaults (model_overrides.._default and + # model_overrides._default): apply ONLY to models the catalog + # does not know. They never displace catalog data for known + # models, so a _default cannot accidentally clamp every model + # of a provider. # - # An unknown model id (not in models.dev) inherits base metadata from - # its family/dated-snapshot entry before patching, so overriding a - # model the catalog doesn't know yet is the supported self-unblock - # path (#84482). + # An unknown model id (not in models.dev) starts from safe defaults + # (200K context, tools on, vision/reasoning off) and the override + # patches the fields it sets — overriding a model the catalog + # doesn't know yet is the supported self-unblock path (#84482, + # #8731). + # + # Provider keys accept the Hermes provider id (as used elsewhere in + # this file) or the models.dev provider id; model ids match + # case-insensitively. # # Example: # model_overrides: @@ -2566,7 +2581,7 @@ DEFAULT_CONFIG = { # supports_vision: true # supports_reasoning: false # supports_tools: true - # _default: + # _default: # fill-gap only: models not in the catalog # context_window: 128000 "model_overrides": {}, diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index ee34dee110..1bc3365b69 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -9,8 +9,10 @@ import pytest from agent.models_dev import ( PROVIDER_TO_MODELS_DEV, _extract_context, + _default_model_override, + _explicit_model_override, _override_context_window, - _resolve_model_override, + _override_for, fetch_models_dev, get_model_capabilities, get_model_info, @@ -407,7 +409,7 @@ class TestModelOverrides: import agent.models_dev as md return patch.object(md, "_load_model_overrides", return_value=overrides_dict) - # --- _resolve_model_override --- + # --- override resolution --- def test_per_provider_model_override(self): """Per-provider+model override is found first.""" @@ -417,39 +419,90 @@ class TestModelOverrides: }, } with self._setup_overrides(overrides): - result = _resolve_model_override("upstage", "solar-pro4") + result = _explicit_model_override("upstage", "solar-pro4") assert result is not None assert result["context_window"] == 524288 - def test_per_provider_default_fallback(self): - """Per-provider _default is used when model not found.""" + def test_explicit_override_case_insensitive_model(self): + """Model ids match case-insensitively, mirroring catalog lookup.""" + overrides = { + "upstage": { + "Solar-Pro4": {"context_window": 524288}, + }, + } + with self._setup_overrides(overrides): + result = _explicit_model_override("upstage", "solar-pro4") + assert result is not None + assert result["context_window"] == 524288 + + def test_provider_key_accepts_either_id_space(self): + """Override keyed by Hermes id resolves for models.dev id and back.""" + overrides = { + "copilot": { + "my-model": {"context_window": 111111}, + }, + } + with self._setup_overrides(overrides): + # Caller passes the models.dev id; config keyed by Hermes id. + result = _explicit_model_override("github-copilot", "my-model") + assert result is not None + assert result["context_window"] == 111111 + + overrides = { + "github-copilot": { + "my-model": {"context_window": 222222}, + }, + } + with self._setup_overrides(overrides): + # Caller passes the Hermes id; config keyed by models.dev id. + result = _explicit_model_override("copilot", "my-model") + assert result is not None + assert result["context_window"] == 222222 + + def test_default_fills_gap_for_unknown_model(self): + """_default applies to models the catalog does not know.""" overrides = { "upstage": { "_default": {"context_window": 128000}, }, } with self._setup_overrides(overrides): - result = _resolve_model_override("upstage", "unknown-model") + result = _override_for("upstage", "unknown-model", catalog_hit=False) assert result is not None assert result["context_window"] == 128000 + def test_default_does_not_clamp_catalog_known_model(self): + """FILL-GAP semantics: _default never displaces catalog data. + + A `_default: {context_window: 128000}` must not clamp every + catalog-known model of the provider — it only fills catalog misses. + """ + overrides = { + "upstage": { + "_default": {"context_window": 128000}, + }, + "_default": {"context_window": 65536}, + } + with self._setup_overrides(overrides): + result = _override_for("upstage", "known-model", catalog_hit=True) + assert result is None + def test_global_default_fallback(self): - """Global _default is used when provider not found.""" + """Global _default is used when provider has no section.""" overrides = { "_default": {"context_window": 65536}, } with self._setup_overrides(overrides): - result = _resolve_model_override("unknown-provider", "unknown-model") + result = _default_model_override("unknown-provider") assert result is not None assert result["context_window"] == 65536 def test_no_override_returns_none(self): - """No override found returns None.""" with self._setup_overrides({}): - result = _resolve_model_override("anthropic", "claude-sonnet-4") - assert result is None + assert _explicit_model_override("anthropic", "claude-sonnet-4") is None + assert _default_model_override("anthropic") is None - def test_per_provider_model_beats_default(self): + def test_explicit_beats_default(self): """Per-provider+model wins over per-provider _default.""" overrides = { "upstage": { @@ -458,12 +511,11 @@ class TestModelOverrides: }, } with self._setup_overrides(overrides): - result = _resolve_model_override("upstage", "solar-pro4") + result = _override_for("upstage", "solar-pro4", catalog_hit=False) assert result is not None assert result["context_window"] == 524288 def test_per_provider_default_beats_global(self): - """Per-provider _default wins over global _default.""" overrides = { "upstage": { "_default": {"context_window": 128000}, @@ -471,11 +523,11 @@ class TestModelOverrides: "_default": {"context_window": 65536}, } with self._setup_overrides(overrides): - result = _resolve_model_override("upstage", "unknown-model") + result = _default_model_override("upstage") assert result is not None assert result["context_window"] == 128000 - # --- _override_context_window --- + # --- _override_context_window (explicit-only, early-chain) --- def test_override_context_window_returns_value(self): overrides = { @@ -502,6 +554,35 @@ class TestModelOverrides: ctx = _override_context_window("upstage", "bad-model") assert ctx is None + def test_override_context_window_ignores_default(self): + """Early-chain lookup is explicit-only: a _default must not preempt + more specific sources (custom_providers, live probes).""" + overrides = { + "upstage": { + "_default": {"context_window": 128000}, + }, + } + with self._setup_overrides(overrides): + ctx = _override_context_window("upstage", "syn-pro") + assert ctx is None + + def test_malformed_context_window_warns_once(self, caplog): + """Garbage values are rejected with a one-shot warning, not silence.""" + import logging + + import agent.models_dev as md + md._OVERRIDE_WARNED_KEYS.clear() + overrides = { + "upstage": { + "bad-model": {"context_window": "512k"}, + }, + } + with self._setup_overrides(overrides), caplog.at_level(logging.WARNING): + assert _override_context_window("upstage", "bad-model") is None + assert _override_context_window("upstage", "bad-model") is None + warnings = [r for r in caplog.records if "model_overrides" in r.message] + assert len(warnings) == 1 + # --- get_model_capabilities with overrides --- def test_caps_override_unknown_model(self): @@ -526,7 +607,7 @@ class TestModelOverrides: assert caps.supports_tools is True def test_caps_override_patches_existing_catalog_entry(self): - """Override patches specific fields on a known catalog entry (#84482).""" + """Explicit override patches specific fields on a known entry (#84482).""" overrides = { "anthropic": { "claude-sonnet-4": { @@ -544,8 +625,20 @@ class TestModelOverrides: assert caps.supports_vision is True assert caps.supports_tools is True + def test_caps_default_does_not_clamp_catalog_model(self): + """A _default must not displace catalog data for known models.""" + overrides = { + "anthropic": { + "_default": {"context_window": 1000}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=CAPS_REGISTRY): + caps = get_model_capabilities("anthropic", "claude-sonnet-4") + assert caps is not None + assert caps.context_window != 1000 + def test_caps_no_override_no_catalog_returns_none(self): - """No override and no catalog entry → None.""" with self._setup_overrides({}), \ patch("agent.models_dev.fetch_models_dev", return_value={}): caps = get_model_capabilities("anthropic", "unknown-model") @@ -571,7 +664,6 @@ class TestModelOverrides: # --- lookup_models_dev_context with overrides --- def test_context_lookup_override_wins_over_catalog(self): - """Override context_window wins over models.dev catalog value.""" overrides = { "anthropic": { "claude-opus-4-6": {"context_window": 500000}, @@ -583,7 +675,6 @@ class TestModelOverrides: assert ctx == 500000 def test_context_lookup_override_for_unknown_provider(self): - """Override works for providers not in PROVIDER_TO_MODELS_DEV.""" overrides = { "upstage": { "solar-pro4": {"context_window": 524288}, @@ -594,18 +685,46 @@ class TestModelOverrides: ctx = lookup_models_dev_context("upstage", "solar-pro4") assert ctx == 524288 - # --- get_model_info with overrides --- + def test_context_lookup_default_fills_catalog_miss(self): + """_default supplies context for a model the catalog lacks.""" + overrides = { + "anthropic": { + "_default": {"context_window": 77777}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): + ctx = lookup_models_dev_context("anthropic", "model-not-in-catalog") + assert ctx == 77777 + + def test_context_lookup_default_does_not_clamp_catalog(self): + """_default must not beat a catalog-known model's real context.""" + overrides = { + "anthropic": { + "_default": {"context_window": 1000}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): + ctx = lookup_models_dev_context("anthropic", "claude-opus-4-6") + assert ctx == 1000000 # catalog value, not the _default + + # --- get_model_info with overrides (canonical schema) --- def test_model_info_override_for_unknown_model(self): - """Override provides full metadata for a model not in the catalog.""" + """Canonical-schema override provides metadata for an unknown model. + + Same key space as every other consumer — context_window, + max_output_tokens, supports_* — NOT the internal catalog shape. + """ overrides = { "custom:my-vllm": { "my-llava-model": { - "name": "My LLaVA Model", - "family": "llava", - "reasoning": False, - "tool_call": True, - "limit": {"context": 8192, "output": 4096}, + "model_family": "llava", + "supports_reasoning": False, + "supports_tools": True, + "context_window": 8192, + "max_output_tokens": 4096, }, }, } @@ -613,7 +732,6 @@ class TestModelOverrides: patch("agent.models_dev.fetch_models_dev", return_value={}): info = get_model_info("custom:my-vllm", "my-llava-model") assert info is not None - assert info.name == "My LLaVA Model" assert info.family == "llava" assert info.context_window == 8192 assert info.max_output == 4096 @@ -621,11 +739,15 @@ class TestModelOverrides: assert info.reasoning is False def test_model_info_override_merges_with_catalog(self): - """Override patches specific fields on a known catalog entry.""" + """Override patches context without clobbering the catalog's output. + + The limit sub-dict is MERGED: an override setting only + context_window preserves the catalog's limit.output. + """ overrides = { "anthropic": { "claude-sonnet-4-6": { - "limit": {"context": 500000, "output": 64000}, + "context_window": 500000, }, }, } @@ -633,8 +755,79 @@ class TestModelOverrides: patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): info = get_model_info("anthropic", "claude-sonnet-4-6") assert info is not None - # Override wins + # Override wins for the field it sets assert info.context_window == 500000 + # Sub-dict merge: catalog's limit.output survives + assert info.max_output == 64000 # Non-overridden fields preserved from catalog assert info.name == "claude-sonnet-4-6" + def test_model_info_default_does_not_clamp_catalog(self): + """_default fills gaps only — known models keep catalog metadata.""" + overrides = { + "anthropic": { + "_default": {"context_window": 1000}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=SAMPLE_REGISTRY): + info = get_model_info("anthropic", "claude-sonnet-4-6") + assert info is not None + assert info.context_window == 1000000 + + # --- e2e config plumbing (real config.yaml, no _load_model_overrides mock) --- + + def test_e2e_overrides_load_from_real_config_yaml(self, tmp_path, monkeypatch): + """The real config path works end-to-end: config.yaml on disk -> + load_config_readonly -> cfg_get -> override applied. + + Every other test mocks _load_model_overrides; this one exercises + the actual wiring (key name, cfg accessor, cache invalidation). + """ + import importlib + + import agent.models_dev as md + import hermes_cli.config as hc + + home = tmp_path / "hermes" + home.mkdir() + (home / "config.yaml").write_text( + "model_overrides:\n" + " upstage:\n" + " solar-pro4:\n" + " context_window: 524288\n", + encoding="utf-8", + ) + monkeypatch.setenv("HERMES_HOME", str(home)) + + # Reset caches that memoize config identity/paths. + monkeypatch.setattr(md, "_OVERRIDE_CACHE", None) + monkeypatch.setattr(md, "_OVERRIDE_CACHE_CFG_ID", 0) + hc_cache = getattr(hc, "_LOAD_CONFIG_CACHE", None) + if isinstance(hc_cache, dict): + hc_cache.clear() + raw_cache = getattr(hc, "_RAW_CONFIG_CACHE", None) + if isinstance(raw_cache, dict): + raw_cache.clear() + importlib.reload # no-op guard: modules stay loaded, caches cleared + + with patch("agent.models_dev.fetch_models_dev", return_value={}): + ctx = lookup_models_dev_context("upstage", "solar-pro4") + assert ctx == 524288 + + def test_model_info_vision_override_sets_input_modality(self): + """supports_vision: true surfaces as an image input modality.""" + overrides = { + "custom:my-vllm": { + "my-model": { + "supports_vision": True, + "context_window": 8192, + }, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + info = get_model_info("custom:my-vllm", "my-model") + assert info is not None + assert "image" in info.input_modalities + assert info.attachment is True From 67ab2f296862cd2f3ecce615add7b5d72684a811 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:56:30 +0530 Subject: [PATCH 017/748] refactor(models): simplify-pass follow-ups on model_overrides MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Deleted the id(cfg)-keyed _OVERRIDE_CACHE layer: id() is unique only among live objects, so a config reload could serve stale overrides forever when CPython reuses the freed dict's address. The upstream load_config_readonly is already (mtime,size)-cached (~1 stat/hit), so the local layer was redundant state with a correctness risk. - _override_to_catalog_shape returns (patch, vision) instead of smuggling an in-band _vision_override sentinel key through the merged dict; removed the two dead call-site pops. - _find_model_entry gains the :cloud/-cloud suffix fallback that lookup_models_dev_context already had, so 'catalog hit' means the same thing to every consumer — a suffix-keyed model (kimi-k2.6:cloud) now counts as KNOWN and keeps its catalog capabilities instead of being displaced by a fill-gap _default (mutation-checked contract test added). - get_model_info's unknown-model override path seeds the same safe defaults as get_model_capabilities (200K ctx, tools on, 8192 out), so a partial override no longer yields ctx=0/tools-off on that path (contract test added); the DEFAULT_CONFIG defaults claim is now true for both paths. - Activated the previously-dead _MODELS_DEV_TO_PROVIDER reverse map (lazily built, many-to-one aware) and used it in _provider_override_section instead of a per-call linear scan. --- agent/models_dev.py | 94 +++++++++++++++++++++++----------- tests/agent/test_models_dev.py | 50 ++++++++++++++++-- 2 files changed, 112 insertions(+), 32 deletions(-) diff --git a/agent/models_dev.py b/agent/models_dev.py index 804694e7d8..e51b086a5a 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -197,8 +197,20 @@ PROVIDER_TO_MODELS_DEV: Dict[str, str] = { "ollama-cloud": "ollama-cloud", } -# Reverse mapping: models.dev → Hermes (built lazily) -_MODELS_DEV_TO_PROVIDER: Optional[Dict[str, str]] = None +# Reverse mapping: models.dev id → Hermes ids (built lazily; many-to-one, +# e.g. both "meta" and "meta-ai" may map to the same models.dev id). +_MODELS_DEV_TO_PROVIDER: Optional[Dict[str, List[str]]] = None + + +def _models_dev_to_hermes_ids(mdev_id: str) -> List[str]: + """Return the Hermes provider ids that map to *mdev_id* (may be []).""" + global _MODELS_DEV_TO_PROVIDER + if _MODELS_DEV_TO_PROVIDER is None: + reverse: Dict[str, List[str]] = {} + for hermes_id, mapped in PROVIDER_TO_MODELS_DEV.items(): + reverse.setdefault(mapped, []).append(hermes_id) + _MODELS_DEV_TO_PROVIDER = reverse + return _MODELS_DEV_TO_PROVIDER.get(mdev_id, []) @@ -627,29 +639,22 @@ class ModelCapabilities: # config.yaml) or the models.dev provider id. Model ids match exactly, # then case-insensitively (mirroring catalog lookup). -_OVERRIDE_CACHE: Optional[Dict[str, Any]] = None -_OVERRIDE_CACHE_CFG_ID: int = 0 _OVERRIDE_WARNED_KEYS: set = set() def _load_model_overrides() -> Dict[str, Any]: - """Load and cache the ``model_overrides`` config section. + """Load the ``model_overrides`` config section. - Caches by ``id(cfg)`` so a config reload (new dict identity) invalidates - automatically. Returns empty dict on any failure. + No local memoization on purpose: ``load_config_readonly()`` is already + (mtime, size)-cached upstream (a hit is ~one stat, no deepcopy, no + parse), and an ``id(cfg)``-keyed layer here can serve stale overrides + after a config reload when CPython reuses the freed dict's address. + Returns empty dict on any failure. """ - global _OVERRIDE_CACHE, _OVERRIDE_CACHE_CFG_ID try: from hermes_cli.config import cfg_get, load_config_readonly - cfg = load_config_readonly() - cfg_id = id(cfg) - if cfg_id == _OVERRIDE_CACHE_CFG_ID and _OVERRIDE_CACHE is not None: - return _OVERRIDE_CACHE - raw = cfg_get(cfg, "model_overrides", default={}) - overrides = raw if isinstance(raw, dict) else {} - _OVERRIDE_CACHE = overrides - _OVERRIDE_CACHE_CFG_ID = cfg_id - return overrides + raw = cfg_get(load_config_readonly(), "model_overrides", default={}) + return raw if isinstance(raw, dict) else {} except Exception: return {} @@ -673,8 +678,8 @@ def _provider_override_section(provider: str) -> Optional[Dict[str, Any]]: if mapped and mapped != provider_key: candidates.append(mapped) # Reverse: caller passed a models.dev id, config keyed by Hermes id. - for hermes_id, mdev_id in PROVIDER_TO_MODELS_DEV.items(): - if mdev_id == provider_key and hermes_id != provider_key: + for hermes_id in _models_dev_to_hermes_ids(provider_key): + if hermes_id != provider_key: candidates.append(hermes_id) for key in candidates: @@ -780,7 +785,9 @@ def _override_context_window(provider: str, model: str) -> Optional[int]: return _override_int(ov, "context_window") -def _override_to_catalog_shape(override: Dict[str, Any]) -> Dict[str, Any]: +def _override_to_catalog_shape( + override: Dict[str, Any], +) -> Tuple[Dict[str, Any], Optional[bool]]: """Translate canonical override keys into a models.dev-shaped patch. ``get_model_info``/``_parse_model_info`` consume the raw catalog shape @@ -788,6 +795,10 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Dict[str, Any]: ONE canonical schema (the documented ``context_window``/``supports_*`` keys), so this boundary translates rather than forcing users to know the internal catalog shape. + + Returns ``(patch, vision)`` — vision is returned out-of-band (not as + a key in the patch) because it maps onto the catalog's + ``modalities.input`` list rather than a scalar field. """ patch: Dict[str, Any] = {} limit: Dict[str, Any] = {} @@ -803,12 +814,13 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Dict[str, Any]: patch["tool_call"] = bool(override["supports_tools"]) if "supports_reasoning" in override: patch["reasoning"] = bool(override["supports_reasoning"]) + vision: Optional[bool] = None if "supports_vision" in override: - patch["attachment"] = bool(override["supports_vision"]) - patch["_vision_override"] = bool(override["supports_vision"]) + vision = bool(override["supports_vision"]) + patch["attachment"] = vision if "model_family" in override: patch["family"] = str(override["model_family"] or "") - return patch + return patch, vision def _merge_catalog_entry_with_override( @@ -820,7 +832,7 @@ def _merge_catalog_entry_with_override( override setting only ``context_window`` must not wipe the catalog's ``limit.output``. """ - shaped = _override_to_catalog_shape(override) + shaped, vision_override = _override_to_catalog_shape(override) merged = dict(raw) limit_patch = shaped.pop("limit", None) if limit_patch: @@ -828,7 +840,6 @@ def _merge_catalog_entry_with_override( base_limit = dict(base_limit) if isinstance(base_limit, dict) else {} base_limit.update(limit_patch) merged["limit"] = base_limit - vision_override = shaped.pop("_vision_override", None) if vision_override is not None: base_mods = raw.get("modalities") base_mods = dict(base_mods) if isinstance(base_mods, dict) else {} @@ -866,7 +877,15 @@ def _get_provider_models(provider: str) -> Optional[Dict[str, Any]]: def _find_model_entry(models: Dict[str, Any], model: str) -> Optional[Dict[str, Any]]: - """Find a model entry by exact match, then case-insensitive fallback.""" + """Find a model entry: exact, case-insensitive, then suffix fallback. + + The ``:cloud``/``-cloud`` suffix fallback mirrors + ``lookup_models_dev_context`` so "is this model in the catalog" means + the same thing to every consumer — important for ``model_overrides`` + fill-gap ``_default`` semantics, where a suffix-keyed catalog model + (e.g. ``kimi-k2.6:cloud``) must count as KNOWN and keep its catalog + metadata rather than being displaced by a ``_default``. + """ # Exact match entry = models.get(model) if isinstance(entry, dict): @@ -878,6 +897,17 @@ def _find_model_entry(models: Dict[str, Any], model: str) -> Optional[Dict[str, if mid.lower() == model_lower and isinstance(mdata, dict): return mdata + # Suffix-aware fallback (e.g. ollama-cloud stores kimi-k2.6:cloud + # while the live API returns the bare name). + for suffix in (":cloud", "-cloud"): + entry = models.get(model + suffix) + if isinstance(entry, dict): + return entry + suffixed_lower = model_lower + suffix + for mid, mdata in models.items(): + if mid.lower() == suffixed_lower and isinstance(mdata, dict): + return mdata + return None @@ -1195,8 +1225,15 @@ def get_model_info( override = _override_for(provider_id, model_id, catalog_hit=False) if override is None: return None - shaped = _merge_catalog_entry_with_override({}, override) - shaped.pop("_vision_override", None) + # Seed the same safe defaults get_model_capabilities uses for + # unknown models (200K context, tools on) so the two + # unknown-model paths agree; the override patches its fields on + # top. + base = { + "limit": {"context": 200000, "output": 8192}, + "tool_call": True, + } + shaped = _merge_catalog_entry_with_override(base, override) return _parse_model_info(model_id, shaped, mdev_id) data = fetch_models_dev() @@ -1212,7 +1249,6 @@ def get_model_info( override = _override_for(provider_id, model_id, catalog_hit=True) if override is not None: merged = _merge_catalog_entry_with_override(raw, override) - merged.pop("_vision_override", None) return _parse_model_info(mid, merged, mdev_id) return _parse_model_info(mid, raw, mdev_id) diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 1bc3365b69..67e008b659 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -800,9 +800,8 @@ class TestModelOverrides: ) monkeypatch.setenv("HERMES_HOME", str(home)) - # Reset caches that memoize config identity/paths. - monkeypatch.setattr(md, "_OVERRIDE_CACHE", None) - monkeypatch.setattr(md, "_OVERRIDE_CACHE_CFG_ID", 0) + # Reset caches that memoize config paths (the override layer has + # no local cache — it rides load_config_readonly's mtime cache). hc_cache = getattr(hc, "_LOAD_CONFIG_CACHE", None) if isinstance(hc_cache, dict): hc_cache.clear() @@ -815,6 +814,51 @@ class TestModelOverrides: ctx = lookup_models_dev_context("upstage", "solar-pro4") assert ctx == 524288 + def test_suffix_keyed_model_counts_as_catalog_hit(self): + """A suffix-keyed catalog model (kimi-k2.6:cloud) is KNOWN: a + _default must not displace its capabilities.""" + registry = { + "ollama-cloud": { + "id": "ollama-cloud", + "models": { + "kimi-k2.6:cloud": { + "id": "kimi-k2.6:cloud", + "tool_call": True, + "limit": {"context": 262144, "output": 8192}, + }, + }, + }, + } + overrides = { + "ollama-cloud": { + "_default": {"context_window": 1000, "supports_tools": False}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value=registry): + caps = get_model_capabilities("ollama-cloud", "kimi-k2.6") + assert caps is not None + assert caps.context_window == 262144 # catalog, not the _default + assert caps.supports_tools is True + + def test_model_info_unknown_model_gets_safe_defaults(self): + """get_model_info's unknown-model path seeds the same safe + defaults as get_model_capabilities (200K/tools-on), so a partial + override doesn't yield ctx=0/tools-off.""" + overrides = { + "custom:my-vllm": { + "my-model": {"supports_reasoning": True}, + }, + } + with self._setup_overrides(overrides), \ + patch("agent.models_dev.fetch_models_dev", return_value={}): + info = get_model_info("custom:my-vllm", "my-model") + assert info is not None + assert info.context_window == 200000 + assert info.max_output == 8192 + assert info.tool_call is True + assert info.reasoning is True + def test_model_info_vision_override_sets_input_modality(self): """supports_vision: true surfaces as an image input modality.""" overrides = { From ccaaca7e66b01dc0ac57a0b274ced7cd6b19982b Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:40:54 +0530 Subject: [PATCH 018/748] =?UTF-8?q?fix(usage):=20cost=20display=20honesty?= =?UTF-8?q?=20=E2=80=94=20sub-cent=20labels,=20cost=20buckets,=20included?= =?UTF-8?q?=20notes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three cost-display honesty fixes: 1. Sub-cent cost label rendering (#79220) — _format_cost_label() scales precision to magnitude: zero renders as '$0.00', sub-cent (< $0.01) renders at 4 decimal places (e.g. '~$0.0046'), normal costs keep 2dp. This fixes the bug where DeepSeek per-turn costs of $0.004640 rendered as '~$0.00' despite amount_usd carrying full Decimal precision. 2. Cost bucket surfacing (#77223) — insights format_terminal and format_gateway now display three cost buckets: estimated (with dollar figure), included (session count, labeled 'subscription — no provider invoice'), and unknown (session count, labeled 'no pricing data'). Previously, included and unknown sessions silently collapsed to $0 in the aggregate view, hiding 315 of 473 sessions in the reporter's DB. 3. Subscription-included cost notes — estimate_usage_cost now attaches a 'subscription-included; no provider invoice for usage' note to CostResult for subscription-included routes (openai-codex), so consumers can distinguish 'free because subscription' from 'free because $0 pricing'. Fixes #79220 Fixes #77223 --- agent/insights.py | 38 ++++++++++++++ agent/usage_pricing.py | 26 +++++++++- tests/agent/test_insights.py | 83 +++++++++++++++++++++++++++++-- tests/agent/test_usage_pricing.py | 76 ++++++++++++++++++++++++++++ 4 files changed, 219 insertions(+), 4 deletions(-) diff --git a/agent/insights.py b/agent/insights.py index 34e78a6ff9..6962d13427 100644 --- a/agent/insights.py +++ b/agent/insights.py @@ -1000,6 +1000,29 @@ class InsightsEngine: lines.append(f" Avg msgs/session: {o['avg_messages_per_session']:.1f}") lines.append("") + # Cost breakdown — surface the three buckets so subscription-included + # and unknown-cost sessions are visible instead of silently collapsing + # to $0. See #77223. + est_cost = o.get("estimated_cost", 0.0) + included_sessions = o.get("included_cost_sessions", 0) + unknown_sessions = o.get("unknown_cost_sessions", 0) + if est_cost > 0 or included_sessions > 0 or unknown_sessions > 0: + lines.append(" 💰 Cost") + lines.append(" " + "─" * 56) + if est_cost > 0: + lines.append(f" Estimated: ~${est_cost:.2f}") + if included_sessions > 0: + lines.append( + f" Included: {included_sessions} session(s) " + f"(subscription — no provider invoice)" + ) + if unknown_sessions > 0: + lines.append( + f" Unknown: {unknown_sessions} session(s) " + f"(no pricing data)" + ) + lines.append("") + # Model breakdown if report["models"]: lines.append(" 🤖 Models Used") @@ -1114,6 +1137,21 @@ class InsightsEngine: lines.append(f"**Active time:** ~{format_duration_compact(o['total_hours'] * 3600)} | **Avg session:** ~{format_duration_compact(o['avg_session_duration'])}") lines.append("") + # Cost breakdown — surface buckets so included/unknown are visible + est_cost = o.get("estimated_cost", 0.0) + included = o.get("included_cost_sessions", 0) + unknown = o.get("unknown_cost_sessions", 0) + cost_parts: list[str] = [] + if est_cost > 0: + cost_parts.append(f"~${est_cost:.2f} estimated") + if included > 0: + cost_parts.append(f"{included} included (subscription)") + if unknown > 0: + cost_parts.append(f"{unknown} unknown") + if cost_parts: + lines.append(f"**Cost:** {' | '.join(cost_parts)}") + lines.append("") + # Models (top 5) if report["models"]: lines.append("**🤖 Models:**") diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 592f6742b6..6ef29b5ff0 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -18,6 +18,29 @@ _ZERO = Decimal("0") _ONE_MILLION = Decimal("1000000") _NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1" +# Sub-cent cost threshold: below $0.01, render at 4 decimal places so +# the display is non-zero (e.g. $0.0046 instead of $0.00). See #79220. +_SUBCENT_THRESHOLD = Decimal("0.01") + + +def _format_cost_label(amount: Decimal) -> str: + """Format a cost amount as a display label. + + Scales precision to magnitude: + - Zero → "$0.00" + - Sub-cent (< $0.01) → "~$0.0046" (4 dp, always non-zero) + - Normal → "~$1.23" (2 dp) + + This fixes #79220 where sub-cent per-turn costs on cheap models + (DeepSeek, etc.) rendered as "$0.00" despite amount_usd carrying + full Decimal precision. + """ + if amount == _ZERO: + return "$0.00" + if amount < _SUBCENT_THRESHOLD: + return f"~${amount:.4f}" + return f"~${amount:.2f}" + CostStatus = Literal["actual", "estimated", "included", "unknown"] CostSource = Literal[ "provider_cost_api", @@ -1334,6 +1357,7 @@ def estimate_usage_cost( source="none", label="included", pricing_version="included-route", + notes=("subscription-included; no provider invoice for usage",), ) entry = get_pricing_entry(model_name, provider=provider, base_url=base_url, api_key=api_key) @@ -1378,7 +1402,7 @@ def estimate_usage_cost( amount += Decimal(usage.request_count) * entry.request_cost status: CostStatus = "estimated" - label = f"~${amount:.2f}" + label = _format_cost_label(amount) if entry.source == "none" and amount == _ZERO: status = "included" label = "included" diff --git a/tests/agent/test_insights.py b/tests/agent/test_insights.py index 2d42ab62e1..2a3b12157a 100644 --- a/tests/agent/test_insights.py +++ b/tests/agent/test_insights.py @@ -557,7 +557,9 @@ class TestTerminalFormatting: assert "N/A" not in text assert "custom/self-hosted" not in text - assert "Cost" not in text + # Cost section now surfaces unknown-cost sessions (#77223) instead + # of hiding them — a custom model with no pricing data should show + # "Unknown: 1 session(s)" rather than silently reporting $0. class TestGatewayFormatting: @@ -571,12 +573,16 @@ class TestGatewayFormatting: def test_gateway_format_hides_cost(self, populated_db): - """Gateway format omits dollar figures and internal cache details.""" + """Gateway format omits internal cache details. + + Dollar figures now appear when there are estimated/included/unknown + cost buckets (#77223) — the old assertion that '$' is absent is no + longer correct because surfacing cost buckets is the fix. + """ engine = InsightsEngine(populated_db) report = engine.generate(days=30) text = engine.format_gateway(report) - assert "$" not in text assert "cache" not in text.lower() @@ -670,3 +676,74 @@ class TestEdgeCases: # Actually the condition is > 1 platforms OR non-cli, so single cli won't show + def test_cost_buckets_displayed_in_terminal_format(self, db): + """#77223: included/estimated/unknown cost buckets surface in terminal.""" + # Estimated cost session + db.create_session(session_id="est", source="cli", model="model-a") + db.update_token_counts( + "est", input_tokens=100, model="model-a", + billing_provider="custom", + estimated_cost_usd=1.50, actual_cost_usd=1.0, + cost_status="estimated", cost_source="provider", api_call_count=1, + ) + # Included cost session (subscription) + db.create_session(session_id="inc", source="cli", model="gpt-5.4-mini") + db.update_token_counts( + "inc", input_tokens=200, model="gpt-5.4-mini", + billing_provider="openai-codex", + estimated_cost_usd=0.0, actual_cost_usd=0.0, + cost_status="included", cost_source="none", api_call_count=1, + ) + + engine = InsightsEngine(db) + report = engine.generate(days=30) + text = engine.format_terminal(report) + + # The cost section should appear with all three buckets + assert "💰 Cost" in text + assert "~$1.50" in text # estimated + assert "included" in text.lower() + assert "subscription" in text.lower() + + def test_cost_buckets_displayed_in_gateway_format(self, db): + """#77223: included/estimated/unknown cost buckets surface in gateway.""" + db.create_session(session_id="est", source="cli", model="model-a") + db.update_token_counts( + "est", input_tokens=100, model="model-a", + billing_provider="custom", + estimated_cost_usd=2.25, actual_cost_usd=0.0, + cost_status="estimated", cost_source="provider", api_call_count=1, + ) + db.create_session(session_id="inc", source="cli", model="gpt-5.4-mini") + db.update_token_counts( + "inc", input_tokens=200, model="gpt-5.4-mini", + billing_provider="openai-codex", + estimated_cost_usd=0.0, actual_cost_usd=0.0, + cost_status="included", cost_source="none", api_call_count=1, + ) + + engine = InsightsEngine(db) + report = engine.generate(days=30) + text = engine.format_gateway(report) + + assert "**Cost:**" in text + assert "~$2.25" in text + assert "included" in text.lower() + + def test_no_cost_section_when_all_zero(self, db): + """A session with no model still shows unknown cost bucket (#77223). + + The unknown bucket is surfaced so users can see they have sessions + with no pricing data, rather than silently reporting $0. + """ + db.create_session(session_id="s1", source="cli", model="test") + db._conn.commit() + + engine = InsightsEngine(db) + report = engine.generate(days=30) + text = engine.format_terminal(report) + # The session has no cost data, so it falls in the "unknown" bucket. + assert "💰 Cost" in text + assert "Unknown" in text + + diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index 7939faaabf..d2bb8c4df7 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -2,11 +2,13 @@ from types import SimpleNamespace from agent.usage_pricing import ( CanonicalUsage, + _format_cost_label, estimate_usage_cost, get_pricing_entry, normalize_usage, resolve_billing_route, ) +from decimal import Decimal @@ -370,3 +372,77 @@ def test_normalize_usage_native_anthropic_no_cache_observability(caplog): assert result.input_tokens == 100 assert result.cache_read_tokens == 50 assert result.cache_write_tokens == 10 + + +# --------------------------------------------------------------------------- +# Cost label formatting (#79220: sub-cent costs render as $0.00) +# --------------------------------------------------------------------------- + + +class TestFormatCostLabel: + """Tests for magnitude-scaled cost label formatting.""" + + def test_zero_renders_as_dollar_zero(self): + assert _format_cost_label(Decimal("0")) == "$0.00" + + def test_sub_cent_renders_4dp(self): + """Costs below $0.01 render at 4 decimal places (#79220).""" + label = _format_cost_label(Decimal("0.004640")) + assert label == "~$0.0046" + # Must NOT be $0.00 + assert "$0.00" != label + + def test_exactly_one_cent_renders_2dp(self): + """$0.01 renders at 2dp.""" + assert _format_cost_label(Decimal("0.01")) == "~$0.01" + + def test_normal_cost_renders_2dp(self): + assert _format_cost_label(Decimal("1.23")) == "~$1.23" + + def test_large_cost_renders_2dp(self): + assert _format_cost_label(Decimal("42.50")) == "~$42.50" + + def test_very_small_sub_cent(self): + """Even very small costs render non-zero.""" + label = _format_cost_label(Decimal("0.0001")) + assert label == "~$0.0001" + assert label != "$0.00" + + def test_sub_cent_deepseek_scenario(self): + """Reproduce the #79220 reproduction: DeepSeek at $0.004640.""" + # DeepSeek V4 Pro: 8K input + 1.2K output + 32K cache read + # = $0.004640 per turn + amount = Decimal("0.004640") + label = _format_cost_label(amount) + assert "0.0046" in label + assert label != "$0.00" + assert label != "~$0.00" + + +# --------------------------------------------------------------------------- +# Subscription-included cost notes +# --------------------------------------------------------------------------- + + +class TestSubscriptionIncludedNotes: + """Subscription-included costs should carry a note clarifying no invoice.""" + + def test_included_cost_has_note(self): + """estimate_usage_cost for subscription-included route includes a note.""" + # openai-codex is subscription_included + usage = CanonicalUsage( + input_tokens=1000, + output_tokens=500, + cache_read_tokens=0, + cache_write_tokens=0, + reasoning_tokens=0, + ) + result = estimate_usage_cost( + "gpt-5.4-mini", + usage, + provider="openai-codex", + ) + assert result.status == "included" + assert result.amount_usd == Decimal("0") + assert len(result.notes) > 0 + assert any("subscription" in note.lower() for note in result.notes) From 2c068d7680df132f3cb06551bf1d618413ab7ade Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:18:34 +0530 Subject: [PATCH 019/748] fix(usage): close sub-cent gaps found in review MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Insights formatters now route aggregate estimated cost through the shared format_cost_label() instead of hardcoded 2dp — a sub-cent aggregate (one cheap DeepSeek session, ~$0.0046) no longer renders 'Estimated: ~$0.00', the exact bug class this PR fixes (#79220). - format_cost_label: positive amounts below $0.00005 render '~$<0.0001' instead of the zero-looking '~$0.0000' 4dp truncation artifact. - Renamed _format_cost_label -> format_cost_label (now a cross-module shared helper). - Tests: renamed test_gateway_format_hides_cost -> test_gateway_format_hides_cache_details and test_no_cost_section_when_all_zero -> test_unknown_bucket_shown_for_costless_session (names contradicted behavior); restored a real assertion in the custom-models test that had been weakened to a comment; added sub-cent-aggregate and 4dp-floor contract tests (mutation-checked). --- agent/insights.py | 16 +++++++++++-- agent/usage_pricing.py | 16 +++++++++---- tests/agent/test_insights.py | 40 +++++++++++++++++++++++++------ tests/agent/test_usage_pricing.py | 26 +++++++++++++------- 4 files changed, 77 insertions(+), 21 deletions(-) diff --git a/agent/insights.py b/agent/insights.py index 6962d13427..b1b9a06ef6 100644 --- a/agent/insights.py +++ b/agent/insights.py @@ -21,16 +21,28 @@ import sqlite3 import time from collections import Counter, defaultdict from datetime import datetime +from decimal import Decimal from typing import Any, Dict, List, Optional from agent.usage_pricing import ( CanonicalUsage, estimate_usage_cost, + format_cost_label, format_duration_compact, has_known_pricing, ) +def _fmt_est_cost(est_cost: float) -> str: + """Format an aggregate estimated cost via the shared cost-label helper. + + Routes through ``format_cost_label`` so sub-cent aggregates render at + 4dp instead of collapsing to "~$0.00" (#79220 bug class — the same + dishonesty this module's cost buckets exist to fix, #77223). + """ + return format_cost_label(Decimal(str(est_cost))) + + def _estimate_cost( @@ -1010,7 +1022,7 @@ class InsightsEngine: lines.append(" 💰 Cost") lines.append(" " + "─" * 56) if est_cost > 0: - lines.append(f" Estimated: ~${est_cost:.2f}") + lines.append(f" Estimated: {_fmt_est_cost(est_cost)}") if included_sessions > 0: lines.append( f" Included: {included_sessions} session(s) " @@ -1143,7 +1155,7 @@ class InsightsEngine: unknown = o.get("unknown_cost_sessions", 0) cost_parts: list[str] = [] if est_cost > 0: - cost_parts.append(f"~${est_cost:.2f} estimated") + cost_parts.append(f"{_fmt_est_cost(est_cost)} estimated") if included > 0: cost_parts.append(f"{included} included (subscription)") if unknown > 0: diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 6ef29b5ff0..d3e294f7b4 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -23,22 +23,30 @@ _NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1" _SUBCENT_THRESHOLD = Decimal("0.01") -def _format_cost_label(amount: Decimal) -> str: +def format_cost_label(amount: Decimal) -> str: """Format a cost amount as a display label. Scales precision to magnitude: - Zero → "$0.00" - - Sub-cent (< $0.01) → "~$0.0046" (4 dp, always non-zero) + - Sub-cent (< $0.01) → "~$0.0046" (4 dp; below $0.00005 falls back + to "~$<0.0001" so the label never reads as zero) - Normal → "~$1.23" (2 dp) This fixes #79220 where sub-cent per-turn costs on cheap models (DeepSeek, etc.) rendered as "$0.00" despite amount_usd carrying full Decimal precision. + + Shared by per-response cost labels (estimate_usage_cost) and the + insights cost-bucket formatters — keep both surfaces on this one + implementation so sub-cent honesty can't regress on one of them. """ if amount == _ZERO: return "$0.00" if amount < _SUBCENT_THRESHOLD: - return f"~${amount:.4f}" + label = f"~${amount:.4f}" + # 4dp truncation of a positive amount below $0.00005 would render + # "~$0.0000" — a zero-looking label, the exact #79220 dishonesty. + return label if label != "~$0.0000" else "~$<0.0001" return f"~${amount:.2f}" CostStatus = Literal["actual", "estimated", "included", "unknown"] @@ -1402,7 +1410,7 @@ def estimate_usage_cost( amount += Decimal(usage.request_count) * entry.request_cost status: CostStatus = "estimated" - label = _format_cost_label(amount) + label = format_cost_label(amount) if entry.source == "none" and amount == _ZERO: status = "included" label = "included" diff --git a/tests/agent/test_insights.py b/tests/agent/test_insights.py index 2a3b12157a..1c19abf43d 100644 --- a/tests/agent/test_insights.py +++ b/tests/agent/test_insights.py @@ -545,8 +545,8 @@ class TestTerminalFormatting: - def test_terminal_format_hides_cost_for_custom_models(self, db): - """Cost display is hidden entirely — custom models no longer show 'N/A' either.""" + def test_terminal_format_unknown_bucket_for_custom_models(self, db): + """Custom models with no pricing surface as the Unknown bucket (#77223).""" db.create_session(session_id="s1", source="cli", model="my-custom-model") db.update_token_counts("s1", input_tokens=1000, output_tokens=500) db._conn.commit() @@ -557,9 +557,11 @@ class TestTerminalFormatting: assert "N/A" not in text assert "custom/self-hosted" not in text - # Cost section now surfaces unknown-cost sessions (#77223) instead - # of hiding them — a custom model with no pricing data should show - # "Unknown: 1 session(s)" rather than silently reporting $0. + # Cost section surfaces unknown-cost sessions (#77223) instead of + # hiding them — a custom model with no pricing data shows in the + # Unknown bucket rather than silently reporting $0. + assert "Unknown" in text + assert "no pricing data" in text class TestGatewayFormatting: @@ -572,7 +574,7 @@ class TestGatewayFormatting: assert len(gateway_text) < len(terminal_text) - def test_gateway_format_hides_cost(self, populated_db): + def test_gateway_format_hides_cache_details(self, populated_db): """Gateway format omits internal cache details. Dollar figures now appear when there are estimated/included/unknown @@ -705,6 +707,30 @@ class TestEdgeCases: assert "included" in text.lower() assert "subscription" in text.lower() + def test_sub_cent_aggregate_estimated_cost_not_zero(self, db): + """A sub-cent aggregate must not render 'Estimated: ~$0.00' (#79220). + + The insights formatters share format_cost_label with per-response + labels; a cheap-model period totaling $0.0046 shows 4dp, not $0.00. + """ + db.create_session(session_id="est", source="cli", model="model-a") + db.update_token_counts( + "est", input_tokens=100, model="model-a", + billing_provider="custom", + estimated_cost_usd=0.0046, actual_cost_usd=0.0, + cost_status="estimated", cost_source="provider", api_call_count=1, + ) + + engine = InsightsEngine(db) + report = engine.generate(days=30) + terminal_text = engine.format_terminal(report) + gateway_text = engine.format_gateway(report) + + assert "~$0.00\n" not in terminal_text + assert "~$0.0046" in terminal_text + assert "~$0.00 estimated" not in gateway_text + assert "~$0.0046 estimated" in gateway_text + def test_cost_buckets_displayed_in_gateway_format(self, db): """#77223: included/estimated/unknown cost buckets surface in gateway.""" db.create_session(session_id="est", source="cli", model="model-a") @@ -730,7 +756,7 @@ class TestEdgeCases: assert "~$2.25" in text assert "included" in text.lower() - def test_no_cost_section_when_all_zero(self, db): + def test_unknown_bucket_shown_for_costless_session(self, db): """A session with no model still shows unknown cost bucket (#77223). The unknown bucket is surfaced so users can see they have sessions diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index d2bb8c4df7..fa0fcef5a6 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -2,7 +2,7 @@ from types import SimpleNamespace from agent.usage_pricing import ( CanonicalUsage, - _format_cost_label, + format_cost_label, estimate_usage_cost, get_pricing_entry, normalize_usage, @@ -383,37 +383,47 @@ class TestFormatCostLabel: """Tests for magnitude-scaled cost label formatting.""" def test_zero_renders_as_dollar_zero(self): - assert _format_cost_label(Decimal("0")) == "$0.00" + assert format_cost_label(Decimal("0")) == "$0.00" def test_sub_cent_renders_4dp(self): """Costs below $0.01 render at 4 decimal places (#79220).""" - label = _format_cost_label(Decimal("0.004640")) + label = format_cost_label(Decimal("0.004640")) assert label == "~$0.0046" # Must NOT be $0.00 assert "$0.00" != label def test_exactly_one_cent_renders_2dp(self): """$0.01 renders at 2dp.""" - assert _format_cost_label(Decimal("0.01")) == "~$0.01" + assert format_cost_label(Decimal("0.01")) == "~$0.01" def test_normal_cost_renders_2dp(self): - assert _format_cost_label(Decimal("1.23")) == "~$1.23" + assert format_cost_label(Decimal("1.23")) == "~$1.23" def test_large_cost_renders_2dp(self): - assert _format_cost_label(Decimal("42.50")) == "~$42.50" + assert format_cost_label(Decimal("42.50")) == "~$42.50" def test_very_small_sub_cent(self): """Even very small costs render non-zero.""" - label = _format_cost_label(Decimal("0.0001")) + label = format_cost_label(Decimal("0.0001")) assert label == "~$0.0001" assert label != "$0.00" + def test_below_4dp_floor_never_reads_zero(self): + """Amounts below $0.00005 must not render as '~$0.0000' (#79220). + + 4dp truncation of a positive amount would produce a zero-looking + label — the exact dishonesty the formatter exists to fix. + """ + label = format_cost_label(Decimal("0.00004")) + assert label == "~$<0.0001" + assert "0.0000" not in label.replace("<0.0001", "") + def test_sub_cent_deepseek_scenario(self): """Reproduce the #79220 reproduction: DeepSeek at $0.004640.""" # DeepSeek V4 Pro: 8K input + 1.2K output + 32K cache read # = $0.004640 per turn amount = Decimal("0.004640") - label = _format_cost_label(amount) + label = format_cost_label(amount) assert "0.0046" in label assert label != "$0.00" assert label != "~$0.00" From f80f453ae0679347e38abc917c7f94f717bf96c5 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:49:40 +0530 Subject: [PATCH 020/748] refactor(usage): simplify-pass follow-ups - Single-source the included note as _INCLUDED_NOTE and attach it at BOTH status='included' sites (the zero-amount pricing-entry branch previously returned the same status with no note). - Docstring/comment precision on format_cost_label: the fallback triggers on 4dp ROUNDING to 0.0000 (banker's rounding includes the exact $0.00005 boundary), not truncation; note why the rendered-label guard beats a naive Decimal threshold. - Tests: replaced a dead assertion with the exact-boundary case ($0.00005), fixed an overclaiming comment, aligned the terminal cost column. --- agent/insights.py | 2 +- agent/usage_pricing.py | 18 ++++++++++++++---- tests/agent/test_insights.py | 3 ++- tests/agent/test_usage_pricing.py | 4 +++- 4 files changed, 20 insertions(+), 7 deletions(-) diff --git a/agent/insights.py b/agent/insights.py index b1b9a06ef6..c1dec9e073 100644 --- a/agent/insights.py +++ b/agent/insights.py @@ -1022,7 +1022,7 @@ class InsightsEngine: lines.append(" 💰 Cost") lines.append(" " + "─" * 56) if est_cost > 0: - lines.append(f" Estimated: {_fmt_est_cost(est_cost)}") + lines.append(f" Estimated: {_fmt_est_cost(est_cost)}") if included_sessions > 0: lines.append( f" Included: {included_sessions} session(s) " diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index d3e294f7b4..5cacdb04a9 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -22,14 +22,20 @@ _NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1" # the display is non-zero (e.g. $0.0046 instead of $0.00). See #79220. _SUBCENT_THRESHOLD = Decimal("0.01") +# Attached to every CostResult with status="included" so consumers can +# distinguish "free because subscription" from "free because $0 pricing". +_INCLUDED_NOTE = "subscription-included; no provider invoice for usage" + def format_cost_label(amount: Decimal) -> str: """Format a cost amount as a display label. Scales precision to magnitude: - Zero → "$0.00" - - Sub-cent (< $0.01) → "~$0.0046" (4 dp; below $0.00005 falls back - to "~$<0.0001" so the label never reads as zero) + - Sub-cent (< $0.01) → "~$0.0046" (4 dp; amounts that ROUND to + 0.0000 at 4 dp — i.e. at or below $0.00005 under banker's + rounding — fall back to "~$<0.0001" so the label never reads + as zero) - Normal → "~$1.23" (2 dp) This fixes #79220 where sub-cent per-turn costs on cheap models @@ -44,8 +50,11 @@ def format_cost_label(amount: Decimal) -> str: return "$0.00" if amount < _SUBCENT_THRESHOLD: label = f"~${amount:.4f}" - # 4dp truncation of a positive amount below $0.00005 would render + # A positive amount that rounds to 0.0000 at 4 dp would render # "~$0.0000" — a zero-looking label, the exact #79220 dishonesty. + # Comparing the rendered label checks the truth directly (a naive + # `< 0.00005` threshold misses the exact boundary under + # ROUND_HALF_EVEN). return label if label != "~$0.0000" else "~$<0.0001" return f"~${amount:.2f}" @@ -1365,7 +1374,7 @@ def estimate_usage_cost( source="none", label="included", pricing_version="included-route", - notes=("subscription-included; no provider invoice for usage",), + notes=(_INCLUDED_NOTE,), ) entry = get_pricing_entry(model_name, provider=provider, base_url=base_url, api_key=api_key) @@ -1414,6 +1423,7 @@ def estimate_usage_cost( if entry.source == "none" and amount == _ZERO: status = "included" label = "included" + notes.append(_INCLUDED_NOTE) if route.provider == "openrouter": notes.append("OpenRouter cost is estimated from the models API until reconciled.") diff --git a/tests/agent/test_insights.py b/tests/agent/test_insights.py index 1c19abf43d..22de263908 100644 --- a/tests/agent/test_insights.py +++ b/tests/agent/test_insights.py @@ -701,7 +701,8 @@ class TestEdgeCases: report = engine.generate(days=30) text = engine.format_terminal(report) - # The cost section should appear with all three buckets + # The cost section should appear with the buckets this DB has + # (estimated + included; no unknown-cost session is created here) assert "💰 Cost" in text assert "~$1.50" in text # estimated assert "included" in text.lower() diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index fa0fcef5a6..ddb9d4cfbb 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -416,7 +416,9 @@ class TestFormatCostLabel: """ label = format_cost_label(Decimal("0.00004")) assert label == "~$<0.0001" - assert "0.0000" not in label.replace("<0.0001", "") + # Exact boundary: $0.00005 rounds to 0.0000 under ROUND_HALF_EVEN + # and must also take the fallback. + assert format_cost_label(Decimal("0.00005")) == "~$<0.0001" def test_sub_cent_deepseek_scenario(self): """Reproduce the #79220 reproduction: DeepSeek at $0.004640.""" From 17dc773156774c79cb434262b0f4699bf87ec7a0 Mon Sep 17 00:00:00 2001 From: Jeeves Assistant Date: Thu, 13 Aug 2026 07:39:17 -0500 Subject: [PATCH 021/748] fix(plugins): discover entrypoint capabilities --- hermes_cli/plugins.py | 132 +++++++++++++++---- tests/hermes_cli/test_plugin_capabilities.py | 40 ++++++ tests/hermes_cli/test_plugins_cmd_list.py | 33 +++++ 3 files changed, 181 insertions(+), 24 deletions(-) diff --git a/hermes_cli/plugins.py b/hermes_cli/plugins.py index 36b5ba24dc..5124516ca0 100644 --- a/hermes_cli/plugins.py +++ b/hermes_cli/plugins.py @@ -65,6 +65,7 @@ from hermes_cli.config import cfg_get, load_config_readonly from hermes_cli.middleware import OBSERVER_SCHEMA_VERSION, VALID_MIDDLEWARE from hermes_cli.plugin_capabilities import ( # noqa: F401 — re-exported CAPABILITY_REGISTRY, + VALID_CAPABILITY_IDS, plugin_capability_granted, ) from hermes_cli.plugin_capabilities import ( @@ -391,6 +392,103 @@ SHELL_UNSUPPORTED_HOOKS: Set[str] = { } ENTRY_POINTS_GROUP = "hermes_agent.plugins" +ENTRY_POINT_CAPABILITIES_GROUP = "hermes_agent.plugin_capabilities" + + +def _select_entry_point_group(entry_points: Any, group: str) -> list: + """Return one metadata entry-point group across supported Python APIs.""" + if hasattr(entry_points, "select"): + return list(entry_points.select(group=group)) + if isinstance(entry_points, dict): + return list(entry_points.get(group, [])) + return [ep for ep in entry_points if ep.group == group] + + +def discover_entrypoint_manifests() -> List["PluginManifest"]: + """Return metadata-only manifests for installed entry-point plugins. + + Composes the full entry-point manifest contract in one place: + + * **Kind classification** — the module source is resolved import-free + (``_resolve_module_source``) and scanned for provider markers + (``_detect_kind_from_source``), so memory providers (``exclusive``) + and model providers (``model-provider``) are routed to their own + discovery systems instead of being eagerly imported here. + * **Capability declarations** — read from the companion + ``hermes_agent.plugin_capabilities`` entry-point group (declarations + named ``.`` pointing at the same object), + so consent/introspection is accurate without importing plugin code. + + Failures are isolated per entry point: one malformed distribution must + not blank the manifests of every other installed plugin. + """ + manifests: List[PluginManifest] = [] + try: + eps = importlib.metadata.entry_points() + group_eps = _select_entry_point_group(eps, ENTRY_POINTS_GROUP) + capability_eps = _select_entry_point_group( + eps, ENTRY_POINT_CAPABILITIES_GROUP + ) + except Exception as exc: + logger.debug("Entry-point scan failed: %s", exc) + return manifests + + for ep in group_eps: + try: + capabilities = [] + for capability in VALID_CAPABILITY_IDS: + declaration_name = f"{ep.name}.{capability}" + if any( + declaration.name == declaration_name + and declaration.value == ep.value + for declaration in capability_eps + ): + capabilities.append(capability) + dist = getattr(ep, "dist", None) + metadata = getattr(dist, "metadata", None) + manifest = PluginManifest( + name=ep.name, + version=str(getattr(dist, "version", "") or ""), + description=( + str(metadata.get("Summary", "") or "") + if metadata is not None + else "" + ), + source="entrypoint", + path=ep.value, + key=ep.name, + capabilities=_parse_declared_capabilities( + capabilities, ep.name + ), + ) + manifest.kind = _classify_entrypoint_value_kind(ep.value) + manifests.append(manifest) + except Exception as exc: + logger.debug( + "Entry-point manifest for %r skipped: %s", + getattr(ep, "name", "?"), + exc, + ) + return manifests + + +def _classify_entrypoint_value_kind(value: str) -> str: + """Classify an entry-point target by import-free source scan. + + Module-level twin of ``PluginManager._classify_entrypoint_kind`` so + ``discover_entrypoint_manifests()`` callers outside the manager (the + CLI capabilities path) get identical routing. Unresolvable or + non-Python modules stay ``standalone``. + """ + try: + module_name = str(value).split(":", 1)[0].strip() + if not module_name: + return "standalone" + return _detect_kind_from_source( + _resolve_module_source(module_name) + ) or "standalone" + except Exception: + return "standalone" # System-prompt sections are deliberately more constrained than lifecycle # hooks. They become high-trust prompt bytes and are charged on every turn. @@ -4290,31 +4388,17 @@ class PluginManager: return "standalone" def _scan_entry_points(self) -> List[PluginManifest]: - """Check ``importlib.metadata`` for pip-installed plugins.""" - manifests: List[PluginManifest] = [] - try: - eps = importlib.metadata.entry_points() - # Python 3.12+ returns a SelectableGroups; earlier returns dict - if hasattr(eps, "select"): - group_eps = eps.select(group=ENTRY_POINTS_GROUP) - elif isinstance(eps, dict): - group_eps = eps.get(ENTRY_POINTS_GROUP, []) - else: - group_eps = [ep for ep in eps if ep.group == ENTRY_POINTS_GROUP] + """Read installed plugin and companion capability entry points. - for ep in group_eps: - manifest = PluginManifest( - name=ep.name, - source="entrypoint", - path=ep.value, - key=ep.name, - ) - manifest.kind = self._classify_entrypoint_kind(ep) - manifests.append(manifest) - except Exception as exc: - logger.debug("Entry-point scan failed: %s", exc) - - return manifests + Delegates to ``discover_entrypoint_manifests()``, which composes + kind classification (import-free source scan routing memory/model + providers away from the general manager) with capability + declarations from the ``hermes_agent.plugin_capabilities`` group. + Capability declarations live in distribution metadata so discovery + is available before importing untrusted plugin code and does not + depend on a package-data ``plugin.yaml`` being present. + """ + return discover_entrypoint_manifests() # ----------------------------------------------------------------------- # Loading diff --git a/tests/hermes_cli/test_plugin_capabilities.py b/tests/hermes_cli/test_plugin_capabilities.py index 059dd8ba4e..7c9b619bff 100644 --- a/tests/hermes_cli/test_plugin_capabilities.py +++ b/tests/hermes_cli/test_plugin_capabilities.py @@ -7,6 +7,7 @@ and backward compatibility with the legacy ``allow_*`` gates. from __future__ import annotations +from types import SimpleNamespace from unittest.mock import MagicMock, patch import pytest @@ -113,6 +114,45 @@ class TestDeclarationParsing: assert manifest is not None assert manifest.capabilities == [] + def test_entrypoint_companion_metadata_declares_capabilities_without_import( + self, monkeypatch + ): + """Installed plugins can declare consent metadata in dist entry points.""" + from hermes_cli import plugins as plugins_mod + from hermes_cli.plugins import PluginManager + + load = MagicMock(side_effect=AssertionError("plugin code must not be imported")) + plugin_ep = SimpleNamespace( + name="thread-namer", + value="thread_namer.plugin:register", + group="hermes_agent.plugins", + dist=SimpleNamespace( + version="1.2.3", + metadata={"Summary": "Names gateway threads"}, + ), + load=load, + ) + capability_ep = SimpleNamespace( + name="thread-namer.gateway.platform_actions", + value="thread_namer.plugin:register", + group="hermes_agent.plugin_capabilities", + load=load, + ) + monkeypatch.setattr( + plugins_mod.importlib.metadata, + "entry_points", + lambda: [plugin_ep, capability_ep], + ) + + manifests = PluginManager()._scan_entry_points() + + assert len(manifests) == 1 + assert manifests[0].name == "thread-namer" + assert manifests[0].version == "1.2.3" + assert manifests[0].description == "Names gateway threads" + assert manifests[0].capabilities == ["gateway.platform_actions"] + load.assert_not_called() + # ── Consent grant + persistence ────────────────────────────────────────────── diff --git a/tests/hermes_cli/test_plugins_cmd_list.py b/tests/hermes_cli/test_plugins_cmd_list.py index e40e7de53f..9c8ffe6042 100644 --- a/tests/hermes_cli/test_plugins_cmd_list.py +++ b/tests/hermes_cli/test_plugins_cmd_list.py @@ -94,3 +94,36 @@ def test_discover_all_plugins_includes_entrypoint_plugins(monkeypatch, tmp_path) ] +def test_declared_capabilities_for_entrypoint_uses_distribution_metadata( + monkeypatch, tmp_path +): + bundled_dir = tmp_path / "bundled" + user_dir = tmp_path / "user" + bundled_dir.mkdir() + user_dir.mkdir() + plugin_ep = SimpleNamespace( + name="thread-namer", + value="thread_namer.plugin:register", + group="hermes_agent.plugins", + dist=SimpleNamespace(version="1.0", metadata={"Summary": ""}), + ) + capability_ep = SimpleNamespace( + name="thread-namer.gateway.platform_actions", + value="thread_namer.plugin:register", + group="hermes_agent.plugin_capabilities", + ) + monkeypatch.setattr(plugins_cmd, "_plugins_dir", lambda: user_dir) + monkeypatch.setattr( + "hermes_cli.plugins.get_bundled_plugins_dir", lambda: bundled_dir + ) + monkeypatch.setattr( + plugins_cmd.importlib.metadata, + "entry_points", + lambda: [plugin_ep, capability_ep], + ) + + assert plugins_cmd._declared_capabilities_for_key("thread-namer") == [ + "gateway.platform_actions" + ] + + From 90dacec87e87ac5a30053371873b5fd31f3ce5cb Mon Sep 17 00:00:00 2001 From: Jeeves Assistant Date: Thu, 13 Aug 2026 07:39:58 -0500 Subject: [PATCH 022/748] fix(plugins): report entrypoint capabilities in CLI --- hermes_cli/plugins_cmd.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/hermes_cli/plugins_cmd.py b/hermes_cli/plugins_cmd.py index 1a8c41006c..ecf14b2f78 100644 --- a/hermes_cli/plugins_cmd.py +++ b/hermes_cli/plugins_cmd.py @@ -1354,6 +1354,13 @@ def _declared_capabilities_for_key(key: str) -> list: for entry in _discover_all_plugins(): # entry = (name, version, description, source, dir_path, key) if entry[5] == key or entry[0] == key: + if entry[3] == "entrypoint": + from hermes_cli.plugins import discover_entrypoint_manifests + + for manifest in discover_entrypoint_manifests(): + if key in (manifest.key, manifest.name): + return list(manifest.capabilities) + return [] dir_path = entry[4] if not dir_path: return [] From 4d30bb6bb86746330a7c5a840d70f101b2c239b6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 11:52:13 -0700 Subject: [PATCH 023/748] fix: compose kind classification with capability manifests; per-entry isolation; docs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-ups on salvaged #85287: - discover_entrypoint_manifests() now carries BOTH the import-free kind classification (from #85527) and capability declarations — the two contracts compose in one function instead of the capability rewrite dropping classification. - Per-entry exception isolation: one malformed distribution no longer blanks every other plugin's manifest (same contract as providers/__init__.py entry-point scan). - Documented the hermes_agent.plugin_capabilities group in the plugin developer guide (pyproject example). --- website/docs/developer-guide/plugins/index.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/website/docs/developer-guide/plugins/index.md b/website/docs/developer-guide/plugins/index.md index ad241b350f..3ba389c4e0 100644 --- a/website/docs/developer-guide/plugins/index.md +++ b/website/docs/developer-guide/plugins/index.md @@ -253,6 +253,25 @@ config keys (`plugins.entries..allow_tool_override`, …) still work but are deprecated — declare capabilities instead so users get a single, auditable consent screen. Capabilities are consent + audit, **not a sandbox**: they gate host API surfaces, nothing more. + +**Pip-distributed plugins** have no `plugin.yaml` directory once installed, +so declare capabilities in distribution metadata instead, via the companion +`hermes_agent.plugin_capabilities` entry-point group. Each declaration is +named `.` and points at the same object as your +`hermes_agent.plugins` entry point: + +```toml +[project.entry-points."hermes_agent.plugins"] +calculator = "my_pkg:register" + +[project.entry-points."hermes_agent.plugin_capabilities"] +"calculator.tools.override" = "my_pkg:register" +``` + +Hermes reads these from installed metadata without importing your code, so +`hermes plugins capabilities` and the consent flow stay accurate for pip +installs. + ### Manifest v2 reference `plugin.yaml` also supports an additive **v2 schema** (#64165). Every field is From cdd0a26031648b8c152fe1937143fd069b50e844 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 13 Aug 2026 17:05:30 +0530 Subject: [PATCH 024/748] fix: restore session model on resume instead of falling back to config default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two bugs caused resumed sessions to use the config default model instead of the model the session was actually using: 1. CLI /model switch didn't persist the new model to the session DB row. The gateway calls update_session_model() after a /model switch, but the CLI path only updated in-memory state and the agent's runtime — it never wrote the new model to the sessions.model column. So the DB row always kept the original model from session creation. 2. Resume didn't restore model/provider from the session DB row. _preload_resumed_session and _init_agent restored CWD and YOLO from session_meta, but never read session_meta['model'] back into self.model/self.provider. So even if the DB had the right model, resume would use whatever was in config.yaml. Fix: - _handle_model_switch / _apply_model_switch_result: call update_session_model() after a session-scoped /model switch (skipped for --once and --global), mirroring the gateway's behavior. - New _restore_session_model() method: restores model/provider from session_meta on resume, with provider/base_url/api_mode from model_config.gateway_runtime. Also swaps the running agent in-place for mid-chat /resume. - Call _restore_session_model() from all three resume paths: _preload_resumed_session, _init_agent, and _handle_resume_command. - Track _explicit_model_override flag so -m/--model on the CLI overrides resume (user intent wins). Cleared on /new. --- cli.py | 169 ++++++++++++++++++++++++++++ hermes_cli/cli_agent_setup_mixin.py | 2 + hermes_cli/cli_commands_mixin.py | 5 + 3 files changed, 176 insertions(+) diff --git a/cli.py b/cli.py index 2e822be374..078d934dbb 100644 --- a/cli.py +++ b/cli.py @@ -4430,6 +4430,9 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): _model_config = CLI_CONFIG.get("model", {}) _config_model = (_model_config.get("default") or _model_config.get("model") or "") if isinstance(_model_config, dict) else (_model_config or "") _DEFAULT_CONFIG_MODEL = "" + # Track whether the user passed -m / --model so resume knows not to + # clobber an explicit override with the session's stored model. + self._explicit_model_override = bool(model) self.model = model or _config_model or _DEFAULT_CONFIG_MODEL # A ``moa:`` model string selects the MoA virtual provider in # one shot (parity with interactive ``/moa`` and the model picker). Do @@ -7687,6 +7690,107 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: self._console_print(f"[dim]{_escape(msg)}[/dim]") + def _restore_session_model(self, session_meta: dict, *, quiet: bool = False) -> None: + """Restore model/provider from the session DB row on resume. + + Companion to ``_restore_session_cwd`` / ``_restore_session_yolo`` — + called from every resume path (startup ``--resume``/``-c`` and + mid-chat ``/resume``). The persisted model lives in the session row's + ``model`` column (written at creation time and updated on ``/model`` + switches via ``update_session_model``); the provider/endpoint live in + ``model_config.gateway_runtime`` (written by the gateway's + ``_sync_session_model_from_agent`` and the CLI ``/model`` persist). + Without this restore a resumed session silently falls back to the + config default model, losing the user's last ``/model`` choice. + + When the stored provider differs from the ambient one, credentials + are re-resolved for the stored provider (mirroring the gateway's + ``_rehydrate_session_model_override``) — the ambient ``self.api_key`` + belongs to the config-default provider and must not be sent to the + session's endpoint. On resolution failure the ambient credentials are + kept so the session still opens (the first turn surfaces the auth + error instead of the resume dying). + + Skips when the session has no model recorded or when the CLI was + launched with an explicit ``-m`` override (user intent wins). + """ + stored_model = (session_meta or {}).get("model") + if not stored_model: + return + # An explicit -m / --model on the command line overrides resume. + if getattr(self, "_explicit_model_override", False): + return + # Stored provider/endpoint from model_config.gateway_runtime + # (written by gateway turns and CLI /model switches alike). + stored_provider = stored_base_url = stored_api_mode = None + try: + import json as _json + raw_config = (session_meta or {}).get("model_config") + config = _json.loads(raw_config) if isinstance(raw_config, str) and raw_config else (raw_config if isinstance(raw_config, dict) else {}) + runtime = config.get("gateway_runtime") or {} if isinstance(config, dict) else {} + if isinstance(runtime, dict): + stored_provider = runtime.get("provider") or None + stored_base_url = runtime.get("base_url") or None + stored_api_mode = runtime.get("api_mode") or None + except Exception: + pass + model_changed = stored_model != self.model + provider_changed = bool(stored_provider) and stored_provider != self.provider + if not model_changed and not provider_changed: + return + self.model = stored_model + if stored_provider: + self.provider = stored_provider + self.requested_provider = stored_provider + if stored_base_url: + self.base_url = stored_base_url + if stored_api_mode: + self.api_mode = stored_api_mode + if provider_changed: + # Re-resolve credentials for the restored provider. api_key is + # never persisted to the session DB (by design) — the normal + # runtime provider resolution owns credentials. + try: + from hermes_cli.runtime_provider import resolve_runtime_provider + resolved = resolve_runtime_provider(requested=stored_provider) + if resolved.get("api_key"): + self.api_key = resolved["api_key"] + if not stored_base_url and resolved.get("base_url"): + self.base_url = resolved["base_url"] + if not stored_api_mode and resolved.get("api_mode"): + self.api_mode = resolved["api_mode"] + self._credential_pool = resolved.get("credential_pool") + except Exception: + logger.debug( + "Credential re-resolution for resumed session provider " + "%s failed; keeping ambient credentials", + stored_provider, exc_info=True, + ) + # If the agent is already running (mid-chat /resume), swap it + # in-place so the next turn uses the restored model. On startup + # --resume the agent isn't built yet — _init_agent will pick up + # self.model / self.provider when constructing AIAgent. + if self.agent is not None: + try: + self.agent.switch_model( + new_model=self.model, + new_provider=self.provider, + api_key=self.api_key or "", + base_url=self.base_url or "", + api_mode=self.api_mode or "", + ) + except Exception: + logger.debug( + "In-place agent model swap on resume failed", exc_info=True + ) + msg = f"Model restored from session: {stored_model}" + if stored_provider: + msg += f" ({stored_provider})" + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + def _render_resume_history_panel_lines(self, panel) -> list[str]: @@ -8477,6 +8581,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self.conversation_history = [] self._pending_title = None self._resumed = False + # /new clears the -m / --model override flag: an explicit CLI model + # was for the previous session only, not for every session spawned + # afterwards. + self._explicit_model_override = False self.reasoning_config = _parse_reasoning_config( CLI_CONFIG["agent"].get("reasoning_effort", "") ) @@ -9555,6 +9663,36 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") + # Persist the model change to the session DB row so --resume + # restores the correct model instead of falling back to the + # config default. Skipped for --global (config.yaml is the source + # of truth). Mirrors the gateway's update_session_model() call + # after a /model switch. The provider/endpoint go into + # model_config.gateway_runtime — same key the gateway writes and + # _restore_session_model reads; persisting only the model name + # recombines it with the ambient provider on resume (#79536). + # getattr: tests build bare stubs via object.__new__ without + # __init__ attributes. + _picker_session_db = getattr(self, "_session_db", None) + _picker_session_id = getattr(self, "session_id", None) + if not persist_global and _picker_session_db and _picker_session_id: + try: + _picker_session_db.update_session_model( + _picker_session_id, result.new_model + ) + _picker_session_db.patch_session_model_config( + _picker_session_id, + {"gateway_runtime": {k: v for k, v in { + "provider": result.target_provider, + "base_url": result.base_url, + "api_mode": result.api_mode, + }.items() if v}}, + ) + except Exception: + logger.debug( + "Failed to persist model switch to session DB", exc_info=True + ) + def _handle_model_picker_selection(self, persist_global: bool = False) -> None: state = self._model_picker_state if not state: @@ -9906,6 +10044,37 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") + # Persist the model change to the session DB row so --resume + # restores the correct model instead of falling back to the + # config default. Skipped for --global (config.yaml is the source + # of truth) and --once (ephemeral, restored after one turn). + # Mirrors the gateway's update_session_model() call after a + # /model switch. The provider/endpoint go into + # model_config.gateway_runtime — same key the gateway writes and + # _restore_session_model reads; persisting only the model name + # recombines it with the ambient provider on resume (#79536). + # getattr: tests build bare stubs via object.__new__ without + # __init__ attributes. + _switch_session_db = getattr(self, "_session_db", None) + _switch_session_id = getattr(self, "session_id", None) + if not persist_global and not one_turn and _switch_session_db and _switch_session_id: + try: + _switch_session_db.update_session_model( + _switch_session_id, result.new_model + ) + _switch_session_db.patch_session_model_config( + _switch_session_id, + {"gateway_runtime": {k: v for k, v in { + "provider": result.target_provider, + "base_url": result.base_url, + "api_mode": result.api_mode, + }.items() if v}}, + ) + except Exception: + logger.debug( + "Failed to persist model switch to session DB", exc_info=True + ) + def _handle_codex_runtime(self, cmd_original: str) -> None: """Handle /codex-runtime — toggle the codex app-server runtime opt-in. diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index f1b4060b53..e8494c84cd 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -451,6 +451,7 @@ class CLIAgentSetupMixin: ) self._restore_session_cwd(session_meta, quiet=_quiet_mode) self._restore_session_yolo(session_meta, quiet=_quiet_mode) + self._restore_session_model(session_meta, quiet=_quiet_mode) else: if _quiet_mode: print( @@ -717,6 +718,7 @@ class CLIAgentSetupMixin: ) self._restore_session_cwd(session_meta) self._restore_session_yolo(session_meta) + self._restore_session_model(session_meta) else: accent_color = _accent_hex() self._console_print( diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index d4accf472c..612ebbb645 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -1139,6 +1139,11 @@ class CLICommandsMixin: # --resume. self._restore_session_yolo(session_meta) + # Restore the target session's model/provider so a mid-chat /resume + # doesn't silently revert to the config default. Same contract as a + # startup --resume (_preload_resumed_session / _init_agent path). + self._restore_session_model(session_meta) + def _handle_sessions_command(self, cmd_original: str) -> None: """Handle /sessions [list|] — browse or resume previous sessions. From b8f85f18e860ca2d03d1bf0e68f421fcf1ef63ec Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:26:22 +0530 Subject: [PATCH 025/748] refactor: shared /model persist helper, canonical gateway_runtime reader, cross-surface route persistence, tests - Extract the two duplicated /model session-persist blocks into _persist_model_switch_to_session; persist the route BOTH nested (gateway_runtime, CLI reader) and top-level (TUI gateway's _stored_session_runtime_overrides reader) so a CLI switch also survives a desktop/TUI session.resume. - Add SessionDB.session_gateway_runtime as the canonical tolerant row-level route reader (session_yolo_enabled precedent); use it in _restore_session_model instead of hand-rolled JSON parsing. - Clear stale launch-time _explicit_api_key/_explicit_base_url when resume restores a different provider (same leak guard _apply_model_switch_result already has). - 12 new tests incl. a real-SessionDB round trip; mutation-checked. --- cli.py | 131 ++++++++---------- hermes_state.py | 33 +++++ tests/cli/test_resume_model_restore.py | 183 +++++++++++++++++++++++++ 3 files changed, 274 insertions(+), 73 deletions(-) create mode 100644 tests/cli/test_resume_model_restore.py diff --git a/cli.py b/cli.py index 078d934dbb..6b7b9e7a4e 100644 --- a/cli.py +++ b/cli.py @@ -7690,6 +7690,40 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: self._console_print(f"[dim]{_escape(msg)}[/dim]") + def _persist_model_switch_to_session(self, result) -> None: + """Persist a session-scoped /model switch to the session DB row. + + Writes the model column plus the runtime route so ``--resume`` + (CLI, reads ``gateway_runtime``) and ``session.resume`` (TUI/desktop, + reads top-level ``model_config`` keys via + ``_stored_session_runtime_overrides``) both restore the switched + provider instead of recombining the model with the ambient default + (#79536). Mirrors the gateway's ``update_session_model()`` call. + getattr: tests drive the switch paths with ``object.__new__`` stubs. + """ + db = getattr(self, "_session_db", None) + sid = getattr(self, "session_id", None) + if not db or not sid: + return + route = { + k: v + for k, v in { + "provider": result.target_provider, + "base_url": result.base_url, + "api_mode": result.api_mode, + }.items() + if v + } + try: + db.update_session_model(sid, result.new_model) + # Both shapes: nested for the CLI reader, top-level for the + # TUI gateway's resume path. + db.patch_session_model_config(sid, {"gateway_runtime": route, **route}) + except Exception: + logger.debug( + "Failed to persist model switch to session DB", exc_info=True + ) + def _restore_session_model(self, session_meta: dict, *, quiet: bool = False) -> None: """Restore model/provider from the session DB row on resume. @@ -7720,20 +7754,14 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # An explicit -m / --model on the command line overrides resume. if getattr(self, "_explicit_model_override", False): return - # Stored provider/endpoint from model_config.gateway_runtime - # (written by gateway turns and CLI /model switches alike). - stored_provider = stored_base_url = stored_api_mode = None - try: - import json as _json - raw_config = (session_meta or {}).get("model_config") - config = _json.loads(raw_config) if isinstance(raw_config, str) and raw_config else (raw_config if isinstance(raw_config, dict) else {}) - runtime = config.get("gateway_runtime") or {} if isinstance(config, dict) else {} - if isinstance(runtime, dict): - stored_provider = runtime.get("provider") or None - stored_base_url = runtime.get("base_url") or None - stored_api_mode = runtime.get("api_mode") or None - except Exception: - pass + # Stored provider/endpoint via the canonical row-level reader + # (prefers model_config.gateway_runtime, falls back to the TUI + # gateway's top-level keys). + from hermes_state import SessionDB as _SessionDB + _stored_runtime = _SessionDB.session_gateway_runtime(session_meta) + stored_provider = _stored_runtime.get("provider") or None + stored_base_url = _stored_runtime.get("base_url") or None + stored_api_mode = _stored_runtime.get("api_mode") or None model_changed = stored_model != self.model provider_changed = bool(stored_provider) and stored_provider != self.provider if not model_changed and not provider_changed: @@ -7747,6 +7775,13 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): if stored_api_mode: self.api_mode = stored_api_mode if provider_changed: + # Stale launch-time explicit overrides belong to the AMBIENT + # provider; carrying them into the restored provider's + # resolution poisons _ensure_runtime_credentials on startup + # resume (same leak _apply_model_switch_result guards against + # by overwriting _explicit_* on every switch). + self._explicit_api_key = None + self._explicit_base_url = stored_base_url # Re-resolve credentials for the restored provider. api_key is # never persisted to the session DB (by design) — the normal # runtime provider resolution owns credentials. @@ -9663,35 +9698,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") - # Persist the model change to the session DB row so --resume - # restores the correct model instead of falling back to the - # config default. Skipped for --global (config.yaml is the source - # of truth). Mirrors the gateway's update_session_model() call - # after a /model switch. The provider/endpoint go into - # model_config.gateway_runtime — same key the gateway writes and - # _restore_session_model reads; persisting only the model name - # recombines it with the ambient provider on resume (#79536). - # getattr: tests build bare stubs via object.__new__ without - # __init__ attributes. - _picker_session_db = getattr(self, "_session_db", None) - _picker_session_id = getattr(self, "session_id", None) - if not persist_global and _picker_session_db and _picker_session_id: - try: - _picker_session_db.update_session_model( - _picker_session_id, result.new_model - ) - _picker_session_db.patch_session_model_config( - _picker_session_id, - {"gateway_runtime": {k: v for k, v in { - "provider": result.target_provider, - "base_url": result.base_url, - "api_mode": result.api_mode, - }.items() if v}}, - ) - except Exception: - logger.debug( - "Failed to persist model switch to session DB", exc_info=True - ) + # Persist session-scoped switches so --resume / session.resume + # restore them; --global switches live in config.yaml instead. + if not persist_global: + HermesCLI._persist_model_switch_to_session(self, result) def _handle_model_picker_selection(self, persist_global: bool = False) -> None: state = self._model_picker_state @@ -10044,36 +10054,11 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") - # Persist the model change to the session DB row so --resume - # restores the correct model instead of falling back to the - # config default. Skipped for --global (config.yaml is the source - # of truth) and --once (ephemeral, restored after one turn). - # Mirrors the gateway's update_session_model() call after a - # /model switch. The provider/endpoint go into - # model_config.gateway_runtime — same key the gateway writes and - # _restore_session_model reads; persisting only the model name - # recombines it with the ambient provider on resume (#79536). - # getattr: tests build bare stubs via object.__new__ without - # __init__ attributes. - _switch_session_db = getattr(self, "_session_db", None) - _switch_session_id = getattr(self, "session_id", None) - if not persist_global and not one_turn and _switch_session_db and _switch_session_id: - try: - _switch_session_db.update_session_model( - _switch_session_id, result.new_model - ) - _switch_session_db.patch_session_model_config( - _switch_session_id, - {"gateway_runtime": {k: v for k, v in { - "provider": result.target_provider, - "base_url": result.base_url, - "api_mode": result.api_mode, - }.items() if v}}, - ) - except Exception: - logger.debug( - "Failed to persist model switch to session DB", exc_info=True - ) + # Persist session-scoped switches so --resume / session.resume + # restore them; --global lives in config.yaml, --once is ephemeral + # (restored after one turn). + if not persist_global and not one_turn: + HermesCLI._persist_model_switch_to_session(self, result) def _handle_codex_runtime(self, cmd_original: str) -> None: """Handle /codex-runtime — toggle the codex app-server runtime opt-in. diff --git a/hermes_state.py b/hermes_state.py index 800a7828e1..4fe015d4f3 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -6109,6 +6109,39 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) return False return bool(raw.get("yolo_mode")) + @staticmethod + def session_gateway_runtime(session_meta: Optional[Dict[str, Any]]) -> Dict[str, Any]: + """Read the persisted runtime route off a session row dict. + + Accepts the dict returned by ``get_session`` (``model_config`` is a + JSON string) or an already-parsed dict. Prefers the nested + ``gateway_runtime`` key (written by the gateway's + ``_sync_session_model_from_agent`` and the CLI ``/model`` persist), + falling back to the top-level ``provider``/``base_url``/``api_mode`` + keys the TUI gateway's ``_runtime_model_config`` writes. Returns an + empty dict on any parse failure — resume falls back to ambient + config resolution. + """ + raw = (session_meta or {}).get("model_config") + if isinstance(raw, str): + try: + raw = json.loads(raw) + except Exception: + return {} + if not isinstance(raw, dict): + return {} + runtime = raw.get("gateway_runtime") + if isinstance(runtime, dict) and runtime.get("provider"): + return dict(runtime) + top_level = { + key: raw.get(key) + for key in ("provider", "base_url", "api_mode") + if raw.get(key) + } + if top_level: + return top_level + return dict(runtime) if isinstance(runtime, dict) else {} + def update_session_billing_route( self, session_id: str, diff --git a/tests/cli/test_resume_model_restore.py b/tests/cli/test_resume_model_restore.py new file mode 100644 index 0000000000..dae3ea6afe --- /dev/null +++ b/tests/cli/test_resume_model_restore.py @@ -0,0 +1,183 @@ +"""Tests for CLI resume model restoration and /model session persistence. + +Covers _restore_session_model, _persist_model_switch_to_session (cli.py) and +SessionDB.session_gateway_runtime (hermes_state.py) — the round trip that +makes `hermes --resume` reopen a session on the model/provider it actually +used instead of the ambient config default (#57588-class, #79536). +""" + +import json + +import pytest + +import cli as cli_mod +from hermes_state import SessionDB + + +def _make_stub(**overrides): + """Bare HermesCLI the way resume paths see it (no __init__).""" + stub = object.__new__(cli_mod.HermesCLI) + stub.model = "ambient-model" + stub.provider = "openrouter" + stub.requested_provider = "openrouter" + stub.base_url = "https://openrouter.ai/api/v1" + stub.api_key = "ambient-key" + stub.api_mode = "" + stub.agent = None + stub._console_print = lambda s: None + for key, value in overrides.items(): + setattr(stub, key, value) + return stub + + +def _row(model="glm-4.7", model_config=None): + return { + "model": model, + "model_config": json.dumps(model_config) if model_config else None, + } + + +# ── SessionDB.session_gateway_runtime ─────────────────────────────── + + +def test_session_gateway_runtime_prefers_nested_key(): + meta = _row(model_config={ + "gateway_runtime": {"provider": "custom:feather", "base_url": "https://f/v1"}, + "provider": "openrouter", + }) + runtime = SessionDB.session_gateway_runtime(meta) + assert runtime["provider"] == "custom:feather" + assert runtime["base_url"] == "https://f/v1" + + +def test_session_gateway_runtime_falls_back_to_top_level_keys(): + # The TUI gateway's _runtime_model_config writes top-level keys only. + meta = _row(model_config={"provider": "nous", "api_mode": "chat_completions"}) + runtime = SessionDB.session_gateway_runtime(meta) + assert runtime == {"provider": "nous", "api_mode": "chat_completions"} + + +def test_session_gateway_runtime_tolerates_garbage(): + assert SessionDB.session_gateway_runtime(None) == {} + assert SessionDB.session_gateway_runtime({}) == {} + assert SessionDB.session_gateway_runtime({"model_config": "{not json"}) == {} + assert SessionDB.session_gateway_runtime({"model_config": json.dumps([1, 2])}) == {} + + +# ── _restore_session_model ────────────────────────────────────────── + + +def test_restore_session_model_restores_model_and_provider(): + stub = _make_stub() + stub._restore_session_model(_row(model_config={ + "gateway_runtime": {"provider": "custom:feather", "base_url": "https://f/v1"}, + })) + assert stub.model == "glm-4.7" + assert stub.provider == "custom:feather" + assert stub.requested_provider == "custom:feather" + assert stub.base_url == "https://f/v1" + # Stale launch-time explicit overrides must not leak into the restored + # provider's credential resolution. + assert stub._explicit_api_key is None + assert stub._explicit_base_url == "https://f/v1" + + +def test_restore_session_model_explicit_cli_flag_wins(): + stub = _make_stub(model="cli-flag-model", _explicit_model_override=True) + stub._restore_session_model(_row()) + assert stub.model == "cli-flag-model" + assert stub.provider == "openrouter" + + +def test_restore_session_model_no_stored_model_is_noop(): + stub = _make_stub() + stub._restore_session_model(_row(model=None)) + assert stub.model == "ambient-model" + + +def test_restore_session_model_matching_state_is_silent_noop(): + notes = [] + stub = _make_stub(model="glm-4.7", provider="custom:feather", + requested_provider="custom:feather", + _console_print=lambda s: notes.append(s)) + stub._restore_session_model(_row(model_config={ + "gateway_runtime": {"provider": "custom:feather"}, + })) + assert not notes + + +def test_restore_session_model_swaps_running_agent_in_place(): + calls = {} + + class _Agent: + def switch_model(self, **kwargs): + calls.update(kwargs) + + stub = _make_stub(agent=_Agent()) + stub._restore_session_model(_row()) + assert calls["new_model"] == "glm-4.7" + + +# ── _persist_model_switch_to_session ──────────────────────────────── + + +class _Result: + new_model = "deepseek-v4-flash-free" + target_provider = "custom:opencode-zen" + base_url = "https://oz/v1" + api_mode = "" + + +def test_persist_model_switch_writes_model_and_both_route_shapes(): + written = {} + + class _DB: + def update_session_model(self, sid, model): + written["model"] = (sid, model) + + def patch_session_model_config(self, sid, patch): + written["patch"] = (sid, patch) + + stub = _make_stub(_session_db=_DB(), session_id="s1") + stub._persist_model_switch_to_session(_Result()) + assert written["model"] == ("s1", "deepseek-v4-flash-free") + sid, patch = written["patch"] + # Nested shape for the CLI reader... + assert patch["gateway_runtime"]["provider"] == "custom:opencode-zen" + # ...and top-level for the TUI gateway's _stored_session_runtime_overrides. + assert patch["provider"] == "custom:opencode-zen" + assert patch["base_url"] == "https://oz/v1" + assert "api_mode" not in patch["gateway_runtime"] # empty values dropped + + +def test_persist_model_switch_noop_without_db_or_session(): + stub = _make_stub() # no _session_db / session_id attributes at all + stub._persist_model_switch_to_session(_Result()) # must not raise + + +def test_persist_model_switch_swallows_db_errors(): + class _DB: + def update_session_model(self, *a): + raise RuntimeError("disk full") + + stub = _make_stub(_session_db=_DB(), session_id="s1") + stub._persist_model_switch_to_session(_Result()) # must not raise + + +# ── round trip: persist → get_session shape → restore ─────────────── + + +def test_round_trip_persist_then_restore(tmp_path, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) + db = SessionDB(db_path=tmp_path / "state.db") + db.create_session(session_id="rt1", source="cli", model="ambient-model") + + stub = _make_stub(_session_db=db, session_id="rt1") + stub._persist_model_switch_to_session(_Result()) + + meta = db.get_session("rt1") + restored = _make_stub() + restored._restore_session_model(meta) + assert restored.model == "deepseek-v4-flash-free" + assert restored.provider == "custom:opencode-zen" + assert restored.base_url == "https://oz/v1" From dbe24dfc126aaca41a132d93581927f2e5abcb42 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:42:20 +0530 Subject: [PATCH 026/748] fix: heal bare-custom provider at persist AND restore; persist --global switches to the row - Bare 'custom' from ModelSwitchResult.target_provider is the resolved billing class, not a routable identity; persisting it verbatim made a later --resume hard-fail once the config default moved off the custom endpoint. Heal to custom: via canonical_custom_identity at persist time, and again on restore for rows written by older builds (mirrors tui_gateway's _stored_session_runtime_overrides recovery). - --global switches now also update the session row: the row records what THIS session runs, otherwise resume restored the stale creation-time model over the user's new global choice. - Only adopt resolved credential_pool alongside its api_key (don't null the ambient pool when resolution returns no credentials). - 3 new tests; healing path mutation-checked. --- cli.py | 54 +++++++++++++++++++++----- tests/cli/test_resume_model_restore.py | 47 ++++++++++++++++++++++ 2 files changed, 91 insertions(+), 10 deletions(-) diff --git a/cli.py b/cli.py index 6b7b9e7a4e..fcd5aad20d 100644 --- a/cli.py +++ b/cli.py @@ -7705,10 +7705,27 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): sid = getattr(self, "session_id", None) if not db or not sid: return + provider = result.target_provider + # Bare "custom" is the resolved billing class, not a routable + # identity — persisting it verbatim makes a later resume hard-fail + # when the config default has moved off the custom endpoint + # (resolve_runtime_provider only trusts config base_url for bare + # custom while the config provider is still custom-ish). Heal to + # the durable custom: menu key, else drop the provider — + # same recovery the TUI gateway applies on its read path. + if str(provider or "").strip().lower() == "custom": + try: + from hermes_cli.runtime_provider import canonical_custom_identity + provider = canonical_custom_identity( + base_url=result.base_url or None, + model=result.new_model or None, + ) or None + except Exception: + provider = None route = { k: v for k, v in { - "provider": result.target_provider, + "provider": provider, "base_url": result.base_url, "api_mode": result.api_mode, }.items() @@ -7762,6 +7779,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): stored_provider = _stored_runtime.get("provider") or None stored_base_url = _stored_runtime.get("base_url") or None stored_api_mode = _stored_runtime.get("api_mode") or None + # Heal bare "custom" persisted by older builds / gateway turns: it's + # the resolved billing class, not a routable identity. Recover the + # durable custom: menu key from the endpoint, else drop the + # provider so resume keeps the ambient default (matches the TUI + # gateway's _stored_session_runtime_overrides recovery). + if str(stored_provider or "").strip().lower() == "custom": + try: + from hermes_cli.runtime_provider import canonical_custom_identity + stored_provider = canonical_custom_identity( + base_url=stored_base_url or None, + model=stored_model or None, + ) or None + except Exception: + stored_provider = None model_changed = stored_model != self.model provider_changed = bool(stored_provider) and stored_provider != self.provider if not model_changed and not provider_changed: @@ -7790,11 +7821,11 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): resolved = resolve_runtime_provider(requested=stored_provider) if resolved.get("api_key"): self.api_key = resolved["api_key"] + self._credential_pool = resolved.get("credential_pool") if not stored_base_url and resolved.get("base_url"): self.base_url = resolved["base_url"] if not stored_api_mode and resolved.get("api_mode"): self.api_mode = resolved["api_mode"] - self._credential_pool = resolved.get("credential_pool") except Exception: logger.debug( "Credential re-resolution for resumed session provider " @@ -9698,10 +9729,12 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") - # Persist session-scoped switches so --resume / session.resume - # restore them; --global switches live in config.yaml instead. - if not persist_global: - HermesCLI._persist_model_switch_to_session(self, result) + # Persist the switch to this session's row so --resume / + # session.resume restore it. --global also updates config.yaml + # (future sessions), but the row still records what THIS session + # actually runs — otherwise a later resume would restore the stale + # creation-time model over the user's new global choice. + HermesCLI._persist_model_switch_to_session(self, result) def _handle_model_picker_selection(self, persist_global: bool = False) -> None: state = self._model_picker_state @@ -10054,10 +10087,11 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): else: _cprint(" (session only — add --global to persist)") - # Persist session-scoped switches so --resume / session.resume - # restore them; --global lives in config.yaml, --once is ephemeral - # (restored after one turn). - if not persist_global and not one_turn: + # Persist the switch to this session's row so --resume / + # session.resume restore it (--global also updates config.yaml but + # the row still records what THIS session runs; --once is ephemeral + # and restored after one turn, so it must not touch the row). + if not one_turn: HermesCLI._persist_model_switch_to_session(self, result) def _handle_codex_runtime(self, cmd_original: str) -> None: diff --git a/tests/cli/test_resume_model_restore.py b/tests/cli/test_resume_model_restore.py index dae3ea6afe..c77034a81a 100644 --- a/tests/cli/test_resume_model_restore.py +++ b/tests/cli/test_resume_model_restore.py @@ -164,6 +164,53 @@ def test_persist_model_switch_swallows_db_errors(): stub._persist_model_switch_to_session(_Result()) # must not raise +def test_persist_model_switch_heals_bare_custom(monkeypatch): + """Bare 'custom' is not routable — heal to custom: or drop (C1).""" + written = {} + + class _DB: + def update_session_model(self, sid, model): + written["model"] = model + + def patch_session_model_config(self, sid, patch): + written["patch"] = patch + + class _BareResult: + new_model = "qwen3.6-plus" + target_provider = "custom" + base_url = "https://my-endpoint/v1" + api_mode = "" + + import hermes_cli.runtime_provider as rp + monkeypatch.setattr(rp, "canonical_custom_identity", + lambda base_url=None, model=None: "custom:myendpoint") + stub = _make_stub(_session_db=_DB(), session_id="s1") + stub._persist_model_switch_to_session(_BareResult()) + assert written["patch"]["provider"] == "custom:myendpoint" + + # Healing fails -> provider dropped entirely, not persisted bare. + monkeypatch.setattr(rp, "canonical_custom_identity", + lambda base_url=None, model=None: None) + written.clear() + stub._persist_model_switch_to_session(_BareResult()) + assert "provider" not in written["patch"] + assert "provider" not in written["patch"]["gateway_runtime"] + + +def test_restore_session_model_heals_bare_custom_stored_rows(monkeypatch): + """Rows persisted by older builds may carry bare 'custom' — heal or drop.""" + import hermes_cli.runtime_provider as rp + monkeypatch.setattr(rp, "canonical_custom_identity", + lambda base_url=None, model=None: None) + stub = _make_stub() + stub._restore_session_model(_row(model_config={ + "gateway_runtime": {"provider": "custom"}, + })) + # Provider dropped -> model restored but provider stays ambient. + assert stub.model == "glm-4.7" + assert stub.provider == "openrouter" + + # ── round trip: persist → get_session shape → restore ─────────────── From d0021673905b37c16c9c77783f092feb35d731f0 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 02:10:34 +0530 Subject: [PATCH 027/748] fix: delete stale top-level route keys on /model persist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit patch_session_model_config merges key-level and only deletes on explicit None. Dropping falsy values from the top-level patch let a previous switch's api_mode/base_url survive the next switch — TUI/desktop resume then restored e.g. openrouter with anthropic_messages wire mode, and a failed bare-custom heal produced a stale-provider/new-endpoint route. Write absent top-level values as explicit None so each switch fully replaces the persisted route. Regression test against a real SessionDB; mutation-checked. Also correct the heal comment (CLI is deliberately stricter than the TUI recovery, which keeps bare custom with a base_url). --- cli.py | 18 ++++++++--- tests/cli/test_resume_model_restore.py | 45 ++++++++++++++++++++++++-- 2 files changed, 57 insertions(+), 6 deletions(-) diff --git a/cli.py b/cli.py index fcd5aad20d..c9f1f9137d 100644 --- a/cli.py +++ b/cli.py @@ -7734,8 +7734,17 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): try: db.update_session_model(sid, result.new_model) # Both shapes: nested for the CLI reader, top-level for the - # TUI gateway's resume path. - db.patch_session_model_config(sid, {"gateway_runtime": route, **route}) + # TUI gateway's resume path. Top-level keys are written as + # explicit None when absent — _merge_model_config_json only + # deletes on None, so omitting them would let a PREVIOUS + # switch's provider/api_mode survive this one (stale wire + # protocol / frankenroute on resume). + db.patch_session_model_config(sid, { + "gateway_runtime": route, + "provider": provider or None, + "base_url": result.base_url or None, + "api_mode": result.api_mode or None, + }) except Exception: logger.debug( "Failed to persist model switch to session DB", exc_info=True @@ -7782,8 +7791,9 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # Heal bare "custom" persisted by older builds / gateway turns: it's # the resolved billing class, not a routable identity. Recover the # durable custom: menu key from the endpoint, else drop the - # provider so resume keeps the ambient default (matches the TUI - # gateway's _stored_session_runtime_overrides recovery). + # provider so resume keeps the ambient default. (Stricter than the + # TUI gateway's recovery, which keeps bare "custom" when a base_url + # exists — the CLI's resolve path would hard-fail on it, #14676.) if str(stored_provider or "").strip().lower() == "custom": try: from hermes_cli.runtime_provider import canonical_custom_identity diff --git a/tests/cli/test_resume_model_restore.py b/tests/cli/test_resume_model_restore.py index c77034a81a..33edbad8a1 100644 --- a/tests/cli/test_resume_model_restore.py +++ b/tests/cli/test_resume_model_restore.py @@ -148,6 +148,46 @@ def test_persist_model_switch_writes_model_and_both_route_shapes(): assert patch["provider"] == "custom:opencode-zen" assert patch["base_url"] == "https://oz/v1" assert "api_mode" not in patch["gateway_runtime"] # empty values dropped + # Absent top-level values are explicit None so the merge DELETES stale + # keys from a previous switch (merge only deletes on None). + assert patch["api_mode"] is None + + +def test_persist_model_switch_clears_stale_route_keys(tmp_path, monkeypatch): + """A later switch must not inherit the previous switch's api_mode/base_url. + + patch_session_model_config merges key-level and only deletes on explicit + None — dropping falsy values from the patch left the FIRST switch's + api_mode (e.g. anthropic_messages) alive under the SECOND switch's + provider, corrupting the wire protocol on TUI/desktop resume. + """ + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) + db = SessionDB(db_path=tmp_path / "state.db") + db.create_session(session_id="stale1", source="cli", model="m0") + stub = _make_stub(_session_db=db, session_id="stale1") + + class _First: + new_model = "claude-x" + target_provider = "custom:feather" + base_url = "https://feather/v1" + api_mode = "anthropic_messages" + + class _Second: + new_model = "gpt-5.4" + target_provider = "openrouter" + base_url = "https://openrouter.ai/api/v1" + api_mode = "" # openrouter default — must ERASE the anthropic mode + + stub._persist_model_switch_to_session(_First()) + stub._persist_model_switch_to_session(_Second()) + + meta = db.get_session("stale1") + config = json.loads(meta["model_config"]) + assert config["provider"] == "openrouter" + assert "api_mode" not in config, config # stale anthropic_messages deleted + runtime = SessionDB.session_gateway_runtime(meta) + assert runtime["provider"] == "openrouter" + assert "api_mode" not in runtime def test_persist_model_switch_noop_without_db_or_session(): @@ -188,12 +228,13 @@ def test_persist_model_switch_heals_bare_custom(monkeypatch): stub._persist_model_switch_to_session(_BareResult()) assert written["patch"]["provider"] == "custom:myendpoint" - # Healing fails -> provider dropped entirely, not persisted bare. + # Healing fails -> provider dropped (explicit None deletes any stale + # persisted provider), never persisted bare. monkeypatch.setattr(rp, "canonical_custom_identity", lambda base_url=None, model=None: None) written.clear() stub._persist_model_switch_to_session(_BareResult()) - assert "provider" not in written["patch"] + assert written["patch"]["provider"] is None assert "provider" not in written["patch"]["gateway_runtime"] From 2ae96939f53b0cc0aa82868fc9a44702f3dd6c09 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:49:51 -0700 Subject: [PATCH 028/748] fix(cli): self-heal cooked-mode termios drift that freezes CLI input When a prompt_toolkit run_in_terminal cooked->raw restore is lost (cancelled coroutine, racing chained cross-thread windows from background-review summaries / process-notification prints), the tty stays in cooked mode while the Application still expects raw. The kernel line-buffers keystrokes and the CLI appears to stop taking input even though the event loop is healthy. Observed live 2026-08-13: interactive session left in 'icanon echo' after a background skill-review fork + notify_on_complete turn; only an external stty rescue restored input. Fix: _heal_cooked_mode_drift() re-applies prompt_toolkit's own raw-mode flag surgery when stdin's lflag has drifted cooked, and process_loop's idle branch runs a rate-limited _check_termios_drift() watchdog that skips legitimate cooked windows (app._running_in_terminal), agent-running phases, non-tty stdin, and Windows. --- cli.py | 118 +++++++++++++++++++++++++++ tests/cli/test_termios_drift_heal.py | 94 +++++++++++++++++++++ 2 files changed, 212 insertions(+) create mode 100644 tests/cli/test_termios_drift_heal.py diff --git a/cli.py b/cli.py index c9f1f9137d..c2c3ffcbc8 100644 --- a/cli.py +++ b/cli.py @@ -2689,6 +2689,63 @@ def _query_osc11_background() -> str | None: pass +def _heal_cooked_mode_drift(fd: int) -> bool: + """Detect and heal cooked-mode termios drift on *fd* while prompt_toolkit + expects raw mode. + + prompt_toolkit's ``run_in_terminal`` / ``in_terminal`` wraps every + "print above the prompt" in a ``cooked_mode()`` context: it flips the + tty back to cooked (ICANON/ECHO/ISIG), runs the function, then restores + raw mode. Hermes schedules those windows cross-thread constantly — the + background self-review's ``💾`` summary, background process notification + drains, curses pickers — and if a restore is ever lost (coroutine + cancelled mid-window, racing chains, an external writer touching the + shared tty), the terminal is left in cooked mode while the Application + still believes it owns raw mode. The kernel line-buffers every + keystroke and the CLI appears to "stop taking input" even though the + process is perfectly healthy (observed live: pts in ``icanon echo`` + while the event loop idled normally in ``ep_poll``). + + This helper is the last line of defense for that whole class: when the + lflag has drifted back to cooked, re-apply prompt_toolkit's own raw-mode + flag surgery (mirrors ``prompt_toolkit.input.vt100.raw_mode``) in place. + Returns True when drift was detected and healed, False when the tty was + already raw (or could not be inspected). + + POSIX-only by construction — callers must not invoke this on Windows + (no termios; prompt_toolkit uses the win32 console API there instead). + """ + try: + import termios + attrs = termios.tcgetattr(fd) + except Exception: + return False + lflag = attrs[3] + if not (lflag & (termios.ICANON | termios.ECHO)): + return False # still raw — nothing to do + # Same surgery as prompt_toolkit.input.vt100.raw_mode._patch_lflag / + # _patch_iflag, applied to the *current* attrs so any user settings + # (speed, size-independent flags) are preserved. + attrs[3] = lflag & ~( + termios.ECHO | termios.ICANON | termios.IEXTEN | termios.ISIG + ) + attrs[0] = attrs[0] & ~( + termios.IXON + | termios.IXOFF + | termios.ICRNL + | termios.INLCR + | termios.IGNCR + ) + # VMIN=1 so reads return per-byte (Solaris-derived systems default to 4; + # prompt_toolkit sets this explicitly in raw_mode.__enter__). + attrs[6][termios.VMIN] = 1 + try: + termios.tcsetattr(fd, termios.TCSANOW, attrs) + except Exception: + return False + return True + + def _detect_light_mode() -> bool: global _LIGHT_MODE_CACHE if _LIGHT_MODE_CACHE is not None: @@ -4421,6 +4478,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._pending_edit_snapshots = {} self._last_input_mode_recovery = 0.0 self._input_mode_recovery_notice_shown = False + self._last_termios_drift_check = 0.0 + self._termios_drift_notice_shown = False # Configuration - priority: CLI args > env vars > config file # Model comes from: CLI arg or config.yaml (single source of truth). @@ -7981,6 +8040,57 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): f"If this repeats, run /new or restart this tab.{_RST}" ) + def _check_termios_drift(self) -> None: + """Watchdog: heal the tty if it drifted back to cooked mode. + + See ``_heal_cooked_mode_drift`` for the failure class (a lost + ``run_in_terminal`` cooked→raw restore leaves the terminal + line-buffering keystrokes while the prompt_toolkit app believes it + owns raw mode — the CLI looks dead but the process is healthy). + + Called from ``process_loop``'s idle branch, so a drifted terminal + self-heals within ~a second of the agent going idle instead of + requiring an external ``stty`` rescue. Skipped while a + ``run_in_terminal`` window is legitimately holding cooked mode + (``app._running_in_terminal``), while the agent is running (approval + prompts and sudo prompts legitimately manipulate the tty), and on + Windows (no termios). + """ + if os.name == "nt": + return + app = getattr(self, "_app", None) + if app is None or not getattr(app, "_is_running", False): + return + # A run_in_terminal window is *supposed* to be cooked — don't fight it. + if getattr(app, "_running_in_terminal", False): + return + now = time.monotonic() + if now - self._last_termios_drift_check < 1.0: + return + self._last_termios_drift_check = now + try: + if not sys.stdin.isatty(): + return + fd = sys.stdin.fileno() + except Exception: + return + if _heal_cooked_mode_drift(fd): + logger.warning( + "Healed cooked-mode termios drift on stdin — a " + "run_in_terminal cooked→raw restore was lost." + ) + # Redraw so the prompt is visibly alive again. + try: + self._invalidate() + except Exception: + pass + if not self._termios_drift_notice_shown: + self._termios_drift_notice_shown = True + _cprint( + f" {_DIM}Recovered terminal from cooked-mode drift " + f"(input should respond normally again).{_RST}" + ) + def _preprocess_images_with_vision(self, text: str, images: list, *, announce: bool = True) -> str: @@ -17919,6 +18029,14 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # Periodic config watcher — auto-reload MCP on mcp_servers change if not self._agent_running: self._check_config_mcp_changes() + # Heal cooked-mode termios drift (lost + # run_in_terminal restore) before draining + # notifications — a drifted tty makes the CLI + # look dead even though the loop is healthy. + try: + self._check_termios_drift() + except Exception: + pass # Check for background process notifications (completions # and watch pattern matches) while agent is idle. try: diff --git a/tests/cli/test_termios_drift_heal.py b/tests/cli/test_termios_drift_heal.py new file mode 100644 index 0000000000..9cee9c0913 --- /dev/null +++ b/tests/cli/test_termios_drift_heal.py @@ -0,0 +1,94 @@ +"""Regression tests for cooked-mode termios drift healing (cli._heal_cooked_mode_drift). + +Failure class: prompt_toolkit's ``run_in_terminal`` flips the tty to cooked +mode for every "print above the prompt" window and restores raw mode after. +If that restore is lost (cancelled coroutine, racing chained windows, an +external actor), the terminal stays in cooked mode while the prompt_toolkit +Application still believes it owns raw mode — the kernel line-buffers +keystrokes and the CLI appears to stop taking input even though the process +is healthy. Observed live on 2026-08-13 (session 20260813_113539_0f5b00): +pts left in ``icanon echo`` after a background-review + bg-notification +sequence, healed externally with stty. + +These tests exercise the healer against a REAL pty pair — no mocks. +""" + +import os +import sys + +import pytest + +if sys.platform == "win32": # pragma: no cover + pytest.skip("termios drift healing is POSIX-only", allow_module_level=True) + +import termios +import tty as _tty + +from cli import _heal_cooked_mode_drift + + +@pytest.fixture() +def pty_fd(): + master, slave = os.openpty() + try: + yield slave + finally: + os.close(master) + os.close(slave) + + +def _is_cooked(fd: int) -> bool: + lflag = termios.tcgetattr(fd)[3] + return bool(lflag & (termios.ICANON | termios.ECHO)) + + +def test_heals_cooked_drift_back_to_raw(pty_fd): + # ptys start in cooked mode — that IS the drifted state. + assert _is_cooked(pty_fd), "precondition: fresh pty must be cooked" + + healed = _heal_cooked_mode_drift(pty_fd) + + assert healed is True + attrs = termios.tcgetattr(pty_fd) + lflag, iflag, cc = attrs[3], attrs[0], attrs[6] + assert not (lflag & termios.ICANON) + assert not (lflag & termios.ECHO) + assert not (lflag & termios.ISIG) + assert not (lflag & termios.IEXTEN) + # iflag surgery matches prompt_toolkit raw_mode._patch_iflag + for flag in (termios.IXON, termios.IXOFF, termios.ICRNL, + termios.INLCR, termios.IGNCR): + assert not (iflag & flag) + assert cc[termios.VMIN] == 1 + + +def test_noop_when_already_raw(pty_fd): + _tty.setraw(pty_fd) + before = termios.tcgetattr(pty_fd) + + healed = _heal_cooked_mode_drift(pty_fd) + + assert healed is False, "already-raw tty must not be treated as drifted" + assert termios.tcgetattr(pty_fd) == before, "raw tty attrs must be untouched" + + +def test_returns_false_on_non_tty(): + r, w = os.pipe() + try: + assert _heal_cooked_mode_drift(r) is False + assert _heal_cooked_mode_drift(w) is False + finally: + os.close(r) + os.close(w) + + +def test_heal_preserves_unrelated_flags(pty_fd): + # Set a flag the healer must not clobber (OPOST in oflag). + attrs = termios.tcgetattr(pty_fd) + attrs[1] |= termios.OPOST + termios.tcsetattr(pty_fd, termios.TCSANOW, attrs) + + assert _heal_cooked_mode_drift(pty_fd) is True + + after = termios.tcgetattr(pty_fd) + assert after[1] & termios.OPOST, "oflag must be preserved by the heal" From acd8737c10f8fdfa6d3eaba6b22d6fbe6ba2a413 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:25:31 +0530 Subject: [PATCH 029/748] fix(models): ETag conditional GET, no-network hot-path invariant, mirror URL override for models.dev catalog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Harden the models.dev catalog refresh path (#35838) with three missing pieces: 1. ETag conditional GET — every network request sends If-None-Match with the last-known ETag (persisted alongside the cache file). A 304 Not Modified re-confirms the existing cache without re-downloading the full ~2 MB registry. This makes the 4-hour TTL effectively free to maintain. 2. No-network-on-hot-paths invariant — allow_network=False is now the default for every query function called on the conversation hot path: get_model_capabilities, get_model_info, lookup_models_dev_context, _get_provider_models. These are called during vision routing, image routing, cost-guard checks, and context-length resolution on every turn — they must never block on the network. Interactive flows (model picker, model switch) explicitly pass allow_network=True. 3. Mirror URL override — models_dev.url in config.yaml lets deployments point at a self-hosted mirror without code changes. Follows the same pattern as model_catalog.url. Additional hardening: - Cache TTL bumped from 1h to 4h (ETag makes refresh cheap) - Corrupt/empty disk cache is rejected with a warning instead of being served as {} and silently breaking provider/model resolution - _validate_registry() guards against non-dict and empty-dict payloads Fixes #35838 --- agent/models_dev.py | 230 +++++++++++++++++--- hermes_cli/config_defaults.py | 10 + hermes_cli/model_switch.py | 4 +- tests/agent/test_models_dev.py | 377 ++++++++++++++++++++++++++++++++- 4 files changed, 585 insertions(+), 36 deletions(-) diff --git a/agent/models_dev.py b/agent/models_dev.py index e51b086a5a..1f174d5706 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -15,8 +15,23 @@ Data resolution order: served rather than blocking callers on the network) 3. Network fetch (https://models.dev/api.json) — only when no cache exists at all; failed refreshes back off for 5 minutes process-wide -Latency-sensitive callers (gateway route-identity checks) pass -``allow_network=False`` and never touch the network. + +Network hardening: + +- **ETag conditional GET**: every network request sends ``If-None-Match`` + with the last-known ETag. A 304 Not Modified response is a no-op — the + existing cache is re-confirmed fresh without re-downloading the full + registry (≈2 MB). The ETag is persisted alongside the cache file. +- **No-network-on-hot-paths invariant**: resolution, picker, and resume + paths NEVER perform network I/O. ``allow_network=False`` is threaded + through every query function, and hot-path callers (vision routing, + image routing, cost guard, context-length lookup) pass it explicitly. +- **Corrupt-cache rejection**: a disk cache that fails to parse, is not a + dict, or is empty is ignored with a warning rather than served as + ``{}`` and silently breaking provider/model resolution. +- **Mirror URL override**: ``models_dev.url`` in config.yaml lets + deployments point at a mirror (e.g. a self-hosted copy) without code + changes. Other modules should import the dataclasses and query functions from here rather than parsing the raw JSON themselves. @@ -36,8 +51,9 @@ import requests logger = logging.getLogger(__name__) -MODELS_DEV_URL = "https://models.dev/api.json" -_MODELS_DEV_CACHE_TTL = 3600 # 1 hour in-memory +_DEFAULT_MODELS_DEV_URL = "https://models.dev/api.json" +MODELS_DEV_URL = _DEFAULT_MODELS_DEV_URL +_MODELS_DEV_CACHE_TTL = 4 * 3600 # 4 hours — ETag conditional GET makes refresh cheap _MODELS_DEV_RETRY_DELAY = 300 # 5 minutes after a failed refresh # In-memory cache @@ -220,15 +236,81 @@ def _get_cache_path() -> Path: return get_hermes_home() / "models_dev_cache.json" +def _get_etag_path() -> Path: + """Return path to the ETag sidecar file for conditional GET.""" + from hermes_constants import get_hermes_home + return get_hermes_home() / "models_dev_cache.etag" + + +def _load_etag() -> str: + """Load the last-known ETag from disk, or empty string if missing.""" + try: + etag_path = _get_etag_path() + if etag_path.exists(): + return etag_path.read_text(encoding="utf-8").strip() + except Exception as e: + logger.debug("Failed to load models.dev ETag: %s", e) + return "" + + +def _save_etag(etag: str) -> None: + """Persist an ETag to the sidecar file atomically.""" + try: + etag_path = _get_etag_path() + etag_path.parent.mkdir(parents=True, exist_ok=True) + tmp = etag_path.with_suffix(".tmp") + tmp.write_text(etag, encoding="utf-8") + tmp.replace(etag_path) + except Exception as e: + logger.debug("Failed to save models.dev ETag: %s", e) + + +def _get_models_dev_url() -> str: + """Resolve the models.dev API URL, honoring a config.yaml override. + + The ``models_dev.url`` config key lets deployments point at a mirror + (e.g. a self-hosted copy behind a corporate proxy) without code changes. + Falls back to the default public URL when unset or empty. + """ + try: + from hermes_cli.config import cfg_get, load_config_readonly + cfg = load_config_readonly() + url = cfg_get(cfg, "models_dev", "url", default="") + if isinstance(url, str) and url.strip(): + return url.strip() + except Exception: + pass + return _DEFAULT_MODELS_DEV_URL + + +def _validate_registry(data: Any) -> bool: + """Return True if *data* is a non-empty dict suitable for serving.""" + return isinstance(data, dict) and len(data) > 0 + + def _load_disk_cache() -> Dict[str, Any]: - """Load models.dev data from disk cache.""" + """Load models.dev data from disk cache. + + A corrupt cache (invalid JSON, not a dict, or empty) is rejected with + a warning so it doesn't silently masquerade as ``{}`` and break + provider/model resolution for every caller. + """ try: cache_path = _get_cache_path() if cache_path.exists(): with open(cache_path, encoding="utf-8") as f: - return json.load(f) + data = json.load(f) + if not _validate_registry(data): + logger.warning( + "models.dev disk cache is corrupt or empty; ignoring " + "(will refetch from network)" + ) + return {} + return data except Exception as e: - logger.debug("Failed to load models.dev disk cache: %s", e) + logger.warning( + "Failed to load models.dev disk cache; ignoring: %s", e + ) return {} @@ -258,30 +340,62 @@ def _disk_cache_age_seconds() -> Optional[float]: return None -def _save_disk_cache(data: Dict[str, Any]) -> None: - """Save models.dev data to disk cache atomically.""" +def _save_disk_cache(data: Dict[str, Any], etag: str = "") -> None: + """Save models.dev data to disk cache atomically. + + Also persists the ETag sidecar when *etag* is non-empty so the next + refresh can issue a conditional GET. + """ try: cache_path = _get_cache_path() atomic_json_write(cache_path, data, indent=None, separators=(",", ":")) except Exception as e: logger.debug("Failed to save models.dev disk cache: %s", e) + if etag: + _save_etag(etag) + + +class _NotModified(Exception): + """Server returned 304 Not Modified — existing cache is still valid.""" def _fetch_models_dev_from_network() -> Dict[str, Any]: """Fetch the live models.dev registry without touching local caches. + Uses ETag conditional GET: sends ``If-None-Match`` when a cached ETag + exists. A 304 Not Modified response means the cached registry is still + current; this raises ``_NotModified`` so the caller can re-confirm the + existing cache's freshness without re-downloading the full payload. + Raises on network errors and on an empty/invalid registry payload. """ + url = _get_models_dev_url() + headers: Dict[str, str] = {} + etag = _load_etag() + if etag: + headers["If-None-Match"] = etag + # Tuple (connect, read): a flat timeout=15 let a blackholed connect # stall the first-turn critical path for the full 15 s. 5 s connect # fails fast on unreachable hosts; 10 s read still tolerates a slow # registry response (matches the OpenRouter fetch convention in # agent/model_metadata.py). - response = requests.get(MODELS_DEV_URL, timeout=(5, 10)) + response = requests.get(url, headers=headers, timeout=(5, 10)) + + if response.status_code == 304: + raise _NotModified() + response.raise_for_status() data = response.json() if not isinstance(data, dict) or not data: raise ValueError("models.dev returned an empty or invalid registry") + + # Persist the new ETag alongside the cache so the next conditional + # GET can short-circuit. + new_etag = response.headers.get("ETag", "") + if new_etag: + _save_etag(new_etag) + return data @@ -319,6 +433,24 @@ def _commit_registry(data: Dict[str, Any], *, where: str) -> None: ) +def _confirm_cache_not_modified(*, where: str) -> None: + """Re-confirm the existing cache as fresh after a 304 Not Modified. + + Callers must hold ``_models_dev_fetch_lock``. Clears the backoff and + resets the in-memory cache timestamp so the next caller hits the fast + path. The disk cache itself is not rewritten — its contents are + unchanged, only its freshness marker is advanced. + """ + global _models_dev_cache_time, _models_dev_retry_after + _models_dev_cache_time = time.time() + _models_dev_retry_after = 0 + logger.debug( + "models.dev registry unchanged (304 Not Modified, %s); " + "cache re-confirmed fresh", + where, + ) + + def _note_refresh_failure(exc: Exception, *, where: str) -> None: """Record a failed refresh: arm the process-wide 5-minute backoff. @@ -341,6 +473,9 @@ def _background_refresh_models_dev() -> None: data = _fetch_models_dev_from_network() with _models_dev_fetch_lock: _commit_registry(data, where="background") + except _NotModified: + with _models_dev_fetch_lock: + _confirm_cache_not_modified(where="background") except Exception as e: with _models_dev_fetch_lock: _note_refresh_failure(e, where="background") @@ -384,6 +519,11 @@ def fetch_models_dev( Returns the full registry dict keyed by provider ID, or empty dict on failure. + Network requests use ETag conditional GET: when a cached ETag exists, + an ``If-None-Match`` header is sent. A 304 Not Modified response + re-confirms the existing cache's freshness without re-downloading the + full (~2 MB) registry. + Cache hierarchy (when ``force_refresh=False``): 1. Fresh in-memory cache → return immediately. 2. Stale in-memory cache → return immediately and refresh in a single @@ -392,6 +532,7 @@ def fetch_models_dev( new models, so stale data is preferable to a foreground timeout. 3. Disk cache file (any age) → load, populate in-mem, return immediately. Stale disk caches trigger the same background refresh. + A corrupt or empty disk cache is rejected with a warning. 4. No cache at all → singleflight foreground network fetch. On success, save to disk + in-mem and return. 5. Any failed refresh (foreground or background) suppresses further @@ -402,8 +543,9 @@ def fetch_models_dev( backoff are bypassed; the function hits the network and only falls back to cached data if the call fails. When ``allow_network=False``, any memory or disk cache is returned regardless of age and no request is - made — used by latency-sensitive paths (gateway route-identity checks) - that must never wait on the network. + made — used by latency-sensitive paths (gateway route-identity checks, + vision routing, context-length lookup) that must never wait on the + network. """ global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after @@ -488,6 +630,11 @@ def fetch_models_dev( data = _fetch_models_dev_from_network() _commit_registry(data, where="foreground") return data + except _NotModified: + # Server confirmed our cache is still valid. Re-confirm freshness + # without re-downloading the full registry. + _confirm_cache_not_modified(where="foreground") + return _models_dev_cache except Exception as e: _note_refresh_failure(e, where="foreground") @@ -506,7 +653,9 @@ def fetch_models_dev( return _models_dev_cache -def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: +def lookup_models_dev_context( + provider: str, model: str, *, allow_network: bool = False +) -> Optional[int]: """Look up context_length for a provider+model combo in models.dev. Returns the context window in tokens, or None if not found. @@ -516,6 +665,10 @@ def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: wins over the catalog value; ``_default`` entries fill the gap only when the catalog has no answer — the supported self-unblock path for models with wrong or missing context in models.dev (#84482). + + ``allow_network`` defaults to False — context-length lookup is a + hot path (called during every conversation turn) and must never block + on the network. Pass True only from explicit refresh flows. """ # Explicit config override — checked before catalog so it always wins. override_ctx = _override_context_window(provider, model) @@ -526,7 +679,7 @@ def lookup_models_dev_context(provider: str, model: str) -> Optional[int]: if not mdev_provider_id: return _default_override_context(provider) - data = fetch_models_dev() + data = fetch_models_dev(allow_network=allow_network) provider_data = data.get(mdev_provider_id) if not isinstance(provider_data, dict): return _default_override_context(provider) @@ -855,16 +1008,21 @@ def _merge_catalog_entry_with_override( return merged -def _get_provider_models(provider: str) -> Optional[Dict[str, Any]]: +def _get_provider_models( + provider: str, *, allow_network: bool = False +) -> Optional[Dict[str, Any]]: """Resolve a Hermes provider ID to its models dict from models.dev. Returns the models dict or None if the provider is unknown or has no data. + + ``allow_network`` defaults to False — this is called from hot paths + (vision routing, image routing, capability checks) and must never block. """ mdev_provider_id = PROVIDER_TO_MODELS_DEV.get(provider) if not mdev_provider_id: return None - data = fetch_models_dev() + data = fetch_models_dev(allow_network=allow_network) provider_data = data.get(mdev_provider_id) if not isinstance(provider_data, dict): return None @@ -911,7 +1069,9 @@ def _find_model_entry(models: Dict[str, Any], model: str) -> Optional[Dict[str, return None -def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilities]: +def get_model_capabilities( + provider: str, model: str, *, allow_network: bool = False +) -> Optional[ModelCapabilities]: """Look up full capability metadata from models.dev cache. Uses the existing fetch_models_dev() and PROVIDER_TO_MODELS_DEV mapping. @@ -925,6 +1085,9 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit of fields; unspecified fields fall through to the catalog value (or sensible defaults when the model is absent from the catalog). + ``allow_network`` defaults to False — capability lookup is a hot path + (vision routing, image routing) and must never block on the network. + Extracts from model entry fields: - reasoning (bool) → supports_reasoning - tool_call (bool) → supports_tools @@ -933,7 +1096,7 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit - limit.output (int) → max_output_tokens - family (str) → model_family """ - models = _get_provider_models(provider) + models = _get_provider_models(provider, allow_network=allow_network) entry = _find_model_entry(models, model) if models is not None else None # Select the override AFTER the catalog lookup: explicit overrides @@ -1010,15 +1173,21 @@ def get_model_capabilities(provider: str, model: str) -> Optional[ModelCapabilit ) -def list_provider_models(provider: str) -> List[str]: +def list_provider_models( + provider: str, *, allow_network: bool = True +) -> List[str]: """Return all model IDs for a provider from models.dev. Returns an empty list if the provider is unknown or has no data. + + ``allow_network`` defaults to True — this is called from the model + picker (``hermes model``), which is an interactive user-facing flow + where a fresh catalog is worth a short network wait. """ from hermes_cli.models import normalize_provider provider = normalize_provider(provider) or provider - models = _get_provider_models(provider) + models = _get_provider_models(provider, allow_network=allow_network) if models is None: return [] return [ @@ -1074,14 +1243,19 @@ def _should_hide_from_provider_catalog(provider: str, model_id: str) -> bool: return False -def list_agentic_models(provider: str) -> List[str]: +def list_agentic_models( + provider: str, *, allow_network: bool = True +) -> List[str]: """Return model IDs suitable for agentic use from models.dev. Filters for tool_call=True and excludes noise (TTS, embedding, dated preview snapshots, live/streaming, image-only models). Returns an empty list on any failure. + + ``allow_network`` defaults to True — like ``list_provider_models``, + this is called from interactive model selection flows. """ - models = _get_provider_models(provider) + models = _get_provider_models(provider, allow_network=allow_network) if models is None: return [] @@ -1180,6 +1354,11 @@ def get_provider_info( Accepts either a Hermes provider ID (e.g. "kilocode") or a models.dev ID (e.g. "kilo"). Returns None if the provider is not in the catalog. + + ``allow_network`` defaults to True — the primary caller is + ``resolve_provider_full`` during interactive setup, where a fresh + catalog is worth a short network wait. Hot-path callers should pass + ``allow_network=False``. """ # Resolve Hermes ID → models.dev ID mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) @@ -1204,7 +1383,7 @@ def get_provider_info( # --------------------------------------------------------------------------- def get_model_info( - provider_id: str, model_id: str + provider_id: str, model_id: str, *, allow_network: bool = False ) -> Optional[ModelInfo]: """Get full model metadata from models.dev. @@ -1218,6 +1397,9 @@ def get_model_info( ``modalities``) are merged rather than clobbered. EXPLICIT entries patch known catalog models; ``_default`` entries fill the gap only for models the catalog does not know (#8731, #84482). + + ``allow_network`` defaults to False — model info lookup is a hot path + (cost guard, inventory) and must never block on the network. """ mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) @@ -1236,7 +1418,7 @@ def get_model_info( shaped = _merge_catalog_entry_with_override(base, override) return _parse_model_info(model_id, shaped, mdev_id) - data = fetch_models_dev() + data = fetch_models_dev(allow_network=allow_network) pdata = data.get(mdev_id) if not isinstance(pdata, dict): return _from_override_alone() diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index e52290d05b..56936f1b61 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2585,6 +2585,16 @@ DEFAULT_CONFIG = { # context_window: 128000 "model_overrides": {}, + # models.dev registry — provider/model metadata (context windows, + # capabilities, pricing, modalities). The agent fetches this on startup + # and serves from cache; a background daemon refreshes stale data. + # Override ``url`` to point at a mirror (e.g. a self-hosted copy behind + # a corporate proxy). ETag conditional GET ensures refreshes are + # cheap (304 = no download). + "models_dev": { + "url": "", # empty = default https://models.dev/api.json + }, + # Network settings — workarounds for connectivity issues. "network": { # Force IPv4 connections. On servers with broken or unreachable IPv6, diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index aea8efba48..d6cd312c99 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -1911,10 +1911,10 @@ def switch_model( base_url = normalize_opencode_base_url(target_provider, api_mode, base_url) # --- Get capabilities (legacy) --- - capabilities = get_model_capabilities(target_provider, new_model) + capabilities = get_model_capabilities(target_provider, new_model, allow_network=True) # --- Get full model info from models.dev --- - model_info = get_model_info(target_provider, new_model) + model_info = get_model_info(target_provider, new_model, allow_network=True) # --- Collect warnings --- warnings: list[str] = [] diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 67e008b659..52db2653b3 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -13,6 +13,8 @@ from agent.models_dev import ( _explicit_model_override, _override_context_window, _override_for, + _NotModified, + _validate_registry, fetch_models_dev, get_model_capabilities, get_model_info, @@ -161,6 +163,15 @@ class TestFetchModelsDev: md._models_dev_retry_after = 0 md._models_dev_refresh_in_flight = False + def _mock_response(self, data, etag="", status_code=200): + """Build a MagicMock response with optional ETag header.""" + resp = MagicMock() + resp.status_code = status_code + resp.json.return_value = data + resp.headers = {"ETag": etag} if etag else {} + resp.raise_for_status = MagicMock() + return resp + @@ -176,6 +187,7 @@ class TestFetchModelsDev: with patch.object(md, "_disk_cache_age_seconds", return_value=md._MODELS_DEV_CACHE_TTL + 60), \ patch.object(md, "_load_disk_cache", return_value=SAMPLE_REGISTRY), \ + patch.object(md, "_load_etag", return_value=""), \ patch.object(md, "_start_background_refresh_models_dev") as mock_refresh: result = fetch_models_dev() @@ -197,7 +209,8 @@ class TestFetchModelsDev: md, "_disk_cache_age_seconds", return_value=md._MODELS_DEV_CACHE_TTL + 60, - ), patch.object(md, "_load_disk_cache", return_value=SAMPLE_REGISTRY): + ), patch.object(md, "_load_disk_cache", return_value=SAMPLE_REGISTRY), \ + patch.object(md, "_load_etag", return_value=""): first = fetch_models_dev() # Join the background refresh worker so its failure backoff is # observable and requests.get stays patched for its lifetime. @@ -225,20 +238,22 @@ class TestFetchModelsDev: """The bg worker must save disk + swap mem cache + clear backoff.""" import agent.models_dev as md - response = MagicMock() - response.json.return_value = SAMPLE_REGISTRY + response = self._mock_response(SAMPLE_REGISTRY, etag='"abc123"') mock_get.return_value = response md._models_dev_cache = {"stale": {}} md._models_dev_cache_time = 0 md._models_dev_retry_after = time.time() - 1 - with patch.object(md, "_save_disk_cache") as mock_save: + with patch.object(md, "_save_disk_cache") as mock_save, \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag") as mock_save_etag: # Run the worker synchronously — deterministic, no thread. md._models_dev_refresh_in_flight = True md._background_refresh_models_dev() mock_save.assert_called_once_with(SAMPLE_REGISTRY) + mock_save_etag.assert_called_once_with('"abc123"') assert md._models_dev_cache == SAMPLE_REGISTRY assert md._models_dev_cache_time > 0 assert md._models_dev_retry_after == 0 @@ -251,8 +266,7 @@ class TestFetchModelsDev: request_started = threading.Event() release_request = threading.Event() - response = MagicMock() - response.json.return_value = SAMPLE_REGISTRY + response = self._mock_response(SAMPLE_REGISTRY) def blocking_get(*_args, **_kwargs): request_started.set() @@ -262,7 +276,9 @@ class TestFetchModelsDev: mock_get.side_effect = blocking_get with patch.object(md, "_disk_cache_age_seconds", return_value=None), patch.object( md, "_save_disk_cache" - ), ThreadPoolExecutor(max_workers=6) as pool: + ), patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"), \ + ThreadPoolExecutor(max_workers=6) as pool: futures = [pool.submit(fetch_models_dev) for _ in range(6)] assert request_started.wait(timeout=2) release_request.set() @@ -275,13 +291,14 @@ class TestFetchModelsDev: def test_force_refresh_bypasses_failure_backoff(self, mock_get): import agent.models_dev as md - response = MagicMock() - response.json.return_value = SAMPLE_REGISTRY + response = self._mock_response(SAMPLE_REGISTRY) mock_get.side_effect = [OSError("models.dev unreachable"), response] with patch.object(md, "_disk_cache_age_seconds", return_value=None), patch.object( md, "_load_disk_cache", return_value={} - ), patch.object(md, "_save_disk_cache"): + ), patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"): assert fetch_models_dev() == {} assert fetch_models_dev(force_refresh=True) == SAMPLE_REGISTRY @@ -322,6 +339,346 @@ class TestFetchModelsDev: +# --------------------------------------------------------------------------- +# ETag conditional GET +# --------------------------------------------------------------------------- + + +class TestETagConditionalGet: + """Tests for ETag-based conditional GET (If-None-Match / 304 handling).""" + + @pytest.fixture(autouse=True) + def _reset_fetch_state(self): + import agent.models_dev as md + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + md._models_dev_retry_after = 0 + md._models_dev_refresh_in_flight = False + yield + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + md._models_dev_retry_after = 0 + md._models_dev_refresh_in_flight = False + + @patch("agent.models_dev.requests.get") + def test_etag_sent_when_cached(self, mock_get): + """If-None-Match header is sent when a cached ETag exists.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {"ETag": '"v2"'} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value='"v1"'), \ + patch.object(md, "_save_etag"): + fetch_models_dev() + + call_kwargs = mock_get.call_args + headers = call_kwargs.kwargs.get("headers", {}) + assert headers.get("If-None-Match") == '"v1"' + + @patch("agent.models_dev.requests.get") + def test_304_reconfirms_cache_freshness(self, mock_get): + """A 304 Not Modified re-confirms the existing cache without download.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 304 + mock_get.return_value = response + + md._models_dev_cache = SAMPLE_REGISTRY + md._models_dev_cache_time = 0 + md._models_dev_retry_after = time.time() + 100 # backoff was armed + + with patch.object(md, "_load_etag", return_value='"v1"'), \ + patch.object(md, "_save_etag"): + # Run the background worker synchronously + md._models_dev_refresh_in_flight = True + md._background_refresh_models_dev() + + # Cache content unchanged + assert md._models_dev_cache == SAMPLE_REGISTRY + # Freshness timestamp advanced + assert md._models_dev_cache_time > 0 + # Backoff cleared + assert md._models_dev_retry_after == 0 + assert not md._models_dev_refresh_in_flight + # response.json() was never called — no body to parse + response.json.assert_not_called() + + @patch("agent.models_dev.requests.get") + def test_foreground_304_returns_existing_cache(self, mock_get): + """Foreground fetch with 304 returns the existing cache.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 304 + mock_get.return_value = response + + md._models_dev_cache = SAMPLE_REGISTRY + md._models_dev_cache_time = 0 + md._models_dev_retry_after = 0 + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_load_etag", return_value='"v1"'), \ + patch.object(md, "_save_etag"): + result = fetch_models_dev(force_refresh=True) + + assert result == SAMPLE_REGISTRY + assert md._models_dev_cache_time > 0 + + @patch("agent.models_dev.requests.get") + def test_new_etag_persisted_after_successful_fetch(self, mock_get): + """A successful fetch with an ETag in the response persists it.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {"ETag": '"new-etag"'} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag") as mock_save_etag: + fetch_models_dev() + + mock_save_etag.assert_called_once_with('"new-etag"') + + @patch("agent.models_dev.requests.get") + def test_no_etag_header_sent_without_cached_etag(self, mock_get): + """No If-None-Match header when no cached ETag exists.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"): + fetch_models_dev() + + call_kwargs = mock_get.call_args + headers = call_kwargs.kwargs.get("headers", {}) + assert "If-None-Match" not in headers + + +# --------------------------------------------------------------------------- +# Corrupt / invalid cache rejection +# --------------------------------------------------------------------------- + + +class TestCorruptCacheRejection: + """A corrupt or empty disk cache must be rejected, not served as {}.""" + + def test_validate_registry_rejects_empty_dict(self): + assert not _validate_registry({}) + + def test_validate_registry_rejects_non_dict(self): + assert not _validate_registry("not a dict") + assert not _validate_registry(None) + assert not _validate_registry([]) + + def test_validate_registry_accepts_populated_dict(self): + assert _validate_registry({"anthropic": {}}) + + @patch("agent.models_dev.requests.get") + def test_corrupt_json_rejected_with_warning(self, mock_get, caplog): + """Invalid JSON on disk is ignored, not served as {}.""" + import agent.models_dev as md + import json as _json + + mock_get.side_effect = OSError("unreachable") + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + + with patch.object(md, "_disk_cache_age_seconds", return_value=0), \ + patch.object(md, "_get_cache_path") as mock_path, \ + patch.object(md, "_load_etag", return_value=""): + mock_path.return_value.exists.return_value = True + mock_path.return_value.open.return_value.__enter__.return_value.read.return_value = "not json" + # json.load will raise on invalid JSON + with patch("builtins.open", side_effect=_json.JSONDecodeError("msg", "doc", 0)): + with patch.object(md, "_load_disk_cache", wraps=md._load_disk_cache): + result = fetch_models_dev() + + # Returns empty dict, not the corrupt data + assert result == {} + + @patch("agent.models_dev.requests.get") + def test_empty_dict_cache_rejected(self, mock_get, caplog): + """An empty dict in the cache file is rejected with a warning.""" + import agent.models_dev as md + import logging + + mock_get.side_effect = OSError("unreachable") + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + + with patch.object(md, "_disk_cache_age_seconds", return_value=0), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_disk_cache"): + with caplog.at_level(logging.WARNING): + # _load_disk_cache returns {} for empty dict, which is correct + result = fetch_models_dev() + + assert result == {} + + +# --------------------------------------------------------------------------- +# Mirror URL override via config +# --------------------------------------------------------------------------- + + +class TestMirrorUrlOverride: + """models_dev.url config key overrides the API endpoint.""" + + @pytest.fixture(autouse=True) + def _reset_fetch_state(self): + import agent.models_dev as md + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + md._models_dev_retry_after = 0 + md._models_dev_refresh_in_flight = False + yield + md._models_dev_cache = {} + md._models_dev_cache_time = 0 + md._models_dev_retry_after = 0 + md._models_dev_refresh_in_flight = False + + @patch("agent.models_dev.requests.get") + def test_mirror_url_used_when_configured(self, mock_get): + """When config has models_dev.url, requests.get hits that URL.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + fake_config = {"models_dev": {"url": "https://mirror.example.com/api.json"}} + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"), \ + patch("hermes_cli.config.load_config_readonly", return_value=fake_config): + fetch_models_dev() + + call_args = mock_get.call_args + assert "mirror.example.com" in call_args.args[0] + + @patch("agent.models_dev.requests.get") + def test_default_url_used_when_not_configured(self, mock_get): + """Without config override, the default models.dev URL is used.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"), \ + patch("hermes_cli.config.load_config_readonly", return_value={}): + fetch_models_dev() + + call_args = mock_get.call_args + assert "models.dev" in call_args.args[0] + + @patch("agent.models_dev.requests.get") + def test_empty_url_falls_back_to_default(self, mock_get): + """An empty string URL in config falls back to the default.""" + import agent.models_dev as md + + response = MagicMock() + response.status_code = 200 + response.json.return_value = SAMPLE_REGISTRY + response.headers = {} + response.raise_for_status = MagicMock() + mock_get.return_value = response + + fake_config = {"models_dev": {"url": ""}} + + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ + patch.object(md, "_load_disk_cache", return_value={}), \ + patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_load_etag", return_value=""), \ + patch.object(md, "_save_etag"), \ + patch("hermes_cli.config.load_config_readonly", return_value=fake_config): + fetch_models_dev() + + call_args = mock_get.call_args + assert "models.dev" in call_args.args[0] + + +# --------------------------------------------------------------------------- +# No-network-on-hot-paths invariant +# --------------------------------------------------------------------------- + + +class TestNoNetworkOnHotPaths: + """Query functions must default to allow_network=False on hot paths.""" + + @patch("agent.models_dev.requests.get") + def test_get_model_capabilities_default_no_network(self, mock_get): + """get_model_capabilities defaults to allow_network=False.""" + with patch("agent.models_dev.fetch_models_dev") as mock_fetch: + mock_fetch.return_value = CAPS_REGISTRY + get_model_capabilities("anthropic", "claude-sonnet-4") + # fetch_models_dev was called with allow_network=False + mock_fetch.assert_called_once_with(allow_network=False) + + @patch("agent.models_dev.requests.get") + def test_get_model_info_default_no_network(self, mock_get): + """get_model_info defaults to allow_network=False.""" + with patch("agent.models_dev.fetch_models_dev") as mock_fetch: + mock_fetch.return_value = SAMPLE_REGISTRY + get_model_info("anthropic", "claude-opus-4-6") + mock_fetch.assert_called_once_with(allow_network=False) + + @patch("agent.models_dev.requests.get") + def test_lookup_models_dev_context_default_no_network(self, mock_get): + """lookup_models_dev_context defaults to allow_network=False.""" + with patch("agent.models_dev.fetch_models_dev") as mock_fetch: + mock_fetch.return_value = SAMPLE_REGISTRY + lookup_models_dev_context("anthropic", "claude-opus-4-6") + mock_fetch.assert_called_once_with(allow_network=False) + + @patch("agent.models_dev.requests.get") + def test_get_model_capabilities_explicit_network(self, mock_get): + """get_model_capabilities can opt into network.""" + with patch("agent.models_dev.fetch_models_dev") as mock_fetch: + mock_fetch.return_value = CAPS_REGISTRY + get_model_capabilities("anthropic", "claude-sonnet-4", allow_network=True) + mock_fetch.assert_called_once_with(allow_network=True) + + # --------------------------------------------------------------------------- # get_model_capabilities — vision via modalities.input # --------------------------------------------------------------------------- From b1ce502535b413a6308137cc214cd739036f762c Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:26:18 +0530 Subject: [PATCH 030/748] fix(models): close review findings on the ETag refresh path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Conditional GET now requires a servable in-memory registry: an If-None-Match sent while holding no cache invited a 304 against nothing, permanently serving {} with a blocking foreground fetch on every call (the exact #35838 class this PR fixes). Empirically repro'd and verified fixed (corrupt cache + stale sidecar: was 3 calls -> {} forever; now 1 unconditional fetch -> real data). - ETag persists atomically WITH the cache body via _commit_registry -> _save_disk_cache(data, etag), wiring up the previously-dead etag param; the sidecar can no longer get ahead of the registry it vouches for. _save_etag now uses utils.atomic_write_text (unique tempnames + fsync) instead of a hand-rolled fixed-name .tmp replace. - Corrupt/unreadable disk cache clears the ETag sidecar so the refetch is unconditional; _confirm_cache_not_modified keeps a defense-in-depth guard (clear sidecar + arm backoff) should a 304 ever land on an empty registry. - allow_network=True paths use the zero-arg fetch_models_dev() call shape at all sites (was 1 of 5) — ~46 test sites monkeypatch it with zero-arg lambdas; the unconditional kwarg broke test_xiaomi_provider (verified fail->pass). - _get_models_dev_url falls back to the MODELS_DEV_URL module global (not the constant) so existing patch sites keep working. - Tests: replaced two mock-riddled corrupt-cache tests with real tmp_path file tests; added regression tests for the 304/empty-cache loop, sidecar clearing, and conditional-GET gating. --- agent/models_dev.py | 117 ++++++++++++++++++------- tests/agent/test_models_dev.py | 153 ++++++++++++++++++++++++--------- 2 files changed, 199 insertions(+), 71 deletions(-) diff --git a/agent/models_dev.py b/agent/models_dev.py index 1f174d5706..6bd607bcc4 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -256,15 +256,28 @@ def _load_etag() -> str: def _save_etag(etag: str) -> None: """Persist an ETag to the sidecar file atomically.""" try: + from utils import atomic_write_text + etag_path = _get_etag_path() etag_path.parent.mkdir(parents=True, exist_ok=True) - tmp = etag_path.with_suffix(".tmp") - tmp.write_text(etag, encoding="utf-8") - tmp.replace(etag_path) + atomic_write_text(etag_path, etag) except Exception as e: logger.debug("Failed to save models.dev ETag: %s", e) +def _clear_etag() -> None: + """Delete the ETag sidecar so the next fetch is unconditional. + + Called when the cached registry the ETag vouches for is gone or + unusable — sending If-None-Match without a servable cache invites a + 304 that would leave the process with no data at all. + """ + try: + _get_etag_path().unlink(missing_ok=True) + except Exception as e: + logger.debug("Failed to clear models.dev ETag: %s", e) + + def _get_models_dev_url() -> str: """Resolve the models.dev API URL, honoring a config.yaml override. @@ -280,7 +293,9 @@ def _get_models_dev_url() -> str: return url.strip() except Exception: pass - return _DEFAULT_MODELS_DEV_URL + # Fall back to the module global (not the constant) so existing + # code/tests that patch MODELS_DEV_URL keep working. + return MODELS_DEV_URL def _validate_registry(data: Any) -> bool: @@ -305,12 +320,17 @@ def _load_disk_cache() -> Dict[str, Any]: "models.dev disk cache is corrupt or empty; ignoring " "(will refetch from network)" ) + # The sidecar vouches for a registry we no longer hold — + # drop it so the refetch is unconditional (a 304 against + # a missing cache would leave us with no data at all). + _clear_etag() return {} return data except Exception as e: logger.warning( "Failed to load models.dev disk cache; ignoring: %s", e ) + _clear_etag() return {} @@ -359,21 +379,30 @@ class _NotModified(Exception): """Server returned 304 Not Modified — existing cache is still valid.""" -def _fetch_models_dev_from_network() -> Dict[str, Any]: +def _fetch_models_dev_from_network() -> Tuple[Dict[str, Any], str]: """Fetch the live models.dev registry without touching local caches. Uses ETag conditional GET: sends ``If-None-Match`` when a cached ETag - exists. A 304 Not Modified response means the cached registry is still - current; this raises ``_NotModified`` so the caller can re-confirm the - existing cache's freshness without re-downloading the full payload. + exists AND the process holds a servable registry the 304 can + re-confirm. A conditional request without a cache invites a 304 that + leaves the process with no data at all (and, before this guard, a + permanent empty-registry loop when the sidecar outlived a corrupt + cache file). A 304 raises ``_NotModified`` so the caller can + re-confirm the existing cache's freshness without re-downloading the + full payload. - Raises on network errors and on an empty/invalid registry payload. + Returns ``(registry, etag)``; the etag is empty when the server sent + none. The caller persists it together with the cache body + (``_commit_registry``) so the sidecar can never get ahead of the data + it vouches for. Raises on network errors and on an empty/invalid + registry payload. """ url = _get_models_dev_url() headers: Dict[str, str] = {} - etag = _load_etag() - if etag: - headers["If-None-Match"] = etag + if _models_dev_cache: + etag = _load_etag() + if etag: + headers["If-None-Match"] = etag # Tuple (connect, read): a flat timeout=15 let a blackholed connect # stall the first-turn critical path for the full 15 s. 5 s connect @@ -387,16 +416,10 @@ def _fetch_models_dev_from_network() -> Dict[str, Any]: response.raise_for_status() data = response.json() - if not isinstance(data, dict) or not data: + if not _validate_registry(data): raise ValueError("models.dev returned an empty or invalid registry") - # Persist the new ETag alongside the cache so the next conditional - # GET can short-circuit. - new_etag = response.headers.get("ETag", "") - if new_etag: - _save_etag(new_etag) - - return data + return data, response.headers.get("ETag", "") def _mark_stale_cache_grace() -> None: @@ -412,7 +435,7 @@ def _mark_stale_cache_grace() -> None: _models_dev_cache_time = grace_time -def _commit_registry(data: Dict[str, Any], *, where: str) -> None: +def _commit_registry(data: Dict[str, Any], *, etag: str = "", where: str) -> None: """Persist a freshly fetched registry: disk + in-mem + clear backoff. Callers must hold ``_models_dev_fetch_lock`` so a failing refresh on one @@ -421,7 +444,7 @@ def _commit_registry(data: Dict[str, Any], *, where: str) -> None: immediately after a successful ``force_refresh``). """ global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after - _save_disk_cache(data) + _save_disk_cache(data, etag) _models_dev_cache = data _models_dev_cache_time = time.time() _models_dev_retry_after = 0 @@ -442,6 +465,21 @@ def _confirm_cache_not_modified(*, where: str) -> None: unchanged, only its freshness marker is advanced. """ global _models_dev_cache_time, _models_dev_retry_after + if not _models_dev_cache: + # Pathological: a 304 arrived but we hold no registry. Should be + # unreachable now that conditional GETs require a servable cache + # (see _fetch_models_dev_from_network); kept as defense in depth + # because this state previously caused a permanent empty-registry + # loop. Drop the sidecar so the next attempt is unconditional and + # arm the normal failure backoff instead of marking {} "fresh". + _clear_etag() + _models_dev_retry_after = time.time() + _MODELS_DEV_RETRY_DELAY + logger.warning( + "models.dev returned 304 but no cached registry is held (%s); " + "cleared ETag sidecar, will refetch unconditionally", + where, + ) + return _models_dev_cache_time = time.time() _models_dev_retry_after = 0 logger.debug( @@ -470,9 +508,9 @@ def _background_refresh_models_dev() -> None: """Best-effort refresh after serving stale cache data.""" global _models_dev_refresh_in_flight try: - data = _fetch_models_dev_from_network() + data, etag = _fetch_models_dev_from_network() with _models_dev_fetch_lock: - _commit_registry(data, where="background") + _commit_registry(data, etag=etag, where="background") except _NotModified: with _models_dev_fetch_lock: _confirm_cache_not_modified(where="background") @@ -627,8 +665,8 @@ def fetch_models_dev( return _models_dev_cache try: - data = _fetch_models_dev_from_network() - _commit_registry(data, where="foreground") + data, etag = _fetch_models_dev_from_network() + _commit_registry(data, etag=etag, where="foreground") return data except _NotModified: # Server confirmed our cache is still valid. Re-confirm freshness @@ -679,7 +717,14 @@ def lookup_models_dev_context( if not mdev_provider_id: return _default_override_context(provider) - data = fetch_models_dev(allow_network=allow_network) + # NOTE: keep the zero-argument call on the allow_network path. Dozens + # of test sites monkeypatch fetch_models_dev with zero-arg lambdas; + # passing the kwarg unconditionally breaks them all (TypeError). + data = ( + fetch_models_dev() + if allow_network + else fetch_models_dev(allow_network=False) + ) provider_data = data.get(mdev_provider_id) if not isinstance(provider_data, dict): return _default_override_context(provider) @@ -1022,7 +1067,14 @@ def _get_provider_models( if not mdev_provider_id: return None - data = fetch_models_dev(allow_network=allow_network) + # NOTE: keep the zero-argument call on the allow_network path. Dozens + # of test sites monkeypatch fetch_models_dev with zero-arg lambdas; + # passing the kwarg unconditionally breaks them all (TypeError). + data = ( + fetch_models_dev() + if allow_network + else fetch_models_dev(allow_network=False) + ) provider_data = data.get(mdev_provider_id) if not isinstance(provider_data, dict): return None @@ -1418,7 +1470,14 @@ def get_model_info( shaped = _merge_catalog_entry_with_override(base, override) return _parse_model_info(model_id, shaped, mdev_id) - data = fetch_models_dev(allow_network=allow_network) + # NOTE: keep the zero-argument call on the allow_network path. Dozens + # of test sites monkeypatch fetch_models_dev with zero-arg lambdas; + # passing the kwarg unconditionally breaks them all (TypeError). + data = ( + fetch_models_dev() + if allow_network + else fetch_models_dev(allow_network=False) + ) pdata = data.get(mdev_id) if not isinstance(pdata, dict): return _from_override_alone() diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 52db2653b3..222c87098b 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -252,8 +252,10 @@ class TestFetchModelsDev: md._models_dev_refresh_in_flight = True md._background_refresh_models_dev() - mock_save.assert_called_once_with(SAMPLE_REGISTRY) - mock_save_etag.assert_called_once_with('"abc123"') + # ETag is committed together with the cache body so the sidecar + # can never get ahead of the data it vouches for. + mock_save.assert_called_once_with(SAMPLE_REGISTRY, '"abc123"') + mock_save_etag.assert_not_called() assert md._models_dev_cache == SAMPLE_REGISTRY assert md._models_dev_cache_time > 0 assert md._models_dev_retry_after == 0 @@ -372,12 +374,17 @@ class TestETagConditionalGet: response.raise_for_status = MagicMock() mock_get.return_value = response + # Conditional GET requires a servable in-memory registry — an + # If-None-Match without one invites a 304 against nothing. + md._models_dev_cache = SAMPLE_REGISTRY + md._models_dev_cache_time = 0 + with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ patch.object(md, "_load_disk_cache", return_value={}), \ patch.object(md, "_save_disk_cache"), \ patch.object(md, "_load_etag", return_value='"v1"'), \ patch.object(md, "_save_etag"): - fetch_models_dev() + fetch_models_dev(force_refresh=True) call_kwargs = mock_get.call_args headers = call_kwargs.kwargs.get("headers", {}) @@ -448,12 +455,14 @@ class TestETagConditionalGet: with patch.object(md, "_disk_cache_age_seconds", return_value=None), \ patch.object(md, "_load_disk_cache", return_value={}), \ - patch.object(md, "_save_disk_cache"), \ + patch.object(md, "_save_disk_cache") as mock_save, \ patch.object(md, "_load_etag", return_value=""), \ patch.object(md, "_save_etag") as mock_save_etag: fetch_models_dev() - mock_save_etag.assert_called_once_with('"new-etag"') + # ETag rides along with the cache body into _save_disk_cache. + mock_save.assert_called_once_with(SAMPLE_REGISTRY, '"new-etag"') + mock_save_etag.assert_not_called() @patch("agent.models_dev.requests.get") def test_no_etag_header_sent_without_cached_etag(self, mock_get): @@ -498,48 +507,105 @@ class TestCorruptCacheRejection: def test_validate_registry_accepts_populated_dict(self): assert _validate_registry({"anthropic": {}}) - @patch("agent.models_dev.requests.get") - def test_corrupt_json_rejected_with_warning(self, mock_get, caplog): - """Invalid JSON on disk is ignored, not served as {}.""" - import agent.models_dev as md - import json as _json - - mock_get.side_effect = OSError("unreachable") - md._models_dev_cache = {} - md._models_dev_cache_time = 0 - - with patch.object(md, "_disk_cache_age_seconds", return_value=0), \ - patch.object(md, "_get_cache_path") as mock_path, \ - patch.object(md, "_load_etag", return_value=""): - mock_path.return_value.exists.return_value = True - mock_path.return_value.open.return_value.__enter__.return_value.read.return_value = "not json" - # json.load will raise on invalid JSON - with patch("builtins.open", side_effect=_json.JSONDecodeError("msg", "doc", 0)): - with patch.object(md, "_load_disk_cache", wraps=md._load_disk_cache): - result = fetch_models_dev() - - # Returns empty dict, not the corrupt data - assert result == {} - - @patch("agent.models_dev.requests.get") - def test_empty_dict_cache_rejected(self, mock_get, caplog): - """An empty dict in the cache file is rejected with a warning.""" - import agent.models_dev as md + def test_corrupt_json_on_disk_rejected_with_warning(self, tmp_path, caplog): + """Invalid JSON in a REAL cache file is rejected with a warning.""" import logging - mock_get.side_effect = OSError("unreachable") - md._models_dev_cache = {} - md._models_dev_cache_time = 0 + import agent.models_dev as md - with patch.object(md, "_disk_cache_age_seconds", return_value=0), \ - patch.object(md, "_load_disk_cache", return_value={}), \ - patch.object(md, "_load_etag", return_value=""), \ - patch.object(md, "_save_disk_cache"): + cache = tmp_path / "models_dev_cache.json" + cache.write_text("not json{{{", encoding="utf-8") + with patch.object(md, "_get_cache_path", return_value=cache), \ + patch.object(md, "_get_etag_path", return_value=tmp_path / "models_dev_cache.etag"): with caplog.at_level(logging.WARNING): - # _load_disk_cache returns {} for empty dict, which is correct - result = fetch_models_dev() + result = md._load_disk_cache() assert result == {} + assert any("disk cache" in r.message for r in caplog.records) + + def test_empty_dict_on_disk_rejected_with_warning(self, tmp_path, caplog): + """A REAL cache file containing {} is rejected with a warning.""" + import logging + + import agent.models_dev as md + + cache = tmp_path / "models_dev_cache.json" + cache.write_text("{}", encoding="utf-8") + with patch.object(md, "_get_cache_path", return_value=cache), \ + patch.object(md, "_get_etag_path", return_value=tmp_path / "models_dev_cache.etag"): + with caplog.at_level(logging.WARNING): + result = md._load_disk_cache() + + assert result == {} + assert any("corrupt or empty" in r.message for r in caplog.records) + + def test_corrupt_cache_clears_etag_sidecar(self, tmp_path): + """Rejecting a corrupt cache must drop the ETag sidecar (#35838 loop). + + If the sidecar outlives the registry it vouches for, the next + conditional GET draws a 304 against nothing and the process serves + {} forever. Clearing the sidecar forces an unconditional refetch. + """ + import agent.models_dev as md + + cache = tmp_path / "models_dev_cache.json" + etag = tmp_path / "models_dev_cache.etag" + cache.write_text("corrupt!!", encoding="utf-8") + etag.write_text("stale-etag", encoding="utf-8") + + with patch.object(md, "_get_cache_path", return_value=cache), \ + patch.object(md, "_get_etag_path", return_value=etag): + result = md._load_disk_cache() + + assert result == {} + assert not etag.exists() + + def test_conditional_get_skipped_without_servable_cache(self): + """No If-None-Match header when the process holds no registry. + + A conditional GET without a servable cache invites a 304 that + leaves the process with no data at all — the permanent + empty-registry loop. The header is only sent when _models_dev_cache + is populated. + """ + import agent.models_dev as md + + captured: dict = {} + + def fake_get(url, headers=None, timeout=None): + captured["headers"] = dict(headers or {}) + resp = MagicMock() + resp.status_code = 200 + resp.json.return_value = {"anthropic": {"models": {}}} + resp.headers = {"ETag": "fresh"} + return resp + + with patch.object(md.requests, "get", side_effect=fake_get), \ + patch.object(md, "_load_etag", return_value="stale-etag"), \ + patch.object(md, "_models_dev_cache", {}): + data, etag = md._fetch_models_dev_from_network() + + assert "If-None-Match" not in captured["headers"] + assert data == {"anthropic": {"models": {}}} + assert etag == "fresh" + + def test_304_with_empty_cache_arms_backoff_and_clears_etag(self, tmp_path): + """Defense in depth: a 304 landing on an empty registry must not + mark {} as fresh — it clears the sidecar and arms the backoff.""" + import agent.models_dev as md + + etag = tmp_path / "models_dev_cache.etag" + etag.write_text("stale", encoding="utf-8") + + with patch.object(md, "_get_etag_path", return_value=etag), \ + patch.object(md, "_models_dev_cache", {}): + before = md._models_dev_retry_after + try: + md._confirm_cache_not_modified(where="test") + assert not etag.exists() + assert md._models_dev_retry_after > time.time() - 1 + finally: + md._models_dev_retry_after = before # --------------------------------------------------------------------------- @@ -676,7 +742,10 @@ class TestNoNetworkOnHotPaths: with patch("agent.models_dev.fetch_models_dev") as mock_fetch: mock_fetch.return_value = CAPS_REGISTRY get_model_capabilities("anthropic", "claude-sonnet-4", allow_network=True) - mock_fetch.assert_called_once_with(allow_network=True) + # allow_network=True uses the zero-arg call shape so the dozens of + # test sites that monkeypatch fetch_models_dev with zero-arg + # lambdas keep working. + mock_fetch.assert_called_once_with() # --------------------------------------------------------------------------- From 5b4c91f1dbaa3d446273504218fc8d237011e039 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 01:52:58 +0530 Subject: [PATCH 031/748] refactor(models): simplify-pass follow-ups on the refresh path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Cold force_refresh (fresh CLI process, e.g. hermes config refresh) now hydrates the memory cache from disk before fetching, so the conditional GET actually fires on the flow the feature was built for instead of silently re-downloading the full ~2 MB registry (empirically probed: If-None-Match sent, 304 serves disk data). - Conditional-GET decision is passed in explicitly (_fetch_models_dev_from_network(conditional=...)) by callers holding the fetch lock, removing the hidden read of module globals inside the fetch; the background worker now fetches INSIDE the lock, symmetric with foreground (true singleflight — no concurrent double-download, no fetching against mid-commit etag state). - Corrupt disk cache is QUARANTINED (renamed to .json.corrupt) rather than left in place: rejection becomes a one-time event instead of a re-read + re-parse + warning + unlink on every hot-path call while offline (probed: 1 warning across 5 calls, was 5). - Dropped the dead _DEFAULT_MODELS_DEV_URL constant; module and function docstrings updated to match the servable-cache conditional semantics. --- agent/models_dev.py | 103 +++++++++++++++++++++++---------- tests/agent/test_models_dev.py | 4 ++ 2 files changed, 76 insertions(+), 31 deletions(-) diff --git a/agent/models_dev.py b/agent/models_dev.py index 6bd607bcc4..fca9201eb6 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -18,10 +18,12 @@ Data resolution order: Network hardening: -- **ETag conditional GET**: every network request sends ``If-None-Match`` - with the last-known ETag. A 304 Not Modified response is a no-op — the - existing cache is re-confirmed fresh without re-downloading the full - registry (≈2 MB). The ETag is persisted alongside the cache file. +- **ETag conditional GET**: network refreshes send ``If-None-Match`` + with the last-known ETag whenever a servable registry is held (memory, + hydrated from disk on cold force-refresh). A 304 Not Modified response + is a no-op — the existing cache is re-confirmed fresh without + re-downloading the full registry (≈2 MB). The ETag is persisted + atomically alongside the cache file. - **No-network-on-hot-paths invariant**: resolution, picker, and resume paths NEVER perform network I/O. ``allow_network=False`` is threaded through every query function, and hot-path callers (vision routing, @@ -51,8 +53,7 @@ import requests logger = logging.getLogger(__name__) -_DEFAULT_MODELS_DEV_URL = "https://models.dev/api.json" -MODELS_DEV_URL = _DEFAULT_MODELS_DEV_URL +MODELS_DEV_URL = "https://models.dev/api.json" _MODELS_DEV_CACHE_TTL = 4 * 3600 # 4 hours — ETag conditional GET makes refresh cheap _MODELS_DEV_RETRY_DELAY = 300 # 5 minutes after a failed refresh @@ -317,23 +318,40 @@ def _load_disk_cache() -> Dict[str, Any]: data = json.load(f) if not _validate_registry(data): logger.warning( - "models.dev disk cache is corrupt or empty; ignoring " - "(will refetch from network)" + "models.dev disk cache is corrupt or empty; " + "quarantining (will refetch from network)" ) - # The sidecar vouches for a registry we no longer hold — - # drop it so the refetch is unconditional (a 304 against - # a missing cache would leave us with no data at all). - _clear_etag() + _quarantine_corrupt_cache(cache_path) return {} return data except Exception as e: logger.warning( - "Failed to load models.dev disk cache; ignoring: %s", e + "Failed to load models.dev disk cache; quarantining: %s", e ) - _clear_etag() + try: + _quarantine_corrupt_cache(_get_cache_path()) + except Exception: + pass return {} +def _quarantine_corrupt_cache(cache_path: Path) -> None: + """Move a rejected cache aside and drop its ETag sidecar. + + Renaming (rather than leaving the file in place) makes the rejection + a one-time event: without it, every hot-path call that finds the + in-memory cache empty re-reads and re-parses the corrupt file and + re-emits the warning until a network fetch succeeds. The sidecar is + cleared because it vouches for a registry we no longer hold — a 304 + against a missing cache would leave the process with no data at all. + """ + try: + cache_path.rename(cache_path.with_suffix(".json.corrupt")) + except Exception as e: + logger.debug("Could not quarantine corrupt models.dev cache: %s", e) + _clear_etag() + + def _disk_cache_age_seconds() -> Optional[float]: """Return age (in seconds) of the disk cache file, or None if missing. @@ -379,17 +397,19 @@ class _NotModified(Exception): """Server returned 304 Not Modified — existing cache is still valid.""" -def _fetch_models_dev_from_network() -> Tuple[Dict[str, Any], str]: - """Fetch the live models.dev registry without touching local caches. +def _fetch_models_dev_from_network( + *, conditional: bool = False +) -> Tuple[Dict[str, Any], str]: + """Fetch the live models.dev registry. - Uses ETag conditional GET: sends ``If-None-Match`` when a cached ETag - exists AND the process holds a servable registry the 304 can - re-confirm. A conditional request without a cache invites a 304 that - leaves the process with no data at all (and, before this guard, a - permanent empty-registry loop when the sidecar outlived a corrupt - cache file). A 304 raises ``_NotModified`` so the caller can - re-confirm the existing cache's freshness without re-downloading the - full payload. + ``conditional`` enables ETag conditional GET (``If-None-Match`` with + the sidecar's ETag). Callers must pass True ONLY while holding + ``_models_dev_fetch_lock`` AND holding a servable registry the 304 + can re-confirm — a conditional request without one invites a 304 + that leaves the process with no data at all (previously a permanent + empty-registry loop when the sidecar outlived a corrupt cache file). + A 304 raises ``_NotModified`` so the caller can re-confirm the + existing cache's freshness without re-downloading the full payload. Returns ``(registry, etag)``; the etag is empty when the server sent none. The caller persists it together with the cache body @@ -399,7 +419,7 @@ def _fetch_models_dev_from_network() -> Tuple[Dict[str, Any], str]: """ url = _get_models_dev_url() headers: Dict[str, str] = {} - if _models_dev_cache: + if conditional: etag = _load_etag() if etag: headers["If-None-Match"] = etag @@ -508,8 +528,15 @@ def _background_refresh_models_dev() -> None: """Best-effort refresh after serving stale cache data.""" global _models_dev_refresh_in_flight try: - data, etag = _fetch_models_dev_from_network() + # Fetch INSIDE the lock: symmetric with the foreground path, so + # conditional-GET inputs (memory cache + etag sidecar) can't be + # mutated mid-fetch by a concurrent force_refresh, and the two + # paths can't double-download concurrently. Hot-path callers are + # unaffected — they return stale data without touching this lock. with _models_dev_fetch_lock: + data, etag = _fetch_models_dev_from_network( + conditional=bool(_models_dev_cache) + ) _commit_registry(data, etag=etag, where="background") except _NotModified: with _models_dev_fetch_lock: @@ -557,10 +584,11 @@ def fetch_models_dev( Returns the full registry dict keyed by provider ID, or empty dict on failure. - Network requests use ETag conditional GET: when a cached ETag exists, - an ``If-None-Match`` header is sent. A 304 Not Modified response - re-confirms the existing cache's freshness without re-downloading the - full (~2 MB) registry. + Network requests use ETag conditional GET when a cached ETag exists + AND a servable registry is held (on a cold ``force_refresh`` the + memory cache is hydrated from disk first). A 304 Not Modified + response re-confirms the existing cache's freshness without + re-downloading the full (~2 MB) registry. Cache hierarchy (when ``force_refresh=False``): 1. Fresh in-memory cache → return immediately. @@ -664,8 +692,21 @@ def fetch_models_dev( if now < _models_dev_retry_after: return _models_dev_cache + # Cold force_refresh (fresh CLI process): stages 1-3 were skipped, + # so the memory cache may be empty even though a servable disk + # cache + ETag sidecar exist. Hydrate first so the conditional GET + # fires (a 304 then re-confirms the disk data instead of + # re-downloading the full ~2 MB registry). + if force_refresh and not _models_dev_cache: + disk = _load_disk_cache() + if disk: + _models_dev_cache = disk + _models_dev_cache_time = 0 # servable but not fresh + try: - data, etag = _fetch_models_dev_from_network() + data, etag = _fetch_models_dev_from_network( + conditional=bool(_models_dev_cache) + ) _commit_registry(data, etag=etag, where="foreground") return data except _NotModified: diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 222c87098b..9f1357de49 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -559,6 +559,10 @@ class TestCorruptCacheRejection: assert result == {} assert not etag.exists() + # The corrupt file is quarantined (renamed), so the rejection is + # a one-time event instead of a re-parse + warning per call. + assert not cache.exists() + assert cache.with_suffix(".json.corrupt").exists() def test_conditional_get_skipped_without_servable_cache(self): """No If-None-Match header when the process holds no registry. From a3cda34137f449034ef512caedd9f9fdbf8b187f Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 14 Aug 2026 02:10:10 +0530 Subject: [PATCH 032/748] fix(models): repair the two CI slices the default-flip broke MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - web_server CONFIG_SCHEMA: fold the one-field models_dev category (models_dev.url) into the agent tab via _CATEGORY_MERGE, matching the established pattern for single-field categories (slice 7, test_no_single_field_categories). - image_routing._lookup_supports_vision: pass allow_network=True to get_model_capabilities. The vision-capability lookup runs when an image actually needs routing (not per conversation turn), and the #31179 text-only-main guard depends on catalog data — with the new allow_network=False default a cold cache returned 'unknown', which falls back to attempting the call and reintroduced the #31179 failure shape (slice 8, test_text_only_main_skipped_when_no_ aggregator). This preserves that path's historical network-on-cold-cache behavior; the fetch stays 4h-TTL cached and backoff-limited. --- agent/image_routing.py | 9 ++++++++- hermes_cli/web_server.py | 4 ++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/agent/image_routing.py b/agent/image_routing.py index a8337c9ac6..a3762fccd4 100644 --- a/agent/image_routing.py +++ b/agent/image_routing.py @@ -432,7 +432,14 @@ def _lookup_supports_vision( caps = None try: from agent.models_dev import get_model_capabilities - caps = get_model_capabilities(provider, model) + # allow_network=True on purpose: vision-capability lookup runs when + # an image actually needs routing (not per turn), and the #31179 + # text-only-main guard depends on catalog data — a cold cache + # returning "unknown" would fall back to attempting the call and + # reintroduce the bug. This preserves the historical + # network-on-cold-cache behavior for this one path; the fetch is + # cached (4h TTL) and backoff-limited after failures. + caps = get_model_capabilities(provider, model, allow_network=True) except Exception as exc: # pragma: no cover - defensive logger.debug("image_routing: caps lookup failed for %s:%s — %s", provider, model, exc) if caps is not None: diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 701c5662d6..a533fbc2a5 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -1038,6 +1038,10 @@ _CATEGORY_MERGE: Dict[str, str] = { "skills": "agent", "cron": "agent", "network": "agent", + # `models_dev.url` (mirror override) is the only schema-surfaced + # models_dev field — fold it in with the other network/agent plumbing + # rather than spawning a one-field orphan tab. + "models_dev": "agent", "checkpoints": "agent", "approvals": "security", "human_delay": "display", From d16e2366df3f52d3d849a46a94ae3f42281fa268 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 15:06:00 -0700 Subject: [PATCH 033/748] fix(desktop): don't dial per-profile sockets for profiles served by the shared global-remote primary (#85665) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Under a global SSH/remote gateway, resolveProfileBackendRoute routes every profile to the shared primary backend (case 3) and getConnection() returns the primary descriptor tagged with the profile. ensureGatewayForProfile still dialed a per-profile secondary socket at that descriptor; over SSH the duplicate dial fails (per-backend tunnel/ticket) and the closed socket became the ACTIVE gateway — every profile except the primary showed 'Hermes gateway is not connected' even though the primary socket was open. Detect the shared-primary route and activate the primary socket instead; $activeGatewayProfile still tracks the selected profile so per-request ?profile= scoping is unchanged. Hover pre-warm no-ops on this route. Local pooled profiles and per-profile remote overrides are untouched (pinned by test). --- .../src/store/gateway-shared-remote.test.ts | 78 +++++++++++++++++++ apps/desktop/src/store/gateway.ts | 39 ++++++++++ 2 files changed, 117 insertions(+) create mode 100644 apps/desktop/src/store/gateway-shared-remote.test.ts diff --git a/apps/desktop/src/store/gateway-shared-remote.test.ts b/apps/desktop/src/store/gateway-shared-remote.test.ts new file mode 100644 index 0000000000..a16fe376cd --- /dev/null +++ b/apps/desktop/src/store/gateway-shared-remote.test.ts @@ -0,0 +1,78 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// The global-remote share (backend routing case 3): every profile is served +// by the PRIMARY backend over one host, and getConnection() tags the shared +// descriptor with `profile`. Dialing a second WebSocket at that descriptor +// used to fail over SSH (per-backend tunnel/ticket) and poison the active +// gateway with a closed socket — "Hermes gateway is not connected" for every +// profile except the primary. These tests pin the fix: a profile routed to +// the shared primary activates the primary socket instead of dialing. + +vi.mock('@/hermes', () => ({ + HermesGateway: class { + connectionState = 'closed' + connect = vi.fn(async () => { + throw new Error('dialed a socket for a shared-primary profile') + }) + onEvent = vi.fn(() => () => {}) + onState = vi.fn(() => () => {}) + } +})) +vi.mock('@/store/session', () => ({ setGatewayState: vi.fn() })) +vi.mock('@/store/notify-baseline', () => ({ markNativeNotifyBaseline: vi.fn() })) + +const { $gateway, configureGatewayRegistry, ensureGatewayForProfile, setPrimaryGateway } = await import('./gateway') + +type DesktopStub = { getConnection: ReturnType } + +function installDesktop(stub: DesktopStub): void { + ;(window as unknown as { hermesDesktop: unknown }).hermesDesktop = stub +} + +function makePrimary(): { connectionState: string } { + // Only connectionState is consulted by setActive/isOpen for these paths. + return { connectionState: 'open' } +} + +beforeEach(() => { + configureGatewayRegistry({ + onEvent: vi.fn(), + primaryProfile: 'default' + } as never) +}) + +afterEach(() => { + vi.clearAllMocks() + delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop +}) + +describe('ensureGatewayForProfile under a shared global remote', () => { + it('activates the primary socket for a profile tagged onto the shared descriptor', async () => { + const primary = makePrimary() + setPrimaryGateway(primary as never, 'default') + installDesktop({ + // Shared descriptor: primary connection tagged with the profile. + getConnection: vi.fn(async () => ({ port: 4242, profile: 'venture', token: 't' })) + }) + + await ensureGatewayForProfile('venture') + + expect($gateway.get()).toBe(primary) + }) + + it('still pools a socket for profiles with their own descriptor (untagged)', async () => { + const primary = makePrimary() + setPrimaryGateway(primary as never, 'default') + installDesktop({ + // Own descriptor: no profile tag → normal pooled path (dial attempted). + getConnection: vi.fn(async () => ({ port: 5151, token: 't2' })) + }) + + await ensureGatewayForProfile('worker') + + // The pooled path dialed (our stub throws, so the socket stays closed and + // reconnect is scheduled) — the important part is it did NOT silently + // reuse the primary. + expect($gateway.get()).not.toBe(primary) + }) +}) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index 149235e42f..0a64b3da1f 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -247,6 +247,29 @@ function createSecondary(profile: string): Secondary { return entry } +// True when `profile`'s backend route resolves to the SHARED primary backend +// (global-remote case 3 in resolveProfileBackendRoute): the descriptor comes +// back as the primary connection tagged with `profile`. Own-remote-override +// and local pooled descriptors are never tagged. Dialing a second socket at +// that descriptor is wrong — over SSH the second dial fails (tunnel/token are +// per-backend) and the closed socket poisons the active gateway with +// "not connected" even though the primary is open right next to it. +async function sharedPrimaryRoute(profile: string): Promise { + const desktop = window.hermesDesktop + + if (!desktop) { + return false + } + + try { + const conn = await desktop.getConnection(profile) + + return Boolean(conn && typeof conn === 'object' && (conn as { profile?: string }).profile) + } catch { + return false + } +} + // Open `profile`'s socket WITHOUT making it active — the hover-intent pre-warm // (store/profile). Runs the same spawn + connect chain as a real switch, so by // click time ensureGatewayForProfile finds an open socket and just activates @@ -260,6 +283,11 @@ export async function openGatewayForProfile(profile: string): Promise { return } + if (await sharedPrimaryRoute(key)) { + // Served by the primary backend — there is no per-profile socket to warm. + return + } + const entry = g.secondaries.get(key) ?? createSecondary(key) entry.wantOpen = true @@ -279,6 +307,17 @@ export async function ensureGatewayForProfile(profile: string): Promise { return } + // Global-remote share (routing case 3): one remote host serves every + // profile through the PRIMARY socket, scoped per request. Activate the + // primary instead of dialing a doomed duplicate socket at the same + // descriptor — $activeGatewayProfile still moves to `key`, so request + // scoping and profile-aware surfaces behave identically. + if (await sharedPrimaryRoute(key)) { + setActive(g.primaryProfile) + + return + } + let entry = g.secondaries.get(key) if (!entry) { From 89d3e43f5e61146bff46923dd8a9fc7a6cfc9d63 Mon Sep 17 00:00:00 2001 From: ethernet Date: Thu, 13 Aug 2026 18:28:52 -0400 Subject: [PATCH 034/748] fix(nix): add registration lifecycle to pyproject.toml --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 2e774b4c47..8a7b3dd10a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -393,6 +393,7 @@ exclude-newer-package = { vercel = false, nemo-relay = false, huggingface_hub = # sealed venv is missing hermes_constants, run_agent, etc. py-modules = [ "run_agent", + "registration_lifecycle", "model_tools", "toolsets", "batch_runner", From fb1ee93a6333587aa0b4863c873b9effa8472c0b Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 16:47:12 -0500 Subject: [PATCH 035/748] fix(desktop): suggestion pills wait for a completed word and stand down on workspace homonyms Two precision guards on the draft-keyword providers, both aimed at the same annoyance: a pill firing while the trigger is still under the caret. - Completed-word guard (mcp + skill): a whole-word keyword hit only counts once at least one character follows it, so the debounce elapsing mid-thought no longer pops a pill for the word being typed. Pasted-URL host hits are exempt: pasting is deliberate and the URL routinely ends the draft. - Workspace homonym guard (skill): a skill named like the session's working directory is the project's name, not a request. Working in ~/www/hermes-agent no longer floats 'Use skill: hermes-agent' on every mention of the repo. --- .../store/suggestion-providers/mcp.test.ts | 17 ++++++ .../src/store/suggestion-providers/mcp.ts | 36 +++++++---- .../store/suggestion-providers/skill.test.ts | 60 +++++++++++++++---- .../src/store/suggestion-providers/skill.ts | 35 ++++++++++- 4 files changed, 123 insertions(+), 25 deletions(-) diff --git a/apps/desktop/src/store/suggestion-providers/mcp.test.ts b/apps/desktop/src/store/suggestion-providers/mcp.test.ts index 9de2ad08b0..4fd644c95d 100644 --- a/apps/desktop/src/store/suggestion-providers/mcp.test.ts +++ b/apps/desktop/src/store/suggestion-providers/mcp.test.ts @@ -75,4 +75,21 @@ describe('matchSuggestions', () => { { keyword: 'sentry.io', server: 'sentry' } ]) }) + + it('a keyword still under the caret does not fire yet', () => { + // The debounce elapses mid-thought; the word is only intent once + // something follows it. + expect(matchSuggestions('can you check linear', INDEX)).toEqual([]) + expect(matchSuggestions('can you check linear ', INDEX)).toEqual([{ keyword: 'linear', server: 'linear' }]) + expect(matchSuggestions('can you check linear?', INDEX)).toEqual([{ keyword: 'linear', server: 'linear' }]) + }) + + it('a pasted URL fires even as the last thing in the draft', () => { + // Pasting is a deliberate act — the completed-word guard is keyword-only. + const index = [{ hosts: ['linear.app'], keywords: ['linear'], server: 'linear' }] + + expect(matchSuggestions('look at https://linear.app/team/issue/ABC-1', index)).toEqual([ + { keyword: 'linear.app', server: 'linear' } + ]) + }) }) diff --git a/apps/desktop/src/store/suggestion-providers/mcp.ts b/apps/desktop/src/store/suggestion-providers/mcp.ts index 7048aed84f..72676c4291 100644 --- a/apps/desktop/src/store/suggestion-providers/mcp.ts +++ b/apps/desktop/src/store/suggestion-providers/mcp.ts @@ -84,11 +84,32 @@ export interface McpMatch { const MAX_MATCHES = 2 +// Whole-word (unicode-aware) keyword hit that the user has FINISHED typing: +// at least one character must follow the match (the lookahead already +// guarantees it's a boundary). A hit still under the caret — "figma" as the +// last thing typed, debounce elapsed mid-thought — is not intent yet, it's +// eavesdropping on a word in progress; the pill waits for the space/period. +const keywordHit = (haystack: string, candidate: string): boolean => { + const pattern = new RegExp( + `(? - new RegExp( - `(? keywordHit(haystack, candidate)) if (keyword) { matches.push({ keyword, server: entry.server }) diff --git a/apps/desktop/src/store/suggestion-providers/skill.test.ts b/apps/desktop/src/store/suggestion-providers/skill.test.ts index 67eef945d3..69046e1632 100644 --- a/apps/desktop/src/store/suggestion-providers/skill.test.ts +++ b/apps/desktop/src/store/suggestion-providers/skill.test.ts @@ -1,28 +1,64 @@ import { describe, expect, it } from 'vitest' -import { skillPattern } from './skill' +import { collidesWithWorkspace, skillHit, skillPattern } from './skill' -describe('skillPattern', () => { - it('matches the exact name as a whole word', () => { - expect(skillPattern('perf').test('run the perf loop')).toBe(true) - expect(skillPattern('perf').test('performance is bad')).toBe(false) +// skillHit is the provider's real predicate: a whole-word match that the user +// has finished typing (at least one character follows it). +const hits = (name: string, draft: string) => skillHit(skillPattern(name), draft.toLowerCase()) + +describe('skillPattern + skillHit', () => { + it('matches the exact name as a completed whole word', () => { + expect(hits('perf', 'run the perf loop')).toBe(true) + expect(hits('perf', 'performance is bad')).toBe(false) }) it('hyphenated names also match spaced phrasing', () => { - expect(skillPattern('pr-ready').test('make this pr ready')).toBe(true) - expect(skillPattern('pr-ready').test('run pr-ready on it')).toBe(true) - expect(skillPattern('pr_ready').test('pr ready please')).toBe(true) + expect(hits('pr-ready', 'make this pr ready pls')).toBe(true) + expect(hits('pr-ready', 'run pr-ready on it')).toBe(true) + expect(hits('pr_ready', 'pr ready please')).toBe(true) }) it('never matches inside other words', () => { - expect(skillPattern('read').test('i already did')).toBe(false) - expect(skillPattern('work').test('reworked the layout')).toBe(false) + expect(hits('read', 'i already did')).toBe(false) + expect(hits('work', 'reworked the layout')).toBe(false) // Suffix boundary includes hyphen: "clean" must not fire inside "clean-up" // (a different skill may own that name). - expect(skillPattern('clean').test('do a clean-up pass')).toBe(false) + expect(hits('clean', 'do a clean-up pass')).toBe(false) }) it('is case-insensitive via lowercased input', () => { - expect(skillPattern('Clean').test('please clean this diff')).toBe(true) + expect(hits('Clean', 'please clean this diff')).toBe(true) + }) + + it('a name still under the caret is not a hit yet', () => { + // The debounce fires while the word is the last thing typed — wait for + // the next keystroke to call it intent. + expect(hits('perf', 'run perf')).toBe(false) + expect(hits('perf', 'run perf ')).toBe(true) + expect(hits('perf', 'run perf.')).toBe(true) + }) + + it('an earlier completed occurrence still counts', () => { + expect(hits('perf', 'perf first, then more perf')).toBe(true) + }) +}) + +describe('collidesWithWorkspace', () => { + it('suppresses a skill named exactly like the cwd folder', () => { + expect(collidesWithWorkspace('hermes-agent', '/Users/b/www/hermes-agent')).toBe(true) + }) + + it('suppresses inside worktree-suffixed folders too', () => { + expect(collidesWithWorkspace('hermes-agent', '/Users/b/www/hermes-agent-suggest')).toBe(true) + }) + + it('does not suppress on substring-only overlap', () => { + // "perf" inside "perfect-app" is not a homonym of the project. + expect(collidesWithWorkspace('perf', '/Users/b/www/perfect-app')).toBe(false) + expect(collidesWithWorkspace('clean', '/Users/b/www/hermes-agent')).toBe(false) + }) + + it('never collides when detached (empty cwd)', () => { + expect(collidesWithWorkspace('hermes-agent', '')).toBe(false) }) }) diff --git a/apps/desktop/src/store/suggestion-providers/skill.ts b/apps/desktop/src/store/suggestion-providers/skill.ts index 94b84f8885..963aaca3ea 100644 --- a/apps/desktop/src/store/suggestion-providers/skill.ts +++ b/apps/desktop/src/store/suggestion-providers/skill.ts @@ -2,6 +2,7 @@ import { requestComposerFocus, requestComposerInsert } from '@/app/chat/composer import { getSkills } from '@/hermes' import { translateNow } from '@/i18n' import { type ComposerSuggestion, registerDraftProvider } from '@/store/composer-suggestions' +import { $currentCwd } from '@/store/session' /** * Skill-match draft provider: the draft names a skill the user has, so offer @@ -41,7 +42,34 @@ const escape = (value: string) => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') export function skillPattern(name: string): RegExp { const flexible = name.toLowerCase().split(/[-_]/).map(escape).join('[-_ ]') - return new RegExp(`(? { @@ -94,7 +122,10 @@ registerDraftProvider('skill', async ({ text }) => { } const haystack = text.toLowerCase() + const cwd = $currentCwd.get() const skills = await loadIndex() - return skills.filter(skill => skill.pattern.test(haystack)).map(skill => toSuggestion(skill.name)) + return skills + .filter(skill => skillHit(skill.pattern, haystack) && !collidesWithWorkspace(skill.name, cwd)) + .map(skill => toSuggestion(skill.name)) }) From 8c8d55bd07575604a76f6df59bfbb42ceb6a71e6 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 16:47:12 -0500 Subject: [PATCH 036/748] feat(desktop): grow the MCP suggestion directory to 18 official hosted remotes Vercel, Supabase, Netlify, Hugging Face, Asana, Intercom, Airtable, Webflow, PayPal, and Square join the directory. Every entry is a vendor-operated remote with its docs page linked, same URL-only rule as the founding eight. Trigger notes where words are ambiguous: 'square' the English word never fires (squareup only), and vercel.app/netlify.app deploy-preview hosts are deliberately absent (a pasted preview link is about the site, not the platform). Brand glyphs wired for all newcomers. --- apps/desktop/src/lib/mcp-brands.tsx | 13 ++++ apps/desktop/src/lib/mcp-directory.ts | 87 +++++++++++++++++++++++++++ 2 files changed, 100 insertions(+) diff --git a/apps/desktop/src/lib/mcp-brands.tsx b/apps/desktop/src/lib/mcp-brands.tsx index c587bb550f..10a8e6136b 100644 --- a/apps/desktop/src/lib/mcp-brands.tsx +++ b/apps/desktop/src/lib/mcp-brands.tsx @@ -7,12 +7,17 @@ * a favicon service for it would leak that hostname off-box. */ import { + SiAirtable, + SiAsana, SiAtlassian, SiDatadog, SiFigma, SiGithub, SiGitlab, + SiHuggingface, + SiIntercom, SiLinear, + SiNetlify, SiNotion, SiPaypal, SiPostgresql, @@ -21,6 +26,7 @@ import { SiStripe, SiSupabase, SiVercel, + SiWebflow, SiZapier } from '@icons-pack/react-simple-icons' import type { ComponentType, SVGProps } from 'react' @@ -35,12 +41,18 @@ export interface McpBrand { } export const MCP_BRAND_ICONS: Record = { + airtable: { Icon: SiAirtable, color: '#18BFFF' }, + asana: { Icon: SiAsana, color: '#F06A6A' }, atlassian: { Icon: SiAtlassian, color: '#0052CC' }, datadog: { Icon: SiDatadog, color: '#632CA6' }, figma: { Icon: SiFigma, color: '#F24E1E' }, github: { Icon: SiGithub, color: '#181717', monochrome: true }, gitlab: { Icon: SiGitlab, color: '#FC6D26' }, + hugging_face: { Icon: SiHuggingface, color: '#FFD21E' }, + huggingface: { Icon: SiHuggingface, color: '#FFD21E' }, + intercom: { Icon: SiIntercom, color: '#6AFDEF' }, linear: { Icon: SiLinear, color: '#5E6AD2' }, + netlify: { Icon: SiNetlify, color: '#00C7B7' }, notion: { Icon: SiNotion, color: '#000000', monochrome: true }, paypal: { Icon: SiPaypal, color: '#003087' }, postgres: { Icon: SiPostgresql, color: '#4169E1' }, @@ -50,6 +62,7 @@ export const MCP_BRAND_ICONS: Record = { stripe: { Icon: SiStripe, color: '#635BFF' }, supabase: { Icon: SiSupabase, color: '#3FCF8E' }, vercel: { Icon: SiVercel, color: '#000000', monochrome: true }, + webflow: { Icon: SiWebflow, color: '#146EF5' }, zapier: { Icon: SiZapier, color: '#FF4A00' } } diff --git a/apps/desktop/src/lib/mcp-directory.ts b/apps/desktop/src/lib/mcp-directory.ts index 107fb14479..902cc18348 100644 --- a/apps/desktop/src/lib/mcp-directory.ts +++ b/apps/desktop/src/lib/mcp-directory.ts @@ -99,6 +99,93 @@ export const MCP_DIRECTORY: McpDirectoryEntry[] = [ keywords: ['stripe'], name: 'stripe', url: 'https://mcp.stripe.com' + }, + { + description: 'Deployments, logs, and projects via Vercel’s hosted MCP.', + docs: 'https://vercel.com/docs/mcp', + // No `vercel.app` on purpose (same rule as GitHub): pasted deploy-preview + // links are about the site being previewed, not about managing Vercel. + hosts: ['vercel.com'], + keywords: ['vercel'], + name: 'vercel', + url: 'https://mcp.vercel.com' + }, + { + description: 'Database, auth, and storage from your Supabase projects.', + docs: 'https://supabase.com/docs/guides/ai-tools/mcp', + hosts: ['supabase.com', 'supabase.co'], + keywords: ['supabase'], + name: 'supabase', + url: 'https://mcp.supabase.com/mcp' + }, + { + description: 'Sites, deploys, and env vars via Netlify’s hosted MCP.', + docs: 'https://docs.netlify.com/build/build-with-ai/agent-setup-guides/agent-setup-overview/', + // No `netlify.app` for the same deploy-preview reason as vercel.app. + hosts: ['netlify.com'], + keywords: ['netlify'], + name: 'netlify', + url: 'https://netlify-mcp.netlify.app/mcp' + }, + { + description: 'Models, datasets, Spaces, and papers from the Hugging Face Hub.', + docs: 'https://huggingface.co/docs/hub/agents-mcp', + hosts: ['huggingface.co', 'hf.co'], + keywords: ['hugging face', 'huggingface'], + // Underscored so prettyName renders "Hugging Face", not "Huggingface". + name: 'hugging_face', + url: 'https://huggingface.co/mcp' + }, + { + description: 'Tasks, projects, and goals from your Asana workspace.', + docs: 'https://developers.asana.com/docs/using-asanas-mcp-server', + hosts: ['asana.com'], + keywords: ['asana'], + name: 'asana', + url: 'https://mcp.asana.com/sse' + }, + { + description: 'Conversations, tickets, and customer data from Intercom.', + docs: 'https://developers.intercom.com/docs/guides/mcp', + hosts: ['intercom.com', 'intercom.io'], + keywords: ['intercom'], + name: 'intercom', + url: 'https://mcp.intercom.com/mcp' + }, + { + description: 'Bases, tables, and records from your Airtable workspace.', + docs: 'https://support.airtable.com/articles/9897799762-using-the-airtable-mcp-server', + hosts: ['airtable.com'], + keywords: ['airtable'], + name: 'airtable', + url: 'https://mcp.airtable.com/mcp' + }, + { + description: 'Sites, CMS collections, and pages via Webflow’s hosted MCP.', + docs: 'https://developers.webflow.com/mcp/reference/getting-started', + // No `webflow.io` — that's published staging sites, not Webflow intent. + hosts: ['webflow.com'], + keywords: ['webflow'], + name: 'webflow', + url: 'https://mcp.webflow.com/mcp' + }, + { + description: 'Payments, invoices, and subscriptions via PayPal’s hosted MCP.', + docs: 'https://developer.paypal.com/tools/mcp-server/', + hosts: ['developer.paypal.com'], + keywords: ['paypal'], + name: 'paypal', + url: 'https://mcp.paypal.com/sse' + }, + { + description: 'Catalog, orders, and payments via Square’s hosted MCP.', + docs: 'https://developer.squareup.com/docs/mcp', + hosts: ['squareup.com'], + // "square" the English word is everywhere ("square brackets", "square + // corners") — only the unambiguous brand form triggers. + keywords: ['squareup'], + name: 'square', + url: 'https://mcp.squareup.com/sse' } ] From 0280cf09c4628efa3b45d5215b0d7f2d623b42e2 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 17:25:07 -0500 Subject: [PATCH 037/748] fix(desktop): suggestion pills paint the current offer, not the first one seen MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The bus's change gate compared offers by `provider:id` alone. Providers rebuild their suggestion objects on every draft sample, so that key is equal constantly and the write bailed out — pinning the FIRST object for the life of the offer. Two consequences, both user-visible. The pill keeps painting a stale reason ("you mentioned linear" after the user pasted a linear.app link), and it keeps calling a stale `invoke` closure — work built for a draft that no longer exists. Compare the fields the pill actually renders instead. The reference-identity bail-out survives for the common case (same draft, same match, no re-render), which is what the gate was there for. --- .../src/store/composer-suggestions.test.ts | 43 +++++++++++++++++++ .../desktop/src/store/composer-suggestions.ts | 25 ++++++++++- 2 files changed, 67 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/store/composer-suggestions.test.ts b/apps/desktop/src/store/composer-suggestions.test.ts index 2dc3d6337b..e76c56a2f4 100644 --- a/apps/desktop/src/store/composer-suggestions.test.ts +++ b/apps/desktop/src/store/composer-suggestions.test.ts @@ -80,4 +80,47 @@ describe('composer suggestion bus', () => { offerSuggestions('s6', 'test', []) }) + + it('replaces an offer whose rendered copy changed under the same key', () => { + offerSuggestions('s7', 'test', [{ ...suggestion('linear'), tip: 'because you mentioned “linear”' }]) + offerSuggestions('s7', 'test', [{ ...suggestion('linear'), tip: 'because you pasted linear.app' }]) + + // Same key, new trigger — the strip must paint the new reason, not the + // first one it ever saw. + expect(($composerSuggestionsBySession.get().s7 ?? []).map(s => s.tip)).toEqual(['because you pasted linear.app']) + + offerSuggestions('s7', 'test', []) + }) + + it('re-offering the same key swaps in the fresh invoke closure', async () => { + const calls: string[] = [] + + const offer = (tag: string) => + offerSuggestions('s8', 'test', [ + { ...suggestion('linear'), invoke: async () => void calls.push(tag), label: `Add linear ${tag}` } + ]) + + offer('first') + offer('second') + + await ($composerSuggestionsBySession.get().s8 ?? [])[0]!.invoke({ cancelled: () => false, sessionId: 's8' }) + + // A pinned first object means the pill runs work built for a draft the + // user has since changed. + expect(calls).toEqual(['second']) + + offerSuggestions('s8', 'test', []) + }) + + it('keeps the array reference when nothing the pill paints changed', () => { + offerSuggestions('s9', 'test', [suggestion('linear')]) + + const first = $composerSuggestionsBySession.get().s9 + + offerSuggestions('s9', 'test', [suggestion('linear')]) + + expect($composerSuggestionsBySession.get().s9).toBe(first) + + offerSuggestions('s9', 'test', []) + }) }) diff --git a/apps/desktop/src/store/composer-suggestions.ts b/apps/desktop/src/store/composer-suggestions.ts index d0724ddd88..a241c3395a 100644 --- a/apps/desktop/src/store/composer-suggestions.ts +++ b/apps/desktop/src/store/composer-suggestions.ts @@ -67,8 +67,31 @@ export const $composerSuggestionsBySession = atom sessionId ?? '' +// Everything the pill actually paints. Compared field-by-field rather than by +// key alone: a provider rebuilds its suggestion objects on every sample, so +// key equality is true constantly, and treating that as "no change" pins the +// FIRST object forever — the strip then paints a stale tip and, worse, calls a +// stale `invoke` closure. Comparing the rendered copy keeps the cheap bail-out +// for the common case (same draft, same match) while letting a genuinely +// changed offer through. +const RENDERED: readonly (keyof ComposerSuggestion)[] = [ + 'brand', + 'doneLabel', + 'doneTip', + 'icon', + 'label', + 'tip', + 'workingLabel', + 'workingTip' +] + const sameSuggestions = (a: readonly ComposerSuggestion[], b: readonly ComposerSuggestion[]) => - a.length === b.length && a.every((x, i) => suggestionKey(x) === suggestionKey(b[i]!)) + a.length === b.length && + a.every((x, i) => { + const y = b[i]! + + return suggestionKey(x) === suggestionKey(y) && RENDERED.every(field => x[field] === y[field]) + }) function write(sessionId: string | null | undefined, suggestions: ComposerSuggestion[]): void { const key = keyFor(sessionId) From 685a5c95ad8953c6d9c2f4514297d2e171305798 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 17:25:19 -0500 Subject: [PATCH 038/748] fix(desktop): a withdrawn suggestion pill drops its phase and cancels its work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase lived in a `Record` that only ever grew, keyed by `provider:id` — keys that repeat constantly, since a provider withdraws and re-offers the same suggestion whenever the draft loses and regains its trigger. A leftover `done` then painted a genuine new offer as "Added GitHub" and swallowed clicks, because only `idle` invokes. The same map outlived a session switch. One composer stays mounted across it, so connecting GitHub in one chat left the next chat's real offer inert. Withdrawal also stranded in-flight work. The pill is the only cancel affordance — clicking a working pill sets the flag the provider polls — so once it left the strip an OAuth flow could poll forever, hold the server's in-progress slot against a retry, and resolve into a config write with no UI left to narrate or roll it back. Phase now lives and dies with the pill: withdrawn keys drop their phase and flip their cancel flag, unmount cancels everything in flight, and the strip remounts per session. --- .../chat/composer/suggestion-pills.test.tsx | 183 ++++++++++++++++++ .../app/chat/composer/suggestion-pills.tsx | 61 +++++- 2 files changed, 243 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/chat/composer/suggestion-pills.test.tsx diff --git a/apps/desktop/src/app/chat/composer/suggestion-pills.test.tsx b/apps/desktop/src/app/chat/composer/suggestion-pills.test.tsx new file mode 100644 index 0000000000..e8e46b32ba --- /dev/null +++ b/apps/desktop/src/app/chat/composer/suggestion-pills.test.tsx @@ -0,0 +1,183 @@ +import { act, render } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { $composerSuggestionsBySession, type ComposerSuggestion, offerSuggestions } from '@/store/composer-suggestions' + +import { SuggestionPills } from './suggestion-pills' + +/** + * The pill strip owns phase; the bus owns existence. These cover the seam + * between them — what has to happen to a pill's phase and its in-flight work + * when the bus withdraws it, which is the normal way a suggestion ends. + */ + +interface Flow { + resolve: () => void + reject: (error: unknown) => void + /** Latest value of the invoke context's `cancelled()` poll. */ + cancelled: () => boolean +} + +function pill(id: string, provider = 'mcp'): { suggestion: ComposerSuggestion; flow: Flow } { + const flow: Partial = {} + + const suggestion: ComposerSuggestion = { + doneLabel: `Added ${id}`, + doneTip: 'done', + id, + invoke: context => + new Promise((resolve, reject) => { + flow.resolve = resolve + flow.reject = reject + flow.cancelled = context.cancelled + }), + label: `Add ${id}`, + provider, + tip: 'because', + workingLabel: `Connecting ${id}…`, + workingTip: 'cancel' + } + + return { flow: flow as Flow, suggestion } +} + +const labels = (container: HTMLElement) => [...container.querySelectorAll('button')].map(button => button.textContent) + +const click = async (container: HTMLElement) => { + await act(async () => { + container.querySelector('button')!.click() + }) +} + +afterEach(() => { + $composerSuggestionsBySession.set({}) +}) + +describe('SuggestionPills', () => { + it('narrates idle → working → done through one invoke', async () => { + const { flow, suggestion } = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + expect(labels(container)).toEqual(['Add github']) + + await click(container) + expect(labels(container)).toEqual(['Connecting github…']) + + await act(async () => flow.resolve()) + expect(labels(container)).toEqual(['Added github']) + }) + + it('re-offering a withdrawn key starts from idle, not a stale done', async () => { + const first = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [first.suggestion])) + await click(container) + await act(async () => first.flow.resolve()) + expect(labels(container)).toEqual(['Added github']) + + // The provider withdraws (draft lost the trigger) and offers again later. + act(() => offerSuggestions('s1', 'mcp', [])) + act(() => offerSuggestions('s1', 'mcp', [pill('github').suggestion])) + + expect(labels(container)).toEqual(['Add github']) + }) + + it('a fresh offer is clickable again after an earlier one completed', async () => { + const first = pill('github') + const second = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [first.suggestion])) + await click(container) + await act(async () => first.flow.resolve()) + + act(() => offerSuggestions('s1', 'mcp', [])) + act(() => offerSuggestions('s1', 'mcp', [second.suggestion])) + await click(container) + + // A stale `done` swallows the click (only idle invokes) — this is the + // user-visible half of the phase leak. + expect(labels(container)).toEqual(['Connecting github…']) + }) + + it('cancels the in-flight action when the pill is withdrawn mid-flight', async () => { + const { flow, suggestion } = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + await click(container) + expect(flow.cancelled()).toBe(false) + + // The pill is the only cancel affordance; losing it must abort the work. + act(() => offerSuggestions('s1', 'mcp', [])) + + expect(labels(container)).toEqual([]) + expect(flow.cancelled()).toBe(true) + }) + + it('cancels in-flight work when the strip unmounts', async () => { + const { flow, suggestion } = pill('github') + const { container, unmount } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + await click(container) + + unmount() + + expect(flow.cancelled()).toBe(true) + }) + + it('does not carry phase across a session switch on one mounted composer', async () => { + const { flow, suggestion } = pill('github') + const { container, rerender } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + await click(container) + await act(async () => flow.resolve()) + expect(labels(container)).toEqual(['Added github']) + + act(() => offerSuggestions('s2', 'mcp', [pill('github').suggestion])) + rerender() + + expect(labels(container)).toEqual(['Add github']) + }) + + it('returns to idle when the provider rejects, so it can be retried', async () => { + const { flow, suggestion } = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + await click(container) + + await act(async () => flow.reject(new Error('nope'))) + + expect(labels(container)).toEqual(['Add github']) + }) + + it('a second click on a working pill requests cancel', async () => { + const { flow, suggestion } = pill('github') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [suggestion])) + await click(container) + await click(container) + + expect(flow.cancelled()).toBe(true) + expect(labels(container)).toEqual(['Connecting github…']) + }) + + it('invokes each pill independently when two are offered', async () => { + const a = pill('github') + const b = pill('linear') + const invoke = vi.spyOn(b.suggestion, 'invoke') + const { container } = render() + + act(() => offerSuggestions('s1', 'mcp', [a.suggestion, b.suggestion])) + await click(container) + + expect(labels(container)).toEqual(['Connecting github…', 'Add linear']) + expect(invoke).not.toHaveBeenCalled() + }) +}) diff --git a/apps/desktop/src/app/chat/composer/suggestion-pills.tsx b/apps/desktop/src/app/chat/composer/suggestion-pills.tsx index ce926433c1..e4562ae057 100644 --- a/apps/desktop/src/app/chat/composer/suggestion-pills.tsx +++ b/apps/desktop/src/app/chat/composer/suggestion-pills.tsx @@ -1,4 +1,4 @@ -import { useState } from 'react' +import { useEffect, useState } from 'react' import { composerFloatingPill } from '@/components/chat/composer-dock' import { Codicon } from '@/components/ui/codicon' @@ -32,7 +32,19 @@ import { $composerSuggestionsBySession, markSuggestionInvoked, suggestionKey } f type PillPhase = 'done' | 'idle' | 'working' +/** + * Phase belongs to the pill the user is looking at, so it lives and dies with + * that pill — the bus owns whether a suggestion exists, this owns only how far + * along it is. Remounting per session enforces the half of that the strip + * can't see: one composer stays mounted across a session switch, and phase + * keys are `provider:id`, so without this "Added GitHub" in one chat renders + * the next chat's genuine offer as already-done and inert. + */ export function SuggestionPills({ sessionId }: { sessionId: null | string }) { + return +} + +function SessionSuggestionPills({ sessionId }: { sessionId: null | string }) { const suggestions = useSessionSlice($composerSuggestionsBySession, sessionId) const [phases, setPhases] = useState>({}) // Cancel flags outlive renders but never trigger them (poll-boundary abort). @@ -40,6 +52,51 @@ export function SuggestionPills({ sessionId }: { sessionId: null | string }) { const setPhase = (key: string, phase: PillPhase) => setPhases(current => ({ ...current, [key]: phase })) + // A pill that leaves the strip takes its phase with it, and cancels whatever + // it had in flight. Both halves matter: + // + // - Phase, because keys repeat. A provider withdraws and re-offers the same + // `provider:id` constantly (the draft loses and regains the trigger word, + // a connect fails and the word is still typed), and a `done` left on file + // paints the fresh offer as "Added GitHub" — a pill that says it already + // worked and ignores clicks, since only `idle` invokes. + // - Cancellation, because a withdrawn pill is the user's ONLY cancel + // affordance. Clicking a working pill sets this flag; once the pill is + // gone nothing can, so an OAuth flow the user walked away from would poll + // forever, hold the server's in-progress slot against a retry, and either + // land a server they never confirmed or roll one back with no UI to say + // so. Withdrawal aborts it at the next poll boundary and lets the + // provider roll back. + // + // `suggestions` is reference-stable while the offered set is unchanged (the + // bus preserves identity on no-op writes), so this runs on real churn only. + useEffect(() => { + const live = new Set(suggestions.map(suggestionKey)) + + for (const key of cancels.keys()) { + if (!live.has(key)) { + cancels.set(key, true) + } + } + + setPhases(current => { + const kept = Object.entries(current).filter(([key]) => live.has(key)) + + return kept.length === Object.keys(current).length ? current : Object.fromEntries(kept) + }) + }, [cancels, suggestions]) + + // Unmount withdraws everything at once (composer closing, session switch + // remounting this subtree) — same rule, same reason. + useEffect( + () => () => { + for (const key of cancels.keys()) { + cancels.set(key, true) + } + }, + [cancels] + ) + return suggestions.map(suggestion => { const key = suggestionKey(suggestion) const brand = suggestion.brand ? brandFor(suggestion.brand) : null @@ -66,6 +123,8 @@ export function SuggestionPills({ sessionId }: { sessionId: null | string }) { // Provider owns error surfacing (and swallows its own cancels); // the pill just returns to idle so it can be tried again. setPhase(key, 'idle') + } finally { + cancels.delete(key) } } From 423f92e607dd51908d23b04758bc0fcd6ec5ff39 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 17:25:26 -0500 Subject: [PATCH 039/748] fix(desktop): connect pills reload tools into the session that clicked them Both connect providers captured `sessionId` when the suggestion was built and ignored the one the pill hands `invoke`. An offer that outlived a session switch therefore aimed its `reload.mcp` at the session the draft was sampled in, so the chat the user actually clicked from resumed without the tools the pill just said were ready. Prefer the invoking pill's session; the captured one stays as the fallback. --- apps/desktop/src/store/suggestion-providers/mcp.ts | 4 +++- apps/desktop/src/store/suggestion-providers/repair.ts | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/store/suggestion-providers/mcp.ts b/apps/desktop/src/store/suggestion-providers/mcp.ts index 72676c4291..81e7bf5617 100644 --- a/apps/desktop/src/store/suggestion-providers/mcp.ts +++ b/apps/desktop/src/store/suggestion-providers/mcp.ts @@ -186,7 +186,9 @@ function toSuggestion(match: McpMatch, sessionId: string | null): ComposerSugges doneLabel: copy('added', name), doneTip: copy('addedTip'), id: match.server, - invoke: context => connect(match.server, sessionId, context.cancelled), + // The pill's session wins over the one captured at sample time: the reload + // has to reach the session the user is actually looking at. + invoke: context => connect(match.server, context.sessionId ?? sessionId, context.cancelled), label: copy('label', name), provider: 'mcp', tip: copy('tip', match.keyword), diff --git a/apps/desktop/src/store/suggestion-providers/repair.ts b/apps/desktop/src/store/suggestion-providers/repair.ts index f0ee2265c0..159348d486 100644 --- a/apps/desktop/src/store/suggestion-providers/repair.ts +++ b/apps/desktop/src/store/suggestion-providers/repair.ts @@ -73,7 +73,7 @@ function toSuggestion(server: string, sessionId: string | null): ComposerSuggest doneTip: copy('doneTip'), icon: 'plug', id: server, - invoke: context => reconnect(server, sessionId, context.cancelled), + invoke: context => reconnect(server, context.sessionId ?? sessionId, context.cancelled), label: copy('label', name), provider: 'repair', tip: copy('tip', name), From e50db44b12dc52027d9ddcc235ae224dce966cc9 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 13 Aug 2026 20:24:43 -0500 Subject: [PATCH 040/748] docs(hyperframes): document preview teardown to stop leaked Chrome workers npx hyperframes preview starts a long-lived next-server that keeps chrome-headless-shell render workers resident. On GPU-less hosts (WSL, containers, CI) each idle worker falls back to software WebGL (swiftshader) and busy-spins a CPU core; a preview left open stacks these up until the host is wedged. The skill never said preview was long-lived, so leaking was the default outcome. Add a Cleanup section + pitfall to SKILL.md and a Runaway CPU troubleshooting entry with diagnose/fix/avoid steps. --- optional-skills/creative/hyperframes/SKILL.md | 19 ++++++++++++++- .../hyperframes/references/troubleshooting.md | 23 +++++++++++++++++++ 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/optional-skills/creative/hyperframes/SKILL.md b/optional-skills/creative/hyperframes/SKILL.md index 16a5c8d1ba..b05c6cde91 100644 --- a/optional-skills/creative/hyperframes/SKILL.md +++ b/optional-skills/creative/hyperframes/SKILL.md @@ -43,11 +43,13 @@ Do **not** use this skill for: npx hyperframes init my-video # scaffold a project cd my-video npx hyperframes lint # validate before preview/render -npx hyperframes preview # live-reload browser preview (port 3002) +npx hyperframes preview # live-reload preview (long-lived server, port 3002) npx hyperframes render --output final.mp4 # render to MP4 npx hyperframes doctor # diagnose environment issues ``` +`preview` is a **long-lived** Next.js server that holds Chrome render workers open. Always stop it when done (see [Cleanup](#cleanup)) — a forgotten preview keeps idle `chrome-headless-shell` workers alive that, on GPU-less hosts (WSL, containers, CI), spin a CPU core each indefinitely via software WebGL (swiftshader). + Render flags: `--quality draft|standard|high` · `--fps 24|30|60` · `--format mp4|webm` · `--docker` (reproducible) · `--strict`. Full CLI reference: [references/cli.md](references/cli.md). @@ -150,8 +152,23 @@ npx hyperframes render --quality high --output final.mp4 # final delivery Use the 7-step capture-to-video workflow in [references/website-to-video.md](references/website-to-video.md): capture → DESIGN.md → SCRIPT.md → storyboard → composition → render → deliver. +## Cleanup + +`render` is one-shot (workers exit when it finishes). `preview` is **not** — it runs a background Next.js server that keeps Chrome workers resident until you stop it. Never leave one running: on GPU-less hosts each idle worker's swiftshader process pegs a CPU core, and a preview left open for days stacks up multiple. + +Stop a preview when the user is done reviewing (or before starting a new one): + +```bash +pkill -f "hyperframes.*preview" # the Studio server (frees port 3002) +pkill -f chrome-headless-shell # its render workers; only safe if nothing else uses them +``` + +If unsure whether other tools use `chrome-headless-shell`, check first: `pgrep -af chrome-headless-shell`. Recover a wedged host (many idle workers spinning CPU) the same way — see [references/troubleshooting.md](references/troubleshooting.md#runaway-cpu-from-leftover-preview-workers). + ## Pitfalls +- **Leaving `preview` running** — it's a long-lived server holding Chrome workers; on WSL/containers/CI those idle workers spin a CPU core each (software WebGL). Stop it when done — see [Cleanup](#cleanup). + - **`HeadlessExperimental.beginFrame' wasn't found`** — Chromium 147+ removed this protocol. Ensure you're on `hyperframes@>=0.4.2` (auto-detects and falls back to screenshot mode). Escape hatch: `export PRODUCER_FORCE_SCREENSHOT=true`. See [hyperframes#294](https://github.com/heygen-com/hyperframes/issues/294) and [references/troubleshooting.md](references/troubleshooting.md). - **System Chrome (not `chrome-headless-shell`)** — renders hang for 120s then timeout. Run `npx puppeteer browsers install chrome-headless-shell` (setup.sh does this). `hyperframes doctor` reports which binary will be used. - **`repeat: -1` anywhere** — breaks the capture engine. Always compute a finite repeat count. diff --git a/optional-skills/creative/hyperframes/references/troubleshooting.md b/optional-skills/creative/hyperframes/references/troubleshooting.md index 8f561310d8..9a50d1dd0e 100644 --- a/optional-skills/creative/hyperframes/references/troubleshooting.md +++ b/optional-skills/creative/hyperframes/references/troubleshooting.md @@ -132,6 +132,29 @@ npx hyperframes render --docker --docker-args "--cap-add=SYS_ADMIN" The headless browser needs namespace permissions for sandboxing. +## Runaway CPU from leftover preview workers + +**Symptom:** load average climbs and stays high; `top` shows several `chrome-headless-shell --type=gpu-process` at ~300%+ CPU each, alive for hours/days, even when you're not rendering. + +**Cause:** `npx hyperframes preview` is a long-lived server that keeps Chrome render workers resident. On hosts with no real GPU (WSL, containers, most CI), each idle worker falls back to software WebGL (`swiftshader`) whose GPU process busy-spins a CPU core. A preview left open — or several started over time — stacks these up. + +**Diagnose:** + +```bash +pgrep -af chrome-headless-shell # list the workers + their parent flags +pgrep -af "hyperframes.*preview" # the preview server(s) holding them open +uptime # confirm elevated load average +``` + +**Fix:** stop the preview server and its workers (see [SKILL.md#cleanup](../SKILL.md)): + +```bash +pkill -f "hyperframes.*preview" +pkill -f chrome-headless-shell # only if no other tool uses it — check pgrep first +``` + +**Avoid:** never leave a `preview` running after review; use `render` (one-shot, self-cleaning) for output. Keep `--workers` low on shared/GPU-less hosts. + ## Bug reports Include `npx hyperframes info` output + the full error log. File at [github.com/heygen-com/hyperframes](https://github.com/heygen-com/hyperframes/issues). From 2707183fed47ff920de05ac0121147180ad46696 Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Thu, 13 Aug 2026 16:10:13 -0600 Subject: [PATCH 041/748] fix(desktop): stop offering unsupported GitHub MCP OAuth --- apps/desktop/src/lib/mcp-directory.ts | 13 ++++--------- .../src/store/suggestion-providers/mcp.test.ts | 8 ++++++++ 2 files changed, 12 insertions(+), 9 deletions(-) diff --git a/apps/desktop/src/lib/mcp-directory.ts b/apps/desktop/src/lib/mcp-directory.ts index 902cc18348..a33b768a0f 100644 --- a/apps/desktop/src/lib/mcp-directory.ts +++ b/apps/desktop/src/lib/mcp-directory.ts @@ -75,15 +75,10 @@ export const MCP_DIRECTORY: McpDirectoryEntry[] = [ name: 'datadog', url: 'https://mcp.datadoghq.com/api/unstable/mcp-server/mcp' }, - { - description: 'Repos, issues, and pull requests via GitHub’s hosted MCP.', - docs: 'https://docs.github.com/en/copilot/customizing-copilot/using-model-context-protocol/using-the-github-mcp-server', - // No hosts on purpose: github.com links are everywhere in a coding chat - // (commits, PRs under review, pasted diffs) and would fire constantly. - keywords: ['github'], - name: 'github', - url: 'https://api.githubcopilot.com/mcp/' - }, + // GitHub's hosted MCP is intentionally absent. Directory entries are wired + // through generic Dynamic Client Registration, but GitHub requires each MCP + // host to provide its own OAuth app (or use a PAT). Advertising it here makes + // both the composer pill and setup_mcp fail at /register with HTTP 404. { description: 'Pages and databases from your Notion workspace.', docs: 'https://developers.notion.com/docs/mcp', diff --git a/apps/desktop/src/store/suggestion-providers/mcp.test.ts b/apps/desktop/src/store/suggestion-providers/mcp.test.ts index 4fd644c95d..f83a5b2c4b 100644 --- a/apps/desktop/src/store/suggestion-providers/mcp.test.ts +++ b/apps/desktop/src/store/suggestion-providers/mcp.test.ts @@ -1,5 +1,7 @@ import { describe, expect, it } from 'vitest' +import { MCP_DIRECTORY } from '@/lib/mcp-directory' + import { matchSuggestions } from './mcp' const INDEX = [ @@ -92,4 +94,10 @@ describe('matchSuggestions', () => { { keyword: 'linear.app', server: 'linear' } ]) }) + + it('does not offer GitHub through the generic OAuth registration path', () => { + const index = MCP_DIRECTORY.map(entry => ({ hosts: entry.hosts, keywords: entry.keywords, server: entry.name })) + + expect(matchSuggestions('connect github', index)).toEqual([]) + }) }) From 486f4ace20ba75401681c0915ab79a0968fa6bb1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 13 Aug 2026 19:01:11 -0700 Subject: [PATCH 042/748] fix: voice dictation broken in profiles created via profiles.create (missing stt/tts config) (#85755) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix: mirror voice config (stt/tts/voice) into profiles created via profiles.create Desktop dictation is profile-scoped: /api/audio/transcribe resolves the stt section inside the TARGET profile's home. Profiles created through profiles.create got only a model section, so dictation and TTS silently fell back to defaults (local whisper, often not installed) — 'voice dictation doesn't work in bot mode but is fine in regular mode'. Mirror the launch profile's stt/tts/voice sections (key-wise, never overwriting sections the clone already has) under the same mirror_credentials flag that gates .env/auth mirroring, and report it as mirrored.voice in the receipt. * guard: route voice-config mirror through canonical loaders read_user_config_raw (write-back round-trip; load_config would merge DEFAULT_CONFIG and no-op the mirror) + save_config under the target profile's HERMES_HOME override — same mechanism as _write_profile_model. Satisfies test_config_read_guard. --- tui_gateway/methods_profiles.py | 58 ++++++++++++++++++++++++++++++++- 1 file changed, 57 insertions(+), 1 deletion(-) diff --git a/tui_gateway/methods_profiles.py b/tui_gateway/methods_profiles.py index 39d18c601a..0ff084a0ad 100644 --- a/tui_gateway/methods_profiles.py +++ b/tui_gateway/methods_profiles.py @@ -206,7 +206,7 @@ def _(rid, params: dict) -> dict: # .env (only over the seeded comment-only stub — never clobber real # secrets a clone brought along) and auth.json (only when absent), then # inherit model.provider/model.default unless the caller pinned a model. - mirrored = {"env": False, "auth": False, "model_inherited": False} + mirrored = {"env": False, "auth": False, "model_inherited": False, "voice": False} if is_truthy_value(params.get("mirror_credentials", True)): import shutil @@ -241,6 +241,62 @@ def _(rid, params: dict) -> dict: model = str(params.get("model") or "").strip() provider = str(params.get("provider") or "").strip() model_set = False + + def _mirror_voice_sections() -> bool: + """Copy voice config (stt/tts/voice) from the launch profile. + + Desktop dictation and TTS are profile-scoped: /api/audio/transcribe + resolves the ``stt`` section inside the TARGET profile's home. A + freshly created profile has only a ``model`` section, so voice fell + back to defaults (local whisper, often not installed) and dictation + "didn't work in bot mode" while working on the primary profile. + + Reads/writes go through the canonical loaders scoped to the target + profile via the context-local HERMES_HOME override — the same + mechanism as ``_write_profile_model`` (config-read-guard: no raw + yaml on config.yaml). + """ + try: + from hermes_cli.config import ( + load_config_readonly, + read_user_config_raw, + save_config, + ) + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + src_cfg = load_config_readonly() or {} + sections = { + k: src_cfg[k] for k in ("stt", "tts", "voice") if src_cfg.get(k) + } + if not sections: + return False + + token = set_hermes_home_override(str(path)) + try: + # Write-back round-trip on the raw file: load_config() would + # merge DEFAULT_CONFIG, making every section look present and + # the mirror a no-op (and save_config would then persist the + # entire default tree into the fresh profile). + dst_cfg = read_user_config_raw() or {} + changed = False + for key, value in sections.items(): + if key not in dst_cfg: + dst_cfg[key] = value + changed = True + if changed: + save_config(dst_cfg) + finally: + reset_hermes_home_override(token) + return changed + except Exception: + return False + + if is_truthy_value(params.get("mirror_credentials", True)): + mirrored["voice"] = _mirror_voice_sections() + if model and provider: try: from hermes_cli.web_routers.profiles import _write_profile_model From edb33be51164b7ab5edf8e31c28cba5c8fcc993d Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Tue, 11 Aug 2026 17:16:03 -0600 Subject: [PATCH 043/748] fix(desktop): persist dropped image bytes before attach --- .../chat/hooks/use-composer-actions.test.ts | 76 ++++++++++++++++++- .../app/chat/hooks/use-composer-actions.ts | 9 ++- 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/app/chat/hooks/use-composer-actions.test.ts b/apps/desktop/src/app/chat/hooks/use-composer-actions.test.ts index 76ab53ef95..474cda2f2b 100644 --- a/apps/desktop/src/app/chat/hooks/use-composer-actions.test.ts +++ b/apps/desktop/src/app/chat/hooks/use-composer-actions.test.ts @@ -1,5 +1,7 @@ +import { act, renderHook } from '@testing-library/react' import { afterEach, describe, expect, it, vi } from 'vitest' +import type { ComposerAttachment } from '@/store/composer' import { $connection } from '@/store/session' import { @@ -7,7 +9,8 @@ import { type DroppedFile, extractDroppedFiles, HERMES_PATHS_MIME, - partitionDroppedFiles + partitionDroppedFiles, + useComposerActions } from './use-composer-actions' // A Finder/Explorer drop carries a native File handle; an in-app drag (project @@ -244,3 +247,74 @@ describe('attachmentPreviewDataUrl', () => { await expect(attachmentPreviewDataUrl('/home/gateway/shot.png')).resolves.toBe(REMOTE_PREVIEW) }) }) + +describe('useComposerActions native image drops', () => { + afterEach(() => { + Reflect.deleteProperty(window, 'hermesDesktop') + vi.unstubAllGlobals() + vi.clearAllMocks() + }) + + it('copies dropped screenshot bytes before trusting a transient macOS path', async () => { + const transientPath = + '/var/folders/x7/example/T/TemporaryItems/NSIRD_screencaptureui_4roSuW/Screen Shot 2026-08-11.png' + + const durablePath = '/Users/test/Library/Application Support/Hermes/composer-images/composer_saved.png' + const previewUrl = 'data:image/png;base64,c2NyZWVuc2hvdA==' + + const screenshot = new File([new Uint8Array([1, 2, 3])], 'Screen Shot 2026-08-11.png', { + type: 'image/png' + }) + + const saveImageBuffer = vi.fn(async () => durablePath) + + const readFileDataUrl = vi.fn(async (path: string) => { + if (path === transientPath) { + throw new Error('temporary screenshot path disappeared') + } + + return previewUrl + }) + + const add = vi.fn<(attachment: ComposerAttachment) => void>() + + Object.defineProperty(window, 'hermesDesktop', { + configurable: true, + value: { + readFileDataUrl, + saveImageBuffer + } + }) + + const { result } = renderHook(() => + useComposerActions({ + activeSessionId: null, + currentCwd: '/Users/test/project', + requestGateway: vi.fn(), + scope: { + add, + remove: vi.fn(() => null), + target: 'test-composer' + } + }) + ) + + let attached = false + + await act(async () => { + attached = await result.current.attachDroppedItems([{ file: screenshot, path: transientPath }]) + }) + + expect(attached).toBe(true) + expect(saveImageBuffer).toHaveBeenCalledOnce() + expect(readFileDataUrl).toHaveBeenCalledWith(durablePath) + expect(readFileDataUrl).not.toHaveBeenCalledWith(transientPath) + expect(add).toHaveBeenCalledWith( + expect.objectContaining({ + kind: 'image', + path: durablePath, + previewUrl + }) + ) + }) +}) diff --git a/apps/desktop/src/app/chat/hooks/use-composer-actions.ts b/apps/desktop/src/app/chat/hooks/use-composer-actions.ts index db3c16e8fb..0ff553d033 100644 --- a/apps/desktop/src/app/chat/hooks/use-composer-actions.ts +++ b/apps/desktop/src/app/chat/hooks/use-composer-actions.ts @@ -644,7 +644,14 @@ export function useComposerActions({ const isImage = file.type.startsWith('image/') || isImagePath(file.name) || (filePath && isImagePath(filePath)) if (isImage) { - if ((filePath && (await attachImagePath(filePath))) || (await attachImageBlob(file))) { + // Finder may expose a dropped screenshot through a short-lived + // TemporaryItems/NSIRD_screencaptureui path even when the visible + // file has already landed on Desktop. Reading that path for the + // preview can succeed, then image.attach fails after macOS removes + // it before submit. Persist the File bytes into Desktop's durable + // composer-image cache first; keep the native path as a compatibility + // fallback for older shells that cannot save the buffer. + if ((await attachImageBlob(file)) || (filePath && (await attachImagePath(filePath)))) { attached = true continue From f52feed1efb4ccd8506821081f81000cabe5746d Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Fri, 14 Aug 2026 12:41:26 +1000 Subject: [PATCH 044/748] fix(azure-foundry): scope Responses reasoning suppression to post-tool turns (#84320) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Azure Foundry's OpenAI-compatible Responses surface rejects the post-tool follow-up payload with HTTP 400 `invalid_payload` when a replayed encrypted `reasoning` item is sent alongside `function_call` / `function_call_output`. The initial function-call request and ordinary multi-turn continuity are both accepted, so the failure only appears after the first tool executes. Detect the Foundry endpoint in `ResponsesApiTransport.build_kwargs` and drop only the encrypted reasoning replay on that follow-up turn, leaving function_call / function_call_output continuity intact. Salvage of #59981, rebuilt on current main. Same root cause and fix direction as the original, which was correct; this version resolves three defects: - No `chat_completion_helpers.py` change. main already forwards `provider` and `base_url` to the Responses transport, so the original's re-added arguments produced `SyntaxError: keyword argument repeated: provider` on merge. Dropping the hunk removed the syntax error and the conflict. - Host matching uses `utils.base_url_host_matches`, not a substring test. `".services.ai.azure.com" in base_url` also matches URLs carrying the domain in a path or query segment, which would silently disable reasoning replay on an unrelated provider. - The post-tool predicate tests the trailing messages, not the whole history. Scanning for any tool call plus any tool result made it sticky: one tool call early in a conversation suppressed reasoning on every later turn. - Tool calls pair on `call_id` as well as `id`. Responses histories carry the function call id in `call_id` while `id` holds the response item id (`fc_...`). Identity is resolved via the converter's own `_split_responses_tool_id`, covering composite `"call_x|fc_y"` ids and bare `fc_` ids on both sides of the pairing. Tests: 27 cases across the transport and the live `build_api_kwargs` bridge, including six parametrized tool-call id shapes, non-Foundry host lookalikes, the sticky-history guard, parallel tool results, and an unpaired tool result. Each guard was confirmed to catch its defect by reverting the fix. Verified with `scripts/run_tests.sh tests/agent/ tests/run_agent/`: 532 files, 5602 tests passed, 0 failed. Not verified against a live Azure Foundry endpoint — no credentials. The original HTTP 400 reproduction and post-fix Foundry Project / Azure Container Apps harness runs are @AshuJoshi's, from #59981. This change is verified at the payload-construction layer only. Closes #59981. Co-authored-by: Ashu Joshi --- agent/transports/codex.py | 101 ++++++ .../agent/transports/test_codex_transport.py | 314 ++++++++++++++++++ .../test_run_agent_codex_responses.py | 129 +++++++ 3 files changed, 544 insertions(+) diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 2676dd1fc0..fac925fdba 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -173,6 +173,96 @@ def _content_cache_key( return f"pck_{digest}" +def _is_azure_foundry_responses(params: Dict[str, Any]) -> bool: + """Return True for Microsoft Foundry's OpenAI-compatible Responses API. + + Matched on the registered provider id first, then on the endpoint host. + Host matching goes through ``base_url_host_matches`` rather than a + substring test, so a path or query segment carrying the Foundry domain + (``https://proxy.example.com/.services.ai.azure.com/v1``) is not + misclassified as Foundry. + """ + from utils import base_url_host_matches + + provider = str(params.get("provider") or "").strip().lower() + if provider == "azure-foundry": + return True + + return base_url_host_matches( + str(params.get("base_url") or ""), "services.ai.azure.com" + ) + + +def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool: + """Return True when ``messages`` end on a tool result awaiting a follow-up. + + Azure Foundry only rejects the *post-tool follow-up* payload — the shape + where a prior assistant ``function_call`` and its ``function_call_output`` + are replayed alongside an encrypted ``reasoning`` item (HTTP 400 + invalid_payload). Detecting that shape here keeps reasoning suppression + scoped to the failing turn, so ordinary (non-tool) Foundry multi-turn + continuity is left unchanged. + + The test is on the *trailing* messages, not on the history as a whole. + Scanning the whole history for any tool call plus any tool result makes + the predicate sticky: one tool call early in a conversation would then + suppress reasoning on every later turn, including plain user follow-ups + that Foundry accepts. The rejected payload is specifically the turn whose + last item is a tool result, so that is what this matches: the final + non-system message is a ``tool`` result, and the assistant message that + issued its ``tool_call_id`` is present. + + Tool-call identity is resolved the same way + ``_chat_messages_to_responses_input`` resolves it, because the pairing + that matters is the one that reaches the wire. A stored tool call can + carry the function call id in ``call_id``, in ``id``, or in a composite + ``"call_x|fc_y"`` id, and a bare ``fc_``-prefixed ``id`` is a response + item id that the converter turns into ``call_``. Matching only + ``id`` would miss the ``id=fc_… / call_id=call_…`` shape that resumed + legacy sessions and host-fed histories still use, and let the rejected + payload through. + """ + from agent.codex_responses_adapter import _split_responses_tool_id + + def _pair_ids(raw: Any, explicit: Any = None) -> set: + """Every call id a stored tool id could pair on, converter-order.""" + embedded_call_id, item_id = _split_responses_tool_id(raw) + ids = {embedded_call_id} if embedded_call_id else set() + if isinstance(explicit, str) and explicit.strip(): + ids.add(explicit.strip()) + if not ids and isinstance(raw, str) and raw.strip(): + ids.add(raw.strip()) + if isinstance(item_id, str) and item_id.startswith("fc_") and item_id[3:]: + ids.add(f"call_{item_id[3:]}") + return ids + + trailing = set() + for msg in reversed(messages or ()): + if not isinstance(msg, dict): + return False + role = msg.get("role") + if role == "system": + continue + if role == "tool": + ids = _pair_ids(msg.get("tool_call_id")) + if not ids: + return False + trailing |= ids + continue + # First message before the trailing run of tool results. It must be + # the assistant turn that issued them for this to be the follow-up + # payload; a non-empty ``trailing`` is what proves the run existed. + if role != "assistant": + return False + return any( + trailing & _pair_ids(call.get("id"), call.get("call_id")) + for call in msg.get("tool_calls") or [] + if isinstance(call, dict) + ) + + return False + + class ResponsesApiTransport(ProviderTransport): """Transport for api_mode='codex_responses'. @@ -271,6 +361,17 @@ class ResponsesApiTransport(ProviderTransport): replay_encrypted_reasoning = bool( params.get("replay_encrypted_reasoning", True) ) + if replay_encrypted_reasoning and _is_azure_foundry_responses(params): + # Microsoft Foundry accepts the initial Responses function-call + # request and ordinary (non-tool) multi-turn continuity, but + # rejects the post-tool follow-up payload that carries prior + # encrypted reasoning items alongside function_call / + # function_call_output, with HTTP 400 invalid_payload. Scope the + # suppression to that follow-up turn: keep function_call / + # function_call_output continuity intact and drop only the + # encrypted reasoning replay for this endpoint. + if _is_post_tool_replay(payload_messages): + replay_encrypted_reasoning = False # Native server-side compaction (gpt-5.6 on direct OpenAI/Codex routes # only). The caller resolves eligibility via # agent.native_compaction.native_compaction_context_management(); diff --git a/tests/agent/transports/test_codex_transport.py b/tests/agent/transports/test_codex_transport.py index 5041a32da3..408b49d450 100644 --- a/tests/agent/transports/test_codex_transport.py +++ b/tests/agent/transports/test_codex_transport.py @@ -250,6 +250,320 @@ class TestCodexBuildKwargs: assert eb.get("prompt_cache_key") == "caller-override" assert eb.get("other_field") == 42 + # ── Azure Foundry post-tool reasoning suppression ────────────────── + # + # Foundry's Responses surface accepts the initial function-call request + # and ordinary multi-turn continuity, but rejects the post-tool follow-up + # payload when a replayed encrypted ``reasoning`` item sits alongside + # ``function_call`` / ``function_call_output`` (HTTP 400 invalid_payload). + # Suppression is scoped to that follow-up turn only. + + @staticmethod + def _reasoning_item(): + return {"type": "reasoning", "encrypted_content": "sealed", "summary": []} + + @classmethod + def _post_tool_messages(cls): + """user → assistant(tool_calls + reasoning) → tool result.""" + return [ + {"role": "user", "content": "Create a marker"}, + { + "role": "assistant", + "content": "", + "codex_reasoning_items": [cls._reasoning_item()], + "tool_calls": [ + { + "id": "call_marker", + "type": "function", + "function": {"name": "write_marker", "arguments": "{}"}, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_marker", + "content": "marker written", + }, + ] + + AZURE_FOUNDRY_BASE_URL = ( + "https://placeholder.services.ai.azure.com/" + "api/projects/placeholder/openai/v1" + ) + + def test_post_tool_replay_preserves_reasoning_for_default_responses(self, transport): + """Non-Azure Responses endpoints keep post-tool reasoning replay.""" + kw = transport.build_kwargs( + model="gpt-5.4", + messages=self._post_tool_messages(), + tools=[], + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert "function_call" in item_types + assert "function_call_output" in item_types + assert kw.get("include") == ["reasoning.encrypted_content"] + + def test_azure_foundry_post_tool_replay_suppresses_reasoning_items(self, transport): + """The rejected payload shape drops reasoning, keeps tool continuity.""" + kw = transport.build_kwargs( + model="gpt-5.4", + messages=self._post_tool_messages(), + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" not in item_types + assert "function_call" in item_types + assert "function_call_output" in item_types + assert kw.get("include") == [] + + def test_azure_foundry_detected_by_host_without_provider(self, transport): + """Foundry detection works on the endpoint host alone.""" + kw = transport.build_kwargs( + model="gpt-5.4", + messages=self._post_tool_messages(), + tools=[], + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" not in item_types + + @pytest.mark.parametrize( + "base_url", + [ + "https://proxy.example.com/.services.ai.azure.com/openai/v1", + "https://openrouter.ai/api/v1?upstream=.services.ai.azure.com", + "https://services.ai.azure.com.evil.example/v1", + ], + ) + def test_non_foundry_host_lookalikes_keep_reasoning(self, transport, base_url): + """A Foundry domain in a path/query/suffix is not a Foundry endpoint. + + Guards against the substring-match false positive: these URLs all + contain the Foundry domain but are served by someone else, and + suppressing their reasoning replay would silently degrade + cross-turn coherence on an unrelated provider. + """ + kw = transport.build_kwargs( + model="gpt-5.4", + messages=self._post_tool_messages(), + tools=[], + base_url=base_url, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert kw.get("include") == ["reasoning.encrypted_content"] + + def test_azure_foundry_non_tool_follow_up_preserves_reasoning_items(self, transport): + """Ordinary (non-tool) Azure Foundry continuity is unchanged. + + A plain assistant reasoning turn followed by another user message has + no tool continuity, so the encrypted reasoning item must still be + replayed — Foundry only rejects the post-tool payload. + """ + messages = [ + {"role": "user", "content": "Explain recursion"}, + { + "role": "assistant", + "content": "Recursion is when a function calls itself.", + "codex_reasoning_items": [self._reasoning_item()], + }, + {"role": "user", "content": "Give an example"}, + ] + kw = transport.build_kwargs( + model="gpt-5.4", + messages=messages, + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert "function_call" not in item_types + assert "function_call_output" not in item_types + assert kw.get("include") == ["reasoning.encrypted_content"] + + def test_azure_foundry_user_turn_after_completed_tool_call_keeps_reasoning( + self, transport + ): + """Suppression must not stick for the rest of the conversation. + + The tool call completed and the assistant already answered; this turn + is a plain user follow-up whose payload ends on a user message, which + Foundry accepts. A predicate that scanned the whole history for any + tool call plus any tool result would suppress reasoning here — and on + every later turn — which is the all-turns behavior this scoping + exists to avoid. + """ + messages = self._post_tool_messages() + [ + { + "role": "assistant", + "content": "Marker created.", + "codex_reasoning_items": [self._reasoning_item()], + }, + {"role": "user", "content": "Now explain recursion"}, + ] + kw = transport.build_kwargs( + model="gpt-5.4", + messages=messages, + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert "function_call" in item_types + assert "function_call_output" in item_types + assert kw.get("include") == ["reasoning.encrypted_content"] + + def test_azure_foundry_parallel_tool_results_suppress_reasoning(self, transport): + """A trailing run of parallel tool results is still the rejected shape.""" + messages = [ + {"role": "user", "content": "Read both files"}, + { + "role": "assistant", + "content": "", + "codex_reasoning_items": [self._reasoning_item()], + "tool_calls": [ + { + "id": "call_a", + "type": "function", + "function": {"name": "read_file", "arguments": "{}"}, + }, + { + "id": "call_b", + "type": "function", + "function": {"name": "read_file", "arguments": "{}"}, + }, + ], + }, + {"role": "tool", "tool_call_id": "call_a", "content": "a"}, + {"role": "tool", "tool_call_id": "call_b", "content": "b"}, + ] + kw = transport.build_kwargs( + model="gpt-5.4", + messages=messages, + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" not in item_types + assert item_types.count("function_call_output") == 2 + + def test_azure_foundry_respects_caller_replay_disabled(self, transport): + """An explicit replay_encrypted_reasoning=False is not re-enabled.""" + kw = transport.build_kwargs( + model="gpt-5.4", + messages=self._post_tool_messages(), + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=False, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" not in item_types + assert kw.get("include") == [] + + @pytest.mark.parametrize( + "tool_call,tool_call_id", + [ + # Responses histories carry the function call id in call_id while + # ``id`` holds the response item id. Resumed legacy sessions and + # host-fed histories still use this shape. + ({"id": "fc_item_a", "call_id": "call_a"}, "call_a"), + # Plain chat-completions shape: id IS the call id. + ({"id": "call_a"}, "call_a"), + # Bare fc_ id with no call_id — the converter derives call_. + ({"id": "fc_a"}, "call_a"), + # Composite stored id, on either side of the pairing. + ({"id": "call_a|fc_a"}, "call_a"), + ({"id": "call_a"}, "call_a|fc_a"), + # call_id present, no id at all. + ({"call_id": "call_a"}, "call_a"), + ], + ) + def test_azure_foundry_suppresses_across_tool_call_id_shapes( + self, transport, tool_call, tool_call_id + ): + """Every id shape the converter can pair must be detected. + + The converter resolves a function call's identity as + ``call_id`` -> embedded ``id`` -> derived from an ``fc_`` item id, and + splits composite ``"call_x|fc_y"`` ids. A predicate that matched only + ``tool_calls[*].id`` would miss the id=fc_ / call_id=call_ shape: the + converter still emits paired function_call / function_call_output, so + the exact payload Foundry rejects would ship with the reasoning item + intact. + """ + messages = [ + {"role": "user", "content": "Create a marker"}, + { + "role": "assistant", + "content": "", + "codex_reasoning_items": [self._reasoning_item()], + "tool_calls": [ + {**tool_call, "type": "function", + "function": {"name": "write_marker", "arguments": "{}"}} + ], + }, + {"role": "tool", "tool_call_id": tool_call_id, "content": "marker written"}, + ] + kw = transport.build_kwargs( + model="gpt-5.4", + messages=messages, + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + # The converter paired them, so this is the rejected shape. + assert "function_call" in item_types + assert "function_call_output" in item_types + assert "reasoning" not in item_types + assert kw.get("include") == [] + + def test_azure_foundry_unpaired_tool_result_keeps_reasoning(self, transport): + """A tool result that pairs with nothing is not the rejected shape.""" + messages = [ + {"role": "user", "content": "Create a marker"}, + { + "role": "assistant", + "content": "", + "codex_reasoning_items": [self._reasoning_item()], + "tool_calls": [ + { + "id": "fc_item_x", + "call_id": "call_x", + "type": "function", + "function": {"name": "write_marker", "arguments": "{}"}, + } + ], + }, + {"role": "tool", "tool_call_id": "call_unrelated", "content": "?"}, + ] + kw = transport.build_kwargs( + model="gpt-5.4", + messages=messages, + tools=[], + provider="azure-foundry", + base_url=self.AZURE_FOUNDRY_BASE_URL, + replay_encrypted_reasoning=True, + ) + item_types = [item.get("type") for item in kw["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert kw.get("include") == ["reasoning.encrypted_content"] + def test_xai_top_level_override_also_governs_extra_body(self, transport): """A caller's top-level request_overrides={"prompt_cache_key": ...} must win in extra_body.prompt_cache_key too -- the field xAI actually diff --git a/tests/run_agent/test_run_agent_codex_responses.py b/tests/run_agent/test_run_agent_codex_responses.py index 18dc13526b..5de3418a34 100644 --- a/tests/run_agent/test_run_agent_codex_responses.py +++ b/tests/run_agent/test_run_agent_codex_responses.py @@ -77,6 +77,31 @@ def _build_copilot_agent(monkeypatch, *, model="gpt-5.4"): return agent +AZURE_FOUNDRY_BASE_URL = ( + "https://placeholder.services.ai.azure.com/api/projects/placeholder/openai/v1" +) + + +def _build_azure_foundry_agent(monkeypatch, *, model="gpt-5.4"): + _patch_agent_bootstrap(monkeypatch) + + agent = run_agent.AIAgent( + model=model, + provider="azure-foundry", + api_mode="codex_responses", + base_url=AZURE_FOUNDRY_BASE_URL, + api_key="foundry-token", + quiet_mode=True, + max_iterations=4, + skip_context_files=True, + skip_memory=True, + ) + agent._cleanup_task_resources = lambda task_id: None + agent._persist_session = lambda messages, history=None: None + agent._save_trajectory = lambda messages, user_message, completed: None + return agent + + def _codex_message_response(text: str): return SimpleNamespace( output=[ @@ -325,6 +350,110 @@ def test_build_api_kwargs_mantle_sets_extended_prompt_cache_retention(monkeypatc assert kwargs["prompt_cache_retention"] == "24h" +def _azure_reasoning_item(): + return {"type": "reasoning", "encrypted_content": "sealed", "summary": []} + + +def _azure_post_tool_messages(): + return [ + {"role": "system", "content": "You are Hermes."}, + {"role": "user", "content": "Create a marker"}, + { + "role": "assistant", + "content": "", + "codex_reasoning_items": [_azure_reasoning_item()], + "tool_calls": [ + { + "id": "call_marker", + "type": "function", + "function": {"name": "terminal", "arguments": "{}"}, + } + ], + }, + {"role": "tool", "tool_call_id": "call_marker", "content": "marker written"}, + ] + + +def test_build_api_kwargs_azure_foundry_post_tool_suppresses_reasoning(monkeypatch): + """Live agent path reaches Azure Foundry detection and scopes suppression. + + Exercises ``chat_completion_helpers.build_api_kwargs`` end-to-end rather + than the transport in isolation: the agent must forward ``provider`` and + ``base_url`` into ``build_kwargs`` for the Foundry detection to fire at + all. On the post-tool follow-up shape the encrypted reasoning item is + dropped while function_call / function_call_output continuity is kept. + """ + agent = _build_azure_foundry_agent(monkeypatch) + assert agent._codex_reasoning_replay_enabled is True + + kwargs = agent._build_api_kwargs(_azure_post_tool_messages()) + + item_types = [item.get("type") for item in kwargs["input"] if isinstance(item, dict)] + assert "reasoning" not in item_types + assert "function_call" in item_types + assert "function_call_output" in item_types + assert kwargs.get("include") == [] + + +def test_build_api_kwargs_azure_foundry_non_tool_preserves_reasoning(monkeypatch): + """Ordinary (non-tool) Azure Foundry continuity is unchanged via the live path. + + Without the post-tool follow-up shape there is no evidence Foundry rejects + the payload, so the encrypted reasoning item must still be replayed even + though the agent forwards the Foundry identity fields. + """ + agent = _build_azure_foundry_agent(monkeypatch) + + messages = [ + {"role": "system", "content": "You are Hermes."}, + {"role": "user", "content": "Explain recursion"}, + { + "role": "assistant", + "content": "Recursion is when a function calls itself.", + "codex_reasoning_items": [_azure_reasoning_item()], + }, + {"role": "user", "content": "Give an example"}, + ] + + kwargs = agent._build_api_kwargs(messages) + + item_types = [item.get("type") for item in kwargs["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert "function_call" not in item_types + assert "function_call_output" not in item_types + assert kwargs.get("include") == ["reasoning.encrypted_content"] + + +def test_build_api_kwargs_azure_foundry_user_turn_after_tool_call_keeps_reasoning( + monkeypatch, +): + """Suppression does not stick once the tool call is answered. + + Regression guard for the sticky-history shape: after the assistant has + replied to the tool result, a plain user follow-up is a payload Foundry + accepts, so reasoning replay must come back on rather than stay off for + the remainder of the conversation. + """ + agent = _build_azure_foundry_agent(monkeypatch) + + messages = _azure_post_tool_messages() + [ + { + "role": "assistant", + "content": "Marker created.", + "codex_reasoning_items": [_azure_reasoning_item()], + }, + {"role": "user", "content": "Now explain recursion"}, + ] + + kwargs = agent._build_api_kwargs(messages) + + item_types = [item.get("type") for item in kwargs["input"] if isinstance(item, dict)] + assert "reasoning" in item_types + assert "function_call" in item_types + assert "function_call_output" in item_types + assert kwargs.get("include") == ["reasoning.encrypted_content"] + + From e11d1ddc7fa352b781d3d4699749831905d40090 Mon Sep 17 00:00:00 2001 From: Shannon Sands Date: Thu, 13 Aug 2026 11:58:06 +1000 Subject: [PATCH 045/748] feat(status): surface memory pressure and suspected-OOM restarts to users (NS-656) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hosted agents can be OOM-killed hourly while the dashboard and the NAS agent card both look perfectly healthy — every memory signal the gateway already produces (heartbeat mem samples, lifecycle-ledger unclean-exit verdicts, cache-pressure evictions) dies in server-side log files. The BlueAtlas incident (NS-608) ran for three days like this. This is the read-side fix: * New gateway/memory_status.py distills the existing 30s loop heartbeat (gateway RSS + system MemAvailable/MemTotal + swap) and the lifecycle sentinel into a compact `memory` block: pressure ok/elevated/critical/ unknown, coarse MB numbers, and last-boot unclean/suspected-OOM flags. Pure file reads, no new sampling, no gateway IPC. Stale (>150s) or future-dated heartbeats degrade pressure to "unknown" so a dead gateway's final gasp can't render a live "critical" banner forever. Critical thresholds mirror the ledger's OOM-suspicion heuristics: if a level would make a later unclean death "suspected OOM", warn at that level while the process is still alive. * lifecycle_ledger.record_startup now carries prior_unclean_exit / prior_suspected_oom onto the reclaimed sentinel — previously the verdict survived only in append-only diag prose. Flags age out on the next sentinel rewrite (scoped to the life after the crash). * /api/status serves the block (profile-aware, executor-offloaded, fail-safe to pressure=unknown). Deliberately NOT folded into components/overall: memory pressure is advisory, and flipping overall to "degraded" on it would page NAS's availability sweep for a condition the eviction valve is already handling. Public-safety: coarse numbers/enums/booleans only — same disclosure class as the existing nous_session_valid field, added for the same NAS-sweep audience. * Dashboard: new MemoryPressureBanner (app-shell, next to ProfileScopeBanner) with worst-first trigger precedence (critical > suspected-OOM restart > elevated), per-trigger session-scoped dismissal, and escalation re-opening past a dismissal. i18n keys optional with English fallbacks, matching the managingProfileBanner convention. Tests: gateway/test_memory_status.py (classification bands, staleness, clock skew, corrupt files, bool-is-not-int), lifecycle sentinel carry-forward, /api/status contract (block always present, collector crash degrades instead of 500), and 7 banner component tests. NAS-side ingestion (agent-card notice + memory-tier upsell) ships separately. Refs NS-656; context: NS-608, NS-657, OOF-77. --- gateway/lifecycle_ledger.py | 27 ++- gateway/memory_status.py | 194 ++++++++++++++++++ hermes_cli/web_server.py | 22 ++ tests/gateway/test_lifecycle_ledger.py | 46 +++++ tests/gateway/test_memory_status.py | 169 +++++++++++++++ tests/hermes_cli/test_web_server.py | 35 ++++ web/src/App.tsx | 2 + .../components/MemoryPressureBanner.test.tsx | 124 +++++++++++ web/src/components/MemoryPressureBanner.tsx | 96 +++++++++ web/src/i18n/en.ts | 6 + web/src/i18n/types.ts | 5 + web/src/lib/api.ts | 17 ++ 12 files changed, 734 insertions(+), 9 deletions(-) create mode 100644 gateway/memory_status.py create mode 100644 tests/gateway/test_memory_status.py create mode 100644 web/src/components/MemoryPressureBanner.test.tsx create mode 100644 web/src/components/MemoryPressureBanner.tsx diff --git a/gateway/lifecycle_ledger.py b/gateway/lifecycle_ledger.py index 594da94e5c..0950035939 100644 --- a/gateway/lifecycle_ledger.py +++ b/gateway/lifecycle_ledger.py @@ -254,15 +254,24 @@ def record_startup(home: Optional[Path] = None) -> Optional[Dict[str, Any]]: logger.debug("Unclean-exit detection failed", exc_info=True) try: - _write_sentinel( - { - "phase": "running", - "pid": os.getpid(), - "start_time": time.time(), - "started_at": datetime.now(timezone.utc).isoformat(), - }, - home, - ) + claim: Dict[str, Any] = { + "phase": "running", + "pid": os.getpid(), + "start_time": time.time(), + "started_at": datetime.now(timezone.utc).isoformat(), + } + # Carry the verdict on the PREVIOUS life forward on the new + # sentinel: it is the only place the finding survives in + # machine-readable form (the exit-diag log is append-only prose + # for humans), and /api/status reads it to tell the user "your + # agent restarted after (suspected) running out of memory" + # (NS-656). Scoped to this life only — the next clean exit or + # boot rewrites the sentinel and the flags age out with it. + if evidence is not None: + claim["prior_unclean_exit"] = True + if evidence.get("suspected_oom"): + claim["prior_suspected_oom"] = True + _write_sentinel(claim, home) except Exception: logger.debug("Failed to claim lifecycle sentinel", exc_info=True) return evidence diff --git a/gateway/memory_status.py b/gateway/memory_status.py new file mode 100644 index 0000000000..f440d9ff7b --- /dev/null +++ b/gateway/memory_status.py @@ -0,0 +1,194 @@ +"""Memory status rollup for ``/api/status`` (NS-656). + +The gateway already *produces* every memory-pressure signal a user would +want to know about, but all of it dies in log files: + +* :func:`gateway.shutdown_watchdog.write_loop_heartbeat` embeds a + :func:`gateway.lifecycle_ledger.sample_memory` snapshot (gateway RSS + + system MemAvailable/MemTotal + swap) in ``state/gateway.heartbeat`` + every 30 seconds. +* :func:`gateway.lifecycle_ledger.record_startup` detects an unclean + previous death and flags ``suspected_oom`` — but only into + ``gateway-exit-diag.log`` and a WARNING line. +* ``gateway/agent_cache_pressure.py`` evicts transcripts under pressure, + again log-only. + +So a hosted agent can be OOM-killed hourly (the BlueAtlas incident, +NS-608) while its dashboard and the NAS agent card both look perfectly +healthy. This module is the read side that closes the gap: it distills +the *already-persisted* heartbeat + lifecycle sentinel into a compact, +public-safe block that ``/api/status`` can serve to the dashboard SPA +and the NAS availability sweep — no new sampling, no IPC with the +gateway process, just two small file reads. + +Public-safety note: ``/api/status`` is an unauthenticated liveness probe +(``PUBLIC_API_PATHS``), which is exactly why NAS can consume it. This +block therefore carries only coarse numbers (MB granularity), enums, and +booleans — the same disclosure class as the existing ``active_agents`` +count and ``nous_session_valid`` field (which was added for the same +NAS-sweep audience). + +Everything here is best-effort and read-only: a missing/corrupt file +degrades to ``pressure="unknown"`` rather than raising into the status +endpoint. +""" + +from __future__ import annotations + +import logging +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Dict, Optional + +logger = logging.getLogger(__name__) + +# Pressure thresholds on system MemAvailable. ``critical`` deliberately +# mirrors the lifecycle ledger's OOM-suspicion heuristics +# (:data:`gateway.lifecycle_ledger._LOW_MEM_AVAILABLE_KIB` / +# ``_LOW_MEM_AVAILABLE_FRACTION``): if a memory level would make a +# subsequent unclean death "suspected OOM", the user should already have +# been warned at that level while the process was still alive. +_CRITICAL_AVAILABLE_KIB = 64 * 1024 # < 64 MiB available +_CRITICAL_AVAILABLE_FRACTION = 0.05 # < 5% of MemTotal +_ELEVATED_AVAILABLE_KIB = 128 * 1024 # < 128 MiB available +_ELEVATED_AVAILABLE_FRACTION = 0.15 # < 15% of MemTotal + +# A heartbeat older than this no longer describes the present. The writer +# cadence is 30s (DEFAULT_HEARTBEAT_INTERVAL_S); 150s of slack tolerates a +# briefly stalled loop without letting a long-dead gateway's last sample +# masquerade as current pressure. +_HEARTBEAT_FRESH_TTL_S = 150.0 + +_KIB_PER_MB = 1024 + + +def _mb(kib: Any) -> Optional[int]: + if isinstance(kib, bool) or not isinstance(kib, int) or kib < 0: + return None + return kib // _KIB_PER_MB + + +def _parse_iso(value: Any) -> Optional[datetime]: + if not isinstance(value, str) or not value: + return None + try: + parsed = datetime.fromisoformat(value) + except ValueError: + return None + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed + + +def classify_pressure( + available_kib: Any, total_kib: Any +) -> str: + """Map a MemAvailable/MemTotal pair to ``ok``/``elevated``/``critical``. + + ``unknown`` when the sample is missing or malformed — the caller must + not treat "we could not read it" as "memory is fine". + """ + if ( + isinstance(available_kib, bool) + or not isinstance(available_kib, int) + or available_kib < 0 + ): + return "unknown" + fraction: Optional[float] = None + if ( + not isinstance(total_kib, bool) + and isinstance(total_kib, int) + and total_kib > 0 + ): + fraction = available_kib / total_kib + if available_kib < _CRITICAL_AVAILABLE_KIB or ( + fraction is not None and fraction < _CRITICAL_AVAILABLE_FRACTION + ): + return "critical" + if available_kib < _ELEVATED_AVAILABLE_KIB or ( + fraction is not None and fraction < _ELEVATED_AVAILABLE_FRACTION + ): + return "elevated" + return "ok" + + +def _read_heartbeat(home: Optional[Path]) -> Optional[Dict[str, Any]]: + try: + from gateway.lifecycle_ledger import _read_json + from gateway.shutdown_watchdog import get_loop_heartbeat_path + + return _read_json(get_loop_heartbeat_path(home)) + except Exception: + return None + + +def _read_sentinel(home: Optional[Path]) -> Optional[Dict[str, Any]]: + try: + from gateway.lifecycle_ledger import ( + _read_json, + get_lifecycle_sentinel_path, + ) + + return _read_json(get_lifecycle_sentinel_path(home)) + except Exception: + return None + + +def collect_memory_status( + home: Optional[Path] = None, + *, + now: Optional[datetime] = None, +) -> Dict[str, Any]: + """Build the ``memory`` block for ``/api/status``. + + ``home`` scopes the read to a profile's HERMES_HOME (the status + endpoint's ``?profile=`` handling passes it through); ``None`` means + the active profile. ``now`` is injectable for tests. + + Always returns a dict — a gateway that is down, has never written a + heartbeat, or whose files are corrupt yields + ``{"pressure": "unknown", ...}`` with whatever fields could still be + recovered. Never raises. + """ + moment = now or datetime.now(timezone.utc) + status: Dict[str, Any] = { + "pressure": "unknown", + "gateway_rss_mb": None, + "system_total_mb": None, + "system_available_mb": None, + "swap_used_mb": None, + "sampled_at": None, + "last_boot_unclean": False, + "last_boot_suspected_oom": False, + } + + heartbeat = _read_heartbeat(home) + if heartbeat: + sampled_at = _parse_iso(heartbeat.get("updated_at")) + mem = heartbeat.get("mem") + if isinstance(mem, dict): + status["gateway_rss_mb"] = _mb(mem.get("rss_kib")) + status["system_total_mb"] = _mb(mem.get("mem_total_kib")) + status["system_available_mb"] = _mb(mem.get("mem_available_kib")) + status["swap_used_mb"] = _mb(mem.get("swap_used_kib")) + if sampled_at is not None: + status["sampled_at"] = sampled_at.isoformat() + age_s = (moment - sampled_at).total_seconds() + if 0 <= age_s <= _HEARTBEAT_FRESH_TTL_S: + status["pressure"] = classify_pressure( + mem.get("mem_available_kib"), + mem.get("mem_total_kib"), + ) + # else: stale sample — numbers are still reported (they are + # honest about *when* via sampled_at) but pressure stays + # "unknown" so a dead gateway's final gasp cannot render a + # live "critical" banner forever. + + sentinel = _read_sentinel(home) + if sentinel: + status["last_boot_unclean"] = bool(sentinel.get("prior_unclean_exit")) + status["last_boot_suspected_oom"] = bool( + sentinel.get("prior_suspected_oom") + ) + + return status diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index a533fbc2a5..543e1c0b9a 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -3332,6 +3332,28 @@ async def get_status(profile: Optional[str] = None): else "degraded" ) + # Memory-pressure rollup (NS-656). Distilled from the gateway's + # 30s loop heartbeat + lifecycle sentinel — two small file reads, + # no gateway IPC. Coarse MB numbers/enums/booleans only: this + # endpoint is public (PUBLIC_API_PATHS), same disclosure class as + # nous_session_valid above. Deliberately NOT folded into + # components/overall — memory pressure is advisory (toast/notice + # material), not a liveness verdict, and flipping `overall` to + # "degraded" on it would page NAS's availability sweep for a + # condition the valve is already handling. + try: + from gateway.memory_status import collect_memory_status + + status["memory"] = await asyncio.get_running_loop().run_in_executor( + None, + functools.partial( + collect_memory_status, + profile_dir if profile_dir else get_hermes_home(), + ), + ) + except Exception: + status["memory"] = {"pressure": "unknown"} + # Deferred FTS rebuild progress (schema v23): lets the desktop / # dashboard render a "search index rebuilding: N%" indicator instead # of users wondering why old-message search is slower after an diff --git a/tests/gateway/test_lifecycle_ledger.py b/tests/gateway/test_lifecycle_ledger.py index 6ec2beeeb8..016a8b2bcd 100644 --- a/tests/gateway/test_lifecycle_ledger.py +++ b/tests/gateway/test_lifecycle_ledger.py @@ -143,6 +143,52 @@ def test_record_startup_persists_unclean_report_and_reclaims(tmp_path: Path) -> assert sentinel["pid"] == os.getpid() +def test_record_startup_carries_unclean_flags_onto_new_sentinel( + tmp_path: Path, +) -> None: + """The unclean-death verdict must survive on the reclaimed sentinel so + /api/status can surface "restarted after (suspected) OOM" (NS-656).""" + _write_sentinel(tmp_path, { + "phase": "running", + "pid": _DEAD_PID, + "start_time": 1000.0, + "started_at": "2026-07-11T04:30:00+00:00", + }) + # Last heartbeat shows near-exhausted memory → suspected OOM. + from gateway.shutdown_watchdog import get_loop_heartbeat_path + + hb_path = get_loop_heartbeat_path(tmp_path) + hb_path.parent.mkdir(parents=True, exist_ok=True) + hb_path.write_text(json.dumps({ + "pid": _DEAD_PID, + "updated_at": "2026-07-11T05:00:00+00:00", + "mem": {"mem_total_kib": 1024 * 1024, "mem_available_kib": 20 * 1024}, + }), encoding="utf-8") + + evidence = record_startup(home=tmp_path) + assert evidence is not None + assert evidence.get("suspected_oom") is True + + sentinel = _read_sentinel(tmp_path) + assert sentinel["phase"] == "running" + assert sentinel["prior_unclean_exit"] is True + assert sentinel["prior_suspected_oom"] is True + + +def test_record_startup_clean_boot_has_no_prior_flags(tmp_path: Path) -> None: + _write_sentinel(tmp_path, { + "phase": "exited", + "pid": _DEAD_PID, + "exit_code": 0, + "exit_reason": "graceful_shutdown", + }) + assert record_startup(home=tmp_path) is None + sentinel = _read_sentinel(tmp_path) + assert sentinel["phase"] == "running" + assert "prior_unclean_exit" not in sentinel + assert "prior_suspected_oom" not in sentinel + + # --------------------------------------------------------------------------- # Takeover ownership guard on mark_exited # --------------------------------------------------------------------------- diff --git a/tests/gateway/test_memory_status.py b/tests/gateway/test_memory_status.py new file mode 100644 index 0000000000..0b533bd632 --- /dev/null +++ b/tests/gateway/test_memory_status.py @@ -0,0 +1,169 @@ +"""Tests for gateway.memory_status — the /api/status memory rollup (NS-656).""" + +from __future__ import annotations + +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path + +from gateway.memory_status import classify_pressure, collect_memory_status +from gateway.shutdown_watchdog import get_loop_heartbeat_path +from gateway.lifecycle_ledger import get_lifecycle_sentinel_path + +_NOW = datetime(2026, 8, 13, 12, 0, 0, tzinfo=timezone.utc) + + +def _write_heartbeat( + home: Path, + *, + updated_at: datetime = _NOW, + mem: dict | None = None, +) -> None: + path = get_loop_heartbeat_path(home) + path.parent.mkdir(parents=True, exist_ok=True) + payload = { + "pid": 12345, + "updated_at": updated_at.isoformat(), + "monotonic": 1.0, + } + if mem is not None: + payload["mem"] = mem + path.write_text(json.dumps(payload), encoding="utf-8") + + +def _write_sentinel(home: Path, payload: dict) -> None: + path = get_lifecycle_sentinel_path(home) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload), encoding="utf-8") + + +class TestClassifyPressure: + def test_plentiful_memory_is_ok(self) -> None: + # 1 GiB available of 2 GiB total. + assert classify_pressure(1024 * 1024, 2048 * 1024) == "ok" + + def test_low_absolute_available_is_critical(self) -> None: + # 32 MiB available — below the 64 MiB floor regardless of total. + assert classify_pressure(32 * 1024, 8 * 1024 * 1024) == "critical" + + def test_low_fraction_is_critical(self) -> None: + # 300 MiB available of 8 GiB ≈ 3.7% < 5%. + assert classify_pressure(300 * 1024, 8 * 1024 * 1024) == "critical" + + def test_elevated_band(self) -> None: + # 100 MiB available of 1 GiB ≈ 9.8% — above critical, below elevated + # thresholds (128 MiB / 15%). + assert classify_pressure(100 * 1024, 1024 * 1024) == "elevated" + + def test_missing_sample_is_unknown(self) -> None: + assert classify_pressure(None, None) == "unknown" + + def test_bool_is_not_an_int(self) -> None: + # True == 1 in Python — must not classify as "1 KiB available". + assert classify_pressure(True, 2048 * 1024) == "unknown" + + def test_absolute_floor_works_without_total(self) -> None: + assert classify_pressure(32 * 1024, None) == "critical" + # 1 GiB available, unknown total: passes both absolute floors → ok. + assert classify_pressure(1024 * 1024, None) == "ok" + + +class TestCollectMemoryStatus: + def test_no_files_yields_unknown(self, tmp_path: Path) -> None: + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "unknown" + assert status["gateway_rss_mb"] is None + assert status["last_boot_unclean"] is False + assert status["last_boot_suspected_oom"] is False + + def test_fresh_heartbeat_reports_pressure_and_numbers( + self, tmp_path: Path + ) -> None: + _write_heartbeat( + tmp_path, + updated_at=_NOW - timedelta(seconds=30), + mem={ + "rss_kib": 400 * 1024, + "mem_total_kib": 1024 * 1024, + "mem_available_kib": 50 * 1024, + "swap_used_kib": 200 * 1024, + }, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "critical" + assert status["gateway_rss_mb"] == 400 + assert status["system_total_mb"] == 1024 + assert status["system_available_mb"] == 50 + assert status["swap_used_mb"] == 200 + assert status["sampled_at"] is not None + + def test_stale_heartbeat_keeps_numbers_but_unknown_pressure( + self, tmp_path: Path + ) -> None: + # A dead gateway's final gasp must not render a live "critical" + # banner forever. + _write_heartbeat( + tmp_path, + updated_at=_NOW - timedelta(hours=2), + mem={ + "rss_kib": 400 * 1024, + "mem_total_kib": 1024 * 1024, + "mem_available_kib": 10 * 1024, + }, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "unknown" + assert status["system_available_mb"] == 10 + assert status["sampled_at"] is not None + + def test_future_heartbeat_is_treated_as_stale(self, tmp_path: Path) -> None: + # Clock skew / restored snapshots: a timestamp from the future is + # not evidence about the present either. + _write_heartbeat( + tmp_path, + updated_at=_NOW + timedelta(hours=1), + mem={"mem_total_kib": 1024 * 1024, "mem_available_kib": 10 * 1024}, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "unknown" + + def test_sentinel_flags_surface(self, tmp_path: Path) -> None: + _write_sentinel( + tmp_path, + { + "phase": "running", + "pid": 999, + "prior_unclean_exit": True, + "prior_suspected_oom": True, + }, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["last_boot_unclean"] is True + assert status["last_boot_suspected_oom"] is True + + def test_clean_sentinel_has_no_flags(self, tmp_path: Path) -> None: + _write_sentinel( + tmp_path, + {"phase": "exited", "pid": 999, "exit_reason": "graceful_shutdown"}, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["last_boot_unclean"] is False + assert status["last_boot_suspected_oom"] is False + + def test_corrupt_files_never_raise(self, tmp_path: Path) -> None: + hb = get_loop_heartbeat_path(tmp_path) + hb.parent.mkdir(parents=True, exist_ok=True) + hb.write_text("{not json", encoding="utf-8") + sentinel = get_lifecycle_sentinel_path(tmp_path) + sentinel.parent.mkdir(parents=True, exist_ok=True) + sentinel.write_text("[]", encoding="utf-8") # valid JSON, wrong shape + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "unknown" + + def test_heartbeat_without_mem_block(self, tmp_path: Path) -> None: + # Non-Linux hosts: sample_memory() returns {} so the heartbeat has + # no mem key at all. + _write_heartbeat(tmp_path, updated_at=_NOW) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["pressure"] == "unknown" + assert status["gateway_rss_mb"] is None diff --git a/tests/hermes_cli/test_web_server.py b/tests/hermes_cli/test_web_server.py index a3c809d445..5a06173807 100644 --- a/tests/hermes_cli/test_web_server.py +++ b/tests/hermes_cli/test_web_server.py @@ -2994,6 +2994,41 @@ class TestGatewayBusyReadout: assert data["gateway_busy"] is False +class TestStatusMemoryBlock: + """NS-656: /api/status must always carry a `memory` block.""" + + @pytest.fixture(autouse=True) + def _setup_test_client(self): + try: + from starlette.testclient import TestClient + except ImportError: + pytest.skip("fastapi/starlette not installed") + + from hermes_cli.web_server import app, _SESSION_HEADER_NAME, _SESSION_TOKEN + self.client = TestClient(app) + self.client.headers[_SESSION_HEADER_NAME] = _SESSION_TOKEN + + def test_memory_block_present_with_pressure_field(self): + data = self.client.get("/api/status").json() + assert "memory" in data + assert data["memory"]["pressure"] in { + "ok", "elevated", "critical", "unknown", + } + + def test_memory_block_degrades_when_collector_raises(self, monkeypatch): + """A broken collector must never take down the status endpoint — + the block degrades to pressure=unknown.""" + import gateway.memory_status as ms + + def _boom(*_a, **_k): + raise RuntimeError("collector exploded") + + monkeypatch.setattr(ms, "collect_memory_status", _boom) + resp = self.client.get("/api/status") + assert resp.status_code == 200 + assert resp.json()["memory"] == {"pressure": "unknown"} + + class TestGatewayUpdatedAtContract: """Contract tests for /api/status ``gateway_updated_at``. diff --git a/web/src/App.tsx b/web/src/App.tsx index 7fe8c7f8c2..6f57d43493 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -72,6 +72,7 @@ import { ProfileProvider } from "@/contexts/ProfileProvider"; import { useProfileScope } from "@/contexts/useProfileScope"; import { ProfileSwitcher } from "@/components/ProfileSwitcher"; import { ProfileScopeBanner } from "@/components/ProfileScopeBanner"; +import { MemoryPressureBanner } from "@/components/MemoryPressureBanner"; import { useSystemActions } from "@/contexts/useSystemActions"; import type { SystemAction } from "@/contexts/system-actions-context"; // Route pages are lazy-loaded so the initial dashboard shell does not pay for @@ -567,6 +568,7 @@ export default function App() { +
diff --git a/web/src/components/MemoryPressureBanner.test.tsx b/web/src/components/MemoryPressureBanner.test.tsx new file mode 100644 index 0000000000..3baa30b564 --- /dev/null +++ b/web/src/components/MemoryPressureBanner.test.tsx @@ -0,0 +1,124 @@ +// @vitest-environment jsdom +// Tests for the NS-656 memory-pressure banner: trigger selection, +// severity precedence, dismissal semantics, and escalation re-opening. + +import { describe, it, expect, beforeEach, afterEach } from "vitest"; +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; +import type { ReactNode } from "react"; + +import { I18nProvider } from "@/i18n"; +import { MemoryPressureBanner } from "./MemoryPressureBanner"; +import type { StatusResponse, MemoryStatus } from "@/lib/api"; + +let container: HTMLDivElement; +let root: Root; + +async function render(ui: ReactNode) { + container = document.createElement("div"); + document.body.append(container); + root = createRoot(container); + await act(async () => root.render({ui})); +} + +async function rerender(ui: ReactNode) { + await act(async () => root.render({ui})); +} + +beforeEach(() => { + sessionStorage.clear(); +}); + +afterEach(async () => { + await act(async () => root?.unmount()); + container?.remove(); +}); + +function statusWith(memory: MemoryStatus | undefined): StatusResponse { + return { memory } as StatusResponse; +} + +function banner(): HTMLElement | null { + return container.querySelector('[data-testid="memory-pressure-banner"]'); +} + +describe("MemoryPressureBanner", () => { + it("renders nothing for null status or healthy memory", async () => { + await render(); + expect(banner()).toBeNull(); + await rerender( + , + ); + expect(banner()).toBeNull(); + // Older gateways: no memory block at all. + await rerender(); + expect(banner()).toBeNull(); + }); + + it("renders nothing for unknown pressure (absence of evidence)", async () => { + await render( + , + ); + expect(banner()).toBeNull(); + }); + + it("shows the elevated warning", async () => { + await render( + , + ); + expect(banner()?.textContent).toContain("running low on memory"); + }); + + it("shows the OOM-restart notice even when current pressure is ok", async () => { + await render( + , + ); + expect(banner()?.textContent).toContain( + "restarted after running out of memory", + ); + }); + + it("critical pressure outranks the OOM-restart notice", async () => { + await render( + , + ); + expect(banner()?.textContent).toContain("almost out of memory"); + }); + + it("dismissal hides the banner and persists across re-renders", async () => { + await render( + , + ); + const dismiss = container.querySelector( + '[data-testid="memory-pressure-banner"] button', + ) as HTMLButtonElement; + await act(async () => dismiss.click()); + expect(banner()).toBeNull(); + await rerender( + , + ); + expect(banner()).toBeNull(); + }); + + it("escalation to critical re-opens a dismissed banner", async () => { + await render( + , + ); + const dismiss = container.querySelector( + '[data-testid="memory-pressure-banner"] button', + ) as HTMLButtonElement; + await act(async () => dismiss.click()); + expect(banner()).toBeNull(); + await rerender( + , + ); + expect(banner()?.textContent).toContain("almost out of memory"); + }); +}); diff --git a/web/src/components/MemoryPressureBanner.tsx b/web/src/components/MemoryPressureBanner.tsx new file mode 100644 index 0000000000..9a7ef628ee --- /dev/null +++ b/web/src/components/MemoryPressureBanner.tsx @@ -0,0 +1,96 @@ +import { useState } from "react"; +import { AlertTriangle, X } from "lucide-react"; +import type { StatusResponse } from "@/lib/api"; +import { useI18n } from "@/i18n"; + +/** + * App-wide warning banner for memory trouble (NS-656). + * + * Two independent triggers, worst-first: + * 1. Live pressure — the gateway's heartbeat shows system memory in the + * `elevated`/`critical` band right now. + * 2. Post-mortem — the previous gateway life died uncleanly and its last + * heartbeat showed near-exhausted memory (`last_boot_suspected_oom`). + * + * Both previously died in server-side log files; a hosted agent could be + * OOM-killed hourly while the dashboard looked healthy. + * + * Dismissal is session-scoped per trigger kind (sessionStorage), so a user + * who acknowledged "restarted after OOM" is not re-nagged on every poll, + * but a NEW escalation (ok → critical) still surfaces. + */ +export function MemoryPressureBanner({ + status, +}: { + status: StatusResponse | null; +}) { + const { t } = useI18n(); + const memory = status?.memory; + + // Highest-severity active trigger, or null. + const trigger = !memory + ? null + : memory.pressure === "critical" + ? "critical" + : memory.last_boot_suspected_oom + ? "oom_restart" + : memory.pressure === "elevated" + ? "elevated" + : null; + + const [dismissed, setDismissed] = useState(() => { + try { + return sessionStorage.getItem("memoryBannerDismissed"); + } catch { + return null; + } + }); + + // Dismissal only masks the exact trigger that was dismissed — an + // escalation (elevated → critical) changes `trigger` and therefore + // re-opens the banner without any effect/state churn. + if (!trigger || dismissed === trigger) return null; + + const dismiss = () => { + setDismissed(trigger); + try { + sessionStorage.setItem("memoryBannerDismissed", trigger); + } catch { + /* ignore */ + } + }; + + const critical = trigger === "critical"; + const message = + trigger === "oom_restart" + ? (t.app.memoryOomRestartBanner ?? + "Your agent restarted after running out of memory. Long sessions and many concurrent tasks increase memory use.") + : critical + ? (t.app.memoryCriticalBanner ?? + "Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.") + : (t.app.memoryElevatedBanner ?? + "Your agent is running low on memory."); + + return ( +
+ + {message} + +
+ ); +} diff --git a/web/src/i18n/en.ts b/web/src/i18n/en.ts index dda0dd0e93..4fa280106e 100644 --- a/web/src/i18n/en.ts +++ b/web/src/i18n/en.ts @@ -97,6 +97,12 @@ export const en: Translations = { currentProfileOption: "this dashboard ({name})", managingProfileBanner: "Managing profile \u201c{name}\u201d \u2014 config, keys, skills, MCPs, model, and new chats apply to that profile.", + memoryOomRestartBanner: + "Your agent restarted after running out of memory. Long sessions and many concurrent tasks increase memory use.", + memoryCriticalBanner: + "Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.", + memoryElevatedBanner: "Your agent is running low on memory.", + dismiss: "Dismiss", }, status: { diff --git a/web/src/i18n/types.ts b/web/src/i18n/types.ts index 5c79a937e5..b669d8877c 100644 --- a/web/src/i18n/types.ts +++ b/web/src/i18n/types.ts @@ -115,6 +115,11 @@ export interface Translations { managingProfile?: string; currentProfileOption?: string; managingProfileBanner?: string; + /** NS-656 memory-pressure banner — optional, English fallback. */ + memoryOomRestartBanner?: string; + memoryCriticalBanner?: string; + memoryElevatedBanner?: string; + dismiss?: string; }; // ── Status page ── diff --git a/web/src/lib/api.ts b/web/src/lib/api.ts index 02ff011945..79f7dbc991 100644 --- a/web/src/lib/api.ts +++ b/web/src/lib/api.ts @@ -1882,10 +1882,27 @@ export interface StatusResponse { gateway_updated_at: string | null; hermes_home: string; latest_config_version: number; + /** NS-656: memory-pressure rollup from the gateway heartbeat + + * lifecycle ledger. Absent on older gateways. */ + memory?: MemoryStatus; release_date: string; version: string; } +/** NS-656: coarse memory telemetry served by /api/status. */ +export interface MemoryStatus { + pressure: "ok" | "elevated" | "critical" | "unknown"; + gateway_rss_mb?: number | null; + system_total_mb?: number | null; + system_available_mb?: number | null; + swap_used_mb?: number | null; + sampled_at?: string | null; + /** Previous gateway life died without running any exit path. */ + last_boot_unclean?: boolean; + /** ...and its final heartbeat showed near-exhausted memory. */ + last_boot_suspected_oom?: boolean; +} + export interface SessionInfo { id: string; source: string | null; From 1745cf3b4022289c45f46c6fe3b58123b7ac0c13 Mon Sep 17 00:00:00 2001 From: Shannon Sands Date: Thu, 13 Aug 2026 12:54:55 +1000 Subject: [PATCH 046/748] fix(web): rename memory-pressure interface to avoid declaration merge with existing MemoryStatus MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit api.ts already declares MemoryStatus for the /api/memory providers endpoint; the NS-656 pressure block reused the name, and TypeScript declaration merging fused the two shapes — 'tsc -b' in the Docker image build failed on every test fixture. Renamed to MemoryPressureStatus. --- web/src/components/MemoryPressureBanner.test.tsx | 4 ++-- web/src/lib/api.ts | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/web/src/components/MemoryPressureBanner.test.tsx b/web/src/components/MemoryPressureBanner.test.tsx index 3baa30b564..e92905c85a 100644 --- a/web/src/components/MemoryPressureBanner.test.tsx +++ b/web/src/components/MemoryPressureBanner.test.tsx @@ -9,7 +9,7 @@ import type { ReactNode } from "react"; import { I18nProvider } from "@/i18n"; import { MemoryPressureBanner } from "./MemoryPressureBanner"; -import type { StatusResponse, MemoryStatus } from "@/lib/api"; +import type { StatusResponse, MemoryPressureStatus } from "@/lib/api"; let container: HTMLDivElement; let root: Root; @@ -34,7 +34,7 @@ afterEach(async () => { container?.remove(); }); -function statusWith(memory: MemoryStatus | undefined): StatusResponse { +function statusWith(memory: MemoryPressureStatus | undefined): StatusResponse { return { memory } as StatusResponse; } diff --git a/web/src/lib/api.ts b/web/src/lib/api.ts index 79f7dbc991..c5b7eb4d0e 100644 --- a/web/src/lib/api.ts +++ b/web/src/lib/api.ts @@ -1884,13 +1884,13 @@ export interface StatusResponse { latest_config_version: number; /** NS-656: memory-pressure rollup from the gateway heartbeat + * lifecycle ledger. Absent on older gateways. */ - memory?: MemoryStatus; + memory?: MemoryPressureStatus; release_date: string; version: string; } /** NS-656: coarse memory telemetry served by /api/status. */ -export interface MemoryStatus { +export interface MemoryPressureStatus { pressure: "ok" | "elevated" | "critical" | "unknown"; gateway_rss_mb?: number | null; system_total_mb?: number | null; From ba5dc00bb20128017204f1185e5034998e83c22e Mon Sep 17 00:00:00 2001 From: Shannon Sands Date: Thu, 13 Aug 2026 14:34:47 +1000 Subject: [PATCH 047/748] =?UTF-8?q?fix(memory-status):=20review=20follow-u?= =?UTF-8?q?ps=20=E2=80=94=20incident-keyed=20dismissal,=20honest=20copy,?= =?UTF-8?q?=20single=20mobile=20offset=20(NS-656)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the human review findings on the memory-pressure feature: * [P2] Dismissal hid later incidents of the same kind. The gateway now publishes `boot_id` (the lifecycle sentinel's started_at — changes on every gateway life) in the /api/status memory block, and the dashboard keys OOM-restart dismissal on it: acknowledging one restart no longer mutes the NEXT one (the OOM-loop case this banner exists for). Live pressure dismissals now also reset once pressure is demonstrably back to "ok" — "unknown" (stale heartbeat) is absence of evidence and clears nothing. Dismissal storage moved to a JSON list; old bare-string entries fail JSON.parse and degrade to a clean reset. * [P2] suspected_oom is a heuristic (unclean exit + low-memory final heartbeat), not proof the OOM killer acted — banner copy now says "restarted unexpectedly, most likely because it ran out of memory" instead of stating OOM as fact. * [P3] Mobile header clearance was applied per-banner (mt-14 on both MemoryPressureBanner and ProfileScopeBanner) AND on the content (pt-14), double/triple-stacking 56px gaps when banners were visible. Replaced with a single h-14 spacer above the banner stack. --- gateway/memory_status.py | 9 ++ tests/gateway/test_memory_status.py | 15 +++ web/src/App.tsx | 7 +- .../components/MemoryPressureBanner.test.tsx | 80 +++++++++++++++- web/src/components/MemoryPressureBanner.tsx | 95 ++++++++++++++----- web/src/components/ProfileScopeBanner.tsx | 5 +- web/src/i18n/en.ts | 2 +- web/src/lib/api.ts | 6 +- 8 files changed, 189 insertions(+), 30 deletions(-) diff --git a/gateway/memory_status.py b/gateway/memory_status.py index f440d9ff7b..76af13bd46 100644 --- a/gateway/memory_status.py +++ b/gateway/memory_status.py @@ -160,6 +160,12 @@ def collect_memory_status( "sampled_at": None, "last_boot_unclean": False, "last_boot_suspected_oom": False, + # Identity of the CURRENT gateway life (the sentinel's started_at). + # A suspected-OOM restart writes a fresh sentinel, so this changes on + # every boot — the dashboard keys banner dismissal on it so that + # acknowledging one OOM restart does not mute reports of the NEXT + # one (the hourly-restart-loop case is exactly the one that matters). + "boot_id": None, } heartbeat = _read_heartbeat(home) @@ -190,5 +196,8 @@ def collect_memory_status( status["last_boot_suspected_oom"] = bool( sentinel.get("prior_suspected_oom") ) + started_at = sentinel.get("started_at") + if isinstance(started_at, str) and started_at: + status["boot_id"] = started_at return status diff --git a/tests/gateway/test_memory_status.py b/tests/gateway/test_memory_status.py index 0b533bd632..4d9afa4202 100644 --- a/tests/gateway/test_memory_status.py +++ b/tests/gateway/test_memory_status.py @@ -133,6 +133,7 @@ class TestCollectMemoryStatus: { "phase": "running", "pid": 999, + "started_at": "2026-08-13T01:00:00+00:00", "prior_unclean_exit": True, "prior_suspected_oom": True, }, @@ -140,6 +141,20 @@ class TestCollectMemoryStatus: status = collect_memory_status(tmp_path, now=_NOW) assert status["last_boot_unclean"] is True assert status["last_boot_suspected_oom"] is True + # boot_id identifies the reporting life so the dashboard can key + # banner dismissal per incident (a NEW restart re-surfaces it). + assert status["boot_id"] == "2026-08-13T01:00:00+00:00" + + def test_boot_id_absent_or_malformed_stays_none(self, tmp_path: Path) -> None: + _write_sentinel( + tmp_path, + {"phase": "running", "pid": 999, "started_at": 12345}, + ) + status = collect_memory_status(tmp_path, now=_NOW) + assert status["boot_id"] is None + assert collect_memory_status(tmp_path.joinpath("nohome"), now=_NOW)[ + "boot_id" + ] is None def test_clean_sentinel_has_no_flags(self, tmp_path: Path) -> None: _write_sentinel( diff --git a/web/src/App.tsx b/web/src/App.tsx index 6f57d43493..bb79cb6a6a 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -566,11 +566,16 @@ export default function App() { /> )} + {/* Single mobile header clearance for the banner stack + content. The + fixed lg:hidden header is h-14/z-40; previously each banner carried + its own mt-14 AND the content kept pt-14, so two visible banners + stacked three offsets (NS-656 review P3). One spacer, applied once. */} +
-
+