diff --git a/hermes_cli/azure_detect.py b/hermes_cli/azure_detect.py index 7638ed6aca..d5e24c2039 100644 --- a/hermes_cli/azure_detect.py +++ b/hermes_cli/azure_detect.py @@ -1,38 +1,8 @@ """Azure Foundry endpoint auto-detection. -Inspect a Microsoft Foundry / Azure OpenAI endpoint to determine: - - API transport (OpenAI-style ``chat_completions`` vs - Anthropic-style ``anthropic_messages``) - - Available models (best effort — Azure does not expose a deployment - listing via the inference API key, but Azure OpenAI v1 endpoints - return the resource's model catalog via ``GET /models``) - - Context length for each discovered/entered model, via the existing - :func:`agent.model_metadata.get_model_context_length` resolver. - -Rationale: - -Azure has no pure-API-key deployment-listing endpoint — per Microsoft, -deployment enumeration requires ARM management-plane auth. Azure -OpenAI v1 endpoints ``{resource}.openai.azure.com/openai/v1`` do return -a ``/models`` list, but it reflects the resource's *available* models -rather than the user's *deployed* deployment names. In practice it is -still a useful hint — the user picks a familiar model name and we look -up its context length from the catalog. - -Authentication modes: - - ``api_key`` (default): the wizard passes an ``api_key`` string; the - probe sends both ``api-key:`` and ``Authorization: Bearer`` headers - so we hit any Azure deployment regardless of which header it expects. - - ``entra_id``: the wizard passes a ``token_provider`` callable from - :mod:`agent.azure_identity_adapter`. The probe mints exactly one - bearer JWT, sends **only** ``Authorization: Bearer `` (never - ``api-key:``), and never persists the token. This matches Microsoft's - documented contract for keyless inference. - -The detector never crashes on errors (every HTTP call is wrapped in a -broad try/except). Callers get a :class:`DetectionResult` with whatever -information could be gathered, and fall back to manual entry for the -rest. +The detector never crashes on errors (every HTTP call is wrapped in a broad try/except). Callers get +a :class:`DetectionResult` with whatever information could be gathered, and fall back to manual +entry for the rest. """ from __future__ import annotations @@ -96,55 +66,40 @@ def _resolve_credential(api_key: Any, ) -> tuple[Optional[str], str]: """Coerce wizard inputs into a (token, mode) pair. - Returns ``(token_or_None, mode)`` where ``mode`` is: - - ``"entra_id"`` when a callable token provider was supplied — the - returned token is a freshly minted bearer JWT, sent ONLY in - ``Authorization: Bearer``. - - ``"api_key"`` when a string key was supplied — the returned token - is the raw API key, sent in BOTH ``api-key:`` and - ``Authorization: Bearer`` headers (preserves the original - broad-compat probe behaviour). - - ``("", "api_key")`` when neither yields a value. - - Bearer minting failures degrade to ``("", "entra_id")`` so the caller - can still report "detection incomplete" rather than crashing. + Returns ``(token_or_None, mode)`` where ``mode`` is: - ``"entra_id"`` when a callable token + provider was supplied — the returned token is a freshly minted bearer JWT, sent ONLY in + ``Authorization: Bearer``. """ # Token-provider path (callable wins when both supplied). - if token_provider is not None and callable(token_provider): - try: - token = token_provider() - return (str(token) if token else None), "entra_id" - except Exception as exc: - logger.debug("azure_detect: token_provider failed: %s", exc) - return None, "entra_id" - if callable(api_key) and not isinstance(api_key, str): - try: - token = api_key() - return (str(token) if token else None), "entra_id" - except Exception as exc: - logger.debug("azure_detect: api_key callable failed: %s", exc) - return None, "entra_id" + for provider, label in ((token_provider, "token_provider"), (api_key, "api_key callable")): + if callable(provider) and not isinstance(provider, str): + try: + token = provider() + return (str(token) if token else None), "entra_id" + except Exception as exc: + logger.debug("azure_detect: %s failed: %s", label, exc) + return None, "entra_id" # API-key path. if isinstance(api_key, str) and api_key: return api_key, "api_key" return None, "api_key" -def _apply_auth_headers(req: urllib_request.Request, - token: Optional[str], - mode: str) -> None: - """Attach the right auth headers to ``req`` based on credential mode.""" - if not token: - return - if mode == "entra_id": - # Bearer-only: do NOT also set api-key, which would log a JWT in - # a header slot intended for static keys. - req.add_header("Authorization", f"Bearer {token}") - else: - # Legacy broad-compat behaviour: send both headers so we land on - # any Azure resource regardless of which it accepts. - req.add_header("api-key", token) +def _authed_request(url: str, api_key: Any, token_provider, *, method: str = "GET", + data: Optional[bytes] = None) -> urllib_request.Request: + """Build a request carrying the right auth headers for the credential mode.""" + token, mode = _resolve_credential(api_key, token_provider) + req = urllib_request.Request(url, method=method, data=data) + if token: + if mode != "entra_id": + # Legacy broad-compat behaviour: send both headers so we land on + # any Azure resource regardless of which it accepts. + req.add_header("api-key", token) + # Bearer-only in entra_id mode: do NOT also set api-key, which would + # log a JWT in a header slot intended for static keys. req.add_header("Authorization", f"Bearer {token}") + req.add_header("User-Agent", "hermes-agent/azure-detect") + return req def _http_get_json(url: str, @@ -155,10 +110,7 @@ def _http_get_json(url: str, ) -> tuple[int, Optional[dict]]: """GET a URL with the appropriate auth headers. Return ``(status_code, parsed_json_or_None)``. Never raises.""" - token, mode = _resolve_credential(api_key, token_provider) - req = urllib_request.Request(url, method="GET") - _apply_auth_headers(req, token, mode) - req.add_header("User-Agent", "hermes-agent/azure-detect") + req = _authed_request(url, api_key, token_provider) try: with open_credentialed_url(req, timeout=timeout) as resp: body = resp.read() @@ -182,9 +134,9 @@ def _strip_trailing_v1(url: str) -> str: def _looks_like_anthropic_path(url: str) -> bool: - """Return True when the URL's path ends in ``/anthropic`` or - contains a ``/anthropic/`` segment. Used by Azure Foundry - resources that route Claude traffic through a dedicated path.""" + """Return True when the URL's path ends in ``/anthropic`` or contains a ``/anthropic/`` segment. + Used by Azure Foundry resources that route Claude traffic through a dedicated path. + """ try: parsed = urlparse(url) path = (parsed.path or "").lower().rstrip("/") @@ -215,20 +167,15 @@ def _probe_openai_models(base_url: str, *, token_provider: Optional[Callable[[], str]] = None, ) -> tuple[bool, list[str]]: - """Probe ``/models`` for an OpenAI-shaped response. - - Returns ``(ok, models)``. ``ok`` is True iff the endpoint accepted - us as an OpenAI-style caller (200 OK + OpenAI-shaped JSON body). - """ + """Probe ``/models`` for an OpenAI-shaped response.""" base_url = base_url.rstrip("/") # Azure OpenAI v1: {resource}.openai.azure.com/openai/v1 — no # api-version required for GA paths, so probe without first. - candidates = [f"{base_url}/models"] # Fallback: explicit api-version for pre-v1 resources - for v in _AZURE_OPENAI_PROBE_API_VERSIONS: - candidates.append(f"{base_url}/models?api-version={v}") - + candidates = [f"{base_url}/models"] + [ + f"{base_url}/models?api-version={v}" for v in _AZURE_OPENAI_PROBE_API_VERSIONS + ] for url in candidates: status, body = _http_get_json(url, api_key, token_provider=token_provider) if status == 200 and body is not None: @@ -251,11 +198,9 @@ def _probe_anthropic_messages(base_url: str, *, token_provider: Optional[Callable[[], str]] = None, ) -> bool: - """Send a zero-token request to ``/v1/messages`` and check - whether the endpoint at least *recognises* the Anthropic Messages - shape (any 4xx that mentions ``messages`` or ``model``, or a 400 - ``invalid_request`` with an Anthropic error shape). Never completes - a real chat. + """Send a zero-token request to ``/v1/messages`` and check whether the endpoint at least + *recognises* the Anthropic Messages shape (any 4xx that mentions ``messages`` or ``model``, or a + 400 ``invalid_request`` with an Anthropic error shape). Never completes a real chat. """ base = _strip_trailing_v1(base_url) url = f"{base}/v1/messages?api-version={_AZURE_ANTHROPIC_API_VERSION}" @@ -264,12 +209,9 @@ def _probe_anthropic_messages(base_url: str, "max_tokens": 1, "messages": [{"role": "user", "content": "ping"}], }).encode("utf-8") - req = urllib_request.Request(url, method="POST", data=payload) - token, mode = _resolve_credential(api_key, token_provider) - _apply_auth_headers(req, token, mode) + req = _authed_request(url, api_key, token_provider, method="POST", data=payload) req.add_header("anthropic-version", "2023-06-01") req.add_header("content-type", "application/json") - req.add_header("User-Agent", "hermes-agent/azure-detect") try: with open_credentialed_url(req, timeout=6.0) as resp: # Should never 200 — "probe" isn't a real deployment. But @@ -303,16 +245,14 @@ def detect(base_url: str, ) -> DetectionResult: """Inspect an Azure endpoint and describe its transport + models. - Call this from the wizard before asking the user to pick an API - mode manually. The caller should treat the returned - :class:`DetectionResult` as *advisory* — if ``api_mode`` is None, - fall back to asking the user. + Call this from the wizard before asking the user to pick an API mode manually. The caller should + treat the returned :class:`DetectionResult` as *advisory* — if ``api_mode`` is None, fall back + to asking the user. - ``api_key`` may be a string (legacy API-key auth — sends both - ``api-key:`` and ``Authorization: Bearer``) or a callable returning - a bearer JWT (Entra ID auth — sends ONLY ``Authorization: Bearer``). - ``token_provider`` is an alternative explicit name for the callable - form; if both are supplied the callable wins. + ``api_key`` may be a string (legacy API-key auth — sends both ``api-key:`` and ``Authorization: + Bearer``) or a callable returning a bearer JWT (Entra ID auth — sends ONLY ``Authorization: + Bearer``). ``token_provider`` is an alternative explicit name for the callable form; if both are + supplied the callable wins. """ result = DetectionResult() @@ -367,16 +307,9 @@ def lookup_context_length(model: str, *, token_provider: Optional[Callable[[], str]] = None, ) -> Optional[int]: - """Thin wrapper around :func:`agent.model_metadata.get_model_context_length` - that returns ``None`` when only the fallback default (128k) would - fire, so the wizard can distinguish "we actually know this" from - "we guessed. - - For Entra-ID mode pass a callable as ``api_key`` (or via - ``token_provider=``); the wrapped resolver expects a string, so we - mint one bearer JWT here for the single lookup. The resolver itself - only reads catalog metadata over HTTP — no SDK client is built — so - the minted token is consumed for at most one /models probe. + """Thin wrapper around :func:`agent.model_metadata.get_model_context_length` that returns ``None`` + when only the fallback default (128k) would fire, so the wizard can distinguish "we actually + know this" from "we guessed. """ model_id = str(model or "").strip() if not model_id: @@ -391,11 +324,10 @@ def lookup_context_length(model: str, # Resolve the credential once. For Entra mode this calls the token # provider; for legacy api_key this is a no-op string pass-through. - token, mode = _resolve_credential(api_key, token_provider) - effective_key = token or "" + token, _mode = _resolve_credential(api_key, token_provider) try: - n = get_model_context_length(model_id, base_url=base_url, api_key=effective_key) + n = get_model_context_length(model_id, base_url=base_url, api_key=token or "") except Exception as exc: logger.debug("azure_detect: context length lookup failed: %s", exc) return None diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index d69faf1c92..02d7439a82 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -1,7 +1,4 @@ -"""Welcome banner, ASCII art, skills summary, and update check for the CLI. - -Pure display functions with no HermesCLI state dependency. -""" +"""Welcome banner, ASCII art, skills summary, and update check for the CLI.""" import json import logging import os @@ -40,13 +37,10 @@ def cprint(text: str): """Print ANSI-colored text through prompt_toolkit's renderer.""" from prompt_toolkit import print_formatted_text as _pt_print from prompt_toolkit.formatted_text import ANSI as _PT_ANSI - try: - _pt_print(_PT_ANSI(text)) - except Exception: - # prompt_toolkit needs a real console. On Windows, a redirected or - # absent stdout (pythonw.exe, CI, `hermes ... > file`) raises - # NoConsoleScreenBufferError from its Win32Output — display helpers - # must never crash the caller over that, so degrade to plain print. + # prompt_toolkit needs a real console. On Windows, a redirected or absent stdout (pythonw.exe, + # CI, `hermes ... > file`) raises NoConsoleScreenBufferError from its Win32Output — display + # helpers must never crash the caller over that, so degrade to plain print. + if _quiet(lambda: _pt_print(_PT_ANSI(text)) or True) is None: print(text) @@ -54,13 +48,23 @@ def cprint(text: str): # Skin-aware color helpers # ========================================================================= +def _quiet(fn, default=None): + """``fn()``, or ``default`` on any exception — for best-effort display inputs.""" + try: + return fn() + except Exception: + return default + + +def _active_skin(): + """The active skin object, or None when the skin engine is unavailable.""" + from hermes_cli.skin_engine import get_active_skin + return get_active_skin() + + def _skin_color(key: str, fallback: str) -> str: """Get a color from the active skin, or return fallback.""" - try: - from hermes_cli.skin_engine import get_active_skin - return get_active_skin().get_color(key, fallback) - except Exception: - return fallback + return _quiet(lambda: _active_skin().get_color(key, fallback), fallback) # ========================================================================= # ASCII Art & Branding # ========================================================================= @@ -91,40 +95,53 @@ HERMES_CADUCEUS = """[#CD7F32]⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⢀⣀⡀⠀⣀⣀ [#B8860B]⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠈⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀⠀[/]""" - # ========================================================================= # Skills scanning # ========================================================================= -_available_skills_cache: Optional[tuple] = None # (result,) once computed +# Per-process caches: ``None`` until computed, then a 1-tuple ``(value,)`` so a computed ``None`` +# is distinguishable from "not yet computed". Reset by assigning ``None`` (tests, ``hermes skills``). +_available_skills_cache: Optional[tuple] = None +_git_banner_state_cache: Optional[tuple] = None +_latest_release_cache: Optional[tuple] = None + + +_UNCACHED = object() # compute() result that must not be memoized + + +def _memo(cache_name: str, compute): + """Return the cached value under module global ``cache_name``, computing (and storing) it once.""" + cached = globals()[cache_name] + if cached is not None: + return cached[0] + value = compute() + if value is not _UNCACHED: + globals()[cache_name] = (value,) + return value def get_available_skills() -> Dict[str, List[str]]: """Return skills grouped by category, filtered by platform and disabled state. - Delegates to ``_find_all_skills()`` from ``tools/skills_tool`` which already - handles platform gating (``platforms:`` frontmatter) and respects the - user's ``skills.disabled`` config list. - - Cached per-process: this feeds only the startup banner, whose snapshot - is taken once anyway, and the underlying skills-tree walk costs ~100ms. - ``prefetch_banner_data()`` uses the cache to pay that walk off-thread. + Cached per-process: this feeds only the startup banner, whose snapshot is taken once anyway, and + the underlying skills-tree walk costs ~100ms. ``prefetch_banner_data()`` uses the cache to pay + that walk off-thread. A failed scan yields ``{}`` and is not cached. """ - global _available_skills_cache - if _available_skills_cache is not None: - return _available_skills_cache[0] - try: + def _scan(): from tools.skills_tool import _find_all_skills - all_skills = _find_all_skills() # already filtered - except Exception: - return {} + return _find_all_skills() # already filtered - skills_by_category: Dict[str, List[str]] = {} - for skill in all_skills: - category = skill.get("category") or "general" - skills_by_category.setdefault(category, []).append(skill["name"]) - _available_skills_cache = (skills_by_category,) - return skills_by_category + def _compute(): + all_skills = _quiet(_scan) + if all_skills is None: + return _UNCACHED + skills_by_category: Dict[str, List[str]] = {} + for skill in all_skills: + skills_by_category.setdefault(skill.get("category") or "general", []).append(skill["name"]) + return skills_by_category + + result = _memo("_available_skills_cache", _compute) + return {} if result is _UNCACHED else result # ========================================================================= @@ -147,62 +164,80 @@ def _canonical_github_remote(url: str | None) -> str: if not url: return "" value = url.strip() - if value.startswith("git@github.com:"): - value = "github.com/" + value[len("git@github.com:"):] - elif value.startswith("ssh://git@github.com/"): - value = "github.com/" + value[len("ssh://git@github.com/"):] + for ssh_prefix in ("git@github.com:", "ssh://git@github.com/"): + if value.startswith(ssh_prefix): + value = "github.com/" + value[len(ssh_prefix):] + break else: parsed = urlparse(value) if parsed.netloc and parsed.path: value = f"{parsed.netloc}{parsed.path}" - value = value.strip().rstrip("/") - if value.endswith(".git"): - value = value[:-4] - return value.lower() - - -def _is_ssh_remote(url: str | None) -> bool: - if not url: - return False - value = url.strip().lower() - return value.startswith("git@") or value.startswith("ssh://") + return value.strip().rstrip("/").removesuffix(".git").lower() def _is_official_ssh_remote(url: str | None) -> bool: - return _is_ssh_remote(url) and _canonical_github_remote(url) == _OFFICIAL_REPO_CANONICAL + if not url or not url.strip().lower().startswith(("git@", "ssh://")): + return False + return _canonical_github_remote(url) == _OFFICIAL_REPO_CANONICAL -def _git_stdout(args: list[str], *, cwd: Path, timeout: int = 5) -> Optional[str]: +_GIT_TEXT_KW = {"text": True, "encoding": "utf-8", "errors": "replace"} + + +def _git_run( + args: list[str], *, cwd: Optional[Path] = None, timeout: int = 5, text: bool = True, network: bool = False +): + """Run ``git `` with the shared subprocess boilerplate; None on any exception. + + git output is UTF-8; on Windows ``text=True`` defaults to the ANSI code page and bytes like 0x90 + (3rd byte of 🐛 in a commit subject) crash the stdlib reader thread (#52649), hence the explicit + encoding. ``network=True`` (ls-remote/fetch) detaches stdin and disables git/GCM prompts so a + passive update check can never hang on a ``Username for 'https://github.com':`` prompt. + """ + kwargs: dict = {} + if network: + from hermes_cli._subprocess_compat import noninteractive_git_env + + kwargs = {"stdin": subprocess.DEVNULL, "env": noninteractive_git_env()} try: - result = subprocess.run( + return subprocess.run( ["git", *args], capture_output=True, - text=True, - # git output is UTF-8; on Windows text=True defaults to the ANSI - # code page and bytes like 0x90 (3rd byte of 🐛 in a commit - # subject) crash the stdlib reader thread (#52649). - encoding="utf-8", - errors="replace", timeout=timeout, - cwd=str(cwd), + cwd=str(cwd) if cwd is not None else None, + **(_GIT_TEXT_KW if text else {}), + **kwargs, ) except Exception: return None - if result.returncode != 0: + + +def _git_stdout(args: list[str], *, cwd: Path, timeout: int = 5) -> Optional[str]: + result = _git_run(args, cwd=cwd, timeout=timeout) + if result is None or result.returncode != 0: return None return (result.stdout or "").strip() +def _git_count(args: list[str], *, cwd: Path) -> Optional[int]: + """``int`` of a successful ``git rev-list --count``-style command, else None. + + Deliberately bypasses ``_git_stdout`` so tests can stub the two layers independently. + """ + result = _git_run(args, cwd=cwd) + try: + if result is not None and result.returncode == 0: + return int(result.stdout.strip()) + except Exception: + pass + return None + + def _github_compare_behind(current_rev: str, target_rev: str) -> Optional[int]: """Exact behind-count via the GitHub compare API for uncountable graphs. - Shallow installer clones and ls-remote-only probes know the two tip SHAs - but have no local history to run ``rev-list --count`` across. GitHub's - ``GET /repos///compare/...`` knows the full - graph regardless of local clone depth and returns ``ahead_by`` — exactly - the behind count the local graph lost. Unauthenticated, bounded, and - best-effort: any failure (offline, rate limit, diverged/unknown SHAs) - returns None so callers keep the honest UPDATE_AVAILABLE_NO_COUNT. + Shallow installer clones and ls-remote-only probes know the two tip SHAs but have no local + history to run ``rev-list --count`` across. """ if not (_is_full_sha(current_rev) and _is_full_sha(target_rev)): return None @@ -210,108 +245,84 @@ def _github_compare_behind(current_rev: str, target_rev: str) -> Optional[int]: "https://api.github.com/repos/nousresearch/hermes-agent/" f"compare/{current_rev}...{target_rev}" ) - try: + def _fetch(): import urllib.request + # api.github.com 403s requests without a User-Agent. req = urllib.request.Request( - url, - headers={ - "Accept": "application/vnd.github+json", - # api.github.com 403s requests without a User-Agent. - "User-Agent": "hermes-cli-update-check", - }, + url, headers={"Accept": "application/vnd.github+json", "User-Agent": "hermes-cli-update-check"}, ) with urllib.request.urlopen(req, timeout=10) as resp: - payload = json.loads(resp.read().decode("utf-8")) - except Exception: - return None + return json.loads(resp.read().decode("utf-8")) + + payload = _quiet(_fetch) ahead = payload.get("ahead_by") if isinstance(payload, dict) else None if isinstance(ahead, int) and not isinstance(ahead, bool) and ahead >= 0: return ahead return None +def _behind_count_or_sentinel(local_rev: str, upstream_rev: str) -> int: + """Exact behind-count via the compare API, else the honest no-count sentinel. + + ``ahead_by == 0`` with differing tips means the remote tip is reachable from our HEAD — a + local-ahead checkout, i.e. NOT behind. A local-only HEAD 404s on the API, which safely degrades + to ``UPDATE_AVAILABLE_NO_COUNT`` — never a fabricated 1. + """ + counted = _github_compare_behind(local_rev, upstream_rev) + return counted if counted is not None else UPDATE_AVAILABLE_NO_COUNT + + +def _tips_behind(head_rev: Optional[str], target_rev: Optional[str], repo_dir: Optional[Path] = None) -> Optional[int]: + """Behind-count from two tip SHAs: None if either is unknown, 0 when equal, else count/sentinel. + + With ``repo_dir``, a target that is already an ancestor of HEAD (local-ahead checkout) is 0 too. + """ + if not head_rev or not target_rev: + return None + if head_rev == target_rev: + return 0 + if repo_dir is not None: + ancestor = _git_run(["merge-base", "--is-ancestor", target_rev, "HEAD"], cwd=repo_dir, text=False) + if ancestor is not None and ancestor.returncode == 0: + return 0 + return _behind_count_or_sentinel(head_rev, target_rev) + + def _is_full_sha(value: Optional[str]) -> bool: - return ( - isinstance(value, str) - and len(value) == 40 - and all(c in "0123456789abcdefABCDEF" for c in value) - ) + return isinstance(value, str) and len(value) == 40 and all(c in "0123456789abcdefABCDEF" for c in value) def _upstream_main_sha() -> Optional[str]: """Tip SHA of upstream main via HTTPS ls-remote (no auth, no prompts).""" - from hermes_cli._subprocess_compat import noninteractive_git_env - - try: - result = subprocess.run( - ["git", "ls-remote", _UPSTREAM_REPO_URL, "refs/heads/main"], - capture_output=True, text=True, encoding="utf-8", errors="replace", - timeout=10, - stdin=subprocess.DEVNULL, - env=noninteractive_git_env(), - ) - except Exception: + result = _git_run(["ls-remote", _UPSTREAM_REPO_URL, "refs/heads/main"], timeout=10, network=True) + if result is None or result.returncode != 0 or not result.stdout: return None - if result.returncode != 0 or not result.stdout: - return None - upstream_rev = result.stdout.split()[0] - return upstream_rev or None + return result.stdout.split()[0] or None def _check_via_rev(local_rev: str) -> Optional[int]: """Compare an embedded git revision to upstream main via ls-remote. - Returns 0 if up-to-date, the exact behind-count when the GitHub compare - API can recover it, ``UPDATE_AVAILABLE_NO_COUNT`` if behind by an unknown - amount, or ``None`` on failure. + Returns 0 if up-to-date, the exact behind-count when the GitHub compare API can recover it, + ``UPDATE_AVAILABLE_NO_COUNT`` if behind by an unknown amount, or ``None`` on failure. """ - upstream_rev = _upstream_main_sha() - if not upstream_rev: - return None - if upstream_rev == local_rev: - return 0 - # Behind, but ls-remote only knows tip SHAs. Try to recover the exact - # count from the GitHub compare API before falling back to the sentinel. - # ahead_by == 0 with differing tips means the remote tip is reachable from - # our HEAD — a local-ahead checkout, i.e. NOT behind. - counted = _github_compare_behind(local_rev, upstream_rev) - return counted if counted is not None else UPDATE_AVAILABLE_NO_COUNT + return _tips_behind(local_rev, _upstream_main_sha()) def _check_via_local_git(repo_dir: Path) -> Optional[int]: """Count commits behind origin/main in a local checkout.""" - from hermes_cli._subprocess_compat import noninteractive_git_env - origin_url = _git_stdout(["remote", "get-url", "origin"], cwd=repo_dir) if _is_official_ssh_remote(origin_url): head_rev = _git_stdout(["rev-parse", "HEAD"], cwd=repo_dir) if not head_rev: return None - # Passive probe via HTTPS ls-remote (never SSH — no hardware-key - # prompts). Tip SHAs alone can't distinguish "behind" from a local - # carried commit sitting AHEAD of origin/main, and misreporting an - # ahead checkout as behind nudges the user into `hermes update`, - # which can wipe their carried work. - upstream_rev = _upstream_main_sha() - if upstream_rev is None: - return None - if upstream_rev == head_rev: - return 0 - # Local-ahead: the remote tip is an ancestor of HEAD. Checked against - # the FRESH upstream SHA (not the possibly stale origin/main tracking - # ref) so a stale ref can't fake an up-to-date report. - ancestor = subprocess.run( - ["git", "merge-base", "--is-ancestor", upstream_rev, "HEAD"], - capture_output=True, timeout=5, cwd=str(repo_dir), - ) - if ancestor.returncode == 0: - return 0 - # Genuinely behind (or diverged). Recover the exact count via the - # GitHub compare API; a local-only HEAD 404s there, which safely - # degrades to the honest no-count sentinel — never a fabricated 1. - counted = _github_compare_behind(head_rev, upstream_rev) - return counted if counted is not None else UPDATE_AVAILABLE_NO_COUNT + # Passive probe via HTTPS ls-remote (never SSH — no hardware-key prompts). Tip SHAs alone + # can't distinguish "behind" from a local carried commit sitting AHEAD of origin/main, and + # misreporting an ahead checkout as behind nudges the user into `hermes update`, which can + # wipe their carried work — hence the ancestor check, against the FRESH upstream SHA (not + # the possibly stale origin/main tracking ref) so a stale ref can't fake an up-to-date report. + return _tips_behind(head_rev, _upstream_main_sha(), repo_dir) # Installer checkouts are shallow (`git clone --depth 1`). On a shallow # clone the history stops at a single commit, so a plain `git fetch` would @@ -321,45 +332,31 @@ def _check_via_local_git(repo_dir: Path) -> Optional[int]: # --depth 1 to preserve the boundary and compare tip SHAs instead of # counting. Full clones (developers, Docker dev images) keep the exact # count path unchanged. Mirrors the desktop fix in apps/desktop/electron/main.cjs. - shallow = _git_stdout(["rev-parse", "--is-shallow-repository"], cwd=repo_dir) - is_shallow = shallow == "true" + is_shallow = _git_stdout(["rev-parse", "--is-shallow-repository"], cwd=repo_dir) == "true" - try: - # Self-heal abandoned git lock files before fetching. A stale - # .git/shallow.lock from a crashed fetch makes the fetch fail, the - # exception below is swallowed, and stale refs get compared against - # HEAD — silently degrading the passive check until a human removes - # the lock (git never self-heals these). + def _fetch() -> bool: + # Self-heal abandoned git lock files before fetching. A stale .git/shallow.lock from a + # crashed fetch makes the fetch fail, the exception is swallowed, and stale refs get + # compared against HEAD — silently degrading the passive check until a human removes the + # lock (git never self-heals these). The passive check is also the main tmp_pack GENERATOR + # on flaky lines (several aborted fetches per day), so it must be the janitor too, or debris + # accumulates unbounded between manual updates (#93732). from hermes_cli.gitlock import clear_stale_git_locks, clear_stale_tmp_packs clear_stale_git_locks(repo_dir) - # The passive check is the main tmp_pack GENERATOR on flaky lines - # (several aborted fetches per day) — it must also be the janitor, - # or debris accumulates unbounded between manual updates (#93732). clear_stale_tmp_packs(repo_dir) - # Scope the fetch to the one branch the behind-count compares against. - # An unscoped ``git fetch origin`` transfers every remote head (~1,400 - # on this repo — measured 3.0 s vs 0.55 s scoped) and can burn the full - # 10 s timeout on slow links. ``cmd_update`` already scopes its fetch - # for the same reason. Modern git updates the ``origin/main`` tracking - # ref on a scoped fetch, so the ``HEAD..origin/main`` count below is - # unaffected; the shallow path compares against FETCH_HEAD, which a - # scoped fetch also updates. - fetch_args = ["git", "fetch", "origin", "main"] - if is_shallow: - fetch_args += ["--depth", "1"] - fetch_args.append("--quiet") - fetch_proc = subprocess.run( - fetch_args, - capture_output=True, timeout=10, - cwd=str(repo_dir), - stdin=subprocess.DEVNULL, - env=noninteractive_git_env(), - ) - fetch_ok = fetch_proc.returncode == 0 - except Exception: - fetch_ok = False # Offline or timeout — don't use stale refs + # Scope the fetch to the one branch the behind-count compares against. An unscoped + # ``git fetch origin`` transfers every remote head (~1,400 on this repo — measured 3.0 s vs + # 0.55 s scoped) and can burn the full 10 s timeout on slow links. Modern git updates the + # ``origin/main`` tracking ref on a scoped fetch, so the ``HEAD..origin/main`` count below + # is unaffected; the shallow path compares against FETCH_HEAD, which a scoped fetch also + # updates. ``--depth 1`` preserves the shallow boundary. + fetch_args = ["fetch", "origin", "main", *(["--depth", "1"] if is_shallow else []), "--quiet"] + fetch_proc = _git_run(fetch_args, cwd=repo_dir, timeout=10, text=False, network=True) + return fetch_proc is not None and fetch_proc.returncode == 0 + + fetch_ok = _quiet(_fetch, False) # Offline or timeout — don't use stale refs # When the fetch fails, the local origin/main tracking ref is stale. It # cannot prove *currentness* (a 0 behind-count may just mean the stale ref @@ -367,67 +364,37 @@ def _check_via_local_git(repo_dir: Path) -> Optional[int]: # evidence an update exists — the ref was good at some point in the past. # Return the positive stale count; return None (inconclusive) otherwise so # the caller doesn't cache a false "up to date". (#82166, review #92578) - if not fetch_ok: - if not is_shallow: - try: - result = subprocess.run( - ["git", "rev-list", "--count", "HEAD..origin/main"], - capture_output=True, text=True, encoding="utf-8", errors="replace", - timeout=5, - cwd=str(repo_dir), - ) - if result.returncode == 0: - behind = int(result.stdout.strip()) - if behind > 0: - return behind - except Exception: - pass - return None - if is_shallow: - # No history to count across the shallow boundary. `origin/main` may not - # be a tracking ref in a `clone --depth 1`, so prefer FETCH_HEAD (just - # updated by the fetch above) and fall back to origin/main. + if not fetch_ok: + return None + # No history to count across the shallow boundary. `origin/main` may not be a tracking ref + # in a `clone --depth 1`, so prefer FETCH_HEAD (just updated by the fetch above) and fall + # back to origin/main. Tips differ but the shallow boundary hides the history between them. head_rev = _git_stdout(["rev-parse", "HEAD"], cwd=repo_dir) target_rev = ( _git_stdout(["rev-parse", "FETCH_HEAD"], cwd=repo_dir) or _git_stdout(["rev-parse", "origin/main"], cwd=repo_dir) ) - if not head_rev or not target_rev: - return None - if head_rev == target_rev: - return 0 - # Tips differ but the shallow boundary hides the history between them. - # Recover the exact count from the GitHub compare API when possible - # (ahead_by == 0 means local-ahead ⇒ up to date); otherwise report the - # honest "update available, count unknown" sentinel. - counted = _github_compare_behind(head_rev, target_rev) - return counted if counted is not None else UPDATE_AVAILABLE_NO_COUNT + return _tips_behind(head_rev, target_rev) - try: - result = subprocess.run( - ["git", "rev-list", "--count", "HEAD..origin/main"], - capture_output=True, text=True, encoding="utf-8", errors="replace", - timeout=5, - cwd=str(repo_dir), - ) - if result.returncode == 0: - return int(result.stdout.strip()) - except Exception: - pass + behind = _git_count(["rev-list", "--count", "HEAD..origin/main"], cwd=repo_dir) + if fetch_ok or (behind is not None and behind > 0): + return behind return None +def _read_json(path: Path) -> Optional[dict]: + """Parse ``path`` as a JSON object; None when missing, unreadable, or not a dict.""" + blob = _quiet(lambda: json.loads(path.read_text(encoding="utf-8"))) + return blob if isinstance(blob, dict) else None + + def check_for_updates() -> Optional[int]: """Check whether a Hermes update is available. - Two paths: if ``HERMES_REVISION`` is set (nix builds embed it), compare - it to upstream main via ``git ls-remote``. Otherwise look for a local - git checkout and count commits behind ``origin/main``. - - Returns the number of commits behind, ``UPDATE_AVAILABLE_NO_COUNT`` (-1) - if behind but the count is unknown, ``0`` if up-to-date, or ``None`` if - the check failed or doesn't apply. Cached for 6 hours. + Two paths: if ``HERMES_REVISION`` is set (nix builds embed it), compare it to upstream main via + ``git ls-remote``. Otherwise look for a local git checkout and count commits behind + ``origin/main``. """ hermes_home = get_hermes_home() cache_file = hermes_home / ".update_check" @@ -440,58 +407,43 @@ def check_for_updates() -> Optional[int]: # on `typeof === 'number' && > 0`) show nothing. The dashboard's REST # `/api/hermes/update/check` endpoint short-circuits docker the same way # (web_server.py); mirror that here so the banner/TUI surfaces agree. - try: + def _install_method(): from hermes_cli.config import detect_install_method, get_project_root - if detect_install_method(get_project_root()) in {"docker", "apt"}: - return None - except Exception: - pass + return detect_install_method(get_project_root()) + + if _quiet(_install_method) in {"docker", "apt"}: + return None # Read cache — invalidate if the embedded rev OR installed version has # changed since the last check. now = time.time() - try: - if cache_file.exists(): - cached = json.loads(cache_file.read_text(encoding="utf-8")) - if ( - now - cached.get("ts", 0) < _UPDATE_CHECK_CACHE_SECONDS - and cached.get("rev") == embedded_rev - and cached.get("ver") == VERSION - ): - return cached.get("behind") - except Exception: - pass + cached = _read_json(cache_file) + if ( + cached is not None + and now - cached.get("ts", 0) < _UPDATE_CHECK_CACHE_SECONDS + and cached.get("rev") == embedded_rev + and cached.get("ver") == VERSION + ): + return cached.get("behind") if embedded_rev: behind = _check_via_rev(embedded_rev) else: - # Prefer the running code's location over the profile-scoped path. - # $HERMES_HOME/hermes-agent/ may be a stale copy from --clone-all; - # Path(__file__) always resolves to the actual installed checkout. - repo_dir = Path(__file__).parent.parent.resolve() - if not (repo_dir / ".git").exists(): - repo_dir = hermes_home / "hermes-agent" - if not (repo_dir / ".git").exists(): - # No git checkout and no embedded revision — can't determine - # update status. This is the Docker path (already short-circuited - # above) or an unsupported install without a source tree. - behind = None - else: - behind = _check_via_local_git(repo_dir) + # No git checkout and no embedded revision — can't determine update + # status (the Docker path, already short-circuited above, or an + # unsupported install without a source tree). + repo_dir = _resolve_repo_dir() + behind = _check_via_local_git(repo_dir) if repo_dir is not None else None - try: - # Don't cache inconclusive results (None). A None means the check - # could not run — typically a failed git fetch. Caching None would - # suppress retries for the full 6-hour cache window, leaving the - # user with a stale "up to date" or no information for hours after - # connectivity is restored (#82166). - if behind is not None: - cache_file.write_text( - json.dumps({"ts": now, "behind": behind, "rev": embedded_rev, "ver": VERSION}), - encoding="utf-8", - ) - except Exception: - pass + # Don't cache inconclusive results (None). A None means the check could not run — typically a + # failed git fetch. Caching None would suppress retries for the full 6-hour cache window, + # leaving the user with a stale "up to date" or no information for hours after connectivity + # is restored (#82166). + if behind is not None: + _quiet(lambda: cache_file.write_text( + json.dumps({"ts": now, "behind": behind, "rev": embedded_rev, "ver": VERSION}), + encoding="utf-8", + )) return behind @@ -499,9 +451,8 @@ def check_for_updates() -> Optional[int]: def _resolve_repo_dir() -> Optional[Path]: """Return the active Hermes git checkout, or None if this isn't a git install. - Prefers the running code's location over the profile-scoped path - because ``$HERMES_HOME/hermes-agent/`` may be a stale copy carried - over by ``--clone-all``. + Prefers the running code's location over the profile-scoped path because ``$HERMES_HOME/hermes- + agent/`` may be a stale copy carried over by ``--clone-all``. """ repo_dir = Path(__file__).parent.parent.resolve() if not (repo_dir / ".git").exists(): @@ -510,149 +461,59 @@ def _resolve_repo_dir() -> Optional[Path]: return repo_dir if (repo_dir / ".git").exists() else None -def _git_short_hash(repo_dir: Path, rev: str) -> Optional[str]: - """Resolve a git revision to an 8-character short hash.""" - try: - result = subprocess.run( - ["git", "rev-parse", "--short=8", rev], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - timeout=5, - cwd=str(repo_dir), - ) - except Exception: - return None - if result.returncode != 0: - return None - value = (result.stdout or "").strip() - return value or None - - -_git_banner_state_cache: Optional[tuple] = None # (state_or_None,) once computed - - def get_git_banner_state(repo_dir: Optional[Path] = None) -> Optional[dict]: """Return upstream/local git hashes for the startup banner. - For source installs and dev images this runs ``git rev-parse`` against - the active checkout. When no checkout is available — the canonical case - is the published Docker image, which excludes ``.git`` from the build - context — we fall back to the baked-in build SHA (see - ``hermes_cli/build_info.py``) and return it as a frozen - ``upstream == local`` state with ``ahead=0``. A built image is by - definition pinned to one commit, so "ahead" is always zero and the - banner correctly shows ``· upstream `` with no carried-commits - annotation. - - Cached per-process (default ``repo_dir`` only): the state costs 2-3 git - subprocesses (~100ms) and the checkout revision cannot change under a - running CLI in a way the banner needs to observe live. The cache also - lets ``prefetch_banner_data()`` pay this cost off-thread before the - banner renders. + Cached per-process (default ``repo_dir`` only): the state costs 2-3 git subprocesses (~100ms) + and the checkout revision cannot change under a running CLI in a way the banner needs to observe + live. The cache also lets ``prefetch_banner_data()`` pay this cost off-thread before the banner + renders. """ - global _git_banner_state_cache - if repo_dir is None and _git_banner_state_cache is not None: - return _git_banner_state_cache[0] - state = _compute_git_banner_state(repo_dir) - if repo_dir is None: - _git_banner_state_cache = (state,) - return state + if repo_dir is not None: + return _compute_git_banner_state(repo_dir) + return _memo("_git_banner_state_cache", _compute_git_banner_state) + + +def _baked_banner_state() -> Optional[dict]: + """Banner state from the baked build SHA (Docker image path), or None.""" + def _baked(): + from hermes_cli.build_info import get_build_sha + return get_build_sha(short=8) + + baked = _quiet(_baked) + return {"upstream": baked, "local": baked, "ahead": 0} if baked else None def _compute_git_banner_state(repo_dir: Optional[Path] = None) -> Optional[dict]: repo_dir = repo_dir or _resolve_repo_dir() if repo_dir is None: - # No git checkout — try the baked build SHA (Docker image path). - try: - from hermes_cli.build_info import get_build_sha - baked = get_build_sha(short=8) - if baked: - return {"upstream": baked, "local": baked, "ahead": 0} - except Exception: - pass - return None + return _baked_banner_state() - upstream = _git_short_hash(repo_dir, "origin/main") - local = _git_short_hash(repo_dir, "HEAD") + upstream, local = (_git_stdout(["rev-parse", "--short=8", rev], cwd=repo_dir) for rev in ("origin/main", "HEAD")) if not upstream or not local: # Live-git lookup failed (e.g. shallow clone without origin/main). - # Fall back to the baked build SHA if available. - try: - from hermes_cli.build_info import get_build_sha - baked = get_build_sha(short=8) - if baked: - return {"upstream": baked, "local": baked, "ahead": 0} - except Exception: - pass - return None - - ahead = 0 - try: - result = subprocess.run( - ["git", "rev-list", "--count", "origin/main..HEAD"], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - timeout=5, - cwd=str(repo_dir), - ) - if result.returncode == 0: - ahead = int((result.stdout or "0").strip() or "0") - except Exception: - ahead = 0 + return _baked_banner_state() + ahead = _git_count(["rev-list", "--count", "origin/main..HEAD"], cwd=repo_dir) or 0 return {"upstream": upstream, "local": local, "ahead": max(ahead, 0)} _RELEASE_URL_BASE = "https://github.com/NousResearch/hermes-agent/releases/tag" -_latest_release_cache: Optional[tuple] = None # (tag, url) once resolved def get_latest_release_tag(repo_dir: Optional[Path] = None) -> Optional[tuple]: """Return ``(tag, release_url)`` for the latest git tag, or None. - Local-only — runs ``git describe --tags --abbrev=0`` against the - Hermes checkout. Cached per-process. Release URL always points at the - canonical NousResearch/hermes-agent repo (forks don't get a link). + Local-only — runs ``git describe --tags --abbrev=0`` against the Hermes checkout. Cached per- + process (a miss is cached too). Release URL always points at the canonical + NousResearch/hermes-agent repo (forks don't get a link). """ - global _latest_release_cache - if _latest_release_cache is not None: - return _latest_release_cache or None + def _compute(): + rd = repo_dir or _resolve_repo_dir() + tag = _git_stdout(["describe", "--tags", "--abbrev=0"], cwd=rd, timeout=3) if rd else None + return (tag, f"{_RELEASE_URL_BASE}/{tag}") if tag else None - repo_dir = repo_dir or _resolve_repo_dir() - if repo_dir is None: - _latest_release_cache = () # falsy sentinel — skip future lookups - return None - - try: - result = subprocess.run( - ["git", "describe", "--tags", "--abbrev=0"], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - timeout=3, - cwd=str(repo_dir), - ) - except Exception: - _latest_release_cache = () - return None - - if result.returncode != 0: - _latest_release_cache = () - return None - - tag = (result.stdout or "").strip() - if not tag: - _latest_release_cache = () - return None - - url = f"{_RELEASE_URL_BASE}/{tag}" - _latest_release_cache = (tag, url) - return _latest_release_cache + return _memo("_latest_release_cache", _compute) def format_banner_version_label() -> str: @@ -662,15 +523,11 @@ def format_banner_version_label() -> str: if not state: return base - upstream = state["upstream"] - local = state["local"] + upstream, local = state["upstream"], state["local"] ahead = int(state.get("ahead") or 0) - if ahead <= 0 or upstream == local: return f"{base} · upstream {upstream}" - - carried_word = "commit" if ahead == 1 else "commits" - return f"{base} · upstream {upstream} · local {local} (+{ahead} carried {carried_word})" + return f"{base} · upstream {upstream} · local {local} (+{ahead} carried {_plural(ahead, 'commit')})" # ========================================================================= @@ -681,14 +538,18 @@ _update_result: Optional[int] = None _update_check_done = threading.Event() +def _daemon(name: Optional[str], target) -> None: + """Start a daemon thread running ``target`` with any exception swallowed.""" + threading.Thread(target=lambda: _quiet(target), name=name, daemon=True).start() + + def prefetch_update_check(): """Kick off update check in a background daemon thread.""" def _run(): global _update_result _update_result = check_for_updates() _update_check_done.set() - t = threading.Thread(target=_run, daemon=True) - t.start() + _daemon(None, _run) _banner_data_prefetch_started = False @@ -697,13 +558,10 @@ _banner_data_prefetch_started = False def prefetch_banner_data(): """Warm the banner's subprocess/I/O-heavy inputs in a daemon thread. - ``build_welcome_banner`` needs git state (2-4 ``git rev-parse``/ - ``describe`` subprocesses, ~130ms) and the skills index (a skills-tree - rglob, ~110ms). Both are cached per-process by their own modules, so - warming them here while the main thread pays the CPU-bound ``cli`` / - prompt_toolkit imports overlaps subprocess waits and file I/O (which - release the GIL) with import work. Idempotent; failures are irrelevant - because the banner recomputes anything missing. + Git state (~130ms of subprocesses) and the skills index (~110ms rglob) are cached per-process + by their own modules, so warming them while the main thread pays the CPU-bound ``cli`` / + prompt_toolkit imports overlaps GIL-releasing I/O with import work. Idempotent; failures + don't matter because the banner recomputes anything missing. """ global _banner_data_prefetch_started if _banner_data_prefetch_started: @@ -711,20 +569,10 @@ def prefetch_banner_data(): _banner_data_prefetch_started = True def _run() -> None: - try: - get_git_banner_state() - except Exception: - pass - try: - get_latest_release_tag() - except Exception: - pass - try: - get_available_skills() - except Exception: - pass + for warm in (get_git_banner_state, get_latest_release_tag, get_available_skills): + _quiet(warm) - threading.Thread(target=_run, name="banner-data-prefetch", daemon=True).start() + _daemon("banner-data-prefetch", _run) def get_update_result(timeout: float = 0.5) -> Optional[int]: @@ -737,19 +585,16 @@ def _format_update_notice(behind: int) -> str: """Render the update warning line for a non-zero ``behind`` result.""" from hermes_cli.config import get_managed_update_command, recommended_update_command if behind > 0: - commits_word = "commit" if behind == 1 else "commits" return ( - f"[bold yellow]⚠ {behind} {commits_word} behind[/]" + f"[bold yellow]⚠ {behind} {_plural(behind, 'commit')} behind[/]" f"[dim yellow] — run [bold]{recommended_update_command()}[/bold] to update[/]" ) # UPDATE_AVAILABLE_NO_COUNT: nix-built hermes; we know an update # exists but not by how much, and we don't know how the user # installed it (nix run, profile, system flake, home-manager). managed_cmd = get_managed_update_command() - line = "[bold yellow]⚠ update available[/]" - if managed_cmd: - line += f"[dim yellow] — run [bold]{managed_cmd}[/bold][/]" - return line + suffix = f"[dim yellow] — run [bold]{managed_cmd}[/bold][/]" if managed_cmd else "" + return f"[bold yellow]⚠ update available[/]{suffix}" _deferred_update_notice_started = False @@ -758,8 +603,8 @@ _deferred_update_notice_started = False def _defer_update_notice(console: "Console", max_wait: float = 30.0) -> None: """Print the update warning once the prefetched check completes. - Used when the banner rendered before the update prefetch finished so - startup never blocks on git/network. Prints at most once per process. + Used when the banner rendered before the update prefetch finished so startup never blocks on + git/network. Prints at most once per process. """ global _deferred_update_notice_started if _deferred_update_notice_started: @@ -767,39 +612,27 @@ def _defer_update_notice(console: "Console", max_wait: float = 30.0) -> None: _deferred_update_notice_started = True def _wait_and_print() -> None: - try: - if not _update_check_done.wait(timeout=max_wait): - return - behind = _update_result - if behind is None or behind == 0: - return - console.print(_format_update_notice(behind)) - except Exception: - pass # never break the session over an update notice + if _update_check_done.wait(timeout=max_wait) and _update_result: + console.print(_format_update_notice(_update_result)) - threading.Thread( - target=_wait_and_print, name="update-notice", daemon=True - ).start() + _daemon("update-notice", _wait_and_print) # never break the session over an update notice # ========================================================================= # Welcome banner # ========================================================================= +def _plural(n: int, word: str) -> str: + return word if n == 1 else f"{word}s" + + def _format_context_length(tokens: int) -> str: """Format a token count for display (e.g. 128000 → '128K', 1048576 → '1M').""" - if tokens >= 1_000_000: - val = tokens / 1_000_000 - rounded = round(val) - if abs(val - rounded) < 0.05: - return f"{rounded}M" - return f"{val:.1f}M" - elif tokens >= 1_000: - val = tokens / 1_000 - rounded = round(val) - if abs(val - rounded) < 0.05: - return f"{rounded}K" - return f"{val:.1f}K" + for unit, div in (("M", 1_000_000), ("K", 1_000)): + if tokens >= div: + val = tokens / div + rounded = round(val) + return f"{rounded}{unit}" if abs(val - rounded) < 0.05 else f"{val:.1f}{unit}" return str(tokens) @@ -807,11 +640,12 @@ def _display_toolset_name(toolset_name: str) -> str: """Normalize internal/legacy toolset identifiers for banner display.""" if not toolset_name: return "unknown" - return ( - toolset_name[:-6] - if toolset_name.endswith("_tools") - else toolset_name - ) + return toolset_name[:-6] if toolset_name.endswith("_tools") else toolset_name + + +def _short_label(name: str) -> str: + """Truncate a model/preset slug to fit the banner's left column.""" + return name[:25] + "..." if len(name) > 28 else name # ========================================================================= @@ -838,17 +672,18 @@ def _banner_snapshot_path() -> Path: def banner_snapshot_fingerprint() -> Optional[str]: """Fingerprint the inputs the banner tool panel depends on.""" import hashlib - parts = [f"v{_BANNER_SNAPSHOT_VERSION}"] - try: + + def _inputs(): from hermes_cli.config import get_config_path - for p in (get_config_path(), get_hermes_home() / ".env"): - try: - st = p.stat() - parts.append(f"{p.name}:{st.st_mtime_ns}:{st.st_size}") - except OSError: - parts.append(f"{p.name}:absent") - except Exception: + return (get_config_path(), get_hermes_home() / ".env") + + paths = _quiet(_inputs) + if paths is None: return None + parts = [f"v{_BANNER_SNAPSHOT_VERSION}"] + for p in paths: + st = _quiet(p.stat) + parts.append(f"{p.name}:{st.st_mtime_ns}:{st.st_size}" if st else f"{p.name}:absent") # Code checkout: version + git HEAD when available (post-update change). parts.append(str(VERSION)) state = get_git_banner_state() @@ -859,24 +694,17 @@ def banner_snapshot_fingerprint() -> Optional[str]: def load_banner_snapshot(enabled_toolsets: List[str] = None) -> Optional[Dict[str, Any]]: """Return the stored banner snapshot when its fingerprint is current.""" - try: - blob = json.loads(_banner_snapshot_path().read_text(encoding="utf-8")) - except Exception: - return None - if not isinstance(blob, dict): + blob = _read_json(_banner_snapshot_path()) + if blob is None: return None fp = banner_snapshot_fingerprint() if not fp or blob.get("fingerprint") != fp: return None if blob.get("enabled_toolsets") != sorted(enabled_toolsets or []): return None - tools = blob.get("tools") - toolset_map = blob.get("toolset_map") - availability = blob.get("availability") - if not isinstance(tools, list) or not isinstance(toolset_map, dict) \ - or not isinstance(availability, dict): - return None - if not isinstance(blob.get("skills_by_category"), dict): + if not isinstance(blob.get("tools"), list) or not all( + isinstance(blob.get(k), dict) for k in ("toolset_map", "availability", "skills_by_category") + ): return None return blob @@ -902,32 +730,27 @@ def save_banner_snapshot( "toolset_map": toolset_map, "availability": { "unavailable_toolsets": availability.get("unavailable_toolsets", []), - "lazy_tools": list(availability.get("lazy_tools", [])), - "disabled_tools": list(availability.get("disabled_tools", [])), + **{k: list(availability.get(k, [])) for k in ("lazy_tools", "disabled_tools")}, }, "skills_by_category": get_available_skills(), } - path = _banner_snapshot_path() - try: - import os as _os - import tempfile as _tempfile + def _write(): + import tempfile + path = _banner_snapshot_path() path.parent.mkdir(parents=True, exist_ok=True) - fd, tmp = _tempfile.mkstemp(dir=str(path.parent), prefix=".banner_snap.") - with _os.fdopen(fd, "w", encoding="utf-8") as fh: + fd, tmp = tempfile.mkstemp(dir=str(path.parent), prefix=".banner_snap.") + with os.fdopen(fd, "w", encoding="utf-8") as fh: json.dump(payload, fh) - _os.replace(tmp, path) - except Exception: - pass + os.replace(tmp, path) + + _quiet(_write) def compute_toolset_availability(enabled_toolsets: List[str] = None) -> Dict[str, Any]: """Compute the banner's toolset-availability payload. - Returns ``{"unavailable_toolsets": [...], "lazy_tools": [...], - "disabled_tools": [...]}`` — the exact inputs ``build_welcome_banner`` - needs to annotate disabled/lazy tools. Split out so the result can be - snapshotted to disk and replayed on the next launch without importing - ``model_tools`` (see ``load_banner_snapshot``). + Returns ``{"unavailable_toolsets", "lazy_tools", "disabled_tools"}``. Split out so the result + can be snapshotted and replayed on the next launch without importing ``model_tools``. """ from model_tools import check_tool_availability, TOOLSET_REQUIREMENTS @@ -945,19 +768,12 @@ def compute_toolset_availability(enabled_toolsets: List[str] = None) -> Dict[str item for item in unavailable_toolsets if str(item.get("id", item.get("name", ""))) in _enabled_ts ] - disabled_tools = set() - # Tools whose toolset has a check_fn are lazy-initialized (e.g. honcho, - # homeassistant) — they show as unavailable at banner time because the - # check hasn't run yet, but they aren't misconfigured. - lazy_tools = set() + # Tools whose toolset has a check_fn are lazy-initialized (e.g. honcho, homeassistant) — they + # show as unavailable at banner time because the check hasn't run yet, but aren't misconfigured. + lazy_tools, disabled_tools = set(), set() for item in unavailable_toolsets: - toolset_name = item.get("name", "") - ts_req = TOOLSET_REQUIREMENTS.get(toolset_name, {}) - tools_in_ts = item.get("tools", []) - if ts_req.get("check_fn"): - lazy_tools.update(tools_in_ts) - else: - disabled_tools.update(tools_in_ts) + is_lazy = TOOLSET_REQUIREMENTS.get(item.get("name", ""), {}).get("check_fn") + (lazy_tools if is_lazy else disabled_tools).update(item.get("tools", [])) return { "unavailable_toolsets": unavailable_toolsets, "lazy_tools": sorted(lazy_tools), @@ -965,6 +781,99 @@ def compute_toolset_availability(enabled_toolsets: List[str] = None) -> Dict[str } +def _mcp_server_line(srv: dict, *, dim: str, text: str) -> str: + """One banner line for an MCP server status entry.""" + name, transport = srv["name"], srv["transport"] + if srv["connected"]: + return f"[dim {dim}]{name}[/] [{text}]({transport})[/] [dim {dim}]—[/] [{text}]{srv['tools']} tool(s)[/]" + status = "disabled" if srv.get("disabled") else srv.get("status") + suffix = { + "disabled": f"[dim {dim}]— disabled[/]", + "connecting": "[yellow]— connecting[/]", + "configured": f"[dim {dim}]— configured[/]", + }.get(status) + if suffix is not None: + return f"[dim {dim}]{name}[/] [dim]({transport})[/] {suffix}" + return f"[red]{name}[/] [dim]({transport})[/] [red]— failed[/]" + + +def _truncate_tool_names(tool_names: List[str]) -> List[Optional[str]]: + """Cut a toolset's tool list to ~42 columns; ``None`` marks the elided tail.""" + if len(", ".join(tool_names)) <= 45: + return list(tool_names) + short_names: List[Optional[str]] = [] + length = 0 + for name in tool_names: + if length + len(name) + 2 > 42: + short_names.append(None) + break + short_names.append(name) + length += len(name) + 2 + return short_names + + +def _pack_skill_names(skill_names: List[str], avail: int) -> str: + """Join skill names into ``avail`` columns, ending with ``+N more`` when they don't all fit.""" + parts: List[str] = [] + length = 0 + for i, name in enumerate(skill_names): + needed = (2 if parts else 0) + len(name) + # Indicator size IF we were to add this skill then stop. + after = len(skill_names) - (i + 1) + ind_len = len(f", +{after} more") if after > 0 else 0 + if parts and length + needed + ind_len > avail: + parts.append(f"+{len(skill_names) - len(parts)} more") + break + parts.append(name) + length += needed + return ", ".join(parts) + + +def _moa_aggregator_label(preset_name: str) -> str: + """Short aggregator-model label for a MoA preset ("" when the preset has none).""" + from hermes_cli.config import load_config + from hermes_cli.moa_config import normalize_moa_config + + preset = normalize_moa_config(load_config().get("moa") or {}).get("presets", {}).get(preset_name) + model = str(((preset or {}).get("aggregator") or {}).get("model") or "") + return model.split("/")[-1] + + +def _mcp_configured() -> bool: + """Cheap probe: does config.yaml or the persisted plugin key cache name any MCP server? + + The full ``get_mcp_status()`` path resolves portable plugin MCP servers, which JOINS the in-flight + background plugin discovery (~100ms on the startup path), so skip it when nothing is configured. + When either probe can't tell, take the full path. + """ + def _native(): + from hermes_cli.config import load_config + return bool((load_config() or {}).get("mcp_servers")) + + def _portable(): + from hermes_cli.plugins import get_portable_mcp_server_names_nowait + return bool(get_portable_mcp_server_names_nowait()) + + return _quiet(_native, True) or _quiet(_portable, True) + + +def _probe_mcp_status() -> list: + from tools.mcp_tool import get_mcp_status + return get_mcp_status() + + +def _codex_runtime_active() -> bool: + """True when the codex_app_server runtime is active (tool counts then live inside codex).""" + from hermes_cli.codex_runtime_switch import get_current_runtime + from hermes_cli.config import load_config + return get_current_runtime(load_config()) == "codex_app_server" + + +def _active_profile_name() -> Optional[str]: + from hermes_cli.profiles import get_active_profile_name + return get_active_profile_name() + + def build_welcome_banner(console: "Console", model: str, cwd: str, tools: List[dict] = None, enabled_toolsets: List[str] = None, @@ -976,22 +885,9 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, skills_by_category: Dict[str, List[str]] = None): """Build and print a welcome banner with caduceus on left and info on right. - Args: - console: Rich Console instance. - model: Current model name. - cwd: Current working directory. - tools: List of tool definitions. - enabled_toolsets: List of enabled toolset names. - session_id: Session identifier. - get_toolset_for_tool: Callable to map tool name -> toolset name. - context_length: Model's context window size in tokens. - provider: Active provider id. When ``"moa"``, ``model`` is a MoA - preset name and the banner renders the aggregator instead of a - bare model slug. - availability: Optional precomputed result of - ``compute_toolset_availability`` (e.g. replayed from the banner - snapshot). When provided together with ``get_toolset_for_tool``, - this function performs no ``model_tools`` import at all. + When ``provider == "moa"``, ``model`` is a MoA preset name and the aggregator is rendered. + Passing a precomputed ``availability`` together with ``get_toolset_for_tool`` avoids any + ``model_tools`` import (banner snapshot replay). """ from rich.panel import Panel from rich.table import Table @@ -1016,62 +912,39 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, accent = _skin_color("banner_accent", "#FFBF00") dim = _skin_color("banner_dim", "#B8860B") text = _skin_color("banner_text", "#FFF8DC") - session_color = _skin_color("session_border", "#8B8682") # Use skin's custom caduceus art if provided - try: - from hermes_cli.skin_engine import get_active_skin - _bskin = get_active_skin() - _hero = _bskin.banner_hero if hasattr(_bskin, 'banner_hero') and _bskin.banner_hero else HERMES_CADUCEUS - except Exception: - _bskin = None - _hero = HERMES_CADUCEUS - left_lines = ["", _hero, ""] + _bskin = _quiet(_active_skin) + left_lines = ["", getattr(_bskin, "banner_hero", None) or HERMES_CADUCEUS, ""] + + def _dim_sep(label: str) -> str: + return f" [dim {dim}]·[/] [dim {dim}]{label}[/]" + + ctx_str = _dim_sep(f"{_format_context_length(context_length)} context") if context_length else "" + nous_str = _dim_sep("Nous Research") if (provider or "").strip().lower() == "moa": # MoA virtual provider: ``model`` is a preset name. Show the preset and # its aggregator so the banner is meaningful instead of a bare slug. - preset_name = model - agg_label = "" - try: - from hermes_cli.config import load_config - from hermes_cli.moa_config import normalize_moa_config - - _moa = normalize_moa_config(load_config().get("moa") or {}) - _preset = _moa.get("presets", {}).get(preset_name) - if _preset: - _agg = _preset.get("aggregator") or {} - _am = str(_agg.get("model") or "") - agg_label = _am.split("/")[-1] if "/" in _am else _am - except Exception: - agg_label = "" - if len(preset_name) > 28: - preset_name = preset_name[:25] + "..." - agg_str = f" [dim {dim}]·[/] [dim {dim}]agg {agg_label}[/]" if agg_label else "" - ctx_str = f" [dim {dim}]·[/] [dim {dim}]{_format_context_length(context_length)} context[/]" if context_length else "" - left_lines.append(f"[{accent}]MoA: {preset_name}[/]{agg_str}{ctx_str} [dim {dim}]·[/] [dim {dim}]Nous Research[/]") + agg_label = _quiet(lambda: _moa_aggregator_label(model), "") + agg_str = _dim_sep(f"agg {agg_label}") if agg_label else "" + left_lines.append(f"[{accent}]MoA: {_short_label(model)}[/]{agg_str}{ctx_str}{nous_str}") + elif not (model or "").strip() or (model or "").strip().lower() == "unknown": + # Unconfigured install: say so in red instead of a blank/"unknown" + # slug — this is the single clearest place to tell the user what + # is wrong and how to fix it. + left_lines.append( + f"[bold red]no model configured[/] " + f"[dim {dim}]— run /model or hermes setup[/]" + ) else: - if not (model or "").strip() or (model or "").strip().lower() == "unknown": - # Unconfigured install: say so in red instead of a blank/"unknown" - # slug — this is the single clearest place to tell the user what - # is wrong and how to fix it. - left_lines.append( - f"[bold red]no model configured[/] " - f"[dim {dim}]— run /model or hermes setup[/]" - ) - else: - model_short = model.split("/")[-1] if "/" in model else model - if model_short.endswith(".gguf"): - model_short = model_short[:-5] - if len(model_short) > 28: - model_short = model_short[:25] + "..." - ctx_str = f" [dim {dim}]·[/] [dim {dim}]{_format_context_length(context_length)} context[/]" if context_length else "" - left_lines.append(f"[{accent}]{model_short}[/]{ctx_str} [dim {dim}]·[/] [dim {dim}]Nous Research[/]") + model_short = model.split("/")[-1].removesuffix(".gguf") + left_lines.append(f"[{accent}]{_short_label(model_short)}[/]{ctx_str}{nous_str}") if os.getenv("HERMES_YOLO_MODE"): left_lines.append(f"[bold red]⚠ YOLO mode[/] [dim {dim}]— all approval prompts bypassed[/]") left_lines.append(f"[dim {dim}]{cwd}[/]") if session_id: - left_lines.append(f"[dim {session_color}]Session: {session_id}[/]") + left_lines.append(f"[dim {_skin_color('session_border', '#8B8682')}]Session: {session_id}[/]") left_content = "\n".join(left_lines) right_lines = [f"[bold {accent}]Available Tools[/]"] @@ -1083,126 +956,51 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, toolsets_dict.setdefault(toolset, []).append(tool_name) for item in unavailable_toolsets: - toolset_id = item.get("id", item.get("name", "unknown")) - display_name = _display_toolset_name(toolset_id) - if display_name not in toolsets_dict: - toolsets_dict[display_name] = [] + names = toolsets_dict.setdefault(_display_toolset_name(item.get("id", item.get("name", "unknown"))), []) for tool_name in item.get("tools", []): - if tool_name not in toolsets_dict[display_name]: - toolsets_dict[display_name].append(tool_name) + if tool_name not in names: + names.append(tool_name) sorted_toolsets = sorted(toolsets_dict.keys()) - display_toolsets = sorted_toolsets[:8] - remaining_toolsets = len(sorted_toolsets) - 8 - for toolset in display_toolsets: - tool_names = toolsets_dict[toolset] - colored_names = [] - for name in sorted(tool_names): - if name in disabled_tools: - colored_names.append(f"[red]{name}[/]") - elif name in lazy_tools: - colored_names.append(f"[yellow]{name}[/]") - else: - colored_names.append(f"[{text}]{name}[/]") + def _color_tool(name: Optional[str]) -> str: + if name is None: # truncation marker + return "[dim]...[/]" + if name in disabled_tools: + return f"[red]{name}[/]" + if name in lazy_tools: + return f"[yellow]{name}[/]" + return f"[{text}]{name}[/]" - tools_str = ", ".join(colored_names) - if len(", ".join(sorted(tool_names))) > 45: - short_names = [] - length = 0 - for name in sorted(tool_names): - if length + len(name) + 2 > 42: - short_names.append("...") - break - short_names.append(name) - length += len(name) + 2 - colored_names = [] - for name in short_names: - if name == "...": - colored_names.append("[dim]...[/]") - elif name in disabled_tools: - colored_names.append(f"[red]{name}[/]") - elif name in lazy_tools: - colored_names.append(f"[yellow]{name}[/]") - else: - colored_names.append(f"[{text}]{name}[/]") - tools_str = ", ".join(colored_names) + for toolset in sorted_toolsets[:8]: + tool_names = _truncate_tool_names(sorted(toolsets_dict[toolset])) + right_lines.append(f"[dim {dim}]{toolset}:[/] {', '.join(_color_tool(n) for n in tool_names)}") - right_lines.append(f"[dim {dim}]{toolset}:[/] {tools_str}") - - if remaining_toolsets > 0: - right_lines.append(f"[dim {dim}](and {remaining_toolsets} more toolsets...)[/]") + if len(sorted_toolsets) > 8: + right_lines.append(f"[dim {dim}](and {len(sorted_toolsets) - 8} more toolsets...)[/]") # MCP Servers section (only if configured). Probe cheaply first: the # full get_mcp_status() path resolves portable plugin MCP servers, # which JOINS the in-flight background plugin discovery (~100ms on the # startup path). When neither config.yaml nor the persisted plugin # key cache mentions any MCP server, skip the section outright. - mcp_status = [] - try: - from hermes_cli.config import load_config as _load_cfg - _has_native_mcp = bool((_load_cfg() or {}).get("mcp_servers")) - except Exception: - _has_native_mcp = True # can't tell — take the full path - _has_portable_mcp = False - if not _has_native_mcp: - try: - from hermes_cli.plugins import get_portable_mcp_server_names_nowait - _has_portable_mcp = bool(get_portable_mcp_server_names_nowait()) - except Exception: - _has_portable_mcp = True # can't tell — take the full path - if _has_native_mcp or _has_portable_mcp: - try: - from tools.mcp_tool import get_mcp_status - mcp_status = get_mcp_status() - except Exception: - mcp_status = [] + mcp_status = _quiet(_probe_mcp_status, []) if _mcp_configured() else [] if mcp_status: - right_lines.append("") - right_lines.append(f"[bold {accent}]MCP Servers[/]") - for srv in mcp_status: - status = srv.get("status") - if srv["connected"]: - right_lines.append( - f"[dim {dim}]{srv['name']}[/] [{text}]({srv['transport']})[/] " - f"[dim {dim}]—[/] [{text}]{srv['tools']} tool(s)[/]" - ) - elif srv.get("disabled") or status == "disabled": - right_lines.append( - f"[dim {dim}]{srv['name']}[/] [dim]({srv['transport']})[/] " - f"[dim {dim}]— disabled[/]" - ) - elif status == "connecting": - right_lines.append( - f"[dim {dim}]{srv['name']}[/] [dim]({srv['transport']})[/] " - f"[yellow]— connecting[/]" - ) - elif status == "configured": - right_lines.append( - f"[dim {dim}]{srv['name']}[/] [dim]({srv['transport']})[/] " - f"[dim {dim}]— configured[/]" - ) - else: - right_lines.append( - f"[red]{srv['name']}[/] [dim]({srv['transport']})[/] " - f"[red]— failed[/]" - ) + right_lines += ["", f"[bold {accent}]MCP Servers[/]"] + right_lines.extend(_mcp_server_line(srv, dim=dim, text=text) for srv in mcp_status) - right_lines.append("") - right_lines.append(f"[bold {accent}]Available Skills[/]") + right_lines += ["", f"[bold {accent}]Available Skills[/]"] # The skills catalog is only reachable when the `skills` toolset is enabled # (it exposes skill_view / skill_manage). When it's disabled — e.g. a Blank # Slate install — the agent literally cannot load any skill, so advertising # the on-disk catalog here is misleading. Reflect the real state instead. _skills_enabled = (not _enabled_ts) or ("skills" in _enabled_ts) - if _skills_enabled: - if skills_by_category is None: - skills_by_category = get_available_skills() - total_skills = sum(len(s) for s in skills_by_category.values()) - else: + if not _skills_enabled: skills_by_category = {} - total_skills = 0 + elif skills_by_category is None: + skills_by_category = get_available_skills() + total_skills = sum(len(s) for s in skills_by_category.values()) # Dynamically size skills display based on terminal width. # Rich grid with 2 columns; right column gets roughly 60% of terminal. @@ -1213,31 +1011,14 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, right_lines.append(f"[dim {dim}]Skills toolset disabled[/]") elif skills_by_category: for category in sorted(skills_by_category.keys()): - skill_names = sorted(skills_by_category[category]) - # Account for "category: " prefix - _prefix_len = len(category) + 2 - _avail = max(_right_col_width - _prefix_len, 20) - # Accumulate skills until we run out of space - parts, length = [], 0 - for i, name in enumerate(skill_names): - _sep = ", " if parts else "" - _needed = len(_sep) + len(name) - # Estimate indicator size IF we were to add this skill then stop - _after = len(skill_names) - (i + 1) # remaining after adding this - _ind_len = len(f", +{_after} more") if _after > 0 else 0 - if parts and length + _needed + _ind_len > _avail: - remaining = len(skill_names) - len(parts) - parts.append(f"+{remaining} more") - break - parts.append(name) - length += _needed - skills_str = ", ".join(parts) + # Account for the "category: " prefix. + skills_str = _pack_skill_names(sorted(skills_by_category[category]), max(_right_col_width - len(category) - 2, 20)) right_lines.append(f"[dim {dim}]{category}:[/] [{text}]{skills_str}[/]") else: right_lines.append(f"[dim {dim}]No skills installed[/]") right_lines.append("") - mcp_connected = sum(1 for s in mcp_status if s["connected"]) if mcp_status else 0 + mcp_connected = sum(1 for s in mcp_status if s["connected"]) summary_parts = [f"{len(tools)} tools", f"{total_skills} skills"] if mcp_connected: summary_parts.append(f"{mcp_connected} MCP servers") @@ -1245,24 +1026,15 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, # Indicate when the codex_app_server runtime is active so users # understand why tool counts may not match what's actually reachable # (codex builds its own tool list inside the spawned subprocess). - try: - from hermes_cli.codex_runtime_switch import get_current_runtime - from hermes_cli.config import load_config as _load_cfg - if get_current_runtime(_load_cfg()) == "codex_app_server": - right_lines.append( - f"[bold {accent}]Runtime:[/] [{text}]codex app-server[/] " - f"[dim {dim}](terminal/file ops/MCP run inside codex)[/]" - ) - except Exception: - pass - # Show active profile name when not 'default' - try: - from hermes_cli.profiles import get_active_profile_name - _profile_name = get_active_profile_name() - if _profile_name and _profile_name != "default": - right_lines.append(f"[bold {accent}]Profile:[/] [{text}]{_profile_name}[/]") - except Exception: - pass # Never break the banner over a profiles.py bug + if _quiet(_codex_runtime_active, False): + right_lines.append( + f"[bold {accent}]Runtime:[/] [{text}]codex app-server[/] " + f"[dim {dim}](terminal/file ops/MCP run inside codex)[/]" + ) + # Show active profile name when not 'default'. Never break the banner over a profiles.py bug. + _profile_name = _quiet(_active_profile_name) + if _profile_name and _profile_name != "default": + right_lines.append(f"[bold {accent}]Profile:[/] [{text}]{_profile_name}[/]") right_lines.append(f"[dim {dim}]{' · '.join(summary_parts)}[/]") @@ -1273,38 +1045,30 @@ def build_welcome_banner(console: "Console", model: str, cwd: str, # result isn't ready yet, defer the warning line: a daemon thread waits # for the prefetch and prints the same notice above the prompt when it # lands (prompt_toolkit's patch_stdout renders late prints safely). - try: + def _update_line(): behind = get_update_result(timeout=0.05) if behind is None and not _update_check_done.is_set(): _defer_update_notice(console) elif behind is not None and behind != 0: right_lines.append(_format_update_notice(behind)) - except Exception: - pass # Never break the banner over an update check - right_content = "\n".join(right_lines) - layout_table.add_row(left_content, right_content) + _quiet(_update_line) # Never break the banner over an update check + + layout_table.add_row(left_content, "\n".join(right_lines)) - title_color = _skin_color("banner_title", "#FFD700") - border_color = _skin_color("banner_border", "#CD7F32") version_label = format_banner_version_label() release_info = get_latest_release_tag() if release_info: - _tag, _url = release_info - title_markup = f"[bold {title_color}][link={_url}]{version_label}[/link][/]" - else: - title_markup = f"[bold {title_color}]{version_label}[/]" + version_label = f"[link={release_info[1]}]{version_label}[/link]" outer_panel = Panel( layout_table, - title=title_markup, - border_style=border_color, + title=f"[bold {_skin_color('banner_title', '#FFD700')}]{version_label}[/]", + border_style=_skin_color("banner_border", "#CD7F32"), padding=(0, 2), ) console.print() - term_width = shutil.get_terminal_size().columns - if term_width >= 95: - _logo = _bskin.banner_logo if _bskin and hasattr(_bskin, 'banner_logo') and _bskin.banner_logo else HERMES_AGENT_LOGO - console.print(_logo) + if shutil.get_terminal_size().columns >= 95: + console.print(getattr(_bskin, "banner_logo", None) or HERMES_AGENT_LOGO) console.print() console.print(outer_panel) diff --git a/hermes_cli/codex_models.py b/hermes_cli/codex_models.py index 2b17802d87..d51b0d18bc 100644 --- a/hermes_cli/codex_models.py +++ b/hermes_cli/codex_models.py @@ -65,39 +65,34 @@ _FORWARD_COMPAT_TEMPLATE_MODELS: List[tuple[str, tuple[str, ...]]] = [ ] +def _dedupe(model_ids) -> List[str]: + """Order-preserving dedupe.""" + return list(dict.fromkeys(model_ids)) + + def _add_forward_compat_models(model_ids: List[str]) -> List[str]: """Add Clawdbot-style synthetic forward-compat Codex models. - If a newer Codex slug isn't returned by live discovery, surface it when an - older compatible template model is present. This mirrors Clawdbot's - synthetic catalog / forward-compat behavior for GPT-5 Codex variants. + If a newer Codex slug isn't returned by live discovery, surface it when an older compatible + template model is present. This mirrors Clawdbot's synthetic catalog / forward-compat behavior + for GPT-5 Codex variants. """ - ordered: List[str] = [] - seen: set[str] = set() - for model_id in model_ids: - if model_id not in seen: - ordered.append(model_id) - seen.add(model_id) - + ordered = _dedupe(model_ids) + seen = set(ordered) for synthetic_model, template_models in _FORWARD_COMPAT_TEMPLATE_MODELS: - if synthetic_model in seen: - continue - if any(template in seen for template in template_models): + if synthetic_model not in seen and any(template in seen for template in template_models): ordered.append(synthetic_model) seen.add(synthetic_model) - return ordered def _add_context_variants(model_ids: List[str]) -> List[str]: """Insert ``-900k`` large-context picker variants after eligible base slugs. - The ChatGPT Codex backend advertises 272K for the gpt-5.4 / gpt-5.6 - families but accepts ~911K (live-verified Aug 2026). The base slugs keep - the cheaper advertised 272K limit by default; each verified slug gets an - explicit ``-900k`` picker entry that opts into the large window. - The suffix is Hermes-side only — it is stripped before the model id hits - the wire (agent/transports/codex.py, agent/auxiliary_client.py). + The base slugs keep the cheaper advertised 272K limit by default; each verified slug gets an + explicit ``-900k`` picker entry that opts into the large window. The suffix is Hermes-side + only — it is stripped before the model id hits the wire (agent/transports/codex.py, + agent/auxiliary_client.py). """ from agent.model_metadata import ( CODEX_CONTEXT_VARIANT_SUFFIX, @@ -124,16 +119,13 @@ def _finalize_codex_models(model_ids: List[str]) -> List[str]: def _extract_chatgpt_account_id(access_token: str) -> Optional[str]: """Best-effort extraction of ``chatgpt_account_id`` from the OAuth JWT. - The Codex backend requires the ``ChatGPT-Account-Id`` header for the - per-account catalog. Without it, ``GET /backend-api/codex/models`` - returns ``{"models":[]}`` (HTTP 200) — which masquerades as "no - models available" and silently degrades the picker to the curated - fallback list. The request-side path in ``auxiliary_client.py`` - already extracts the same claim; this mirrors that logic here so the - probe sees the same catalog the request path will actually use. + The Codex backend requires the ``ChatGPT-Account-Id`` header for the per-account catalog. + Without it, ``GET /backend-api/codex/models`` returns ``{"models":[]}`` (HTTP 200) — which + masquerades as "no models available" and silently degrades the picker to the curated fallback + list. - Returns ``None`` on any parse error — the probe then degrades - gracefully to the unauthenticated fallback list instead of crashing. + Returns ``None`` on any parse error — the probe then degrades gracefully to the unauthenticated + fallback list instead of crashing. """ try: parts = access_token.split(".") @@ -151,6 +143,31 @@ def _extract_chatgpt_account_id(access_token: str) -> Optional[str]: return None +def _ranked_slugs(entries: object) -> List[str]: + """Visible model slugs from a Codex catalog ``models`` list, sorted by (priority, slug), deduped. + + Does not filter on ``supported_in_api``: that flag describes the public OpenAI API, while + Hermes openai-codex talks to the same OAuth-backed Codex backend as Codex CLI, which still + accepts slugs marked false there (for example gpt-5.3-codex-spark). + """ + sortable = [] + for item in entries: + if not isinstance(item, dict): + continue + slug = item.get("slug") + if not isinstance(slug, str) or not slug.strip(): + continue + visibility = item.get("visibility") + if isinstance(visibility, str) and visibility.strip().lower() in {"hide", "hidden"}: + continue + priority = item.get("priority") + rank = int(priority) if isinstance(priority, (int, float)) else 10_000 + sortable.append((rank, slug.strip())) + + sortable.sort() + return _dedupe(slug for _, slug in sortable) + + def _fetch_models_from_api(access_token: str) -> List[str]: """Fetch available models from the Codex API. Returns visible models sorted by priority.""" try: @@ -172,27 +189,7 @@ def _fetch_models_from_api(access_token: str) -> List[str]: logger.debug("Failed to fetch Codex models from API: %s", exc) return [] - sortable = [] - for item in entries: - if not isinstance(item, dict): - continue - slug = item.get("slug") - if not isinstance(slug, str) or not slug.strip(): - continue - slug = slug.strip() - # Codex CLI's catalog uses ``supported_in_api`` for the public OpenAI - # API, not for the OAuth-backed Codex backend that this provider uses. - # Some valid Codex CLI models (for example gpt-5.3-codex-spark) are - # marked false here but are still accepted by the Codex route. - visibility = item.get("visibility", "") - if isinstance(visibility, str) and visibility.strip().lower() in {"hide", "hidden"}: - continue - priority = item.get("priority") - rank = int(priority) if isinstance(priority, (int, float)) else 10_000 - sortable.append((rank, slug)) - - sortable.sort(key=lambda x: (x[0], x[1])) - return _finalize_codex_models([slug for _, slug in sortable]) + return _finalize_codex_models(_ranked_slugs(entries)) def _read_default_model(codex_home: Path) -> Optional[str]: @@ -201,16 +198,12 @@ def _read_default_model(codex_home: Path) -> Optional[str]: return None try: import tomllib - except Exception: - return None - try: + payload = tomllib.loads(config_path.read_text(encoding="utf-8")) except Exception: return None model = payload.get("model") if isinstance(payload, dict) else None - if isinstance(model, str) and model.strip(): - return model.strip() - return None + return model.strip() if isinstance(model, str) and model.strip() else None def _read_cache_models(codex_home: Path) -> List[str]: @@ -223,42 +216,16 @@ def _read_cache_models(codex_home: Path) -> List[str]: return [] entries = raw.get("models") if isinstance(raw, dict) else None - sortable = [] - if isinstance(entries, list): - for item in entries: - if not isinstance(item, dict): - continue - slug = item.get("slug") - if not isinstance(slug, str) or not slug.strip(): - continue - slug = slug.strip() - # Do not filter on ``supported_in_api`` here. It describes the - # public OpenAI API, while Hermes openai-codex talks to the same - # OAuth-backed Codex backend as Codex CLI. - visibility = item.get("visibility") - if isinstance(visibility, str) and visibility.strip().lower() in {"hide", "hidden"}: - continue - priority = item.get("priority") - rank = int(priority) if isinstance(priority, (int, float)) else 10_000 - sortable.append((rank, slug)) - - sortable.sort(key=lambda item: (item[0], item[1])) - deduped: List[str] = [] - for _, slug in sortable: - if slug not in deduped: - deduped.append(slug) - return deduped + return _ranked_slugs(entries if isinstance(entries, list) else []) def get_codex_model_ids(access_token: Optional[str] = None) -> List[str]: """Return available Codex model IDs, trying API first, then local sources. - - Resolution order: API (live, if token provided) > config.toml default > - local cache > hardcoded defaults. + + Resolution order: API (live, if token provided) > config.toml default > local cache > hardcoded + defaults. """ - codex_home_str = os.getenv("CODEX_HOME", "").strip() or str(Path.home() / ".codex") - codex_home = Path(codex_home_str).expanduser() - ordered: List[str] = [] + codex_home = Path(os.getenv("CODEX_HOME", "").strip() or str(Path.home() / ".codex")).expanduser() # Try live API if we have a token if access_token: @@ -268,15 +235,8 @@ def get_codex_model_ids(access_token: Optional[str] = None) -> List[str]: # Fall back to local sources default_model = _read_default_model(codex_home) - if default_model: - ordered.append(default_model) - - for model_id in _read_cache_models(codex_home): - if model_id not in ordered: - ordered.append(model_id) - - for model_id in DEFAULT_CODEX_MODELS: - if model_id not in ordered: - ordered.append(model_id) - - return _finalize_codex_models(ordered) + return _finalize_codex_models(_dedupe([ + *([default_model] if default_model else []), + *_read_cache_models(codex_home), + *DEFAULT_CODEX_MODELS, + ])) diff --git a/hermes_cli/codex_runtime_plugin_migration.py b/hermes_cli/codex_runtime_plugin_migration.py index 4b30d3ebf2..310867762f 100644 --- a/hermes_cli/codex_runtime_plugin_migration.py +++ b/hermes_cli/codex_runtime_plugin_migration.py @@ -1,37 +1,5 @@ -"""Migrate Hermes' MCP server config and Codex's installed curated plugins -to the format Codex expects in ~/.codex/config.toml. - -When the user enables the codex_app_server runtime, the codex subprocess -runs its own MCP client and its own plugin runtime (Linear, Atlassian, -Asana, plus per-account ChatGPT apps via app/list). For both of those to -be useful, the user's choices need to be visible to codex too. This -module: - - 1. Reads Hermes' YAML and writes equivalent [mcp_servers.] - entries to ~/.codex/config.toml. - 2. Queries codex's `plugin/list` for the openai-curated marketplace - and writes [plugins."@"] entries for any plugin - the user has installed=true on their codex CLI. (This is what - OpenClaw calls "migrate native codex plugins" — the YouTube-video- - worthy bit Pash highlighted: Canva, GitHub, Calendar, Gmail - pre-configured.) - 3. Writes a [permissions] default profile so users on this runtime - don't get an approval prompt on every write attempt. - -What translates (MCP servers): - Hermes mcp_servers..command/args/env → codex stdio transport - Hermes mcp_servers..url/headers → codex streamable_http transport - Hermes mcp_servers..timeout → codex tool_timeout_sec - Hermes mcp_servers..connect_timeout → codex startup_timeout_sec - -What does NOT translate (warned + skipped): - Hermes-specific keys (sampling, etc.) — codex's MCP client has no - equivalent. Listed in the per-server skipped[] field of the report. - -What's NOT migrated (intentional): - AGENTS.md — codex respects this file natively in its cwd. Hermes' own - AGENTS.md (project-level) is already in the worktree, so codex picks - it up without translation. No code needed. +"""Migrate Hermes' MCP server config and Codex's installed curated plugins to the format Codex expects +in ~/.codex/config.toml. """ from __future__ import annotations @@ -127,11 +95,9 @@ _KEYS_DROPPED_WITH_WARNING = { def _translate_one_server( name: str, hermes_cfg: dict ) -> tuple[Optional[dict], list[str]]: - """Translate one Hermes MCP server config to the codex inline-table dict - representation. Returns (codex_entry, skipped_keys). - - codex_entry is a dict ready for TOML serialization, or None when the - server can't be translated (e.g. neither command nor url present).""" + """Translate one Hermes MCP server config to the codex inline-table dict representation. Returns + (codex_entry, skipped_keys). + """ if not isinstance(hermes_cfg, dict): return None, [] @@ -199,8 +165,9 @@ def _translate_one_server( def _format_toml_value(value: Any) -> str: """Minimal TOML value formatter for the value types we emit. - We only emit strings, numbers, booleans, and tables of those — no nested - arrays of tables. This covers everything codex's MCP schema accepts.""" + We only emit strings, numbers, booleans, and tables of those — no nested arrays of tables. This + covers everything codex's MCP schema accepts. + """ if isinstance(value, bool): return "true" if value else "false" if isinstance(value, (int, float)): @@ -248,18 +215,11 @@ def render_codex_toml_section( plugins: Optional[list[dict]] = None, default_permission_profile: Optional[str] = None, ) -> str: - """Render the managed [mcp_servers.] / [plugins.] / [permissions] - block for ~/.codex/config.toml. + """Render the managed [mcp_servers.] / [plugins.] / [permissions] block for + ~/.codex/config.toml. - Args: - servers: dict of MCP server name → translated codex inline-table - plugins: optional list of {name, marketplace, enabled} for native - Codex plugins to enable. (E.g. the Linear / Atlassian / Asana - curated plugins, or per-account ChatGPT apps.) - default_permission_profile: when set, write `[permissions] default` - so the user doesn't get an approval prompt on every write - attempt. Common values: "workspace-write", "read-only", - "full-access". + ``default_permission_profile`` (e.g. "workspace-write", "read-only", "full-access") writes + ``[permissions] default`` so the user is not prompted on every write attempt. """ out = [MIGRATION_MARKER] if not servers and not plugins and not default_permission_profile: @@ -307,11 +267,9 @@ def render_codex_toml_section( def _insert_managed_block_at_top_level(user_text: str, managed_block: str) -> str: """Insert Hermes' managed Codex TOML block while keeping root keys root-scoped. - TOML has no syntax to return to the document root after a table header. - Therefore appending a root key like `default_permissions = ...` after a - user table such as `[features]` actually creates `features.default_permissions`, - which Codex rejects. Insert the managed block before the first table header - so its root keys remain top-level, while preserving user content verbatim. + TOML has no syntax to return to the document root after a table header. Therefore appending a + root key like `default_permissions = ...` after a user table such as `[features]` actually + creates `features.default_permissions`, which Codex rejects. """ if not user_text.strip(): return managed_block @@ -336,28 +294,16 @@ def _insert_managed_block_at_top_level(user_text: str, managed_block: str) -> st def _strip_unmanaged_plugin_tables(toml_text: str) -> str: - """Remove ``[plugins."@"]`` tables that live OUTSIDE the - managed block. + """Remove ``[plugins."@"]`` tables that live OUTSIDE the managed block. - Codex itself writes these tables when the user runs ``codex plugins enable`` - directly (i.e. before Hermes' migrate has ever touched the file). When we - later run migrate, ``_query_codex_plugins()`` reports the same plugins via - the live ``plugin/list`` RPC and we re-emit them inside the managed block. - The result without this strip is duplicate ``[plugins."X@Y"]`` table - headers — codex's strict TOML parser then refuses to load the file. + Codex itself writes these tables when the user runs ``codex plugins enable`` directly (i.e. + before Hermes' migrate has ever touched the file). When we later run migrate, + ``_query_codex_plugins()`` reports the same plugins via the live ``plugin/list`` RPC and we re- + emit them inside the managed block. - We own the ``[plugins.*]`` namespace once migrate has run, so dropping any - pre-existing ``[plugins.*]`` tables is safe: ``plugin/list`` is the source - of truth for what's actually installed. The caller is expected to only - invoke this strip when ``plugin/list`` succeeded — otherwise we'd lose - plugins the user installed via ``codex`` without a way to re-emit them. - - Behavior: - * Lines beginning with ``[plugins.`` start a swallow region that ends at - the next non-``[plugins.`` table header or end-of-file. - * Content inside the managed block is untouched (callers should run - ``_strip_existing_managed_block`` first so the managed block has - already been removed when this runs). + We own the ``[plugins.*]`` namespace once migrate has run, so dropping any pre-existing + ``[plugins.*]`` tables is safe: ``plugin/list`` is the source of truth for what's actually + installed. """ lines = toml_text.splitlines(keepends=True) out: list[str] = [] @@ -384,11 +330,9 @@ def _strip_unmanaged_plugin_tables(toml_text: str) -> str: def _looks_like_table_header(stripped_line: str) -> bool: """Return True if ``stripped_line`` is a TOML table header. - A header has the shape ``[name]`` or ``[[name]]`` (array-of-tables), - optionally followed by a comment. The closing ``]`` (or ``]]``) must - appear on the same line, and no key-assignment ``=`` can precede it. - This distinguishes real headers from multi-line array continuation - lines that also start with ``[`` after ``lstrip()``. + A header has the shape ``[name]`` or ``[[name]]`` (array-of-tables), optionally followed by a + comment. The closing ``]`` (or ``]]``) must appear on the same line, and no key-assignment ``=`` + can precede it. """ if not stripped_line.startswith("["): return False @@ -404,16 +348,11 @@ def _looks_like_table_header(stripped_line: str) -> bool: def _strip_existing_managed_block(toml_text: str) -> str: """Remove any prior managed section so re-runs idempotently replace it. - The managed section is everything between MIGRATION_MARKER (start) and - MIGRATION_END_MARKER (end), inclusive of both markers. User-edited - sections above or below are preserved verbatim. - - Backward compatibility: if the start marker is found but no end marker - follows, we fall back to the heuristic that swallows lines until we - hit a section that's not [mcp_servers.*]/[plugins.*]/[permissions]/ - a `default_permissions =` key. This matches what older versions of - this code wrote so re-runs don't break configs from prior Hermes - versions.""" + The managed section spans MIGRATION_MARKER through MIGRATION_END_MARKER inclusive; user text + outside it is preserved verbatim. If the start marker exists without an end marker (older + writers), fall back to swallowing lines until a section that is not [mcp_servers.*]/ + [plugins.*]/[permissions]/``default_permissions =``, so prior-version configs still migrate. + """ lines = toml_text.splitlines(keepends=True) out: list[str] = [] in_managed = False @@ -453,14 +392,9 @@ def _query_codex_plugins( ) -> tuple[list[dict], Optional[str]]: """Query codex's `plugin/list` for installed curated plugins. - Spawns `codex app-server` briefly, sends initialize + plugin/list, - extracts plugins where installed=true. Returns (plugins, error). - Plugins is a list of {name, marketplace, enabled} dicts ready for - render_codex_toml_section(). - - On any failure (codex not installed, RPC error, timeout) returns - ([], error_message). Migration treats this as non-fatal — MCP - servers and permissions still write through. + Spawns ``codex app-server`` briefly, sends initialize + plugin/list and keeps installed=true. + Returns ``(plugins, error)``; any failure (not installed, RPC error, timeout) yields ``([], + error)`` and is non-fatal — MCP servers and permissions still write through. """ try: from agent.transports.codex_app_server import CodexAppServerClient @@ -531,16 +465,10 @@ def _query_codex_plugins( def _looks_like_test_tempdir(path: str) -> bool: """Heuristic: does ``path`` look like a pytest/transient tempdir? - pytest tempdirs live under ``pytest-of-/pytest-/`` (created via - ``tmp_path`` / ``tmp_path_factory``) and are reaped between sessions. - macOS routes ``/tmp`` through ``/private/var/folders/<…>/T`` which is - what pytest's tempdir factory uses by default. If a HERMES_HOME pointing - at one of those paths is burned into ``~/.codex/config.toml``, every - codex-routed hermes-tools call fails silently once the directory is GC'd. - - We err on the side of refusing — losing a (very unlikely) real - ``~/.hermes`` symlink that happens to live under ``/private/var/folders`` - is much less harmful than silently bricking codex's tool surface. + pytest tempdirs (``pytest-of-/pytest-/``, on macOS under ``/private/var/folders/…/T``) + are reaped between sessions; a HERMES_HOME pointing there burned into ``~/.codex/config.toml`` + makes every codex-routed call fail silently once GC'd. Err on refusing: a false positive is far + less harmful than silently bricking codex's tool surface. """ if not path: return False @@ -555,14 +483,10 @@ def _looks_like_test_tempdir(path: str) -> bool: def _build_hermes_tools_mcp_entry() -> dict: - """Build the codex stdio-transport entry that launches Hermes' own - tool surface as an MCP server. Codex's subprocess will call back into - this for browser/web/delegate_task/vision/memory/skills tools. - - The command runs the worktree's Python via the current sys.executable - so a hermes installed under /opt/, /usr/local/, or a venv all work. - HERMES_HOME and PYTHONPATH are passed through so the spawned process - sees the same config + module layout the user is running.""" + """Build the codex stdio-transport entry that launches Hermes' own tool surface as an MCP server. + Codex's subprocess will call back into this for browser/web/delegate_task/vision/memory/skills + tools. + """ import sys env: dict[str, str] = {} @@ -615,31 +539,13 @@ def migrate( default_permission_profile: Optional[str] = ":workspace", expose_hermes_tools: bool = True, ) -> MigrationReport: - """Translate Hermes mcp_servers config + Codex curated plugins into - ~/.codex/config.toml. + """Translate Hermes mcp_servers config + Codex curated plugins into ~/.codex/config.toml. - Args: - hermes_config: full ~/.hermes/config.yaml dict - codex_home: override CODEX_HOME (defaults to ~/.codex) - dry_run: skip the actual write; report what would happen - discover_plugins: when True (default), query `plugin/list` against - the live codex CLI to migrate any installed curated plugins - into [plugins."@"] entries. Set False to - skip the subprocess spawn (for tests or restricted environments). - default_permission_profile: when set (default ":workspace"), write - top-level `default_permissions = ""` so users on this - runtime don't get an approval prompt on every write attempt. - Built-in codex profile names are ":workspace", ":read-only", - ":danger-no-sandbox" (note the leading ":"). Also accepts a - user-defined profile name (no leading ":") that the user has - configured in their own [permissions.] table. Set None - to leave permissions unset and let codex use its compiled-in - default (which is read-only). - expose_hermes_tools: when True (default), register Hermes' own - tool surface (web_search, browser_*, delegate_task, vision, - memory, skills, etc.) as an MCP server in ~/.codex/config.toml - so the codex subprocess can call back into Hermes for tools - codex doesn't have built in. Set False to opt out. + ``discover_plugins`` spawns the live codex CLI (set False in tests). + ``default_permission_profile`` (default ":workspace"; built-ins carry a leading ":", user + profiles do not; None leaves codex's read-only default) avoids an approval prompt on every + write. ``expose_hermes_tools`` registers Hermes' own tool surface as an MCP server so the codex + subprocess can call back for tools it lacks. """ report = MigrationReport(dry_run=dry_run) codex_home = codex_home or Path.home() / ".codex" diff --git a/hermes_cli/codex_runtime_switch.py b/hermes_cli/codex_runtime_switch.py index 06bff58e2e..18f5224b80 100644 --- a/hermes_cli/codex_runtime_switch.py +++ b/hermes_cli/codex_runtime_switch.py @@ -1,18 +1,12 @@ """Shared logic for the /codex-runtime slash command. -Toggles `model.openai_runtime` between "auto" (= chat_completions, Hermes' -default) and "codex_app_server" (= hand turns to a codex subprocess). - -Both CLI (cli.py) and gateway (gateway/run.py) call into this module so the -behavior stays identical across surfaces. - -The actual runtime resolution happens in hermes_cli.runtime_provider's -_maybe_apply_codex_app_server_runtime() helper, which reads the persisted -config value. This module just persists the value and reports the change. +Both CLI (cli.py) and gateway (gateway/run.py) call into this module so the behavior stays identical +across surfaces. """ from __future__ import annotations +import functools import logging from dataclasses import dataclass from typing import Optional @@ -33,16 +27,13 @@ class CodexRuntimeStatus: old_value: Optional[str] = None message: str = "" requires_new_session: bool = False - codex_binary_ok: bool = True - codex_version: Optional[str] = None def parse_args(arg_string: str) -> tuple[Optional[str], list[str]]: """Parse the slash-command argument string. Returns (value, errors). - No args → return current state (value=None) - 'auto' / 'codex_app_server' / 'on' / 'off' → return that value - anything else → error + No args returns the current state (value=None); ``auto``/``codex_app_server``/``on``/``off`` + return that value; anything else is an error. """ raw = (arg_string or "").strip().lower() if not raw: @@ -106,27 +97,15 @@ def apply( ) -> CodexRuntimeStatus: """Top-level entry point used by both CLI and gateway handlers. - Args: - config: in-memory config dict (will be mutated when new_value is set) - new_value: desired runtime; None means "show current state only" - persist_callback: optional callable taking the mutated config dict - and persisting it to disk. Skipped when None (used by tests). - - Returns: CodexRuntimeStatus describing the outcome. + ``config`` is mutated in place when ``new_value`` is set (None means show current state + only). ``persist_callback`` receives the mutated dict to write it to disk; skipped when None + (tests). """ current = get_current_runtime(config) - # Cache the codex binary check for this apply() call. Subprocess spawn - # is cheap (~50ms for `codex --version`), but we'd otherwise call it up - # to 3 times in the enable path (read-only/state, gate, success message). - # None = not yet checked; (bool, str) = result. - _binary_check: Optional[tuple[bool, Optional[str]]] = None - - def _check_binary_cached() -> tuple[bool, Optional[str]]: - nonlocal _binary_check - if _binary_check is None: - _binary_check = check_codex_binary_ok() - return _binary_check + # Cache the codex binary check for this apply() call: the enable path + # would otherwise spawn `codex --version` up to 3 times (state, gate, message). + _check_binary_cached = functools.cache(check_codex_binary_ok) # Read-only call: just report state if new_value is None: @@ -140,8 +119,6 @@ def apply( new_value=current, old_value=current, message=msg, - codex_binary_ok=ok, - codex_version=ver if ok else None, ) # No-config-change paths. For `auto` we return immediately — disabling @@ -177,8 +154,6 @@ def apply( f"{ver_or_msg or 'codex CLI not available'}\n" "Install with: npm i -g @openai/codex" ), - codex_binary_ok=False, - codex_version=None, ) if not reapplying_enable: @@ -195,12 +170,11 @@ def apply( message=f"updated config in memory but persist failed: {exc}", ) - if reapplying_enable: - msg_lines = [ - f"openai_runtime already set to {current} — re-applying migration" - ] - else: - msg_lines = [f"openai_runtime: {current} → {new_value}"] + msg_lines = [ + f"openai_runtime already set to {current} — re-applying migration" + if reapplying_enable + else f"openai_runtime: {current} → {new_value}" + ] if new_value == "codex_app_server": ok, ver = _check_binary_cached() if ok: diff --git a/hermes_cli/fallback_cmd.py b/hermes_cli/fallback_cmd.py index 56c9021cce..3b00430937 100644 --- a/hermes_cli/fallback_cmd.py +++ b/hermes_cli/fallback_cmd.py @@ -1,20 +1,13 @@ -""" -hermes fallback — manage the fallback provider chain. +"""hermes fallback — manage the fallback provider chain. -Fallback providers are tried in order when the primary model fails with -rate-limit, overload, or connection errors. See: -https://hermes-agent.nousresearch.com/docs/user-guide/features/fallback-providers +Fallback providers are tried in order when the primary model fails with rate-limit, overload, or +connection errors. See: https://hermes-agent.nousresearch.com/docs/user-guide/features/fallback- +providers -Subcommands: - hermes fallback [list] Show the current fallback chain (default when no subcommand) - hermes fallback add Pick provider + model via the same picker as `hermes model`, - then append the selection to the chain - hermes fallback remove Pick an entry to delete from the chain - hermes fallback clear Remove all fallback entries - -Storage: ``fallback_providers`` in ``~/.hermes/config.yaml`` (top-level, list of -``{provider, model, base_url?, api_mode?}`` dicts). The legacy single-dict -``fallback_model`` format is migrated to the new list format on first add. +Subcommands: hermes fallback [list] Show the current fallback chain (default when no subcommand) +hermes fallback add Pick provider + model via the same picker as `hermes model`, then append the +selection to the chain hermes fallback remove Pick an entry to delete from the chain hermes fallback +clear Remove all fallback entries """ from __future__ import annotations @@ -24,6 +17,17 @@ from typing import Any, Dict, List, Optional from hermes_cli.fallback_config import get_fallback_chain +def _identity(entry: Dict[str, Any]): + """BackendIdentity for a ``{provider, model, base_url?}`` entry.""" + from agent.backend_identity import BackendIdentity + + return BackendIdentity.build( + provider=entry.get("provider"), + model=entry.get("model"), + base_url=entry.get("base_url"), + ) + + # --------------------------------------------------------------------------- # Helpers # --------------------------------------------------------------------------- @@ -31,10 +35,10 @@ from hermes_cli.fallback_config import get_fallback_chain def _read_chain(config: Dict[str, Any]) -> List[Dict[str, Any]]: """Return the normalized fallback chain as a list of dicts. - Accepts both the new list format (``fallback_providers``) and the legacy - ``fallback_model`` format. When both are present, the effective chain is - merged with ``fallback_providers`` entries kept first. The returned list is - always a fresh copy — callers can mutate without touching the config dict. + Accepts both the new list format (``fallback_providers``) and the legacy ``fallback_model`` + format. When both are present, the effective chain is merged with ``fallback_providers`` entries + kept first. The returned list is always a fresh copy — callers can mutate without touching the + config dict. """ return get_fallback_chain(config) @@ -43,17 +47,13 @@ def _write_chain(config: Dict[str, Any], chain: List[Dict[str, Any]]) -> None: """Persist the chain to ``fallback_providers`` and clear legacy key.""" config["fallback_providers"] = chain # Drop the legacy single-dict key on write so there's only one source of truth. - if "fallback_model" in config: - config.pop("fallback_model", None) + config.pop("fallback_model", None) def _format_entry(entry: Dict[str, Any]) -> str: """One-line human-readable rendering of a fallback entry.""" - provider = entry.get("provider", "?") - model = entry.get("model", "?") base = entry.get("base_url") - suffix = f" [{base}]" if base else "" - return f"{model} (via {provider}){suffix}" + return f"{entry.get('model', '?')} (via {entry.get('provider', '?')}){f' [{base}]' if base else ''}" def _extract_fallback_from_model_cfg(model_cfg: Any) -> Optional[Dict[str, Any]]: @@ -66,12 +66,10 @@ def _extract_fallback_from_model_cfg(model_cfg: Any) -> Optional[Dict[str, Any]] if not provider or not model: return None entry: Dict[str, Any] = {"provider": provider, "model": model} - base_url = (model_cfg.get("base_url") or "").strip() - if base_url: - entry["base_url"] = base_url - api_mode = (model_cfg.get("api_mode") or "").strip() - if api_mode: - entry["api_mode"] = api_mode + for key in ("base_url", "api_mode"): + value = (model_cfg.get(key) or "").strip() + if value: + entry[key] = value return entry @@ -79,8 +77,7 @@ def _snapshot_auth_active_provider() -> Any: """Return the current ``active_provider`` in auth.json, or a sentinel if unavailable.""" try: from hermes_cli.auth import _load_auth_store - store = _load_auth_store() - return store.get("active_provider") + return _load_auth_store().get("active_provider") except Exception: return None @@ -104,29 +101,45 @@ def _restore_auth_active_provider(value: Any) -> None: # Subcommand handlers # --------------------------------------------------------------------------- -def cmd_fallback_list(args) -> None: # noqa: ARG001 - """Print the current fallback chain.""" +def _entries(n: int) -> str: + return f"{n} {'entry' if n == 1 else 'entries'}" + + +def _print_chain(heading: str, chain: List[Dict[str, Any]]) -> None: + print(f" {heading} ({_entries(len(chain))}):") + for i, entry in enumerate(chain, 1): + print(f" {i}. {_format_entry(entry)}") + print() + + +def _load_chain(empty_message: str): + """Load config + chain; print ``empty_message`` block and return ``(config, None)`` when empty.""" from hermes_cli.config import load_config config = load_config() chain = _read_chain(config) - - print() if not chain: - print(" No fallback providers configured.") print() + print(empty_message) + print() + return config, None + return config, chain + + +def cmd_fallback_list(args) -> None: # noqa: ARG001 + """Print the current fallback chain.""" + config, chain = _load_chain(" No fallback providers configured.") + if chain is None: print(" Add one with: hermes fallback add") print() return + print() primary = _describe_primary(config) if primary: print(f" Primary: {primary}") print() - print(f" Fallback chain ({len(chain)} {'entry' if len(chain) == 1 else 'entries'}):") - for i, entry in enumerate(chain, 1): - print(f" {i}. {_format_entry(entry)}") - print() + _print_chain("Fallback chain", chain) print(" Tried in order when the primary fails (rate-limit, 5xx, connection errors).") print(" Docs: https://hermes-agent.nousresearch.com/docs/user-guide/features/fallback-providers") print() @@ -162,12 +175,15 @@ def cmd_fallback_add(args) -> None: print(" `hermes model` — select the provider + model you want as a fallback.") print() + def _restore() -> None: + _restore_model_cfg(model_before) + _restore_auth_active_provider(active_provider_before) + try: select_provider_and_model(args=args) except SystemExit: # Some provider flows exit on auth failure — restore state and re-raise. - _restore_model_cfg(model_before) - _restore_auth_active_provider(active_provider_before) + _restore() raise # Read the post-picker state to see what the user selected. @@ -177,8 +193,7 @@ def cmd_fallback_add(args) -> None: new_entry = _extract_fallback_from_model_cfg(model_after) if not new_entry: # Picker didn't complete (user cancelled or flow bailed). Nothing to do. - _restore_model_cfg(model_before) - _restore_auth_active_provider(active_provider_before) + _restore() print() print(" No fallback added.") return @@ -188,24 +203,12 @@ def cmd_fallback_add(args) -> None: # semantics owned by agent.backend_identity (#54250/#57584/#62984): same # provider+model on a DIFFERENT explicit base_url is a different backend # (multi-endpoint pool) and is a legitimate fallback. - from agent.backend_identity import BackendIdentity, same_deployment + from agent.backend_identity import same_deployment - new_ident = BackendIdentity.build( - provider=new_entry.get("provider"), - model=new_entry.get("model"), - base_url=new_entry.get("base_url"), - ) + new_ident = _identity(new_entry) primary_entry = _extract_fallback_from_model_cfg(model_before) - if primary_entry and same_deployment( - BackendIdentity.build( - provider=primary_entry.get("provider"), - model=primary_entry.get("model"), - base_url=primary_entry.get("base_url"), - ), - new_ident, - ): - _restore_model_cfg(model_before) - _restore_auth_active_provider(active_provider_before) + if primary_entry and same_deployment(_identity(primary_entry), new_ident): + _restore() print() print(f" Selected model matches the current primary ({_format_entry(new_entry)}).") print(" A provider cannot be a fallback for itself — no change.") @@ -215,26 +218,17 @@ def cmd_fallback_add(args) -> None: # to ``fallback_providers``. We deliberately re-load (rather than mutating # ``after_cfg``) because the picker may have touched other top-level keys # (custom_providers, providers credentials) that we want to keep. - _restore_model_cfg(model_before) - _restore_auth_active_provider(active_provider_before) + _restore() final_cfg = load_config() chain = _read_chain(final_cfg) # Reject exact-duplicate fallback entries (same deployment; a different # explicit base_url is a different endpoint and NOT a duplicate). - for existing in chain: - if same_deployment( - BackendIdentity.build( - provider=existing.get("provider"), - model=existing.get("model"), - base_url=existing.get("base_url"), - ), - new_ident, - ): - print() - print(f" {_format_entry(new_entry)} is already in the fallback chain — skipped.") - return + if any(same_deployment(_identity(existing), new_ident) for existing in chain): + print() + print(f" {_format_entry(new_entry)} is already in the fallback chain — skipped.") + return chain.append(new_entry) _write_chain(final_cfg, chain) @@ -242,7 +236,7 @@ def cmd_fallback_add(args) -> None: print() print(f" Added fallback: {_format_entry(new_entry)}") - print(f" Chain is now {len(chain)} {'entry' if len(chain) == 1 else 'entries'} long.") + print(f" Chain is now {_entries(len(chain))} long.") print() print(" Run `hermes fallback list` to view, or `hermes fallback remove` to delete.") @@ -261,19 +255,13 @@ def _restore_model_cfg(model_before: Any) -> None: def cmd_fallback_remove(args) -> None: # noqa: ARG001 """Pick an entry from the chain and remove it.""" - from hermes_cli.config import load_config, save_config + from hermes_cli.config import save_config - config = load_config() - chain = _read_chain(config) - - if not chain: - print() - print(" No fallback providers configured — nothing to remove.") - print() + config, chain = _load_chain(" No fallback providers configured — nothing to remove.") + if chain is None: return - choices = [_format_entry(e) for e in chain] - choices.append("Cancel") + choices = [_format_entry(e) for e in chain] + ["Cancel"] try: from hermes_cli.setup import _curses_prompt_choice @@ -292,31 +280,20 @@ def cmd_fallback_remove(args) -> None: # noqa: ARG001 print() print(f" Removed fallback: {_format_entry(removed)}") - if chain: - print(f" Chain is now {len(chain)} {'entry' if len(chain) == 1 else 'entries'} long.") - else: - print(" Fallback chain is now empty.") + print(f" Chain is now {_entries(len(chain))} long." if chain else " Fallback chain is now empty.") print() def cmd_fallback_clear(args) -> None: # noqa: ARG001 """Remove all fallback entries (with confirmation).""" - from hermes_cli.config import load_config, save_config + from hermes_cli.config import save_config - config = load_config() - chain = _read_chain(config) - - if not chain: - print() - print(" No fallback providers configured — nothing to clear.") - print() + config, chain = _load_chain(" No fallback providers configured — nothing to clear.") + if chain is None: return print() - print(f" Current fallback chain ({len(chain)} {'entry' if len(chain) == 1 else 'entries'}):") - for i, entry in enumerate(chain, 1): - print(f" {i}. {_format_entry(entry)}") - print() + _print_chain("Current fallback chain", chain) try: resp = input(" Clear all entries? [y/N]: ").strip().lower() except (KeyboardInterrupt, EOFError): @@ -363,15 +340,17 @@ def _numbered_pick(question: str, choices: List[str]) -> Optional[int]: def cmd_fallback(args) -> None: """Top-level dispatcher for ``hermes fallback [subcommand]``.""" sub = getattr(args, "fallback_command", None) - if sub in {None, "", "list", "ls"}: - cmd_fallback_list(args) - elif sub == "add": - cmd_fallback_add(args) - elif sub in {"remove", "rm"}: - cmd_fallback_remove(args) - elif sub == "clear": - cmd_fallback_clear(args) - else: + handler = _SUBCOMMANDS.get(sub) + if handler is None: print(f"Unknown fallback subcommand: {sub}") print("Use one of: list, add, remove, clear") raise SystemExit(2) + handler(args) + + +_SUBCOMMANDS = { + None: cmd_fallback_list, "": cmd_fallback_list, "list": cmd_fallback_list, "ls": cmd_fallback_list, + "add": cmd_fallback_add, + "remove": cmd_fallback_remove, "rm": cmd_fallback_remove, + "clear": cmd_fallback_clear, +} diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index f9b0de4d90..46c9428574 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -1,34 +1,5 @@ -"""Provider/model inventory context — shared substrate for the dashboard -``/api/model/options``, the TUI ``model.options``/``model.save_key`` -JSON-RPC handlers, and the interactive picker. - -Before this module the three call-sites each duplicated: - -1. The 17-LOC config-slice that pulls ``model.{default,name,provider,base_url}``, - ``providers:``, and ``custom_providers:`` out of ``load_config()``; -2. The call into ``list_authenticated_providers`` with the resulting kwargs; -3. (TUI only) a 45-LOC post-pass that merges authenticated rows with - unconfigured ``CANONICAL_PROVIDERS`` rows and emits ``authenticated``/ - ``auth_type``/``key_env``/``warning`` hints for the picker UI. - -Consolidating those three steps into one entry point eliminates two bugs -the duplicates were hiding: - -- The dashboard read ``cfg.get("custom_providers")`` directly, missing the - v12+ keyed ``providers:`` form (which the TUI handled via - ``get_compatible_custom_providers``). -- The TUI's canonical-merge keyed on ``is_user_defined`` to decide - ordering. Section 3 of ``list_authenticated_providers`` sets - ``is_user_defined=True`` even for canonical slugs that appear in the - ``providers:`` config dict, which silently demoted them to the tail of - the picker. ``_reorder_canonical`` keys on slug membership instead. - -Substrate facts (verified May 2026): -- ``list_authenticated_providers`` already populates each row's - ``models`` from the curated catalog (same source as the picker). Do - NOT call ``provider_model_ids()`` per row to "freshen" — that bypasses - curation and pulls in non-agentic models (Nous /models returns ~400 - IDs including TTS, embeddings, rerankers, image/video generators). +"""Provider/model inventory context — shared substrate for the dashboard ``/api/model/options``, the +TUI ``model.options``/``model.save_key`` JSON-RPC handlers, and the interactive picker. """ from __future__ import annotations @@ -42,9 +13,9 @@ from typing import Any, Optional @dataclass(frozen=True) class ConfigContext: - """Snapshot of the model + provider config every inventory caller - needs. Built once via ``load_picker_context()``; the TUI overlays - live agent state via ``with_overrides()`` before passing through. + """Snapshot of the model + provider config every inventory caller needs. Built once via + ``load_picker_context()``; the TUI overlays live agent state via ``with_overrides()`` before + passing through. """ current_provider: str @@ -63,26 +34,23 @@ class ConfigContext: ) -> "ConfigContext": """Return a copy with truthy overrides applied. - Truthy-only because the TUI reads agent attributes that may be - empty strings before an agent is spawned — empties must NOT - clobber the disk-config values. + Truthy-only because the TUI reads agent attributes that may be empty strings before an agent + is spawned — empties must NOT clobber the disk-config values. """ - kw: dict = {} - if current_provider: - kw["current_provider"] = current_provider - if current_model: - kw["current_model"] = current_model - if current_base_url: - kw["current_base_url"] = current_base_url + kw = { + k: v + for k, v in ( + ("current_provider", current_provider), + ("current_model", current_model), + ("current_base_url", current_base_url), + ) + if v + } return replace(self, **kw) if kw else self def load_picker_context() -> ConfigContext: - """Load the disk-config snapshot every consumer needs. - - Replaces the inline 17-LOC config-slice that ``web_server.py`` and - ``tui_gateway/server.py`` (×2 sites) used to do. - """ + """Load the disk-config snapshot every consumer needs.""" from hermes_cli.config import ( coerce_provider_id, get_compatible_custom_providers, @@ -114,6 +82,14 @@ def load_picker_context() -> ConfigContext: ) +def _slug(row: dict) -> str: + return str(row.get("slug") or "").strip().lower() + + +def _without_slug(rows: list[dict], slug: str) -> list[dict]: + return [r for r in rows if _slug(r) != slug] + + # ─── Public: payload builder ──────────────────────────────────────────── @@ -134,60 +110,11 @@ def build_models_payload( for_picker: bool = False, max_models: int | None = None, ) -> dict: - """Build the ``{providers, model, provider}`` shape every consumer - needs from a single substrate call. + """Build the ``{providers, model, provider}`` shape every consumer needs from a single substrate call. - Flags: - - ``explicit_only``: keep only providers the user explicitly configured - (current provider, providers from config, or providers backed by - provider-specific env vars). This hides ambient / auto-seeded - credentials from desktop chat pickers. - - ``include_unconfigured``: append ``CANONICAL_PROVIDERS`` rows that - ``list_authenticated_providers`` didn't emit (TUI uses this to show - the full provider universe in the picker). - - ``picker_hints``: add ``authenticated``/``auth_type``/``key_env``/ - ``warning`` per row (TUI ``ModelPickerDialog`` shape). - - ``canonical_order``: reorder canonical-slug rows to - ``CANONICAL_PROVIDERS`` declaration order; truly-custom rows go - last (TUI display order). - - ``pricing``: enrich each row with formatted per-model pricing and, - for Nous, ``free_tier``/``unavailable_models`` so the GUI picker can - show $/Mtok columns and gate paid models on free accounts — - mirroring the ``hermes model`` CLI picker. Adds network calls - (pricing fetch + Nous tier check); only set for interactive pickers. - - ``capabilities``: add a per-row ``capabilities`` map - ``{model: {fast, reasoning}}`` so pickers can gate the model-options - controls (fast toggle / reasoning) to what each model actually - supports, instead of offering knobs the backend would reject. - - ``featured``: add a per-row ``featured_models`` list — the newest few - models per lab (by models.dev release_date, ranked within the row's own - models; see ``_FEATURED_PER_LAB``) for aggregator providers that serve - dozens of models across many labs. Pickers default their visible set to - these; the rest of ``models`` stays reachable via search / show-all. Empty - for single-lab providers (callers fall back to top-N). Derived live from - models.dev — no allowlist. - - ``force_fresh_nous_tier``: bypass the short Nous free-tier cache when - selecting Portal-recommended Nous models and applying tier gating. Keep - this false for UI picker opens; explicit auth/model flows can opt in - when they need freshly-purchased credits to show up immediately. - - ``refresh``: bust the per-provider model-id disk cache so every row - re-fetches its live catalog. Set only for an explicit user-triggered - "refresh models" action; normal picker opens leave it false to stay - snappy on the 1h cache. - - ``probe_custom_providers``: allow saved custom/provider endpoints to - run live ``/models`` discovery while building the payload. GUI picker - opens should leave this false unless the user explicitly refreshes; the - row can still render its configured model immediately, and slow/offline - local endpoints no longer block the dialog. - - ``probe_current_custom_provider``: when ``probe_custom_providers`` is - false, still live-probe the current custom endpoint. This keeps normal - GUI/TUI picker opens fast while making the active custom provider's model - list match the classic CLI picker. - - ``for_picker``: interactive-picker visibility. Keeps providers whose - credential pool exists but is entirely rate-limited (exhausted) in the - list. Rate limits are per-model, so a different model under the same - provider may still work; hiding the provider strands the user. Set for - any surface a human is choosing from, not for programmatic resolution. + Flags: - ``explicit_only``: keep only providers the user explicitly configured (current + provider, providers from config, or providers backed by provider-specific env vars). This hides + ambient / auto-seeded credentials from desktop chat pickers. """ from hermes_cli.model_switch import list_authenticated_providers @@ -214,8 +141,7 @@ def build_models_payload( # through the llamacpp alias -> managed/detected server resolution. local_row = _local_runtime_row(ctx) if local_row is not None: - rows = [r for r in rows if str(r.get("slug", "")).lower() != "llamacpp"] - rows.append(local_row) + rows = _without_slug(rows, "llamacpp") + [local_row] # A live session on the managed server reports provider "custom" # (the resolution seam's generic label for a raw base_url), which # would otherwise materialize a duplicate "Custom endpoint" row @@ -223,17 +149,17 @@ def build_models_payload( # Local row owns the managed server's identity — drop custom rows # that point at the managed endpoint. if local_row.get("is_current"): + staged = set(local_row["models"]) + def _is_managed_custom(row: dict) -> bool: - if str(row.get("slug", "")).lower() != "custom": - return False models = {str(m) for m in (row.get("models") or [])} - return bool(models) and models <= set(local_row["models"]) + return _slug(row) == "custom" and bool(models) and models <= staged rows = [r for r in rows if not _is_managed_custom(r)] moa_row = _moa_provider_row(ctx.current_provider) if moa_row is not None: - rows = [moa_row] + [r for r in rows if str(r.get("slug", "")).lower() != "moa"] + rows = [moa_row] + _without_slug(rows, "moa") if explicit_only: rows = _filter_explicit_provider_rows(rows, ctx) @@ -248,9 +174,7 @@ def build_models_payload( _local_owns_current = bool(local_row and local_row.get("is_current") and (ctx.current_provider or "").lower() == "custom") if not _local_owns_current: - rows = list(rows) + _append_unconfigured_rows( - rows, ctx, current_only=True - ) + rows = list(rows) + _append_unconfigured_rows(rows, ctx, current_only=True) # --- Deduplicate: remove models from aggregators that overlap with # user-defined providers. When a local proxy (e.g. litellm-proxy) @@ -261,44 +185,10 @@ def build_models_payload( # breaking the call. Filtering at the payload level keeps the # aggregator rows honest: they only show models the user can't get # from a more-specific provider. (#45954) - try: - from hermes_cli.providers import is_routing_aggregator as _is_routing_aggregator - except Exception: - _is_routing_aggregator = None # type: ignore[assignment] - - if _is_routing_aggregator is not None: - user_models: set[str] = set() - for row in rows: - if row.get("is_user_defined"): - user_models.update(m.lower() for m in (row.get("models") or [])) - if user_models: - for row in rows: - # A user's own configured provider is never an "aggregator - # duplicate" of itself: user_models is built from these very - # rows, and is_routing_aggregator() reports True for every - # custom:* slug. Without this guard the dedup strips a - # user-defined custom provider's entire model list (all of it - # lives in user_models), emptying its picker row. - if row.get("is_user_defined"): - continue - slug = row.get("slug", "") - # Only strip overlaps from TRUE routing aggregators (OpenRouter, - # custom:* proxies). Flat-namespace resellers (opencode-go / - # opencode-zen) serve every listed model as a first-party model, - # so their rows must keep models that a user's proxy happens to - # share a name with — otherwise a subscription provider's own - # catalog (minimax-m3, glm-5, deepseek-v4-flash, ...) is silently - # gutted in the picker. (#47077) - if not _is_routing_aggregator(slug): - continue - original = row.get("models") or [] - filtered = [m for m in original if m.lower() not in user_models] - if len(filtered) < len(original): - row["models"] = filtered - row["total_models"] = len(filtered) + _strip_aggregator_overlaps(rows) if include_unconfigured: - rows = list(rows) + [r for r in _append_unconfigured_rows(rows, ctx) if str(r.get("slug", "")).lower() != "moa"] + rows = list(rows) + _without_slug(_append_unconfigured_rows(rows, ctx), "moa") if picker_hints: _apply_picker_hints(rows) if canonical_order: @@ -318,6 +208,37 @@ def build_models_payload( } +def _strip_aggregator_overlaps(rows: list[dict]) -> None: + """Drop models from TRUE routing aggregators (OpenRouter, custom:* proxies) that a user-defined + provider also serves, so the picker never lists them under both (#45954). + + A user's own configured provider is never an "aggregator duplicate" of itself: user_models is + built from these very rows, and is_routing_aggregator() reports True for every custom:* slug — + without that guard the dedup would empty a user-defined custom provider's row. Flat-namespace + resellers (opencode-go / opencode-zen) serve every listed model first-party and must keep models + that a user's proxy happens to share a name with (#47077). + """ + try: + from hermes_cli.providers import is_routing_aggregator + except Exception: + return + + user_models: set[str] = set() + for row in rows: + if row.get("is_user_defined"): + user_models.update(m.lower() for m in (row.get("models") or [])) + if not user_models: + return + for row in rows: + if row.get("is_user_defined") or not is_routing_aggregator(row.get("slug", "")): + continue + original = row.get("models") or [] + filtered = [m for m in original if m.lower() not in user_models] + if len(filtered) < len(original): + row["models"] = filtered + row["total_models"] = len(filtered) + + def build_model_options_payload( ctx: ConfigContext, *, @@ -327,13 +248,12 @@ def build_model_options_payload( ) -> dict: """Build the shared API-server/dashboard/TUI model-options payload. - This wraps ``build_models_payload`` with the stable picker shape and the - safe custom-provider probe policy used for normal GUI/TUI opens: + This wraps ``build_models_payload`` with the stable picker shape and the safe custom-provider + probe policy used for normal GUI/TUI opens: - - normal open: probe only the current custom provider so offline saved - endpoints do not block the picker - - explicit refresh: probe every custom provider while busting the model - cache so live catalogs repopulate fully + - normal open: probe only the current custom provider so offline saved endpoints do not block + the picker - explicit refresh: probe every custom provider while busting the model cache so live + catalogs repopulate fully """ refresh = bool(refresh) return build_models_payload( @@ -363,32 +283,15 @@ def build_aux_picker_rows( ) -> list[dict]: """Provider rows for any auxiliary-task picker (vision, compression, …). - THE entry point for every aux picker — present and future. Call this - instead of ``list_authenticated_providers()`` directly. + - user-defined ``providers:`` and saved ``custom_providers:`` entries - + ``model_catalog.excluded_providers`` honoured, matching ``/model`` - exhausted-credential-pool + providers stay visible (``for_picker``) - the active custom endpoint is probed, offline saved + ones are not, so the picker never blocks on a dead local server - Aux pickers kept re-deriving their own kwargs and each one silently - dropped a different slice of the user's configuration. Two independent - contributor PRs landed against the same two call sites for exactly this: - #52642 (user ``providers:`` / ``custom_providers:`` entries never - appeared) and #66624 (providers with an exhausted credential pool were - hidden). Both were per-site kwarg patches, so the next aux picker would - have reintroduced the same gap. Routing through one function makes the - correct behaviour the default that a new caller cannot forget: - - - user-defined ``providers:`` and saved ``custom_providers:`` entries - - ``model_catalog.excluded_providers`` honoured, matching ``/model`` - - exhausted-credential-pool providers stay visible (``for_picker``) - - the active custom endpoint is probed, offline saved ones are not, so - the picker never blocks on a dead local server - - The virtual ``moa`` row is excluded: auxiliary tasks must not run the - MoA reference fan-out, and ``auxiliary_client`` unwraps a ``moa`` - provider to its aggregator slot anyway (see ``_resolve_auto``), so - offering it here would be a choice silently rewritten behind the user's - back. Mirrors the same filter in ``hermes_cli/moa_cmd.py``. - - Rows are the standard ``list_authenticated_providers`` shape. Pair with - :func:`format_aux_picker_entries` to render them. + The virtual ``moa`` row is excluded: auxiliary tasks must not run the MoA reference fan-out, and + ``auxiliary_client`` unwraps a ``moa`` provider to its aggregator slot anyway (see + ``_resolve_auto``), so offering it here would be a choice silently rewritten behind the user's + back. """ ctx = load_picker_context().with_overrides( current_provider=current_provider, @@ -402,7 +305,7 @@ def build_aux_picker_rows( probe_current_custom_provider=True, max_models=max_models, )["providers"] - return [r for r in rows if str(r.get("slug") or "").strip().lower() != "moa"] + return _without_slug(rows, "moa") def format_aux_picker_entries( @@ -413,14 +316,9 @@ def format_aux_picker_entries( ) -> list[tuple[str, str, list[str]]]: """Render aux-picker rows as ``(slug, label, models)`` menu entries. - Owns the label text and the ``← current`` marker so every aux picker - presents providers identically. Callers add their own leading/trailing - entries (``auto``, ``Custom endpoint``, ``Back``) around this list. - - A custom endpoint set via a raw ``base_url`` is "current" only through - that URL — never through a provider slug — so when ``current_base_url`` - is set no provider row is marked, matching the pre-existing behaviour of - both call sites. + A custom endpoint set via a raw ``base_url`` is "current" only through that URL — never through + a provider slug — so when ``current_base_url`` is set no provider row is marked, matching the + pre-existing behaviour of both call sites. """ entries: list[tuple[str, str, list[str]]] = [] current_slug = str(current_provider or "").strip().lower() @@ -442,9 +340,9 @@ def format_aux_picker_entries( def _reasoning_catalog_reader(slug: str): """Per-model reasoning-capability reader for aggregators that publish one. - Cache-only — building the picker payload must never block on HTTP. A cold - cache warms in the background so the next open is accurate; until then the - model reports no restriction and the UI offers the full scale. + Cache-only — building the picker payload must never block on HTTP. A cold cache warms in the + background so the next open is accurate; until then the model reports no restriction and the UI + offers the full scale. """ try: from hermes_cli.models import ( @@ -456,37 +354,24 @@ def _reasoning_catalog_reader(slug: str): except Exception: return None - if slug == "nous": - warm_nous_reasoning_caps_async() - return nous_model_reasoning_capabilities - if slug == "openrouter": - warm_openrouter_reasoning_caps_async() - return openrouter_model_reasoning_capabilities - return None + readers = { + "nous": (warm_nous_reasoning_caps_async, nous_model_reasoning_capabilities), + "openrouter": (warm_openrouter_reasoning_caps_async, openrouter_model_reasoning_capabilities), + } + if slug not in readers: + return None + warm, read = readers[slug] + warm() + return read def _apply_capabilities(rows: list[dict]) -> None: """Attach a ``{model: {fast, reasoning, ...}}`` map to each provider row. - `fast` mirrors ``model_supports_fast_mode`` (the same gate the runtime - enforces). `reasoning` comes from the models.dev catalog when known and - defaults to True otherwise — the effort dial is broadly accepted and a - no-op on models that ignore it, whereas hiding it from a capable-but- - uncatalogued model is the worse failure. - - Aggregators that publish per-model reasoning detail add - `can_disable_reasoning`, False on reasoning-mandatory routes whose upstream - answers a disable with HTTP 400. Omitted when the catalog doesn't say, - which the UI reads as "no restriction known". Such a catalog also overrides - `reasoning` itself when it reports a route that takes no reasoning - parameter — a definitive negative from the provider actually serving the - model outranks the models.dev inference. - - The catalog's `supported_efforts` list is deliberately NOT forwarded: it - under-reports. The Portal accepts and honors levels a route doesn't - advertise (``z-ai/glm-5.3`` publishes ``max, high, low`` yet serves - ``minimal`` at its lowest thinking), so filtering the picker by that list - would hide levels that demonstrably work. + ``fast`` mirrors the runtime gate. ``reasoning`` defaults True when the catalog is silent: + the dial is a no-op on models that ignore it, and hiding it from a capable model is worse. + A serving aggregator's per-model detail overrides models.dev (adds ``can_disable_reasoning``). + ``supported_efforts`` is deliberately NOT forwarded -- it under-reports levels that work. """ from hermes_cli.models import model_supports_fast_mode @@ -543,20 +428,10 @@ _FEATURED_PER_LAB = 5 def _apply_featured(rows: list[dict]) -> None: """Attach a ``featured_models`` shortlist to each aggregator provider row. - Aggregator providers (nous, openrouter) serve dozens of models across many - labs, so a flat "top-N" default would drop whole labs from the picker. - Instead we surface the ``_FEATURED_PER_LAB`` newest models per lab (the - vendor segment of a ``vendor/model`` id), ranked by models.dev - ``release_date`` among that row's OWN models — never against the current - date, so the choice is stable as models age. Same-date ties (and labs whose - models lack a date) fall back to the row's curated order, which is already - flagship-first, so a lab keeps its headliners rather than an arbitrary slice. - - Derived live from the models.dev catalog already loaded on this path (same - source as pricing/capabilities) — there is no hand-maintained allowlist to - keep in sync. Non-aggregator providers (a single lab, local endpoints, - custom proxies) get an empty list and callers fall back to their existing - top-N behaviour; splitting one lab into a shortlist would just hide models. + Aggregators serve many labs, so a flat top-N would drop whole labs; instead surface the + newest ``_FEATURED_PER_LAB`` per vendor, ranked by models.dev ``release_date`` within the + row's own models (never vs. today, so the choice is stable). Ties fall back to the curated + flagship-first order. Non-aggregators get an empty list and keep top-N behaviour. """ try: from agent.models_dev import get_model_info @@ -602,13 +477,10 @@ def _apply_featured(rows: list[dict]) -> None: def _apply_custom_aliases(rows: list[dict]) -> None: """Attach the accepted identity set to each user-defined provider row. - A session's ``model.options`` reports the canonical ``custom:`` - identity (via ``canonical_custom_identity``), while catalog rows carry - the bare config key as ``slug``. GUI pickers compare the two to decide - which row is active; exact equality never matches for custom providers - (#87035). Exposing ``aliases`` — every current and legacy spelling from - :func:`hermes_cli.providers.custom_provider_aliases` — lets the frontend - do a membership check instead. + A session's ``model.options`` reports the canonical ``custom:`` identity (via + ``canonical_custom_identity``), while catalog rows carry the bare config key as ``slug``. GUI + pickers compare the two to decide which row is active; exact equality never matches for custom + providers (#87035). """ from hermes_cli.providers import custom_provider_aliases @@ -628,6 +500,28 @@ def _apply_custom_aliases(rows: list[dict]) -> None: # ─── Internal: row post-processing ────────────────────────────────────── +def _provider_auth_hint(slug: str) -> tuple[str, str]: + """``(auth_type, key_env)`` for a canonical provider (``("api_key", "")`` when unregistered).""" + from hermes_cli.auth import PROVIDER_REGISTRY + + cfg = PROVIDER_REGISTRY.get(slug) + auth_type = cfg.auth_type if cfg else "api_key" + key_env = cfg.api_key_env_vars[0] if (cfg and cfg.api_key_env_vars) else "" + return auth_type, key_env + + +def _canonical_row(entry, cur: str, **extra: Any) -> dict: + from hermes_cli.models import _PROVIDER_LABELS + + return { + "slug": entry.slug, + "name": _PROVIDER_LABELS.get(entry.slug, entry.label), + "is_current": entry.slug.lower() == cur, + "is_user_defined": False, + **extra, + } + + def _append_unconfigured_rows( rows: list[dict], ctx: ConfigContext, @@ -636,13 +530,12 @@ def _append_unconfigured_rows( ) -> list[dict]: """Build fallback rows for canonical providers missing from ``rows``. - Most missing canonical providers become empty setup skeletons. The one - exception is the *current* configured provider: if config.yaml still points - at it but credentials are presently unavailable, keep a visible row carrying - the saved model so GUI pickers don't silently snap to some other provider. + Most missing canonical providers become empty setup skeletons. The one exception is the + *current* configured provider: if config.yaml still points at it but credentials are presently + unavailable, keep a visible row carrying the saved model so GUI pickers don't silently snap to + some other provider. """ - from hermes_cli.auth import PROVIDER_REGISTRY - from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_LABELS + from hermes_cli.models import CANONICAL_PROVIDERS seen = {r["slug"].lower() for r in rows} cur = (ctx.current_provider or "").lower() @@ -654,13 +547,7 @@ def _append_unconfigured_rows( if current_only and entry.slug.lower() != cur: continue if entry.slug.lower() == cur: - cfg = PROVIDER_REGISTRY.get(entry.slug) - auth_type = cfg.auth_type if cfg else "api_key" - key_env = ( - cfg.api_key_env_vars[0] - if (cfg and cfg.api_key_env_vars) - else "" - ) + auth_type, key_env = _provider_auth_hint(entry.slug) warning = ( f"Configured provider missing usable credentials; paste {key_env} to reactivate. " "Showing the saved model only." @@ -668,47 +555,27 @@ def _append_unconfigured_rows( else "Configured provider is not authenticated; run `hermes model` to reactivate. " "Showing the saved model only." ) - extras.append( - { - "slug": entry.slug, - "name": _PROVIDER_LABELS.get(entry.slug, entry.label), - "is_current": True, - "is_user_defined": False, - "models": [cur_model] if cur_model else [], - "total_models": 1 if cur_model else 0, - "source": "configured-current", - "authenticated": False, - "auth_type": auth_type, - "key_env": key_env, - "warning": warning, - } - ) + extras.append(_canonical_row( + entry, cur, + models=[cur_model] if cur_model else [], + total_models=1 if cur_model else 0, + source="configured-current", + authenticated=False, + auth_type=auth_type, + key_env=key_env, + warning=warning, + )) continue - extras.append( - { - "slug": entry.slug, - "name": _PROVIDER_LABELS.get(entry.slug, entry.label), - "is_current": entry.slug.lower() == cur, - "is_user_defined": False, - "models": [], - "total_models": 0, - "source": "canonical", - } - ) + extras.append(_canonical_row(entry, cur, models=[], total_models=0, source="canonical")) return extras def _anthropic_oauth_credentials_present() -> bool: """True when the user explicitly authenticated Anthropic via OAuth. - Two deliberate flows leave no trace in active_provider / - model.provider / API-key env vars: Hermes' own Anthropic device flow - (token in auth.json) and a Claude Code login (~/.claude/.credentials.json). - ``list_authenticated_providers`` already accepts both readers as real - credentials when discovering rows; this mirrors that acceptance so the - desktop explicit-only filter does not silently drop a provider the user - deliberately signed into. Unlike ambient CLI tokens (gh -> copilot), - an OAuth access token only exists after an interactive login. + Two deliberate flows leave no trace in active_provider / model.provider / API-key env vars: + Hermes' own Anthropic device flow (token in auth.json) and a Claude Code login + (~/.claude/.credentials.json). """ try: from agent.anthropic_adapter import ( @@ -716,11 +583,10 @@ def _anthropic_oauth_credentials_present() -> bool: read_hermes_oauth_credentials, ) - hermes_creds = read_hermes_oauth_credentials() or {} - if hermes_creds.get("accessToken"): - return True - cc_creds = read_claude_code_credentials() or {} - if cc_creds.get("accessToken"): + if any( + (read() or {}).get("accessToken") + for read in (read_hermes_oauth_credentials, read_claude_code_credentials) + ): return True except Exception: return False @@ -748,66 +614,57 @@ def _anthropic_oauth_credentials_present() -> bool: def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list[dict]: """Keep only rows backed by explicit user configuration. - ``list_authenticated_providers`` intentionally discovers ambient / auto- - seeded credentials (for example GitHub CLI -> Copilot). Desktop chat model - pickers want the narrower subset the user explicitly configured for Hermes. + ``list_authenticated_providers`` also discovers ambient credentials (e.g. GitHub CLI -> + Copilot); Desktop chat pickers want only what the user configured for Hermes. """ from hermes_cli.auth import is_provider_explicitly_configured current_slug = str(ctx.current_provider or "").strip().lower() - kept: list[dict] = [] - for row in rows: - slug = str(row.get("slug", "")).strip().lower() - if not slug: - continue - if row.get("is_user_defined"): - kept.append(row) - continue - if current_slug and slug == current_slug: - kept.append(row) - continue - if row.get("source") == "local-runtime": + + def _is_explicit(row: dict, slug: str) -> bool: + if ( + row.get("is_user_defined") + or (current_slug and slug == current_slug) # Managed local models are explicit configuration by existence: # the user downloaded gigabytes into the machine-scoped models # dir. There is deliberately no config credential to find # (credential is reachability), so without this clause the row # only survives on the profile where Use was last clicked — # every other profile loses local models from its picker. - kept.append(row) - continue + or row.get("source") == "local-runtime" + ): + return True if slug == "moa": # MoA is a virtual routing mode, not an independently configured # provider. Hide it from explicit-only pickers unless it is the # current provider (handled above) or the user explicitly wrote an # enabled MoA preset into config.yaml. Use raw config so the # DEFAULT_CONFIG preset does not make every desktop picker show MoA. - if _raw_config_has_enabled_moa_preset(): - kept.append(row) - continue - if _provider_is_keyless(slug): + return _raw_config_has_enabled_moa_preset() + return ( # Keyless providers (opencode-free) require no configuration at # all — there is nothing to "explicitly configure", and hiding # them would defeat their purpose (zero-setup discoverability). - kept.append(row) - continue - if slug == "anthropic" and _anthropic_oauth_credentials_present(): + _provider_is_keyless(slug) # Anthropic OAuth logins (Hermes device flow / Claude Code) are # deliberate sign-ins that leave no trace in active_provider, # model.provider, or API-key env vars. The strict gate below # would drop the row even though list_authenticated_providers # just accepted those same credentials when building it. - kept.append(row) - continue - if _external_process_signed_in(slug): - # External-process providers (copilot-acp) authenticate through - # their own CLI (`copilot login`), which — like the Anthropic - # OAuth case above — leaves no trace in active_provider, - # model.provider, or env vars. Verified CLI credentials are a - # deliberate sign-in; without this the desktop picker drops the - # row the picker-discovery side just accepted. - kept.append(row) - continue - if is_provider_explicitly_configured(slug): + or (slug == "anthropic" and _anthropic_oauth_credentials_present()) + # External-process providers (copilot-acp) authenticate through their + # own CLI (`copilot login`) — same class as Anthropic OAuth above: + # verified CLI credentials are a deliberate sign-in with no trace in + # active_provider/model.provider/env, so keep the row the + # picker-discovery side just accepted. + or _external_process_signed_in(slug) + or is_provider_explicitly_configured(slug) + ) + + kept: list[dict] = [] + for row in rows: + slug = str(row.get("slug", "")).strip().lower() + if slug and _is_explicit(row, slug): kept.append(row) return kept @@ -840,10 +697,10 @@ def _provider_is_keyless(slug: str) -> bool: def _raw_config_has_enabled_moa_preset() -> bool: """Return True when the user's raw config explicitly enables MoA. - ``load_config()`` includes ``DEFAULT_CONFIG["moa"].presets.default`` for - everyone. Explicit-only model pickers must not treat that default as a user - choice, but they should keep MoA visible once the user has saved at least - one enabled preset (or an older flat MoA config) in their own config.yaml. + ``load_config()`` includes ``DEFAULT_CONFIG["moa"].presets.default`` for everyone. Explicit-only + model pickers must not treat that default as a user choice, but they should keep MoA visible + once the user has saved at least one enabled preset (or an older flat MoA config) in their own + config.yaml. """ try: from hermes_cli.config import read_raw_config @@ -860,14 +717,11 @@ def _raw_config_has_enabled_moa_preset() -> bool: presets = moa.get("presets") if isinstance(presets, dict): - for name, preset in presets.items(): - if not str(name or "").strip(): - continue - if not isinstance(preset, dict): - return True - if preset.get("enabled", True): - return True - return False + return any( + not isinstance(preset, dict) or preset.get("enabled", True) + for name, preset in presets.items() + if str(name or "").strip() + ) legacy_keys = { "reference_models", @@ -882,15 +736,7 @@ def _raw_config_has_enabled_moa_preset() -> bool: def _apply_picker_hints(rows: list[dict]) -> None: - """Add ``authenticated``/``auth_type``/``key_env``/``warning`` per row. - - Mutates ``rows`` in-place. Rows already from - ``list_authenticated_providers`` are marked ``authenticated=True``; - the unconfigured skeleton rows from ``_append_unconfigured_rows`` get - the picker's setup-hint shape. - """ - from hermes_cli.auth import PROVIDER_REGISTRY - + """Add ``authenticated``/``auth_type``/``key_env``/``warning`` per row.""" for row in rows: if "authenticated" in row: continue @@ -903,13 +749,7 @@ def _apply_picker_hints(rows: list[dict]) -> None: row["authenticated"] = not is_skeleton if not is_skeleton or row.get("is_user_defined"): continue - cfg = PROVIDER_REGISTRY.get(row["slug"]) - auth_type = cfg.auth_type if cfg else "api_key" - key_env = ( - cfg.api_key_env_vars[0] - if (cfg and cfg.api_key_env_vars) - else "" - ) + auth_type, key_env = _provider_auth_hint(row["slug"]) row["auth_type"] = auth_type row["key_env"] = key_env row["warning"] = ( @@ -920,14 +760,11 @@ def _apply_picker_hints(rows: list[dict]) -> None: def _reorder_canonical(rows: list[dict]) -> list[dict]: - """Canonical slugs in ``CANONICAL_PROVIDERS`` declaration order; - truly-custom rows last. + """Canonical slugs in ``CANONICAL_PROVIDERS`` declaration order; truly-custom rows last. - Keys on slug membership, NOT ``is_user_defined`` — section 3 of - ``list_authenticated_providers`` sets ``is_user_defined=True`` on - rows from the ``providers:`` config dict even when the slug is - canonical. Keying on the flag would silently demote canonical - providers configured via the new keyed schema. + Keys on slug membership, NOT ``is_user_defined``: rows from the ``providers:`` config dict + carry that flag even for canonical slugs, so keying on it would demote canonical providers + configured via the keyed schema. """ from hermes_cli.models import CANONICAL_PROVIDERS @@ -947,20 +784,11 @@ def _apply_pricing( ) -> None: """Enrich each provider row with per-model pricing + Nous tier gating. - Mutates ``rows`` in-place. For every row whose provider supports live - pricing (openrouter / nous / novita) adds:: + row["pricing"] = {model_id: {"input": "$3.00", "output": "$15.00", "cache": "$0.30" | None, + "free": bool}} - row["pricing"] = {model_id: {"input": "$3.00", "output": "$15.00", - "cache": "$0.30" | None, "free": bool}} - - For Nous additionally adds:: - - row["free_tier"] = bool # current account is free-tier - row["unavailable_models"] = [...] # paid models a free user can't pick - - Prices are pre-formatted via ``_format_price_per_mtok`` so the GUI just - renders strings — identical formatting to the CLI picker. All failures - are swallowed (best-effort): a row simply gets no ``pricing`` key. + row["free_tier"] = bool # current account is free-tier row["unavailable_models"] = [...] # paid + models a free user can't pick """ from hermes_cli.models import ( _format_price_per_mtok, @@ -995,14 +823,12 @@ def _apply_pricing( cache_raw = p.get("input_cache_read", "") inp = _format_price_per_mtok(inp_raw) if inp_raw != "" else "" out = _format_price_per_mtok(out_raw) if out_raw != "" else "" - cache = _format_price_per_mtok(cache_raw) if cache_raw else None - # A model is "free" when both input and output cost nothing. - is_free = inp == "free" and (out == "free" or out == "") entry: dict = { "input": inp, "output": out, - "cache": cache, - "free": is_free, + "cache": _format_price_per_mtok(cache_raw) if cache_raw else None, + # A model is "free" when both input and output cost nothing. + "free": inp == "free" and out in ("free", ""), } # Sale chrome is Nous Portal-only. Other providers (OpenRouter, # Novita, …) never get discount_percent / was_* even if a nested @@ -1016,14 +842,9 @@ def _apply_pricing( if sale is not None: discount_percent, was_prompt_raw, was_out_raw = sale entry["discount_percent"] = discount_percent - if was_prompt_raw != "": - entry["was_input"] = _format_price_per_mtok( - was_prompt_raw - ) - if was_out_raw != "": - entry["was_output"] = _format_price_per_mtok( - was_out_raw - ) + for key, was_raw in (("was_input", was_prompt_raw), ("was_output", was_out_raw)): + if was_raw != "": + entry[key] = _format_price_per_mtok(was_raw) formatted[mid] = entry if formatted: @@ -1036,13 +857,11 @@ def _apply_pricing( force_fresh=force_fresh_nous_tier ) row["free_tier"] = bool(nous_free_tier) - if nous_free_tier: - _selectable, unavailable = partition_nous_models_by_tier( - list(models), raw_pricing, free_tier=True - ) - row["unavailable_models"] = unavailable - else: - row["unavailable_models"] = [] + row["unavailable_models"] = ( + partition_nous_models_by_tier(list(models), raw_pricing, free_tier=True)[1] + if nous_free_tier + else [] + ) except Exception: # Tier detection failed — fail open (no gating) so the user # is never blocked from picking a model. @@ -1053,10 +872,9 @@ def _apply_pricing( def _local_runtime_row(ctx: "ConfigContext") -> dict | None: """Build the ``llamacpp`` provider row from staged local models. - Present whenever GGUFs are staged in the managed models directory — - downloaded models must be selectable even before the server is running - (selection starts it via the runtime_provider seam / activate flow). - Returns ``None`` when nothing is staged. + Present whenever GGUFs are staged in the managed models directory — downloaded models must be + selectable even before the server is running (selection starts it via the runtime_provider seam + / activate flow). Returns ``None`` when nothing is staged. """ try: from hermes_cli.local_runtime.bootstrap import staged_model_ids @@ -1103,9 +921,8 @@ def _local_runtime_row(ctx: "ConfigContext") -> dict | None: def _moa_provider_row(current_provider: str = "") -> dict | None: """Build the virtual ``moa`` provider row for model pickers. - Shared by the CLI inventory (:func:`build_models_payload`) and the gateway - picker path (:func:`hermes_cli.model_switch.list_picker_providers`) so the - row shape stays in one place. Returns ``None`` when no MoA presets exist. + Shared by the CLI inventory and the gateway picker path so the row shape lives in one place. + Returns ``None`` when no MoA presets exist. """ try: from hermes_cli.config import load_config diff --git a/hermes_cli/local_runtime/__init__.py b/hermes_cli/local_runtime/__init__.py index 9793970b5f..3186263840 100644 --- a/hermes_cli/local_runtime/__init__.py +++ b/hermes_cli/local_runtime/__init__.py @@ -1,21 +1,9 @@ """Managed llama.cpp runtime. -Hermes downloads, verifies, supervises, and updates one llama-server, and -decides per machine which model build and context window to run. Key -modules: - -- ``binaries`` — resolve/download/verify official llama.cpp release zips - into ``$HERMES_HOME/runtimes/llamacpp//``. -- ``supervisor``— spawn and supervise one llama-server in router mode; - readiness is a touch generation, never health-200 alone. -- ``detect`` — find an already-running llama-server (external or ours). -- ``estimator`` / ``context_policy`` / ``growth`` — price context memory - per architecture and run the window ladder (zero-spill start, grow - toward native max, compress only at the top). -- ``catalog`` / ``presets`` — the curated model list and the per-model - launch flags that carry policy decisions to the router. - -Everything is driven by the ``local_runtime`` section of config.yaml. +- ``binaries`` — resolve/download/verify official llama.cpp release zips into +``$HERMES_HOME/runtimes/llamacpp//``. - ``supervisor``— spawn and supervise one llama-server in +router mode; readiness is a touch generation, never health-200 alone. - ``detect`` — find an +already-running llama-server (external or ours). """ from hermes_cli.local_runtime.binaries import ( # noqa: F401 diff --git a/hermes_cli/local_runtime/binaries.py b/hermes_cli/local_runtime/binaries.py index bcd7d32a1d..ea67131e2e 100644 --- a/hermes_cli/local_runtime/binaries.py +++ b/hermes_cli/local_runtime/binaries.py @@ -1,15 +1,4 @@ -"""Binary acquisition for the managed llama.cpp runtime. - -llama.cpp publishes per-tag assets (rolling ``bNNNN`` tags, no semver). -Backends are dlopen'd plugins, so a runtime = CPU/base zip + backend zip -extracted into one directory, plus the cudart runtime zip on Windows CUDA -(end users have no CUDA toolkit). We pin the tag in config, sha256-verify -every download, and keep the previous tag for rollback (N-1). - -Layout: ``$HERMES_HOME/runtimes/llamacpp///`` -with a ``manifest.json`` recording zips, sha256s, and the verified -llama-server version string. -""" +"""Binary acquisition for the managed llama.cpp runtime.""" from __future__ import annotations @@ -25,7 +14,6 @@ from dataclasses import dataclass, field from pathlib import Path from typing import Callable -from hermes_constants import get_hermes_home logger = logging.getLogger(__name__) @@ -68,12 +56,11 @@ class AssetPlan: def runtimes_root() -> Path: - """Machine-scoped, deliberately NOT profile-scoped. Engine binaries, - presets, and server state describe this machine's hardware and its one - managed server (stable port) — a second profile re-downloading the - engine or fighting over the port would be the bug. Profile-scoped - things (which model is the default, enabled) live in each profile's - config.yaml as ever.""" + """Machine-scoped, deliberately NOT profile-scoped. Engine binaries, presets, and server state + describe this machine's hardware and its one managed server (stable port) — a second profile + re-downloading the engine or fighting over the port would be the bug. Profile-scoped things + (which model is the default, enabled) live in each profile's config.yaml as ever. + """ from hermes_constants import get_default_hermes_root return get_default_hermes_root() / "runtimes" / "llamacpp" @@ -108,9 +95,9 @@ def installed_tags() -> list[str]: def _host_os_arch() -> tuple[str, str]: """(os, arch) normalized to release-asset vocabulary. - PITFALL: PROCESSOR_ARCHITECTURE lies under x64 emulation on - ARM64 Windows. platform.machine() reads the same env on some Pythons, so - on Windows prefer PROCESSOR_IDENTIFIER's text when present. + PITFALL: PROCESSOR_ARCHITECTURE lies under x64 emulation on ARM64 Windows. platform.machine() + reads the same env on some Pythons, so on Windows prefer PROCESSOR_IDENTIFIER's text when + present. """ system = platform.system().lower() os_name = {"windows": "win", "darwin": "macos", "linux": "ubuntu"}.get(system, system) @@ -118,8 +105,8 @@ def _host_os_arch() -> tuple[str, str]: arch = "arm64" if machine in ("arm64", "aarch64") else "x64" if os_name == "win": import os as _os - ident = _os.environ.get("PROCESSOR_IDENTIFIER", "") - if "armv8" in ident.lower() or "arm " in ident.lower(): + ident = _os.environ.get("PROCESSOR_IDENTIFIER", "").lower() + if "armv8" in ident or "arm " in ident: arch = "arm64" return os_name, arch @@ -140,57 +127,52 @@ def select_backend(gpu_vendor: str | None, os_name: str | None = None) -> str: return "cpu" +# Per-OS (human label, {backend: asset-name templates}). Windows CUDA pairs the +# runtime zip with its cudart zip; ubuntu ships tarballs, win ships zips. +_ASSET_TEMPLATES = { + "ubuntu": ("linux", { + "vulkan": ["llama-{tag}-bin-ubuntu-vulkan-{arch}.tar.gz"], + "hip": ["llama-{tag}-bin-ubuntu-rocm-7.2-{arch}.tar.gz"], + "cpu": ["llama-{tag}-bin-ubuntu-{arch}.tar.gz"], + }), + "win": ("windows", { + "cuda": ["llama-{tag}-bin-win-cuda-{cuda_ver}-{arch}.zip", + "cudart-llama-bin-win-cuda-{cuda_ver}-{arch}.zip"], + "vulkan": ["llama-{tag}-bin-win-vulkan-x64.zip"], + "hip": ["llama-{tag}-bin-win-hip-radeon-x64.zip"], + "cpu": ["llama-{tag}-bin-win-cpu-{arch}.zip"], + }), +} + + def resolve_assets(tag: str, backend: str, os_name: str | None = None, arch: str | None = None) -> AssetPlan: """Compose the asset list for (tag, backend, platform). - Raises BinaryResolutionError for combinations the release does not ship - (a platform/backend pair upstream publishes no artifact for). Callers - fall back down the backend ladder: cuda -> vulkan -> cpu. + Raises BinaryResolutionError for platform/backend pairs the release ships no artifact for; + callers fall back down the backend ladder cuda -> vulkan -> cpu. """ host_os, host_arch = _host_os_arch() os_name = os_name or host_os arch = arch or host_arch - plan = AssetPlan(tag=tag, backend=backend) - if os_name == "macos": # macOS tarballs are unified (Metal built in). - plan.assets = [f"llama-{tag}-bin-macos-{arch}.tar.gz"] - return plan - - if os_name == "ubuntu": - if backend == "cuda": - # No prebuilt Linux CUDA zips at current tags — Linux CUDA users - # build from source or use vulkan; resolver is honest about it. - raise BinaryResolutionError( - f"no prebuilt linux CUDA asset at {tag}; use vulkan/cpu or a source build") - suffix = {"vulkan": f"vulkan-{arch}", "hip": f"rocm-7.2-{arch}", - "cpu": arch}.get(backend) - if suffix is None: - raise BinaryResolutionError(f"unsupported linux backend {backend}") - plan.assets = [f"llama-{tag}-bin-ubuntu-{suffix}.tar.gz"] - return plan - - if os_name == "win": - if backend == "cuda": - cuda_ver = _WIN_CUDA_VERSION_ARM64 if arch == "arm64" else _WIN_CUDA_VERSION - plan.assets = [ - f"llama-{tag}-bin-win-cuda-{cuda_ver}-{arch}.zip", - f"cudart-llama-bin-win-cuda-{cuda_ver}-{arch}.zip", - ] - elif backend == "vulkan": - if arch == "arm64": - raise BinaryResolutionError(f"no win-vulkan-arm64 asset at {tag}") - plan.assets = [f"llama-{tag}-bin-win-vulkan-x64.zip"] - elif backend == "hip": - plan.assets = [f"llama-{tag}-bin-win-hip-radeon-x64.zip"] - elif backend == "cpu": - plan.assets = [f"llama-{tag}-bin-win-cpu-{arch}.zip"] - else: - raise BinaryResolutionError(f"unsupported windows backend {backend}") - return plan - - raise BinaryResolutionError(f"unsupported platform {os_name}-{arch}") + return AssetPlan(tag, backend, [f"llama-{tag}-bin-macos-{arch}.tar.gz"]) + if os_name not in _ASSET_TEMPLATES: + raise BinaryResolutionError(f"unsupported platform {os_name}-{arch}") + if os_name == "ubuntu" and backend == "cuda": + # No prebuilt Linux CUDA zips at current tags — Linux CUDA users + # build from source or use vulkan; resolver is honest about it. + raise BinaryResolutionError( + f"no prebuilt linux CUDA asset at {tag}; use vulkan/cpu or a source build") + if os_name == "win" and backend == "vulkan" and arch == "arm64": + raise BinaryResolutionError(f"no win-vulkan-arm64 asset at {tag}") + label, templates = _ASSET_TEMPLATES[os_name] + if backend not in templates: + raise BinaryResolutionError(f"unsupported {label} backend {backend}") + cuda_ver = _WIN_CUDA_VERSION_ARM64 if arch == "arm64" else _WIN_CUDA_VERSION + return AssetPlan(tag, backend, [t.format(tag=tag, arch=arch, cuda_ver=cuda_ver) + for t in templates[backend]]) def _sha256(path: Path) -> str: @@ -227,26 +209,21 @@ def _extract(archive: Path, dest: Path, """Extract member by member so ``progress(done, total)`` can tick in uncompressed bytes — big archives take real time on laptop disks.""" if archive.name.endswith(".zip"): - with zipfile.ZipFile(archive) as z: - members = z.infolist() - total = sum(m.file_size for m in members) - done = 0 - for m in members: - z.extract(m, dest) - done += m.file_size - if progress is not None: - progress(done, total) + opener, list_members, size = zipfile.ZipFile, "infolist", "file_size" + kwargs = {} else: import tarfile - with tarfile.open(archive) as t: - members = t.getmembers() - total = sum(m.size for m in members) - done = 0 - for m in members: - t.extract(m, dest, filter="data") - done += m.size - if progress is not None: - progress(done, total) + opener, list_members, size = tarfile.open, "getmembers", "size" + kwargs = {"filter": "data"} + with opener(archive) as ar: + members = getattr(ar, list_members)() + total = sum(getattr(m, size) for m in members) + done = 0 + for m in members: + ar.extract(m, dest, **kwargs) + done += getattr(m, size) + if progress is not None: + progress(done, total) def server_binary(install_dir: Path) -> Path: @@ -295,13 +272,10 @@ def ensure_runtime_installed(tag: str, backend: str, progress: "Callable[[str, int, int, str], None] | None" = None) -> Path: """Idempotent: resolve, download, verify, extract, version-check. - ``expected_sha256`` maps asset name -> hash when the catalog pins them; - without pins the computed hash is recorded in the manifest (trust on - first download, verified on every reinstall). - ``progress(stage, done_bytes, total_bytes, label)`` ticks through the - slow parts — stage is "download" | "extract" | "verify", label is the - asset counter ("1/2") when the plan has several archives. - Returns the install directory containing llama-server. + ``expected_sha256`` pins hashes per asset when the catalog has them; without pins the computed + hash is recorded in the manifest (trust on first download, verified on every reinstall). + ``progress(stage, done, total, label)`` ticks through download/extract/verify. Returns the + install dir containing llama-server. """ plan = resolve_assets(tag, backend) install_dir = plan.install_dir diff --git a/hermes_cli/local_runtime/bootstrap.py b/hermes_cli/local_runtime/bootstrap.py index 79402142ae..d204b48ee3 100644 --- a/hermes_cli/local_runtime/bootstrap.py +++ b/hermes_cli/local_runtime/bootstrap.py @@ -1,25 +1,23 @@ -"""Bootstrap for the managed runtime: config -> installed binaries -> -running supervised server. +"""Bootstrap for the managed runtime: config -> installed binaries -> running supervised server. -One public call, ``ensure_local_runtime(config)``, safe to call at any -session start: -- disabled or already-running (state file answers /health) -> no-op -- enabled -> install binaries if missing (idempotent), spawn supervisor +One public call, ``ensure_local_runtime(config)``, safe to call at any session start: - disabled or +already-running (state file answers /health) -> no-op - enabled -> install binaries if missing +(idempotent), spawn supervisor -Kept import-light: callers gate on config before importing this module so -sessions with local_runtime disabled never pay the import. +Kept import-light: callers gate on config before importing this module so sessions with +local_runtime disabled never pay the import. """ from __future__ import annotations import logging import os +import re +import signal import subprocess import time from pathlib import Path -from hermes_constants import get_hermes_home # noqa: F401 — config paths - from hermes_cli.local_runtime.binaries import runtimes_root logger = logging.getLogger(__name__) @@ -28,10 +26,10 @@ _SUPERVISOR = None # process-wide singleton; one router per Hermes process def _detect_gpu_vendor() -> str | None: - """Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi - (resolved by the hardware probe's PATH-independent ladder — a stripped - service PATH must not demote an NVIDIA box to vulkan/cpu); anything - else defers to select_backend's fallback ladder.""" + """Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi (resolved by the hardware + probe's PATH-independent ladder — a stripped service PATH must not demote an NVIDIA box to + vulkan/cpu); anything else defers to select_backend's fallback ladder. + """ from hermes_cli.local_runtime.hardware import _nvidia_smi_path smi = _nvidia_smi_path() @@ -65,13 +63,10 @@ def assets_dir() -> Path: def staged_models() -> "list[Path]": - """Servable staged models: single-file GGUFs count when present; a - split GGUF counts once, by its first part, and only when EVERY part - is on disk — a mid-download split is not servable and must not - surface anywhere as a model. Continuation parts and assets/ never - count.""" - import re - + """Servable staged models: single-file GGUFs count when present; a split GGUF counts once, by its + first part, and only when EVERY part is on disk — a mid-download split is not servable and must + not surface anywhere as a model. Continuation parts and assets/ never count. + """ part = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$") files = sorted(models_dir().glob("*.gguf")) names = {p.name for p in files} @@ -92,8 +87,6 @@ def staged_models() -> "list[Path]": def staged_model_ids() -> "list[str]": - import re - return [re.sub(r"-\d{5}-of-\d{5}$", "", p.stem) for p in staged_models()] @@ -110,21 +103,18 @@ def _presets_stale() -> bool: def _stop_state_server(state: dict) -> None: - """Best-effort stop of the server the state file points at (an - incumbent this process doesn't supervise). The state pid is ours by - contract — the file only ever describes the managed server.""" + """Best-effort stop of the server the state file points at (an incumbent this process doesn't + supervise). The state pid is ours by contract — the file only ever describes the managed server. + """ from hermes_cli.local_runtime.endpoint import _pid_alive - pid = state.get("pid") try: - pid = int(pid) + pid = int(state.get("pid")) except (TypeError, ValueError): return if pid <= 0: return try: - import signal - os.kill(pid, signal.SIGTERM) except (OSError, ValueError): return @@ -140,16 +130,9 @@ def _stop_state_server(state: dict) -> None: def refresh_local_runtime() -> bool: """Restart the managed server so it rescans the models directory. - The router's model list is SPAWN-ONLY: a GGUF added after start is - invisible to GET /models and 400s on completion, so anything that - changes the staged set while the server runs must bounce it. Covers - both ownership shapes: a supervised server restarts in-process; an - ADOPTED server (started by a previous backend session — the normal - shape after any restart) is stopped via its state-file pid and - replaced with a supervised boot. Without the adopted branch, every - download/delete in a post-restart session silently no-ops the bounce - and the router serves a stale catalog. Returns False when there is - nothing to refresh (no server anywhere; next boot scans fresh). + The router's model list is SPAWN-ONLY: a GGUF added after start is invisible to GET /models and + 400s on completion, so anything that changes the staged set while the server runs must bounce + it. """ global _SUPERVISOR try: @@ -173,13 +156,9 @@ def refresh_local_runtime() -> bool: def ensure_local_runtime(config: dict, force: bool = False) -> "object | None": - """Idempotent boot of the managed runtime. Returns the supervisor (or - None when disabled/unavailable). Never raises into a session start — - failures log and return None; chat falls back to configured providers. - - ``force=True`` skips the enabled gate — used by the explicit "Use this - model" action, where the click IS the opt-in (the caller records it in - config so future boots auto-start). + """Idempotent boot of the managed runtime. Returns the supervisor (or None when + disabled/unavailable). Never raises into a session start — failures log and return None; chat + falls back to configured providers. """ global _SUPERVISOR section = (config or {}).get("local_runtime") or {} @@ -216,7 +195,9 @@ def ensure_local_runtime(config: dict, force: bool = False) -> "object | None": try: from hermes_cli.local_runtime.binaries import ( + default_tag, ensure_runtime_installed, + installed_tags, select_backend, ) from hermes_cli.local_runtime.hardware import probe_budget @@ -233,8 +214,6 @@ def ensure_local_runtime(config: dict, force: bool = False) -> "object | None": # endpoint reports the pending update — the download is a deliberate # button click in the pane, not a boot-path surprise (a multi-minute # inline download here is exactly how the onboarding bounce returns). - from hermes_cli.local_runtime.binaries import default_tag, installed_tags - tag = section.get("tag") or default_tag() have = installed_tags() if tag not in have: diff --git a/hermes_cli/local_runtime/capabilities.py b/hermes_cli/local_runtime/capabilities.py index b30152fadf..084d7a190a 100644 --- a/hermes_cli/local_runtime/capabilities.py +++ b/hermes_cli/local_runtime/capabilities.py @@ -1,22 +1,9 @@ """Capability answers for models served by the managed runtime. -Capability lookups (vision, and whatever comes next) consult cloud-shaped -catalogs that have never heard of a local GGUF, so a vision-capable local -model reads as text-only and images detour to an auxiliary cloud model — -the wrong behavior twice over for a local-first user (broken feature, and -a screenshot silently leaving the machine). - -The managed runtime can answer from ground truth instead, best source -first: - -1. The RUNNING child's /props: llama-server reports a ``modalities`` block - when a vision projector is loaded. The server that will receive the - image says whether it can see — no inference, no catalog. -2. The catalog entry's declared capability (the ``vision`` tag + mmproj - asset) for staged-but-unloaded models: what the model WILL support once - its projector loads beside it. -3. None — not one of ours, or nothing known; the caller falls through to - its other sources. +Capability lookups (vision, and whatever comes next) consult cloud-shaped catalogs that have never +heard of a local GGUF, so a vision-capable local model reads as text-only and images detour to an +auxiliary cloud model — the wrong behavior twice over for a local-first user (broken feature, and a +screenshot silently leaving the machine). """ from __future__ import annotations @@ -27,7 +14,7 @@ import urllib.request logger = logging.getLogger(__name__) -_LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"}) +from hermes_cli.local_runtime.endpoint import LLAMACPP_ALIASES as _LLAMACPP_ALIASES # Image formats the managed server's decoder actually handles. llama.cpp # decodes with stb_image: PNG/JPEG/GIF/BMP yes, WebP NO — and a WebP part @@ -45,20 +32,20 @@ def is_managed_provider(provider: str, base_url: str = "") -> bool: p = (provider or "").strip().lower() if p in _LLAMACPP_ALIASES: return True - if p == "custom" and base_url: - try: - from hermes_cli.local_runtime.growth import is_managed_endpoint + if p != "custom" or not base_url: + return False + try: + from hermes_cli.local_runtime.growth import is_managed_endpoint - return is_managed_endpoint(base_url) - except Exception: # noqa: BLE001 - return False - return False + return is_managed_endpoint(base_url) + except Exception: # noqa: BLE001 + return False def _props_modalities(model_id: str) -> "bool | None": - """Ask the running server whether this loaded child sees images. - None when the server is down, the model isn't loaded, or the build - doesn't report modalities.""" + """Ask the running server whether this loaded child sees images. None when the server is down, the + model isn't loaded, or the build doesn't report modalities. + """ try: from hermes_cli.local_runtime.endpoint import _state_endpoint diff --git a/hermes_cli/local_runtime/catalog.py b/hermes_cli/local_runtime/catalog.py index 715d83c07f..09beef35f1 100644 --- a/hermes_cli/local_runtime/catalog.py +++ b/hermes_cli/local_runtime/catalog.py @@ -1,50 +1,12 @@ """Curated starter catalog for the managed local runtime. -Small and honest: every entry carries the estimator inputs (measured on -real GGUFs) so the picker can price a model BEFORE the user downloads -gigabytes. Once a file is on disk, profile_from_gguf() is the authority -and the catalog numbers are only used for the download decision. Entries -whose base config is gated upstream carry a same-family conservative -prior (commented) — the GGUF header corrects it at load time. +Small and honest: every entry carries the estimator inputs (measured on real GGUFs) so the picker +can price a model BEFORE the user downloads gigabytes. Once a file is on disk, profile_from_gguf() +is the authority and the catalog numbers are only used for the download decision. -Each model ships ONE build, Q4-class (UD-Q4_K_M where the repo has it, -UD-Q4_K_XL elsewhere). Q4 is the quant class current engines optimize -for and the sweet spot of the size/quality curve, so there is no quant -ladder: headroom buys a bigger context window, never a bigger quant, -and every machine runs the same well-tested build. Below Q4 the quality -loss is too severe to ship as someone's first local-AI experience; the -fit policy prices the build honestly (zero-spill, spilled, or refused by -the physics check). - -Validation lifecycle: builds proven end-to-end on real hardware are -marked validated. Day-0 entries ship before that proof (they simply lack -the validated flag) — ensure_model_ready's touch generation still gates -every first load at runtime. - -Multi-file models: variants may carry split-GGUF parts (llama-server loads -from the first part; all parts download together). Entries may carry an -mmproj (vision projector) and a speculative-decode draft model — both -download alongside the weights. MTP-integrated models run spec decode -wherever they load; a separate draft model attaches only when the launch -decision spills, where its speedup is largest. - -File sizes come from HF LFS metadata and feed the estimator, the fit -pills, and download progress. There is no download-time integrity check -by design: a corrupt or truncated file surfaces as a llama.cpp -load error at first use, and the reachability test catches upstream -re-uploads by size drift before users do. - -This is deliberately not a live registry feed: entries are reviewed like a -version bump (the same policy governs vendor recipe ingestion — parsed -data, never executed commands). - -Vendor recipes overlay: a per-SKU recipes repo may SUPPLEMENT these -entries where applicable — vendor SKUs only, never the base layer for -other platforms. A recipe may enrich identity (GGUF/quant/sha), perf -hints (-b/-ub, spec-decode), and sampling defaults; it never carries -context/slots/placement/serving flags (the fit policy owns those). -Resolution: exact SKU -> GPU-class bucket -> fit-only. Snapshot-synced, -reviewed like a tag bump. +Validation lifecycle: builds proven end-to-end on real hardware are marked validated. Day-0 entries +ship before that proof (they simply lack the validated flag) — ensure_model_ready's touch generation +still gates every first load at runtime. """ from __future__ import annotations @@ -73,19 +35,17 @@ from hermes_cli.local_runtime.estimator import ( logger = logging.getLogger(__name__) -_GIB = 1 << 30 _PART_SUFFIX = re.compile(r"-\d{5}-of-\d{5}$") @dataclass(frozen=True) class AssetFile: - """One downloadable file: repo-relative path and exact bytes (the size - feeds the estimator and the download progress bar; there is no - download-time integrity check by design — a corrupt file surfaces as a - llama.cpp load error). ``local`` overrides the on-disk name (repos - reuse generic names like mmproj-BF16.gguf across models). Non-model - extras live under the models dir's assets/ subdirectory so the router - never lists them.""" + """One downloadable file: repo-relative path and exact bytes (the size feeds the estimator and the + download progress bar; there is no download-time integrity check by design — a corrupt file + surfaces as a llama.cpp load error). ``local`` overrides the on-disk name (repos reuse generic + names like mmproj-BF16.gguf across models). Non-model extras live under the models dir's assets/ + subdirectory so the router never lists them. + """ path: str # repo-relative (may include a subdir) size_bytes: int @@ -116,9 +76,9 @@ class QuantVariant: @property def weights_bytes(self) -> int: - """Pre-download weights estimate: GGUF bytes ≈ tensor bytes + a - small header (<2%) — a safe, slightly conservative stand-in until - profile_from_gguf reads the real table.""" + """Pre-download weights estimate: GGUF bytes ≈ tensor bytes + a small header (<2%) — a safe, + slightly conservative stand-in until profile_from_gguf reads the real table. + """ return self.size_bytes @@ -202,14 +162,12 @@ class VariantChoice: def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice | None: """Fit the entry's one build (Q4-class) to this machine. - Every entry ships exactly one variant (see the module docstring for - why there is no quant ladder); headroom buys a bigger window, never - a bigger quant. The fit shapes: + Every entry ships exactly one variant (see the module docstring for why there is no quant + ladder); headroom buys a bigger window, never a bigger quant. The fit shapes: - - "best-large-window": zero-spills at TARGET_WINDOW - - "best-fits": zero-spills at the 64K floor - - "smallest-fits-spilled": weights spill to host RAM, priced honestly - - None: even spilled, physics refuses (the machine can't run it) + - "best-large-window": zero-spills at TARGET_WINDOW - "best-fits": zero-spills at the 64K floor + - "smallest-fits-spilled": weights spill to host RAM, priced honestly - None: even spilled, + physics refuses (the machine can't run it) """ overhead = (RUNTIME_OVERHEAD_BYTES + (entry.mmproj.size_bytes if entry.mmproj else 0) @@ -218,17 +176,14 @@ def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice variant = entry.variants[-1] profile = entry.profile(variant) need = variant.weights_bytes + overhead - if (need + ctx_bytes(profile, min(TARGET_WINDOW, native)) - <= budget.usable_vram_bytes): - return VariantChoice(variant=variant, zero_spill=True, - reason_key="best-large-window") + vram = budget.usable_vram_bytes + if need + ctx_bytes(profile, min(TARGET_WINDOW, native)) <= vram: + return VariantChoice(variant, zero_spill=True, reason_key="best-large-window") floor_kv = ctx_bytes(profile, min(FLOOR, native)) - if need + floor_kv <= budget.usable_vram_bytes: - return VariantChoice(variant=variant, zero_spill=True, - reason_key="best-fits") - if need + floor_kv <= budget.usable_vram_bytes + budget.ram_available_bytes: - return VariantChoice(variant=variant, zero_spill=False, - reason_key="smallest-fits-spilled") + if need + floor_kv <= vram: + return VariantChoice(variant, zero_spill=True, reason_key="best-fits") + if need + floor_kv <= vram + budget.ram_available_bytes: + return VariantChoice(variant, zero_spill=False, reason_key="smallest-fits-spilled") return None @@ -277,61 +232,35 @@ def recommended_entry(budget: HardwareBudget, ) -> "tuple[CatalogEntry, str] | None": """The catalog's default pick for THIS machine, with its reason. - Callers pass pre-filtered entries when some are ineligible for - reasons the catalog can't know (engine too old); default is the full - catalog. Returns (entry, reason) — the reason is a key the UI turns - into the Recommended badge's tooltip, so the rationale shown to the - user is the branch that actually fired, never a parallel explanation - that can drift: + Callers pass pre-filtered entries when some are ineligible for reasons the catalog can't know + (engine too old); default is the full catalog. - best-quality-resident quality won among resident entries that - clear the pleasant floor - speed-gated-quality same, but the floor eliminated a HIGHER - quality candidate — the exact 'why not the - big model?' a unified-memory owner asks - fastest-resident nothing resident clears the floor; the - quickest resident entry wins - least-painful-spilled nothing runs resident; fastest from host - memory (MoE by construction) - - Returns None only when nothing fits at all. + best-quality-resident quality won among resident entries that clear the pleasant floor speed- + gated-quality same, but the floor eliminated a HIGHER quality candidate — the exact 'why not the + big model?' a unified-memory owner asks fastest-resident nothing resident clears the floor; the + quickest resident entry wins least-painful-spilled nothing runs resident; fastest from host + memory (MoE by construction) """ pool = CATALOG if entries is None else entries - fitting: list[tuple[CatalogEntry, VariantChoice]] = [] - for entry in pool: - choice = select_variant(entry, budget) - if choice is not None: - fitting.append((entry, choice)) + fitting = [(e, c) for e in pool if (c := select_variant(e, budget)) is not None] if not fitting: return None + def speed(t, spilled=False): + return predicted_decode_tok_s(t[0], t[1].variant, budget, spilled=spilled) + resident = [(e, c) for e, c in fitting if c.zero_spill] - pleasant = [ - (e, c) for e, c in resident - if predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S - ] + pleasant = [t for t in resident if speed(t) >= PLEASANT_FLOOR_TOK_S] if pleasant: pick = max(pleasant, key=lambda t: (t[0].quality, -t[1].variant.size_bytes))[0] floor_gated = any(e.quality > pick.quality for e, _ in resident) - return (pick, "speed-gated-quality" if floor_gated - else "best-quality-resident") + return (pick, "speed-gated-quality" if floor_gated else "best-quality-resident") if resident: - pick = max(resident, - key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget))[0] - return (pick, "fastest-resident") + return (max(resident, key=speed)[0], "fastest-resident") # Everything spills: take the least painful — fastest predicted decode # from host memory (MoE wins here by construction; a dense spill # streams every weight over the host bus). - pick = max(fitting, - key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget, - spilled=True))[0] - return (pick, "least-painful-spilled") - - -def recommended_id(budget: HardwareBudget, - entries: "tuple[CatalogEntry, ...] | None" = None) -> str | None: - picked = recommended_entry(budget, entries) - return picked[0].id if picked is not None else None + return (max(fitting, key=lambda t: speed(t, spilled=True))[0], "least-painful-spilled") # ── catalog data: packaged JSON, refreshed from GitHub in memory ─ @@ -360,6 +289,18 @@ def _asset_from(d: "dict | None") -> "AssetFile | None": local=d.get("local")) +# Scalar CatalogEntry fields parsed from JSON: key -> (coerce, default); +# a None default means the key is required. +_SCALAR_FIELDS = { + "n_ctx_train": (int, None), "full_layers": (int, None), + "recurrent_layers": (int, None), "per_layer_f16": (int, None), + "swa_layers": (int, 0), "swa_window": (int, 0), + "moe": (bool, False), "mtp": (bool, False), "mtp_draft_depth": (int, 3), + "n_vocab": (int, 0), "sampling": (dict, {}), "min_engine": (str, ""), + "quality": (int, 0), "decode_fraction": (float, 1.0), +} + + def _load_catalog(doc: dict) -> "tuple[CatalogEntry, ...]": """Parse a catalog document into entries. Unknown fields are ignored (newer catalogs stay readable by older apps); a major schema bump is @@ -374,25 +315,13 @@ def _load_catalog(doc: dict) -> "tuple[CatalogEntry, ...]": files=tuple(_asset_from(f) for f in v["files"]), validated=bool(v.get("validated"))) for v in m["variants"]) + scalars = {k: coerce(m[k] if default is None else m.get(k, default)) + for k, (coerce, default) in _SCALAR_FIELDS.items()} entries.append(CatalogEntry( id=m["id"], display_name=m["display_name"], description=m["description"], repo=m["repo"], variants=variants, - n_ctx_train=int(m["n_ctx_train"]), - full_layers=int(m["full_layers"]), - recurrent_layers=int(m["recurrent_layers"]), - per_layer_f16=int(m["per_layer_f16"]), - swa_layers=int(m.get("swa_layers", 0)), - swa_window=int(m.get("swa_window", 0)), - moe=bool(m.get("moe")), mtp=bool(m.get("mtp")), - mtp_draft_depth=int(m.get("mtp_draft_depth", 3)), - n_vocab=int(m.get("n_vocab", 0)), - mmproj=_asset_from(m.get("mmproj")), - draft=_asset_from(m.get("draft")), - sampling=dict(m.get("sampling", {})), - min_engine=str(m.get("min_engine", "")), - quality=int(m.get("quality", 0)), - decode_fraction=float(m.get("decode_fraction", 1.0)), - )) + mmproj=_asset_from(m.get("mmproj")), draft=_asset_from(m.get("draft")), + **scalars)) return tuple(entries) @@ -410,9 +339,10 @@ CATALOG: "tuple[CatalogEntry, ...]" = _packaged_catalog() def refresh_catalog(force: bool = False) -> bool: """Fetch the current catalog from the repo and swap it in memory. - Best-effort by design: any failure (offline, GitHub down, unreadable - schema) leaves the running catalog untouched and retries after the - TTL. Returns True when a fetched document replaced the catalog.""" + Best-effort by design: any failure (offline, GitHub down, unreadable schema) leaves the running + catalog untouched and retries after the TTL. Returns True when a fetched document replaced the + catalog. + """ global CATALOG, _last_refresh_attempt now = time.monotonic() @@ -435,9 +365,9 @@ def refresh_catalog(force: bool = False) -> bool: def refresh_catalog_soon() -> None: - """TTL-gated background refresh; returns immediately. The caller's - current request serves the catalog it already has — the refresh - lands for the next one.""" + """TTL-gated background refresh; returns immediately. The caller's current request serves the + catalog it already has — the refresh lands for the next one. + """ if time.monotonic() - _last_refresh_attempt < _REFRESH_TTL_S: return threading.Thread(target=refresh_catalog, daemon=True, @@ -448,13 +378,6 @@ def catalog_by_id() -> dict[str, CatalogEntry]: return {entry.id: entry for entry in CATALOG} -def find_variant(entry_id: str, model_id: str) -> QuantVariant | None: - entry = catalog_by_id().get(entry_id) - if entry is None: - return None - return next((v for v in entry.variants if v.model_id == model_id), None) - - def find_entry_for_model(model_id: str) -> "tuple[CatalogEntry, QuantVariant] | None": """Locate the entry + variant that owns a staged model id.""" for entry in CATALOG: diff --git a/hermes_cli/local_runtime/context_policy.py b/hermes_cli/local_runtime/context_policy.py index 1f08ae354b..682580f8f7 100644 --- a/hermes_cli/local_runtime/context_policy.py +++ b/hermes_cli/local_runtime/context_policy.py @@ -1,35 +1,10 @@ """Context policy — the window ladder for managed local models. -One contract: any model runs at any window up to its native max; hardware -and session depth only change tokens/s. Constants, not knobs — nothing in -this module reads config. +One contract: any model runs at any window up to its native max; hardware and session depth only +change tokens/s. Constants, not knobs — nothing in this module reads config. -The policy encodes behavior measured on real hardware (llama.cpp, -discrete NVIDIA GPUs on Windows/WDDM, and unified-memory devices): - -- Windows never over-allocates VRAM ahead of need. On WDDM, allocating - past residency slows decode roughly 9x even at identical conversation - depth — the driver silently demotes pages instead of failing. Every - window grant therefore re-fits against live memory at grant time. -- Models launch at the largest window that fits entirely in GPU memory - (zero-spill) and grow toward their native max as the session needs - room, at request boundaries only. -- Growth re-prefills the conversation into the larger window. Measured - cost is comparable to save/restore on discrete GPUs, and recurrent or - hybrid-attention models cannot rewind mid-sequence anyway, so - re-prefill is the only mechanism that works for every architecture. -- Every recommended model gets at least a 64K window. When weights alone - exceed VRAM, the fit deliberately spills weights to host RAM to - protect that floor (measured: an explicit context size makes the fit - spill weights and hold the window rather than shrink it). -- Below ~6 tok/s decode, growth stops and compression becomes the - default; deeper context is an explicit per-session choice. The deepest - measured host-spilled configuration bottomed out near this rate. -- Spilled mixture-of-experts configs pin expert/FFN weights to host so - attention and KV stay GPU-resident — measured ~1.75x faster than - spilling layers naively at the same host byte count. -- Speculative decoding (MTP) defaults on only for spilled configs, where - its speedup is largest (measured 1.43x spilled vs 1.35x resident). +The policy encodes behavior measured on real hardware (llama.cpp, discrete NVIDIA GPUs on +Windows/WDDM, and unified-memory devices): """ from __future__ import annotations @@ -97,14 +72,10 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, overhead_bytes: int = 0) -> WindowDecision | PhysicsRefusal: """The launch decision: largest cheap rung, never below the floor. - Zero-spill rung: weights + ctx + overhead fit usable VRAM entirely. - Bounded-early-cost rung: weights already exceed VRAM; take the largest - rung whose ctx stays <= ~15% of usable VRAM. - Floor everywhere, capped at native. - - ``overhead_bytes``: runtime cost beyond weights+KV (RUNTIME_OVERHEAD - plus the vision projector when one loads). Zero keeps this function - pure physics for decision-table tests; production callers pass it. + Zero-spill rung: weights + ctx + overhead fit usable VRAM entirely. Bounded-early-cost rung + (weights already exceed VRAM): largest rung whose ctx stays <= ~15% of usable VRAM. Floor + everywhere, capped at native. ``overhead_bytes`` is runtime cost beyond weights+KV; zero + keeps this pure physics for decision-table tests, production callers pass it. """ refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention) if refusal: @@ -160,23 +131,13 @@ def growth_decision(profile: ModelProfile, budget: HardwareBudget, *, server_idle: bool, flash_attention: bool = True, occupancy_confirmed: bool = False) -> GrowthDecision: - """One growth evaluation, END-OF-TURN ONLY (caller guarantees the turn - boundary; recurrent state cannot rewind mid-sequence). + """One growth evaluation, END-OF-TURN ONLY (recurrent state cannot rewind mid-sequence). - Gate ordering: - 1. occupancy (~85%) — nothing to do before the edge; - 2. native cap — the contract tops out at trained context; - 3. idleness — growth re-grants only on an otherwise-idle - server (concurrency design); - 4. speed floor — below it, compression becomes the default and deeper - is an explicit user choice; - 5. re-fit against LIVE free memory (the rung must fit residency - NOW, not at launch time — over-allocation is the slow path). - - ``occupancy_confirmed``: the caller has independently established that - the session is at its window's edge (the agent's compression gate fired - on its own threshold). Skips gate 1 so two separately-derived edge - definitions can't deadlock into compress-before-grow. + Gate order: occupancy (~85%) → native cap → idleness (growth only on an otherwise-idle + server) → speed floor (below it compression is the default) → re-fit against LIVE free + memory (the rung must fit NOW, not at launch). ``occupancy_confirmed`` skips gate 1 when the + caller's own compression gate already fired, so two edge definitions can't deadlock into + compress-before-grow. """ if not occupancy_confirmed and session_tokens < current_window * _GROW_AT_OCCUPANCY: return GrowthDecision("hold", reason="session below growth occupancy") @@ -211,10 +172,10 @@ def growth_decision(profile: ModelProfile, budget: HardwareBudget, *, def spill_overrides(profile: ModelProfile) -> list[str]: - """-ot placement for spilled configs: expert/FFN weights to host so - attention + KV stay GPU-resident. MoE gets the expert pattern; - hybrids push recurrent-layer FFNs (their n_head_kv==0 layers carry no - KV worth protecting).""" + """-ot placement for spilled configs: expert/FFN weights to host so attention + KV stay + GPU-resident. MoE gets the expert pattern; hybrids push recurrent-layer FFNs (their n_head_kv==0 + layers carry no KV worth protecting). + """ if profile.moe: return ["-ot", r"blk\.\d+\.ffn_.*_exps\.weight=CPU"] if profile.recurrent_layer_count: @@ -228,27 +189,10 @@ def launch_args(profile: ModelProfile, decision: WindowDecision, *, mtp_draft_depth: int = 3, uma: bool = False, mtp_prefill: bool = False) -> list[str]: - """Per-model launch flags from a window decision. Explicit -c puts fit - into spill-weights-and-hold-ctx; q8 KV cache wherever flash attention - exists; -ot placement on spilled configs — DISCRETE cards only. - - ``uma``: on unified memory there is no bus to protect tensors from — - "CPU" and "GPU" are the same silicon, and pinning FFN weights to the - host path just forces CPU compute (measured well over 2x slower than - letting the allocator place everything). The discrete - ~1.75x win the -ot pattern encodes does not transfer; a spilled UMA - config runs unpinned. - - MTP and the large prefill microbatch both win, and whether they may - STACK is a fit question, not a rule: backend sampling keeps a - ubatch x vocab x fp32 logits buffer on the GPU and MTP's draft - context doubles it, so the stacked posture costs a few GiB extra at - large vocab. Where it fits, it measures best on both axes (Qwen3.8 - Q4 on a 32 GiB card: 93.3 tok/s decode vs 89.5 at ub512, prefill - slightly better too); where it doesn't, ub512 keeps the decode win - without packing the card. ``mtp_prefill`` is that fit verdict — - presets decide it against the priced margin, and ub_logits_bytes() - prices the same choice so the flag and its cost travel together.""" + """Per-model launch flags from a window decision. Explicit -c puts fit into + spill-weights-and-hold-ctx; q8 KV cache wherever flash attention exists; -ot placement on + spilled configs — DISCRETE cards only. + """ args = ["-c", str(decision.window)] if mtp_capable: args += ["--spec-type", "draft-mtp", @@ -267,17 +211,10 @@ def launch_args(profile: ModelProfile, decision: WindowDecision, *, def ub_logits_bytes(n_vocab: int, *, mtp_capable: bool, mtp_prefill: bool = False) -> int: - """GPU logits/compute-buffer cost of the microbatch posture chosen by - launch_args, priced from the model's own vocab and calibrated against - measured server RSS (Qwen3.8 Q4, both postures, three windows): - - stacked (MTP + ub2048): ubatch x vocab x fp32 x 1.5 (~2.9 GiB at - 248K vocab; fitted 2.5, rounded up) - decode (MTP + ub512): ubatch x vocab x fp32 x 2 (~1.0 GiB) - plain (ub2048): ubatch x vocab x fp32 (~1.9 GiB) - - Callers add this to RUNTIME_OVERHEAD per model — the flag and its - price travel together or the fit lies.""" + """GPU logits/compute-buffer cost of the microbatch posture chosen by launch_args, priced from the + model's own vocab and calibrated against measured server RSS (Qwen3.8 Q4, both postures, three + windows): + """ v = max(0, int(n_vocab)) if mtp_capable and mtp_prefill: return int(2048 * v * 4 * 1.5) diff --git a/hermes_cli/local_runtime/detect.py b/hermes_cli/local_runtime/detect.py index 02131255ed..376054555a 100644 --- a/hermes_cli/local_runtime/detect.py +++ b/hermes_cli/local_runtime/detect.py @@ -1,9 +1,8 @@ """Detection of running llama-server instances. -Probes well-known local roots and fingerprints genuine llama-server via -/props (build_info + model fields — Ollama and LM Studio answer /v1/models -but not /props). The credential is reachability; detection never needs a -key, but honors one if the probed server requires it (401 -> detected, +Probes well-known local roots and fingerprints genuine llama-server via /props (build_info + model +fields — Ollama and LM Studio answer /v1/models but not /props). The credential is reachability; +detection never needs a key, but honors one if the probed server requires it (401 -> detected, auth_required=True). """ @@ -51,10 +50,8 @@ def probe_port(port: int) -> DetectedServer | None: build = str(props.get("build_info", "")) if not build: return None # answers /props but isn't llama-server - n_ctx = None dgs = props.get("default_generation_settings") - if isinstance(dgs, dict): - n_ctx = dgs.get("n_ctx") + n_ctx = dgs.get("n_ctx") if isinstance(dgs, dict) else None models_status, models = _get(f"{root}/models") return DetectedServer( base_url=f"{root}/v1", @@ -69,11 +66,7 @@ def probe_port(port: int) -> DetectedServer | None: def detect_server(extra_ports: tuple[int, ...] = ()) -> DetectedServer | None: """First hit across default + extra ports (managed port, config port).""" - seen = set() - for port in (*DEFAULT_PROBE_PORTS, *extra_ports): - if port in seen: - continue - seen.add(port) + for port in dict.fromkeys((*DEFAULT_PROBE_PORTS, *extra_ports)): hit = probe_port(port) if hit: return hit diff --git a/hermes_cli/local_runtime/endpoint.py b/hermes_cli/local_runtime/endpoint.py index b80a663d03..01734465bf 100644 --- a/hermes_cli/local_runtime/endpoint.py +++ b/hermes_cli/local_runtime/endpoint.py @@ -1,14 +1,7 @@ """Endpoint resolution for llamacpp-alias requests (provider integration). -The seam between the existing provider mechanism and the managed runtime: -``provider: llamacpp`` with no explicit base_url resolves, in order, to - -1. the managed server this Hermes is supervising (state file written by - LlamaServerSupervisor.start, removed on stop, staleness-checked), or -2. a detected external llama-server. - -Returns None when neither exists — the caller falls through to the normal -custom-provider path and its own error reporting. +The seam between the existing provider mechanism and the managed runtime: ``provider: llamacpp`` +with no explicit base_url resolves, in order, to """ from __future__ import annotations @@ -28,9 +21,8 @@ logger = logging.getLogger(__name__) def _pid_alive(pid: int) -> bool: """Liveness for the state file's supervisor-child pid. - psutil when available; otherwise fall back to True (optimistic) — on - Windows ``os.kill(pid, 0)`` TERMINATES the process, so it must never be - used as a probe (windows-git-bash interop pitfall). + psutil when available; otherwise fall back to True (optimistic) — on Windows ``os.kill(pid, 0)`` + TERMINATES the process, so it must never be used as a probe (windows-git-bash interop pitfall). """ if not pid or pid < 0: return False @@ -78,25 +70,16 @@ def _state_endpoint() -> dict | None: # optimistically so readiness probes racing the boot see a configured # provider, not missing credentials. A dead pid is a crashed-without- # cleanup leftover — ignore it so requests don't blackhole. - if pid_ok: - return endpoint - return None + return endpoint if pid_ok else None def resolve_llamacpp_endpoint(config: dict | None = None, wait_for_boot_s: float = 8.0) -> dict | None: """Managed-first, detection-second endpoint for llamacpp aliases. - Returns {"base_url", "api_key"} or None. api_key is empty for keyless - external servers (callers substitute the SDK placeholder). - - Boot-race rung: on a fresh backend start there is NO state file yet — - the lifespan boot thread is still spawning the server (config load + - preset generation + spawn ≈ 1-3 s) while the desktop's readiness probe - fires the moment the WebSocket connects. When the runtime is enabled - and installed, a missing endpoint means BOOTING, not unconfigured: - poll briefly for the state file instead of failing the probe (twice - observed as 'no usable credentials' → onboarding on restart). + Boot-race rung: on a fresh backend start there is NO state file yet — the lifespan boot thread + is still spawning the server (config load + preset generation + spawn ≈ 1-3 s) while the + desktop's readiness probe fires the moment the WebSocket connects. """ managed = _state_endpoint() if managed: @@ -104,11 +87,8 @@ def resolve_llamacpp_endpoint(config: dict | None = None, from hermes_cli.local_runtime.detect import detect_server - extra = () - if config: - ports = (config.get("local_runtime") or {}).get("detect_ports") or [] - extra = tuple(int(p) for p in ports) - hit = detect_server(extra_ports=extra) + ports = ((config or {}).get("local_runtime") or {}).get("detect_ports") or [] + hit = detect_server(extra_ports=tuple(int(p) for p in ports)) if hit and not hit.auth_required: return {"base_url": hit.base_url, "api_key": ""} @@ -126,33 +106,28 @@ def resolve_llamacpp_endpoint(config: dict | None = None, _KICK_LOCK = threading.Lock() +def _load_config_if_none(config: dict | None) -> dict | None: + if config is not None: + return config + from hermes_cli.config import load_config + + return load_config() + + def _kick_managed_boot(config: dict | None) -> None: """Actively start the managed server when resolution finds it missing. - The wait loop above assumes some OTHER thread is bringing the server - up — true only at backend start (the lifespan boot thread). A router - that dies LATER leaves no boot in flight: the backend process was - killed with the router as part of its tree, or another install took - the stable port and the ownership guard rightly refused it. In those - states the wait just expired and agent init failed with 'no provider - configured', even though the fix is the same idempotent ensure call - the lifespan makes. Kick it here, off-thread (the resolver's wait - stays bounded; ensure's own state checks make a concurrent lifespan - boot harmless) and non-reentrant (racing resolutions kick once). + The wait loop above assumes some OTHER thread is bringing the server up — true only at backend + start (the lifespan boot thread). """ if not _KICK_LOCK.acquire(blocking=False): return # a kick is already in flight def _boot() -> None: try: - cfg = config - if cfg is None: - from hermes_cli.config import load_config - - cfg = load_config() from hermes_cli.local_runtime.bootstrap import ensure_local_runtime - ensure_local_runtime(cfg) + ensure_local_runtime(_load_config_if_none(config)) except Exception: # noqa: BLE001 — best-effort; resolution falls back logger.warning("on-demand managed-server boot failed", exc_info=True) finally: @@ -164,28 +139,20 @@ def _kick_managed_boot(config: dict | None) -> None: def _boot_in_flight(config: dict | None) -> bool: """True when the managed runtime is enabled and installed — the state - a lifespan boot thread is (or is about to be) bringing up. - Installed-ness is a verified-manifest scan under runtimes_root(), NOT a - server_binary() call — that helper requires an install_dir argument, and - calling it bare made this gate throw-and-return-False forever, silently - disabling the boot wait (the regression - test had monkeypatched this function instead of exercising it). + Installed-ness is a verified-manifest scan under runtimes_root(), NOT a bare ``server_binary()`` + call: that helper needs an install_dir, and calling it bare made this gate throw-and-return + False forever, silently disabling the boot wait. """ try: - if config is None: - from hermes_cli.config import load_config - - config = load_config() + config = _load_config_if_none(config) if not ((config or {}).get("local_runtime") or {}).get("enabled"): return False - import json as _json - from hermes_cli.local_runtime.binaries import runtimes_root for manifest in runtimes_root().glob("*/*/manifest.json"): try: - if _json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"): + if json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"): return True except (ValueError, OSError): continue diff --git a/hermes_cli/local_runtime/estimator.py b/hermes_cli/local_runtime/estimator.py index 6778b88ed4..988a88a2a1 100644 --- a/hermes_cli/local_runtime/estimator.py +++ b/hermes_cli/local_runtime/estimator.py @@ -1,18 +1,7 @@ """Per-layer context-memory estimator + physics check. -The whole-model dense formula misprices 1M-context hybrids by ~100x; the -per-layer walk fixes that, and every column is measured on real GGUFs: - -- full-attention layer: linear in T (B1: 144.0 KiB/tok on Qwen3-4B - f16 — formula-exact) -- SWA layer: capped at the sliding window -- recurrent layer (n_head_kv == 0): constant (state is ~context-free) -- q8_0 KV = exactly 34/64 of f16 (holds on CUDA and CPU) -- weights: exact from the tensor table (within 0.01% of the loader) - -The estimator is ADVISORY: fit's allocation is authoritative at launch and -the touch generation is ground truth after it. Unknown shapes round UP -(never underestimate memory). +The estimator is ADVISORY: fit's allocation is authoritative at launch and the touch generation is +ground truth after it. Unknown shapes round UP (never underestimate memory). """ from __future__ import annotations @@ -46,9 +35,9 @@ class LayerKind(Enum): @dataclass class ModelProfile: - """Everything the policy needs, decoupled from GGUF parsing so the - decision-table tests can construct profiles directly (design's - verification plan).""" + """Everything the policy needs, decoupled from GGUF parsing so the decision-table tests can + construct profiles directly (design's verification plan). + """ name: str weights_bytes: int @@ -82,11 +71,9 @@ class ModelProfile: class HardwareBudget: """Memory the physics check may budget against. - Budget-source rule: discrete cards may trust the device query - (measured honest); unified-memory devices must budget from OS free - physical memory minus headroom — their device queries have been - observed off by 3x. Callers construct - this accordingly; the estimator just consumes it. + Budget-source rule: discrete cards may trust the device query (measured honest); unified-memory + devices must budget from OS free physical memory minus headroom — their device queries have been + observed off by 3x. Callers construct this accordingly; the estimator just consumes it. """ usable_vram_bytes: int # live free (discrete) / derived (UMA) @@ -138,9 +125,9 @@ def kv_dtype_factor(flash_attention: bool) -> float: def ctx_bytes(profile: ModelProfile, window: int, *, flash_attention: bool = True) -> int: - """Context memory for one window: full layers linear in T, SWA layers - capped at the sliding window, recurrent layers constant. Scaled by - profile.kv_scale (MTP draft context).""" + """Context memory for one window: full layers linear in T, SWA layers capped at the sliding window, + recurrent layers constant. Scaled by profile.kv_scale (MTP draft context). + """ factor = kv_dtype_factor(flash_attention) total = 0.0 for kind, per_token_f16 in profile.layers: diff --git a/hermes_cli/local_runtime/gguf.py b/hermes_cli/local_runtime/gguf.py index a5159b9d56..e73e67dd2d 100644 --- a/hermes_cli/local_runtime/gguf.py +++ b/hermes_cli/local_runtime/gguf.py @@ -1,12 +1,7 @@ """GGUF metadata + tensor-table reader (stdlib only). -Feeds the per-layer context estimator: architecture, layer count, per-layer -KV head counts (0 = recurrent layer — the hybrid discriminator), head dims, -sliding-window config, trained context, and exact weight bytes summed from -the tensor table (validated to within 0.01% of the loader's buffer). - -Reads the header only (metadata + tensor infos); never touches tensor data, -so it is fast enough to run at picker time on multi-GB files. +Reads the header only (metadata + tensor infos); never touches tensor data, so it is fast enough to +run at picker time on multi-GB files. """ from __future__ import annotations @@ -65,9 +60,9 @@ class GGUFHeader: @property def n_vocab(self) -> int: - """Vocabulary size: prices the GPU logits buffers (they scale - ubatch x vocab). vocab_size metadata when present, else the - tokenizer list length.""" + """Vocabulary size: prices the GPU logits buffers (they scale ubatch x vocab). vocab_size + metadata when present, else the tokenizer list length. + """ v = self._arch_key("vocab_size") if v: return int(v) @@ -82,12 +77,10 @@ class GGUFHeader: def sampling_defaults(self) -> dict: """Upstream's recommended sampling, when the file carries it. - Model publishers bake general.sampling.* keys into the GGUF - (llama-server reads them as that model's default generation - settings), so the file itself is the source of truth for how its - publisher wants it run — it arrives with the download and updates - with every re-upload, no catalog required. Returned as preset INI - keys; empty when the file carries none. + Publishers bake ``general.sampling.*`` keys into the GGUF (llama-server reads them as that + model's defaults), so the file is the source of truth -- it ships with the download and + updates with every re-upload, no catalog needed. Returned as preset INI keys; empty if + absent. """ ini_key = {"temp": "temp", "temperature": "temp", "top_p": "top-p", "top_k": "top-k", "min_p": "min-p", @@ -121,16 +114,12 @@ class GGUFHeader: return int(self._arch_key("full_attention_interval") or 0) def head_counts_kv(self) -> list[int]: - """Per-layer KV head counts; 0 marks a recurrent/linear layer (the - n_head_kv == 0 discriminator). + """Per-layer KV head counts; 0 marks a recurrent/linear layer (n_head_kv == 0). - Three GGUF shapes, each verified against real files: - - per-layer array (nemotron_h_moe): use as-is; - - scalar + full_attention_interval (qwen35): the scalar applies to - every INTERVAL-th layer (1-indexed: layers where (i+1) % N == 0), - zero elsewhere — pricing all layers as attention was a 4x - overestimate on Qwen3.6-27B; - - plain scalar (dense): broadcast to every layer. + Three GGUF shapes: a per-layer array (nemotron_h_moe) is used as-is; a scalar plus + ``full_attention_interval`` (qwen35) applies to every N-th layer (1-indexed) and is zero + elsewhere -- pricing all layers as attention was a 4x overestimate; a plain scalar (dense) + broadcasts to every layer. """ v = self._arch_key("attention.head_count_kv") if isinstance(v, list): diff --git a/hermes_cli/local_runtime/growth.py b/hermes_cli/local_runtime/growth.py index 484f2ff9f5..43cca00d0b 100644 --- a/hermes_cli/local_runtime/growth.py +++ b/hermes_cli/local_runtime/growth.py @@ -1,20 +1,7 @@ """In-session context growth for the managed llama.cpp runtime. -The live half of the window ladder (context_policy.growth_decision): when a -session reaches the edge of its granted window, Hermes grows the window -toward the model's native max INSTEAD of compressing. Compression becomes -what the design says it is — the move of last resort, once the window is at -native (or the speed floor / physics say stop). - -Mechanism: growth is re-prefill. A per-model window -override is persisted, presets regenerate with the bigger window, the -supervised server bounces, and the next request autoloads the model at the -new window and re-prefills the conversation. Nothing about the Hermes -conversation mutates — no prompt-cache or role-alternation risk; the whole -operation is server-side. - -Scope guard: only a server THIS process supervises grows. Detected external -servers and other-process supervisors keep their own policies. +Scope guard: only a server THIS process supervises grows. Detected external servers and other- +process supervisors keep their own policies. """ from __future__ import annotations @@ -75,13 +62,12 @@ def is_managed_endpoint(base_url: str) -> bool: def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int, current_window: int, measured_decode_tok_s: float | None = None) -> int | None: - """One growth evaluation + execution. Returns the NEW window when the - ladder granted a bigger one, else None (hold / compress / not ours). + """One growth evaluation + execution. Returns the NEW window when the ladder granted a bigger + one, else None (hold / compress / not ours). - The caller sits at a request boundary by construction (the pre-API - compression gate), so re-prefill growth is safe at any call: the next - request rebuilds server state from scratch in the larger window — - nothing rewinds. + The caller sits at a request boundary by construction (the pre-API compression gate), so + re-prefill growth is safe at any call: the next request rebuilds server state in the larger + window — nothing rewinds. """ from hermes_cli.local_runtime.bootstrap import ( get_supervisor, diff --git a/hermes_cli/local_runtime/hardware.py b/hermes_cli/local_runtime/hardware.py index c6a316776d..0b309f0748 100644 --- a/hermes_cli/local_runtime/hardware.py +++ b/hermes_cli/local_runtime/hardware.py @@ -1,27 +1,11 @@ """Live hardware budget probe. -Budget-source rule: discrete cards may trust the device query (measured -honest within rounding); unified-memory devices must budget from OS free -physical memory minus headroom — their device queries have been observed -off by 3x in both directions. The probe classifies the device and -constructs the right HardwareBudget for the estimator. +Budget-source rule: discrete cards may trust the device query (measured honest within rounding); +unified-memory devices must budget from OS free physical memory minus headroom — their device +queries have been observed off by 3x in both directions. -Vendor probe quirk (WDDM carve-out): on unified-memory NVIDIA devices -under Windows, nvidia-smi answers from the legacy dedicated-VRAM -carve-out — a fraction of the pool the CUDA allocator actually -addresses uniformly at full bandwidth. The CUDA driver API is -the tiebreaker: cuDeviceGetAttribute(INTEGRATED) is the vendor's own -declaration and always wins — 1 budgets unified, 0 stays discrete no -matter what any other number says. Only when the driver API is -unreachable does the engine's --list-devices view apply, and then only -behind two independent conditions no discrete card can meet. - -Every probe here must work under a stripped PATH — gateway and service -sessions don't inherit the interactive environment. nvcuda/libcuda load -through the system loader (PATH plays no part), so classification never -depends on PATH; nvidia-smi resolves through an explicit candidate -ladder (PATH first, then the driver's known install locations) and its -absence only softens the live number, never the verdict. +Every probe here must work under a stripped PATH — gateway and service sessions don't inherit the +interactive environment. """ from __future__ import annotations @@ -81,21 +65,21 @@ _pool_probe_cache: tuple[float, "tuple[int, bool | None] | None"] | None = None _DEVICE_LINE_RE = re.compile(r"CUDA\d+:.*\((\d+)\s*MiB,\s*\d+\s*MiB free\)\s*$") +def _stdout(*argv: str) -> str: + return subprocess.run(list(argv), capture_output=True, text=True, timeout=5).stdout + + def _ram_bytes() -> tuple[int, int]: """(total, available) physical memory, cross-platform stdlib.""" try: import ctypes class MEMORYSTATUSEX(ctypes.Structure): - _fields_ = [("dwLength", ctypes.c_ulong), - ("dwMemoryLoad", ctypes.c_ulong), - ("ullTotalPhys", ctypes.c_ulonglong), - ("ullAvailPhys", ctypes.c_ulonglong), - ("ullTotalPageFile", ctypes.c_ulonglong), - ("ullAvailPageFile", ctypes.c_ulonglong), - ("ullTotalVirtual", ctypes.c_ulonglong), - ("ullAvailVirtual", ctypes.c_ulonglong), - ("ullAvailExtendedVirtual", ctypes.c_ulonglong)] + _fields_ = ([("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong)] + + [(name, ctypes.c_ulonglong) for name in ( + "ullTotalPhys", "ullAvailPhys", "ullTotalPageFile", + "ullAvailPageFile", "ullTotalVirtual", "ullAvailVirtual", + "ullAvailExtendedVirtual")]) stat = MEMORYSTATUSEX() stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX) @@ -103,48 +87,35 @@ def _ram_bytes() -> tuple[int, int]: return stat.ullTotalPhys, stat.ullAvailPhys except (AttributeError, OSError): pass - if sys.platform == "darwin": - # macOS getconf has no _PHYS_PAGES/_AVPHYS_PAGES (exit 64, "no such - # configuration parameter") — the POSIX branch below returns (0, 0) - # and every model reads unavailable. sysctl is the platform truth. - try: - total = int(subprocess.run( - ["/usr/sbin/sysctl", "-n", "hw.memsize"], - capture_output=True, text=True, timeout=5).stdout.strip() or 0) + try: + if sys.platform == "darwin": + # macOS getconf has no _PHYS_PAGES/_AVPHYS_PAGES (exit 64, "no such + # configuration parameter") — the POSIX branch below returns (0, 0) + # and every model reads unavailable. sysctl is the platform truth. + total = int(_stdout("/usr/sbin/sysctl", "-n", "hw.memsize").strip() or 0) if total <= 0: return 0, 0 avail = total // 2 # conservative fallback try: - out = subprocess.run(["/usr/bin/vm_stat"], capture_output=True, - text=True, timeout=5).stdout + out = _stdout("/usr/bin/vm_stat") page_m = re.search(r"page size of (\d+)", out) page = int(page_m.group(1)) if page_m else 16384 - pages = 0 # free + inactive + purgeable ≈ reclaimable-on-demand; the # speculative pool is dropped by the OS under pressure too. - for key in ("Pages free", "Pages inactive", "Pages purgeable", - "Pages speculative"): - m = re.search(rf"{key}:\s+(\d+)\.", out) - if m: - pages += int(m.group(1)) + pages = sum(int(m.group(1)) for key in ( + "Pages free", "Pages inactive", "Pages purgeable", "Pages speculative") + if (m := re.search(rf"{key}:\s+(\d+)\.", out))) if pages > 0: avail = pages * page except (OSError, ValueError): pass return total, avail - except (OSError, ValueError): - return 0, 0 - # POSIX - try: - page = int(subprocess.run(["getconf", "PAGE_SIZE"], capture_output=True, - text=True, timeout=5).stdout or 4096) - total = int(subprocess.run(["getconf", "_PHYS_PAGES"], capture_output=True, - text=True, timeout=5).stdout or 0) * page + # POSIX + page = int(_stdout("getconf", "PAGE_SIZE") or 4096) + total = int(_stdout("getconf", "_PHYS_PAGES") or 0) * page avail = total // 2 # conservative when _AVPHYS is unavailable try: - avail = int(subprocess.run(["getconf", "_AVPHYS_PAGES"], - capture_output=True, text=True, - timeout=5).stdout or 0) * page or avail + avail = int(_stdout("getconf", "_AVPHYS_PAGES") or 0) * page or avail except (OSError, ValueError): pass return total, avail @@ -160,25 +131,23 @@ _smi_path_cache: "tuple[str | None] | None" = None def _nvidia_smi_path() -> str | None: - """Absolute path to nvidia-smi, or None. PATH first (respects user - overrides), then the driver's known install locations on Windows; - on Linux/WSL the PATH lookup is the whole ladder.""" + """Absolute path to nvidia-smi, or None. PATH first (respects user overrides), then the driver's + known install locations on Windows; on Linux/WSL the PATH lookup is the whole ladder. + """ global _smi_path_cache if _smi_path_cache is not None: return _smi_path_cache[0] found = shutil.which("nvidia-smi") if found is None and os.name == "nt": windir = os.environ.get("SystemRoot", r"C:\Windows") - for candidate in ( + candidates = ( # DCH drivers (every modern install) place it in System32. Path(windir) / "System32" / "nvidia-smi.exe", # Legacy standalone drivers used NVSMI, never on PATH. Path(os.environ.get("ProgramFiles", r"C:\Program Files")) / "NVIDIA Corporation" / "NVSMI" / "nvidia-smi.exe", - ): - if candidate.exists(): - found = str(candidate) - break + ) + found = next((str(c) for c in candidates if c.exists()), None) _smi_path_cache = (found,) return found @@ -202,12 +171,11 @@ def _nvidia_vram() -> tuple[int, int] | None: def _cuda_driver_pool() -> "tuple[int, bool | None] | None": - """(allocator_total_bytes, integrated_or_None) from the CUDA driver - API, or None when unreachable. ctypes against the driver's own DLL/SO - — no toolkit, no subprocess, ~ms. INTEGRATED is the vendor's own - unified-memory declaration; total is the pool the allocator will - actually hand out (on carve-out devices, several times what - nvidia-smi reports).""" + """(allocator_total_bytes, integrated_or_None) from the CUDA driver API, or None when unreachable. + ctypes against the driver's own DLL/SO — no toolkit, no subprocess, ~ms. INTEGRATED is the + vendor's own unified-memory declaration; total is the pool the allocator will actually hand out + (on carve-out devices, several times what nvidia-smi reports). + """ import ctypes for name in ("nvcuda.dll", "libcuda.so.1", "libcuda.so"): @@ -239,10 +207,10 @@ def _cuda_driver_pool() -> "tuple[int, bool | None] | None": def _engine_device_pool() -> "tuple[int, bool | None] | None": - """(engine_total_bytes, None) from the installed runtime's own - --list-devices, or None. The fallback truth source when the driver - API is unreachable: asks the exact binary that will do the - allocating. Carries no integrated verdict — callers must gate it.""" + """(engine_total_bytes, None) from the installed runtime's own --list-devices, or None. The + fallback truth source when the driver API is unreachable: asks the exact binary that will do the + allocating. Carries no integrated verdict — callers must gate it. + """ try: from hermes_cli.local_runtime.binaries import ( installed_tags, @@ -272,9 +240,9 @@ def _engine_device_pool() -> "tuple[int, bool | None] | None": def _device_pool_view() -> "tuple[int, bool | None] | None": - """Best available allocator-side view, cached: a hit is permanent for - the process, a miss retries after a short TTL (the engine binary can - appear mid-session via a pane install).""" + """Best available allocator-side view, cached: a hit is permanent for the process, a miss retries + after a short TTL (the engine binary can appear mid-session via a pane install). + """ global _pool_probe_cache now = time.monotonic() if _pool_probe_cache is not None: @@ -288,13 +256,10 @@ def _device_pool_view() -> "tuple[int, bool | None] | None": def _unified_pool_bytes(smi_total: int, ram_total: int) -> int | None: """The real pool size when this NVIDIA device is unified memory behind - a WDDM carve-out, else None (trust nvidia-smi as ever). - The driver's INTEGRATED attribute decides when readable — in BOTH - directions (0 pins discrete even if the numbers look weird; a driver - that declares integrated is believed even at modest pool sizes). Only - an attribute-less view (engine fallback) needs the two numeric gates; - both must hold and no discrete card meets either. + The driver's INTEGRATED attribute decides in BOTH directions when readable (0 pins discrete + even if numbers look odd; a declared-integrated driver is believed at modest pool sizes). Only + the attribute-less engine fallback needs the two numeric gates, both of which must hold. """ view = _device_pool_view() if view is None: @@ -310,20 +275,19 @@ def _unified_pool_bytes(smi_total: int, ram_total: int) -> int | None: return None +def _uma_budget(base: int, total: int) -> HardwareBudget: + usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION))) + return HardwareBudget(usable_vram_bytes=usable, total_device_bytes=total, + ram_available_bytes=0, uma=True) + + def probe_budget(*, planning: bool = False) -> HardwareBudget: """Construct the budget per the source rules above. - ``planning=False`` (default): LIVE budget — free VRAM right now. The - right input for launch-time fit decisions and growth re-grants. - - ``planning=True``: CAPACITY budget — what this machine can run once - the runtime manages placement (total device memory minus the margin). - The right input for catalog pricing and quant selection: pricing - against live-free while a model is already loaded made every row read - 'larger than your GPU memory' and degraded quant picks to Q2 on a - 32 GiB card. The managed server - unloads/relaunches models itself, so at load time the capacity is - genuinely available. + ``planning=False``: LIVE budget (free VRAM now) for launch-time fit and growth re-grants. + ``planning=True``: CAPACITY budget (total minus margin) for catalog pricing and quant selection; + pricing against live-free while a model was loaded made every row read as too large and + degraded quant picks. The managed server unloads/relaunches itself, so capacity is real. """ ram_total, ram_avail = _ram_bytes() vram = _nvidia_vram() @@ -355,25 +319,17 @@ def probe_budget(*, planning: bool = False) -> HardwareBudget: # Without smi, OS-available alone is the honest floor. live = (vram[1] + ram_avail) if vram else ram_avail base = min(unified, live) - usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION))) - return HardwareBudget(usable_vram_bytes=usable, - total_device_bytes=unified, - ram_available_bytes=0, uma=True) + return _uma_budget(base, unified) if vram is None: # No NVIDIA device visible: Metal/Vulkan/CPU paths budget from RAM # as UMA (Apple Silicon) — conservative for discrete AMD until a # vendor probe lands (E3 hardware). - base = ram_total if planning else ram_avail - usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION))) - return HardwareBudget(usable_vram_bytes=usable, - total_device_bytes=ram_total, - ram_available_bytes=0, uma=True) + return _uma_budget(ram_total if planning else ram_avail, ram_total) total, free = vram margin = max(_MARGIN_FLOOR, int(total * _MARGIN_FRACTION)) - base = total if planning else free - return HardwareBudget(usable_vram_bytes=max(0, base - margin), + return HardwareBudget(usable_vram_bytes=max(0, (total if planning else free) - margin), total_device_bytes=total, - ram_available_bytes=ram_avail if not planning else ram_total, + ram_available_bytes=ram_total if planning else ram_avail, uma=False) diff --git a/hermes_cli/local_runtime/hf_browse.py b/hermes_cli/local_runtime/hf_browse.py index e554ee1fc4..d69e1fbc85 100644 --- a/hermes_cli/local_runtime/hf_browse.py +++ b/hermes_cli/local_runtime/hf_browse.py @@ -1,19 +1,11 @@ """Browse Hugging Face for GGUF models the user can run. -The curated catalog is the front page; this module is the firehose behind -it — day-0 models not yet in the catalog, community quants, -anything. Three rules keep it safe and honest: +The curated catalog is the front page; this module is the firehose behind it — day-0 models not yet +in the catalog, community quants, anything. Three rules keep it safe and honest: -1. Acquisition only. Nothing here serves a model: a browsed download - lands in the machine-scoped models dir and from that moment the - normal machinery owns it — staleness bounce, preset generation from - the real GGUF header, fit policy, placement pills. -2. The fit verdict shown BEFORE download is a rough cut priced from file - size alone (weights dominate; KV/overhead use conservative fill-ins). - After download the GGUF header is the authority, as everywhere. -3. HF is queried directly with short timeouts and a small in-process - cache. No third-party proxy service; if HF rate limits ever bite at - fleet scale, revisit with a caching proxy then. +1. Acquisition only. Nothing here serves a model: a browsed download lands in the machine-scoped +models dir and from that moment the normal machinery owns it — staleness bounce, preset generation +from the real GGUF header, fit policy, placement pills. 2. """ from __future__ import annotations @@ -24,7 +16,7 @@ import re import time import urllib.parse import urllib.request -from dataclasses import dataclass, field +from dataclasses import dataclass, replace logger = logging.getLogger(__name__) @@ -87,16 +79,12 @@ def search_models(query: str, limit: int = 20) -> list[HFModelHit]: q = urllib.parse.quote(query.strip()) url = (f"{_HF}/api/models?search={q}&filter=gguf&sort=downloads" f"&direction=-1&limit={max(1, min(int(limit), 50))}") - out: list[HFModelHit] = [] - for m in _get_json(url): - out.append(HFModelHit( - repo=str(m.get("id", "")), - downloads=int(m.get("downloads") or 0), - likes=int(m.get("likes") or 0), - updated=str(m.get("lastModified") or ""), - gated=bool(m.get("gated")), - )) - return out + return [HFModelHit(repo=str(m.get("id", "")), + downloads=int(m.get("downloads") or 0), + likes=int(m.get("likes") or 0), + updated=str(m.get("lastModified") or ""), + gated=bool(m.get("gated"))) + for m in _get_json(url)] def _quant_label(filename: str) -> str: @@ -128,10 +116,8 @@ def repo_files(repo: str) -> list[HFFileGroup]: else: singles.append((path, size)) - groups: list[HFFileGroup] = [] - for path, size in singles: - groups.append(HFFileGroup(label=_quant_label(path), paths=(path,), - total_bytes=size)) + groups = [HFFileGroup(label=_quant_label(path), paths=(path,), total_bytes=size) + for path, size in singles] for stem, parts in splits.items(): parts.sort() groups.append(HFFileGroup( @@ -143,9 +129,9 @@ def repo_files(repo: str) -> list[HFFileGroup]: def rough_fit(total_bytes: int, budget) -> str: - """Coarse pre-download verdict from file size alone. The GGUF header - refines this after download; bands match the catalog pills' language. - File size ≈ in-memory weights for GGUF (mmap'd as-is).""" + """Coarse pre-download verdict from file size alone. The GGUF header refines this after download; + bands match the catalog pills' language. File size ≈ in-memory weights for GGUF (mmap'd as-is). + """ need = total_bytes + _ROUGH_KV_AND_OVERHEAD if need <= budget.usable_vram_bytes: return "fits-gpu" @@ -155,7 +141,5 @@ def rough_fit(total_bytes: int, budget) -> str: def priced_repo_files(repo: str, budget) -> list[HFFileGroup]: - from dataclasses import replace - return [replace(g, fit=rough_fit(g.total_bytes, budget)) for g in repo_files(repo)] diff --git a/hermes_cli/local_runtime/load_progress.py b/hermes_cli/local_runtime/load_progress.py index 379955c6e8..da86a6ba94 100644 --- a/hermes_cli/local_runtime/load_progress.py +++ b/hermes_cli/local_runtime/load_progress.py @@ -1,24 +1,12 @@ """Live model-load progress from the managed llama-server router. -llama-server's child processes emit per-tensor load progress -({stages, current, value}, throttled upstream to ~200ms) which the -router relays ONLY over its /models/sse stream — GET /models carries -just the coarse status string. This module owns one lazy background -watcher on that stream and keeps an in-memory snapshot other code can -poll cheaply: +llama-server's child processes emit per-tensor load progress ({stages, current, value}, throttled +upstream to ~200ms) which the router relays ONLY over its /models/sse stream — GET /models carries +just the coarse status string. - get_loading_progress() -> {model_id: {"stage", "value", "percent"}} - -"percent" is a composite across stages so a bar doesn't sprint 0->100 -once per stage: the text model dominates load time (its weights dwarf -the mmproj/spec extras), so it gets the lion's share of the range and -the extras split the remainder. - -The watcher starts on first call, reconnects with backoff (the router -bounces on model download/eject), and never raises into callers — no -router, no state file, or no SSE support (older engines) all read as -"nothing loading". Safe from any process on the machine: the endpoint -comes from the supervisor's machine-scoped state file. +The watcher starts on first call, reconnects with backoff (the router bounces on model +download/eject), and never raises into callers — no router, no state file, or no SSE support (older +engines) all read as "nothing loading". """ from __future__ import annotations @@ -58,11 +46,11 @@ def _composite_percent(stages: list[str], current: str, value: float) -> int: def _endpoint() -> "tuple[str, str] | None": """(base_root, api_key) of the managed router, or None. - Resolved through the endpoint module's ownership-guarded reader, not - a raw state-file read: on the shared stable port, a foreign install's - server answers /health for anyone, and a raw read would attach this - watcher to someone else's SSE stream (or spin on 401s against it). - The guard's dead-pid check is the ownership proof.""" + Resolved through the endpoint module's ownership-guarded reader, not a raw state-file read: on + the shared stable port a foreign install's server answers /health for anyone, and a raw read + would attach this watcher to someone else's SSE stream. The dead-pid check is the ownership + proof. + """ try: from hermes_cli.local_runtime.endpoint import _state_endpoint @@ -160,17 +148,13 @@ def get_loading_progress() -> dict[str, dict]: def get_prefill_progress(model: str) -> "dict | None": - """{"processed": tokens} while the managed server is prompt-processing - for ``model``, or None (idle, decoding, unreachable, or foreign server). + """{"processed": tokens} while the managed server is prompt-processing for ``model``, or None + (idle, decoding, unreachable, or foreign server). - llama-server's /slots reports ``n_prompt_tokens_processed`` climbing in - real time during prefill, but exposes no total — callers supply their - own denominator (the request's estimated token count). Busiest - processing slot wins when several are active: a parallel small request - (title generation) freezes its counter during decode while a live - prefill keeps climbing past it. One authenticated HTTP call per poll; - every failure reads as "no prefill" — this is garnish, never load- - bearing. + llama-server's /slots exposes ``n_prompt_tokens_processed`` but no total, so callers supply the + denominator. The busiest processing slot wins: a parallel small request freezes its counter + during decode while a live prefill keeps climbing. One HTTP call per poll; every failure reads + as "no prefill" — garnish, never load-bearing. """ ep = _endpoint() if ep is None: diff --git a/hermes_cli/local_runtime/presets.py b/hermes_cli/local_runtime/presets.py index a5aa96cfa3..a4008c22a9 100644 --- a/hermes_cli/local_runtime/presets.py +++ b/hermes_cli/local_runtime/presets.py @@ -1,9 +1,5 @@ -"""Per-model preset generation (--models-preset INI) — the router-side -carrier for context-policy launch decisions. - -The INI shape is what the router itself generates per child: a -[model-id] section whose keys are long-form -llama-server flag names without the leading dashes. +"""Per-model preset generation (--models-preset INI) — the router-side carrier for context-policy +launch decisions. """ from __future__ import annotations @@ -69,16 +65,10 @@ def _args_to_keys(args: list[str]) -> dict[str, str]: def generate_presets(models_dir: Path, budget: HardwareBudget, preset_path: Path, mtp_capable: set[str] | None = None) -> list[PresetEntry]: - """Walk the staged models, run the launch decision per model, and - write one INI. Refused models get no section (the router simply won't - have policy for them; the picker surfaces the refusal + smaller-quant - suggestion from the returned entries). - - Catalog-declared companions merge in here: sampling defaults (policy - keys always win), the vision projector when present, and a spec-decode - draft model iff the decision spilled — the rule: speculative - decode is a spill amplifier, so a resident draft accelerates a spilled - main model; a zero-spill model doesn't pay the draft's memory.""" + """Walk the staged models, run the launch decision per model, and write one INI. Refused models get + no section (the router simply won't have policy for them; the picker surfaces the refusal + + smaller-quant suggestion from the returned entries). + """ from hermes_cli.local_runtime.bootstrap import assets_dir from hermes_cli.local_runtime.catalog import find_entry_for_model @@ -221,9 +211,9 @@ def generate_presets(models_dir: Path, budget: HardwareBudget, def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEntry]: - """The launch decisions the running server was actually given, read - back from the preset INI (the INI is the record — it's what spawned - the children). Missing/unparseable file returns {}.""" + """The launch decisions the running server was actually given, read back from the preset INI (the + INI is the record — it's what spawned the children). Missing/unparseable file returns {}. + """ import configparser if preset_path is None: diff --git a/hermes_cli/local_runtime/supervisor.py b/hermes_cli/local_runtime/supervisor.py index 44f162cf48..23d2fb258b 100644 --- a/hermes_cli/local_runtime/supervisor.py +++ b/hermes_cli/local_runtime/supervisor.py @@ -1,20 +1,12 @@ """Supervision of one llama-server in router mode. -The router process is ours (restart with backoff on crash); router children -are its problem — child failures surface via GET /models exit_code, never -auto-retried here. +The router process is ours (restart with backoff on crash); router children are its problem — child +failures surface via GET /models exit_code, never auto-retried here. -Readiness rules (each learned the hard way on real hardware): -- health-200 is NOT readiness; every readiness claim requires a touch - generation (temp-0, expected token, generous budget, reasoning_content - scanned). -- Always dial 127.0.0.1 — resolving localhost adds ~2s per request on - Windows via IPv6 fallback. -- /metrics is opt-in (--metrics) and carries no KV-usage metric; - idleness = requests_processing == 0 and no slot is_processing. -- The router's LRU eviction has no pin for the primary model: until an - upstream pin exists, keep_primary_loaded re-touches the primary after - any other model load. +Readiness rules (each learned the hard way on real hardware): - health-200 is NOT readiness; every +readiness claim requires a touch generation (temp-0, expected token, generous budget, +reasoning_content scanned). - Always dial 127.0.0.1 — resolving localhost adds ~2s per request on +Windows via IPv6 fallback. """ from __future__ import annotations @@ -60,9 +52,9 @@ _DEFAULT_PORT = 18434 def _stable_port() -> int: - """The stable default port, falling back to an ephemeral one only when - something else already listens there (and it isn't a leftover managed - server, which stop() would have cleaned up).""" + """The stable default port, falling back to an ephemeral one only when something else already + listens there (and it isn't a leftover managed server, which stop() would have cleaned up). + """ try: with socket.socket() as s: s.bind(("127.0.0.1", _DEFAULT_PORT)) @@ -77,12 +69,9 @@ def _stable_port() -> int: def _stable_api_key() -> str: """One key for the life of the install, persisted beside the runtimes. - Endpoint identity must survive restarts as a UNIT — sessions persist the - resolved base_url + api_key, so a per-boot key strands every resumed - session on HTTP 401 exactly the way a per-boot port would strand them - on connection errors. Rotating it buys nothing: the key exists to stop - other loopback processes free-riding, and it lives on the same disk as - the state file that would leak it. Delete the file to rotate manually. + Endpoint identity must survive restarts as a UNIT — sessions persist the resolved base_url + + api_key, so a per-boot key strands every resumed session on HTTP 401 exactly the way a per-boot + port would strand them on connection errors. """ key_path = runtimes_root() / ".api_key" try: @@ -102,16 +91,7 @@ def _stable_api_key() -> str: class LlamaServerSupervisor: - """Own one llama-server router process for the life of a Hermes session. - - Usage:: - - sup = LlamaServerSupervisor(install_dir, models_dir) - sup.start() # spawn + wait healthy - sup.ensure_model_ready(name) # load + touch-generate - ... sup.base_url is the /v1 endpoint, sup.api_key its key ... - sup.stop() - """ + """Own one llama-server router process for the life of a Hermes session.""" def __init__(self, install_dir: Path, models_dir: Path, *, models_max: int = 4, port: int | None = None, @@ -143,14 +123,20 @@ class LlamaServerSupervisor: def _url(self, route: str) -> str: return f"http://127.0.0.1:{self.port}{route}" - def _request(self, route: str, body: dict | None = None, timeout_s: int = 30) -> dict: + def _open(self, route: str, body: dict | None = None, timeout_s: int = 30, + *, json_type: bool = True): + headers = {"Authorization": f"Bearer {self.api_key}"} + if json_type: + headers["Content-Type"] = "application/json" req = urllib.request.Request( self._url(route), data=json.dumps(body).encode() if body is not None else None, - headers={"Content-Type": "application/json", - "Authorization": f"Bearer {self.api_key}"}, + headers=headers, ) - with urllib.request.urlopen(req, timeout=timeout_s) as r: + return urllib.request.urlopen(req, timeout=timeout_s) + + def _request(self, route: str, body: dict | None = None, timeout_s: int = 30) -> dict: + with self._open(route, body, timeout_s) as r: raw = r.read() return json.loads(raw) if raw else {} @@ -182,9 +168,7 @@ class LlamaServerSupervisor: ] if self.preset_path and self.preset_path.exists(): cmd += ["--models-preset", str(self.preset_path)] - cmd += [ - *self.extra_args, - ] + cmd += self.extra_args self.log_path.parent.mkdir(parents=True, exist_ok=True) if self._log_handle is not None: # The crash-restart loop calls _spawn repeatedly; without @@ -279,14 +263,10 @@ class LlamaServerSupervisor: def _terminate_tree(proc: subprocess.Popen) -> None: """Terminate the router AND its model children. - The router spawns one child llama-server per loaded model, each - holding gigabytes of VRAM. Terminating only the router (on - Windows, TerminateProcess — no signal handlers, no cleanup pass) - orphans those children: the port goes quiet but the weights stay - resident, and the next spawn re-loads models alongside a ghost - still holding the memory. Enumerate children FIRST (the parent - must be alive to walk them), then terminate parent and children - together, escalating to kill for stragglers. + The router spawns one child llama-server per loaded model, each holding gigabytes of VRAM. + Terminating only the router (on Windows, TerminateProcess with no cleanup) orphans them: the + port goes quiet but the weights stay resident. Enumerate children FIRST (the parent must be + alive to walk them), then terminate all together, escalating to kill for stragglers. """ children: list = [] try: @@ -315,12 +295,10 @@ class LlamaServerSupervisor: def _reap_orphaned_children(self) -> None: """Kill model children orphaned by a router crash, before respawn. - A crashed router can't clean up its children, and a dead parent - can't be walked — so match by identity instead: any process - running OUR llama-server binary whose parent is gone is an - orphan of a previous router. Their VRAM must come back before - the new router loads models next to the ghosts. External - llama-servers (different binary path) never match. + A crashed router can't clean up its children, and a dead parent can't be walked — so match + by identity instead: any process running OUR llama-server binary whose parent is gone is an + orphan of a previous router. Their VRAM must come back before the new router loads models + next to the ghosts. """ try: import psutil @@ -350,31 +328,13 @@ class LlamaServerSupervisor: return {m["id"]: m.get("status", {}).get("value", "unknown") for m in data.get("data", [])} - def model_failures(self) -> dict: - """{model_id: exit_code} for children that died — surfaced to the - UI, never auto-retried (design: router children are its problem).""" - data = self._request("/models") - out = {} - for m in data.get("data", []): - status = m.get("status", {}) - if status.get("value") == "failed" or status.get("exit_code"): - out[m["id"]] = status.get("exit_code") - return out - def load_model(self, model_id: str, timeout_s: int = 600) -> None: self._request("/models/load", {"model": model_id}, timeout_s=timeout_s) def unload_model(self, model_id: str) -> None: - """Free the child's VRAM now. Route existence verified empirically - on b10290 (POST /models/unload; bogus name -> 400 'model is not - found'). Momentary action: never touches primary_model — the - declaration is durable, an eject is not (residency design). - - Settle before returning: for a few seconds after unload returns, - the router still routes to the dying child and answers chat with - 500 'proxy error: Could not establish connection' (probed on - b10362). Waiting for the model to report unloaded means the next - message autoloads cleanly instead of racing the teardown. + """Free the child's VRAM now. Route existence verified empirically on b10290 (POST + /models/unload; bogus name -> 400 'model is not found'). Momentary action: never touches + primary_model — the declaration is durable, an eject is not (residency design). """ self._request("/models/unload", {"model": model_id}, timeout_s=120) deadline = time.monotonic() + 15 @@ -406,10 +366,7 @@ class LlamaServerSupervisor: except Exception: # noqa: BLE001 return unloaded for model_id, status in statuses.items(): - if status not in ("loaded", "ready"): - self._idle_since.pop(model_id, None) - continue - if not self.is_idle(model_id): + if status not in ("loaded", "ready") or not self.is_idle(model_id): self._idle_since.pop(model_id, None) continue first_idle = self._idle_since.setdefault(model_id, now) @@ -425,9 +382,9 @@ class LlamaServerSupervisor: return unloaded def touch_generate(self, model_id: str, timeout_s: int = 300) -> bool: - """The readiness proof. Generous budget + reasoning_content scan — - small token budgets false-fail reasoning models, which spend their - first tokens thinking.""" + """The readiness proof. Generous budget + reasoning_content scan — small token budgets + false-fail reasoning models, which spend their first tokens thinking. + """ try: resp = self._request("/v1/chat/completions", { "model": model_id, @@ -450,32 +407,13 @@ class LlamaServerSupervisor: self.load_model(model_id, timeout_s=timeout_s) return self.touch_generate(model_id) - def actual_n_ctx(self, model_id: str) -> int | None: - """/props reconciliation: the granted window as the child reports - it — the compressor's budget and the picker's 'running at 87K of - 262K' both read THIS value, never the request (design step 4).""" - try: - props = self._request(f"/props?model={model_id}") - return props.get("default_generation_settings", {}).get("n_ctx") - except Exception: # noqa: BLE001 - return None - - def keep_primary_loaded(self) -> None: - """The router's LRU eviction has no pin, so after any other load - re-touch the primary to keep it most-recently-used. Best-effort - under bursty multi-model load — replaced when an upstream pin - exists.""" - if self.primary_model and self.models().get(self.primary_model) in ( - "loaded", "ready"): - self.touch_generate(self.primary_model, timeout_s=60) - # ── telemetry ──────────────────────────────────────────── def is_idle(self, model_id: str | None = None) -> bool: - """No processing requests and no busy slots. Router quirk: /slots - and /metrics are per-child and require ?model= (bare calls 400), - and no KV-usage metric exists. With ``model_id`` checks that one - child; without, every loaded child.""" + """No processing requests and no busy slots. Router quirk: /slots and /metrics are per-child + and require ?model= (bare calls 400), and no KV-usage metric exists. With ``model_id`` + checks that one child; without, every loaded child. + """ try: if model_id is not None: loaded = [model_id] @@ -486,10 +424,7 @@ class LlamaServerSupervisor: slots = self._request(f"/slots?model={mid}") if any(s.get("is_processing") for s in slots): return False - req = urllib.request.Request( - self._url(f"/metrics?model={mid}"), - headers={"Authorization": f"Bearer {self.api_key}"}) - with urllib.request.urlopen(req, timeout=10) as r: + with self._open(f"/metrics?model={mid}", timeout_s=10, json_type=False) as r: text = r.read().decode() for line in text.splitlines(): if line.startswith("llamacpp:requests_processing"): diff --git a/hermes_cli/model_catalog.py b/hermes_cli/model_catalog.py index 1e869b746a..029e2bcd72 100644 --- a/hermes_cli/model_catalog.py +++ b/hermes_cli/model_catalog.py @@ -1,45 +1,13 @@ """Remote model catalog fetcher. -The Hermes docs site hosts a JSON manifest of curated models for providers -we want to update without shipping a release (currently OpenRouter and -Nous Portal). This module fetches, validates, and caches that manifest, -falling back to the in-repo hardcoded lists when the network is unavailable. +Pipeline -------- 1. ``get_catalog()`` — returns a parsed manifest dict. - Checks in-process cache +(invalidated by TTL). - Reads disk cache at ``~/.hermes/cache/model_catalog.json``. - Fetches the +master URL if disk cache is stale or missing. - On any fetch failure, keeps using the stale cache +(or empty dict). -Pipeline --------- -1. ``get_catalog()`` — returns a parsed manifest dict. - - Checks in-process cache (invalidated by TTL). - - Reads disk cache at ``~/.hermes/cache/model_catalog.json``. - - Fetches the master URL if disk cache is stale or missing. - - On any fetch failure, keeps using the stale cache (or empty dict). - -2. ``get_curated_openrouter_models()`` / ``get_curated_nous_models()`` — - thin accessors returning the shapes existing callers expect. Each - falls back to the in-repo hardcoded list on any lookup failure. - -Schema (version 1) ------------------- -:: - - { - "version": 1, - "updated_at": "2026-04-25T22:00:00Z", - "metadata": {...}, # free-form - "providers": { - "openrouter": { - "metadata": {...}, # free-form - "models": [ - {"id": "vendor/model", "description": "recommended", - "metadata": {...}} # free-form, model-level - ] - }, - "nous": {...} - } - } - -Unknown fields are ignored — extra metadata can be added at either level -without bumping ``version``. ``version`` bumps are reserved for -breaking changes (renaming ``providers``, changing ``models`` shape). +2. ``get_curated_openrouter_models()`` / ``get_curated_nous_models()`` — thin accessors returning +the shapes existing callers expect. Each falls back to the in-repo hardcoded list on any lookup +failure. """ from __future__ import annotations @@ -177,10 +145,9 @@ def _fetch_manifest_with_fallback( ) -> dict[str, Any] | None: """Try ``primary_url`` first, then walk ``fallback_urls``. - Returns the first manifest that fetches and validates, or None when - every URL fails. Skips fallback URLs identical to the primary so an - operator who configured the catalog URL to point at the raw GitHub - copy doesn't double-fetch. + Returns the first manifest that fetches and validates, or None when every URL fails. Skips + fallback URLs identical to the primary so an operator who configured the catalog URL to point at + the raw GitHub copy doesn't double-fetch. """ data = _fetch_manifest(primary_url, timeout) if data is not None: @@ -289,8 +256,8 @@ def _spawn_catalog_swr_refresh(url: str) -> None: def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]: """Return the parsed model catalog manifest, or an empty dict on failure. - Callers should treat a missing provider/model as "use the in-repo fallback" - — never raise from this function so the CLI keeps working offline. + Callers should treat a missing provider/model as "use the in-repo fallback" — never raise from + this function so the CLI keeps working offline. """ global _catalog_cache, _catalog_cache_source_mtime @@ -359,11 +326,9 @@ def refresh_interval_seconds() -> float: def refresh_catalogs() -> bool: """Force-refresh every remote model catalog the picker reads from. - Fetches the curated manifest, the OpenRouter live list (tool-support / - free-pricing filter) and the Nous Portal recommendations, writing each - to its disk cache so the next ``/model`` open in ANY process on this - machine sees the new lists. Blocking; run it off the event loop. - Returns True when the manifest refresh succeeded. + Fetches the curated manifest, the OpenRouter live list (tool-support / free-pricing filter) and + the Nous Portal recommendations, writing each to its disk cache so the next ``/model`` open in + ANY process on this machine sees the new lists. Blocking; run it off the event loop. """ if not _load_catalog_config()["enabled"]: return False @@ -411,11 +376,7 @@ def _get_provider_block(provider: str) -> dict[str, Any] | None: def get_curated_openrouter_models() -> list[tuple[str, str]] | None: - """Return OpenRouter's curated ``[(id, description), ...]`` from the manifest. - - Returns ``None`` when the manifest is unavailable, so callers can fall - back to their hardcoded list. - """ + """Return OpenRouter's curated ``[(id, description), ...]`` from the manifest.""" block = _get_provider_block("openrouter") if not block: return None @@ -430,10 +391,7 @@ def get_curated_openrouter_models() -> list[tuple[str, str]] | None: def get_curated_nous_models() -> list[str] | None: - """Return Nous Portal's curated list of model ids from the manifest. - - Returns ``None`` when the manifest is unavailable. - """ + """Return Nous Portal's curated list of model ids from the manifest.""" block = _get_provider_block("nous") if not block: return None @@ -460,14 +418,8 @@ def _default_model_from_block(block: dict[str, Any] | None) -> str | None: def get_default_model_from_cache(provider: str) -> str | None: """Return the catalog's labeled default model for ``provider`` — cache only. - The manifest marks exactly one model entry per provider with - ``"default": true``; that entry is the model Hermes silently lands on when - the user never picked one. This accessor reads ONLY the in-process copy or - the disk cache — it NEVER triggers a network fetch, so it is safe on hot - resolution paths (agent build, gateway session setup) that must stay - network-free. The cache is kept fresh by the picker/`hermes update` paths; - when no cached manifest exists (fresh install, offline), returns None and - the caller falls back to the in-repo constant. + The manifest marks exactly one model entry per provider with ``"default": true``; that entry is + the model Hermes silently lands on when the user never picked one. """ if _catalog_cache is not None: block = _catalog_cache.get("providers", {}).get(provider) @@ -484,18 +436,13 @@ def get_default_model_from_cache(provider: str) -> str | None: def seed_cache_from_checkout(project_root: "Path | str") -> bool: """Overwrite the disk cache with the catalog shipped in a local checkout. - ``hermes update`` pulls the latest repo, so the freshly-pulled - ``website/static/api/model-catalog.json`` IS the newest catalog — no - network round-trip needed. Copying it straight over the disk cache keeps - the model picker current even when the remote manifest fetch is bot-gated + ``hermes update`` pulls the latest repo, so the freshly-pulled ``website/static/api/model- + catalog.json`` IS the newest catalog — no network round-trip needed. Copying it straight over + the disk cache keeps the model picker current even when the remote manifest fetch is bot-gated or the Portal hiccups. - Reads the shipped manifest, validates it against the schema, and writes it - to ``~/.hermes/cache/model_catalog.json`` via the same atomic writer the - network path uses. Returns ``True`` on success, ``False`` if the file is - missing, malformed, or fails validation (caller should treat a ``False`` - as non-fatal — the network fetch path still applies on the next picker - open). + Reads the shipped manifest, validates it against the schema, and writes it to + ``~/.hermes/cache/model_catalog.json`` via the same atomic writer the network path uses. """ src = Path(project_root) / "website" / "static" / "api" / "model-catalog.json" try: diff --git a/hermes_cli/model_cost_guard.py b/hermes_cli/model_cost_guard.py index 1ed8835b5b..16d4f0fd51 100644 --- a/hermes_cli/model_cost_guard.py +++ b/hermes_cli/model_cost_guard.py @@ -37,9 +37,7 @@ def _to_decimal(value: object) -> Optional[Decimal]: def _format_money(value: Optional[Decimal]) -> str: - if value is None: - return "unknown" - return f"${value:.2f}/M" + return "unknown" if value is None else f"${value:.2f}/M" def _pricing_from_model_info( @@ -56,9 +54,7 @@ def _pricing_from_model_info( def _known_models_dev_provider(provider: Optional[str]) -> Optional[str]: normalized = (provider or "").strip().lower() - if not normalized: - return None - return PROVIDER_TO_MODELS_DEV.get(normalized) + return PROVIDER_TO_MODELS_DEV.get(normalized) if normalized else None def _can_trust_model_info_pricing( @@ -98,8 +94,8 @@ def expensive_model_warning( ) -> Optional[ExpensiveModelWarning]: """Return a warning payload when known pricing exceeds safety thresholds. - The guard only triggers when pricing is known. Callers should use this after - model resolution so aliases and provider-specific model IDs have settled. + The guard only triggers when pricing is known. Callers should use this after model resolution so + aliases and provider-specific model IDs have settled. """ model = (model_name or "").strip() if not model: @@ -112,11 +108,10 @@ def expensive_model_warning( if _can_trust_model_info_pricing(provider, model_info): input_cost, output_cost, source = _pricing_from_model_info(model_info) - if ( - input_cost is None - and output_cost is None - and _known_models_dev_provider(provider) - ): + def _unpriced() -> bool: + return input_cost is None and output_cost is None + + if _unpriced() and _known_models_dev_provider(provider): try: from agent.models_dev import get_model_info @@ -126,11 +121,7 @@ def expensive_model_warning( except Exception: pass - if ( - input_cost is None - and output_cost is None - and _can_trust_pricing_lookup(model, provider=provider, base_url=base_url) - ): + if _unpriced() and _can_trust_pricing_lookup(model, provider=provider, base_url=base_url): try: from agent.usage_pricing import get_pricing_entry @@ -149,12 +140,8 @@ def expensive_model_warning( is_known_gpt55_pro_confusion = model.lower() == GPT55_PRO_OPENROUTER_ID - over_input = ( - input_cost is not None and input_cost > INPUT_COST_WARNING_THRESHOLD - ) - over_output = ( - output_cost is not None and output_cost > OUTPUT_COST_WARNING_THRESHOLD - ) + over_input = input_cost is not None and input_cost > INPUT_COST_WARNING_THRESHOLD + over_output = output_cost is not None and output_cost > OUTPUT_COST_WARNING_THRESHOLD if not over_input and not over_output and not is_known_gpt55_pro_confusion: return None @@ -164,10 +151,7 @@ def expensive_model_warning( f"{model} has known pricing above Hermes' safety threshold.", f"Input tokens: {_format_money(input_cost)}", f"Output tokens: {_format_money(output_cost)}", - ( - "Threshold: more than $20/M input tokens or more than " - "$100/M output tokens." - ), + "Threshold: more than $20/M input tokens or more than $100/M output tokens.", ] if source: lines.append(f"Pricing source: {source}.") diff --git a/hermes_cli/model_data_policy_guard.py b/hermes_cli/model_data_policy_guard.py index 6cb86991dc..68c04d3e98 100644 --- a/hermes_cli/model_data_policy_guard.py +++ b/hermes_cli/model_data_policy_guard.py @@ -1,22 +1,11 @@ """Data-policy confirmation helpers for model selection surfaces. -Some inference tiers are cheap *because* the vendor trains future models on your -prompts and completions. Selecting one for the low price without realising the -data trade-off is a real footgun. This guard mirrors -``hermes_cli.model_cost_guard`` — it returns a warning payload that the CLI and -web model-selection flows surface as an explicit confirm step. +Some inference tiers are cheap *because* the vendor trains future models on your prompts and +completions. Selecting one for the low price without realising the data trade-off is a real footgun. -Why a static table (not a ProviderProfile hook): the guard runs inside core -selection code (``auth.py`` / ``web_server.py``), which never calls into the -active provider profile for a selection-time warning. Keeping the rule set here -also means it renders regardless of which provider plugin happens to be loaded, -and it stays testable without importing arbitrary third-party plugin code into -the selection path. - -The status is NOT machine-readable anywhere today: neither models.dev nor the -Meta ``/v1/models`` payload exposes a training/retention flag (verified -2026-08-07). The only reliable signals are the vendor-documented model id and -its anomalously low pricing, so the rule keys on the id. +Why a static table (not a ProviderProfile hook): the guard runs inside core selection code +(``auth.py`` / ``web_server.py``), which never calls into the active provider profile for a +selection-time warning. """ from __future__ import annotations @@ -84,9 +73,9 @@ def data_training_warning( ) -> Optional[DataTrainingWarning]: """Return a warning payload when *model_name* selects a data-training tier. - Returns ``None`` when no rule matches (the common case). Callers should run - this after model resolution so aliases / provider-specific ids have settled, - and surface ``.message`` as a confirm prompt. + Returns ``None`` when no rule matches (the common case). Callers should run this after model + resolution so aliases / provider-specific ids have settled, and surface ``.message`` as a + confirm prompt. """ model = (model_name or "").strip() if not model: diff --git a/hermes_cli/model_normalize.py b/hermes_cli/model_normalize.py index 21d10c5e6f..0838cb207b 100644 --- a/hermes_cli/model_normalize.py +++ b/hermes_cli/model_normalize.py @@ -1,33 +1,4 @@ -"""Per-provider model name normalization. - -Different LLM providers expect model identifiers in different formats: - -- **Aggregators** (OpenRouter, Nous, AI Gateway, Kilo Code) need - ``vendor/model`` slugs like ``anthropic/claude-sonnet-4.6``. -- **Anthropic** native API expects bare names with dots replaced by - hyphens: ``claude-sonnet-4-6``. -- **Copilot** expects bare names *with* dots preserved: - ``claude-sonnet-4.6``. -- **OpenCode Zen** preserves dots for GPT/GLM/Gemini/Kimi/MiniMax-style - model IDs, but Claude still uses hyphenated native names like - ``claude-sonnet-4-6``. -- **OpenCode Go** preserves dots in model names: ``minimax-m2.7``. -- **DeepSeek** accepts only the first-class V-series IDs - (``deepseek-v4-pro``, ``deepseek-v4-flash``, and any future - ``deepseek-v-*``). The legacy aliases ``deepseek-chat`` and - ``deepseek-reasoner`` were retired on 2026-07-24 and are remapped to - ``deepseek-v4-flash`` (official non-thinking / thinking shims). Older - Hermes revisions folded every non-reasoner input into - ``deepseek-chat``, which on aggregators routes to V3 — so a user - picking V4 Pro was silently downgraded. -- **Custom** and remaining providers pass the name through as-is. - -This module centralises that translation so callers can simply write:: - - api_model = normalize_model_for_provider(user_input, provider) - -Inspired by Clawdbot's ``normalizeAnthropicModelId`` pattern. -""" +"""Per-provider model name normalization.""" from __future__ import annotations @@ -172,22 +143,9 @@ _DEEPSEEK_V_SERIES_RE = re.compile(r"^deepseek-v\d+([-.].+)?$") def _normalize_for_deepseek(model_name: str) -> str: """Map a model input to a DeepSeek-accepted identifier. - Rules: - - Retired aliases ``deepseek-chat`` / ``deepseek-reasoner`` (cut off - 2026-07-24) -> ``deepseek-v4-flash``. - - Already a known canonical (``deepseek-v4-pro``/``deepseek-v4-flash``) - -> pass through. - - Matches the V-series pattern ``deepseek-v...`` -> pass through - (covers future ``deepseek-v5-*`` and dated variants without a release). - - Contains a reasoner keyword (r1, think, reasoning, cot, reasoner) - -> ``deepseek-v4-flash``. - - Everything else -> ``deepseek-v4-flash``. - - Args: - model_name: The bare model name (vendor prefix already stripped). - - Returns: - A DeepSeek-accepted model identifier. + Retired aliases ``deepseek-chat``/``deepseek-reasoner`` and known canonicals map as expected; + anything matching ``deepseek-v...`` passes through so future V-series ids work without a + release; reasoner keywords and everything else fall back to ``deepseek-v4-flash``. """ bare = _strip_vendor_prefix(model_name).lower() @@ -216,28 +174,14 @@ def _normalize_for_deepseek(model_name: str) -> str: # --------------------------------------------------------------------------- def _strip_vendor_prefix(model_name: str) -> str: - """Remove a ``vendor/`` prefix if present. - - Examples:: - - >>> _strip_vendor_prefix("anthropic/claude-sonnet-4.6") - 'claude-sonnet-4.6' - >>> _strip_vendor_prefix("claude-sonnet-4.6") - 'claude-sonnet-4.6' - >>> _strip_vendor_prefix("meta-llama/llama-4-scout") - 'llama-4-scout' - """ + """Remove a ``vendor/`` prefix if present.""" if "/" in model_name: return model_name.split("/", 1)[1] return model_name def _dots_to_hyphens(model_name: str) -> str: - """Replace dots with hyphens in a model name. - - Anthropic's native API uses hyphens where marketing names use dots: - ``claude-sonnet-4.6`` -> ``claude-sonnet-4-6``. - """ + """Replace dots with hyphens in a model name.""" return model_name.replace(".", "-") @@ -257,17 +201,10 @@ def _normalize_provider_alias(provider_name: str) -> str: def _strip_matching_provider_prefix(model_name: str, target_provider: str) -> str: """Strip ``provider/`` only when the prefix matches the target provider. - This prevents arbitrary slash-bearing model IDs from being mangled on - native providers while still repairing manual config values like - ``zai/glm-5.1`` for the ``zai`` provider. - - ``custom`` is a generic bucket for arbitrary user-defined endpoints, not - a vendor identity like ``zai``/``gemini``/``xai``. An alias that merely - *resolves to* ``custom`` (e.g. ``ollama``, via ``_PROVIDER_ALIASES``) - does not mean a ``ollama/`` prefix is redundant -- it may be the actual - routing prefix a proxy in front of the custom endpoint (e.g. LiteLLM) - requires, as in ``ollama/glm-5.2``. Only a literal ``custom/`` prefix -- - the bucket's own name -- is treated as redundant here. + Prevents arbitrary slash-bearing ids from being mangled on native providers while still + repairing config values like ``zai/glm-5.1`` for ``zai``. ``custom`` is a bucket, not a vendor: + an alias that merely resolves to it (e.g. ``ollama``) may be a real routing prefix required by a + proxy such as LiteLLM, so only a literal ``custom/`` prefix is treated as redundant. """ if "/" not in model_name: return model_name @@ -289,31 +226,7 @@ def _strip_matching_provider_prefix(model_name: str, target_provider: str) -> st def detect_vendor(model_name: str) -> Optional[str]: - """Detect the vendor slug from a bare model name. - - Uses the first hyphen-delimited token of the model name to look up - the corresponding vendor in ``_VENDOR_PREFIXES``. Also handles - case-insensitive matching and special patterns. - - Args: - model_name: A model name, optionally already including a - ``vendor/`` prefix. If a prefix is present it is used - directly. - - Returns: - The vendor slug (e.g. ``"anthropic"``, ``"openai"``) or ``None`` - if no vendor can be confidently detected. - - Examples:: - - >>> detect_vendor("claude-sonnet-4.6") - 'anthropic' - >>> detect_vendor("gpt-5.4-mini") - 'openai' - >>> detect_vendor("anthropic/claude-sonnet-4.6") - 'anthropic' - >>> detect_vendor("my-custom-model") - """ + """Detect the vendor slug from a bare model name.""" name = model_name.strip() if not name: return None @@ -341,19 +254,8 @@ def detect_vendor(model_name: str) -> Optional[str]: def _prepend_vendor(model_name: str) -> str: """Prepend the detected ``vendor/`` prefix if missing. - Used for aggregator providers that require ``vendor/model`` format. - If the name already contains a ``/``, it is returned as-is. - If no vendor can be detected, the name is returned unchanged - (aggregators may still accept it or return an error). - - Examples:: - - >>> _prepend_vendor("claude-sonnet-4.6") - 'anthropic/claude-sonnet-4.6' - >>> _prepend_vendor("anthropic/claude-sonnet-4.6") - 'anthropic/claude-sonnet-4.6' - >>> _prepend_vendor("my-custom-thing") - 'my-custom-thing' + For aggregators that require ``vendor/model``. Names already containing ``/`` or with no + detectable vendor are returned unchanged (the aggregator may still accept them). """ if "/" in model_name: return model_name @@ -367,18 +269,8 @@ def _prepend_vendor(model_name: str) -> str: def _repair_prefix_from_catalogue(model_name: str, provider: str) -> str: """Restore a dropped ``vendor/`` prefix using the provider's catalogue. - Unlike :func:`_prepend_vendor`, this never guesses from the model's name - shape — it only repairs a bare id that matches **exactly one** curated - entry for this provider modulo the prefix. That keeps self-hosted models - behind the same provider id (local NIM containers, proxies) untouched, - since they aren't in the catalogue. - - Examples:: - - >>> _repair_prefix_from_catalogue("nemotron-3-ultra-550b-a55b", "nvidia") - 'nvidia/nemotron-3-ultra-550b-a55b' - >>> _repair_prefix_from_catalogue("my-local-nim", "nvidia") - 'my-local-nim' + Unlike :func:`_prepend_vendor`, this never guesses from the model's name shape — it only repairs + a bare id that matches **exactly one** curated entry for this provider modulo the prefix. """ if "/" in model_name: return model_name @@ -404,11 +296,10 @@ def _repair_prefix_from_catalogue(model_name: str, provider: str) -> str: def suggest_prefixed_model_id(provider: str, model_name: str) -> Optional[str]: """Return the prefixed catalogue id for a bare *model_name*, if unambiguous. - The diagnostic counterpart to :func:`_repair_prefix_from_catalogue`: used - to explain a provider's content-free 404 when the configured id lost its - ``vendor/`` prefix. Returns ``None`` when the name already has a prefix, - the provider has no curated catalogue, or nothing matches — so callers can - stay silent rather than guess (#78796). + Diagnostic counterpart to :func:`_repair_prefix_from_catalogue`, used to explain a provider's + content-free 404 when the configured id lost its ``vendor/`` prefix. Returns ``None`` when the + name already has a prefix, the provider has no catalogue, or nothing matches — so callers stay + silent rather than guess. """ name = (model_name or "").strip() if not name or "/" in name: @@ -428,64 +319,9 @@ def suggest_prefixed_model_id(provider: str, model_name: str) -> Optional[str]: def normalize_model_for_provider(model_input: str, target_provider: str) -> str: """Translate a model name into the format the target provider's API expects. - This is the primary entry point for model name normalisation. It - accepts any user-facing model identifier and transforms it for the - specific provider that will receive the API call. - - Args: - model_input: The model name as provided by the user or config. - Can be bare (``"claude-sonnet-4.6"``), vendor-prefixed - (``"anthropic/claude-sonnet-4.6"``), or already in native - format (``"claude-sonnet-4-6"``). - target_provider: The canonical Hermes provider id, e.g. - ``"openrouter"``, ``"anthropic"``, ``"copilot"``, - ``"deepseek"``, ``"custom"``. Should already be normalised - via ``hermes_cli.models.normalize_provider()``. - - Returns: - The model identifier string that the target provider's API - expects. - - Raises: - No exceptions -- always returns a best-effort string. - - Examples:: - - >>> normalize_model_for_provider("claude-sonnet-4.6", "openrouter") - 'anthropic/claude-sonnet-4.6' - - >>> normalize_model_for_provider("anthropic/claude-sonnet-4.6", "anthropic") - 'claude-sonnet-4-6' - - >>> normalize_model_for_provider("anthropic/claude-sonnet-4.6", "copilot") - 'claude-sonnet-4.6' - - >>> normalize_model_for_provider("openai/gpt-5.4", "copilot") - 'gpt-5.4' - - >>> normalize_model_for_provider("claude-sonnet-4.6", "opencode-zen") - 'claude-sonnet-4-6' - - >>> normalize_model_for_provider("minimax-m2.5-free", "opencode-zen") - 'minimax-m2.5-free' - - >>> normalize_model_for_provider("deepseek-v3", "deepseek") - 'deepseek-v4-flash' - - >>> normalize_model_for_provider("deepseek-r1", "deepseek") - 'deepseek-v4-flash' - - >>> normalize_model_for_provider("deepseek-reasoner", "deepseek") - 'deepseek-v4-flash' - - >>> normalize_model_for_provider("my-model", "custom") - 'my-model' - - >>> normalize_model_for_provider("claude-sonnet-4.6", "zai") - 'claude-sonnet-4.6' - - >>> normalize_model_for_provider("MiMo-V2.5-Pro", "xiaomi") - 'mimo-v2.5-pro' + Primary entry point for model-name normalisation. Accepts bare, vendor-prefixed or native ids; + ``target_provider`` should already be normalised via ``normalize_provider()``. Never raises — + always returns a best-effort string. """ name = (model_input or "").strip() if not name: diff --git a/hermes_cli/model_search.py b/hermes_cli/model_search.py index 23d3bae14b..74bb8daa27 100644 --- a/hermes_cli/model_search.py +++ b/hermes_cli/model_search.py @@ -1,12 +1,4 @@ -"""Picker-only search aliases for model ids. - -Wire IDs stay unchanged. Some providers report short or brand-less ids -(Kimi Coding's flagship is literally ``k3``) that users still search for by -the familiar ``kimi-…`` naming of sibling models. - -Keep in sync with ``ui-tui/src/lib/model-search-text.ts`` and -``web/src/lib/model-search-text.ts``. -""" +"""Picker-only search aliases for model ids.""" from __future__ import annotations @@ -30,11 +22,7 @@ _MODEL_ALIAS_CANONICAL: dict[str, str] = { def model_alias_canonical(model: str) -> str: - """Return the canonical public slug for a bare wire-id alias. - - Identity for ids with no alias entry. Lowercases the input so callers - can use the result directly as a dedup key. - """ + """Return the canonical public slug for a bare wire-id alias.""" key = (model or "").strip().lower() return _MODEL_ALIAS_CANONICAL.get(key, key) diff --git a/hermes_cli/model_selection_guards.py b/hermes_cli/model_selection_guards.py index 0d886497cb..8ce700f096 100644 --- a/hermes_cli/model_selection_guards.py +++ b/hermes_cli/model_selection_guards.py @@ -1,21 +1,7 @@ """Unified selection-time guard registry for model switching surfaces. -Hermes has multiple model-selection surfaces (CLI picker, TUI, dashboard, -gateway ``/model``, Telegram/Discord pickers, TUI-gateway RPC). Each of them -previously imported ``model_cost_guard.expensive_model_warning`` directly, so -every new guard class (e.g. the data-training-tier guard) had to be wired into -every surface by hand — and inevitably missed some. - -This module is the single evaluation point: ``selection_warnings()`` runs every -registered guard and returns the warnings that fired. Surfaces render the -result with their own confirm UX (stdin prompt, modal, inline keyboard, -``confirm_required`` JSON) — that half stays per-surface; the *evaluation* half -lives here. Adding a guard to ``_GUARDS`` makes it appear on every surface at -once. - -Guard modules (``model_cost_guard``, ``model_data_policy_guard``) keep their -public APIs — existing tests and mock patch points remain valid; this module -only aggregates them. +Guard modules (``model_cost_guard``, ``model_data_policy_guard``) keep their public APIs — existing +tests and mock patch points remain valid; this module only aggregates them. """ from __future__ import annotations @@ -37,6 +23,23 @@ class SelectionWarning: message: str +def _wrap(kind: str, title: str, warning, model_name: str, provider: Optional[str]): + """Lift a raw guard payload into a :class:`SelectionWarning` (None passes through). + + Duck-typed access: tests (and future guard payloads) may supply objects carrying only + ``.message``. + """ + if warning is None: + return None + return SelectionWarning( + kind=kind, + title=title, + model=getattr(warning, "model", model_name), + provider=getattr(warning, "provider", provider or ""), + message=warning.message, + ) + + def _cost_guard( model_name: str, provider: Optional[str], @@ -47,23 +50,9 @@ def _cost_guard( from hermes_cli.model_cost_guard import expensive_model_warning warning = expensive_model_warning( - model_name, - provider=provider, - base_url=base_url, - api_key=api_key, - model_info=model_info, - ) - if warning is None: - return None - # Duck-typed access: tests (and future guard payloads) may supply objects - # carrying only ``.message``. - return SelectionWarning( - kind="cost", - title="Expensive Model Warning", - model=getattr(warning, "model", model_name), - provider=getattr(warning, "provider", provider or ""), - message=warning.message, + model_name, provider=provider, base_url=base_url, api_key=api_key, model_info=model_info ) + return _wrap("cost", "Expensive Model Warning", warning, model_name, provider) def _data_policy_guard( @@ -75,28 +64,13 @@ def _data_policy_guard( ) -> Optional[SelectionWarning]: from hermes_cli.model_data_policy_guard import data_training_warning - warning = data_training_warning( - model_name, - provider=provider, - base_url=base_url, - ) - if warning is None: - return None - return SelectionWarning( - kind="data_policy", - title="Data-Training Tier Warning", - model=getattr(warning, "model", model_name), - provider=getattr(warning, "provider", provider or ""), - message=warning.message, - ) + warning = data_training_warning(model_name, provider=provider, base_url=base_url) + return _wrap("data_policy", "Data-Training Tier Warning", warning, model_name, provider) # Registry, evaluated in order. Add new guard classes here — never at the # individual surfaces. -_GUARDS = ( - _cost_guard, - _data_policy_guard, -) +_GUARDS = (_cost_guard, _data_policy_guard) def selection_warnings( @@ -110,15 +84,11 @@ def selection_warnings( ) -> List[SelectionWarning]: """Run every registered selection guard and return the warnings that fired. - Returns an empty list in the common case (no guard fired). Callers should - run this after model resolution so aliases / provider-specific ids have - settled, then surface the messages as a confirm step. ``include_kinds`` - optionally restricts which guard kinds run (e.g. auth.py's picker only runs - the cost guard when a provider is known, but always runs the data-policy - guard). + Returns an empty list in the common case (no guard fired). Callers should run this after model + resolution so aliases / provider-specific ids have settled, then surface the messages as a + confirm step. ``include_kinds`` optionally restricts which guard kinds run (e.g. - A misbehaving guard must never break model selection: individual guard - exceptions are swallowed. + A misbehaving guard must never break model selection: individual guard exceptions are swallowed. """ wanted = set(include_kinds) if include_kinds is not None else None results: List[SelectionWarning] = [] @@ -127,20 +97,16 @@ def selection_warnings( warning = guard(model_name, provider, base_url, api_key, model_info) except Exception: continue - if warning is None: - continue - if wanted is not None and warning.kind not in wanted: - continue - results.append(warning) + if warning is not None and (wanted is None or warning.kind in wanted): + results.append(warning) return results def combined_message(warnings: List[SelectionWarning]) -> str: """Join multiple warnings into one confirm-prompt body. - Surfaces that show a single confirm dialog use this when more than one - guard fires (rare) — one prompt showing both blocks beats two sequential - prompts. + Used by surfaces with a single confirm dialog when more than one guard fires (rare) — one + prompt showing both blocks beats two sequential prompts. """ return "\n\n".join(w.message for w in warnings) @@ -155,18 +121,12 @@ def combined_selection_warning( ) -> Optional[SelectionWarning]: """Drop-in replacement for ``expensive_model_warning`` call sites. - Returns ``None`` when no guard fired, a single :class:`SelectionWarning` - when exactly one fired, or a merged warning (``kind="multiple"``) whose - ``message`` stacks every fired guard. Surfaces that render one confirm - dialog with ``warning.message`` can switch to this without reshaping their - control flow. + Returns ``None`` when no guard fired, the single :class:`SelectionWarning` when one fired, + or a merged ``kind="multiple"`` warning stacking every message — so surfaces rendering one + confirm dialog from ``warning.message`` can switch without reshaping control flow. """ warnings = selection_warnings( - model_name, - provider=provider, - base_url=base_url, - api_key=api_key, - model_info=model_info, + model_name, provider=provider, base_url=base_url, api_key=api_key, model_info=model_info ) if not warnings: return None diff --git a/hermes_cli/models.py b/hermes_cli/models.py index fb8371983c..99d7b023fb 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -1,9 +1,4 @@ -""" -Canonical model catalogs and lightweight validation helpers. - -Add, remove, or reorder entries here — both `hermes setup` and -`hermes` provider-selection will pick up the change automatically. -""" +"""Canonical model catalogs and lightweight validation helpers.""" from __future__ import annotations @@ -49,11 +44,10 @@ def _urlopen_model_catalog_request(req: urllib.request.Request, *, timeout: floa def _custom_provider_ssl_context(base_url: str): """Build an ``ssl.SSLContext`` from a custom provider's TLS settings. - Mirrors the httpx/requests TLS resolution so the urllib ``/models`` - discovery probe honors a provider's ``ssl_ca_cert`` / ``ssl_verify`` - instead of falling back to the process-wide ``SSL_CERT_FILE`` / certifi - bundle. Returns None when no per-provider TLS override applies, so the - caller keeps urllib's default policy for public/unconfigured endpoints. + Mirrors the httpx/requests TLS resolution so the urllib ``/models`` probe honors a + provider's ``ssl_ca_cert`` / ``ssl_verify`` instead of the process-wide + ``SSL_CERT_FILE``/certifi bundle. Returns None when no per-provider override applies, so the + caller keeps urllib's default policy. """ if not base_url: return None @@ -180,9 +174,9 @@ _ai_gateway_catalog_cache: list[tuple[str, str]] | None = None def _codex_curated_models() -> list[str]: """Derive the openai-codex curated list from codex_models.py. - Single source of truth: DEFAULT_CODEX_MODELS + forward-compat synthesis. - This keeps the gateway /model picker in sync with the CLI `hermes model` - flow without maintaining a separate static list. + Single source of truth: DEFAULT_CODEX_MODELS + forward-compat synthesis. This keeps the gateway + /model picker in sync with the CLI `hermes model` flow without maintaining a separate static + list. """ from hermes_cli.codex_models import DEFAULT_CODEX_MODELS, _finalize_codex_models return _finalize_codex_models(list(DEFAULT_CODEX_MODELS)) @@ -245,8 +239,8 @@ def _xai_finalize_catalog(ids: list[str]) -> list[str]: def _xai_curated_models() -> list[str]: """Offline curated floor for xAI / xAI OAuth pickers. - Reads $HERMES_HOME/models_dev_cache.json directly (no network). Falls - back to ``_XAI_STATIC_FALLBACK`` when the cache is empty or unreadable. + Reads $HERMES_HOME/models_dev_cache.json directly (no network). Falls back to + ``_XAI_STATIC_FALLBACK`` when the cache is empty or unreadable. """ try: from agent.models_dev import _load_disk_cache @@ -835,37 +829,6 @@ def _is_model_free(model_id: str, pricing: dict[str, dict[str, str]]) -> bool: return False -# --------------------------------------------------------------------------- -# Nous Portal account tier detection -# --------------------------------------------------------------------------- -def is_nous_free_tier(account_info: dict[str, Any]) -> bool: - """Return True if the account info indicates a free (unpaid) tier. - - Prefer the Portal's explicit ``paid_service_access.allowed`` entitlement - decision. Legacy payloads fall back to ``subscription.monthly_charge == 0``. - Returns False when both signals are missing or unparseable. - """ - paid_access = account_info.get("paid_service_access") - if isinstance(paid_access, dict): - allowed = paid_access.get("allowed") - if isinstance(allowed, bool): - return not allowed - paid = paid_access.get("paid_access") - if isinstance(paid, bool): - return not paid - - sub = account_info.get("subscription") - if not isinstance(sub, dict): - return False - charge = sub.get("monthly_charge") - if charge is None: - return False - try: - return float(charge) == 0 - except (TypeError, ValueError): - return False - - def partition_nous_models_by_tier( model_ids: list[str], pricing: dict[str, dict[str, str]], @@ -873,10 +836,8 @@ def partition_nous_models_by_tier( ) -> tuple[list[str], list[str]]: """Split Nous models into (selectable, unavailable) based on user tier. - For paid-tier users: all models are selectable, none unavailable. - - For free-tier users: only free models are selectable; paid models - are returned as unavailable (shown grayed out in the menu). + For free-tier users: only free models are selectable; paid models are returned as unavailable + (shown grayed out in the menu). """ if not free_tier: return (model_ids, []) @@ -903,26 +864,8 @@ def union_with_portal_free_recommendations( ) -> tuple[list[str], dict[str, dict[str, str]]]: """Augment curated list + pricing with the Portal's ``freeRecommendedModels``. - The Portal's ``/api/nous/recommended-models`` endpoint advertises which - models are free *right now* — independent of what the in-repo - ``_PROVIDER_MODELS["nous"]`` list happens to contain or whether the - docs-hosted catalog manifest has been rebuilt since the last release. - - For free-tier users this is the source of truth: any model the Portal - flags as free should be selectable, even if the user is running an - older Hermes that doesn't ship that model in its hardcoded curated - list. This function returns an augmented ``(model_ids, pricing)`` - pair where: - - * Portal free recommendations missing from ``curated_ids`` are - appended after the curated list (so the in-repo curated models - show first and Portal-only picks follow). - * ``pricing`` gets a synthetic ``{"prompt": "0", "completion": "0"}`` - entry for any free recommendation missing from the live pricing - map, so :func:`partition_nous_models_by_tier` keeps it. - - Failures (network, parse, missing field) are silent and degrade to - returning the inputs unchanged. + * Portal free recommendations missing from ``curated_ids`` are appended after the curated list + (so the in-repo curated models show first and Portal-only picks follow). """ try: payload = fetch_nous_recommended_models( @@ -969,32 +912,12 @@ def union_with_portal_paid_recommendations( ) -> tuple[list[str], dict[str, dict[str, str]]]: """Augment curated list with the Portal's ``paidRecommendedModels``. - Mirror of :func:`union_with_portal_free_recommendations` for paid-tier - users. The Portal's ``/api/nous/recommended-models`` endpoint advertises - which paid models are blessed *right now* — independent of what the - in-repo ``_PROVIDER_MODELS["nous"]`` list happens to contain or whether - the docs-hosted catalog manifest has been rebuilt since the last release. + * Portal paid recommendations missing from ``curated_ids`` are appended after the curated list + (so the in-repo curated models show first and Portal-only picks follow). * ``pricing`` is left + untouched — we deliberately do NOT synthesize pricing entries for paid models. - For paid-tier users this lets newly-launched paid models surface in the - picker even if the user is running an older Hermes that doesn't ship - them in its hardcoded curated list. This function returns an augmented - ``(model_ids, pricing)`` pair where: - - * Portal paid recommendations missing from ``curated_ids`` are - appended after the curated list (so the in-repo curated models - show first and Portal-only picks follow). - * ``pricing`` is left untouched — we deliberately do NOT synthesize - pricing entries for paid models. Live pricing is fetched separately - via :func:`get_pricing_for_provider`; if the live endpoint hasn't - published pricing yet, the picker shows a blank price column rather - than fabricating numbers. (The free helper synthesizes ``$0`` so - :func:`partition_nous_models_by_tier` keeps free models selectable; - no equivalent gating applies on the paid side, so synthesis would - only mislead the user.) - - Failures (network, parse, missing field) are silent and degrade to - returning the inputs unchanged — never block the picker on a - Portal-side hiccup. + Failures (network, parse, missing field) are silent and degrade to returning the inputs + unchanged — never block the picker on a Portal-side hiccup. """ try: payload = fetch_nous_recommended_models( @@ -1037,12 +960,11 @@ _free_tier_cache: tuple[bool, float] | None = None # (result, timestamp) def check_nous_free_tier(*, force_fresh: bool = False) -> bool: """Check if the current Nous Portal user is on a free (unpaid) tier. - Results are cached for ``_FREE_TIER_CACHE_TTL`` seconds to avoid - hitting the Portal API on every call. The cache is short-lived so - that an account upgrade is reflected within a few minutes. + Results are cached for ``_FREE_TIER_CACHE_TTL`` seconds to avoid hitting the Portal API on every + call. The cache is short-lived so that an account upgrade is reflected within a few minutes. - Returns True only when entitlement is known to be free. Unknown/error - states return False so this compatibility wrapper does not block users. + Returns True only when entitlement is known to be free. Unknown/error states return False so + this compatibility wrapper does not block users. """ global _free_tier_cache now = time.monotonic() @@ -1098,8 +1020,7 @@ def _nous_recommended_disk_path() -> "Path": def _read_nous_recommended_disk(base: str) -> dict[str, Any] | None: """Return the last-known-good payload for ``base`` from disk, or None. - The disk file is a JSON object keyed by portal base URL so staging and - prod don't collide: + The disk file is a JSON object keyed by portal base URL so staging and prod don't collide: ``{"": {"data": {...}, "ts": }}``. """ try: @@ -1119,8 +1040,8 @@ def _read_nous_recommended_disk(base: str) -> dict[str, Any] | None: def _write_nous_recommended_disk(base: str, data: dict[str, Any]) -> None: """Persist ``data`` as the last-known-good payload for ``base``. - Merges into any existing per-base map, then writes atomically. Failures - are non-fatal (logged at debug) — the in-process cache still works. + Merges into any existing per-base map, then writes atomically. Failures are non-fatal (logged at + debug) — the in-process cache still works. """ if not data: return @@ -1155,21 +1076,13 @@ def fetch_nous_recommended_models( ) -> dict[str, Any]: """Fetch the Nous Portal's curated recommended-models payload. - Hits ``/api/nous/recommended-models``. The endpoint is public — - no auth is required. Results are cached per portal URL for - ``_NOUS_RECOMMENDED_CACHE_TTL`` seconds in process; pass + Hits ``/api/nous/recommended-models``. The endpoint is public — no auth is required. + Results are cached per portal URL for ``_NOUS_RECOMMENDED_CACHE_TTL`` seconds in process; pass ``force_refresh=True`` to bypass the in-process cache. A successful live fetch is also persisted to a per-base disk cache - (``$HERMES_HOME/cache/nous_recommended_cache.json``) as last-known-good. - When the live fetch fails (network, parse, non-2xx) and the in-process - cache is empty, the disk copy is returned instead of ``{}`` — so a - transient Portal hiccup no longer silently drops the free/paid model - recommendations from the picker. Self-heals on the next successful fetch. - - Returns the parsed JSON dict, or ``{}`` only when neither the network nor - any cache layer can supply data. Callers must treat missing/null fields - as "no recommendation" and fall back to their own default. + (``$HERMES_HOME/cache/nous_recommended_cache.json``) as last-known-good. Self-heals on the next + successful fetch. """ base = (portal_base_url or "https://portal.nousresearch.com").rstrip("/") now = time.monotonic() @@ -1244,24 +1157,9 @@ def get_nous_recommended_aux_model( ) -> Optional[str]: """Return the Portal's recommended model name for an auxiliary task. - Picks the best field from the Portal's recommended-models payload: - - * ``vision=True`` → ``paidRecommendedVisionModel`` (paid tier) or - ``freeRecommendedVisionModel`` (free tier) - * ``vision=False`` → ``paidRecommendedCompactionModel`` or - ``freeRecommendedCompactionModel`` - - When ``free_tier`` is ``None`` (default) the user's tier is auto-detected - via :func:`check_nous_free_tier`. Pass an explicit bool to bypass the - detection — useful for tests or when the caller already knows the tier. - - For paid-tier users we prefer the paid recommendation but gracefully fall - back to the free recommendation if the Portal returned ``null`` for the - paid field (common during the staged rollout of new paid models). - - Returns ``None`` when every candidate is missing, null, or the fetch - fails — callers should fall back to their own default (currently - ``google/gemini-3-flash-preview``). + For paid-tier users we prefer the paid recommendation but gracefully fall back to the free + recommendation if the Portal returned ``null`` for the paid field (common during the staged + rollout of new paid models). """ base = portal_base_url or _resolve_nous_portal_url() payload = fetch_nous_recommended_models(base, force_refresh=force_refresh) @@ -1422,28 +1320,13 @@ def provider_group_for_slug(slug: str) -> str: def group_providers(slugs): """Fold a flat ordered slug iterable into picker rows by provider group. - DISPLAY ONLY. Used by every interactive picker (``hermes model``, the - setup wizard, the Telegram ``/model`` keyboard) so grouping is identical - across surfaces. + DISPLAY ONLY. Used by every interactive picker (``hermes model``, the setup wizard, the Telegram + ``/model`` keyboard) so grouping is identical across surfaces. - Each returned row is a dict:: - - {"kind": "single", "slug": } # ungrouped, or - # 1-member group - {"kind": "group", "group_id": , "label":