From 48751f0916b041b0496e11605bcf7f398d7be5e1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:11:59 -0700 Subject: [PATCH 1/4] refactor(web-profiles/status): shared home scope, DB-read helper, get_status phase helpers, dedupe fallbacks --- hermes_cli/web_routers/profiles.py | 619 +++++++++---------- hermes_cli/web_routers/status.py | 951 ++++++++++++----------------- hermes_cli/web_server_profiles.py | 503 ++++++--------- 3 files changed, 855 insertions(+), 1218 deletions(-) diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 19f28d8d25..402666c7d8 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -1,14 +1,12 @@ """Profiles dashboard routes. -Two routers because route order matters: ``sessions_router`` -(/api/profiles/sessions*, projects/tree, pull-requests) was registered long -before the generic ``/api/profiles/{name}`` routes on ``router``; the original -global registration order is preserved rather than relying on Starlette's -literal-before-param matching. +Two routers because route order matters: ``sessions_router`` (/api/profiles/sessions*, +projects/tree, pull-requests) was registered long before the generic +``/api/profiles/{name}`` routes on ``router``; the original global registration order is +preserved rather than relying on Starlette's literal-before-param matching. -web_server-owned helpers are reached via the late-binding seam in -:mod:`hermes_cli.web_deps` so tests that ``monkeypatch.setattr(web_server, -"_helper", ...)`` keep working. +web_server-owned helpers are reached via the late-binding seam in :mod:`hermes_cli.web_deps` +so tests that ``monkeypatch.setattr(web_server, "_helper", ...)`` keep working. """ import contextlib @@ -24,32 +22,24 @@ import threading import time from collections import OrderedDict from pathlib import Path -from typing import Any, Dict, List, Optional, Tuple +from typing import Any, Callable, Dict, List, Optional, Tuple from fastapi import APIRouter, HTTPException, Query from hermes_cli.web_deps import late from hermes_cli.web_models import ( - ProfileCreate, - ProfileActiveUpdate, - ProfileExport, - ProfileImport, - ProfileRename, - ProfileSoulUpdate, - ProfileDescriptionUpdate, - ProfileModelUpdate, - ProfileDescribeAuto, - SessionPrScanBody, -) + ProfileCreate, ProfileActiveUpdate, ProfileExport, ProfileImport, ProfileRename, + ProfileSoulUpdate, ProfileDescriptionUpdate, ProfileModelUpdate, ProfileDescribeAuto, + SessionPrScanBody) +from hermes_cli.web_server_profiles import _hermes_home_scope # Same logger the handlers used before extraction (identical logger object). _log = logging.getLogger("hermes_cli.web_server") -# Per-profile session reads report failures in the response's ``errors`` -# array, which the desktop sidebar does not surface — an empty sidebar can -# look healthy while nothing logs. Warn once per (profile, message) per -# process so a persistent failure is loud in errors.log without turning every -# sidebar poll into log spam. +# Per-profile session reads report failures in the response's ``errors`` array, which the +# desktop sidebar does not surface — an empty sidebar can look healthy while nothing logs. +# Warn once per (profile, message) per process so a persistent failure is loud in +# errors.log without turning every sidebar poll into log spam. _profile_read_warned: set = set() @@ -64,8 +54,7 @@ def _warn_profile_read_error(profile: str, exc: Exception) -> None: sessions_router = APIRouter() router = APIRouter() -# Late-bound web_server helpers (resolved at call time; cycle-safe, -# monkeypatch-transparent). +# Late-bound web_server helpers (resolved at call time; cycle-safe, monkeypatch-transparent). _cron_profile_home = late("_cron_profile_home") _fallback_profile_dicts = late("_fallback_profile_dicts") _hub_action_name = late("_hub_action_name") @@ -84,30 +73,24 @@ _normalize_main_model_assignment = late("_normalize_main_model_assignment") # --------------------------------------------------------------------------- -def _profile_attr(info, name: str, default: Any = None) -> Any: - try: - return getattr(info, name) - except Exception: - return default - - def _profile_to_dict(info) -> Dict[str, Any]: + attr = functools.partial(getattr, info) return { - "name": _profile_attr(info, "name", ""), - "path": str(_profile_attr(info, "path", "")), - "is_default": bool(_profile_attr(info, "is_default", False)), - "model": _profile_attr(info, "model"), - "provider": _profile_attr(info, "provider"), - "has_env": bool(_profile_attr(info, "has_env", False)), - "skill_count": int(_profile_attr(info, "skill_count", 0) or 0), - "gateway_running": bool(_profile_attr(info, "gateway_running", False)), - "description": _profile_attr(info, "description", "") or "", - "description_auto": bool(_profile_attr(info, "description_auto", False)), - "display_name": _profile_attr(info, "display_name", "") or "", - "distribution_name": _profile_attr(info, "distribution_name"), - "distribution_version": _profile_attr(info, "distribution_version"), - "distribution_source": _profile_attr(info, "distribution_source"), - "has_alias": _profile_attr(info, "alias_path") is not None, + "name": attr("name", ""), + "path": str(attr("path", "")), + "is_default": bool(attr("is_default", False)), + "model": attr("model", None), + "provider": attr("provider", None), + "has_env": bool(attr("has_env", False)), + "skill_count": int(attr("skill_count", 0) or 0), + "gateway_running": bool(attr("gateway_running", False)), + "description": attr("description", "") or "", + "description_auto": bool(attr("description_auto", False)), + "display_name": attr("display_name", "") or "", + "distribution_name": attr("distribution_name", None), + "distribution_version": attr("distribution_version", None), + "distribution_source": attr("distribution_source", None), + "has_alias": attr("alias_path", None) is not None, } @@ -118,67 +101,47 @@ def _profile_setup_command(name: str) -> str: def _write_profile_model(profile_dir: Path, provider: str, model: str) -> None: - """Write the main model assignment into a specific profile's config.yaml. - - Scopes ``load_config``/``save_config`` to ``profile_dir`` via the - context-local HERMES_HOME override so the write lands in the target - profile's config rather than the dashboard process's active profile. - Clears any stale ``base_url`` / ``context_length`` the same way - ``POST /api/model/set`` does, since the new model may differ. - """ + """Write the main model assignment into ``profile_dir``'s config.yaml (HERMES_HOME-scoped, + so it lands in the target profile rather than the dashboard's active one). Clears stale + ``base_url`` / ``context_length`` the same way ``POST /api/model/set`` does.""" from hermes_cli.web_server import load_config, save_config - from hermes_constants import set_hermes_home_override, reset_hermes_home_override - token = set_hermes_home_override(str(profile_dir)) - try: + with _hermes_home_scope(profile_dir): provider, model = _normalize_main_model_assignment(provider, model) cfg = load_config() cfg["model"] = _apply_main_model_assignment(cfg.get("model", {}), provider, model) save_config(cfg) - finally: - reset_hermes_home_override(token) def _disable_unselected_skills(profile_dir: Path, keep: List[str]) -> int: - """Disable every installed skill in ``profile_dir`` not in ``keep``. + """Disable every installed skill in ``profile_dir`` not in ``keep``; returns how many + were newly disabled. - Profiles manage skill activation via a *disabled* list — all installed - skills are active by default and users opt out. The builder's skill step - uses "replace" semantics: the user picks exactly which seeded built-in / - optional skills stay active, and everything else gets added to the disabled - list. (Hub skills are installed separately via subprocess and are active on - install.) Scoped to the profile via the HERMES_HOME override. Returns the - number of skills newly disabled. + Profiles manage activation via a *disabled* list (everything installed is active by + default). The builder's skill step has "replace" semantics: the user picks exactly which + seeded skills stay active. Hub skills are installed separately via subprocess and are + active on install. """ from hermes_cli.web_server import load_config - from hermes_constants import set_hermes_home_override, reset_hermes_home_override from hermes_cli.skills_config import get_disabled_skills, save_disabled_skills keep_set = {s.strip() for s in keep if s and s.strip()} - disabled_count = 0 - token = set_hermes_home_override(str(profile_dir)) - try: - installed: List[str] = [] + with _hermes_home_scope(profile_dir): skills_root = profile_dir / "skills" - if skills_root.is_dir(): - for md in skills_root.rglob("SKILL.md"): - installed.append(md.parent.name) + installed = [md.parent.name for md in skills_root.rglob("SKILL.md")] if skills_root.is_dir() else [] cfg = load_config() disabled = get_disabled_skills(cfg) + newly = 0 for name in installed: if name not in keep_set and name not in disabled: disabled.add(name) - disabled_count += 1 - if disabled_count: + newly += 1 + if newly: save_disabled_skills(cfg, disabled) - finally: - reset_hermes_home_override(token) - return disabled_count + return newly -# Returned by the offloaded file readers below to mean "the file is not there", -# which a plain ``None`` cannot express: ``desktop.json`` may legitimately hold -# the document ``null``, and that is an existing-but-empty overlay rather than -# an absent one. +# Returned by the offloaded file readers below to mean "the file is not there", which a plain +# ``None`` cannot express: ``desktop.json`` may legitimately hold the document ``null``. _MISSING = object() @@ -283,9 +246,27 @@ def _open_profile_db(name: str, home, errors: Optional[List[Dict[str, str]]]): return None -# Bounded cache lifetime for the expensive sidebar scan. Short enough that the -# UI never shows meaningfully stale data, long enough to coalesce the desktop's -# reconnect/focus/change poll bursts into one scan. +def _read_profile_db(name: str, home, errors: Optional[List[Dict[str, str]]], + fn: Callable[[Any], Any]) -> Any: + """``fn(db)`` against the profile's read-only DB, or None when it can't be opened or + ``fn`` raises (warned once, recorded in ``errors`` when given). Always closes the DB.""" + db = _open_profile_db(name, home, errors) + if db is None: + return None + try: + return fn(db) + except Exception as exc: + _warn_profile_read_error(name, exc) + if errors is not None: + errors.append({"profile": name, "error": str(exc)}) + return None + finally: + db.close() + + +# Bounded cache lifetime for the expensive sidebar scan. Short enough that the UI never shows +# meaningfully stale data, long enough to coalesce the desktop's reconnect/focus/change poll +# bursts into one scan. _SIDEBAR_CACHE_TTL_SECONDS = 5.0 _SIDEBAR_CACHE_MAX_ENTRIES = 32 _SIDEBAR_PROFILE_CACHE_MAX_ENTRIES = 256 @@ -320,8 +301,8 @@ def _sidebar_profile_cache_put(key, value): db_path, fingerprint = key[:2] snapshot = copy.deepcopy(value) with _SIDEBAR_PROFILE_CACHE_LOCK: - # A changed DB/WAL makes all older parameter variants for that profile - # obsolete. Remove them eagerly rather than waiting for LRU pressure. + # A changed DB/WAL obsoletes every older parameter variant for that profile; drop + # them eagerly rather than waiting for LRU pressure. for existing in [k for k in _SIDEBAR_PROFILE_CACHE if k[0] == db_path and k[1] != fingerprint]: _SIDEBAR_PROFILE_CACHE.pop(existing, None) @@ -339,12 +320,11 @@ def _sidebar_profile_cache_clear(): def _sidebar_singleflight_cache(func): """Coalesce concurrent sidebar scans and briefly reuse their response. - Every uncached refresh opens every profile database and runs several - session queries per profile. Desktop reconnect/focus/change bursts overlap - identical scans in AnyIO worker threads, amplifying YAML/SQLite work and - starving the uvicorn event loop for the GIL. The short TTL bounds UI - staleness; the single-flight lock guarantees one expensive scan at a time. - Cached values are copied on store and hit so FastAPI serialization or a + Every uncached refresh opens every profile database and runs several session queries + per profile. Desktop reconnect/focus/change bursts overlap identical scans in AnyIO + worker threads, amplifying YAML/SQLite work and starving the uvicorn loop for the GIL. + The short TTL bounds UI staleness; the single-flight lock guarantees one expensive scan + at a time. Cached values are copied on store and hit so FastAPI serialization or a caller cannot mutate shared state. """ signature = inspect.signature(func) @@ -382,16 +362,16 @@ def _sidebar_singleflight_cache(func): if cached is not miss: return cached - # A plain Lock is intentional: FastAPI executes this sync handler in - # the AnyIO worker pool, so contenders sleep without holding the GIL. + # A plain Lock is intentional: FastAPI executes this sync handler in the AnyIO + # worker pool, so contenders sleep without holding the GIL. with refresh_lock: cached = _lookup(key) if cached is not miss: return cached result = func(*args, **kwargs) - # A 200 carrying errors[] is a FAILED profile scan, not a successful - # empty page. Caching it would hold the empty recents in front of a - # store that has already recovered, for the whole TTL. + # A 200 carrying errors[] is a FAILED profile scan, not a successful empty page. + # Caching it would hold the empty recents in front of a store that has already + # recovered, for the whole TTL. if isinstance(result, dict) and result.get("errors"): return result try: @@ -414,12 +394,16 @@ def _sidebar_singleflight_cache(func): return wrapped +def _csv_list(value: Optional[str]) -> List[str]: + return [s.strip() for s in (value or "").split(",") if s.strip()] + + @sessions_router.get("/api/profiles/sessions") def get_profiles_sessions( - # ``le=500`` caps the per-request page size — this endpoint fans out across - # EVERY profile's state.db, so an unbounded limit multiplies the damage. - # 500 (not 100) because real desktop callers use limit=200 and the electron - # remote-merge over-fetches ``limit + offset``. + # ``le=500`` caps the per-request page size — this endpoint fans out across EVERY + # profile's state.db, so an unbounded limit multiplies the damage. 500 (not 100) because + # real desktop callers use limit=200 and the electron remote-merge over-fetches + # ``limit + offset``. limit: int = Query(20, ge=0, le=500), offset: int = Query(0, ge=0), min_messages: int = 0, @@ -433,11 +417,10 @@ def get_profiles_sessions( ): """Unified, read-only session list aggregated across ALL profiles. - Process-light: opens each profile's ``state.db`` directly from disk — it - does NOT spawn a dashboard backend per profile. Each row is tagged with its - owning ``profile`` so the desktop renders one list and only spins up a - backend when the user interacts. Rows omit ``system_prompt`` / - ``model_config`` unless ``full=1`` — same projection as ``/api/sessions``. + Process-light: opens each profile's ``state.db`` directly from disk — it does NOT spawn + a dashboard backend per profile. Each row is tagged with its owning ``profile`` so the + desktop renders one list and only spins up a backend when the user interacts. Rows omit + ``system_prompt`` / ``model_config`` unless ``full=1`` — same projection as ``/api/sessions``. """ if archived not in ("exclude", "only", "include"): raise HTTPException(status_code=400, detail="archived must be one of: exclude, only, include") @@ -449,52 +432,42 @@ def get_profiles_sessions( else: targets = _profile_targets("GET /api/profiles/sessions", lightweight=True) - # Source scoping (see /api/sessions): recents pass exclude_sources=cron, - # the cron-jobs section passes source=cron — two independent lists so - # newest cron sessions can't starve the recents page. + # Source scoping (see /api/sessions): recents pass exclude_sources=cron, the cron-jobs + # section passes source=cron — two independent lists so newest cron sessions can't + # starve the recents page. filters = dict( source=source or None, - sources=[s.strip() for s in (sources or "").split(",") if s.strip()] or None, - exclude_sources=[s.strip() for s in (exclude_sources or "").split(",") if s.strip()] or None, + sources=_csv_list(sources) or None, + exclude_sources=_csv_list(exclude_sources) or None, min_message_count=max(0, min_messages), include_archived=archived == "include", archived_only=archived == "only", ) - # Over-fetch per profile so the merged+sorted window is correct for the - # requested page. Capped so a huge profile can't blow up the response. + # Over-fetch per profile so the merged+sorted window is correct for the requested page. + # Capped so a huge profile can't blow up the response. per_profile = min(max(limit + offset, limit), 500) merged: List[Dict[str, Any]] = [] - total = 0 - profile_totals: Dict[str, int] = {} + totals: Dict[str, int] = {} errors: List[Dict[str, str]] = [] now = time.time() for name, home in targets: - db = _open_profile_db(name, home, errors) - if db is None: - continue - try: + def _read(db, name=name): rows = db.list_sessions_rich( limit=per_profile, offset=0, order_by_last_active=order == "recent", # Same SQL-level blob skip as /api/sessions. compact_rows=not full, include_pinned=True, **filters, ) - profile_total = db.session_count(exclude_children=True, **filters) - total += profile_total - profile_totals[name] = profile_total + totals[name] = db.session_count(exclude_children=True, **filters) merged.extend(_tag_rows(rows, name, now)) - except Exception as exc: - _warn_profile_read_error(name, exc) - errors.append({"profile": name, "error": str(exc)}) - finally: - db.close() + _read_profile_db(name, home, errors, _read) sort_key = "last_active" if order == "recent" else "started_at" merged.sort(key=lambda s: s.get(sort_key) or s.get("started_at") or 0, reverse=True) window = _pinned_window(merged, offset, limit) if not full: _strip_session_list_rows(window) - return {"sessions": window, "total": total, "profile_totals": profile_totals, + return {"sessions": window, "total": sum(totals.values()), "profile_totals": totals, "limit": limit, "offset": offset, "errors": errors} @@ -510,19 +483,16 @@ def get_profiles_sessions_sidebar( ): """Batched sidebar session slices — one profile-DB open per refresh. - The desktop sidebar needs three source-scoped windows per refresh: recents - (local chats), cron sessions, and messaging-platform sessions. Served as - three ``/api/profiles/sessions`` calls they reopened every profile's DB - three times; this opens each once and runs the three queries together. - Same row projection and 300s active heuristic as the per-slice endpoint. + The desktop sidebar needs three source-scoped windows per refresh: recents (local + chats), cron sessions, and messaging-platform sessions. Served as three + ``/api/profiles/sessions`` calls they reopened every profile's DB three times; this + opens each once. Same row projection and 300s active heuristic as the per-slice endpoint. - ``recents_profile`` scopes the WHOLE payload, not just recents — the - sidebar has one scope, so a concrete profile must never show another - profile's Telegram threads or cronjobs; ``all`` asks for everything. - - The caller passes the source taxonomy (``recents_exclude`` / - ``messaging_exclude`` CSV, ``source=cron`` is implicit). All slices use - ``min_messages=1`` / ``archived=exclude`` / recency order. + ``recents_profile`` scopes the WHOLE payload, not just recents — the sidebar has one + scope, so a concrete profile must never show another profile's Telegram threads or + cronjobs; ``all`` asks for everything. The caller passes the source taxonomy + (``recents_exclude`` / ``messaging_exclude`` CSV, ``source=cron`` is implicit). All + slices use ``min_messages=1`` / ``archived=exclude`` / recency order. """ targets = _profile_targets("GET /api/profiles/sessions/sidebar", lightweight=True) @@ -543,14 +513,26 @@ def get_profiles_sessions_sidebar( now = time.time() def _slice(db, *, source=None, exclude=None, cap): - # include_pinned: a pinned conversation must reach the sidebar even - # when it has aged past the window, or its Pinned row renders empty. + # include_pinned: a pinned conversation must reach the sidebar even when it has aged + # past the window, or its Pinned row renders empty. return db.list_sessions_rich( source=source, exclude_sources=exclude or None, limit=cap, offset=0, min_message_count=1, include_archived=False, archived_only=False, order_by_last_active=True, compact_rows=True, include_pinned=True, ) + def _build_slices(db, cache_key): + slices = { + "recents": _slice(db, exclude=recents_exclude_list, cap=recents_cap), + # Aggregated in SQL rather than over the recents window: the window is a page, + # and a total that shrank when you scrolled would be worse than no total at all. + "usage": db.usage_totals(), + "cron": _slice(db, source="cron", cap=cron_cap), + "messaging": _slice(db, exclude=messaging_exclude_list, cap=messaging_cap), + } + _sidebar_profile_cache_put(cache_key, slices) + return slices + for name, home in targets: if recents_scope != "all" and name != recents_scope: continue @@ -562,31 +544,15 @@ def get_profiles_sessions_sidebar( tuple(messaging_exclude_list)) slices = _sidebar_profile_cache_get(profile_cache_key) if slices is None: - db = _open_profile_db(name, home, errors) - if db is None: + slices = _read_profile_db(name, home, errors, + lambda db: _build_slices(db, profile_cache_key)) + if slices is None: continue - try: - slices = { - "recents": _slice(db, exclude=recents_exclude_list, cap=recents_cap), - # Aggregated in SQL rather than over the recents window: the - # window is a page, and a total that shrank when you scrolled - # would be worse than no total at all. - "usage": db.usage_totals(), - "cron": _slice(db, source="cron", cap=cron_cap), - "messaging": _slice(db, exclude=messaging_exclude_list, cap=messaging_cap), - } - _sidebar_profile_cache_put(profile_cache_key, slices) - except Exception as exc: - _warn_profile_read_error(name, exc) - errors.append({"profile": name, "error": str(exc)}) - continue - finally: - db.close() profile_rows = slices["recents"] - # A full window means more rows remain on disk — all "load more" needs, - # at no cost beyond the rows already read. Discount pinned back-fills: - # they arrive past the LIMIT and would fake a full page on a short list. + # A full window means more rows remain on disk — all "load more" needs, at no cost + # beyond the rows already read. Discount pinned back-fills: they arrive past the + # LIMIT and would fake a full page on a short list. unpinned_count = sum(1 for s in profile_rows if not s.get("pinned")) recents_truncated[name] = unpinned_count >= recents_cap recents_rows.extend(_tag_rows(profile_rows, name, now)) @@ -612,9 +578,9 @@ def get_profiles_sessions_sidebar( def _merge_by_id(into: Dict[str, Dict[str, Any]], entries: List[Dict[str, Any]], child_key: str) -> None: """Fold ``entries`` into ``into`` by id, recursing through one child list. - Repos merge their lanes, lanes merge their sessions. Counts add up and the - newest activity wins; everything else is first-writer, since the entries - describe the same path either way. + Repos merge their lanes, lanes merge their sessions. Counts add up and the newest + activity wins; everything else is first-writer, since the entries describe the same + path either way. """ for entry in entries: existing = into.get(entry["id"]) @@ -639,12 +605,11 @@ def _merge_profile_tree( ) -> None: """Fold one profile's projects into the shared tree, keyed by folder. - The same checkout in two profiles is one group, as is ``__no_project__`` - (every profile has one, which would otherwise put a "Home" on screen per - profile). Keying on the path also folds a declared project (``p_``) - with the auto entry another profile grows for the same folder. Sessions - carry the owning profile — the row badge and profile filter read that; a - group header never claims a single owner. + The same checkout in two profiles is one group, as is ``__no_project__`` (every profile + has one, which would otherwise put a "Home" on screen per profile). Keying on the path + also folds a declared project (``p_``) with the auto entry another profile grows + for the same folder. Sessions carry the owning profile — the row badge and profile + filter read that; a group header never claims a single owner. """ for project in projects: lane_sessions = (s for r in project.get("repos") or [] @@ -660,8 +625,8 @@ def _merge_profile_tree( merged[key] = project continue - # A declared project carries the label, color and icon the user chose, - # so it wins the identity when it meets another profile's auto entry. + # A declared project carries the label, color and icon the user chose, so it wins + # the identity when it meets another profile's auto entry. if existing.get("isAuto") and not project.get("isAuto"): existing, project = project, existing merged[key] = existing @@ -681,19 +646,17 @@ def _merge_profile_tree( def get_profiles_projects_tree(preview_limit: int = 3, session_limit: int = 2000): """Project tree for every profile at once, for the all-profiles sidebar. - ``projects.tree`` over JSON-RPC answers for the backend's own profile only. - This runs the same authoritative builder once per profile against that - profile's ``state.db``, scoping its other inputs (projects.db, repo-scan - policy, HERMES_HOME junk filters) through the context-local home override. - Projects merge across profiles so a group stands for a checkout rather than - a checkout-and-owner; the profile shows up per row for the filter. + ``projects.tree`` over JSON-RPC answers for the backend's own profile only. This runs + the same authoritative builder once per profile against that profile's ``state.db``, + scoping its other inputs (projects.db, repo-scan policy, HERMES_HOME junk filters) + through the context-local home override. Projects merge across profiles so a group + stands for a checkout rather than a checkout-and-owner. - Discovery is off: a repo with zero sessions is the same repo in every - profile (the disk scan would multiply empty lanes by the profile count), - and it is the one part of the builder that writes (policy reconciliation), - which this read-only fan-out must not do to a profile the user isn't driving. + Discovery is off: a repo with zero sessions is the same repo in every profile (the disk + scan would multiply empty lanes by the profile count), and it is the one part of the + builder that writes (policy reconciliation), which this read-only fan-out must not do + to a profile the user isn't driving. """ - from hermes_constants import reset_hermes_home_override, set_hermes_home_override from tui_gateway import server as gateway_server merged: Dict[str, Dict[str, Any]] = {} @@ -701,35 +664,26 @@ def get_profiles_projects_tree(preview_limit: int = 3, session_limit: int = 2000 errors: List[Dict[str, str]] = [] for name, home in _profile_targets("GET /api/profiles/projects/tree", lightweight=False): - db = _open_profile_db(name, home, errors) - if db is None: - continue - token = set_hermes_home_override(str(home)) - try: - tree, _active_id = gateway_server._build_project_tree( - db, preview_limit=preview_limit, hydrate=False, - session_limit=session_limit, include_discovered=False) - _merge_profile_tree(merged, tree["projects"], name, preview_limit) - scoped_session_ids.extend(tree["scoped_session_ids"]) - except Exception as exc: - _warn_profile_read_error(name, exc) - errors.append({"profile": name, "error": str(exc)}) - finally: - reset_hermes_home_override(token) - db.close() + def _read(db, name=name, home=home): + with _hermes_home_scope(home): + tree, _active_id = gateway_server._build_project_tree( + db, preview_limit=preview_limit, hydrate=False, + session_limit=session_limit, include_discovered=False) + _merge_profile_tree(merged, tree["projects"], name, preview_limit) + scoped_session_ids.extend(tree["scoped_session_ids"]) + _read_profile_db(name, home, errors, _read) return { "projects": sorted(merged.values(), key=lambda p: p.get("lastActive") or 0, reverse=True), - # Ownership is per profile, so no project is "the active one" here; the - # desktop only reads active_id to bias its overview sort. + # Ownership is per profile, so no project is "the active one" here; the desktop only + # reads active_id to bias its overview sort. "active_id": None, "scoped_session_ids": scoped_session_ids, "errors": errors, } -# `gh pr create` prints the PR url and nothing else, so a tool result whose -# whole output IS a PR url means this session opened that PR. Anything looser — -# a url inside prose, a `gh pr view` payload, an issue link — is a session -# TALKING about a PR, which is not the same claim. +# `gh pr create` prints the PR url and nothing else, so a tool result whose whole output IS +# a PR url means this session opened that PR. Anything looser — a url inside prose, a +# `gh pr view` payload, an issue link — is a session TALKING about a PR, not the same claim. _PR_URL_RE = re.compile(r"^https://github\.com/[\w.-]+/[\w.-]+/pull/(\d+)/?$") @@ -749,37 +703,31 @@ def _pr_url_from_tool_output(content: str) -> Optional[Tuple[int, str]]: def post_profiles_sessions_pull_requests(body: SessionPrScanBody): """The PR each of these sessions opened, recovered from its own transcript. - A session records the branch it started on, but one that starts in the main - checkout and works in a worktree has no branch of its own, so its PR is - invisible to that join. The evidence is in the conversation: ``gh pr - create`` ran and its output is a bare PR url (see - ``_pr_url_from_tool_output``). Read-only across every profile; the caller - asks once per session and remembers the answer. + A session records the branch it started on, but one that starts in the main checkout + and works in a worktree has no branch of its own, so its PR is invisible to that join. + The evidence is in the conversation: ``gh pr create`` ran and its output is a bare PR + url (see ``_pr_url_from_tool_output``). Read-only across every profile; the caller asks + once per session and remembers the answer. """ wanted = list(dict.fromkeys(s for s in (body.ids or []) if s))[:2000] if not wanted: return {"pull_requests": {}, "scanned": []} found: Dict[str, Dict[str, Any]] = {} - for name, home in _profile_targets("POST /api/profiles/sessions/pull-requests", lightweight=False): - db = _open_profile_db(name, home, None) - if db is None: - continue - try: - for pr in db.find_pr_url_messages(wanted): - parsed = _pr_url_from_tool_output(pr["content"]) - if parsed: - # Ordered oldest-first, so a later `gh pr create` in the same - # conversation wins — the replacement PR is the one the - # session ended on. - found[pr["session_id"]] = {"number": parsed[0], "url": parsed[1]} - except Exception as exc: - _warn_profile_read_error(name, exc) - finally: - db.close() - # Every id we looked at, so the caller can remember "asked, nothing there" - # and never scan this session again. + def _read(db): + for pr in db.find_pr_url_messages(wanted): + parsed = _pr_url_from_tool_output(pr["content"]) + if parsed: + # Ordered oldest-first, so a later `gh pr create` in the same conversation + # wins — the replacement PR is the one the session ended on. + found[pr["session_id"]] = {"number": parsed[0], "url": parsed[1]} + + for name, home in _profile_targets("POST /api/profiles/sessions/pull-requests", lightweight=False): + _read_profile_db(name, home, None, _read) + + # Every id we looked at, so the caller can remember "asked, nothing there" and never + # scan this session again. return {"pull_requests": found, "scanned": wanted} @@ -799,12 +747,12 @@ async def create_profile_endpoint(body: ProfileCreate): from hermes_cli import profiles as profiles_mod explicit_source = (body.clone_from or "").strip() if explicit_source: - # Duplicating a specific profile: clone its config/skills/SOUL (or full - # state when clone_all) from the named source rather than "default". + # Duplicating a specific profile: clone its config/skills/SOUL (or full state when + # clone_all) from the named source rather than "default". clone, clone_from, clone_config = True, explicit_source, not body.clone_all elif body.clone_all: - # Historical dashboard clone-all behavior: a full-copy request with no - # explicit dropdown source copies from default. + # Historical dashboard clone-all behavior: a full-copy request with no explicit + # dropdown source copies from default. clone, clone_from, clone_config = True, "default", False else: clone = body.clone_from_default @@ -815,20 +763,19 @@ async def create_profile_endpoint(body: ProfileCreate): path = profiles_mod.create_profile( name=body.name, clone_from=clone_from, clone_all=body.clone_all, clone_config=clone_config, no_skills=body.no_skills, description=body.description) - # Match the CLI's profile-create flow: fresh named profiles get the - # bundled skills. Cloning already copied the source's skills (incl. - # user-installed); no_skills wrote the opt-out marker so seeding no-ops. + # Match the CLI's profile-create flow: fresh named profiles get the bundled skills. + # Cloning already copied the source's skills (incl. user-installed); no_skills wrote + # the opt-out marker so seeding no-ops. if not clone: profiles_mod.seed_profile_skills(path, quiet=True) - # Match the CLI: named profiles get a ~/.local/bin wrapper when the - # alias is safe to create. + # Match the CLI: named profiles get a ~/.local/bin wrapper when the alias is safe. if not profiles_mod.check_alias_collision(body.name): profiles_mod.create_wrapper_script(body.name) - # Everything below is best-effort: the profile already exists, so a hiccup - # must not 500 the whole create — the user can fix it from the relevant - # dashboard page or ` setup` afterward. + # Everything below is best-effort: the profile already exists, so a hiccup must not 500 + # the whole create — the user can fix it from the relevant dashboard page or + # ` setup` afterward. provider = (body.provider or "").strip() model = (body.model or "").strip() model_set = bool(provider and model) and _best_effort( @@ -838,16 +785,16 @@ async def create_profile_endpoint(body: ProfileCreate): "Writing MCP servers for new profile %s failed", body.name, fn=lambda: _write_profile_mcp_servers(path, body.mcp_servers), default=0 ) if body.mcp_servers else 0 - # "keep" skill selection has replace semantics: disable every seeded skill - # not in the list. Skipped when empty (legacy: keep the bundle). + # "keep" skill selection has replace semantics: disable every seeded skill not in the + # list. Skipped when empty (legacy: keep the bundle). skills_disabled = _best_effort( "Applying skill selection for new profile %s failed", body.name, fn=lambda: _disable_unselected_skills(path, body.keep_skills), default=0 ) if body.keep_skills else 0 - # Skills-hub installs are spawned async, scoped to the new profile via - # `-p ` (a fresh subprocess re-binds skills_hub.SKILLS_DIR to the - # profile's HERMES_HOME at import). PIDs go back for the UI to poll. + # Skills-hub installs are spawned async, scoped to the new profile via `-p ` (a + # fresh subprocess re-binds skills_hub.SKILLS_DIR to the profile's HERMES_HOME at + # import). PIDs go back for the UI to poll. def _spawn_install(ident: str): return _spawn_hermes_action(["-p", body.name, "skills", "install", ident, "--yes"], _hub_action_name("install", ident)).pid @@ -866,38 +813,34 @@ async def create_profile_endpoint(body: ProfileCreate): @router.get("/api/profiles/active") async def get_active_profile_endpoint(): - """``active`` is the sticky default written by ``hermes profile use`` (what - new CLI invocations pick up); ``current`` is the profile this running - dashboard/gateway is scoped to (derived from HERMES_HOME).""" + """``active`` is the sticky default written by ``hermes profile use`` (what new CLI + invocations pick up); ``current`` is the profile this running dashboard/gateway is + scoped to (derived from HERMES_HOME).""" from hermes_cli import profiles as profiles_mod def _run(): - # Both reads touch the filesystem: get_active_profile() reads the - # active_profile state file and get_active_profile_name() resolves - # HERMES_HOME against the profiles root. Batched into one hop so the - # sidebar's polling costs a single executor round-trip, not two. - try: - active = profiles_mod.get_active_profile() or "default" - except Exception: - active = "default" - try: - current = profiles_mod.get_active_profile_name() or "default" - except Exception: - current = "default" - return {"active": active, "current": current} + # Both reads touch the filesystem; batched into one hop so the sidebar's polling + # costs a single executor round-trip, not two. + def _or_default(fn): + try: + return fn() or "default" + except Exception: + return "default" + return {"active": _or_default(profiles_mod.get_active_profile), + "current": _or_default(profiles_mod.get_active_profile_name)} return await run_in_threadpool(_run) @router.post("/api/profiles/active") async def set_active_profile_endpoint(body: ProfileActiveUpdate): - """Set the sticky active profile (mirrors ``hermes profile use``). Does not - retarget the already-running dashboard — it changes which profile - subsequent CLI commands and gateways use.""" + """Set the sticky active profile (mirrors ``hermes profile use``). Does not retarget the + already-running dashboard — it changes which profile subsequent CLI commands and + gateways use.""" from hermes_cli import profiles as profiles_mod with _profile_errors("POST /api/profiles/active failed"): - # set_active_profile() stats the target profile, creates the state - # directory and writes active_profile through a temp file + replace. + # set_active_profile() stats the target, creates the state directory and writes + # active_profile through a temp file + replace. await run_in_threadpool(profiles_mod.set_active_profile, body.name) return {"ok": True, "active": profiles_mod.normalize_profile_name(body.name)} @@ -907,20 +850,21 @@ async def get_profile_setup_command(name: str): return {"command": _profile_setup_command(name)} +# (executable, flag) — a flag of None means the emulator takes one quoted `sh -lc '…'` string +# after -e; "" means the argv follows the executable directly (kitty). +_LINUX_TERMINALS = ( + ("x-terminal-emulator", "-e"), ("gnome-terminal", "--"), ("konsole", "-e"), + ("xfce4-terminal", None), ("mate-terminal", None), ("lxterminal", None), + ("tilix", "-e"), ("alacritty", "-e"), ("kitty", ""), ("xterm", "-e"), +) + + def _linux_terminal_commands(command: str) -> list: sh = ["sh", "-lc", command] quoted = f"sh -lc '{command}'" return [ - ("x-terminal-emulator", ["x-terminal-emulator", "-e", *sh]), - ("gnome-terminal", ["gnome-terminal", "--", *sh]), - ("konsole", ["konsole", "-e", *sh]), - ("xfce4-terminal", ["xfce4-terminal", "-e", quoted]), - ("mate-terminal", ["mate-terminal", "-e", quoted]), - ("lxterminal", ["lxterminal", "-e", quoted]), - ("tilix", ["tilix", "-e", *sh]), - ("alacritty", ["alacritty", "-e", *sh]), - ("kitty", ["kitty", *sh]), - ("xterm", ["xterm", "-e", *sh]), + (exe, [exe, "-e", quoted] if flag is None else [exe, *([flag] if flag else []), *sh]) + for exe, flag in _LINUX_TERMINALS ] @@ -951,14 +895,13 @@ async def rename_profile_endpoint(name: str, body: ProfileRename): from hermes_cli import profiles as profiles_mod with _profile_errors("PATCH /api/profiles/%s failed", name, bad_request=(ValueError, FileExistsError)): - # rename_profile() stops a running gateway through the same 10-second - # _stop_gateway_process() poll that delete does, then renames the - # profile directory, rewrites the Honcho host blocks and regenerates - # the wrapper script. + # rename_profile() stops a running gateway through the same 10-second poll that + # delete does, then renames the directory, rewrites the Honcho host blocks and + # regenerates the wrapper script. path = await run_in_threadpool(profiles_mod.rename_profile, name, body.new_name) - # For the default profile the rename lands as a presentation-only - # display_name; the canonical id ("default") is unchanged. Always return - # the canonical id so callers keying on `name` stay correct. + # For the default profile the rename lands as a presentation-only display_name; the + # canonical id ("default") is unchanged and always returned so callers keying on `name` + # stay correct. try: is_default = profiles_mod.normalize_profile_name(name) == "default" except ValueError: @@ -970,13 +913,12 @@ async def rename_profile_endpoint(name: str, body: ProfileRename): @router.delete("/api/profiles/{name}") async def delete_profile_endpoint(name: str): - """The dashboard collects the user's confirmation in its own dialog, so - ``yes=True`` always skips the CLI's interactive prompt.""" + """The dashboard collects the user's confirmation in its own dialog, so ``yes=True`` + always skips the CLI's interactive prompt.""" from hermes_cli import profiles as profiles_mod with _profile_errors("DELETE /api/profiles/%s failed", name): - # delete_profile() stops a running gateway by polling its PID once - # every 500 ms for up to 10 s, then rmtree()s the profile directory; - # on the loop that parks every request past the desktop's 10 s + # delete_profile() polls a running gateway's PID for up to 10 s, then rmtree()s the + # directory; on the loop that parks every request past the desktop's 10 s # WebSocket ready-probe. path = await run_in_threadpool(profiles_mod.delete_profile, name, yes=True) return {"ok": True, "path": str(path)} @@ -987,8 +929,8 @@ async def get_profile_soul(name: str): soul_path = _resolve_profile_dir(name) / "SOUL.md" def _run(): - # Probe and read in the same hop: two round-trips would also widen the - # window between the existence check and the read. + # Probe and read in the same hop: two round-trips would also widen the window + # between the existence check and the read. if not soul_path.exists(): return _MISSING return soul_path.read_text(encoding="utf-8") @@ -1009,22 +951,19 @@ async def update_profile_soul(name: str, body: ProfileSoulUpdate): def _run(): from utils import atomic_write_text - # PUT replaces the whole persona document. A bare write_text() truncates - # SOUL.md before the new body lands, and the paired GET reports an - # unreadable file as ``{"content": "", "exists": False}`` — so an - # interrupted save reads as "never set" and the editor's next Save - # persists that empty document over it. + # PUT replaces the whole persona document. A bare write_text() truncates SOUL.md + # before the new body lands, and the paired GET reports an unreadable file as + # ``{"content": "", "exists": False}`` — so an interrupted save reads as "never set" + # and the editor's next Save persists that empty document over it. # - # preserve_mode keeps an existing file's mode/owner across the replace. - # create_mode=0o644 covers the first save: named profiles seed SOUL.md - # at the umask default (profiles chmods only .env to 0600) and SOUL.md - # is not a secret. (The default profile's seeder runs on every - # load_config, so its file already exists and preserve_mode applies.) + # preserve_mode keeps an existing file's mode/owner across the replace. create_mode + # 0o644 covers the first save: named profiles seed SOUL.md at the umask default + # (profiles chmods only .env to 0600) and SOUL.md is not a secret. atomic_write_text(soul_path, body.content, preserve_mode=True, create_mode=0o644) try: - # atomic_write_text() writes a temp file, fsyncs it and replaces the - # original — blocking for as long as the filesystem takes to commit. + # atomic_write_text() writes a temp file, fsyncs and replaces — blocking for as long + # as the filesystem takes to commit. await run_in_threadpool(_run) except OSError as e: _log.exception("PUT /api/profiles/%s/soul failed", name) @@ -1034,9 +973,9 @@ async def update_profile_soul(name: str, body: ProfileSoulUpdate): @router.put("/api/profiles/{name}/description") async def update_profile_description_endpoint(name: str, body: ProfileDescriptionUpdate): - """Set or clear a profile's role description (kanban routing signal). - Non-empty stores it as user-authored (``description_auto: false``) so the - auto-describer won't overwrite it on a sweep.""" + """Set or clear a profile's role description (kanban routing signal). Non-empty stores + it as user-authored (``description_auto: false``) so the auto-describer won't overwrite + it on a sweep.""" from hermes_cli import profiles as profiles_mod profile_dir = _resolve_profile_dir(name) text = (body.description or "").strip() @@ -1050,10 +989,9 @@ async def update_profile_description_endpoint(name: str, body: ProfileDescriptio @router.put("/api/profiles/{name}/model") async def update_profile_model_endpoint(name: str, body: ProfileModelUpdate): - """Set the main model (``model.default`` + ``model.provider``) for a - specific profile's config.yaml without touching the dashboard's own - active profile. Mirrors ``POST /api/model/set`` (main scope) scoped to - the named profile via the HERMES_HOME override.""" + """Set the main model (``model.default`` + ``model.provider``) for a specific profile's + config.yaml without touching the dashboard's own active profile. Mirrors + ``POST /api/model/set`` (main scope) via the HERMES_HOME override.""" profile_dir = _resolve_profile_dir(name) provider = (body.provider or "").strip() model = (body.model or "").strip() @@ -1061,7 +999,6 @@ async def update_profile_model_endpoint(name: str, body: ProfileModelUpdate): raise HTTPException(status_code=400, detail="provider and model are required") with _profile_errors("PUT /api/profiles/%s/model failed", name, not_found=(), bad_request=()): - # _write_profile_model() reads and rewrites the profile's config.yaml. await run_in_threadpool(_write_profile_model, profile_dir, provider, model) return {"ok": True, "provider": provider, "model": model} @@ -1069,12 +1006,11 @@ async def update_profile_model_endpoint(name: str, body: ProfileModelUpdate): @router.post("/api/profiles/{name}/describe-auto") async def describe_profile_auto_endpoint(name: str, body: ProfileDescribeAuto): """Auto-generate a profile's description via the auxiliary LLM - (``auxiliary.profile_describer``); mirrors ``hermes profile describe - --auto``. A failed generation (no aux client, LLM error, …) is - ``ok: false`` with a reason rather than an HTTP error so the UI can - surface it inline and let the operator fix config and retry.""" - # Resolution stays on the loop: a name check plus one stat, and it owns - # the 400/404 mapping that the 500 fallback below would flatten. + (``auxiliary.profile_describer``); mirrors ``hermes profile describe --auto``. + A failed generation (no aux client, LLM error, …) is ``ok: false`` with a reason rather + than an HTTP error so the UI can surface it inline and let the operator retry.""" + # Resolution stays on the loop: a name check plus one stat, and it owns the 400/404 + # mapping that the 500 fallback below would flatten. _resolve_profile_dir(name) def _run(): @@ -1083,25 +1019,23 @@ async def describe_profile_auto_endpoint(name: str, body: ProfileDescribeAuto): with _profile_errors("POST /api/profiles/%s/describe-auto failed", name, not_found=(), bad_request=()): - # describe_profile() is a synchronous LLM round-trip with a 60 s - # ceiling; held on the loop it stalls every other dashboard request. + # describe_profile() is a synchronous LLM round-trip with a 60 s ceiling; held on + # the loop it stalls every other dashboard request. outcome = await run_in_threadpool(_run) return { "ok": bool(outcome.ok), "reason": outcome.reason, "description": outcome.description, - # Only a successful generation is an auto-authored description. A failed - # sweep leaves any existing description untouched, so don't claim it's - # auto-generated. + # Only a successful generation is an auto-authored description. A failed sweep + # leaves any existing description untouched, so don't claim it's auto-generated. "description_auto": bool(outcome.ok), } # ── Export / Import ────────────────────────────────────────────────────────── -# Profile sharing for the desktop: wraps hermes_cli.profiles.export_profile / -# import_profile (the same machinery behind `hermes profile export|import`). -# Paths are exchanged, not bytes — the desktop's local and pooled backends -# share the filesystem with the native save/open dialogs that produce them. +# Profile sharing for the desktop: wraps hermes_cli.profiles.export_profile / import_profile +# (the machinery behind `hermes profile export|import`). Paths are exchanged, not bytes — +# the desktop's local and pooled backends share the filesystem with the native dialogs. def _read_desktop_overlay(profile_dir: Path) -> Any: @@ -1150,9 +1084,8 @@ async def import_profile_endpoint(body: ProfileImport): profiles_mod.create_wrapper_script(imported) _best_effort("Creating wrapper for imported profile %s failed", imported, fn=_wrapper) - # Surface the bundled desktop appearance overlay (if the archive carried - # one) so the desktop can apply theme/interface prefs without another - # round-trip. + # Surface the bundled desktop appearance overlay (if the archive carried one) so the + # desktop can apply theme/interface prefs without another round-trip. desktop_overlay = None if (profile_dir / "desktop.json").is_file(): desktop_overlay = _best_effort( @@ -1163,13 +1096,13 @@ async def import_profile_endpoint(body: ProfileImport): @router.get("/api/profiles/{name}/desktop-overlay") async def get_profile_desktop_overlay(name: str): - """The desktop appearance/interface overlay bundled with an imported - profile (``desktop.json`` at the profile root), or ``exists: false``.""" + """The desktop appearance/interface overlay bundled with an imported profile + (``desktop.json`` at the profile root), or ``exists: false``.""" profile_dir = _resolve_profile_dir(name) def _run(): - # Probe and read in one hop; _MISSING (not None) because desktop.json - # may legitimately hold the document ``null``. + # Probe and read in one hop; _MISSING (not None) because desktop.json may + # legitimately hold the document ``null``. if not (profile_dir / "desktop.json").is_file(): return _MISSING return _read_desktop_overlay(profile_dir) @@ -1178,8 +1111,6 @@ async def get_profile_desktop_overlay(name: str): overlay = await run_in_threadpool(_run) except Exception as e: raise HTTPException(status_code=500, detail=f"Could not read desktop.json: {e}") - # _MISSING rather than None: an overlay file holding the document ``null`` - # exists, and must not be reported as absent. if overlay is _MISSING: return {"exists": False, "desktop": None} return {"exists": True, "desktop": overlay} diff --git a/hermes_cli/web_routers/status.py b/hermes_cli/web_routers/status.py index 5ebfe74b4f..5e7feddeca 100644 --- a/hermes_cli/web_routers/status.py +++ b/hermes_cli/web_routers/status.py @@ -1,10 +1,12 @@ -"""Status dashboard routes: health, /api/status, system stats, curator, learning graph, portal and diagnostics actions. +"""Status dashboard routes: health, /api/status, system stats, curator, learning graph, +portal and diagnostics actions. Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch on ``web_server`` stay there and are imported lazily at call time (cycle-safe). """ import concurrent.futures +import importlib import logging import re import asyncio @@ -18,6 +20,7 @@ from gateway.status import derive_gateway_busy, derive_gateway_drainable, normal from hermes_cli import __version__, __release_date__ from hermes_cli.config import get_config_path, get_env_path from hermes_cli.web_models import CuratorPause, LearningNodeRef, LearningNodeEdit, DebugShareRequest +from hermes_cli.web_routers._common import scoped_to_thread from pathlib import Path from typing import Any, Dict, Optional @@ -33,7 +36,6 @@ _dashboard_local_update_managed_externally = late("_dashboard_local_update_manag _display_system_platform = late("_display_system_platform") _load_configured_gateway_platforms = late("_load_configured_gateway_platforms") _probe_gateway_health = late("_probe_gateway_health") -_profile_scope = late("_profile_scope") _require_token = late("_require_token") _resolve_profile_dir = late("_resolve_profile_dir") _resolve_restart_drain_timeout = late("_resolve_restart_drain_timeout") @@ -52,21 +54,17 @@ _open_session_db_for_profile = late("_open_session_db_for_profile") _STATUS_ACTIVE_SESSIONS_TIMEOUT = 0.75 +_GATEWAY_HEALTH_ROUTE_TIMEOUT = 1.0 +_HEALTHY_PLATFORM_STATES = {"connected", "running", "ok"} def _count_status_active_sessions() -> int: - """Return the dashboard status active-session count. - - This is best-effort status garnish, not a critical path. Opens read-only - (via the shared stale-schema heal, same as every other dashboard read - path) so /api/status never routinely writes to state.db while another - Hermes process is using it. - """ + """Best-effort status garnish. Opens read-only (via the shared stale-schema heal) so + /api/status never routinely writes to state.db while another Hermes process uses it.""" from hermes_state import _default_db_path - # The heal helper bootstraps a missing store; this garnish must not — on - # a fresh install /api/status polls would otherwise create state.db - # before the user's first session. + # The heal helper bootstraps a missing store; this garnish must not — on a fresh install + # /api/status polls would otherwise create state.db before the user's first session. if not Path(_default_db_path()).exists(): return 0 @@ -83,9 +81,6 @@ def _count_status_active_sessions() -> int: db.close() -_GATEWAY_HEALTH_ROUTE_TIMEOUT = 1.0 - - async def _status_active_sessions() -> int: try: return await asyncio.wait_for( @@ -126,13 +121,10 @@ async def get_health(): } -_PROFILE_PLATFORM_STATUS_KEY_RE = re.compile( - # Profile segment mirrors hermes_cli.profiles._PROFILE_ID_RE. Platform - # segment mirrors the Platform enum's normalized values: built-in members - # plus plugin directory names (lowercased), which allow hyphens as well - # as underscores (e.g. ``reviewer:foo-bar``). - r"^[a-z0-9][a-z0-9_-]{0,63}:[a-z0-9][a-z0-9_-]{0,63}$" -) +# Profile segment mirrors hermes_cli.profiles._PROFILE_ID_RE. Platform segment mirrors the +# Platform enum's normalized values: built-in members plus plugin directory names +# (lowercased), which allow hyphens as well as underscores (e.g. ``reviewer:foo-bar``). +_PROFILE_PLATFORM_STATUS_KEY_RE = re.compile(r"^[a-z0-9][a-z0-9_-]{0,63}:[a-z0-9][a-z0-9_-]{0,63}$") def _is_profile_platform_status_key(key: object) -> bool: @@ -140,17 +132,14 @@ def _is_profile_platform_status_key(key: object) -> bool: return isinstance(key, str) and bool(_PROFILE_PLATFORM_STATUS_KEY_RE.fullmatch(key)) -def _status_platform_key_allowed( - key: object, configured: "set[str] | None" -) -> bool: - """Decide whether a runtime-status platform key may appear publicly. +def _status_platform_key_allowed(key: object, configured: "set[str] | None") -> bool: + """Whether a runtime-status platform key may appear publicly. - Namespaced ``:`` keys are validated against the key - grammar *unconditionally* — the config-set load failing must not fail - open into projecting arbitrary colon-containing keys from a process-local - JSON file onto the public endpoint. Plain platform keys keep the - long-standing behavior: checked against the configured set when it - loaded, passed through when it did not. + Namespaced ``:`` keys are validated against the key grammar + *unconditionally* — the config-set load failing must not fail open into projecting + arbitrary colon-containing keys from a process-local JSON file onto the public + endpoint. Plain platform keys keep the long-standing behavior: checked against the + configured set when it loaded, passed through when it did not. """ if not isinstance(key, str): return False @@ -159,10 +148,9 @@ def _status_platform_key_allowed( return configured is None or key in configured -# Per-entry writer-identity stamps (added by gateway.status.write_runtime_status -# for the aggregation ownership check) are process recon — the same class of -# detail as the auth-gated top-level ``gateway_pid`` — and must not project -# onto the public endpoint. +# Per-entry writer-identity stamps (added by gateway.status.write_runtime_status for the +# aggregation ownership check) are process recon — the same class of detail as the +# auth-gated top-level ``gateway_pid`` — and must not project onto the public endpoint. _PRIVATE_PLATFORM_ENTRY_KEYS = frozenset({"writer_pid", "writer_start_time"}) @@ -173,20 +161,15 @@ def _public_platform_entry(value: Any) -> Any: return {k: v for k, v in value.items() if k not in _PRIVATE_PLATFORM_ENTRY_KEYS} -def _merge_profile_gateway_platforms( - gateway_platforms: dict, profile_platforms: dict -) -> dict: - """Merge independent per-profile gateway platform states (OOF-3). +def _merge_profile_gateway_platforms(gateway_platforms: dict, profile_platforms: dict) -> dict: + """Merge independent per-profile gateway platform states. - Hosts that run separate gateway services per profile (``gateway_mode == - "multiple"``) persist each profile's platform failures in that profile's - own ``gateway_state.json``. The unparameterized ``/api/status`` — the - machine-level probe NAS health monitoring reads — only read the active - profile's file, so those failures were invisible to fleet health. Fold - them in under the same validated ``:`` grammar the - multiplex path uses. The active profile's own map is skipped (its - entries are already present, including any multiplex-namespaced ones), - and existing keys are never overwritten. + Hosts running separate gateway services per profile (``gateway_mode == "multiple"``) + persist each profile's platform failures in that profile's own ``gateway_state.json``. + The unparameterized ``/api/status`` — the machine-level probe NAS health monitoring + reads — only read the active profile's file, so those failures were invisible to fleet + health. Fold them in under the validated ``:`` grammar. The active + profile's own map is skipped (already present) and existing keys are never overwritten. """ try: from hermes_cli.profiles import get_active_profile_name @@ -207,20 +190,247 @@ def _merge_profile_gateway_platforms( return merged +def _bounded_health_probe(): + """Health probe with the route's blocking-call budget preserved. The resolver only + reaches this rung when the local PID probe came up empty, so the timeout is paid at + most once per request and only in the cross-container case that needs it.""" + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: + future = pool.submit(_probe_gateway_health) + try: + return future.result(timeout=_GATEWAY_HEALTH_ROUTE_TIMEOUT) + except concurrent.futures.TimeoutError: + _log.warning( + "/api/status gateway health probe exceeded %.2fs; " + "using local status", + _GATEWAY_HEALTH_ROUTE_TIMEOUT, + ) + return False, None + except Exception: + return False, None + + +def _project_gateway_platforms(gateway_platforms: dict, configured: "set[str] | None", + gateway_running: bool, gateway_state) -> dict: + """Public projection of a runtime's platform map. + + Colon-containing keys are validated against the narrow grammar UNCONDITIONALLY — a + failed config load must not fail open into projecting arbitrary keys from a + process-local JSON file onto this public endpoint (the config set belongs to the + active/default profile, so suffix-checking secondary-profile keys against it would + incorrectly hide them). A cleanly stopped gateway's platform states are stale noise + and are cleared so a dead process can't report "connected"; a startup_failed + gateway's FATAL entries are the diagnosis (per-profile credential collisions, auth + failures) that the single exit_reason string can't express, so they are kept — + upstream writer-identity/freshness filtering already dropped other processes' entries. + """ + platforms = { + key: _public_platform_entry(value) + for key, value in gateway_platforms.items() + if _status_platform_key_allowed(key, configured) + } + if gateway_running: + return platforms + if gateway_state == "startup_failed": + return {key: value for key, value in platforms.items() + if isinstance(value, dict) and value.get("state") == "fatal"} + return {} + + +async def _resolve_gateway_status(profile_dir: Optional[Path], health_url) -> Dict[str, Any]: + """Liveness + runtime-state readout: gateway_running/pid/state/platforms/exit_reason/ + updated_at plus the raw ``runtime`` document used downstream. + + Liveness is delegated to the single shared ladder in gateway.status so this endpoint + and /api/messaging/platforms can never disagree about whether the gateway is up. With + ``?profile=`` PID/state reads are scoped to that profile's directory — gateway + identity files live in the per-profile home, not the process-level HERMES_HOME; plain + /api/status keeps the exact zero-arg call so its behavior (and cache signature) is + unchanged. The module-level probe references are handed to the resolver so the + long-standing ``monkeypatch.setattr(web_server, "get_running_pid_cached", ...)`` seam + still intercepts them. + """ + local_runtime = (read_runtime_status(path=profile_dir / "gateway_state.json") + if profile_dir else read_runtime_status()) + liveness = await run_in_threadpool(lambda: resolve_gateway_liveness( + profile_dir=profile_dir, runtime=local_runtime, + health_probe=_bounded_health_probe if health_url else None, + pid_probe=get_running_pid_cached, runtime_reader=read_runtime_status, + runtime_pid_probe=get_runtime_status_running_pid)) + gateway_running = liveness.running + remote_health_body: dict | None = liveness.health_body + + try: + configured = await run_in_threadpool(_load_configured_gateway_platforms) + except Exception: + configured = None + + # Prefer the detailed health endpoint response (has full state) when the local runtime + # status file is absent or stale (cross-container). + runtime = local_runtime + if runtime is None and remote_health_body and remote_health_body.get("gateway_state"): + runtime = remote_health_body + + gateway_state = None + gateway_platforms: dict = {} + gateway_exit_reason = None + gateway_updated_at = None + if runtime: + gateway_state = runtime.get("gateway_state") + if not gateway_running: + gateway_state = gateway_state if gateway_state in {"stopped", "startup_failed"} else "stopped" + elif remote_health_body is not None and gateway_state in {None, "stopped"}: + # The health probe confirmed the gateway is alive, but the local runtime status + # file may be stale (cross-container): override so the badge is correct. + gateway_state = "running" + gateway_platforms = _project_gateway_platforms( + runtime.get("platforms") or {}, configured, gateway_running, gateway_state) + gateway_exit_reason = runtime.get("exit_reason") + # Contract: gateway_updated_at is RFC3339 string | null, never a number. ``runtime`` + # may be the local gateway_state.json (legacy gateways wrote epoch floats; hand + # edits can inject anything) or a remote /health/detailed body — normalize both. + gateway_updated_at = normalize_updated_at(runtime.get("updated_at")) + + # No runtime info at all but the health probe confirmed alive (no shared volume). + if gateway_running and gateway_state is None and remote_health_body is not None: + gateway_state = "running" + + return { + "runtime": runtime, "gateway_running": gateway_running, "gateway_pid": liveness.pid, + "gateway_state": gateway_state, "gateway_platforms": gateway_platforms, + "gateway_exit_reason": gateway_exit_reason, "gateway_updated_at": gateway_updated_at, + } + + +def _auth_gate_status() -> Dict[str, Any]: + """Dashboard auth gate readout: whether the gate is engaged, which providers are + registered, and the RFC 8252 native-app capability advertisement (``auth_flows``). + + The desktop reads ``auth_flows`` to decide whether it can use the system-browser + + loopback + PKCE flow or must fall back to the embedded-webview cookie flow. "cookie" is + always available in gated mode; "native_pkce" is present when at least one interactive + session provider is registered (OAuth providers broker the IDP round trip, password + providers complete at /login in the system browser where OS password managers can + autofill). Token-only credentials (e.g. drain) don't count. Absent field / missing + "native_pkce" ⇒ older gateway ⇒ desktop falls back automatically. + """ + auth_required = bool(getattr(app.state, "auth_required", False)) + auth_providers: list[str] = [] + auth_flows: list[str] = [] + try: + from hermes_cli.dashboard_auth import ( + list_providers as _list_providers, + list_session_providers as _list_session_providers, + ) + auth_providers = [p.name for p in _list_providers()] + if auth_required: + auth_flows.append("cookie") + if _list_session_providers(): + auth_flows.append("native_pkce") + except Exception: + # Module not importable yet (early startup) — leave as []. + pass + return {"auth_required": auth_required, "auth_providers": auth_providers, "auth_flows": auth_flows} + + +def _nous_session_validity() -> str: + """Nous bootstrap-session validity for the NAS health sweep. A hosted agent whose Nous + auth dies terminally (invalid_grant / quarantine) looks HEALTHY to every liveness probe + yet every inference turn fails; this is the ONLY signal that surfaces it, determinable + with no working token (local auth-store state). NAS re-mints the bootstrap session when + it reads "terminal". Best-effort: never let auth classification break the probe.""" + try: + from hermes_cli.auth import get_nous_session_validity + return get_nous_session_validity() + except Exception: + return "unknown" + + +async def _component_health(gateway: Dict[str, Any]) -> Dict[str, Any]: + """Component-level health rollup: counts and status enums only — this payload is + public (PUBLIC_API_PATHS), so no messages, paths, or other detail that could carry + secrets. The storage probe reuses the gateway readiness state_db check (read-only, + 1s-bounded) in an executor so a wedged DB can't stall the event loop.""" + from hermes_cli.web_server import DASHBOARD_HEALTH + + gateway_running, gateway_state = gateway["gateway_running"], gateway["gateway_state"] + gateway_platforms = gateway["gateway_platforms"] + components: Dict[str, Any] = { + "gateway": { + "status": "ok" if gateway_running and gateway_state in {"running", "draining"} else "degraded", + "state": gateway_state or ("running" if gateway_running else "stopped"), + }, + "dashboard": DASHBOARD_HEALTH.snapshot(), + } + try: + from gateway.readiness import _probe_state_db + + storage_check = await run_in_threadpool(_probe_state_db, get_hermes_home()) + components["storage"] = {"status": storage_check.get("status", "degraded")} + except Exception: + components["storage"] = {"status": "degraded"} + platform_states = [ + str(value.get("state") or value.get("status") or "").lower() + for value in gateway_platforms.values() + if isinstance(value, dict) + ] + connected = sum(1 for state in platform_states if state in _HEALTHY_PLATFORM_STATES) + components["platforms"] = { + "status": "ok" if connected == len(platform_states) else "degraded", + "configured": len(gateway_platforms), + "connected": connected, + } + return components + + +async def _advisory_pressure(status: Dict[str, Any], home: Path) -> None: + """Memory / disk pressure rollups + deferred FTS rebuild progress. + + Coarse MB numbers/enums/booleans only — this endpoint is public, same disclosure class + as nous_session_valid. Deliberately NOT folded into components/overall: pressure is + advisory (toast material), not a liveness verdict, and flipping ``overall`` on it would + page NAS's availability sweep for a condition the valve is already handling. The FTS + probe (schema v23) lets the desktop render "search index rebuilding: N%"; absent when + no rebuild is pending. All read-only, never raise. + """ + for key, mod_name, fn_name in (("memory", "gateway.memory_status", "collect_memory_status"), + ("disk", "gateway.disk_status", "collect_disk_status")): + try: + collect = getattr(importlib.import_module(mod_name), fn_name) + status[key] = await run_in_threadpool(collect, home) + except Exception: + status[key] = {"pressure": "unknown"} + + try: + from hermes_state import SessionDB as _SDB + from hermes_constants import get_hermes_home as _ghh + + _db_path = _ghh() / "state.db" + if _db_path.exists(): + _sdb = _SDB(db_path=_db_path, read_only=True) + try: + _rebuild = _sdb.fts_rebuild_status() + finally: + _sdb.close() + if _rebuild is not None: + status["fts_rebuild"] = _rebuild + except Exception: + pass + + @router.get("/api/status") async def get_status(profile: Optional[str] = None): - from hermes_cli.web_server import DASHBOARD_HEALTH, _GATEWAY_HEALTH_URL + """Public machine-level liveness probe (``PUBLIC_API_PATHS``): version, gateway state, + active session count and the auth-gate shape — no bodies, no session content, no secrets. + + Plain /api/status stays the machine-level probe; the dashboard adds ``?profile=`` when + its management switcher targets another profile so its gateway badge follows. That uses + the config-only (contextvar) scope, NOT _profile_scope: this handler awaits the remote + health probe, and _profile_scope swaps process-global skills-module attributes that a + concurrent request would cross-restore across that await. + """ + from hermes_cli.web_server import _GATEWAY_HEALTH_URL status_scope = None requested_profile = (profile or "").strip() - # Plain /api/status stays the machine-level public liveness probe. The - # dashboard adds ?profile= when its management switcher targets another - # profile, so its gateway badge reflects the selected profile. - # - # Use the config-only (contextvar) scope, NOT _profile_scope: this handler - # awaits the remote-health probe, and _profile_scope swaps process-global - # skills-module attributes that a concurrent request would cross-restore - # across that await. Status only resolves get_hermes_home() at call time - # (config/env/gateway state), which the task-local contextvar covers. profile_dir: Optional[Path] = None if requested_profile and requested_profile.lower() != "current": profile_dir = _resolve_profile_dir(requested_profile) @@ -229,225 +439,33 @@ async def get_status(profile: Optional[str] = None): try: current_ver, latest_ver = check_config_version() - # --- Gateway liveness detection --- - # Delegated to the single shared ladder in gateway.status so this - # endpoint and /api/messaging/platforms can never disagree about - # whether the gateway is up (they used to: sidebar "running" while - # the Channels page rendered "The gateway is not running"). - # - # When ?profile= was given, scope PID and state reads to that - # profile's directory — gateway identity files (PID, lock, runtime - # status) are written to the per-profile home, not the process-level - # HERMES_HOME (see issue #69143). Plain /api/status keeps the exact - # zero-arg call so its behavior (and cache signature) is unchanged. - # - # The module-level probe references are handed to the resolver so the - # long-standing `monkeypatch.setattr(web_server, "get_running_pid_cached", ...)` - # seam used across the test-suite still intercepts them. - def _bounded_health_probe(): - """Health probe with the route's blocking-call budget preserved. + gateway = await _resolve_gateway_status(profile_dir, _GATEWAY_HEALTH_URL) + gateway_running, gateway_state = gateway["gateway_running"], gateway["gateway_state"] - The resolver only reaches this rung when the local PID probe came - up empty, so the timeout is paid at most once per request and only - in the cross-container case that needs it. - """ - with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: - future = pool.submit(_probe_gateway_health) - try: - return future.result(timeout=_GATEWAY_HEALTH_ROUTE_TIMEOUT) - except concurrent.futures.TimeoutError: - _log.warning( - "/api/status gateway health probe exceeded %.2fs; " - "using local status", - _GATEWAY_HEALTH_ROUTE_TIMEOUT, - ) - return False, None - except Exception: - return False, None - - local_runtime = ( - read_runtime_status(path=profile_dir / "gateway_state.json") - if profile_dir - else read_runtime_status() - ) - - liveness = await run_in_threadpool( - lambda: resolve_gateway_liveness( - profile_dir=profile_dir, - runtime=local_runtime, - health_probe=_bounded_health_probe if _GATEWAY_HEALTH_URL else None, - pid_probe=get_running_pid_cached, - runtime_reader=read_runtime_status, - runtime_pid_probe=get_runtime_status_running_pid, - ) - ) - gateway_running = liveness.running - gateway_pid = liveness.pid - remote_health_body: dict | None = liveness.health_body - - gateway_state = None - gateway_platforms: dict = {} - gateway_exit_reason = None - gateway_updated_at = None - configured_gateway_platforms: set[str] | None = None - try: - configured_gateway_platforms = await run_in_threadpool( - _load_configured_gateway_platforms - ) - except Exception: - configured_gateway_platforms = None - - # Prefer the detailed health endpoint response (has full state) when the - # local runtime status file is absent or stale (cross-container). - runtime = local_runtime - if runtime is None and remote_health_body and remote_health_body.get("gateway_state"): - runtime = remote_health_body - - if runtime: - gateway_state = runtime.get("gateway_state") - gateway_platforms = runtime.get("platforms") or {} - # Namespaced entries are emitted by configured secondary-profile - # adapters. The config set here belongs to the active/default - # profile, so suffix-checking against it would incorrectly hide - # secondary-only platforms. Colon-containing keys are validated - # against the narrow key grammar UNCONDITIONALLY — a failed config - # load must not fail open into projecting arbitrary keys from a - # process-local JSON file onto this public endpoint. - gateway_platforms = { - key: _public_platform_entry(value) - for key, value in gateway_platforms.items() - if _status_platform_key_allowed(key, configured_gateway_platforms) - } - gateway_exit_reason = runtime.get("exit_reason") - # Contract: gateway_updated_at is RFC3339 string | null, never a - # number. ``runtime`` here may be the local gateway_state.json - # (legacy gateways wrote epoch floats; hand edits can inject - # anything) or a remote /health/detailed body — normalize both. - gateway_updated_at = normalize_updated_at(runtime.get("updated_at")) - if not gateway_running: - gateway_state = gateway_state if gateway_state in {"stopped", "startup_failed"} else "stopped" - # A cleanly stopped gateway's platform states are stale noise — - # clear them so a dead process can't report "connected". But a - # startup_failed gateway's FATAL entries are the diagnosis: - # they carry per-profile credential collisions and auth - # failures (multiplex entries under ``:``) - # that the single exit_reason string can't express. Writer - # -identity and freshness filtering upstream already dropped - # entries from other/older processes, so keeping fatals here - # cannot leak another gateway's live state (#80451 follow-up). - if gateway_state == "startup_failed": - gateway_platforms = { - key: value - for key, value in gateway_platforms.items() - if isinstance(value, dict) and value.get("state") == "fatal" - } - else: - gateway_platforms = {} - elif gateway_running and remote_health_body is not None: - # The health probe confirmed the gateway is alive, but the local - # runtime status file may be stale (cross-container). Override - # stopped/None state so the dashboard shows the correct badge. - if gateway_state in {None, "stopped"}: - gateway_state = "running" - - # If there was no runtime info at all but the health probe confirmed alive, - # ensure we still report the gateway as running (no shared volume scenario). - if gateway_running and gateway_state is None and remote_health_body is not None: - gateway_state = "running" - - # Profile + gateway topology (cached, TTL 10s): fetched here — before - # the platform rollup — because plain ``/api/status`` is the - # machine-level probe NAS reads, and hosts running independent - # per-profile gateway services (gateway_mode == "multiple") persist - # each profile's platform failures in that profile's own - # gateway_state.json. Fold those in under the validated - # ``:`` grammar so fleet health sees them (OOF-3). - # A ``?profile=`` request targets one profile's view and is left - # unmerged. + # Topology (cached, TTL 10s) is fetched before the platform rollup because plain + # /api/status is the machine-level probe NAS reads: hosts with per-profile gateway + # services persist each profile's platform failures in its own gateway_state.json, + # folded in here under the validated ``:`` grammar. A + # ``?profile=`` request targets one profile's view and is left unmerged. topology = await run_in_threadpool(_collect_profile_gateway_topology_cached) if not requested_profile: - gateway_platforms = _merge_profile_gateway_platforms( - gateway_platforms, topology.get("profile_platforms") or {} - ) + gateway["gateway_platforms"] = _merge_profile_gateway_platforms( + gateway["gateway_platforms"], topology.get("profile_platforms") or {}) active_sessions = await _status_active_sessions() - # Busy/drainable readout (NAS lifecycle-safety gate). active_agents is - # the in-flight gateway-turn count the gateway now persists at every - # turn boundary; gateway_busy/gateway_drainable are derived from it + - # liveness via the single shared contract in gateway.status. Liveness - # keys off gateway_running (a live PID/health probe), NEVER - # gateway_updated_at — a healthy idle gateway never advances that. - active_agents = parse_active_agents((runtime or {}).get("active_agents", 0)) - gateway_busy = derive_gateway_busy( - gateway_running=gateway_running, - gateway_state=gateway_state, - active_agents=active_agents, - ) - gateway_drainable = derive_gateway_drainable( - gateway_running=gateway_running, - gateway_state=gateway_state, - ) - # Resolved drain timeout (seconds) so NAS can size its poll deadline - # without out-of-band knowledge. Offload to a thread: on a cold - # Windows install the first import of hermes_cli.gateway blocks the - # asyncio event loop for 15-30s (.pyc compilation + Defender scans), - # exceeding the desktop handshake's 15s socket timeout. After the - # first call the module is in sys.modules and the worker call returns - # in microseconds. + # Busy/drainable readout (NAS lifecycle-safety gate). active_agents is the in-flight + # gateway-turn count persisted at every turn boundary; busy/drainable derive from it + # + liveness via the shared contract in gateway.status. Liveness keys off + # gateway_running (live PID/health probe), NEVER gateway_updated_at — a healthy idle + # gateway never advances that. + active_agents = parse_active_agents((gateway["runtime"] or {}).get("active_agents", 0)) + # Drain timeout offloaded to a thread: on a cold Windows install the first import of + # hermes_cli.gateway blocks the loop for 15-30s (.pyc compilation + Defender scans), + # exceeding the desktop handshake's 15s socket timeout. restart_drain_timeout = await run_in_threadpool(_resolve_restart_drain_timeout) + auth = _auth_gate_status() - # Dashboard auth gate (Phase 7): surface whether the gate is engaged - # and which providers are registered so ``hermes status`` and the - # SPA's StatusPage can show "OAuth gate ON via Nous Research" or - # "loopback only — no auth gate" with no extra round trips. - auth_required = bool(getattr(app.state, "auth_required", False)) - auth_providers: list[str] = [] - # RFC 8252 native-app capability advertisement. The desktop reads this - # to decide whether it can use the system-browser + loopback + PKCE - # flow (no embedded webview, no session cookies) or must fall back to - # the legacy embedded-webview cookie flow. "cookie" is always available - # in gated mode; "native_pkce" is present when at least one interactive - # session provider is registered — OAuth providers broker the upstream - # IDP round trip, password providers complete interactively at /login - # in the system browser (where OS password managers can autofill; an - # embedded webview cannot reach them). Token-only credentials (e.g. - # drain) don't count. Absent field / missing "native_pkce" ⇒ older - # gateway ⇒ desktop falls back automatically. - auth_flows: list[str] = [] - try: - from hermes_cli.dashboard_auth import ( - list_providers as _list_providers, - list_session_providers as _list_session_providers, - ) - auth_providers = [p.name for p in _list_providers()] - if auth_required: - auth_flows.append("cookie") - if _list_session_providers(): - auth_flows.append("native_pkce") - except Exception: - # Module not importable yet (early startup) — leave as []. - pass - - # Nous bootstrap-session validity for the NAS health sweep. A hosted - # agent whose Nous auth dies terminally (invalid_grant / quarantine) - # looks HEALTHY to every liveness/connectivity probe — the machine, - # relay, and this dashboard all stay up — yet every inference turn - # fails. This is the ONLY signal that surfaces that condition, and it - # is determinable with no working token (local auth-store state). NAS - # re-mints the bootstrap session when it reads "terminal". Best-effort: - # never let auth classification break the public liveness probe. - nous_session_valid = "unknown" - try: - from hermes_cli.auth import get_nous_session_validity - nous_session_valid = get_nous_session_validity() - except Exception: - nous_session_valid = "unknown" - - # Always-public liveness + auth-gate shape. Safe for external uptime - # probes (NAS's wildcard-subdomain liveness probe), the SPA's pre-login - # bootstrap, and anyone who can curl the host — i.e. exactly the audience - # ``PUBLIC_API_PATHS`` documents this endpoint as serving. status = { "version": __version__, "release_date": __release_date__, @@ -456,159 +474,50 @@ async def get_status(profile: Optional[str] = None): "can_update_hermes": not _dashboard_local_update_managed_externally(), "gateway_running": gateway_running, "gateway_state": gateway_state, - "gateway_platforms": gateway_platforms, - "gateway_exit_reason": gateway_exit_reason, - "gateway_updated_at": gateway_updated_at, + "gateway_platforms": gateway["gateway_platforms"], + "gateway_exit_reason": gateway["gateway_exit_reason"], + "gateway_updated_at": gateway["gateway_updated_at"], "active_agents": active_agents, - "gateway_busy": gateway_busy, - "gateway_drainable": gateway_drainable, + "gateway_busy": derive_gateway_busy( + gateway_running=gateway_running, gateway_state=gateway_state, active_agents=active_agents), + "gateway_drainable": derive_gateway_drainable( + gateway_running=gateway_running, gateway_state=gateway_state), "restart_drain_timeout": restart_drain_timeout, "active_sessions": active_sessions, - "auth_required": auth_required, - "auth_providers": auth_providers, - "auth_flows": auth_flows, - "nous_session_valid": nous_session_valid, + **auth, + "nous_session_valid": _nous_session_validity(), } - # Stable per-install identity (see get_install_id above). First call - # may touch disk, so keep it off the event loop; afterwards it is a - # process-global cache hit. Omitted (not null) when unpersistable so - # older-client behavior and the no-identity fallback stay identical. + # Stable per-install identity. First call may touch disk, so keep it off the loop; + # afterwards it is a process-global cache hit. Omitted (not null) when unpersistable + # so older-client behavior and the no-identity fallback stay identical. install_id = await run_in_threadpool(get_install_id) if install_id: status["install_id"] = install_id - # Component-level health rollup. Counts and status enums only — this - # payload is public (PUBLIC_API_PATHS), so no messages, paths, or - # other detail that could carry secrets. The storage probe reuses the - # gateway readiness state_db check (read-only, 1s-bounded) in an - # executor so a wedged DB can't stall the event loop. - components: Dict[str, Any] = { - "gateway": { - "status": "ok" if gateway_running and gateway_state in {"running", "draining"} else "degraded", - "state": gateway_state or ("running" if gateway_running else "stopped"), - }, - "dashboard": DASHBOARD_HEALTH.snapshot(), - } - try: - from gateway.readiness import _probe_state_db - - storage_check = await run_in_threadpool(_probe_state_db, get_hermes_home()) - components["storage"] = {"status": storage_check.get("status", "degraded")} - except Exception: - components["storage"] = {"status": "degraded"} - platform_states = [ - str(value.get("state") or value.get("status") or "").lower() - for value in gateway_platforms.values() - if isinstance(value, dict) - ] - platforms_ok = all( - state in {"connected", "running", "ok"} for state in platform_states - ) - components["platforms"] = { - "status": "ok" if platforms_ok else "degraded", - "configured": len(gateway_platforms), - "connected": sum( - 1 for state in platform_states if state in {"connected", "running", "ok"} - ), - } + components = await _component_health(gateway) status["components"] = components - status["overall"] = ( - "ok" - if all(item.get("status") == "ok" for item in components.values()) - else "degraded" - ) + status["overall"] = ("ok" if all(item.get("status") == "ok" for item in components.values()) + else "degraded") + await _advisory_pressure(status, profile_dir if profile_dir else get_hermes_home()) - # Memory-pressure rollup (NS-656). Distilled from the gateway's - # 30s loop heartbeat + lifecycle sentinel — two small file reads, - # no gateway IPC. Coarse MB numbers/enums/booleans only: this - # endpoint is public (PUBLIC_API_PATHS), same disclosure class as - # nous_session_valid above. Deliberately NOT folded into - # components/overall — memory pressure is advisory (toast/notice - # material), not a liveness verdict, and flipping `overall` to - # "degraded" on it would page NAS's availability sweep for a - # condition the valve is already handling. - try: - from gateway.memory_status import collect_memory_status - - status["memory"] = await run_in_threadpool( - collect_memory_status, - profile_dir if profile_dir else get_hermes_home(), - ) - except Exception: - status["memory"] = {"pressure": "unknown"} - - # Disk-usage rollup (NS-656, same lineage as OOF-2/OOF-107 fleet - # disk-exhaustion incidents). One statvfs call on HERMES_HOME's - # filesystem — coarse MB numbers + enum, same public disclosure - # class as the memory block, and equally advisory: not folded - # into components/overall. - try: - from gateway.disk_status import collect_disk_status - - status["disk"] = await run_in_threadpool( - collect_disk_status, - profile_dir if profile_dir else get_hermes_home(), - ) - except Exception: - status["disk"] = {"pressure": "unknown"} - - # Deferred FTS rebuild progress (schema v23): lets the desktop / - # dashboard render a "search index rebuilding: N%" indicator instead - # of users wondering why old-message search is slower after an - # update. None/absent when no rebuild is pending (the common case). - # Read-only probe, never blocks startup, never raises. - try: - from hermes_state import SessionDB as _SDB - from hermes_constants import get_hermes_home as _ghh - - _db_path = _ghh() / "state.db" - if _db_path.exists(): - _sdb = _SDB(db_path=_db_path, read_only=True) - try: - _rebuild = _sdb.fts_rebuild_status() - finally: - _sdb.close() - if _rebuild is not None: - status["fts_rebuild"] = _rebuild - except Exception: - pass - - # Profile + gateway topology: which profiles exist, whether one - # multiplexed gateway or several per-profile gateways serve them, and - # (gated) which host ports the live gateways' port-binding platforms - # listen on. Enumerating profiles walks the filesystem and probes the - # process table, so keep it off the event loop. - # - # Split by sensitivity: profile NAMES (``profiles``) and the gateway - # ``gateway_mode`` are low-sensitivity PRODUCT surface — Hermes Cloud - # renders the profile list in the Portal, which reads this endpoint over - # the network (a gated bind), so they must survive the auth gate. The - # per-gateway ``gateways[]`` detail carries host ports (deployment - # recon), so it stays gated with the host paths / PID below. - # (``topology`` was already fetched above, before the platform rollup, - # so the per-profile platform merge could use it — the TTL cache makes - # the earlier fetch the only real scan either way.) + # Split by sensitivity: profile NAMES and ``gateway_mode`` are low-sensitivity + # PRODUCT surface — Hermes Cloud renders the profile list in the Portal over a gated + # bind, so they must survive the auth gate. Per-gateway ``gateways[]`` carries host + # ports (deployment recon) and stays gated with the host paths / PID below. status["profiles"] = topology["profiles"] status["gateway_mode"] = topology["gateway_mode"] - # Absolute host paths, the gateway PID, the internal gateway health - # URL, and per-gateway ports are deployment recon a liveness probe never - # needs. ``/api/status`` is in ``PUBLIC_API_PATHS`` so it bypasses - # dashboard auth; on a network-exposed (gated) bind that means *any* - # unauthenticated caller reaches it, and leaking host metadata there - # contradicts the allowlist's own contract ("version, gateway state, - # active session count, and the dashboard auth-gate shape. No bodies, no - # session content, no secrets"). Surface this detail only on a loopback - # / ``--insecure`` bind, where the dashboard is local-only and the - # caller is already inside the trust envelope — the same loopback/gated - # split ``should_require_auth`` draws. - if not auth_required: + # Absolute host paths, the gateway PID, the internal health URL and per-gateway ports + # are deployment recon a liveness probe never needs; on a network-exposed (gated) + # bind *any* unauthenticated caller reaches this endpoint. Surface them only on a + # loopback / ``--insecure`` bind — the same split ``should_require_auth`` draws. + if not auth["auth_required"]: status.update({ "hermes_home": str(get_hermes_home()), "config_path": str(get_config_path()), "env_path": str(get_env_path()), - "gateway_pid": gateway_pid, + "gateway_pid": gateway["gateway_pid"], "gateway_health_url": _GATEWAY_HEALTH_URL, "gateways": topology["gateways"], }) @@ -621,12 +530,9 @@ async def get_status(profile: Optional[str] = None): @router.get("/api/system/stats") async def get_system_stats(): - """Host + process system stats for the System page. - - OS / Python / host identity from stdlib; CPU / memory / disk / uptime from - psutil when available, with graceful degradation when it isn't. Read-only - and non-sensitive (no env values, no paths beyond the hermes home root). - """ + """Host + process system stats for the System page: OS / Python / host identity from + stdlib; CPU / memory / disk / uptime from psutil when available. Read-only and + non-sensitive (no env values, no paths beyond the hermes home root).""" import platform as _platform info: Dict[str, Any] = { @@ -644,54 +550,42 @@ async def get_system_stats(): "cpu_count": os.cpu_count(), } + def _optional(fill): + try: + fill() + except Exception: + pass + + def _disk(): + du = psutil.disk_usage(str(get_hermes_home())) + info["disk"] = {"total": du.total, "used": du.used, "free": du.free, "percent": du.percent} + + def _cpu(): + info["cpu_percent"] = psutil.cpu_percent(interval=0.1) + la = getattr(psutil, "getloadavg", None) + if la: + info["load_avg"] = list(la()) + + def _uptime(): + info["uptime_seconds"] = int(time.time() - psutil.boot_time()) + + def _process(): + proc = psutil.Process() + info["process"] = {"pid": proc.pid, "rss": proc.memory_info().rss, + "create_time": int(proc.create_time()), "num_threads": proc.num_threads()} + # psutil enriches the picture when present; everything below is optional. try: import psutil # type: ignore vm = psutil.virtual_memory() - info["memory"] = { - "total": vm.total, - "available": vm.available, - "used": vm.used, - "percent": vm.percent, - } - try: - du = psutil.disk_usage(str(get_hermes_home())) - info["disk"] = { - "total": du.total, - "used": du.used, - "free": du.free, - "percent": du.percent, - } - except Exception: - pass - try: - info["cpu_percent"] = psutil.cpu_percent(interval=0.1) - la = getattr(psutil, "getloadavg", None) - if la: - info["load_avg"] = list(la()) - except Exception: - pass - try: - boot = psutil.boot_time() - info["uptime_seconds"] = int(time.time() - boot) - except Exception: - pass - try: - proc = psutil.Process() - info["process"] = { - "pid": proc.pid, - "rss": proc.memory_info().rss, - "create_time": int(proc.create_time()), - "num_threads": proc.num_threads(), - } - except Exception: - pass + info["memory"] = {"total": vm.total, "available": vm.available, "used": vm.used, "percent": vm.percent} + for fill in (_disk, _cpu, _uptime, _process): + _optional(fill) info["psutil"] = True except Exception: info["psutil"] = False - # stdlib-only fallbacks for load average + uptime where the kernel - # exposes them. + # stdlib-only fallbacks for load average where the kernel exposes it. try: info["load_avg"] = list(os.getloadavg()) except (OSError, AttributeError): @@ -701,11 +595,8 @@ async def get_system_stats(): # --------------------------------------------------------------------------- -# Curator endpoints — background skill-maintenance status + controls. -# -# The curator periodically reviews skills (archive stale, prune, pin). The -# dashboard surfaces its state and the pause/resume/run-now controls that -# `hermes curator` exposes. +# Curator endpoints — background skill-maintenance (archive stale, prune, pin) status + +# the pause/resume/run-now controls `hermes curator` exposes. # --------------------------------------------------------------------------- @@ -738,81 +629,68 @@ async def set_curator_paused(body: CuratorPause): return {"ok": True, "paused": bool(body.paused)} +def _spawn_action(argv: list, name: str, prefix: str) -> dict: + """Spawn a background ``hermes `` action; a spawn failure is ``500 ": "``.""" + try: + proc = _spawn_hermes_action(argv, name) + except Exception as exc: + raise HTTPException(status_code=500, detail=f"{prefix}: {exc}") + return {"ok": True, "pid": proc.pid, "name": name} + + @router.post("/api/curator/run") async def run_curator(): """Trigger a curator review now (backgrounded; tail via action status).""" - try: - proc = _spawn_hermes_action(["curator", "run"], "curator-run") - except Exception as exc: - raise HTTPException(status_code=500, detail=f"Failed to run curator: {exc}") - return {"ok": True, "pid": proc.pid, "name": "curator-run"} + return _spawn_action(["curator", "run"], "curator-run", "Failed to run curator") @router.get("/api/learning/graph") async def get_learning_graph(profile: Optional[str] = None): - """Learning graph payload for the desktop panel. - - Profile-scoped view of learned, non-base skills plus memory chunks, with - graph links derived from skill relations and memory-skill overlap. - """ + """Learning graph payload for the desktop panel: profile-scoped view of learned, + non-base skills plus memory chunks, with links from skill relations and memory-skill + overlap.""" def _run(): from agent.learning_graph import build_learning_graph - - with _profile_scope(profile): - return build_learning_graph() + return build_learning_graph() try: - # _profile_scope takes _SKILLS_PROFILE_LOCK and the graph build reads - # skills/memories from disk — keep it off the event loop. - return await asyncio.to_thread(_run) + # _profile_scope takes _SKILLS_PROFILE_LOCK and the graph build reads skills/memories + # from disk — keep it off the event loop. + return await scoped_to_thread(profile, _run) except Exception: _log.exception("GET /api/learning/graph failed") raise HTTPException(status_code=500, detail="Failed to build learning graph") +async def _learning_mutation(profile: Optional[str], fn, status: int, fallback: str): + """Run a learning_mutations call under ``_profile_scope`` off-loop; a non-ok result + becomes ``HTTPException(status, message)``.""" + res = await scoped_to_thread(profile, fn) + if not res.get("ok"): + raise HTTPException(status_code=status, detail=res.get("message", fallback)) + return res + + @router.get("/api/learning/node") async def get_learning_node(id: str, profile: Optional[str] = None): """Current content of a journey node (skill SKILL.md or memory chunk), for an edit prefill.""" from agent.learning_mutations import node_detail - - def _run(): - with _profile_scope(profile): - return node_detail(id) - - res = await asyncio.to_thread(_run) - if not res.get("ok"): - raise HTTPException(status_code=404, detail=res.get("message", "not found")) - return res + return await _learning_mutation(profile, lambda: node_detail(id), 404, "not found") @router.delete("/api/learning/node") async def delete_learning_node(body: LearningNodeRef): """Delete a journey node — skills are archived (restorable), memories removed.""" from agent.learning_mutations import delete_node - - def _run(): - with _profile_scope(body.profile): - return delete_node(body.id) - - res = await asyncio.to_thread(_run) - if not res.get("ok"): - raise HTTPException(status_code=400, detail=res.get("message", "delete failed")) - return res + return await _learning_mutation(body.profile, lambda: delete_node(body.id), 400, "delete failed") @router.put("/api/learning/node") async def update_learning_node(body: LearningNodeEdit): """Rewrite a journey node's content (SKILL.md or memory chunk).""" from agent.learning_mutations import edit_node - - def _run(): - with _profile_scope(body.profile): - return edit_node(body.id, body.content) - - res = await asyncio.to_thread(_run) - if not res.get("ok"): - raise HTTPException(status_code=400, detail=res.get("message", "edit failed")) - return res + return await _learning_mutation( + body.profile, lambda: edit_node(body.id, body.content), 400, "edit failed") def _safe_call(mod, fn_name: str, default): @@ -830,12 +708,17 @@ def _safe_call(mod, fn_name: str, default): @router.get("/api/portal") async def get_portal_status(): - # load_config() + auth/subscription snapshots are disk reads — this is a - # polled endpoint, so keep them off the event loop. - def _run(): - return _get_portal_status_sync() + # load_config() + auth/subscription snapshots are disk reads on a polled endpoint — + # keep them off the event loop. + return await asyncio.to_thread(_get_portal_status_sync) - return await asyncio.to_thread(_run) + +def _feature_state(feat) -> str: + if getattr(feat, "managed_by_nous", False): + return "via Nous Portal" + if getattr(feat, "active", False): + return getattr(feat, "current_provider", None) or "active" + return "not configured" def _get_portal_status_sync(): @@ -844,8 +727,8 @@ def _get_portal_status_sync(): try: from hermes_cli.auth import get_nous_auth_status_local - # Read-only dashboard endpoint: refresh-free snapshot so polling - # never performs an OAuth refresh or burns a refresh token. + # Read-only dashboard endpoint: refresh-free snapshot so polling never performs an + # OAuth refresh or burns a refresh token. auth = get_nous_auth_status_local() or {} except Exception: auth = {} @@ -856,16 +739,8 @@ def _get_portal_status_sync(): feats = get_nous_subscription_features(cfg) if feats is not None: - for feat in feats.items(): - if getattr(feat, "managed_by_nous", False): - state = "via Nous Portal" - elif getattr(feat, "active", False) and getattr(feat, "current_provider", None): - state = feat.current_provider - elif getattr(feat, "active", False): - state = "active" - else: - state = "not configured" - features.append({"label": getattr(feat, "label", ""), "state": state}) + features = [{"label": getattr(feat, "label", ""), "state": _feature_state(feat)} + for feat in feats.items()] except Exception: _log.exception("portal features failed") @@ -881,49 +756,34 @@ def _get_portal_status_sync(): # --------------------------------------------------------------------------- -# Diagnostics: prompt-size, support dump, debug upload, config migrate. -# All produce text output, so they spawn background actions tailed via -# /api/actions//status. +# Diagnostics: prompt-size, support dump, debug upload, config migrate. All produce text +# output, so they spawn background actions tailed via /api/actions//status. # --------------------------------------------------------------------------- @router.post("/api/ops/prompt-size") async def run_prompt_size(): - try: - proc = _spawn_hermes_action(["prompt-size"], "prompt-size") - except Exception as exc: - raise HTTPException(status_code=500, detail=f"Failed: {exc}") - return {"ok": True, "pid": proc.pid, "name": "prompt-size"} + return _spawn_action(["prompt-size"], "prompt-size", "Failed") @router.post("/api/ops/dump") async def run_dump(): - try: - proc = _spawn_hermes_action(["dump"], "dump") - except Exception as exc: - raise HTTPException(status_code=500, detail=f"Failed: {exc}") - return {"ok": True, "pid": proc.pid, "name": "dump"} + return _spawn_action(["dump"], "dump", "Failed") @router.post("/api/ops/config-migrate") async def run_config_migrate(): - try: - proc = _spawn_hermes_action(["config", "migrate"], "config-migrate") - except Exception as exc: - raise HTTPException(status_code=500, detail=f"Failed: {exc}") - return {"ok": True, "pid": proc.pid, "name": "config-migrate"} + return _spawn_action(["config", "migrate"], "config-migrate", "Failed") @router.post("/api/ops/debug-share") async def run_debug_share_endpoint(body: DebugShareRequest | None = None): """Upload a redacted debug report + full logs and return the paste URLs. - Unlike the other diagnostics actions (doctor, dump, prompt-size) this is - *synchronous*: the whole point of ``debug share`` is the set of shareable - URLs it produces, so we run the upload in a worker thread and return the - structured ``{urls, failures, redacted, ...}`` payload directly. The - dashboard renders those as real, copyable links instead of scraping a log - tail. Pastes auto-delete after 6 hours (handled inside the share core). + Unlike the other diagnostics actions this is *synchronous*: the point of ``debug share`` + is the set of shareable URLs it produces, so the upload runs in a worker thread and the + structured ``{urls, failures, redacted, ...}`` payload is returned directly for the + dashboard to render as copyable links. Pastes auto-delete after 6 hours (share core). """ from hermes_cli.debug import build_debug_share @@ -977,9 +837,9 @@ async def get_logs( except ImportError: COMPONENT_PREFIXES = {} - # Normalize "ALL" / "all" / empty → no filter. _matches_filters treats an - # empty tuple as "must match a prefix" (startswith(()) is always False), - # so passing () instead of None silently drops every line. + # Normalize "ALL" / "all" / empty → no filter. _matches_filters treats an empty tuple as + # "must match a prefix" (startswith(()) is always False), so passing () instead of None + # silently drops every line. min_level = level if level and level.upper() != "ALL" else None if component and component.lower() != "all": comp_prefixes = COMPONENT_PREFIXES.get(component) @@ -999,9 +859,8 @@ async def get_logs( min_level=min_level, component_prefixes=comp_prefixes, ) - # Post-filter by search term (case-insensitive substring match). - # _read_tail doesn't support free-text search, so we filter here and - # trim to the requested line count afterward. + # _read_tail doesn't support free-text search, so post-filter (case-insensitive + # substring) here and trim to the requested line count afterward. if search: needle = search.lower() result = [l for l in result if needle in l.lower()][-min(lines, 500):] diff --git a/hermes_cli/web_server_profiles.py b/hermes_cli/web_server_profiles.py index dae0d46b30..39a0d8ce85 100644 --- a/hermes_cli/web_server_profiles.py +++ b/hermes_cli/web_server_profiles.py @@ -1,4 +1,5 @@ -"""Profile-scoped helpers: profile discovery fallback, profile dir/MCP-server writes, the profile/config scope context managers, skills-hub and tools/analytics catalog helpers. +"""Profile-scoped helpers: profile discovery fallback, profile dir/MCP-server writes, +the profile/config scope context managers, skills-hub and tools/analytics catalog helpers. Split out of ``hermes_cli.web_server``; every externally used name is re-imported there, so ``web_server.`` keeps resolving (and monkeypatching) as before. @@ -14,7 +15,7 @@ import threading from contextlib import contextmanager from fastapi import HTTPException from pathlib import Path -from typing import Any, Dict, List, Optional +from typing import Any, Callable, Dict, List, Optional from hermes_cli.config import DEFAULT_CONFIG, get_process_hermes_home from hermes_cli.web_models import MCPServerCreate from hermes_cli.web_server_gateway import _ACTION_LOG_FILES @@ -24,14 +25,39 @@ from hermes_cli.web_server_mcp import _normalize_mcp_server_create _log = logging.getLogger("hermes_cli.web_server") +def _safe(callable_: Callable[[], Any], default: Any) -> Any: + try: + return callable_() + except Exception: + return default + + +def _is_current_profile(profile: Optional[str]) -> bool: + """None/""/"current" all mean the dashboard's own profile.""" + requested = (profile or "").strip() + return not requested or requested.lower() == "current" + + +@contextmanager +def _hermes_home_scope(path) -> Any: + """Scope ``load_config``/``save_config`` (anything resolving ``get_hermes_home()`` at call + time) to ``path`` for the block via the context-local HERMES_HOME override.""" + from hermes_constants import set_hermes_home_override, reset_hermes_home_override + + token = set_hermes_home_override(str(path)) + try: + yield + finally: + reset_hermes_home_override(token) + + def _is_other_profile(profile: Optional[str]) -> bool: """True when ``profile`` names a profile other than this process's own.""" from hermes_cli.web_server import _resolve_profile_dir - requested = (profile or "").strip() - if not requested or requested.lower() == "current": + if _is_current_profile(profile): return False try: - target = _resolve_profile_dir(requested) + target = _resolve_profile_dir(profile.strip()) except HTTPException: return True return target.resolve() != get_process_hermes_home().resolve() @@ -40,12 +66,10 @@ def _is_other_profile(profile: Optional[str]) -> bool: def _approval_mode_of(config: Dict[str, Any]) -> str: """Normalize approvals.mode from an in-memory config document. - Both sides of the broadcast comparison use in-memory documents (the raw - on-disk dict and the about-to-be-saved dict): re-reading through the - config cache after a save can serve the pre-save document when the - replacement file collides on the (mtime_ns, size) cache key, which would - suppress the broadcast exactly when the mode changed. Absent block or - key normalizes to the same default the approval gate uses. + Both sides of the broadcast comparison use in-memory documents: re-reading through + the config cache after a save can serve the pre-save document when the replacement + file collides on the (mtime_ns, size) cache key, suppressing the broadcast exactly + when the mode changed. Absent block/key normalizes to the approval gate's default. """ from tools.approval import _normalize_approval_mode @@ -56,11 +80,8 @@ def _approval_mode_of(config: Dict[str, Any]) -> str: def _broadcast_gateway_session_info() -> None: - """Broadcast session.info on the in-process gateway when it's loaded. - - ``sys.modules`` guard, not an import: gateway never imported means no - live sessions in this process to notify. - """ + """Broadcast session.info on the in-process gateway when it's loaded (``sys.modules`` + guard, not an import: gateway never imported means no live sessions to notify).""" server = sys.modules.get("tui_gateway.server") if server is None: return @@ -71,12 +92,9 @@ def _broadcast_gateway_session_info() -> None: def _parse_model_ids(resp: "Any") -> List[str]: - """Extract model ids from an OpenAI-compatible ``/v1/models`` response. - - Tolerant of the common shapes: ``{"data": [{"id": ...}]}`` (OpenAI / vLLM / - llama.cpp) and a bare ``{"data": ["id", ...]}``. Returns ``[]`` on any - parse/HTTP error so a slightly non-standard endpoint never hard-blocks. - """ + """Model ids from an OpenAI-compatible ``/v1/models`` response: ``{"data": [{"id": ..}]}`` + or a bare ``{"data": ["id", ..]}``. ``[]`` on any parse/HTTP error so a slightly + non-standard endpoint never hard-blocks.""" try: if not resp.is_success: return [] @@ -86,81 +104,54 @@ def _parse_model_ids(resp: "Any") -> List[str]: data = payload.get("data") if isinstance(payload, dict) else payload if not isinstance(data, list): return [] - ids: List[str] = [] - for item in data: - if isinstance(item, dict): - mid = str(item.get("id") or "").strip() - else: - mid = str(item or "").strip() - if mid: - ids.append(mid) - return ids + ids = [str((item.get("id") if isinstance(item, dict) else item) or "").strip() for item in data] + return [mid for mid in ids if mid] + + +def _fallback_profile_entry(profiles_mod, name: str, home: Path, *, is_default: bool, + has_env: bool, gateway_running: Callable[[], bool]) -> Dict[str, Any]: + model, provider = _safe(lambda: profiles_mod._read_config_model(home), (None, None)) + + def meta(key, default): + return _safe(lambda: profiles_mod.read_profile_meta(home).get(key, default), default) + + return { + "name": name, "path": str(home), "is_default": is_default, "model": model, "provider": provider, + "has_env": has_env, + "skill_count": _safe(lambda: profiles_mod._count_skills(home), 0), + "gateway_running": _safe(gateway_running, False), + "description": meta("description", ""), "description_auto": meta("description_auto", False), + "distribution_name": None, "distribution_version": None, "distribution_source": None, + "has_alias": False, + } def _fallback_profile_dicts(profiles_mod) -> List[Dict[str, Any]]: - def _safe(callable_, default): - try: - return callable_() - except Exception: - return default - profiles: List[Dict[str, Any]] = [] default_home = profiles_mod._get_default_hermes_home() if default_home.is_dir(): - model, provider = _safe(lambda: profiles_mod._read_config_model(default_home), (None, None)) - profiles.append({ - "name": "default", - "path": str(default_home), - "is_default": True, - "model": model, - "provider": provider, - "has_env": (default_home / ".env").exists(), - "skill_count": _safe(lambda: profiles_mod._count_skills(default_home), 0), - "gateway_running": _safe(lambda: profiles_mod._check_gateway_running(default_home), False), - "description": _safe(lambda: profiles_mod.read_profile_meta(default_home).get("description", ""), ""), - "description_auto": _safe(lambda: profiles_mod.read_profile_meta(default_home).get("description_auto", False), False), - "distribution_name": None, - "distribution_version": None, - "distribution_source": None, - "has_alias": False, - }) + profiles.append(_fallback_profile_entry( + profiles_mod, "default", default_home, is_default=True, + has_env=(default_home / ".env").exists(), + gateway_running=lambda: profiles_mod._check_gateway_running(default_home))) profiles_root = profiles_mod._get_profiles_root() if profiles_root.is_dir(): - # Use os.scandir (context-managed) instead of Path.iterdir to avoid - # leaking directory fds when an exception interrupts iteration — the - # sidebar polls every few seconds so an fd leak exhausts RLIMIT_NOFILE - # within days (#81547). + # os.scandir (context-managed) rather than Path.iterdir: an exception mid-iteration + # must not leak the directory fd — the sidebar polls every few seconds, so a leak + # exhausts RLIMIT_NOFILE within days. with os.scandir(profiles_root) as scan: entries = sorted(scan, key=lambda e: e.name) for entry in entries: - entry_path = Path(entry.path) + home = Path(entry.path) if not entry.is_dir() or not profiles_mod._PROFILE_ID_RE.match(entry.name): continue - model, provider = _safe(lambda entry=entry_path: profiles_mod._read_config_model(entry), (None, None)) - profiles.append({ - "name": entry.name, - "path": str(entry_path), - "is_default": False, - "model": model, - "provider": provider, - "has_env": _safe(lambda entry=entry_path: (entry / ".env").exists(), False), - "skill_count": _safe(lambda entry=entry_path: profiles_mod._count_skills(entry), 0), - "gateway_running": _safe( - lambda entry=entry_path, name=entry.name: ( - profiles_mod._check_gateway_running(entry) - or profiles_mod._served_by_running_multiplexer(name) - ), - False, - ), - "description": _safe(lambda entry=entry_path: profiles_mod.read_profile_meta(entry).get("description", ""), ""), - "description_auto": _safe(lambda entry=entry_path: profiles_mod.read_profile_meta(entry).get("description_auto", False), False), - "distribution_name": None, - "distribution_version": None, - "distribution_source": None, - "has_alias": False, - }) - + profiles.append(_fallback_profile_entry( + profiles_mod, entry.name, home, is_default=False, + has_env=_safe(lambda: (home / ".env").exists(), False), + gateway_running=lambda home=home, name=entry.name: ( + profiles_mod._check_gateway_running(home) + or profiles_mod._served_by_running_multiplexer(name)))) return profiles @@ -177,24 +168,16 @@ def _resolve_profile_dir(name: str) -> Path: def _write_profile_mcp_servers(profile_dir: Path, servers: List["MCPServerCreate"]) -> int: - """Write MCP server entries into a specific profile's config.yaml. + """Write MCP server entries into ``profile_dir``'s config.yaml (HERMES_HOME-scoped). - Scopes ``load_config``/``save_config`` to ``profile_dir`` via the - context-local HERMES_HOME override (same mechanism as - ``_write_profile_model``) so the entries land in the target profile's - config rather than the dashboard process's active profile. - - Mirrors the per-server shape the ``POST /api/mcp/servers`` endpoint builds, - but batched so the whole profile-create write is a single config save. - Returns the number of servers written. + Mirrors the per-server shape ``POST /api/mcp/servers`` builds, batched so the whole + profile-create write is one config save. Returns the number of servers written. """ from hermes_cli.web_server import load_config, save_config - from hermes_constants import set_hermes_home_override, reset_hermes_home_override from hermes_cli.mcp_config import _save_bearer_auth_token written = 0 - token = set_hermes_home_override(str(profile_dir)) - try: + with _hermes_home_scope(profile_dir): cfg = load_config() mcp = cfg.setdefault("mcp_servers", {}) for server in servers: @@ -202,11 +185,7 @@ def _write_profile_mcp_servers(profile_dir: Path, servers: List["MCPServerCreate name, entry, bearer_token = _normalize_mcp_server_create(server) except ValueError as exc: display_name = (server.name or "").strip() or "" - _log.warning( - "Profile-create: skipping MCP server '%s': %s", - display_name, - exc, - ) + _log.warning("Profile-create: skipping MCP server '%s': %s", display_name, exc) continue if bearer_token is not None: entry["headers"] = _save_bearer_auth_token(name, bearer_token) @@ -215,25 +194,17 @@ def _write_profile_mcp_servers(profile_dir: Path, servers: List["MCPServerCreate if written: save_config(cfg) elif not mcp: - # We created an empty mcp_servers dict but wrote nothing — don't - # leave a stray empty key in the new profile's config. + # Don't leave the stray empty key we just created in the new profile's config. cfg.pop("mcp_servers", None) save_config(cfg) - finally: - reset_hermes_home_override(token) return written # --------------------------------------------------------------------------- -# Skills & Tools endpoints -# -# Every read/write below accepts an optional ``profile`` query param so the -# dashboard can manage ANY profile's skills/toolsets, not just the profile -# the dashboard process happens to be running under. Without this, "Set as -# active" on the Profiles page (which only flips the sticky ``active_profile`` -# file for FUTURE CLI/gateway invocations) misled users into thinking skill -# toggles would land in the activated profile — they silently wrote into the -# dashboard's own config instead. See _profile_scope() for the mechanism. +# Skills & Tools endpoints accept an optional ``profile`` query param so the dashboard can +# manage ANY profile's skills/toolsets, not just the one the dashboard process runs under +# ("Set as active" only flips the sticky active_profile file for FUTURE invocations, which +# misled users into thinking toggles landed in the activated profile). # --------------------------------------------------------------------------- @@ -244,145 +215,74 @@ _SKILLS_PROFILE_LOCK = threading.RLock() def _profile_scope(profile: Optional[str]): """Scope config + skill-directory resolution to ``profile`` for one request. - Two seams must be redirected for skills/toolsets endpoints: + Two seams: (1) ``load_config``/``save_config`` resolve ``get_hermes_home()`` at call + time, so the contextvar override (``_config_profile_scope``) reaches them; (2) + ``tools.skills_tool`` / ``tools.skill_manager_tool`` bind ``SKILLS_DIR`` at import + time, so both are retargeted under a lock and restored immediately after (like + ``_call_cron_for_profile`` does for cron). ``tools.skills_sync`` needs no retargeting — + its lookups resolve at call time through the same override. - 1. ``load_config``/``save_config`` resolve ``get_hermes_home()`` at call - time — the context-local override from ``set_hermes_home_override`` - reaches them (same pattern as ``_write_profile_model``). - 2. ``tools.skills_tool`` and ``tools.skill_manager_tool`` bind - ``SKILLS_DIR`` at import time, so the override CANNOT reach them. - Like ``_call_cron_for_profile`` does for cron's module globals, - temporarily retarget both under a lock and restore them - immediately after. - - ``tools.skills_sync`` (reset/diff/list-modified/opt-in/opt-out/ - repair-official) needs NO retargeting: since #65828 its directory - lookups resolve at call time through the same contextvar override - set in step 1. - - ``profile`` of None/""/"current" means "the dashboard's own profile" — - config resolution is untouched, but the skill-module globals are still - retargeted to the *current* ``get_hermes_home()`` so writes land in the - live home even when the import-time binding is stale (e.g. the process - imported the modules before a HERMES_HOME override, or under test - isolation). + For the dashboard's own profile (None/""/"current") config resolution is untouched, but + the skill-module globals are still retargeted to the *current* ``get_hermes_home()`` so + writes land in the live home even when the import-time binding is stale (process + imported the modules before a HERMES_HOME override, or under test isolation). + Yields the profile dir for a named profile, None for the current one. """ - from hermes_cli.web_server import _resolve_profile_dir - requested = (profile or "").strip() - - from hermes_constants import ( - get_hermes_home, - set_hermes_home_override, - reset_hermes_home_override, - ) + from hermes_constants import get_hermes_home from tools import skills_tool as _skills_tool from tools import skill_manager_tool as _skill_mgr - token = None - if not requested or requested.lower() == "current": - profile_dir = get_hermes_home() - else: - profile_dir = _resolve_profile_dir(requested) - token = set_hermes_home_override(str(profile_dir)) - - with _SKILLS_PROFILE_LOCK: - old_home = _skills_tool.HERMES_HOME - old_skills_dir = _skills_tool.SKILLS_DIR - old_mgr_home = _skill_mgr.HERMES_HOME - old_mgr_skills_dir = _skill_mgr.SKILLS_DIR - _skills_tool.HERMES_HOME = profile_dir - _skills_tool.SKILLS_DIR = profile_dir / "skills" - _skill_mgr.HERMES_HOME = profile_dir - _skill_mgr.SKILLS_DIR = profile_dir / "skills" - try: - yield profile_dir if token is not None else None - finally: - _skills_tool.HERMES_HOME = old_home - _skills_tool.SKILLS_DIR = old_skills_dir - _skill_mgr.HERMES_HOME = old_mgr_home - _skill_mgr.SKILLS_DIR = old_mgr_skills_dir - if token is not None: - reset_hermes_home_override(token) + with _config_profile_scope(profile) as scoped: + profile_dir = get_hermes_home() if scoped is None else scoped + modules = (_skills_tool, _skill_mgr) + with _SKILLS_PROFILE_LOCK: + saved = [(m.HERMES_HOME, m.SKILLS_DIR) for m in modules] + for m in modules: + m.HERMES_HOME, m.SKILLS_DIR = profile_dir, profile_dir / "skills" + try: + yield scoped + finally: + for m, (home, skills_dir) in zip(modules, saved): + m.HERMES_HOME, m.SKILLS_DIR = home, skills_dir @contextmanager def _config_profile_scope(profile: Optional[str]): """Await-safe, config-only profile scope for handlers that ``await``. - Unlike ``_profile_scope`` this touches ONLY the context-local - ``set_hermes_home_override`` contextvar — it does NOT swap the - process-global ``skills_tool``/``skill_manager`` module attributes. - Those globals are shared across all event-loop tasks, so holding them - across an ``await`` lets a concurrent skills request restore THIS - request's profile dir on its ``finally`` (cross-contamination). The - contextvar override is task-local and survives an ``await`` cleanly, - which is all endpoints that resolve ``get_hermes_home()`` at call time - (config, env, gateway status) actually need. - - None/""/"current" means the dashboard's own profile — no override. + Touches ONLY the task-local ``set_hermes_home_override`` contextvar — never the + process-global ``skills_tool``/``skill_manager`` attributes ``_profile_scope`` swaps: + holding those across an ``await`` lets a concurrent skills request restore THIS + request's dir on its ``finally``. Enough for endpoints that resolve ``get_hermes_home()`` + at call time (config, env, gateway status). None/""/"current" = no override. """ from hermes_cli.web_server import _resolve_profile_dir - requested = (profile or "").strip() - if not requested or requested.lower() == "current": + if _is_current_profile(profile): yield None return - - from hermes_constants import ( - set_hermes_home_override, - reset_hermes_home_override, - ) - - profile_dir = _resolve_profile_dir(requested) - token = set_hermes_home_override(str(profile_dir)) - try: + profile_dir = _resolve_profile_dir(profile.strip()) + with _hermes_home_scope(profile_dir): yield profile_dir - finally: - reset_hermes_home_override(token) # --------------------------------------------------------------------------- -# Terminal execution backend picker — the GUI counterpart of terminal.backend -# in config.yaml. Each row carries a fast, defensive health probe (Docker -# daemon reachable, SSH host configured, Modal/Daytona credentials present) so -# the Capabilities panel can render Ready / Needs setup guidance instead of a -# bare enum (issues #57738 / #63783). Probes must never raise — a probe -# failure renders as a status, not a 500. +# Terminal execution backend picker — GUI counterpart of terminal.backend. Each row carries a +# fast, defensive health probe so the Capabilities panel renders Ready / Needs setup instead +# of a bare enum. Probes must never raise — a probe failure renders as a status, not a 500. +# Keep in sync with tools/terminal_tool.py::_create_environment and the terminal.backend enum. # --------------------------------------------------------------------------- -# Table-driven backend metadata — kept in sync with the dispatch ladder in -# tools/terminal_tool.py::_create_environment and the terminal.backend enum -# surfaced in the desktop raw-config settings. _TERMINAL_BACKENDS: List[Dict[str, str]] = [ - { - "name": "local", - "label": "Local", - "description": "Run commands directly on this machine. No isolation.", - }, - { - "name": "docker", - "label": "Docker", - "description": "Run commands in an isolated Docker container with a persistent workspace.", - }, - { - "name": "singularity", - "label": "Singularity / Apptainer", - "description": "Run commands in a Singularity/Apptainer container (HPC-friendly, rootless).", - }, - { - "name": "modal", - "label": "Modal", - "description": "Run commands in a Modal cloud sandbox.", - }, - { - "name": "daytona", - "label": "Daytona", - "description": "Run commands in a Daytona cloud sandbox.", - }, - { - "name": "ssh", - "label": "SSH", - "description": "Run commands on a remote host over SSH.", - }, + dict(zip(("name", "label", "description"), row)) for row in ( + ("local", "Local", "Run commands directly on this machine. No isolation."), + ("docker", "Docker", + "Run commands in an isolated Docker container with a persistent workspace."), + ("singularity", "Singularity / Apptainer", + "Run commands in a Singularity/Apptainer container (HPC-friendly, rootless)."), + ("modal", "Modal", "Run commands in a Modal cloud sandbox."), + ("daytona", "Daytona", "Run commands in a Daytona cloud sandbox."), + ("ssh", "SSH", "Run commands on a remote host over SSH."), + ) ] @@ -400,11 +300,8 @@ def _plugin_terminal_backend_rows() -> List[Dict[str, str]]: for provider in list_providers(): try: - rows.append({ - "name": provider.name.strip().lower(), - "label": provider.display_name, - "description": provider.description, - }) + rows.append({"name": provider.name.strip().lower(), "label": provider.display_name, + "description": provider.description}) except Exception: continue except Exception: @@ -416,14 +313,17 @@ def _plugin_terminal_backend_rows() -> List[Dict[str, str]]: # Token / cost analytics endpoint # --------------------------------------------------------------------------- +_AUX_COUNTERS = ("input_tokens", "output_tokens", "estimated_cost", "api_calls") + + +def _token_volume(row: Dict[str, Any]) -> Any: + return (row.get("input_tokens") or 0) + (row.get("output_tokens") or 0) + def _aux_usage_rows(db, cutoff: float) -> List[Dict[str, Any]]: - """Per-(model, task) auxiliary usage within the window (issue #23270). - - Reads the task-dimension rows (task != '') that record_auxiliary_usage - writes into session_model_usage. Returns [] when the table predates the - task column (older DB opened read-only by newer code). - """ + """Per-(model, task) auxiliary usage within the window: the task-dimension rows + (task != '') record_auxiliary_usage writes into session_model_usage. [] when the + table predates the task column (older DB opened read-only by newer code).""" try: cur = db._conn.execute(""" SELECT u.model, @@ -445,8 +345,6 @@ def _aux_usage_rows(db, cutoff: float) -> List[Dict[str, Any]]: """, (cutoff,)) return [dict(r) for r in cur.fetchall()] except Exception: - # Table predates the task column (older DB opened by newer code) — - # aux breakdown is simply unavailable. return [] @@ -455,47 +353,22 @@ def _merge_aux_into_by_model( ) -> List[Dict[str, Any]]: """Fold aux usage rows into the sessions-derived per-model list. - Aux usage lives only in session_model_usage (never in the sessions - counters), so adding it here cannot double-count. Models that ONLY - appear via aux calls (e.g. a dedicated vision model) get their own - entry — previously they were entirely invisible. + Aux usage lives only in session_model_usage (never in the sessions counters), so + adding it here cannot double-count. Models that ONLY appear via aux calls (e.g. a + dedicated vision model) get their own entry. """ if not aux_rows: return by_model - merged: Dict[str, Dict[str, Any]] = {} - for row in by_model: - merged[row.get("model") or "unknown"] = row + merged: Dict[str, Dict[str, Any]] = {row.get("model") or "unknown": row for row in by_model} for aux in aux_rows: model = aux.get("model") or "unknown" - target = merged.get(model) - if target is None: - target = { - "model": model, - "input_tokens": 0, - "output_tokens": 0, - "estimated_cost": 0, - "sessions": 0, - "api_calls": 0, - } - merged[model] = target - target["input_tokens"] = (target.get("input_tokens") or 0) + (aux.get("input_tokens") or 0) - target["output_tokens"] = (target.get("output_tokens") or 0) + (aux.get("output_tokens") or 0) - target["estimated_cost"] = (target.get("estimated_cost") or 0) + (aux.get("estimated_cost") or 0) - target["api_calls"] = (target.get("api_calls") or 0) + (aux.get("api_calls") or 0) - tasks = target.setdefault("aux_tasks", []) - tasks.append({ - "task": aux.get("task") or "", - "input_tokens": aux.get("input_tokens") or 0, - "output_tokens": aux.get("output_tokens") or 0, - "estimated_cost": aux.get("estimated_cost") or 0, - "api_calls": aux.get("api_calls") or 0, - }) - result = list(merged.values()) - result.sort( - key=lambda r: (r.get("input_tokens") or 0) + (r.get("output_tokens") or 0), - reverse=True, - ) - return result + target = merged.setdefault(model, {"model": model, "input_tokens": 0, "output_tokens": 0, + "estimated_cost": 0, "sessions": 0, "api_calls": 0}) + for key in _AUX_COUNTERS: + target[key] = (target.get(key) or 0) + (aux.get(key) or 0) + target.setdefault("aux_tasks", []).append( + {"task": aux.get("task") or "", **{key: aux.get(key) or 0 for key in _AUX_COUNTERS}}) + return sorted(merged.values(), key=_token_volume, reverse=True) def _aux_task_summary(aux_rows: List[Dict[str, Any]]) -> List[Dict[str, Any]]: @@ -503,37 +376,22 @@ def _aux_task_summary(aux_rows: List[Dict[str, Any]]) -> List[Dict[str, Any]]: by_task: Dict[str, Dict[str, Any]] = {} for aux in aux_rows: task = aux.get("task") or "" - d = by_task.setdefault(task, { - "task": task, - "input_tokens": 0, - "output_tokens": 0, - "estimated_cost": 0, - "api_calls": 0, - "models": [], - }) - d["input_tokens"] += aux.get("input_tokens") or 0 - d["output_tokens"] += aux.get("output_tokens") or 0 - d["estimated_cost"] += aux.get("estimated_cost") or 0 - d["api_calls"] += aux.get("api_calls") or 0 + d = by_task.setdefault(task, {"task": task, "input_tokens": 0, "output_tokens": 0, + "estimated_cost": 0, "api_calls": 0, "models": []}) + for key in _AUX_COUNTERS: + d[key] += aux.get(key) or 0 model = aux.get("model") or "unknown" if model not in d["models"]: d["models"].append(model) - result = list(by_task.values()) - result.sort( - key=lambda r: (r.get("input_tokens") or 0) + (r.get("output_tokens") or 0), - reverse=True, - ) - return result + return sorted(by_task.values(), key=_token_volume, reverse=True) def _profile_cli_args(profile: Optional[str]) -> List[str]: - """Return ``["-p", ]`` for a validated non-default profile. + """``["-p", ]`` for a validated non-default profile, else ``[]``. - Hub install/uninstall/update run in a fresh ``hermes`` subprocess, and - ``_apply_profile_override()`` reads ``-p`` from argv in the child — the - only mechanism that reaches import-time-bound globals like - ``skills_hub.SKILLS_DIR``. Empty/"current" means the dashboard's own - profile (no args, legacy behavior). + Hub install/uninstall/update run in a fresh ``hermes`` subprocess whose + ``_apply_profile_override()`` reads ``-p`` from argv — the only mechanism that reaches + import-time-bound globals like ``skills_hub.SKILLS_DIR``. """ requested = (profile or "").strip() if not requested or requested.lower() in {"current", "default"}: @@ -547,9 +405,8 @@ def _hub_action_name(verb: str, key: str) -> str: """Unique per-skill hub action name (+ registered log file). ``_spawn_hermes_action`` tracks one process/log per name, so a shared - "skills-install"/"skills-uninstall" would make concurrent row-level actions - overwrite each other's status/log while the UI polls per identifier. Slug - (readable) + hash (collision-proof) keys each action to its own row. + "skills-install" would make concurrent row-level actions overwrite each other's + status/log while the UI polls per identifier. Slug (readable) + hash (collision-proof). """ slug = re.sub(r"[^a-z0-9]+", "-", key.lower()).strip("-")[:48] or "skill" digest = hashlib.sha1(key.encode()).hexdigest()[:8] @@ -559,31 +416,21 @@ def _hub_action_name(verb: str, key: str) -> str: def _installed_hub_identifiers(profile: Optional[str] = None) -> dict: - """Map identifier -> installed lock entry for hub-installed skills. - - Lets the UI mark search results that are already installed. Scoped to - ``profile``'s skills/.hub/lock.json when provided (HubLockFile takes an - explicit path, sidestepping the import-time LOCK_FILE binding). - Best-effort: returns an empty dict if the lock file can't be read. - """ + """identifier -> installed lock entry for hub-installed skills, so the UI can mark + search results already installed. Scoped to ``profile``'s skills/.hub/lock.json when + given (HubLockFile takes an explicit path, sidestepping the import-time LOCK_FILE + binding). Best-effort: {} when the lock file can't be read.""" try: from tools.skills_hub import HubLockFile - requested = (profile or "").strip() - if requested and requested.lower() != "current": - profile_dir = _resolve_profile_dir(requested) - lock = HubLockFile(profile_dir / "skills" / ".hub" / "lock.json") - else: + if _is_current_profile(profile): lock = HubLockFile() - out = {} - for entry in lock.list_installed(): - ident = entry.get("identifier") - if ident: - out[ident] = { - "name": entry.get("name"), - "trust_level": entry.get("trust_level"), - "scan_verdict": entry.get("scan_verdict"), - } - return out + else: + lock = HubLockFile(_resolve_profile_dir(profile.strip()) / "skills" / ".hub" / "lock.json") + return { + entry["identifier"]: {"name": entry.get("name"), "trust_level": entry.get("trust_level"), + "scan_verdict": entry.get("scan_verdict")} + for entry in lock.list_installed() if entry.get("identifier") + } except Exception: return {} From 976b00861136855759a650cea038df0f0d3415a1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:14:17 -0700 Subject: [PATCH 2/4] refactor(web-profiles/status): AST-neutral bracket hugging --- hermes_cli/web_routers/profiles.py | 48 +++++++----------- hermes_cli/web_routers/status.py | 78 ++++++++++-------------------- hermes_cli/web_server_profiles.py | 13 ++--- 3 files changed, 47 insertions(+), 92 deletions(-) diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 402666c7d8..d123daf639 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -90,8 +90,7 @@ def _profile_to_dict(info) -> Dict[str, Any]: "distribution_name": attr("distribution_name", None), "distribution_version": attr("distribution_version", None), "distribution_source": attr("distribution_source", None), - "has_alias": attr("alias_path", None) is not None, - } + "has_alias": attr("alias_path", None) is not None} def _profile_setup_command(name: str) -> str: @@ -128,7 +127,8 @@ def _disable_unselected_skills(profile_dir: Path, keep: List[str]) -> int: keep_set = {s.strip() for s in keep if s and s.strip()} with _hermes_home_scope(profile_dir): skills_root = profile_dir / "skills" - installed = [md.parent.name for md in skills_root.rglob("SKILL.md")] if skills_root.is_dir() else [] + installed = ([md.parent.name for md in skills_root.rglob("SKILL.md")] + if skills_root.is_dir() else []) cfg = load_config() disabled = get_disabled_skills(cfg) newly = 0 @@ -202,8 +202,7 @@ def _tag_rows(rows: List[Dict[str, Any]], name: str, now: float) -> List[Dict[st s["profile"] = name s["is_default_profile"] = name == "default" s["is_active"] = ( - s.get("ended_at") is None - and (now - s.get("last_active", s.get("started_at", 0))) < 300 + s.get("ended_at") is None and (now - s.get("last_active", s.get("started_at", 0))) < 300 ) s["archived"] = bool(s.get("archived")) s["pinned"] = bool(s.get("pinned")) @@ -413,8 +412,7 @@ def get_profiles_sessions( source: str = None, sources: str = None, exclude_sources: str = None, - full: bool = False, -): + full: bool = False): """Unified, read-only session list aggregated across ALL profiles. Process-light: opens each profile's ``state.db`` directly from disk — it does NOT spawn @@ -441,8 +439,7 @@ def get_profiles_sessions( exclude_sources=_csv_list(exclude_sources) or None, min_message_count=max(0, min_messages), include_archived=archived == "include", - archived_only=archived == "only", - ) + archived_only=archived == "only") # Over-fetch per profile so the merged+sorted window is correct for the requested page. # Capped so a huge profile can't blow up the response. per_profile = min(max(limit + offset, limit), 500) @@ -456,8 +453,7 @@ def get_profiles_sessions( rows = db.list_sessions_rich( limit=per_profile, offset=0, order_by_last_active=order == "recent", # Same SQL-level blob skip as /api/sessions. - compact_rows=not full, include_pinned=True, **filters, - ) + compact_rows=not full, include_pinned=True, **filters) totals[name] = db.session_count(exclude_children=True, **filters) merged.extend(_tag_rows(rows, name, now)) _read_profile_db(name, home, errors, _read) @@ -479,8 +475,7 @@ def get_profiles_sessions_sidebar( recents_exclude: str = None, cron_limit: int = 50, messaging_limit: int = 100, - messaging_exclude: str = None, -): + messaging_exclude: str = None): """Batched sidebar session slices — one profile-DB open per refresh. The desktop sidebar needs three source-scoped windows per refresh: recents (local @@ -518,8 +513,7 @@ def get_profiles_sessions_sidebar( return db.list_sessions_rich( source=source, exclude_sources=exclude or None, limit=cap, offset=0, min_message_count=1, include_archived=False, archived_only=False, - order_by_last_active=True, compact_rows=True, include_pinned=True, - ) + order_by_last_active=True, compact_rows=True, include_pinned=True) def _build_slices(db, cache_key): slices = { @@ -528,8 +522,7 @@ def get_profiles_sessions_sidebar( # and a total that shrank when you scrolled would be worse than no total at all. "usage": db.usage_totals(), "cron": _slice(db, source="cron", cap=cron_cap), - "messaging": _slice(db, exclude=messaging_exclude_list, cap=messaging_cap), - } + "messaging": _slice(db, exclude=messaging_exclude_list, cap=messaging_cap)} _sidebar_profile_cache_put(cache_key, slices) return slices @@ -571,8 +564,7 @@ def get_profiles_sessions_sidebar( "profiles_truncated": recents_truncated, "profiles_usage": profile_totals}, "cron": {"sessions": _window(cron_rows, cron_cap)}, "messaging": {"sessions": _window(messaging_rows, messaging_cap), "total": len(messaging_rows)}, - "errors": errors, - } + "errors": errors} def _merge_by_id(into: Dict[str, Dict[str, Any]], entries: List[Dict[str, Any]], child_key: str) -> None: @@ -601,8 +593,7 @@ def _merge_profile_tree( merged: Dict[str, Dict[str, Any]], projects: List[Dict[str, Any]], profile: str, - preview_limit: int, -) -> None: + preview_limit: int) -> None: """Fold one profile's projects into the shared tree, keyed by folder. The same checkout in two profiles is one group, as is ``__no_project__`` (every profile @@ -677,8 +668,7 @@ def get_profiles_projects_tree(preview_limit: int = 3, session_limit: int = 2000 "projects": sorted(merged.values(), key=lambda p: p.get("lastActive") or 0, reverse=True), # Ownership is per profile, so no project is "the active one" here; the desktop only # reads active_id to bias its overview sort. - "active_id": None, "scoped_session_ids": scoped_session_ids, "errors": errors, - } + "active_id": None, "scoped_session_ids": scoped_session_ids, "errors": errors} # `gh pr create` prints the PR url and nothing else, so a tool result whose whole output IS @@ -803,8 +793,7 @@ async def create_profile_endpoint(body: ProfileCreate): {"identifier": ident, "pid": _best_effort( "Spawning hub-skill install %s for new profile %s failed", ident, body.name, fn=lambda: _spawn_install(ident))} - for ident in ((i or "").strip() for i in body.hub_skills) if ident - ] + for ident in ((i or "").strip() for i in body.hub_skills) if ident] return {"ok": True, "name": body.name, "path": str(path), "model_set": model_set, "mcp_written": mcp_written, "skills_disabled": skills_disabled, @@ -855,8 +844,7 @@ async def get_profile_setup_command(name: str): _LINUX_TERMINALS = ( ("x-terminal-emulator", "-e"), ("gnome-terminal", "--"), ("konsole", "-e"), ("xfce4-terminal", None), ("mate-terminal", None), ("lxterminal", None), - ("tilix", "-e"), ("alacritty", "-e"), ("kitty", ""), ("xterm", "-e"), -) + ("tilix", "-e"), ("alacritty", "-e"), ("kitty", ""), ("xterm", "-e")) def _linux_terminal_commands(command: str) -> list: @@ -864,8 +852,7 @@ def _linux_terminal_commands(command: str) -> list: quoted = f"sh -lc '{command}'" return [ (exe, [exe, "-e", quoted] if flag is None else [exe, *([flag] if flag else []), *sh]) - for exe, flag in _LINUX_TERMINALS - ] + for exe, flag in _LINUX_TERMINALS] @router.post("/api/profiles/{name}/open-terminal") @@ -1028,8 +1015,7 @@ async def describe_profile_auto_endpoint(name: str, body: ProfileDescribeAuto): "description": outcome.description, # Only a successful generation is an auto-authored description. A failed sweep # leaves any existing description untouched, so don't claim it's auto-generated. - "description_auto": bool(outcome.ok), - } + "description_auto": bool(outcome.ok)} # ── Export / Import ────────────────────────────────────────────────────────── diff --git a/hermes_cli/web_routers/status.py b/hermes_cli/web_routers/status.py index 5e7feddeca..f38b535894 100644 --- a/hermes_cli/web_routers/status.py +++ b/hermes_cli/web_routers/status.py @@ -75,8 +75,7 @@ def _count_status_active_sessions() -> int: return sum( 1 for s in sessions if s.get("ended_at") is None - and (now - s.get("last_active", s.get("started_at", 0))) < 300 - ) + and (now - s.get("last_active", s.get("started_at", 0))) < 300) finally: db.close() @@ -85,13 +84,11 @@ async def _status_active_sessions() -> int: try: return await asyncio.wait_for( run_in_threadpool(_count_status_active_sessions), - timeout=_STATUS_ACTIVE_SESSIONS_TIMEOUT, - ) + timeout=_STATUS_ACTIVE_SESSIONS_TIMEOUT) except asyncio.TimeoutError: _log.debug( "/api/status active session count exceeded %.2fs; returning 0", - _STATUS_ACTIVE_SESSIONS_TIMEOUT, - ) + _STATUS_ACTIVE_SESSIONS_TIMEOUT) except Exception as exc: _log.debug("/api/status active session count unavailable: %s", exc) return 0 @@ -107,8 +104,7 @@ async def get_ssh_ownership(request: Request): "ok": True, "sshOwnerNonce": _SSH_OWNER_NONCE, "protocolVersion": 1, - "runtimeIntact": _ssh_runtime_intact(), - } + "runtimeIntact": _ssh_runtime_intact()} @router.get("/api/health") @@ -117,8 +113,7 @@ async def get_health(): return { "ok": True, "version": __version__, - "auth_required": bool(getattr(app.state, "auth_required", False)), - } + "auth_required": bool(getattr(app.state, "auth_required", False))} # Profile segment mirrors hermes_cli.profiles._PROFILE_ID_RE. Platform segment mirrors the @@ -202,8 +197,7 @@ def _bounded_health_probe(): _log.warning( "/api/status gateway health probe exceeded %.2fs; " "using local status", - _GATEWAY_HEALTH_ROUTE_TIMEOUT, - ) + _GATEWAY_HEALTH_ROUTE_TIMEOUT) return False, None except Exception: return False, None @@ -226,8 +220,7 @@ def _project_gateway_platforms(gateway_platforms: dict, configured: "set[str] | platforms = { key: _public_platform_entry(value) for key, value in gateway_platforms.items() - if _status_platform_key_allowed(key, configured) - } + if _status_platform_key_allowed(key, configured)} if gateway_running: return platforms if gateway_state == "startup_failed": @@ -297,8 +290,7 @@ async def _resolve_gateway_status(profile_dir: Optional[Path], health_url) -> Di return { "runtime": runtime, "gateway_running": gateway_running, "gateway_pid": liveness.pid, "gateway_state": gateway_state, "gateway_platforms": gateway_platforms, - "gateway_exit_reason": gateway_exit_reason, "gateway_updated_at": gateway_updated_at, - } + "gateway_exit_reason": gateway_exit_reason, "gateway_updated_at": gateway_updated_at} def _auth_gate_status() -> Dict[str, Any]: @@ -318,9 +310,7 @@ def _auth_gate_status() -> Dict[str, Any]: auth_flows: list[str] = [] try: from hermes_cli.dashboard_auth import ( - list_providers as _list_providers, - list_session_providers as _list_session_providers, - ) + list_providers as _list_providers, list_session_providers as _list_session_providers) auth_providers = [p.name for p in _list_providers()] if auth_required: auth_flows.append("cookie") @@ -357,10 +347,8 @@ async def _component_health(gateway: Dict[str, Any]) -> Dict[str, Any]: components: Dict[str, Any] = { "gateway": { "status": "ok" if gateway_running and gateway_state in {"running", "draining"} else "degraded", - "state": gateway_state or ("running" if gateway_running else "stopped"), - }, - "dashboard": DASHBOARD_HEALTH.snapshot(), - } + "state": gateway_state or ("running" if gateway_running else "stopped")}, + "dashboard": DASHBOARD_HEALTH.snapshot()} try: from gateway.readiness import _probe_state_db @@ -371,14 +359,12 @@ async def _component_health(gateway: Dict[str, Any]) -> Dict[str, Any]: platform_states = [ str(value.get("state") or value.get("status") or "").lower() for value in gateway_platforms.values() - if isinstance(value, dict) - ] + if isinstance(value, dict)] connected = sum(1 for state in platform_states if state in _HEALTHY_PLATFORM_STATES) components["platforms"] = { "status": "ok" if connected == len(platform_states) else "degraded", "configured": len(gateway_platforms), - "connected": connected, - } + "connected": connected} return components @@ -485,8 +471,7 @@ async def get_status(profile: Optional[str] = None): "restart_drain_timeout": restart_drain_timeout, "active_sessions": active_sessions, **auth, - "nous_session_valid": _nous_session_validity(), - } + "nous_session_valid": _nous_session_validity()} # Stable per-install identity. First call may touch disk, so keep it off the loop; # afterwards it is a process-global cache hit. Omitted (not null) when unpersistable @@ -519,8 +504,7 @@ async def get_status(profile: Optional[str] = None): "env_path": str(get_env_path()), "gateway_pid": gateway["gateway_pid"], "gateway_health_url": _GATEWAY_HEALTH_URL, - "gateways": topology["gateways"], - }) + "gateways": topology["gateways"]}) return status finally: @@ -540,15 +524,13 @@ async def get_system_stats(): system=_platform.system(), release=_platform.release(), version=_platform.version(), - platform_label=_platform.platform(), - ), + platform_label=_platform.platform()), "arch": _platform.machine(), "hostname": _platform.node(), "python_version": _platform.python_version(), "python_impl": _platform.python_implementation(), "hermes_version": __version__, - "cpu_count": os.cpu_count(), - } + "cpu_count": os.cpu_count()} def _optional(fill): try: @@ -579,7 +561,8 @@ async def get_system_stats(): import psutil # type: ignore vm = psutil.virtual_memory() - info["memory"] = {"total": vm.total, "available": vm.available, "used": vm.used, "percent": vm.percent} + info["memory"] = {"total": vm.total, "available": vm.available, "used": vm.used, + "percent": vm.percent} for fill in (_disk, _cpu, _uptime, _process): _optional(fill) info["psutil"] = True @@ -617,8 +600,7 @@ async def get_curator_status(): "last_run_at": state.get("last_run_at"), "min_idle_hours": _safe_call(curator, "get_min_idle_hours", None), "stale_after_days": _safe_call(curator, "get_stale_after_days", None), - "archive_after_days": _safe_call(curator, "get_archive_after_days", None), - } + "archive_after_days": _safe_call(curator, "get_archive_after_days", None)} @router.put("/api/curator/paused") @@ -751,8 +733,7 @@ def _get_portal_status_sync(): "inference_url": auth.get("inference_base_url"), "provider": str((model_cfg or {}).get("provider") or ""), "subscription_url": "https://portal.nousresearch.com/manage-subscription", - "features": features, - } + "features": features} # --------------------------------------------------------------------------- @@ -790,10 +771,7 @@ async def run_debug_share_endpoint(body: DebugShareRequest | None = None): req = body or DebugShareRequest() try: result = await asyncio.to_thread( - build_debug_share, - log_lines=max(1, min(int(req.lines), 5000)), - redact=bool(req.redact), - ) + build_debug_share, log_lines=max(1, min(int(req.lines), 5000)), redact=bool(req.redact)) except RuntimeError as exc: # Required summary-report upload failed (offline / paste service down). raise HTTPException(status_code=502, detail=f"Upload failed: {exc}") @@ -806,8 +784,7 @@ async def run_debug_share_endpoint(body: DebugShareRequest | None = None): "urls": result.urls, "failures": result.failures, "redacted": result.redacted, - "auto_delete_seconds": result.auto_delete_seconds, - } + "auto_delete_seconds": result.auto_delete_seconds} # --------------------------------------------------------------------------- @@ -821,8 +798,7 @@ async def get_logs( lines: int = 100, level: Optional[str] = None, component: Optional[str] = None, - search: Optional[str] = None, -): + search: Optional[str] = None): from hermes_cli.logs import _read_tail, LOG_FILES log_name = LOG_FILES.get(file) @@ -847,8 +823,7 @@ async def get_logs( raise HTTPException( status_code=400, detail=f"Unknown component: {component}. " - f"Available: {', '.join(sorted(COMPONENT_PREFIXES))}", - ) + f"Available: {', '.join(sorted(COMPONENT_PREFIXES))}") else: comp_prefixes = None @@ -857,8 +832,7 @@ async def get_logs( log_path, min(lines, 500) if not search else 2000, has_filters=has_filters, min_level=min_level, - component_prefixes=comp_prefixes, - ) + component_prefixes=comp_prefixes) # _read_tail doesn't support free-text search, so post-filter (case-insensitive # substring) here and trim to the requested line count afterward. if search: diff --git a/hermes_cli/web_server_profiles.py b/hermes_cli/web_server_profiles.py index 39a0d8ce85..dfd3c82204 100644 --- a/hermes_cli/web_server_profiles.py +++ b/hermes_cli/web_server_profiles.py @@ -122,8 +122,7 @@ def _fallback_profile_entry(profiles_mod, name: str, home: Path, *, is_default: "gateway_running": _safe(gateway_running, False), "description": meta("description", ""), "description_auto": meta("description_auto", False), "distribution_name": None, "distribution_version": None, "distribution_source": None, - "has_alias": False, - } + "has_alias": False} def _fallback_profile_dicts(profiles_mod) -> List[Dict[str, Any]]: @@ -281,9 +280,7 @@ _TERMINAL_BACKENDS: List[Dict[str, str]] = [ "Run commands in a Singularity/Apptainer container (HPC-friendly, rootless)."), ("modal", "Modal", "Run commands in a Modal cloud sandbox."), ("daytona", "Daytona", "Run commands in a Daytona cloud sandbox."), - ("ssh", "SSH", "Run commands on a remote host over SSH."), - ) -] + ("ssh", "SSH", "Run commands on a remote host over SSH."))] def _plugin_terminal_backend_rows() -> List[Dict[str, str]]: @@ -349,8 +346,7 @@ def _aux_usage_rows(db, cutoff: float) -> List[Dict[str, Any]]: def _merge_aux_into_by_model( - by_model: List[Dict[str, Any]], aux_rows: List[Dict[str, Any]] -) -> List[Dict[str, Any]]: + by_model: List[Dict[str, Any]], aux_rows: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """Fold aux usage rows into the sessions-derived per-model list. Aux usage lives only in session_model_usage (never in the sessions counters), so @@ -430,7 +426,6 @@ def _installed_hub_identifiers(profile: Optional[str] = None) -> dict: return { entry["identifier"]: {"name": entry.get("name"), "trust_level": entry.get("trust_level"), "scan_verdict": entry.get("scan_verdict")} - for entry in lock.list_installed() if entry.get("identifier") - } + for entry in lock.list_installed() if entry.get("identifier")} except Exception: return {} From 7e04ebda8e41b8ac92d693e6dd022fb38f11c924 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:17:14 -0700 Subject: [PATCH 3/4] refactor(web-profiles/status): AST-neutral bracket packing/hugging --- hermes_cli/web_routers/profiles.py | 21 ++++++--------------- hermes_cli/web_routers/status.py | 29 +++++++++++++---------------- hermes_cli/web_server_profiles.py | 10 ++++++---- 3 files changed, 25 insertions(+), 35 deletions(-) diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index d123daf639..375c64c36b 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -434,12 +434,9 @@ def get_profiles_sessions( # section passes source=cron — two independent lists so newest cron sessions can't # starve the recents page. filters = dict( - source=source or None, - sources=_csv_list(sources) or None, - exclude_sources=_csv_list(exclude_sources) or None, - min_message_count=max(0, min_messages), - include_archived=archived == "include", - archived_only=archived == "only") + source=source or None, sources=_csv_list(sources) or None, + exclude_sources=_csv_list(exclude_sources) or None, min_message_count=max(0, min_messages), + include_archived=archived == "include", archived_only=archived == "only") # Over-fetch per profile so the merged+sorted window is correct for the requested page. # Capped so a huge profile can't blow up the response. per_profile = min(max(limit + offset, limit), 500) @@ -470,12 +467,8 @@ def get_profiles_sessions( @sessions_router.get("/api/profiles/sessions/sidebar") @_sidebar_singleflight_cache def get_profiles_sessions_sidebar( - recents_profile: str = "all", - recents_limit: int = 20, - recents_exclude: str = None, - cron_limit: int = 50, - messaging_limit: int = 100, - messaging_exclude: str = None): + recents_profile: str = "all", recents_limit: int = 20, recents_exclude: str = None, + cron_limit: int = 50, messaging_limit: int = 100, messaging_exclude: str = None): """Batched sidebar session slices — one profile-DB open per refresh. The desktop sidebar needs three source-scoped windows per refresh: recents (local @@ -590,9 +583,7 @@ def _merge_by_id(into: Dict[str, Dict[str, Any]], entries: List[Dict[str, Any]], def _merge_profile_tree( - merged: Dict[str, Dict[str, Any]], - projects: List[Dict[str, Any]], - profile: str, + merged: Dict[str, Dict[str, Any]], projects: List[Dict[str, Any]], profile: str, preview_limit: int) -> None: """Fold one profile's projects into the shared tree, keyed by folder. diff --git a/hermes_cli/web_routers/status.py b/hermes_cli/web_routers/status.py index f38b535894..3fd15f45e5 100644 --- a/hermes_cli/web_routers/status.py +++ b/hermes_cli/web_routers/status.py @@ -319,7 +319,8 @@ def _auth_gate_status() -> Dict[str, Any]: except Exception: # Module not importable yet (early startup) — leave as []. pass - return {"auth_required": auth_required, "auth_providers": auth_providers, "auth_flows": auth_flows} + return {"auth_required": auth_required, "auth_providers": auth_providers, + "auth_flows": auth_flows} def _nous_session_validity() -> str: @@ -465,7 +466,8 @@ async def get_status(profile: Optional[str] = None): "gateway_updated_at": gateway["gateway_updated_at"], "active_agents": active_agents, "gateway_busy": derive_gateway_busy( - gateway_running=gateway_running, gateway_state=gateway_state, active_agents=active_agents), + gateway_running=gateway_running, gateway_state=gateway_state, + active_agents=active_agents), "gateway_drainable": derive_gateway_drainable( gateway_running=gateway_running, gateway_state=gateway_state), "restart_drain_timeout": restart_drain_timeout, @@ -521,9 +523,7 @@ async def get_system_stats(): info: Dict[str, Any] = { **_display_system_platform( - system=_platform.system(), - release=_platform.release(), - version=_platform.version(), + system=_platform.system(), release=_platform.release(), version=_platform.version(), platform_label=_platform.platform()), "arch": _platform.machine(), "hostname": _platform.node(), @@ -554,7 +554,8 @@ async def get_system_stats(): def _process(): proc = psutil.Process() info["process"] = {"pid": proc.pid, "rss": proc.memory_info().rss, - "create_time": int(proc.create_time()), "num_threads": proc.num_threads()} + "create_time": int(proc.create_time()), + "num_threads": proc.num_threads()} # psutil enriches the picture when present; everything below is optional. try: @@ -664,7 +665,8 @@ async def get_learning_node(id: str, profile: Optional[str] = None): async def delete_learning_node(body: LearningNodeRef): """Delete a journey node — skills are archived (restorable), memories removed.""" from agent.learning_mutations import delete_node - return await _learning_mutation(body.profile, lambda: delete_node(body.id), 400, "delete failed") + return await _learning_mutation( + body.profile, lambda: delete_node(body.id), 400, "delete failed") @router.put("/api/learning/node") @@ -794,11 +796,8 @@ async def run_debug_share_endpoint(body: DebugShareRequest | None = None): @logs_router.get("/api/logs") async def get_logs( - file: str = "agent", - lines: int = 100, - level: Optional[str] = None, - component: Optional[str] = None, - search: Optional[str] = None): + file: str = "agent", lines: int = 100, level: Optional[str] = None, + component: Optional[str] = None, search: Optional[str] = None): from hermes_cli.logs import _read_tail, LOG_FILES log_name = LOG_FILES.get(file) @@ -829,10 +828,8 @@ async def get_logs( has_filters = bool(min_level or comp_prefixes or search) result = _read_tail( - log_path, min(lines, 500) if not search else 2000, - has_filters=has_filters, - min_level=min_level, - component_prefixes=comp_prefixes) + log_path, min(lines, 500) if not search else 2000, has_filters=has_filters, + min_level=min_level, component_prefixes=comp_prefixes) # _read_tail doesn't support free-text search, so post-filter (case-insensitive # substring) here and trim to the requested line count afterward. if search: diff --git a/hermes_cli/web_server_profiles.py b/hermes_cli/web_server_profiles.py index dfd3c82204..4337feb50b 100644 --- a/hermes_cli/web_server_profiles.py +++ b/hermes_cli/web_server_profiles.py @@ -116,8 +116,8 @@ def _fallback_profile_entry(profiles_mod, name: str, home: Path, *, is_default: return _safe(lambda: profiles_mod.read_profile_meta(home).get(key, default), default) return { - "name": name, "path": str(home), "is_default": is_default, "model": model, "provider": provider, - "has_env": has_env, + "name": name, "path": str(home), "is_default": is_default, + "model": model, "provider": provider, "has_env": has_env, "skill_count": _safe(lambda: profiles_mod._count_skills(home), 0), "gateway_running": _safe(gateway_running, False), "description": meta("description", ""), "description_auto": meta("description_auto", False), @@ -422,9 +422,11 @@ def _installed_hub_identifiers(profile: Optional[str] = None) -> dict: if _is_current_profile(profile): lock = HubLockFile() else: - lock = HubLockFile(_resolve_profile_dir(profile.strip()) / "skills" / ".hub" / "lock.json") + profile_dir = _resolve_profile_dir(profile.strip()) + lock = HubLockFile(profile_dir / "skills" / ".hub" / "lock.json") return { - entry["identifier"]: {"name": entry.get("name"), "trust_level": entry.get("trust_level"), + entry["identifier"]: {"name": entry.get("name"), + "trust_level": entry.get("trust_level"), "scan_verdict": entry.get("scan_verdict")} for entry in lock.list_installed() if entry.get("identifier")} except Exception: From ba1df7dab1133ae54909a43fba3b630c53e43b51 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:49:33 -0700 Subject: [PATCH 4/4] refactor(web-profiles/status): fold db-open into read helper, data-drive sidebar slices, compact docstrings --- hermes_cli/web_routers/profiles.py | 421 +++++++++++------------------ hermes_cli/web_routers/status.py | 255 ++++++----------- hermes_cli/web_server_profiles.py | 102 +++---- 3 files changed, 276 insertions(+), 502 deletions(-) diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 375c64c36b..df5ccc48bd 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -36,10 +36,9 @@ from hermes_cli.web_server_profiles import _hermes_home_scope # Same logger the handlers used before extraction (identical logger object). _log = logging.getLogger("hermes_cli.web_server") -# Per-profile session reads report failures in the response's ``errors`` array, which the -# desktop sidebar does not surface — an empty sidebar can look healthy while nothing logs. -# Warn once per (profile, message) per process so a persistent failure is loud in -# errors.log without turning every sidebar poll into log spam. +# Per-profile session reads report failures only in the response's ``errors`` array, which +# the desktop sidebar does not surface. Warn once per (profile, message) per process so a +# persistent failure is loud in errors.log without turning every sidebar poll into spam. _profile_read_warned: set = set() @@ -76,11 +75,9 @@ _normalize_main_model_assignment = late("_normalize_main_model_assignment") def _profile_to_dict(info) -> Dict[str, Any]: attr = functools.partial(getattr, info) return { - "name": attr("name", ""), - "path": str(attr("path", "")), + "name": attr("name", ""), "path": str(attr("path", "")), "is_default": bool(attr("is_default", False)), - "model": attr("model", None), - "provider": attr("provider", None), + "model": attr("model", None), "provider": attr("provider", None), "has_env": bool(attr("has_env", False)), "skill_count": int(attr("skill_count", 0) or 0), "gateway_running": bool(attr("gateway_running", False)), @@ -100,11 +97,9 @@ def _profile_setup_command(name: str) -> str: def _write_profile_model(profile_dir: Path, provider: str, model: str) -> None: - """Write the main model assignment into ``profile_dir``'s config.yaml (HERMES_HOME-scoped, - so it lands in the target profile rather than the dashboard's active one). Clears stale - ``base_url`` / ``context_length`` the same way ``POST /api/model/set`` does.""" + """Write the main model assignment into ``profile_dir``'s config.yaml (HERMES_HOME-scoped); + clears stale ``base_url`` / ``context_length`` like ``POST /api/model/set`` does.""" from hermes_cli.web_server import load_config, save_config - with _hermes_home_scope(profile_dir): provider, model = _normalize_main_model_assignment(provider, model) cfg = load_config() @@ -113,17 +108,12 @@ def _write_profile_model(profile_dir: Path, provider: str, model: str) -> None: def _disable_unselected_skills(profile_dir: Path, keep: List[str]) -> int: - """Disable every installed skill in ``profile_dir`` not in ``keep``; returns how many - were newly disabled. - - Profiles manage activation via a *disabled* list (everything installed is active by - default). The builder's skill step has "replace" semantics: the user picks exactly which - seeded skills stay active. Hub skills are installed separately via subprocess and are - active on install. - """ + """Disable every installed skill in ``profile_dir`` not in ``keep``; returns how many were + newly disabled. Profiles manage activation via a *disabled* list (everything installed is + active by default); the builder's skill step has "replace" semantics. Hub skills are + installed separately via subprocess and are active on install.""" from hermes_cli.web_server import load_config from hermes_cli.skills_config import get_disabled_skills, save_disabled_skills - keep_set = {s.strip() for s in keep if s and s.strip()} with _hermes_home_scope(profile_dir): skills_root = profile_dir / "skills" @@ -148,9 +138,8 @@ _MISSING = object() @contextlib.contextmanager def _profile_errors(log_msg: str, *args, not_found=(FileNotFoundError,), bad_request=(ValueError,)): - """Map hermes_cli.profiles exceptions to HTTP: ``not_found`` -> 404, - ``bad_request`` -> 400 (checked in that order), anything else is logged - with ``log_msg`` and becomes a 500. ``HTTPException`` passes through.""" + """Map hermes_cli.profiles exceptions to HTTP: ``not_found`` -> 404, ``bad_request`` -> 400 + (in that order), anything else is logged with ``log_msg`` -> 500. HTTPException passes.""" try: yield except HTTPException: @@ -164,9 +153,17 @@ def _profile_errors(log_msg: str, *args, not_found=(FileNotFoundError,), raise HTTPException(status_code=500, detail=str(e)) +async def _read_off_loop(read, label: str, errors): + """``read()`` on a worker thread; ``errors`` become ``500 "Could not read