diff --git a/hermes_cli/web_server_config.py b/hermes_cli/web_server_config.py index 29368507a4..e268794f4d 100644 --- a/hermes_cli/web_server_config.py +++ b/hermes_cli/web_server_config.py @@ -1,8 +1,8 @@ """Dashboard config schema and model-assignment logic: CONFIG_SCHEMA construction, dynamic provider options, web<->config normalisation, main/aux model assignment. -Split out of ``hermes_cli.web_server``; every externally used name is re-imported -there, so ``web_server.`` keeps resolving (and monkeypatching) as before. -Helpers that tests patch on ``web_server`` are reached lazily through it. +Split out of ``hermes_cli.web_server``; every externally used name is re-imported there so +``web_server.`` keeps resolving (and monkeypatching). Helpers that tests patch on +``web_server`` are reached lazily through it. """ import logging @@ -28,16 +28,13 @@ _log = logging.getLogger("hermes_cli.web_server") # Config schema — auto-generated from DEFAULT_CONFIG # --------------------------------------------------------------------------- -# Manual overrides for fields that need select options or custom types def _memory_provider_options() -> List[str]: """Discovered memory providers for the ``memory.provider`` select. - Directory-scan only (no provider imports), so it's safe at module import - time. ``""`` (built-in only) is always first; discovery failures degrade to - the bundled defaults rather than dropping the field. The literal - ``builtin`` alias is deliberately NOT offered — built-in memory is not a - provider plugin, and ``_normalize_memory_provider_name`` already maps any - legacy ``builtin``/``built-in``/``none`` value back to ``""`` (#49513). + Directory-scan only (no provider imports), so safe at module import time. ``""`` + (built-in only) is always first; discovery failures degrade to the bundled defaults. + The literal ``builtin`` alias is deliberately NOT offered — built-in memory is not a + provider plugin; ``_normalize_memory_provider_name`` maps legacy aliases back to ``""``. """ options = [""] try: @@ -46,7 +43,6 @@ def _memory_provider_options() -> List[str]: options.extend(list_memory_provider_names()) except Exception: options.extend(["honcho"]) - # Dedupe, preserve order return list(dict.fromkeys(options)) @@ -232,16 +228,15 @@ _SCHEMA_OVERRIDES: Dict[str, Dict[str, Any]] = { }, } -# Categories with fewer fields get merged into "general" to avoid tab sprawl. +# Small categories fold into a bigger tab to avoid one-field orphan tabs. Several sources +# (models_dev, onboarding, mcp, computer_use, telemetry, plugins, doctor, runtime, session, +# nous, telegram) currently surface a single schema field each. _CATEGORY_MERGE: Dict[str, str] = { "privacy": "security", "context": "agent", "skills": "agent", "cron": "agent", "network": "agent", - # `models_dev.url` (mirror override) is the only schema-surfaced - # models_dev field — fold it in with the other network/agent plumbing - # rather than spawning a one-field orphan tab. "models_dev": "agent", "checkpoints": "agent", "approvals": "security", @@ -249,110 +244,55 @@ _CATEGORY_MERGE: Dict[str, str] = { "dashboard": "display", "code_execution": "agent", "prompt_caching": "agent", - # bot_mode holds a couple of relay tuning knobs — keep it folded into the - # agent tab rather than spawning a tiny standalone category. "bot_mode": "agent", "goals": "agent", "updates": "general", - # `onboarding.profile_build` is the only schema-surfaced onboarding field - # (`onboarding.seen` is an internal latch dict, not a user setting), so fold - # it into the agent tab rather than spawning a one-field orphan category. "onboarding": "agent", - # Only `telegram.reactions` currently lives under telegram — fold it in - # with the other messaging-platform config (discord) so it isn't an - # orphan tab of one field. "telegram": "discord", - # `mcp.auto_reload_on_config_change` is the only schema-surfaced mcp - # runtime field (server definitions live under mcp_servers, edited via - # the MCP tab) — fold it into the agent tab rather than spawning a - # one-field orphan category. "mcp": "agent", - # `computer_use.cua_telemetry` is the only schema-surfaced computer_use - # field — fold it into the agent tab rather than spawning a one-field - # orphan category. "computer_use": "agent", - # `telemetry.shared_metrics.enabled` is the only schema-surfaced telemetry - # field — fold it into security alongside the other privacy-posture toggles. "telemetry": "security", - # `plugins.hook_callback_timeout` is the only schema-surfaced plugins field - # (`enabled`/`disabled` are list allow-lists omitted from DEFAULT_CONFIG) — - # fold it into the agent tab rather than spawning a one-field orphan category. "plugins": "agent", - # `doctor.live_probe_timeout` is the only schema-surfaced doctor field — - # fold it into general rather than spawning a one-field orphan category. "doctor": "general", - # `runtime.nofile_soft_limit` (#78873) is the only schema-surfaced runtime - # field — fold it into the agent tab rather than spawning a one-field - # orphan category. "runtime": "agent", - # `session.terminal_continue` is the only schema-surfaced session field — - # fold it into general rather than spawning a one-field orphan category. "session": "general", - # `nous.keepalive_interval_seconds` is the only schema-surfaced nous field - # (Portal tokens live in auth.json) — fold it into the agent tab. "nous": "agent", } +_UI_TYPES = ((bool, "boolean"), (int, "number"), (float, "number"), (list, "list"), (dict, "object")) + + def _infer_type(value: Any) -> str: """Infer a UI field type from a Python value.""" - if isinstance(value, bool): - return "boolean" - if isinstance(value, int): - return "number" - if isinstance(value, float): - return "number" - if isinstance(value, list): - return "list" - if isinstance(value, dict): - return "object" - return "string" + return next((ui for py, ui in _UI_TYPES if isinstance(value, py)), "string") -def _build_schema_from_config( - config: Dict[str, Any], - prefix: str = "", -) -> Dict[str, Dict[str, Any]]: +def _build_schema_from_config(config: Dict[str, Any], prefix: str = "") -> Dict[str, Dict[str, Any]]: """Walk DEFAULT_CONFIG and produce a flat dot-path → field schema dict.""" schema: Dict[str, Dict[str, Any]] = {} for key, value in config.items(): full_key = f"{prefix}.{key}" if prefix else key - - # Skip internal / version keys - if full_key in {"_config_version"}: + if full_key == "_config_version": continue - - # Category is the first path component for nested keys, or "general" - # for top-level scalar fields (model, toolsets, timezone, etc.). - if prefix: - category = prefix.split(".")[0] - elif isinstance(value, dict): - category = key - else: - category = "general" - if isinstance(value, dict): - # Recurse into nested dicts schema.update(_build_schema_from_config(value, full_key)) - else: - entry: Dict[str, Any] = { - "type": _infer_type(value), - "description": full_key.replace(".", " → ").replace("_", " ").title(), - "category": category, - } - # Apply manual overrides - if full_key in _SCHEMA_OVERRIDES: - entry.update(_SCHEMA_OVERRIDES[full_key]) - # Merge small categories - entry["category"] = _CATEGORY_MERGE.get(entry["category"], entry["category"]) - schema[full_key] = entry + continue + # Category: first path component for nested keys, "general" for top-level scalars. + entry: Dict[str, Any] = { + "type": _infer_type(value), + "description": full_key.replace(".", " → ").replace("_", " ").title(), + "category": prefix.split(".")[0] if prefix else "general", + } + entry.update(_SCHEMA_OVERRIDES.get(full_key, {})) + entry["category"] = _CATEGORY_MERGE.get(entry["category"], entry["category"]) + schema[full_key] = entry return schema def _config_schema_with_virtual_fields() -> Dict[str, Dict[str, Any]]: - """DEFAULT_CONFIG schema plus the virtual fields the normalize/denormalize - cycle surfaces: ``model_context_length`` is inserted right after ``model`` - so it renders adjacent in the frontend.""" + """DEFAULT_CONFIG schema plus the virtual ``model_context_length`` field, inserted right + after ``model`` so it renders adjacent in the frontend.""" ordered: Dict[str, Dict[str, Any]] = {} for key, entry in _build_schema_from_config(DEFAULT_CONFIG).items(): ordered[key] = entry @@ -365,17 +305,12 @@ CONFIG_SCHEMA = _config_schema_with_virtual_fields() def _is_command_provider_block(value: Any) -> bool: - """Return True when *value* declares a command-type voice provider. + """True when *value* declares a command-type voice provider. - Mirrors the runtime discriminators - (``tools.tts_tool._is_command_provider_config`` / - ``tools.transcription_tools._is_command_stt_provider_config``) and the - desktop's ``isCommandProvider`` in - ``apps/desktop/src/app/settings/helpers.ts``: ``type`` is OPTIONAL and - case/space-insensitive (absent or normalizing to ``"command"``), and - ``command`` MUST be a non-empty string. Built-in blocks (which carry - ``voice``/``model`` and no ``command``) and the ``providers`` container - itself are rejected. + Mirrors the runtime discriminators (``tools.tts_tool._is_command_provider_config`` / + ``tools.transcription_tools._is_command_stt_provider_config``) and the desktop's + ``isCommandProvider``: ``type`` is OPTIONAL and case/space-insensitive (absent or + normalizing to ``"command"``); ``command`` MUST be a non-empty string. """ if not isinstance(value, dict): return False @@ -386,48 +321,21 @@ def _is_command_provider_block(value: Any) -> bool: return isinstance(command, str) and bool(command.strip()) -def _custom_provider_options( - kind: str, - builtin_names: List[str], - cfg: Dict[str, Any], -) -> List[str]: - """Return a merged provider option list without hard-coding vendor names. +def _custom_provider_options(kind: str, builtin_names: List[str], cfg: Dict[str, Any]) -> List[str]: + """Merged ``tts``/``stt`` provider options without hard-coding vendor names. - *kind* is ``"tts"`` or ``"stt"``. The result keeps the built-in display - names first (original order — NOT re-sorted), then appends: - - 1. Command-type providers declared under the canonical - ``.providers.`` location, plus the legacy top-level - ``.`` fallback — exactly the dual resolution the runtime - performs in ``_get_named_provider_config`` / - ``_get_named_stt_provider_config``. Names colliding with a RUNTIME - built-in are excluded case-insensitively (the runtime rejects a - built-in name as a command provider before any config lookup), so a - ``providers.EDGE`` command block is not offered. - 2. Plugin-registered provider names from ``agent.tts_registry`` / - ``agent.transcription_registry`` — opportunistic only: plugins - register at runtime via ``ctx.register_tts_provider()``, and this - process does not necessarily call ``discover_plugins()``, so the - registry may legitimately be empty here. (There is no static - ``provides: [tts]`` manifest convention to scan — real manifests only - carry ``provides_tools``/``provides_hooks``.) - 3. The current ``.provider`` value when not already present — a - custom name that only appears as the active provider stays - selectable (matches desktop ``enumOptionsFor``'s current-value - preservation). - - Guard semantics deliberately mirror - ``apps/desktop/src/app/settings/helpers.ts:commandProviderNames`` so the - backend schema (web dashboard) and the desktop client agree on which - names are offered. + Built-in display names first (original order), then, deduped case-insensitively: + 1. Command-type providers from canonical ``.providers.`` and the legacy + top-level ``.`` — the runtime's dual resolution order. Names colliding + with a RUNTIME built-in are excluded (the runtime rejects them before config lookup); + the runtime sets are used rather than the display shortlist, which drifts. + 2. Plugin-registered names from the tts/transcription registries — opportunistic: this + process may never call ``discover_plugins()``, so the registry may be empty. + 3. The current ``.provider`` value, so a custom active name stays selectable. + Guard semantics mirror the desktop's ``commandProviderNames`` so both surfaces agree. """ names = [str(n) for n in builtin_names] seen = {n.strip().lower() for n in names} - - # Guard against the RUNTIME built-in sets, not the display shortlist - # above: the display list drifts from the runtime sets (e.g. omits - # ``deepinfra``), and filtering on it would offer names the runtime - # would never honour as command providers. if kind == "tts": from tools.tts_tool import BUILTIN_TTS_PROVIDERS as _runtime_builtins else: @@ -445,16 +353,9 @@ def _custom_provider_options( section = cfg.get(kind) if not isinstance(section, dict): section = {} - - # Canonical nested location first, then the legacy top-level fallback — - # the same order the runtime resolves them in. - candidate_blocks: List[Any] = [] providers_map = section.get("providers") - if isinstance(providers_map, dict): - candidate_blocks.append(providers_map) - candidate_blocks.append( - {k: v for k, v in section.items() if k != "providers"} - ) + candidate_blocks: List[Any] = [providers_map] if isinstance(providers_map, dict) else [] + candidate_blocks.append({k: v for k, v in section.items() if k != "providers"}) for block in candidate_blocks: for name, value in block.items(): if ( @@ -464,9 +365,6 @@ def _custom_provider_options( ): _add(name) - # Plugin-registered providers (only populated when plugins are loaded in - # this process). Registry names can never collide with built-ins — the - # registries reject such registrations. try: if kind == "tts": from agent.tts_registry import list_providers as _list_voice_providers @@ -477,48 +375,37 @@ def _custom_provider_options( except Exception: # pragma: no cover - registry import should not break schema pass - # Current-value preservation (``cfg_get`` takes *keys*, not dotted paths). + # ``cfg_get`` takes *keys*, not dotted paths. _add(cfg_get(cfg, kind, "provider")) - return names def _memory_provider_schema_options(cfg: Dict[str, Any]) -> List[str]: - """Discovered memory providers for a per-request schema merge. - - Reuses the cheap directory scan of :func:`_memory_provider_options` and - additionally preserves the currently-configured provider, so a value - selected in config but not (yet) discoverable — e.g. a plugin removed from - disk — never silently vanishes from the dropdown. - """ + """Discovered memory providers plus the currently-configured one, so a value that is no + longer discoverable (e.g. plugin removed from disk) never vanishes from the dropdown.""" from hermes_cli.web_server import _memory_provider_options options = _memory_provider_options() - memory = cfg.get("memory") - configured = memory.get("provider") if isinstance(memory, dict) else None - current = _normalize_memory_provider_name(configured) - + current = _normalize_memory_provider_name(memory.get("provider") if isinstance(memory, dict) else None) if current and current not in options: options = [*options, current] - return options +def _schema_select_options(key: str) -> Optional[List[str]]: + entry = CONFIG_SCHEMA.get(key) + options = entry.get("options") if isinstance(entry, dict) else None + return options if isinstance(options, list) else None + + def _schema_with_dynamic_provider_options() -> Dict[str, Dict[str, Any]]: - """Return CONFIG_SCHEMA with per-request discovery-driven options merged. + """CONFIG_SCHEMA with per-request discovery-driven ``*.provider`` options merged. - Some ``*.provider`` selects have options that are discovered at runtime - (voice backends via the tts/stt registries + config.yaml command - providers; memory providers via a plugin-dir scan). The module-level - ``_SCHEMA_OVERRIDES`` freezes those lists at import time, so a provider - installed after the server started never appears. This recomputes them at - request time — reflecting the CURRENT config.yaml, the profile-scoped - config when the request carries a ``profile`` param, and mid-session - plugin installs — for every surface that reads the schema (desktop, CLI, - dashboard), with no extra frontend round-trips. - - The module-level ``CONFIG_SCHEMA`` is never mutated; entries that change - are shallow-copied onto a copied mapping. + ``_SCHEMA_OVERRIDES`` freezes option lists at import time, so a provider installed after + the server started never appears. Recomputing at request time reflects the CURRENT + (possibly profile-scoped) config.yaml and mid-session plugin installs for every surface + that reads the schema. ``CONFIG_SCHEMA`` is never mutated; changed entries are + shallow-copied onto a copied mapping. """ from hermes_cli.web_server import _plugin_terminal_backend_rows, load_config try: @@ -529,76 +416,50 @@ def _schema_with_dynamic_provider_options() -> Dict[str, Dict[str, Any]]: overlay: Dict[str, Dict[str, Any]] = {} def merge(key: str, options: List[str]) -> None: - entry = CONFIG_SCHEMA.get(key) - - if isinstance(entry, dict) and isinstance(entry.get("options"), list) and options != entry["options"]: - overlay[key] = {**entry, "options": options} + if _schema_select_options(key) is not None and options != CONFIG_SCHEMA[key]["options"]: + overlay[key] = {**CONFIG_SCHEMA[key], "options": options} for kind in ("tts", "stt"): - entry = CONFIG_SCHEMA.get(f"{kind}.provider") - existing = entry.get("options") if isinstance(entry, dict) else None - - if isinstance(existing, list): + existing = _schema_select_options(f"{kind}.provider") + if existing is not None: merge(f"{kind}.provider", _custom_provider_options(kind, list(existing), cfg)) merge("memory.provider", _memory_provider_schema_options(cfg)) - tb_entry = CONFIG_SCHEMA.get("terminal.backend") - if isinstance(tb_entry, dict) and isinstance(tb_entry.get("options"), list): + tb_options = _schema_select_options("terminal.backend") + if tb_options is not None: try: - plugin_names = sorted( - {row["name"] for row in _plugin_terminal_backend_rows()} - - set(tb_entry["options"]) - ) + plugin_names = sorted({row["name"] for row in _plugin_terminal_backend_rows()} - set(tb_options)) except Exception: plugin_names = [] if plugin_names: - merge("terminal.backend", [*tb_entry["options"], *plugin_names]) + merge("terminal.backend", [*tb_options, *plugin_names]) - if not overlay: - return CONFIG_SCHEMA - - return {**CONFIG_SCHEMA, **overlay} + return {**CONFIG_SCHEMA, **overlay} if overlay else CONFIG_SCHEMA def _normalize_main_model_assignment(provider: str, model: str) -> tuple[str, str]: """Normalize a main-slot (provider, model) pair before persisting. - The Models page has two assignment paths and only one of them was safe: + The per-card "Use as → Main model" menu can send the model's VENDOR prefix as the + provider (analytics rows with no ``billing_provider``), producing e.g. + ``provider: anthropic`` + ``default: anthropic/claude-opus-4.6`` — an aggregator slug on + the native provider, which 400s. Two repairs at this single chokepoint: - - The "Change" picker sends a real Hermes provider slug — fine. - - The per-card "Use as → Main model" menu sends ``entry.provider`` - from the analytics rows, falling back to the model's VENDOR prefix - (``modelVendor("anthropic/claude-opus-4.6") == "anthropic"``) when - the session row has no ``billing_provider`` (older sessions, NULL - rows). That wrote ``provider: anthropic`` + - ``default: anthropic/claude-opus-4.6`` to config — a vendor-prefixed - OpenRouter slug on the NATIVE Anthropic provider. New sessions then - 400 against api.anthropic.com ("model: anthropic/claude-opus-4.6 not - found") and the user reads it as "changing models does nothing". - - Two repairs, both at this single chokepoint so every caller inherits: - - 1. Vendor-name → Hermes-provider mapping: when the provider string is - not a known Hermes provider/alias (e.g. ``moonshotai``, ``x-ai`` is - known but ``poolside`` isn't) but the model is a vendor-prefixed - aggregator slug, keep the user's CURRENT aggregator if they're on - one, else fall back to openrouter. - - Named custom providers (``custom:litellm``, etc.) are excluded from - this fallback: ``_KNOWN_PROVIDER_NAMES`` only lists the bare - ``"custom"`` bucket, never a specific ``custom:`` slug, so - without this exclusion every named custom provider paired with a - slash-bearing model (e.g. ``ollama/glm-5.2`` behind a LiteLLM proxy) - looked exactly like the stray-vendor-prefix case above and got - silently reassigned to ``openrouter``. + 1. Vendor-name → Hermes-provider: when the provider is not a known provider/alias but the + model is a vendor-prefixed slug, keep the user's CURRENT aggregator if on one, else + openrouter. User-declared ``providers:``/``custom_providers:`` entries resolve first, + and durable named-custom slugs (``custom`` / ``custom:``) are excluded — + ``_KNOWN_PROVIDER_NAMES`` lists only the bare ``custom`` bucket, so without this a + LiteLLM proxy serving ``ollama/glm-5.2`` would be silently reassigned to openrouter. + Matching only that syntax (not ``startswith("custom")``) avoids swallowing + unconfigured vendors like ``customproxy``. 2. Model-format normalization for the resolved provider via - ``normalize_model_for_provider`` (e.g. ``anthropic/claude-opus-4.6`` - on native anthropic → ``claude-opus-4-6``). + ``normalize_model_for_provider`` (custom/user providers keep the model verbatim). """ from hermes_cli.web_server import load_config from hermes_cli.config import get_compatible_custom_providers - from hermes_cli.models import _KNOWN_PROVIDER_NAMES, normalize_provider + from hermes_cli.models import _AGGREGATOR_PROVIDERS, _KNOWN_PROVIDER_NAMES, normalize_provider from hermes_cli.model_normalize import normalize_model_for_provider from hermes_cli.providers import resolve_custom_provider, resolve_user_provider @@ -606,68 +467,38 @@ def _normalize_main_model_assignment(provider: str, model: str) -> tuple[str, st model_in = (model or "").strip() canonical = normalize_provider(prov_in) - # User-declared providers are real routing targets, not analytics vendor - # labels. Resolve them before the unknown-vendor fallback. ``providers:`` - # keeps its declared bare slug; ``custom_providers:`` canonicalizes both a - # bare display name and ``custom:`` to the durable custom slug. try: cfg = load_config() except Exception: cfg = {} - user_providers = cfg.get("providers") if isinstance(cfg, dict) else None - user_provider = resolve_user_provider( - prov_in, user_providers if isinstance(user_providers, dict) else {} - ) - custom_provider = resolve_custom_provider( - prov_in, - get_compatible_custom_providers(cfg) if isinstance(cfg, dict) else [], - ) + if not isinstance(cfg, dict): + cfg = {} + user_providers = cfg.get("providers") + user_provider = resolve_user_provider(prov_in, user_providers if isinstance(user_providers, dict) else {}) + custom_provider = resolve_custom_provider(prov_in, get_compatible_custom_providers(cfg)) if user_provider is not None: return user_provider.id, model_in if custom_provider is not None: return custom_provider.id, model_in - # A named custom provider that didn't resolve above (typo, config - # mismatch, entry missing from custom_providers/providers) must still - # not be treated as a stray vendor prefix -- it isn't a known Hermes - # provider/alias, but it also isn't the analytics-vendor case this - # fallback exists for. Match only the durable named-custom syntax - # (bare "custom" bucket, or "custom:" per - # ``providers.custom_provider_slug``) -- a bare ``startswith("custom")`` - # would also swallow unrelated unconfigured vendor names that merely - # happen to start with "custom" (e.g. "customproxy"). is_custom_provider_slug = canonical == "custom" or canonical.startswith("custom:") - if ( - canonical not in _KNOWN_PROVIDER_NAMES - and not is_custom_provider_slug - and "/" in model_in - ): - # Vendor prefix posing as a provider (analytics fallback). Resolve - # against the user's current provider when it's an aggregator that - # serves vendor-prefixed slugs; otherwise default to openrouter. + if canonical not in _KNOWN_PROVIDER_NAMES and not is_custom_provider_slug and "/" in model_in: try: cur_cfg = cfg.get("model", {}) cur_provider = ( - str(cur_cfg.get("provider", "") or "").strip().lower() - if isinstance(cur_cfg, dict) else "" + str(cur_cfg.get("provider", "") or "").strip().lower() if isinstance(cur_cfg, dict) else "" ) except Exception: cur_provider = "" - from hermes_cli.models import _AGGREGATOR_PROVIDERS if cur_provider and normalize_provider(cur_provider) in _AGGREGATOR_PROVIDERS: canonical = normalize_provider(cur_provider) prov_in = cur_provider else: - canonical = "openrouter" - prov_in = "openrouter" + canonical = prov_in = "openrouter" - # Custom/user-config providers keep the model verbatim — the registry - # normalizer doesn't know their namespaces. if canonical in _KNOWN_PROVIDER_NAMES and not canonical.startswith("custom"): try: - normalized_model = normalize_model_for_provider(model_in, canonical) - if normalized_model: - model_in = normalized_model + model_in = normalize_model_for_provider(model_in, canonical) or model_in except Exception: _log.debug("model normalization failed for %s/%s", prov_in, model_in, exc_info=True) @@ -679,77 +510,48 @@ def _apply_main_model_assignment( ) -> dict: """Apply a main-slot model assignment to a ``model`` config dict in place. - Sets ``provider``/``default``, then reconciles ``base_url``: + Sets ``provider``/``default``, then reconciles endpoint fields. ``base_url`` and the + endpoint key share one lifecycle: an explicit value is always persisted; an existing + value is cleared ONLY when switching to a *different* provider (it belonged to the old + endpoint); a same-provider re-pick preserves it — re-picking a model used to wipe a + user's custom host (e.g. a Xiaomi MiMo Token Plan URL) and break their keys. The + runtime resolver reads ``model.base_url`` from config and only honors it when the + configured provider matches, so preserving it here is what lets the override route. + A stale secret may live under the legacy ``api`` alias with no ``api_key``, so the + switch-clears-the-key path triggers on either field. ``context_length`` is always + dropped (the new model may have a different window). - - An explicitly supplied ``base_url`` is always persisted (covers - ``custom``/local endpoints and any provider whose key is bound to a - non-default host). - - Otherwise, a stale ``base_url`` is cleared ONLY when switching to a - *different* provider — that URL belonged to the old provider. When the - provider is unchanged and no new URL is supplied, the existing - ``base_url`` is preserved. This keeps a user's custom endpoint (e.g. a - Xiaomi MiMo Token Plan host, ``https://token-plan-*.xiaomimimo.com/v1``) - alive when they merely re-pick a model under the same provider — picking - a model previously wiped it, forcing the registry default and breaking - Token Plan keys. - - The runtime resolver reads ``model.base_url`` from config (it ignores - ``OPENAI_BASE_URL``) and only honors it when the configured provider matches - and the pool entry is on the registry default, so preserving it here is what - lets the override actually route. The hardcoded ``context_length`` override - is always dropped since the new model may have a different context window. - - Returns the same dict (coerced to a fresh dict if the input wasn't one) so - callers can assign it straight back onto the model config. + Returns the same dict (a fresh dict if the input wasn't one). """ if not isinstance(model_cfg, dict): model_cfg = {} prev_provider = str(model_cfg.get("provider") or "").strip().lower() new_provider = provider.strip().lower() + switched = new_provider != prev_provider model_cfg["provider"] = provider model_cfg["default"] = model if base_url.strip(): model_cfg["base_url"] = base_url.strip() - elif model_cfg.get("base_url") and new_provider != prev_provider: - # Switching providers: the old URL belonged to the old provider, drop - # it so the new provider's default endpoint is used. Same-provider - # re-assignment keeps the user's configured base_url intact. + elif model_cfg.get("base_url") and switched: model_cfg["base_url"] = "" - # The endpoint key follows the same lifecycle as base_url: an explicit key - # is always persisted; an existing key is dropped only when switching to a - # different provider (it belonged to the old endpoint), and preserved on a - # same-provider re-pick so re-selecting a model doesn't wipe the key. if api_key.strip(): model_cfg["api_key"] = api_key.strip() model_cfg.pop("api", None) - elif (model_cfg.get("api_key") or model_cfg.get("api")) and new_provider != prev_provider: - # A stale endpoint secret can live under the legacy ``api`` alias with - # no ``api_key`` (the resolver still reads ``model.api`` as a key), so - # the switch-clears-the-key path must trigger on either field — else the - # old endpoint's secret survives in config.yaml and contaminates a later - # custom resolution. clear_model_endpoint_credentials scrubs both. + elif (model_cfg.get("api_key") or model_cfg.get("api")) and switched: clear_model_endpoint_credentials(model_cfg, clear_api_mode=False) - if new_provider != prev_provider: + if switched: clear_model_endpoint_credentials(model_cfg, clear_api_key=False) model_cfg.pop("context_length", None) return model_cfg def _normalize_config_for_web(config: Dict[str, Any]) -> Dict[str, Any]: - """Normalize config for the web UI. - - Hermes supports ``model`` as either a bare string (``"anthropic/claude-sonnet-4"``) - or a dict (``{default: ..., provider: ..., base_url: ...}``). The schema is built - from DEFAULT_CONFIG where ``model`` is a string, but user configs often have the - dict form. Normalize to the string form so the frontend schema matches. - - Also surfaces ``model_context_length`` as a top-level field so the web UI can - display and edit it. A value of 0 means "auto-detect". - """ - config = dict(config) # shallow copy + """Flatten a dict-form ``model`` to its string form (the schema is built from + DEFAULT_CONFIG where ``model`` is a string) and surface ``model_context_length`` + as a top-level field (0 = auto-detect).""" + config = dict(config) model_val = config.get("model") if isinstance(model_val, dict): - # Extract context_length before flattening the dict ctx_len = model_val.get("context_length", 0) config["model"] = model_val.get("default", model_val.get("name", "")) config["model_context_length"] = ctx_len if isinstance(ctx_len, int) else 0 @@ -759,45 +561,26 @@ def _normalize_config_for_web(config: Dict[str, Any]) -> Dict[str, Any]: # --------------------------------------------------------------------------- -# Model assignment — pick provider+model for main slot or auxiliary slots. -# Mirrors the model.options JSON-RPC from tui_gateway but uses REST so the -# Models page (which has no chat PTY open) can drive it. +# Model assignment — main slot or auxiliary slots. Mirrors the model.options +# JSON-RPC from tui_gateway but over REST so the Models page can drive it. # --------------------------------------------------------------------------- # Canonical auxiliary task slots. Keep in sync with DEFAULT_CONFIG["auxiliary"] # in hermes_cli/config.py — listed here for deterministic ordering in the UI. _AUX_TASK_SLOTS: Tuple[str, ...] = ( - "vision", - "compression", - "skills_hub", - "approval", - "mcp", - "title_generation", - "review", - "triage_specifier", - "kanban_decomposer", - "profile_describer", - "curator", + "vision", "compression", "skills_hub", "approval", "mcp", "title_generation", "review", + "triage_specifier", "kanban_decomposer", "profile_describer", "curator", ) def _dashboard_code_skew_guard() -> Optional[str]: - """Return a clear \"restart required\" message when this process runs stale code. + """Return a "restart required" message when this process runs stale code, else None. - The dashboard and Desktop-owned ``hermes serve`` are long-lived; their - ``sys.modules`` is frozen at boot. When ``hermes update`` (or a manual - ``git pull``) replaces the checkout underneath them, a first-time lazy - import on a new code path can resolve a freshly-pulled consumer module - against a stale cached dependency -> ImportError — e.g. ``/api/model/options`` - 500 after the update added ``agent.model_metadata.is_grok_46_family`` while - the running process kept serving the pre-update module (#86207). Mirror - the gateway's ``_model_switch_skew_guard``: refuse the risky call with an - actionable, deployment-aware message instead of crashing with a cryptic - import error (#97046). - - Returns None when no drift is detectable (fresh process, or a non-git - install where the boot fingerprint could not be read — never a false - positive). + Long-lived dashboard / Desktop-owned ``hermes serve`` processes freeze ``sys.modules`` + at boot; after ``hermes update`` replaces the checkout, a first-time lazy import can + resolve a fresh consumer module against a stale cached dependency -> ImportError. + Mirrors the gateway's ``_model_switch_skew_guard``: refuse the risky call with an + actionable message. Never a false positive (non-git installs return None). """ from gateway.code_skew import detect_code_skew @@ -813,13 +596,9 @@ def _dashboard_code_skew_guard() -> Optional[str]: def _dashboard_skew_restart_hint() -> str: - """Restart advice that matches how this process is actually owned. - - The same FastAPI app backs the browser dashboard *and* Desktop-owned - ``hermes serve --isolated`` (local or SSH). Hardcoding a systemd unit - misleads macOS/launchd hosts and Desktop SSH backends, which have no - ``hermes-dashboard`` unit (#97046). - """ + """Restart advice matching how this process is owned — the same app backs the browser + dashboard and Desktop-owned ``hermes serve``; naming a systemd unit would mislead + macOS/launchd hosts and Desktop SSH backends.""" if os.environ.get("HERMES_SERVE_HEADLESS") == "1": return ( "restart the Desktop-owned backend to load the new code " @@ -831,171 +610,128 @@ def _dashboard_skew_restart_hint() -> str: ) -def _apply_model_assignment_sync( - scope: str, provider: str, model: str, task: str, base_url: str, api_key: str = "" -): - """Synchronous body of POST /api/model/set. +def _resolve_assignment_credentials(model_cfg: dict, provider: str, provider_entry: Any) -> None: + """Carry the provider's credential POINTER (``key_env`` / raw ``${VAR}``) onto ``model_cfg``. - Runs inside ``_profile_scope`` (in a worker thread) so every - load_config/save_config lands in the requested profile. Raises - HTTPException for validation errors — the async wrapper re-raises them. + ``provider_entry`` comes from ``load_config()``, which expands ``${VAR}`` to plaintext; + copying that into ``model.api_key`` would write the SECRET into config.yaml (and recreate + it on every re-apply). Prefer the raw template; fall back to the expanded value only when + the raw yaml itself stores the key as a literal (no new exposure). """ - from hermes_cli.web_server import load_config, save_config - cfg = load_config() + try: + _stored, raw_entry = find_provider_entry(read_raw_config().get("providers"), provider) + except Exception: + raw_entry = None + if not isinstance(raw_entry, dict): + raw_entry = {} + key_env = str(raw_entry.get("key_env") or "").strip() + if key_env: + model_cfg["key_env"] = key_env + model_cfg.pop("api_key", None) + elif isinstance(provider_entry, dict) and provider_entry.get("api_key"): + raw_key = str(raw_entry.get("api_key") or "").strip() + model_cfg["api_key"] = raw_key if raw_key.startswith("${") and raw_key.endswith("}") else provider_entry["api_key"] - if scope == "main": - if not provider or not model: - raise HTTPException(status_code=400, detail="provider and model required for main") - provider, model = _normalize_main_model_assignment(provider, model) - providers_cfg = cfg.get("providers") - provider_entry = providers_cfg.get(provider) if isinstance(providers_cfg, dict) else None - if not base_url and isinstance(provider_entry, dict) and provider_entry.get("base_url"): - base_url = str(provider_entry.get("base_url") or "").strip() - model_cfg = _apply_main_model_assignment( - cfg.get("model", {}), provider, model, base_url, api_key + +def _apply_nous_gateway_defaults(cfg: dict) -> list: + """Mirror the CLI's post-model-selection behaviour when switching main to Nous: route + *unconfigured* tools through the Nous Tool Gateway. Purely additive — tools with a direct + key or explicit backend are skipped. Failures never block saving the assignment.""" + try: + from hermes_cli.nous_subscription import apply_nous_managed_defaults + from hermes_cli.tools_config import _get_platform_tools + + enabled = _get_platform_tools(cfg, "cli", include_default_mcp_servers=False) + return sorted(apply_nous_managed_defaults(cfg, enabled_toolsets=enabled, force_fresh=True)) + except Exception: + _log.debug("apply_nous_managed_defaults skipped", exc_info=True) + return [] + + +def _register_custom_endpoint(base_url: str, api_key: str, model: str) -> None: + """Register a named ``custom_providers`` entry for a custom/local endpoint (mirrors the + ``hermes model`` custom flow) so the picker gets a proper ready row instead of a "needs + setup" dead-end. Dedups by base_url; never blocks the already-persisted assignment.""" + try: + from hermes_cli.main import _auto_provider_name, _save_custom_provider + + _save_custom_provider(base_url, api_key, model, name=_auto_provider_name(base_url)) + except Exception: + _log.debug("custom_providers registration skipped", exc_info=True) + + +def _stale_aux_pins(cfg: dict, new_provider: str) -> list: + """Aux slots still pinned to a *different* provider than the new main one. + + Switching main never touches aux pins (independent, sticky per-task overrides) — a user + leaving a now-unpaid provider keeps paying 402s on background calls until they reset + them. We never auto-clear (pinning aux is legitimate) but report them so the UI can + offer a "reset to main" nudge. + """ + stale_aux: list[dict] = [] + aux_cfg = cfg.get("auxiliary", {}) + if not isinstance(aux_cfg, dict): + return stale_aux + for slot in _AUX_TASK_SLOTS: + slot_cfg = aux_cfg.get(slot) + if not isinstance(slot_cfg, dict): + continue + slot_provider = str(slot_cfg.get("provider", "") or "").strip() + if slot_provider and slot_provider.lower() not in {"auto", ""} and slot_provider.lower() != new_provider: + stale_aux.append({ + "task": slot, "provider": slot_provider, "model": str(slot_cfg.get("model", "") or ""), + }) + return stale_aux + + +def _cron_model_impact(cfg: dict, provider: str, model: str) -> Any: + from hermes_cli.web_server import load_config + try: + effective_config = load_config() + effective_provider, effective_model = resolve_cron_model_drift_defaults(effective_config) + return build_cron_model_impact( + current_provider=effective_provider or provider, + current_model=effective_model or model, + config=effective_config, ) - _raw_assign_entry = None - try: - _stored, _raw_assign_entry = find_provider_entry( - read_raw_config().get("providers"), provider - ) - except Exception: - _raw_assign_entry = None - _assign_key_env = ( - str(_raw_assign_entry.get("key_env") or "").strip() - if isinstance(_raw_assign_entry, dict) - else "" - ) - if _assign_key_env: - # #88990: carry the credential POINTER, never a resolved secret. - model_cfg["key_env"] = _assign_key_env - model_cfg.pop("api_key", None) - elif isinstance(provider_entry, dict) and provider_entry.get("api_key"): - # #88990: provider_entry comes from load_config(), which expands - # ${VAR} env refs to plaintext. Copying that resolved value into - # model.api_key writes the SECRET into config.yaml (and recreates - # it on every re-apply, even after the user deletes it by hand). - # Prefer the raw ${VAR} template; only fall back to the expanded - # value when the raw yaml itself stores the key as a literal (no - # new exposure in that case). - _raw_key = ( - str(_raw_assign_entry.get("api_key") or "").strip() - if isinstance(_raw_assign_entry, dict) - else "" - ) - if _raw_key.startswith("${") and _raw_key.endswith("}"): - model_cfg["api_key"] = _raw_key - else: - model_cfg["api_key"] = provider_entry["api_key"] - cfg["model"] = model_cfg + except Exception: + _log.debug("cron model impact inspection failed", exc_info=True) + return build_cron_model_impact(config=cfg, jobs={}) - # When switching the main provider to Nous, mirror the CLI's - # post-model-selection behaviour (hermes_cli/main.py - # prompt_enable_tool_gateway / tools_config apply_nous_managed_defaults): - # auto-route any *unconfigured* tools through the Nous Tool Gateway. - # This is purely additive — apply_nous_managed_defaults skips every - # tool where the user already has a direct key (FIRECRAWL_API_KEY, - # FAL_KEY, etc.) or an explicit backend/provider in config, so it - # never overwrites a user's own setup. GUI users thus land on the - # gateway the same way CLI users do, without a separate prompt. - gateway_tools: list[str] = [] - if provider.strip().lower() == "nous": - try: - from hermes_cli.nous_subscription import apply_nous_managed_defaults - from hermes_cli.tools_config import _get_platform_tools - enabled = _get_platform_tools( - cfg, "cli", include_default_mcp_servers=False - ) - changed = apply_nous_managed_defaults( - cfg, - enabled_toolsets=enabled, - force_fresh=True, - ) - gateway_tools = sorted(changed) - except Exception: - # Portal lookup hiccups / non-subscriber / non-nous gating - # must never block saving the model assignment. - _log.debug("apply_nous_managed_defaults skipped", exc_info=True) +def _apply_main_assignment_sync(cfg: dict, provider: str, model: str, base_url: str, api_key: str) -> dict: + from hermes_cli.web_server import save_config + if not provider or not model: + raise HTTPException(status_code=400, detail="provider and model required for main") + provider, model = _normalize_main_model_assignment(provider, model) + providers_cfg = cfg.get("providers") + provider_entry = providers_cfg.get(provider) if isinstance(providers_cfg, dict) else None + if not base_url and isinstance(provider_entry, dict) and provider_entry.get("base_url"): + base_url = str(provider_entry.get("base_url") or "").strip() + model_cfg = _apply_main_model_assignment(cfg.get("model", {}), provider, model, base_url, api_key) + _resolve_assignment_credentials(model_cfg, provider, provider_entry) + cfg["model"] = model_cfg - save_config(cfg) + new_provider = provider.strip().lower() + gateway_tools = _apply_nous_gateway_defaults(cfg) if new_provider == "nous" else [] + save_config(cfg) + if new_provider in {"custom", "local"} and base_url: + _register_custom_endpoint(base_url, api_key, model) - # Register a named ``custom_providers`` entry for a custom/local - # endpoint, mirroring the ``hermes model`` custom flow - # (_save_custom_provider). Without this the endpoint only lives in - # ``model.*`` and the picker has no proper ready row for it — the - # GUI then surfaces a "needs setup" dead-end on the bare ``custom`` - # provider. Dedups by base_url, so re-saving is idempotent. - if provider.strip().lower() in {"custom", "local"} and base_url: - try: - from hermes_cli.main import _auto_provider_name, _save_custom_provider + return { + "ok": True, + "scope": "main", + "provider": provider, + "model": model, + "base_url": model_cfg.get("base_url", ""), + "gateway_tools": gateway_tools, + "stale_aux": _stale_aux_pins(cfg, new_provider), + "cron_model_impact": _cron_model_impact(cfg, provider, model), + } - _save_custom_provider( - base_url, - api_key, - model, - name=_auto_provider_name(base_url), - ) - except Exception: - # Never block the assignment on the bookkeeping write — - # model.* is already persisted and routable. - _log.debug("custom_providers registration skipped", exc_info=True) - # Surface auxiliary slots still pinned to a *different* provider than - # the new main one. Switching the main model does NOT touch aux pins - # (they're independent, sticky per-task overrides — see - # auxiliary_client._resolve_auto). A user who switches main away from - # a now-unpaid provider (e.g. nous with $0 balance) keeps paying 402s - # on every background aux call until they reset those pins. We never - # auto-clear them — pinning aux to a cheaper/different model is a - # legitimate config — but we tell the caller so the UI can offer a - # "reset to main" nudge instead of silently burning credits. - new_provider = provider.strip().lower() - stale_aux: list[dict] = [] - aux_cfg = cfg.get("auxiliary", {}) - if isinstance(aux_cfg, dict): - for slot in _AUX_TASK_SLOTS: - slot_cfg = aux_cfg.get(slot) - if not isinstance(slot_cfg, dict): - continue - slot_provider = str(slot_cfg.get("provider", "") or "").strip() - if ( - slot_provider - and slot_provider.lower() not in {"auto", ""} - and slot_provider.lower() != new_provider - ): - stale_aux.append({ - "task": slot, - "provider": slot_provider, - "model": str(slot_cfg.get("model", "") or ""), - }) - - try: - effective_config = load_config() - effective_provider, effective_model = resolve_cron_model_drift_defaults( - effective_config - ) - cron_model_impact = build_cron_model_impact( - current_provider=effective_provider or provider, - current_model=effective_model or model, - config=effective_config, - ) - except Exception: - _log.debug("cron model impact inspection failed", exc_info=True) - cron_model_impact = build_cron_model_impact(config=cfg, jobs={}) - - return { - "ok": True, - "scope": "main", - "provider": provider, - "model": model, - "base_url": model_cfg.get("base_url", ""), - "gateway_tools": gateway_tools, - "stale_aux": stale_aux, - "cron_model_impact": cron_model_impact, - } - - # scope == "auxiliary" +def _apply_aux_assignment_sync(cfg: dict, provider: str, model: str, task: str, base_url: str, api_key: str) -> dict: + from hermes_cli.web_server import save_config aux = cfg.get("auxiliary") if not isinstance(aux, dict): aux = {} @@ -1019,6 +755,7 @@ def _apply_model_assignment_sync( raise HTTPException(status_code=400, detail="provider required for auxiliary") targets = [task] if task else list(_AUX_TASK_SLOTS) + new_provider = provider.strip().lower() for slot in targets: if slot not in _AUX_TASK_SLOTS: raise HTTPException(status_code=400, detail=f"unknown auxiliary task: {slot}") @@ -1026,18 +763,13 @@ def _apply_model_assignment_sync( if not isinstance(slot_cfg, dict): slot_cfg = {} prev_provider = str(slot_cfg.get("provider") or "").strip().lower() - new_provider = provider.strip().lower() slot_cfg["provider"] = provider slot_cfg["model"] = model if base_url: - # Sibling of the main-slot endpoint handling (#65254): an aux - # assignment for a custom/local endpoint must carry its own - # base_url, or the slot silently rebinds to whatever - # model.base_url happens to hold — and breaks entirely once the - # main slot switches away and clears it. The auxiliary resolver - # already reads auxiliary..base_url/api_key - # (_resolve_task_provider_model), so persisting them here is - # what actually wires the endpoint in. + # Sibling of the main-slot endpoint handling: an aux assignment for a custom/local + # endpoint must carry its own base_url/api_key (the auxiliary resolver reads + # auxiliary..base_url/api_key), or it silently rebinds to model.base_url and + # breaks once the main slot switches away. slot_cfg["base_url"] = base_url if api_key: slot_cfg["api_key"] = api_key @@ -1048,41 +780,38 @@ def _apply_model_assignment_sync( cfg["auxiliary"] = aux save_config(cfg) - return { - "ok": True, - "scope": "auxiliary", - "tasks": targets, - "provider": provider, - "model": model, - } + return {"ok": True, "scope": "auxiliary", "tasks": targets, "provider": provider, "model": model} + + +def _apply_model_assignment_sync( + scope: str, provider: str, model: str, task: str, base_url: str, api_key: str = "" +): + """Synchronous body of POST /api/model/set. + + Runs inside ``_profile_scope`` (worker thread) so every load_config/save_config lands in + the requested profile. Raises HTTPException for validation errors. + """ + from hermes_cli.web_server import load_config + cfg = load_config() + if scope == "main": + return _apply_main_assignment_sync(cfg, provider, model, base_url, api_key) + return _apply_aux_assignment_sync(cfg, provider, model, task, base_url, api_key) def _infer_provider_on_model_change(model_val: str, prev_provider: str) -> tuple[str, str]: - """Infer which provider serves ``model_val`` when the flat Config-page Model - field changes, given the previously-saved ``prev_provider``. + """Infer which provider serves ``model_val`` when the flat Config-page Model field changes. - Returns ``(provider, model)``; ``provider`` is empty when no switch is - warranted (leave the existing provider untouched). Two signals, in order: - - 1. Curated-catalog detection (``detect_provider_for_model``) — handles the - ~28 OpenRouter-curated models and direct provider-static catalogs. - 2. Vendor-slug heuristic — a ``vendor/model`` slug cannot belong to a - single-model / non-aggregator provider (e.g. ``ollama-local``). When the - current provider is not an aggregator that serves vendor-prefixed slugs, - route to an aggregator. ``_normalize_main_model_assignment`` (called by - the caller) keeps the user's current aggregator when they're already on - one, else falls back to openrouter — the same chokepoint logic as - ``POST /api/model/set``. + Returns ``(provider, model)``; ``provider`` is empty when no switch is warranted. Signals, + in order: curated-catalog detection (``detect_provider_for_model``), then the vendor-slug + heuristic — a ``vendor/model`` slug cannot belong to a non-aggregator provider (e.g. + ``ollama-local``), so return the sentinel ``"openrouter"``; the caller's + ``_normalize_main_model_assignment`` resolves the real aggregator (keeps the current one). """ name = (model_val or "").strip() if not name: return "", name try: - from hermes_cli.models import ( - _AGGREGATOR_PROVIDERS, - detect_provider_for_model, - normalize_provider, - ) + from hermes_cli.models import _AGGREGATOR_PROVIDERS, detect_provider_for_model, normalize_provider except Exception: return "", name @@ -1093,9 +822,6 @@ def _infer_provider_on_model_change(model_val: str, prev_provider: str) -> tuple if detected: return detected[0], detected[1] - # Vendor-prefixed slug under a non-aggregator provider → reassign. Use a - # sentinel "openrouter" here; _normalize_main_model_assignment resolves the - # real aggregator (keeps a current aggregator, else openrouter). if "/" in name: try: cur_is_aggregator = normalize_provider(prev_provider) in _AGGREGATOR_PROVIDERS @@ -1103,35 +829,26 @@ def _infer_provider_on_model_change(model_val: str, prev_provider: str) -> tuple cur_is_aggregator = False if not cur_is_aggregator: return "openrouter", name - return "", name def _denormalize_config_from_web(config: Dict[str, Any]) -> Dict[str, Any]: - """Reverse _normalize_config_for_web before saving. + """Reverse ``_normalize_config_for_web`` before saving. - Reconstructs ``model`` as a dict by reading the current on-disk config - to recover model subkeys (provider, base_url, api_mode, etc.) that were - stripped from the GET response. The frontend only sees model as a flat - string; the rest is preserved transparently. + Reconstructs ``model`` as a dict from the on-disk config to recover subkeys (provider, + base_url, api_mode, ...) the GET response stripped. When the model name actually changed, + re-detects the serving provider and routes through the assignment chokepoints (a user + picking an OpenRouter model while on ``ollama-local`` would otherwise keep the stale + provider and 404); saving unrelated fields never overwrites an explicit provider. - Also handles ``model_context_length`` — writes it back into the model dict - as ``context_length``. A value of 0 means "auto-detect" (omitted from the - dict so get_model_context_length() uses its normal resolution). ``config`` - may be a partial update (e.g. the Settings autosave diff) that omits - ``model_context_length`` entirely when the user didn't touch it — that - must leave the on-disk override untouched, not get treated the same as an - explicit 0 and cleared. + ``model_context_length`` is written back as ``context_length`` (0 = auto-detect, key + removed). A partial update (Settings autosave diff) that OMITS the key means "unchanged" + and must leave the on-disk override alone — not be treated as an explicit 0. """ from hermes_cli.web_server import load_config config = dict(config) - # Remove any _model_meta that might have leaked in (shouldn't happen - # with the stripped GET response, but be defensive) config.pop("_model_meta", None) - # Extract and remove model_context_length before processing model, but - # remember whether it was actually present: a partial update omitting the - # key means "unchanged", which is different from an explicit 0. ctx_sent = "model_context_length" in config ctx_override = config.pop("model_context_length", 0) if not isinstance(ctx_override, int): @@ -1141,61 +858,37 @@ def _denormalize_config_from_web(config: Dict[str, Any]) -> Dict[str, Any]: ctx_override = 0 model_val = config.get("model") - if (isinstance(model_val, str) and model_val) or ctx_sent: - # Read the current disk config to recover model subkeys - try: - disk_config = load_config() - disk_model = disk_config.get("model") - if isinstance(disk_model, dict): - if isinstance(model_val, str) and model_val: - prev_default = str(disk_model.get("default") or "").strip() - prev_provider = str(disk_model.get("provider") or "").strip() - # When the model name actually changed, re-detect which - # provider serves it. The Config-page Model field is a flat - # string with no provider info, so without this a user who - # picks an OpenRouter model while their default provider is - # ollama-local keeps the stale provider and 404s. Only fires - # on a real model change so saving unrelated config fields - # never overwrites an explicit provider. - if model_val != prev_default and prev_provider: - new_provider, resolved_model = _infer_provider_on_model_change( - model_val, prev_provider - ) - if new_provider and new_provider.strip().lower() != prev_provider.lower(): - # Route through the canonical assignment chokepoints so - # the model is normalized for the new provider and stale - # base_url/api_mode/api_key are cleared on the switch - # (and preserved on a same-provider re-pick). - norm_provider, norm_model = _normalize_main_model_assignment( - new_provider, resolved_model - ) - disk_model = _apply_main_model_assignment( - disk_model, norm_provider, norm_model - ) - model_val = norm_model - # Preserve all subkeys, update default with the new value - disk_model["default"] = model_val - # Write context_length into the model dict (0 = remove/auto), - # but only when the payload actually carried the key. - if ctx_sent: - if ctx_override > 0: - disk_model["context_length"] = ctx_override - else: - disk_model.pop("context_length", None) - config["model"] = disk_model - # Model was previously a bare string (or absent) — upgrade to a - # dict if the user is setting a context_length override. - elif ctx_sent and ctx_override > 0: - if isinstance(model_val, str) and model_val: - default = model_val - elif isinstance(disk_model, str) and disk_model: - default = disk_model + has_model = isinstance(model_val, str) and bool(model_val) + if not (has_model or ctx_sent): + return config + try: + disk_model = load_config().get("model") + if isinstance(disk_model, dict): + if has_model: + prev_default = str(disk_model.get("default") or "").strip() + prev_provider = str(disk_model.get("provider") or "").strip() + if model_val != prev_default and prev_provider: + new_provider, resolved_model = _infer_provider_on_model_change(model_val, prev_provider) + if new_provider and new_provider.strip().lower() != prev_provider.lower(): + norm_provider, norm_model = _normalize_main_model_assignment(new_provider, resolved_model) + disk_model = _apply_main_model_assignment(disk_model, norm_provider, norm_model) + model_val = norm_model + disk_model["default"] = model_val + if ctx_sent: + if ctx_override > 0: + disk_model["context_length"] = ctx_override else: - default = "" - config["model"] = { - "default": default, - "context_length": ctx_override, - } - except Exception: - pass # can't read disk config — just use the string form + disk_model.pop("context_length", None) + config["model"] = disk_model + elif ctx_sent and ctx_override > 0: + # Model was a bare string (or absent) — upgrade to a dict for the override. + if has_model: + default = model_val + elif isinstance(disk_model, str) and disk_model: + default = disk_model + else: + default = "" + config["model"] = {"default": default, "context_length": ctx_override} + except Exception: + pass # can't read disk config — just use the string form return config diff --git a/hermes_cli/web_server_dashboard.py b/hermes_cli/web_server_dashboard.py index b05a82c5bd..a862e723c3 100644 --- a/hermes_cli/web_server_dashboard.py +++ b/hermes_cli/web_server_dashboard.py @@ -1,8 +1,8 @@ """Dashboard UI assets: SPA mount, theme normalisation/bootstrap CSS, dashboard-plugin discovery and the plugins-hub merge. -Split out of ``hermes_cli.web_server``; every externally used name is re-imported -there, so ``web_server.`` keeps resolving (and monkeypatching) as before. -Helpers that tests patch on ``web_server`` are reached lazily through it. +Split out of ``hermes_cli.web_server``; every externally used name is re-imported there so +``web_server.`` keeps resolving (and monkeypatching). Helpers that tests patch on +``web_server`` are reached lazily through it. """ import logging @@ -26,74 +26,49 @@ _log = logging.getLogger("hermes_cli.web_server") def _normalise_prefix(raw: Optional[str]) -> str: - """Normalise an X-Forwarded-Prefix header value. - - Thin re-export of :func:`hermes_cli.dashboard_auth.prefix.normalise_prefix` - — the single source of truth lives in the dashboard_auth package so - the gate middleware, the OAuth routes, the cookie helpers, and the - SPA mount all agree on validation rules. - """ + """Normalise an X-Forwarded-Prefix header value (single source of truth lives in + ``hermes_cli.dashboard_auth.prefix`` so gate, OAuth, cookies and SPA mount agree).""" from hermes_cli.dashboard_auth.prefix import normalise_prefix return normalise_prefix(raw) +def _layer_hex(palette: Dict[str, Any], key: str, default: str) -> str: + layer = palette.get(key) or {} + return layer.get("hex", default) if isinstance(layer, dict) else default + + def _render_active_theme_bootstrap_css() -> str: - """Critical-CSS shim for the active user theme. + """Critical-CSS ```` escape — current values are well-known - # hex/font strings, but this keeps the helper safe if it is - # later extended to ship user-authored CSS literals. - def _esc(s: str) -> str: + + def _esc(s: str) -> str: # defensive ```` escape return str(s).replace("' ":root{" - f"--background-base:{_esc(bg_hex)};" - f"--midground-base:{_esc(mg_hex)};" + f"--background-base:{_esc(_layer_hex(palette, 'background', '#0a0a0a'))};" + f"--midground-base:{_esc(_layer_hex(palette, 'midground', '#e5e5e5'))};" f"--theme-font-sans:{_esc(font_sans)};" f"--theme-base-size:{_esc(base_size)};" "}" @@ -109,194 +84,107 @@ def _render_active_theme_bootstrap_css() -> str: return "" -# Hashed bundle assets (``/assets/-.``) are immutable -# by construction: any content change produces a new filename, and the entry -# point (index.html) is served ``no-store`` so it always references the -# current hashes. A year-long immutable cache lets browsers skip even the -# revalidation round-trip on every dashboard load. +# Hashed bundle assets are immutable by construction (content hash in the filename; index.html +# is served ``no-store`` and always references the current hashes). _IMMUTABLE_ASSET_CACHE_CONTROL = "public, max-age=31536000, immutable" +_NO_STORE = {"Cache-Control": "no-store, no-cache, must-revalidate"} +_HEADLESS_MSG = ( + "Headless backend (hermes serve): web UI disabled — use " + "`hermes dashboard` for the browser UI." +) def mount_spa(application: FastAPI): - """Mount the built SPA. Falls back to index.html for client-side routing. + """Mount the built SPA; unmatched paths fall back to index.html for client-side routing. - The session token is injected into index.html via a ``" - "Headless backend (hermes serve): web UI disabled — use " - "`hermes dashboard` for the browser UI." - "", - headers={ - "Cache-Control": "no-store, no-cache, must-revalidate" - }, + f"{_HEADLESS_MSG}", + headers=_NO_STORE, ) - return JSONResponse({"error": _msg}, status_code=404) + return JSONResponse({"error": _HEADLESS_MSG}, status_code=404) return - # A missing WEB_DIST is deliberately NOT a mount-time terminal state - # (#82614): a long-lived `hermes dashboard --skip-build` process that - # survives a `git pull` (or starts before the first build) used to - # install a permanent no_frontend catch-all here and could never - # recover — every route answered 404 "Frontend not built" until the - # process was restarted, even after `npm run build` completed. The SPA - # routes below all cope with a missing dist per-request (`_serve_index` - # returns the same 404 JSON when index.html is unreadable; the asset - # mounts use check_dir=False and 404 on missing files), so mounting - # them unconditionally makes the dashboard recover the moment a build - # appears on disk — no restart needed. - - _index_path = WEB_DIST / "index.html" - def _serve_index(prefix: str = ""): - """Return index.html with the session token + base-path injected. + """index.html with the session token + base-path injected. - ``prefix`` is the normalised ``X-Forwarded-Prefix`` (e.g. ``/hermes``) - or empty string when served at root. - - When the OAuth auth gate is active (``app.state.auth_required``), - the legacy ``_SESSION_TOKEN`` is NOT injected — the SPA reads - identity from ``/api/auth/me`` over cookie auth instead. The - ``__HERMES_AUTH_REQUIRED__`` flag lets the SPA pick the right - auth scheme for /api/pty and /api/ws (ticket vs token). + When the OAuth auth gate is active (``app.state.auth_required``), the legacy + ``_SESSION_TOKEN`` is NOT injected — the SPA reads identity from ``/api/auth/me`` over + cookie auth; ``__HERMES_AUTH_REQUIRED__`` tells it which scheme to use for /api/pty + and /api/ws (ticket vs token). """ try: - html = _index_path.read_text(encoding="utf-8") + html = (WEB_DIST / "index.html").read_text(encoding="utf-8") except OSError: - # The dist dir existed at mount time but index.html is missing or - # unreadable now (partial build, wiped dist, permissions). Without - # this guard every request raises FileNotFoundError (500). Return - # the same JSON 404 payload mount_spa uses for a fully-missing - # dist so clients get a clear, consistent signal. - return JSONResponse( - {"error": "Frontend not built. Run: cd web && npm run build"}, - status_code=404, - ) + # Partial build / wiped dist / permissions: same JSON 404 as a fully-missing dist. + return JSONResponse({"error": "Frontend not built. Run: cd web && npm run build"}, status_code=404) chat_js = "true" if _DASHBOARD_EMBEDDED_CHAT_ENABLED else "false" gated = bool(getattr(app.state, "auth_required", False)) - gated_js = "true" if gated else "false" - if gated: - bootstrap_script = ( - f"" - ) - else: - bootstrap_script = ( - f'" - ) + token_js = "" if gated else f'window.__HERMES_SESSION_TOKEN__="{_SESSION_TOKEN}";' + bootstrap_script = ( + f"" + ) if prefix: - # Rewrite absolute asset URLs baked into the Vite build so the - # browser fetches them through the same proxy prefix. - html = html.replace('href="/assets/', f'href="{prefix}/assets/') - html = html.replace('src="/assets/', f'src="{prefix}/assets/') - html = html.replace('href="/favicon.ico"', f'href="{prefix}/favicon.ico"') - html = html.replace('href="/fonts/', f'href="{prefix}/fonts/') - html = html.replace('href="/ds-assets/', f'href="{prefix}/ds-assets/') - html = html.replace('src="/ds-assets/', f'src="{prefix}/ds-assets/') - # Theme flash mitigation: when the active theme is a user theme - # (``HERMES_HOME/dashboard-themes/.yaml``), inject a minimal - # critical-CSS block so the first paint uses the target palette. - # Without this the SPA paints the default Hermes Teal canvas, then - # ``ThemeProvider`` flips the CSS variables once - # ``/api/dashboard/themes`` resolves. Built-in themes are already - # in the bundle's ``presets.ts`` so no shim is needed for them. + # Rewrite absolute asset URLs baked into the Vite build to go through the proxy. + for attr in ('href="/assets/', 'src="/assets/', 'href="/favicon.ico"', 'href="/fonts/', + 'href="/ds-assets/', 'src="/ds-assets/'): + html = html.replace(attr, attr.replace('"/', f'"{prefix}/', 1)) theme_bootstrap = _render_active_theme_bootstrap_css() if theme_bootstrap: html = html.replace("", f"{theme_bootstrap}", 1) html = html.replace("", f"{bootstrap_script}", 1) - return HTMLResponse( - html, - headers={"Cache-Control": "no-store, no-cache, must-revalidate"}, - ) + return HTMLResponse(html, headers=_NO_STORE) - # When served behind a path-prefix proxy, the built CSS contains - # absolute ``url(/fonts/...)`` and ``url(/ds-assets/...)`` references. - # Browsers resolve those against the document origin, which means - # under ``/hermes`` they'd hit ``mission-control.tilos.com/fonts/...`` - # (the MC Pages app), not the Hermes backend. Intercept CSS asset - # requests BEFORE the StaticFiles mount and rewrite the absolute paths - # when a prefix is in play. + # Built CSS contains absolute ``url(/fonts/...)`` / ``url(/ds-assets/...)`` references that + # browsers resolve against the document origin — wrong under a proxy prefix. Intercept CSS + # BEFORE the StaticFiles mount and rewrite when a prefix is in play. @application.get("/assets/{filename}.css") async def serve_css(filename: str, request: Request): css_path = WEB_DIST / "assets" / f"{filename}.css" - if not css_path.is_file() or not css_path.resolve().is_relative_to( - WEB_DIST.resolve() - ): + if not css_path.is_file() or not css_path.resolve().is_relative_to(WEB_DIST.resolve()): return JSONResponse({"error": "not found"}, status_code=404) prefix = _normalise_prefix(request.headers.get("x-forwarded-prefix")) css = css_path.read_text(encoding="utf-8") if prefix: for asset_dir in ("/fonts/", "/fonts-terminal/", "/ds-assets/", "/assets/"): - css = css.replace(f"url({asset_dir}", f"url({prefix}{asset_dir}") - css = css.replace(f"url(\"{asset_dir}", f"url(\"{prefix}{asset_dir}") - css = css.replace(f"url('{asset_dir}", f"url('{prefix}{asset_dir}") + for quote in ("", '"', "'"): + css = css.replace(f"url({quote}{asset_dir}", f"url({quote}{prefix}{asset_dir}") return Response( - content=css, - media_type="text/css", - headers={"Cache-Control": _IMMUTABLE_ASSET_CACHE_CONTROL}, + content=css, media_type="text/css", headers={"Cache-Control": _IMMUTABLE_ASSET_CACHE_CONTROL} ) class _ImmutableAssetFiles(StaticFiles): - """StaticFiles that marks hashed bundle assets immutable. - - Everything under ``/assets/`` carries a Vite content hash in its - filename, so a given URL's bytes can never change — a rebuild - produces a NEW filename referenced by a fresh (``no-store``) - index.html. Without this header every dashboard load re-validated - each chunk; with it the browser serves reloads straight from its - HTTP cache. - """ + """StaticFiles that marks hashed bundle assets immutable so reloads skip revalidation.""" async def get_response(self, path: str, scope): response = await super().get_response(path, scope) @@ -304,29 +192,18 @@ def mount_spa(application: FastAPI): response.headers["Cache-Control"] = _IMMUTABLE_ASSET_CACHE_CONTROL return response + # check_dir=False: the dist may not exist yet; StaticFiles 404s per-request until it does. application.mount( - "/assets", - # check_dir=False: the dist (and its assets/ dir) may not exist yet — - # the whole point of the dynamic recheck (#82614). StaticFiles then - # 404s per-request until a build appears instead of raising at mount. - _ImmutableAssetFiles(directory=WEB_DIST / "assets", check_dir=False), - name="assets", + "/assets", _ImmutableAssetFiles(directory=WEB_DIST / "assets", check_dir=False), name="assets" ) @application.get("/{full_path:path}") async def serve_spa(full_path: str, request: Request): prefix = _normalise_prefix(request.headers.get("x-forwarded-prefix")) - # An unmatched /api/* path is a missing/renamed endpoint, NOT a - # client-side route. Falling through to index.html here returns - # `` with status 200, which makes JSON clients (the - # desktop app's fetchJson, dashboard fetch wrappers) blow up with an - # opaque `SyntaxError: Unexpected token '<'`. Return a real 404 JSON - # so the caller sees a clear "no such endpoint" instead. + # An unmatched /api/* path is a missing endpoint, not a client-side route: return a + # real 404 JSON instead of index.html (which breaks JSON clients with a SyntaxError). if full_path == "api" or full_path.startswith("api/"): - return JSONResponse( - {"detail": f"No such API endpoint: /{full_path}"}, - status_code=404, - ) + return JSONResponse({"detail": f"No such API endpoint: /{full_path}"}, status_code=404) file_path = WEB_DIST / full_path # Prevent path traversal via url-encoded sequences (%2e%2e/) if ( @@ -340,11 +217,10 @@ def mount_spa(application: FastAPI): # --------------------------------------------------------------------------- -# Dashboard theme endpoints +# Dashboard themes # --------------------------------------------------------------------------- -# Built-in dashboard themes — label + description only. The actual color -# definitions live in the frontend (web/src/themes/presets.ts). +# Built-in themes — label + description only; colors live in web/src/themes/presets.ts. _BUILTIN_DASHBOARD_THEMES = [ {"name": "default", "label": "Hermes Teal", "description": "Classic dark teal — the canonical Hermes look"}, {"name": "default-large", "label": "Hermes Teal (Large)", "description": "Hermes Teal with bigger fonts and roomier spacing"}, @@ -358,12 +234,8 @@ _BUILTIN_DASHBOARD_THEMES = [ def _parse_theme_layer(value: Any, default_hex: str, default_alpha: float = 1.0) -> Optional[Dict[str, Any]]: - """Normalise a theme layer spec from YAML into `{hex, alpha}` form. - - Accepts shorthand (a bare hex string) or full dict form. Returns - ``None`` on garbage input so the caller can fall back to a built-in - default rather than blowing up. - """ + """Normalise a theme layer spec (bare hex shorthand or ``{hex, alpha}`` dict); ``None`` on + garbage so the caller falls back to a built-in default.""" if value is None: return {"hex": default_hex, "alpha": default_alpha} if isinstance(value, str): @@ -400,153 +272,110 @@ _THEME_OVERRIDE_KEYS = { "border", "input", "ring", } -# Well-known named asset slots themes can populate. Any other keys under -# ``assets.custom`` are exposed as ``--theme-asset-custom-`` CSS vars -# for plugin/shell use. +# Named asset slots; other keys under ``assets.custom`` become ``--theme-asset-custom-``. _THEME_NAMED_ASSET_KEYS = {"bg", "hero", "logo", "crest", "sidebar", "header"} -# Component-style buckets themes can override. The value under each bucket -# is a mapping from camelCase property name to CSS string; each pair emits -# ``--component--`` on :root. The frontend's shell -# components (Card, App header, Backdrop, etc.) consume these vars so themes -# can restyle chrome (clip-path, border-image, segmented progress, etc.) -# without shipping their own CSS. +# Component-style buckets: each camelCase property under a bucket emits +# ``--component--`` on :root, consumed by shell components. _THEME_COMPONENT_BUCKETS = { "card", "header", "footer", "sidebar", "tab", "progress", "badge", "backdrop", "page", } _THEME_LAYOUT_VARIANTS = {"standard", "cockpit", "tiled"} -# Cap on customCSS length so a malformed/oversized theme YAML can't blow up -# the response payload or the