diff --git a/hermes_cli/web_routers/models.py b/hermes_cli/web_routers/models.py new file mode 100644 index 0000000000..12fddb7e25 --- /dev/null +++ b/hermes_cli/web_routers/models.py @@ -0,0 +1,455 @@ +"""Model assignment dashboard routes: model info/options/recommended default, auxiliary + MoA slots, /api/model/set. + +Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch on +``web_server`` stay there and are imported lazily at call time (cycle-safe). +""" + +import logging +from fastapi import APIRouter +from fastapi import HTTPException +from hermes_cli.web_models import ModelAssignment, MoaModelSlot, MoaPresetPayload, MoaConfigPayload +from typing import Optional + +_log = logging.getLogger("hermes_cli.web_server") +router = APIRouter() + + +@router.get("/api/model/info") +def get_model_info(profile: Optional[str] = None): + """Return resolved model metadata for the currently configured model. + + Calls the same context-length resolution chain the agent uses, so the + frontend can display "Auto-detected: 200K" alongside the override field. + Also returns model capabilities (vision, reasoning, tools) when available. + """ + from hermes_cli.web_server import _EMPTY_MODEL_INFO, _profile_scope, load_config + try: + with _profile_scope(profile): + cfg = load_config() + model_cfg = cfg.get("model", "") + + # Extract model name and provider from the config + if isinstance(model_cfg, dict): + model_name = model_cfg.get("default", model_cfg.get("name", "")) + provider = model_cfg.get("provider", "") + base_url = model_cfg.get("base_url", "") + config_ctx = model_cfg.get("context_length") + else: + model_name = str(model_cfg) if model_cfg else "" + provider = "" + base_url = "" + config_ctx = None + + if not model_name: + return dict(_EMPTY_MODEL_INFO, provider=provider) + + # Resolve auto-detected context length (pass config_ctx=None to get + # purely auto-detected value, then separately report the override) + try: + from agent.model_metadata import get_model_context_length + auto_ctx = get_model_context_length( + model=model_name, + base_url=base_url, + provider=provider, + config_context_length=None, # ignore override — we want auto value + ) + except Exception: + auto_ctx = 0 + + config_ctx_int = 0 + if isinstance(config_ctx, int) and config_ctx > 0: + config_ctx_int = config_ctx + + # Effective is what the agent actually uses + effective_ctx = config_ctx_int if config_ctx_int > 0 else auto_ctx + + # Try to get model capabilities from models.dev + caps = {} + try: + from agent.models_dev import get_model_capabilities + mc = get_model_capabilities(provider=provider, model=model_name) + if mc is not None: + caps = { + "supports_tools": mc.supports_tools, + "supports_vision": mc.supports_vision, + "supports_reasoning": mc.supports_reasoning, + "context_window": mc.context_window, + "max_output_tokens": mc.max_output_tokens, + "model_family": mc.model_family, + } + except Exception: + pass + + return { + "model": model_name, + "provider": provider, + "auto_context_length": auto_ctx, + "config_context_length": config_ctx_int, + "effective_context_length": effective_ctx, + "capabilities": caps, + } + except HTTPException: + # Unknown/invalid profile must surface as 404, not degrade into a + # 200 with empty model info (which would render as "no model set"). + raise + except Exception: + _log.exception("GET /api/model/info failed") + return dict(_EMPTY_MODEL_INFO) + + +@router.get("/api/model/options") +async def get_model_options( + profile: Optional[str] = None, + refresh: bool = False, + include_unconfigured: bool = False, + explicit_only: bool = False, +): + """Return authenticated providers + their curated model lists. + + REST equivalent of the ``model.options`` JSON-RPC on tui_gateway, so the + dashboard Models page can render the picker without a live chat session. + The response shape matches ``model.options`` 1:1 so ``ModelPickerDialog`` + can share the same types. + + ``profile`` scopes the picker context (current model/provider, custom + providers from config, per-profile .env auth state) so the Models page + reads the SAME profile /api/model/set writes. + + ``refresh`` busts the per-provider model-id disk cache so every row + re-fetches its live catalog — used by the picker's explicit "Refresh + Models" control. Normal opens leave it false to stay on the 1h cache. + """ + from hermes_cli.web_server import ( + _config_profile_scope, + _dashboard_code_skew_guard, + run_in_threadpool, + ) + try: + skew_msg = _dashboard_code_skew_guard() + if skew_msg: + _log.warning("GET /api/model/options refused: %s", skew_msg) + raise HTTPException( + status_code=503, detail=f"Restart required: {skew_msg}" + ) + + from hermes_cli.inventory import build_model_options_payload, load_picker_context + + def _build_payload_scoped() -> dict: + # Keep the profile override inside the worker thread so the full + # sync picker build (config load, pricing, refresh probes) runs + # off the event loop under the requested profile. + # Use _config_profile_scope (contextvar only, no skill-module + # lock) — the payload build can block for 15s on a models.dev + # cache miss, and _profile_scope's RLock held across that block + # starves concurrent /api/config and freezes the server (#58576). + with _config_profile_scope(profile): + return build_model_options_payload( + load_picker_context(), + explicit_only=bool(explicit_only), + include_unconfigured=bool(include_unconfigured), + refresh=bool(refresh), + ) + + return await run_in_threadpool(_build_payload_scoped) + except HTTPException: + raise + except Exception: + _log.exception("GET /api/model/options failed") + raise HTTPException(status_code=500, detail="Failed to list model options") + + +@router.get("/api/model/recommended-default") +def get_recommended_default_model(provider: str = ""): + """Return the recommended default model for a freshly-authenticated provider. + + Mirrors the model-curation `hermes model` does so GUI onboarding lands on a + sensible default instead of blindly taking the first curated entry. For + Nous this honors the user's free/paid tier: free users get a free model, + paid users get the full curated default. For any other provider it falls + back to the first curated model (same as before). + + Response: {"provider": str, "model": str, "free_tier": bool | None} + where free_tier is True/False for Nous and None otherwise. `model` may be + empty if nothing could be resolved (caller degrades gracefully). + """ + slug = (provider or "").strip().lower() + + if slug == "nous": + try: + from hermes_cli.models import ( + get_curated_nous_model_ids, + get_pricing_for_provider, + check_nous_free_tier, + nous_policy_allowed_ids, + partition_nous_models_by_tier, + pick_silent_default_model, + restrict_to_nous_policy, + union_with_portal_free_recommendations, + union_with_portal_paid_recommendations, + ) + from hermes_cli.auth import get_provider_auth_state + + model_ids = get_curated_nous_model_ids() + pricing = get_pricing_for_provider("nous") or {} + free_tier = check_nous_free_tier(force_fresh=True) + + portal_url = "" + try: + state = get_provider_auth_state("nous") or {} + portal_url = state.get("portal_base_url", "") or "" + except Exception: + portal_url = "" + + # This endpoint picks the model a user lands on without choosing it, + # so an unreachable one here is worse than in a picker. Narrow before + # the tier split, so a rescued id still has to pass the free/paid + # predicate. + _policy_allowed = nous_policy_allowed_ids() + + if free_tier: + model_ids, pricing = union_with_portal_free_recommendations( + model_ids, pricing, portal_url + ) + model_ids = restrict_to_nous_policy( + model_ids, _policy_allowed, rescue_empty=True, + ) + model_ids, _unavailable = partition_nous_models_by_tier( + model_ids, pricing, free_tier=True + ) + else: + model_ids, pricing = union_with_portal_paid_recommendations( + model_ids, pricing, portal_url + ) + model_ids = restrict_to_nous_policy( + model_ids, _policy_allowed, rescue_empty=True, + ) + + model = pick_silent_default_model(model_ids, provider="nous") + return {"provider": "nous", "model": model, "free_tier": bool(free_tier)} + except Exception: + _log.exception("GET /api/model/recommended-default (nous) failed") + return {"provider": "nous", "model": "", "free_tier": None} + + # Non-Nous: preferred silent default when the provider's curated list + # carries it, else the first curated model. Aggregator lists lead with the + # priciest Anthropic flagship (claude-fable-5), which must never be the + # model a user lands on without explicitly picking it. + try: + from hermes_cli.inventory import build_models_payload, load_picker_context + from hermes_cli.models import pick_silent_default_model + + payload = build_models_payload(load_picker_context()) + for row in payload.get("providers", []): + if str(row.get("slug", "")).lower() == slug: + models = [str(m) for m in (row.get("models") or [])] + return {"provider": slug, "model": pick_silent_default_model(models, provider=slug), "free_tier": None} + return {"provider": slug, "model": "", "free_tier": None} + except Exception: + _log.exception("GET /api/model/recommended-default failed") + return {"provider": slug, "model": "", "free_tier": None} + + +@router.get("/api/model/auxiliary") +def get_auxiliary_models(profile: Optional[str] = None): + """Return current auxiliary task assignments. + + Shape: + { + "tasks": [ + {"task": "vision", "provider": "auto", "model": "", "base_url": ""}, + ... + ], + "main": {"provider": "openrouter", "model": "anthropic/claude-opus-4.7"}, + } + + ``profile`` scopes the read — without it, the Models page would show + the dashboard profile's auxiliary pins while /api/model/set wrote the + selected profile's (read/write asymmetry). + """ + from hermes_cli.web_server import _AUX_TASK_SLOTS, _profile_scope, load_config + try: + with _profile_scope(profile): + cfg = load_config() + aux_cfg = cfg.get("auxiliary", {}) + if not isinstance(aux_cfg, dict): + aux_cfg = {} + + tasks = [] + for slot in _AUX_TASK_SLOTS: + slot_cfg = aux_cfg.get(slot, {}) if isinstance(aux_cfg.get(slot), dict) else {} + tasks.append({ + "task": slot, + "provider": str(slot_cfg.get("provider", "auto") or "auto"), + "model": str(slot_cfg.get("model", "") or ""), + "base_url": str(slot_cfg.get("base_url", "") or ""), + }) + + model_cfg = cfg.get("model", {}) + if isinstance(model_cfg, dict): + main = { + "provider": str(model_cfg.get("provider", "") or ""), + "model": str(model_cfg.get("default", model_cfg.get("name", "")) or ""), + } + else: + main = {"provider": "", "model": str(model_cfg) if model_cfg else ""} + + return {"tasks": tasks, "main": main} + except HTTPException: + raise + except Exception: + _log.exception("GET /api/model/auxiliary failed") + raise HTTPException(status_code=500, detail="Failed to read auxiliary config") + + +@router.get("/api/model/moa") +def get_moa_models(profile: Optional[str] = None): + """Return the configured Mixture-of-Agents provider/model slots.""" + from hermes_cli.web_server import _profile_scope, load_config + try: + from hermes_cli.moa_config import normalize_moa_config + + with _profile_scope(profile): + cfg = load_config() + return normalize_moa_config(cfg.get("moa") if isinstance(cfg, dict) else {}) + except HTTPException: + raise + except Exception: + _log.exception("GET /api/model/moa failed") + raise HTTPException(status_code=500, detail="Failed to read MoA config") + + +@router.put("/api/model/moa") +def set_moa_models(body: MoaConfigPayload, profile: Optional[str] = None): + """Persist the Mixture-of-Agents provider/model slots.""" + from hermes_cli.web_server import _profile_scope, load_config, save_config + try: + from hermes_cli.moa_config import normalize_moa_config, validate_moa_payload + + def _slot_dict(slot: MoaModelSlot) -> dict: + # Drop unset optionals so saved slots stay minimal ({provider, model}). + return {k: v for k, v in slot.dict().items() if v is not None} + + def _preset_dict(preset: MoaPresetPayload) -> dict: + return { + "reference_models": [_slot_dict(slot) for slot in preset.reference_models], + "aggregator": _slot_dict(preset.aggregator), + "reference_temperature": preset.reference_temperature, + "aggregator_temperature": preset.aggregator_temperature, + "reference_timeout": preset.reference_timeout, + "degraded_reference_policy": preset.degraded_reference_policy, + "max_tokens": preset.max_tokens, + "reference_max_tokens": preset.reference_max_tokens, + "fanout": preset.fanout, + "enabled": preset.enabled, + } + + with _profile_scope(body.profile or profile): + cfg = load_config() + if body.presets: + raw = { + "default_preset": body.default_preset, + "active_preset": body.active_preset, + "presets": {name: _preset_dict(preset) for name, preset in body.presets.items()}, + } + else: + raw = _preset_dict( + MoaPresetPayload( + reference_models=body.reference_models, + aggregator=body.aggregator, + reference_temperature=body.reference_temperature, + aggregator_temperature=body.aggregator_temperature, + reference_timeout=body.reference_timeout, + degraded_reference_policy=body.degraded_reference_policy, + max_tokens=body.max_tokens, + reference_max_tokens=body.reference_max_tokens, + fanout=body.fanout, + enabled=body.enabled, + ) + ) + + # Reject-don't-repair: normalize_moa_config() silently swaps any + # preset containing incomplete slots for the hardcoded defaults — + # correct tolerance for hand-edited configs at READ time, silent + # data loss at WRITE time (#64156: desktop autosave of a + # half-filled slot replaced the user's whole preset). Refuse the + # save loudly so no client can corrupt config through this route. + problems = validate_moa_payload(raw) + if problems: + raise HTTPException( + status_code=422, + detail="Invalid MoA config: " + "; ".join(problems), + ) + normalized = normalize_moa_config(raw) + # Merge instead of overwrite so that hand-edited keys not declared + # in MoaConfigPayload (e.g. save_traces, trace_dir) survive a GUI + # save. See issue #58819. + cfg.setdefault("moa", {}).update(normalized) + save_config(cfg) + return {"ok": True, **normalized} + except HTTPException: + raise + except Exception: + _log.exception("PUT /api/model/moa failed") + raise HTTPException(status_code=500, detail="Failed to save MoA config") + + +@router.post("/api/model/set") +async def set_model_assignment(body: ModelAssignment, profile: Optional[str] = None): + """Assign a model to the main slot or an auxiliary task slot. + + Writes to ``~/.hermes/config.yaml`` — applies to **new** sessions only. + The currently running chat PTY (if any) is not affected; use the + ``/model`` slash command inside a chat to hot-swap that specific session. + """ + from hermes_cli.web_server import _apply_model_assignment_sync, _profile_scope, asyncio + scope = (body.scope or "").strip().lower() + provider = (body.provider or "").strip() + model = (body.model or "").strip() + task = (body.task or "").strip().lower() + base_url = (body.base_url or "").strip() + api_key = (body.api_key or "").strip() + + if scope not in {"main", "auxiliary"}: + raise HTTPException(status_code=400, detail="scope must be 'main' or 'auxiliary'") + + try: + # Expensive-model warning runs BEFORE the profile scope is entered: + # _profile_scope must never be held across an await (the RLock is + # reentrant per-thread, so a second coroutine interleaving on the + # event-loop thread could cross-restore the module globals). + if model and not body.confirm_expensive_model: + try: + from hermes_cli.model_selection_guards import combined_selection_warning + + # Pricing lookup can hit models.dev / a /models endpoint on a + # cache miss — keep it off the event loop. + warning = await asyncio.to_thread( + combined_selection_warning, + model, + provider=provider, + base_url=base_url, + ) + except Exception: + warning = None + if warning is not None: + return { + "ok": False, + "scope": scope, + "provider": provider, + "model": model, + "confirm_required": True, + "confirm_message": warning.message, + } + + def _apply_assignment(): + with _profile_scope(body.profile or profile): + return _apply_model_assignment_sync( + scope, provider, model, task, base_url, api_key + ) + + return await asyncio.to_thread(_apply_assignment) + except HTTPException: + raise + except Exception: + _log.exception("POST /api/model/set failed") + raise HTTPException(status_code=500, detail="Failed to save model assignment") diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index f9701194b7..cd2958692f 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -6204,86 +6204,15 @@ _EMPTY_MODEL_INFO: dict = { } -@app.get("/api/model/info") -def get_model_info(profile: Optional[str] = None): - """Return resolved model metadata for the currently configured model. +from hermes_cli.web_routers import models as _models_routes # noqa: E402 - Calls the same context-length resolution chain the agent uses, so the - frontend can display "Auto-detected: 200K" alongside the override field. - Also returns model capabilities (vision, reasoning, tools) when available. - """ - try: - with _profile_scope(profile): - cfg = load_config() - model_cfg = cfg.get("model", "") - - # Extract model name and provider from the config - if isinstance(model_cfg, dict): - model_name = model_cfg.get("default", model_cfg.get("name", "")) - provider = model_cfg.get("provider", "") - base_url = model_cfg.get("base_url", "") - config_ctx = model_cfg.get("context_length") - else: - model_name = str(model_cfg) if model_cfg else "" - provider = "" - base_url = "" - config_ctx = None - - if not model_name: - return dict(_EMPTY_MODEL_INFO, provider=provider) - - # Resolve auto-detected context length (pass config_ctx=None to get - # purely auto-detected value, then separately report the override) - try: - from agent.model_metadata import get_model_context_length - auto_ctx = get_model_context_length( - model=model_name, - base_url=base_url, - provider=provider, - config_context_length=None, # ignore override — we want auto value - ) - except Exception: - auto_ctx = 0 - - config_ctx_int = 0 - if isinstance(config_ctx, int) and config_ctx > 0: - config_ctx_int = config_ctx - - # Effective is what the agent actually uses - effective_ctx = config_ctx_int if config_ctx_int > 0 else auto_ctx - - # Try to get model capabilities from models.dev - caps = {} - try: - from agent.models_dev import get_model_capabilities - mc = get_model_capabilities(provider=provider, model=model_name) - if mc is not None: - caps = { - "supports_tools": mc.supports_tools, - "supports_vision": mc.supports_vision, - "supports_reasoning": mc.supports_reasoning, - "context_window": mc.context_window, - "max_output_tokens": mc.max_output_tokens, - "model_family": mc.model_family, - } - except Exception: - pass - - return { - "model": model_name, - "provider": provider, - "auto_context_length": auto_ctx, - "config_context_length": config_ctx_int, - "effective_context_length": effective_ctx, - "capabilities": caps, - } - except HTTPException: - # Unknown/invalid profile must surface as 404, not degrade into a - # 200 with empty model info (which would render as "no model set"). - raise - except Exception: - _log.exception("GET /api/model/info failed") - return dict(_EMPTY_MODEL_INFO) +app.include_router(_models_routes.router) +from hermes_cli.web_routers.models import ( # noqa: E402,F401 — legacy re-exports; tests call these via web_server. + get_model_info, + get_model_options, + get_recommended_default_model, + set_moa_models, +) # --------------------------------------------------------------------------- @@ -6359,355 +6288,6 @@ def _dashboard_skew_restart_hint() -> str: ) -@app.get("/api/model/options") -async def get_model_options( - profile: Optional[str] = None, - refresh: bool = False, - include_unconfigured: bool = False, - explicit_only: bool = False, -): - """Return authenticated providers + their curated model lists. - - REST equivalent of the ``model.options`` JSON-RPC on tui_gateway, so the - dashboard Models page can render the picker without a live chat session. - The response shape matches ``model.options`` 1:1 so ``ModelPickerDialog`` - can share the same types. - - ``profile`` scopes the picker context (current model/provider, custom - providers from config, per-profile .env auth state) so the Models page - reads the SAME profile /api/model/set writes. - - ``refresh`` busts the per-provider model-id disk cache so every row - re-fetches its live catalog — used by the picker's explicit "Refresh - Models" control. Normal opens leave it false to stay on the 1h cache. - """ - try: - skew_msg = _dashboard_code_skew_guard() - if skew_msg: - _log.warning("GET /api/model/options refused: %s", skew_msg) - raise HTTPException( - status_code=503, detail=f"Restart required: {skew_msg}" - ) - - from hermes_cli.inventory import build_model_options_payload, load_picker_context - - def _build_payload_scoped() -> dict: - # Keep the profile override inside the worker thread so the full - # sync picker build (config load, pricing, refresh probes) runs - # off the event loop under the requested profile. - # Use _config_profile_scope (contextvar only, no skill-module - # lock) — the payload build can block for 15s on a models.dev - # cache miss, and _profile_scope's RLock held across that block - # starves concurrent /api/config and freezes the server (#58576). - with _config_profile_scope(profile): - return build_model_options_payload( - load_picker_context(), - explicit_only=bool(explicit_only), - include_unconfigured=bool(include_unconfigured), - refresh=bool(refresh), - ) - - return await run_in_threadpool(_build_payload_scoped) - except HTTPException: - raise - except Exception: - _log.exception("GET /api/model/options failed") - raise HTTPException(status_code=500, detail="Failed to list model options") - - -@app.get("/api/model/recommended-default") -def get_recommended_default_model(provider: str = ""): - """Return the recommended default model for a freshly-authenticated provider. - - Mirrors the model-curation `hermes model` does so GUI onboarding lands on a - sensible default instead of blindly taking the first curated entry. For - Nous this honors the user's free/paid tier: free users get a free model, - paid users get the full curated default. For any other provider it falls - back to the first curated model (same as before). - - Response: {"provider": str, "model": str, "free_tier": bool | None} - where free_tier is True/False for Nous and None otherwise. `model` may be - empty if nothing could be resolved (caller degrades gracefully). - """ - slug = (provider or "").strip().lower() - - if slug == "nous": - try: - from hermes_cli.models import ( - get_curated_nous_model_ids, - get_pricing_for_provider, - check_nous_free_tier, - nous_policy_allowed_ids, - partition_nous_models_by_tier, - pick_silent_default_model, - restrict_to_nous_policy, - union_with_portal_free_recommendations, - union_with_portal_paid_recommendations, - ) - from hermes_cli.auth import get_provider_auth_state - - model_ids = get_curated_nous_model_ids() - pricing = get_pricing_for_provider("nous") or {} - free_tier = check_nous_free_tier(force_fresh=True) - - portal_url = "" - try: - state = get_provider_auth_state("nous") or {} - portal_url = state.get("portal_base_url", "") or "" - except Exception: - portal_url = "" - - # This endpoint picks the model a user lands on without choosing it, - # so an unreachable one here is worse than in a picker. Narrow before - # the tier split, so a rescued id still has to pass the free/paid - # predicate. - _policy_allowed = nous_policy_allowed_ids() - - if free_tier: - model_ids, pricing = union_with_portal_free_recommendations( - model_ids, pricing, portal_url - ) - model_ids = restrict_to_nous_policy( - model_ids, _policy_allowed, rescue_empty=True, - ) - model_ids, _unavailable = partition_nous_models_by_tier( - model_ids, pricing, free_tier=True - ) - else: - model_ids, pricing = union_with_portal_paid_recommendations( - model_ids, pricing, portal_url - ) - model_ids = restrict_to_nous_policy( - model_ids, _policy_allowed, rescue_empty=True, - ) - - model = pick_silent_default_model(model_ids, provider="nous") - return {"provider": "nous", "model": model, "free_tier": bool(free_tier)} - except Exception: - _log.exception("GET /api/model/recommended-default (nous) failed") - return {"provider": "nous", "model": "", "free_tier": None} - - # Non-Nous: preferred silent default when the provider's curated list - # carries it, else the first curated model. Aggregator lists lead with the - # priciest Anthropic flagship (claude-fable-5), which must never be the - # model a user lands on without explicitly picking it. - try: - from hermes_cli.inventory import build_models_payload, load_picker_context - from hermes_cli.models import pick_silent_default_model - - payload = build_models_payload(load_picker_context()) - for row in payload.get("providers", []): - if str(row.get("slug", "")).lower() == slug: - models = [str(m) for m in (row.get("models") or [])] - return {"provider": slug, "model": pick_silent_default_model(models, provider=slug), "free_tier": None} - return {"provider": slug, "model": "", "free_tier": None} - except Exception: - _log.exception("GET /api/model/recommended-default failed") - return {"provider": slug, "model": "", "free_tier": None} - - -@app.get("/api/model/auxiliary") -def get_auxiliary_models(profile: Optional[str] = None): - """Return current auxiliary task assignments. - - Shape: - { - "tasks": [ - {"task": "vision", "provider": "auto", "model": "", "base_url": ""}, - ... - ], - "main": {"provider": "openrouter", "model": "anthropic/claude-opus-4.7"}, - } - - ``profile`` scopes the read — without it, the Models page would show - the dashboard profile's auxiliary pins while /api/model/set wrote the - selected profile's (read/write asymmetry). - """ - try: - with _profile_scope(profile): - cfg = load_config() - aux_cfg = cfg.get("auxiliary", {}) - if not isinstance(aux_cfg, dict): - aux_cfg = {} - - tasks = [] - for slot in _AUX_TASK_SLOTS: - slot_cfg = aux_cfg.get(slot, {}) if isinstance(aux_cfg.get(slot), dict) else {} - tasks.append({ - "task": slot, - "provider": str(slot_cfg.get("provider", "auto") or "auto"), - "model": str(slot_cfg.get("model", "") or ""), - "base_url": str(slot_cfg.get("base_url", "") or ""), - }) - - model_cfg = cfg.get("model", {}) - if isinstance(model_cfg, dict): - main = { - "provider": str(model_cfg.get("provider", "") or ""), - "model": str(model_cfg.get("default", model_cfg.get("name", "")) or ""), - } - else: - main = {"provider": "", "model": str(model_cfg) if model_cfg else ""} - - return {"tasks": tasks, "main": main} - except HTTPException: - raise - except Exception: - _log.exception("GET /api/model/auxiliary failed") - raise HTTPException(status_code=500, detail="Failed to read auxiliary config") - - -@app.get("/api/model/moa") -def get_moa_models(profile: Optional[str] = None): - """Return the configured Mixture-of-Agents provider/model slots.""" - try: - from hermes_cli.moa_config import normalize_moa_config - - with _profile_scope(profile): - cfg = load_config() - return normalize_moa_config(cfg.get("moa") if isinstance(cfg, dict) else {}) - except HTTPException: - raise - except Exception: - _log.exception("GET /api/model/moa failed") - raise HTTPException(status_code=500, detail="Failed to read MoA config") - - -@app.put("/api/model/moa") -def set_moa_models(body: MoaConfigPayload, profile: Optional[str] = None): - """Persist the Mixture-of-Agents provider/model slots.""" - try: - from hermes_cli.moa_config import normalize_moa_config, validate_moa_payload - - def _slot_dict(slot: MoaModelSlot) -> dict: - # Drop unset optionals so saved slots stay minimal ({provider, model}). - return {k: v for k, v in slot.dict().items() if v is not None} - - def _preset_dict(preset: MoaPresetPayload) -> dict: - return { - "reference_models": [_slot_dict(slot) for slot in preset.reference_models], - "aggregator": _slot_dict(preset.aggregator), - "reference_temperature": preset.reference_temperature, - "aggregator_temperature": preset.aggregator_temperature, - "reference_timeout": preset.reference_timeout, - "degraded_reference_policy": preset.degraded_reference_policy, - "max_tokens": preset.max_tokens, - "reference_max_tokens": preset.reference_max_tokens, - "fanout": preset.fanout, - "enabled": preset.enabled, - } - - with _profile_scope(body.profile or profile): - cfg = load_config() - if body.presets: - raw = { - "default_preset": body.default_preset, - "active_preset": body.active_preset, - "presets": {name: _preset_dict(preset) for name, preset in body.presets.items()}, - } - else: - raw = _preset_dict( - MoaPresetPayload( - reference_models=body.reference_models, - aggregator=body.aggregator, - reference_temperature=body.reference_temperature, - aggregator_temperature=body.aggregator_temperature, - reference_timeout=body.reference_timeout, - degraded_reference_policy=body.degraded_reference_policy, - max_tokens=body.max_tokens, - reference_max_tokens=body.reference_max_tokens, - fanout=body.fanout, - enabled=body.enabled, - ) - ) - - # Reject-don't-repair: normalize_moa_config() silently swaps any - # preset containing incomplete slots for the hardcoded defaults — - # correct tolerance for hand-edited configs at READ time, silent - # data loss at WRITE time (#64156: desktop autosave of a - # half-filled slot replaced the user's whole preset). Refuse the - # save loudly so no client can corrupt config through this route. - problems = validate_moa_payload(raw) - if problems: - raise HTTPException( - status_code=422, - detail="Invalid MoA config: " + "; ".join(problems), - ) - normalized = normalize_moa_config(raw) - # Merge instead of overwrite so that hand-edited keys not declared - # in MoaConfigPayload (e.g. save_traces, trace_dir) survive a GUI - # save. See issue #58819. - cfg.setdefault("moa", {}).update(normalized) - save_config(cfg) - return {"ok": True, **normalized} - except HTTPException: - raise - except Exception: - _log.exception("PUT /api/model/moa failed") - raise HTTPException(status_code=500, detail="Failed to save MoA config") - - -@app.post("/api/model/set") -async def set_model_assignment(body: ModelAssignment, profile: Optional[str] = None): - """Assign a model to the main slot or an auxiliary task slot. - - Writes to ``~/.hermes/config.yaml`` — applies to **new** sessions only. - The currently running chat PTY (if any) is not affected; use the - ``/model`` slash command inside a chat to hot-swap that specific session. - """ - scope = (body.scope or "").strip().lower() - provider = (body.provider or "").strip() - model = (body.model or "").strip() - task = (body.task or "").strip().lower() - base_url = (body.base_url or "").strip() - api_key = (body.api_key or "").strip() - - if scope not in {"main", "auxiliary"}: - raise HTTPException(status_code=400, detail="scope must be 'main' or 'auxiliary'") - - try: - # Expensive-model warning runs BEFORE the profile scope is entered: - # _profile_scope must never be held across an await (the RLock is - # reentrant per-thread, so a second coroutine interleaving on the - # event-loop thread could cross-restore the module globals). - if model and not body.confirm_expensive_model: - try: - from hermes_cli.model_selection_guards import combined_selection_warning - - # Pricing lookup can hit models.dev / a /models endpoint on a - # cache miss — keep it off the event loop. - warning = await asyncio.to_thread( - combined_selection_warning, - model, - provider=provider, - base_url=base_url, - ) - except Exception: - warning = None - if warning is not None: - return { - "ok": False, - "scope": scope, - "provider": provider, - "model": model, - "confirm_required": True, - "confirm_message": warning.message, - } - - def _apply_assignment(): - with _profile_scope(body.profile or profile): - return _apply_model_assignment_sync( - scope, provider, model, task, base_url, api_key - ) - - return await asyncio.to_thread(_apply_assignment) - except HTTPException: - raise - except Exception: - _log.exception("POST /api/model/set failed") - raise HTTPException(status_code=500, detail="Failed to save model assignment") - - def _apply_model_assignment_sync( scope: str, provider: str, model: str, task: str, base_url: str, api_key: str = "" ):