fix(multiplex): per-profile catalog, skin and guest-mint state in hermes_cli

DeepInfra catalog (fetched with the launch env's key via os.getenv), Copilot
context limits (api_key ignored on hit), Nous reasoning caps + once-per-process
guards, the curated OpenRouter list, the model-catalog in-process copy (mtime
without path), banner skills, the guest-mint back-off flag and the active skin
were single slots read under per-profile overrides by the gateway and the TUI
gateway; the SWR refresh thread ran without the caller's ContextVars.

Under an override each lives per home key (hermes_cli/models_profile_cache.py
holds the shared slot helper so models.py does not grow), credentials are read
through the scope-aware dotenv reader and keyed by fingerprint, and background
refreshes run under copy_context(). Unscoped behaviour is byte-identical.
This commit is contained in:
Teknium
2026-09-12 01:03:46 -07:00
parent 4b8c01f691
commit 5ff34f565e
8 changed files with 442 additions and 37 deletions
+25 -6
View File
@@ -284,7 +284,28 @@ def _mint_locked(
# Per-process memo: one failed mint is enough for a process (a 429 or a closed gate must not be hit
# twice); ``clear_dead_guest`` resets it because a retired credential is a reason to mint again.
# The bool is the unscoped (launch profile) slot; routed multiplex profiles each get their own entry
# in the set — profile A's 429 must not stop profile B from ever getting an identity.
_mint_failed = False
_mint_failed_homes: set[str] = set()
def _mint_failed_for_profile() -> bool:
from hermes_constants import get_hermes_home_override, hermes_home_key
if get_hermes_home_override() is None:
return _mint_failed
return hermes_home_key() in _mint_failed_homes
def _set_mint_failed(failed: bool) -> None:
global _mint_failed
from hermes_constants import get_hermes_home_override, hermes_home_key
if get_hermes_home_override() is None:
_mint_failed = failed
elif failed:
_mint_failed_homes.add(hermes_home_key())
else:
_mint_failed_homes.discard(hermes_home_key())
def _reconcile_and_provision(*, timeout_seconds: float, carries_inference: bool = True) -> Optional[Dict[str, Any]]:
@@ -346,16 +367,15 @@ def ensure_portal_identity(
"""
if not explicit:
raise ValueError("ensure_portal_identity: only explicit creators may call this (explicit=True)")
global _mint_failed
if not guest_enabled():
return None
if _mint_failed and not current_nous_state():
return None # this process already tried and failed; do not hammer the portal
if _mint_failed_for_profile() and not current_nous_state():
return None # this profile already tried and failed in this process; do not hammer the portal
try:
return _reconcile_and_provision(
timeout_seconds=timeout_seconds, carries_inference=carries_inference)
except Exception:
_mint_failed = True
_set_mint_failed(True)
raise
@@ -401,8 +421,7 @@ def clear_dead_guest(reason: str, *, dead_token: Optional[str] = None) -> None:
shared = _read_shared_nous_state()
if token and is_guest_state(shared) and shared.get("anon_token") == token:
_clear_shared_nous_state(reason)
global _mint_failed
_mint_failed = False
_set_mint_failed(False)
logger.info("Nous free-tier identity retired (%s); a new one is set up on next use", reason)
+7 -1
View File
@@ -92,7 +92,13 @@ _UNCACHED = object() # compute() result that must not be memoized
def _memo(cache_name: str, compute):
"""Return the cached value under module global ``cache_name``, computing (and storing) it once."""
"""Return the cached value under module global ``cache_name``, computing (and storing) it once.
Not consulted under a routed profile (HERMES_HOME override): every memo here is derived from the
launch home (its skills tree, its checkout), and the TUI gateway calls these per profile."""
from hermes_constants import get_hermes_home_override
if get_hermes_home_override() is not None:
return compute()
cached = globals()[cache_name]
if cached is not None:
return cached[0]
+20 -6
View File
@@ -37,9 +37,12 @@ SUPPORTED_SCHEMA_VERSION = 1
_HERMES_USER_AGENT = f"hermes-cli/{_HERMES_VERSION}"
# In-process cache, invalidated against the disk file's mtime and TTL.
# In-process cache, invalidated against the disk file's path + mtime and TTL. The path matters:
# under a multiplexed gateway each profile has its own ``<home>/cache/model_catalog.json``, and
# mtime alone cannot tell two profiles' files apart.
_catalog_cache: dict[str, Any] | None = None
_catalog_cache_source_mtime: float = 0.0
_catalog_cache_source_path: str = ""
def _load_catalog_config() -> dict[str, Any]:
@@ -193,11 +196,19 @@ def _spawn_catalog_swr_refresh(url: str) -> None:
def _remember(data: dict[str, Any], mtime: float) -> dict[str, Any]:
global _catalog_cache, _catalog_cache_source_mtime
global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path
_catalog_cache, _catalog_cache_source_mtime = data, mtime
_catalog_cache_source_path = str(_cache_path())
return data
def _in_process_catalog() -> dict[str, Any] | None:
"""The in-process copy when it mirrors the ACTIVE profile's cache file, else None."""
if _catalog_cache is not None and _catalog_cache_source_path == str(_cache_path()):
return _catalog_cache
return None
def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]:
"""Parsed model catalog manifest, or ``{}`` on failure — never raises, so the CLI works offline
(callers treat a missing provider/model as "use the in-repo fallback")."""
@@ -210,8 +221,9 @@ def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]:
disk_fresh = disk_data is not None and (now - disk_mtime) < ttl_seconds
if not force_refresh and disk_data is not None:
if disk_fresh and _catalog_cache is not None and disk_mtime == _catalog_cache_source_mtime:
return _catalog_cache
cached = _in_process_catalog()
if disk_fresh and cached is not None and disk_mtime == _catalog_cache_source_mtime:
return cached
if not disk_fresh:
# Stale-while-revalidate: serve the expired disk copy now and refresh off-thread so the
# /model picker (which calls this on every open) never blocks on the manifest fetch.
@@ -303,7 +315,8 @@ def _default_model_from_block(block: dict[str, Any] | None) -> str | None:
def get_default_model_from_cache(provider: str) -> str | None:
"""The manifest's labeled default for ``provider`` (the model Hermes silently lands on when the
user never picked one) — in-process then disk cache only, never a fetch."""
found = _default_model_from_block(_block_of(_catalog_cache, provider)) if _catalog_cache is not None else None
cached = _in_process_catalog()
found = _default_model_from_block(_block_of(cached, provider)) if cached is not None else None
if found:
return found
disk_data, _mtime = _read_disk_cache()
@@ -331,6 +344,7 @@ def seed_cache_from_checkout(project_root: "Path | str") -> bool:
def reset_cache() -> None:
"""Clear the in-process cache. Used by tests and ``hermes model --refresh``."""
global _catalog_cache, _catalog_cache_source_mtime
global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path
_catalog_cache = None
_catalog_cache_source_mtime = 0.0
_catalog_cache_source_path = ""
+48 -19
View File
@@ -8,11 +8,13 @@ Origin module; cohesive clusters live in siblings and are re-imported here so
from __future__ import annotations
import contextvars
import copy
import json
import logging
import os
import re
import sys
import threading
import urllib.parse
import urllib.request
@@ -542,17 +544,21 @@ def _fetch_live_catalog_index(url: str, timeout: float, opener) -> Optional[tupl
def fetch_openrouter_models(
timeout: float = 8.0, *, force_refresh: bool = False) -> list[tuple[str, str]]:
"""Return the curated OpenRouter picker list, refreshed from the live catalog when possible."""
global _openrouter_catalog_cache
# The curated list is filtered from this profile's manifest (``model_catalog.*`` config, its
# ``<home>/cache`` copy), so a routed profile keeps its own slot instead of the module one.
from hermes_cli.models_profile_cache import profile_slot_get, profile_slot_set
_me = sys.modules[__name__]
cached = profile_slot_get(_me, "_openrouter_catalog_cache")
if _openrouter_catalog_cache is not None and not force_refresh:
return list(_openrouter_catalog_cache)
if cached is not None and not force_refresh:
return list(cached)
# Cold process: serve from the persisted disk cache when fresh so the
# picker doesn't re-download the full ~686KB catalog on every open.
if not force_refresh:
disk = _read_openrouter_catalog_disk()
if disk:
_openrouter_catalog_cache = disk
profile_slot_set(_me, "_openrouter_catalog_cache", disk)
return list(disk)
# Remote catalog manifest first, in-repo snapshot when unreachable; the live /v1/models filter
@@ -566,7 +572,7 @@ def fetch_openrouter_models(
live = _fetch_live_catalog_index(_OPENROUTER_CATALOG_URL, timeout, _urlopen_model_catalog_request)
if live is None:
return list(_openrouter_catalog_cache or fallback)
return list(cached or fallback)
live_items, live_by_id = live
# Free warm-up for the reasoning-capability cache: same payload the caps fetch would pull.
@@ -592,10 +598,10 @@ def fetch_openrouter_models(
curated.append((preferred_id, desc))
if not curated:
return list(_openrouter_catalog_cache or fallback)
return list(cached or fallback)
if not curated[0][1]:
curated[0] = (curated[0][0], "recommended")
_openrouter_catalog_cache = curated
profile_slot_set(_me, "_openrouter_catalog_cache", curated)
_write_openrouter_catalog_disk(curated)
return list(curated)
@@ -1479,10 +1485,15 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None:
Failures are swallowed — the stale entry stays served until a later refresh succeeds.
``refresh_fn`` (no-args → fresh entry dict or None) lets ``custom:<base_url>`` keys from
:func:`cached_fetch_api_models` reuse the same inflight-dedupe scaffolding."""
# Under a routed profile the inflight key includes the home: the same provider slug names a
# different disk cache and credential set per profile, so one profile's refresh must not
# suppress another's. Unscoped keeps the bare key (tests inspect the set by slug).
from hermes_constants import get_hermes_home_override, hermes_home_key
inflight_key = cache_key if get_hermes_home_override() is None else (hermes_home_key(), cache_key)
with _swr_refresh_lock:
if cache_key in _swr_refresh_inflight:
if inflight_key in _swr_refresh_inflight:
return
_swr_refresh_inflight.add(cache_key)
_swr_refresh_inflight.add(inflight_key)
def _default_refresh():
live = provider_model_ids(cache_key, force_refresh=True)
@@ -1499,9 +1510,11 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None:
logger.debug("SWR refresh failed for %s", cache_key, exc_info=True)
finally:
with _swr_refresh_lock:
_swr_refresh_inflight.discard(cache_key)
_swr_refresh_inflight.discard(inflight_key)
threading.Thread(target=_refresh, daemon=True, name=f"model-cache-swr-{cache_key}").start()
# copy_context: the refresh must read the calling profile's credentials and write ITS disk cache.
ctx = contextvars.copy_context()
threading.Thread(target=lambda: ctx.run(_refresh), daemon=True, name=f"model-cache-swr-{cache_key}").start()
def _provider_models_cache_path() -> Path:
@@ -1893,15 +1906,21 @@ def fetch_github_model_catalog(
# Module-level cache: {model_id: max_prompt_tokens}
_copilot_context_cache: dict[str, int] = {}
_copilot_context_cache_time: float = 0.0
_copilot_context_cache_key: Optional[str] = None # fingerprint of the api_key the entry was fetched with
_COPILOT_CONTEXT_CACHE_TTL = 3600 # 1 hour
def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> Optional[int]:
"""``max_prompt_tokens`` for a Copilot model from the live /models API (cached in-process 1h; a
miss on a fresh cache does not re-fetch), or None."""
global _copilot_context_cache, _copilot_context_cache_time
global _copilot_context_cache, _copilot_context_cache_time, _copilot_context_cache_key
if _copilot_context_cache and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL):
# Keyed on the credential like fetch_github_model_catalog: the catalog (and its limits) is
# per-account, so another profile's token must not be served this entry.
from agent.credential_persistence import fingerprint_secret_value
key_fp = fingerprint_secret_value(api_key)
if (_copilot_context_cache and _copilot_context_cache_key == key_fp
and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL)):
return _copilot_context_cache.get(model_id)
catalog = fetch_github_model_catalog(api_key=api_key)
@@ -1915,6 +1934,7 @@ def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> O
cache[mid] = max_prompt
_copilot_context_cache = cache
_copilot_context_cache_time = time.time()
_copilot_context_cache_key = key_fp
return cache.get(model_id)
@@ -2355,11 +2375,20 @@ _deepinfra_catalog_neg_cache: dict[str, float] = {}
_DEEPINFRA_CATALOG_NEG_TTL = 60.0 # seconds
def _deepinfra_env(key: str) -> str:
"""Profile-scoped ``.env``/environ read: under a multiplexed turn the launch env is not this profile's."""
from hermes_cli.config import get_env_value_prefer_dotenv
return (get_env_value_prefer_dotenv(key) or "").strip()
def _deepinfra_catalog_url() -> tuple[str, str]:
"""Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint."""
base = os.getenv("DEEPINFRA_BASE_URL", "").strip() or _DEEPINFRA_DEFAULT_BASE_URL
cache_key = base.rstrip("/")
return cache_key, f"{cache_key}/models?{_DEEPINFRA_MODELS_QUERY}"
"""Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint. The key carries the
api-key fingerprint: the catalog is user-scoped (private fine-tunes), so two profiles with
different keys must not share an entry."""
base = (_deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL).rstrip("/")
from agent.credential_persistence import fingerprint_secret_value
fp = fingerprint_secret_value(_deepinfra_env("DEEPINFRA_API_KEY")) or "anon"
return f"{base}#{fp}", f"{base}/models?{_DEEPINFRA_MODELS_QUERY}"
def _fetch_deepinfra_catalog(
@@ -2375,7 +2404,7 @@ def _fetch_deepinfra_catalog(
return None
headers: dict[str, str] = {"User-Agent": _HERMES_USER_AGENT}
api_key = os.getenv("DEEPINFRA_API_KEY", "").strip()
api_key = _deepinfra_env("DEEPINFRA_API_KEY")
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
try:
@@ -2435,7 +2464,7 @@ def deepinfra_model_ids(tag: str, *, force_refresh: bool = False) -> list[str]:
def deepinfra_base_url(section: Optional[dict] = None) -> str:
"""DeepInfra base URL: config-section ``base_url`` → ``DEEPINFRA_BASE_URL`` env → default; stripped."""
candidate = section.get("base_url") if isinstance(section, dict) else None
value = candidate or os.getenv("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL
value = candidate or _deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL
return str(value).strip().rstrip("/")
+32
View File
@@ -0,0 +1,32 @@
"""Per-profile view of ``hermes_cli.models``' module-level catalog slots.
Several caches on the facade (curated OpenRouter list, reasoning-capability catalogs and their
once-per-process guards) hold values derived from ONE profile's config, ``.env`` and ``<home>/cache``
files. In a multiplexed gateway every turn runs under a HERMES_HOME override, so a single module slot
would hand the launch profile's value to every other profile. Under an override the slot is read and
written per home key (routed profiles start cold, never from the launch profile's warmed value);
without one the module attribute stays the slot, so single-profile behaviour and the tests that reset
``models._X = None`` are untouched. Same shape as ``tools.approval._permanent_set``.
"""
from __future__ import annotations
from typing import Any
from hermes_constants import get_hermes_home_override, hermes_home_key
_SLOTS_BY_HOME: dict[tuple[str, str], Any] = {}
def profile_slot_get(module: Any, attr: str, default: Any = None) -> Any:
"""``module.<attr>`` for the active profile; ``default`` is a routed profile's cold value."""
if get_hermes_home_override() is None:
return getattr(module, attr)
return _SLOTS_BY_HOME.get((hermes_home_key(), attr), default)
def profile_slot_set(module: Any, attr: str, value: Any) -> None:
if get_hermes_home_override() is None:
setattr(module, attr, value)
else:
_SLOTS_BY_HOME[(hermes_home_key(), attr)] = value
+17 -4
View File
@@ -14,6 +14,7 @@ malformed).
from __future__ import annotations
import contextvars
import json
import logging
import os
@@ -109,7 +110,8 @@ def _warm_reasoning_caps_async(refresh) -> None:
Callers own the once-per-process guard; the fetch keeps its own failure TTL."""
if os.environ.get("PYTEST_CURRENT_TEST"):
return
threading.Thread(target=refresh, name="reasoning-caps-warm", daemon=True).start()
# copy_context: the Portal URL and disk mirror are the calling profile's, not the launch home's.
threading.Thread(target=contextvars.copy_context().run, args=(refresh,), name="reasoning-caps-warm", daemon=True).start()
def _hydrate_reasoning_caps_from_disk(url: str, refresh) -> Optional[Caps]:
@@ -170,12 +172,23 @@ class _CapsSource:
disk_checked: str
warm_started: str
url: Callable[[], str]
# True when the URL (hence the catalog) follows the active profile's credentials/.env: under a
# routed profile the slots then live per home, and the once-per-process guards must not let the
# launch profile's disk hydrate or warm count as another profile's.
per_profile: bool = False
def get(self, slot: str):
return getattr(_origin(), getattr(self, slot))
if not self.per_profile:
return getattr(_origin(), getattr(self, slot))
from hermes_cli.models_profile_cache import profile_slot_get
return profile_slot_get(_origin(), getattr(self, slot), False if slot in ("disk_checked", "warm_started") else None)
def set(self, slot: str, value) -> None:
setattr(_origin(), getattr(self, slot), value)
if not self.per_profile:
setattr(_origin(), getattr(self, slot), value)
return
from hermes_cli.models_profile_cache import profile_slot_set
profile_slot_set(_origin(), getattr(self, slot), value)
def _fetch_caps(src: _CapsSource, timeout: float = 6.0, *, force: bool = False) -> Optional[Caps]:
@@ -250,7 +263,7 @@ _OPENROUTER_CAPS = _CapsSource(
_NOUS_CAPS = _CapsSource(
"_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at",
"_nous_caps_disk_checked", "_nous_caps_warm_started",
lambda: nous_catalog_url(),
lambda: nous_catalog_url(), per_profile=True,
)
+35 -1
View File
@@ -342,6 +342,23 @@ _BUILTIN_SKINS: Dict[str, Dict[str, Any]] = {
_active_skin: Optional[SkinConfig] = None
_active_skin_name: str = "default"
# Routed multiplex profiles: (name, skin) per home key. ``display.skin`` and ``<home>/skins/*.yaml``
# are per profile, and the relay display name / TUI skin payload are read under each profile's
# override — one module slot would be last-writer-wins across profiles. Unscoped keeps the module slot.
_active_skin_by_home: Dict[str, Tuple[str, SkinConfig]] = {}
def _routed_home_key() -> Optional[str]:
from hermes_constants import get_hermes_home_override, hermes_home_key
return None if get_hermes_home_override() is None else hermes_home_key()
def _profile_config() -> dict:
try:
from hermes_cli.config import load_config_readonly
return load_config_readonly() or {}
except Exception:
return {}
def _skins_dir() -> Path:
@@ -414,6 +431,14 @@ def load_skin(name: str) -> SkinConfig:
def get_active_skin() -> SkinConfig:
"""Currently active skin config (cached)."""
global _active_skin
home_key = _routed_home_key()
if home_key is not None:
entry = _active_skin_by_home.get(home_key)
if entry is None:
# Cold routed profile: its own ``display.skin`` (nobody ran init_skin_from_config for it).
init_skin_from_config(_profile_config())
entry = _active_skin_by_home[home_key]
return entry[1]
if _active_skin is None:
_active_skin = load_skin(_active_skin_name)
return _active_skin
@@ -422,12 +447,21 @@ def get_active_skin() -> SkinConfig:
def set_active_skin(name: str) -> SkinConfig:
"""Switch the active skin. Returns the new SkinConfig."""
global _active_skin, _active_skin_name
skin = load_skin(name)
home_key = _routed_home_key()
if home_key is not None:
_active_skin_by_home[home_key] = (name, skin)
return skin
_active_skin_name = name
_active_skin = load_skin(name)
_active_skin = skin
return _active_skin
def get_active_skin_name() -> str:
home_key = _routed_home_key()
if home_key is not None:
entry = _active_skin_by_home.get(home_key)
return entry[0] if entry else "default"
return _active_skin_name
@@ -0,0 +1,258 @@
"""Multiplexed gateway: hermes_cli's process-wide caches must not hand profile A's value to profile B.
Every test builds two real profile homes (config.yaml / .env / cache files that differ), warms a
cache under ``set_hermes_home_override(A)`` and reads under B. Only the HTTP transport is canned —
its payload depends on the Authorization header or URL so a leaked entry is observable.
"""
from __future__ import annotations
import io
import json
import os
import threading
import time
import httpx
import pytest
from agent.secret_scope import build_profile_secret_scope, reset_secret_scope, set_secret_scope
from hermes_constants import get_hermes_home, reset_hermes_home_override, set_hermes_home_override
class _Resp(io.BytesIO):
def __enter__(self):
return self
def __exit__(self, *a):
return False
def _json_resp(payload) -> _Resp:
return _Resp(json.dumps(payload).encode())
class _Scoped:
"""Run a block as one profile's multiplexed turn (home override + its .env secret scope)."""
def __init__(self, home):
self.home = home
def __enter__(self):
self._t = set_hermes_home_override(self.home)
self._s = set_secret_scope(build_profile_secret_scope(self.home))
return self
def __exit__(self, *a):
reset_secret_scope(self._s)
reset_hermes_home_override(self._t)
@pytest.fixture
def homes(tmp_path, monkeypatch):
a = tmp_path / "hermes"
b = a / "profiles" / "B"
for home in (a, b):
(home / "cache").mkdir(parents=True)
monkeypatch.setenv("HERMES_HOME", str(a))
for var in ("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL", "NOUS_INFERENCE_BASE_URL"):
monkeypatch.delenv(var, raising=False)
return a, b
def test_deepinfra_catalog_is_fetched_with_each_profiles_key(homes, monkeypatch):
a, b = homes
(a / ".env").write_text("DEEPINFRA_API_KEY=key-A\n", encoding="utf-8")
(b / ".env").write_text("DEEPINFRA_API_KEY=key-B\n", encoding="utf-8")
import hermes_cli.models as models
monkeypatch.setattr(models, "_deepinfra_catalog_cache", {})
monkeypatch.setattr(models, "_deepinfra_catalog_neg_cache", {})
def transport(req, *, timeout, **kw):
who = req.headers.get("Authorization", "").rsplit("-", 1)[-1] or "anon"
return _json_resp({"data": [{"id": f"di/model-{who}", "metadata": {"tags": ["chat"]}}]})
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
with _Scoped(a):
assert models._fetch_deepinfra_models() == ["di/model-A"]
with _Scoped(b):
assert models._fetch_deepinfra_models() == ["di/model-B"]
def test_copilot_context_cache_hit_requires_same_api_key(homes, monkeypatch):
import hermes_cli.models as models
monkeypatch.setattr(models, "_copilot_context_cache", {})
monkeypatch.setattr(models, "_copilot_context_cache_time", 0.0)
monkeypatch.setattr(models, "_github_model_catalog_cache", None)
def transport(req, *, timeout, **kw):
limit = 111 if req.headers.get("Authorization", "").endswith("copilot-A") else 222
return _json_resp({"data": [{"id": "gpt-x", "model_picker_enabled": True,
"supported_endpoints": ["/chat/completions"],
"capabilities": {"type": "chat", "limits": {"max_prompt_tokens": limit}}}]})
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111
assert models.get_copilot_model_context("gpt-x", api_key="copilot-B") == 222
assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111
def test_nous_reasoning_caps_follow_each_profiles_portal(homes, monkeypatch):
a, b = homes
(a / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-a.example/v1\n", encoding="utf-8")
(b / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-b.example/v1\n", encoding="utf-8")
import hermes_cli.models as models
import hermes_cli.models_reasoning_caps as caps
for attr, value in (("_nous_reasoning_caps_cache", None), ("_nous_reasoning_caps_failed_at", None),
("_nous_caps_disk_checked", False), ("_nous_caps_warm_started", False)):
monkeypatch.setattr(models, attr, value)
def transport(req, *, timeout, **kw):
effort = "low" if "portal-a" in req.full_url else "high"
return _json_resp({"data": [{"id": "nous/m", "supported_parameters": ["reasoning"],
"reasoning": {"supported_efforts": [effort]}}]})
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
with _Scoped(a):
assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["low"]
with _Scoped(b):
assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["high"]
def test_swr_refresh_runs_as_the_profile_that_spawned_it(homes):
a, b = homes
import hermes_cli.models as models
seen: dict[str, str] = {}
done = threading.Event()
def refresh():
seen["home"] = str(get_hermes_home())
done.set()
return {"fp": "fp", "at": time.time(), "models": ["m"]}
with _Scoped(b):
models._spawn_swr_refresh("custom:https://gw.example/v1#fp", refresh)
assert done.wait(5)
assert seen["home"] == str(b)
deadline = time.monotonic() + 5
while not (b / "provider_models_cache.json").exists() and time.monotonic() < deadline:
time.sleep(0.05)
assert (b / "provider_models_cache.json").exists()
assert not (a / "provider_models_cache.json").exists()
def _write_manifest(home, model_id: str, mtime: float) -> None:
path = home / "cache" / "model_catalog.json"
path.write_text(json.dumps({"version": 1, "providers": {"openrouter": {"models": [
{"id": model_id, "description": "x", "default": True}]}}}), encoding="utf-8")
os.utime(path, (mtime, mtime))
def test_model_catalog_in_process_copy_is_bound_to_its_cache_file(homes, monkeypatch):
a, b = homes
import hermes_cli.model_catalog as mc
for home in (a, b):
(home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8")
same_mtime = time.time() - 5 # identical mtimes: only the path can tell the two files apart
_write_manifest(a, "vendor/a-model", same_mtime)
_write_manifest(b, "vendor/b-model", same_mtime)
mc.reset_cache()
with _Scoped(a):
assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/a-model"]
with _Scoped(b):
assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/b-model"]
assert mc.get_default_model_from_cache("openrouter") == "vendor/b-model"
def test_openrouter_curated_list_is_per_profile(homes, monkeypatch):
a, b = homes
import hermes_cli.models as models
for home in (a, b):
(home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8")
now = time.time()
for home, mid in ((a, "vendor/a-model"), (b, "vendor/b-model")):
(home / "cache" / "openrouter_curated_catalog.json").write_text(
json.dumps({"fetched_at": now, "curated": [[mid, "free"]]}), encoding="utf-8")
monkeypatch.setattr(models, "_openrouter_catalog_cache", None)
with _Scoped(a):
assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/a-model"]
with _Scoped(b):
assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/b-model"]
def test_banner_skills_are_the_routed_profiles(homes):
a, b = homes
import hermes_cli.banner as banner
for home, tag in ((a, "a"), (b, "b")):
skill = home / "skills" / f"skill_{tag}"
skill.mkdir(parents=True)
(skill / "SKILL.md").write_text(f"---\nname: skill_{tag}\ndescription: {tag}\n---\nbody\n", encoding="utf-8")
banner._available_skills_cache = None
try:
with _Scoped(a):
assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_a"]
with _Scoped(b):
assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_b"]
finally:
banner._available_skills_cache = None
def test_failed_guest_mint_only_suppresses_that_profile(homes, monkeypatch, tmp_path):
a, b = homes
monkeypatch.setenv("HERMES_GUEST_ONBOARDING", "1")
monkeypatch.setenv("HERMES_SHARED_AUTH_DIR", str(tmp_path / "shared"))
import hermes_cli.anon_auth as anon
import hermes_cli.auth_nous as auth_nous
monkeypatch.setattr(anon, "_mint_failed", False)
monkeypatch.setattr(anon, "_mint_failed_homes", set(), raising=False)
status = {"code": 429}
attempts: list[str] = []
def client(timeout_seconds, verify):
def handler(request):
attempts.append(str(get_hermes_home()))
if status["code"] == 429:
return httpx.Response(429, json={"error": "rate"})
return httpx.Response(201, json={"token": "anon_b", "user_id": "u", "org_id": "o"})
return httpx.Client(transport=httpx.MockTransport(handler))
monkeypatch.setattr(auth_nous, "_nous_http_client", client)
with _Scoped(a), pytest.raises(Exception):
anon.ensure_portal_identity(explicit=True, timeout_seconds=1)
assert attempts == [str(a)]
status["code"] = 201
with _Scoped(b):
assert anon.ensure_portal_identity(explicit=True, timeout_seconds=1) is not None
assert attempts[-1] == str(b)
def test_active_skin_is_per_profile_and_leaves_launch_slot_alone(homes):
a, b = homes
from hermes_cli import skin_engine
(a / "config.yaml").write_text("display:\n skin: ares\n", encoding="utf-8")
(b / "config.yaml").write_text("display:\n skin: mono\n", encoding="utf-8")
skin_engine._active_skin = None
skin_engine._active_skin_name = "default"
getattr(skin_engine, "_active_skin_by_home", {}).clear()
try:
with _Scoped(a):
skin_engine.init_skin_from_config({"display": {"skin": "ares"}})
assert skin_engine.get_active_skin().name == "ares"
with _Scoped(b):
assert skin_engine.get_active_skin().name == "mono" # B's own display.skin, never A's
with _Scoped(a):
assert skin_engine.get_active_skin().name == "ares"
assert skin_engine.get_active_skin_name() == "default" # routed turns never touch the launch slot
finally:
skin_engine._active_skin = None
skin_engine._active_skin_name = "default"
getattr(skin_engine, "_active_skin_by_home", {}).clear()