fix(multiplex): per-profile catalog, skin and guest-mint state in hermes_cli
DeepInfra catalog (fetched with the launch env's key via os.getenv), Copilot context limits (api_key ignored on hit), Nous reasoning caps + once-per-process guards, the curated OpenRouter list, the model-catalog in-process copy (mtime without path), banner skills, the guest-mint back-off flag and the active skin were single slots read under per-profile overrides by the gateway and the TUI gateway; the SWR refresh thread ran without the caller's ContextVars. Under an override each lives per home key (hermes_cli/models_profile_cache.py holds the shared slot helper so models.py does not grow), credentials are read through the scope-aware dotenv reader and keyed by fingerprint, and background refreshes run under copy_context(). Unscoped behaviour is byte-identical.
This commit is contained in:
+25
-6
@@ -284,7 +284,28 @@ def _mint_locked(
|
||||
|
||||
# Per-process memo: one failed mint is enough for a process (a 429 or a closed gate must not be hit
|
||||
# twice); ``clear_dead_guest`` resets it because a retired credential is a reason to mint again.
|
||||
# The bool is the unscoped (launch profile) slot; routed multiplex profiles each get their own entry
|
||||
# in the set — profile A's 429 must not stop profile B from ever getting an identity.
|
||||
_mint_failed = False
|
||||
_mint_failed_homes: set[str] = set()
|
||||
|
||||
|
||||
def _mint_failed_for_profile() -> bool:
|
||||
from hermes_constants import get_hermes_home_override, hermes_home_key
|
||||
if get_hermes_home_override() is None:
|
||||
return _mint_failed
|
||||
return hermes_home_key() in _mint_failed_homes
|
||||
|
||||
|
||||
def _set_mint_failed(failed: bool) -> None:
|
||||
global _mint_failed
|
||||
from hermes_constants import get_hermes_home_override, hermes_home_key
|
||||
if get_hermes_home_override() is None:
|
||||
_mint_failed = failed
|
||||
elif failed:
|
||||
_mint_failed_homes.add(hermes_home_key())
|
||||
else:
|
||||
_mint_failed_homes.discard(hermes_home_key())
|
||||
|
||||
|
||||
def _reconcile_and_provision(*, timeout_seconds: float, carries_inference: bool = True) -> Optional[Dict[str, Any]]:
|
||||
@@ -346,16 +367,15 @@ def ensure_portal_identity(
|
||||
"""
|
||||
if not explicit:
|
||||
raise ValueError("ensure_portal_identity: only explicit creators may call this (explicit=True)")
|
||||
global _mint_failed
|
||||
if not guest_enabled():
|
||||
return None
|
||||
if _mint_failed and not current_nous_state():
|
||||
return None # this process already tried and failed; do not hammer the portal
|
||||
if _mint_failed_for_profile() and not current_nous_state():
|
||||
return None # this profile already tried and failed in this process; do not hammer the portal
|
||||
try:
|
||||
return _reconcile_and_provision(
|
||||
timeout_seconds=timeout_seconds, carries_inference=carries_inference)
|
||||
except Exception:
|
||||
_mint_failed = True
|
||||
_set_mint_failed(True)
|
||||
raise
|
||||
|
||||
|
||||
@@ -401,8 +421,7 @@ def clear_dead_guest(reason: str, *, dead_token: Optional[str] = None) -> None:
|
||||
shared = _read_shared_nous_state()
|
||||
if token and is_guest_state(shared) and shared.get("anon_token") == token:
|
||||
_clear_shared_nous_state(reason)
|
||||
global _mint_failed
|
||||
_mint_failed = False
|
||||
_set_mint_failed(False)
|
||||
logger.info("Nous free-tier identity retired (%s); a new one is set up on next use", reason)
|
||||
|
||||
|
||||
|
||||
@@ -92,7 +92,13 @@ _UNCACHED = object() # compute() result that must not be memoized
|
||||
|
||||
|
||||
def _memo(cache_name: str, compute):
|
||||
"""Return the cached value under module global ``cache_name``, computing (and storing) it once."""
|
||||
"""Return the cached value under module global ``cache_name``, computing (and storing) it once.
|
||||
|
||||
Not consulted under a routed profile (HERMES_HOME override): every memo here is derived from the
|
||||
launch home (its skills tree, its checkout), and the TUI gateway calls these per profile."""
|
||||
from hermes_constants import get_hermes_home_override
|
||||
if get_hermes_home_override() is not None:
|
||||
return compute()
|
||||
cached = globals()[cache_name]
|
||||
if cached is not None:
|
||||
return cached[0]
|
||||
|
||||
@@ -37,9 +37,12 @@ SUPPORTED_SCHEMA_VERSION = 1
|
||||
|
||||
_HERMES_USER_AGENT = f"hermes-cli/{_HERMES_VERSION}"
|
||||
|
||||
# In-process cache, invalidated against the disk file's mtime and TTL.
|
||||
# In-process cache, invalidated against the disk file's path + mtime and TTL. The path matters:
|
||||
# under a multiplexed gateway each profile has its own ``<home>/cache/model_catalog.json``, and
|
||||
# mtime alone cannot tell two profiles' files apart.
|
||||
_catalog_cache: dict[str, Any] | None = None
|
||||
_catalog_cache_source_mtime: float = 0.0
|
||||
_catalog_cache_source_path: str = ""
|
||||
|
||||
|
||||
def _load_catalog_config() -> dict[str, Any]:
|
||||
@@ -193,11 +196,19 @@ def _spawn_catalog_swr_refresh(url: str) -> None:
|
||||
|
||||
|
||||
def _remember(data: dict[str, Any], mtime: float) -> dict[str, Any]:
|
||||
global _catalog_cache, _catalog_cache_source_mtime
|
||||
global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path
|
||||
_catalog_cache, _catalog_cache_source_mtime = data, mtime
|
||||
_catalog_cache_source_path = str(_cache_path())
|
||||
return data
|
||||
|
||||
|
||||
def _in_process_catalog() -> dict[str, Any] | None:
|
||||
"""The in-process copy when it mirrors the ACTIVE profile's cache file, else None."""
|
||||
if _catalog_cache is not None and _catalog_cache_source_path == str(_cache_path()):
|
||||
return _catalog_cache
|
||||
return None
|
||||
|
||||
|
||||
def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]:
|
||||
"""Parsed model catalog manifest, or ``{}`` on failure — never raises, so the CLI works offline
|
||||
(callers treat a missing provider/model as "use the in-repo fallback")."""
|
||||
@@ -210,8 +221,9 @@ def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]:
|
||||
disk_fresh = disk_data is not None and (now - disk_mtime) < ttl_seconds
|
||||
|
||||
if not force_refresh and disk_data is not None:
|
||||
if disk_fresh and _catalog_cache is not None and disk_mtime == _catalog_cache_source_mtime:
|
||||
return _catalog_cache
|
||||
cached = _in_process_catalog()
|
||||
if disk_fresh and cached is not None and disk_mtime == _catalog_cache_source_mtime:
|
||||
return cached
|
||||
if not disk_fresh:
|
||||
# Stale-while-revalidate: serve the expired disk copy now and refresh off-thread so the
|
||||
# /model picker (which calls this on every open) never blocks on the manifest fetch.
|
||||
@@ -303,7 +315,8 @@ def _default_model_from_block(block: dict[str, Any] | None) -> str | None:
|
||||
def get_default_model_from_cache(provider: str) -> str | None:
|
||||
"""The manifest's labeled default for ``provider`` (the model Hermes silently lands on when the
|
||||
user never picked one) — in-process then disk cache only, never a fetch."""
|
||||
found = _default_model_from_block(_block_of(_catalog_cache, provider)) if _catalog_cache is not None else None
|
||||
cached = _in_process_catalog()
|
||||
found = _default_model_from_block(_block_of(cached, provider)) if cached is not None else None
|
||||
if found:
|
||||
return found
|
||||
disk_data, _mtime = _read_disk_cache()
|
||||
@@ -331,6 +344,7 @@ def seed_cache_from_checkout(project_root: "Path | str") -> bool:
|
||||
|
||||
def reset_cache() -> None:
|
||||
"""Clear the in-process cache. Used by tests and ``hermes model --refresh``."""
|
||||
global _catalog_cache, _catalog_cache_source_mtime
|
||||
global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path
|
||||
_catalog_cache = None
|
||||
_catalog_cache_source_mtime = 0.0
|
||||
_catalog_cache_source_path = ""
|
||||
|
||||
+48
-19
@@ -8,11 +8,13 @@ Origin module; cohesive clusters live in siblings and are re-imported here so
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextvars
|
||||
import copy
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import threading
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
@@ -542,17 +544,21 @@ def _fetch_live_catalog_index(url: str, timeout: float, opener) -> Optional[tupl
|
||||
def fetch_openrouter_models(
|
||||
timeout: float = 8.0, *, force_refresh: bool = False) -> list[tuple[str, str]]:
|
||||
"""Return the curated OpenRouter picker list, refreshed from the live catalog when possible."""
|
||||
global _openrouter_catalog_cache
|
||||
# The curated list is filtered from this profile's manifest (``model_catalog.*`` config, its
|
||||
# ``<home>/cache`` copy), so a routed profile keeps its own slot instead of the module one.
|
||||
from hermes_cli.models_profile_cache import profile_slot_get, profile_slot_set
|
||||
_me = sys.modules[__name__]
|
||||
cached = profile_slot_get(_me, "_openrouter_catalog_cache")
|
||||
|
||||
if _openrouter_catalog_cache is not None and not force_refresh:
|
||||
return list(_openrouter_catalog_cache)
|
||||
if cached is not None and not force_refresh:
|
||||
return list(cached)
|
||||
|
||||
# Cold process: serve from the persisted disk cache when fresh so the
|
||||
# picker doesn't re-download the full ~686KB catalog on every open.
|
||||
if not force_refresh:
|
||||
disk = _read_openrouter_catalog_disk()
|
||||
if disk:
|
||||
_openrouter_catalog_cache = disk
|
||||
profile_slot_set(_me, "_openrouter_catalog_cache", disk)
|
||||
return list(disk)
|
||||
|
||||
# Remote catalog manifest first, in-repo snapshot when unreachable; the live /v1/models filter
|
||||
@@ -566,7 +572,7 @@ def fetch_openrouter_models(
|
||||
|
||||
live = _fetch_live_catalog_index(_OPENROUTER_CATALOG_URL, timeout, _urlopen_model_catalog_request)
|
||||
if live is None:
|
||||
return list(_openrouter_catalog_cache or fallback)
|
||||
return list(cached or fallback)
|
||||
live_items, live_by_id = live
|
||||
|
||||
# Free warm-up for the reasoning-capability cache: same payload the caps fetch would pull.
|
||||
@@ -592,10 +598,10 @@ def fetch_openrouter_models(
|
||||
curated.append((preferred_id, desc))
|
||||
|
||||
if not curated:
|
||||
return list(_openrouter_catalog_cache or fallback)
|
||||
return list(cached or fallback)
|
||||
if not curated[0][1]:
|
||||
curated[0] = (curated[0][0], "recommended")
|
||||
_openrouter_catalog_cache = curated
|
||||
profile_slot_set(_me, "_openrouter_catalog_cache", curated)
|
||||
_write_openrouter_catalog_disk(curated)
|
||||
return list(curated)
|
||||
|
||||
@@ -1479,10 +1485,15 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None:
|
||||
Failures are swallowed — the stale entry stays served until a later refresh succeeds.
|
||||
``refresh_fn`` (no-args → fresh entry dict or None) lets ``custom:<base_url>`` keys from
|
||||
:func:`cached_fetch_api_models` reuse the same inflight-dedupe scaffolding."""
|
||||
# Under a routed profile the inflight key includes the home: the same provider slug names a
|
||||
# different disk cache and credential set per profile, so one profile's refresh must not
|
||||
# suppress another's. Unscoped keeps the bare key (tests inspect the set by slug).
|
||||
from hermes_constants import get_hermes_home_override, hermes_home_key
|
||||
inflight_key = cache_key if get_hermes_home_override() is None else (hermes_home_key(), cache_key)
|
||||
with _swr_refresh_lock:
|
||||
if cache_key in _swr_refresh_inflight:
|
||||
if inflight_key in _swr_refresh_inflight:
|
||||
return
|
||||
_swr_refresh_inflight.add(cache_key)
|
||||
_swr_refresh_inflight.add(inflight_key)
|
||||
|
||||
def _default_refresh():
|
||||
live = provider_model_ids(cache_key, force_refresh=True)
|
||||
@@ -1499,9 +1510,11 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None:
|
||||
logger.debug("SWR refresh failed for %s", cache_key, exc_info=True)
|
||||
finally:
|
||||
with _swr_refresh_lock:
|
||||
_swr_refresh_inflight.discard(cache_key)
|
||||
_swr_refresh_inflight.discard(inflight_key)
|
||||
|
||||
threading.Thread(target=_refresh, daemon=True, name=f"model-cache-swr-{cache_key}").start()
|
||||
# copy_context: the refresh must read the calling profile's credentials and write ITS disk cache.
|
||||
ctx = contextvars.copy_context()
|
||||
threading.Thread(target=lambda: ctx.run(_refresh), daemon=True, name=f"model-cache-swr-{cache_key}").start()
|
||||
|
||||
|
||||
def _provider_models_cache_path() -> Path:
|
||||
@@ -1893,15 +1906,21 @@ def fetch_github_model_catalog(
|
||||
# Module-level cache: {model_id: max_prompt_tokens}
|
||||
_copilot_context_cache: dict[str, int] = {}
|
||||
_copilot_context_cache_time: float = 0.0
|
||||
_copilot_context_cache_key: Optional[str] = None # fingerprint of the api_key the entry was fetched with
|
||||
_COPILOT_CONTEXT_CACHE_TTL = 3600 # 1 hour
|
||||
|
||||
|
||||
def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> Optional[int]:
|
||||
"""``max_prompt_tokens`` for a Copilot model from the live /models API (cached in-process 1h; a
|
||||
miss on a fresh cache does not re-fetch), or None."""
|
||||
global _copilot_context_cache, _copilot_context_cache_time
|
||||
global _copilot_context_cache, _copilot_context_cache_time, _copilot_context_cache_key
|
||||
|
||||
if _copilot_context_cache and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL):
|
||||
# Keyed on the credential like fetch_github_model_catalog: the catalog (and its limits) is
|
||||
# per-account, so another profile's token must not be served this entry.
|
||||
from agent.credential_persistence import fingerprint_secret_value
|
||||
key_fp = fingerprint_secret_value(api_key)
|
||||
if (_copilot_context_cache and _copilot_context_cache_key == key_fp
|
||||
and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL)):
|
||||
return _copilot_context_cache.get(model_id)
|
||||
|
||||
catalog = fetch_github_model_catalog(api_key=api_key)
|
||||
@@ -1915,6 +1934,7 @@ def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> O
|
||||
cache[mid] = max_prompt
|
||||
_copilot_context_cache = cache
|
||||
_copilot_context_cache_time = time.time()
|
||||
_copilot_context_cache_key = key_fp
|
||||
return cache.get(model_id)
|
||||
|
||||
|
||||
@@ -2355,11 +2375,20 @@ _deepinfra_catalog_neg_cache: dict[str, float] = {}
|
||||
_DEEPINFRA_CATALOG_NEG_TTL = 60.0 # seconds
|
||||
|
||||
|
||||
def _deepinfra_env(key: str) -> str:
|
||||
"""Profile-scoped ``.env``/environ read: under a multiplexed turn the launch env is not this profile's."""
|
||||
from hermes_cli.config import get_env_value_prefer_dotenv
|
||||
return (get_env_value_prefer_dotenv(key) or "").strip()
|
||||
|
||||
|
||||
def _deepinfra_catalog_url() -> tuple[str, str]:
|
||||
"""Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint."""
|
||||
base = os.getenv("DEEPINFRA_BASE_URL", "").strip() or _DEEPINFRA_DEFAULT_BASE_URL
|
||||
cache_key = base.rstrip("/")
|
||||
return cache_key, f"{cache_key}/models?{_DEEPINFRA_MODELS_QUERY}"
|
||||
"""Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint. The key carries the
|
||||
api-key fingerprint: the catalog is user-scoped (private fine-tunes), so two profiles with
|
||||
different keys must not share an entry."""
|
||||
base = (_deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL).rstrip("/")
|
||||
from agent.credential_persistence import fingerprint_secret_value
|
||||
fp = fingerprint_secret_value(_deepinfra_env("DEEPINFRA_API_KEY")) or "anon"
|
||||
return f"{base}#{fp}", f"{base}/models?{_DEEPINFRA_MODELS_QUERY}"
|
||||
|
||||
|
||||
def _fetch_deepinfra_catalog(
|
||||
@@ -2375,7 +2404,7 @@ def _fetch_deepinfra_catalog(
|
||||
return None
|
||||
|
||||
headers: dict[str, str] = {"User-Agent": _HERMES_USER_AGENT}
|
||||
api_key = os.getenv("DEEPINFRA_API_KEY", "").strip()
|
||||
api_key = _deepinfra_env("DEEPINFRA_API_KEY")
|
||||
if api_key:
|
||||
headers["Authorization"] = f"Bearer {api_key}"
|
||||
try:
|
||||
@@ -2435,7 +2464,7 @@ def deepinfra_model_ids(tag: str, *, force_refresh: bool = False) -> list[str]:
|
||||
def deepinfra_base_url(section: Optional[dict] = None) -> str:
|
||||
"""DeepInfra base URL: config-section ``base_url`` → ``DEEPINFRA_BASE_URL`` env → default; stripped."""
|
||||
candidate = section.get("base_url") if isinstance(section, dict) else None
|
||||
value = candidate or os.getenv("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL
|
||||
value = candidate or _deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL
|
||||
return str(value).strip().rstrip("/")
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Per-profile view of ``hermes_cli.models``' module-level catalog slots.
|
||||
|
||||
Several caches on the facade (curated OpenRouter list, reasoning-capability catalogs and their
|
||||
once-per-process guards) hold values derived from ONE profile's config, ``.env`` and ``<home>/cache``
|
||||
files. In a multiplexed gateway every turn runs under a HERMES_HOME override, so a single module slot
|
||||
would hand the launch profile's value to every other profile. Under an override the slot is read and
|
||||
written per home key (routed profiles start cold, never from the launch profile's warmed value);
|
||||
without one the module attribute stays the slot, so single-profile behaviour and the tests that reset
|
||||
``models._X = None`` are untouched. Same shape as ``tools.approval._permanent_set``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from hermes_constants import get_hermes_home_override, hermes_home_key
|
||||
|
||||
_SLOTS_BY_HOME: dict[tuple[str, str], Any] = {}
|
||||
|
||||
|
||||
def profile_slot_get(module: Any, attr: str, default: Any = None) -> Any:
|
||||
"""``module.<attr>`` for the active profile; ``default`` is a routed profile's cold value."""
|
||||
if get_hermes_home_override() is None:
|
||||
return getattr(module, attr)
|
||||
return _SLOTS_BY_HOME.get((hermes_home_key(), attr), default)
|
||||
|
||||
|
||||
def profile_slot_set(module: Any, attr: str, value: Any) -> None:
|
||||
if get_hermes_home_override() is None:
|
||||
setattr(module, attr, value)
|
||||
else:
|
||||
_SLOTS_BY_HOME[(hermes_home_key(), attr)] = value
|
||||
@@ -14,6 +14,7 @@ malformed).
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextvars
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
@@ -109,7 +110,8 @@ def _warm_reasoning_caps_async(refresh) -> None:
|
||||
Callers own the once-per-process guard; the fetch keeps its own failure TTL."""
|
||||
if os.environ.get("PYTEST_CURRENT_TEST"):
|
||||
return
|
||||
threading.Thread(target=refresh, name="reasoning-caps-warm", daemon=True).start()
|
||||
# copy_context: the Portal URL and disk mirror are the calling profile's, not the launch home's.
|
||||
threading.Thread(target=contextvars.copy_context().run, args=(refresh,), name="reasoning-caps-warm", daemon=True).start()
|
||||
|
||||
|
||||
def _hydrate_reasoning_caps_from_disk(url: str, refresh) -> Optional[Caps]:
|
||||
@@ -170,12 +172,23 @@ class _CapsSource:
|
||||
disk_checked: str
|
||||
warm_started: str
|
||||
url: Callable[[], str]
|
||||
# True when the URL (hence the catalog) follows the active profile's credentials/.env: under a
|
||||
# routed profile the slots then live per home, and the once-per-process guards must not let the
|
||||
# launch profile's disk hydrate or warm count as another profile's.
|
||||
per_profile: bool = False
|
||||
|
||||
def get(self, slot: str):
|
||||
return getattr(_origin(), getattr(self, slot))
|
||||
if not self.per_profile:
|
||||
return getattr(_origin(), getattr(self, slot))
|
||||
from hermes_cli.models_profile_cache import profile_slot_get
|
||||
return profile_slot_get(_origin(), getattr(self, slot), False if slot in ("disk_checked", "warm_started") else None)
|
||||
|
||||
def set(self, slot: str, value) -> None:
|
||||
setattr(_origin(), getattr(self, slot), value)
|
||||
if not self.per_profile:
|
||||
setattr(_origin(), getattr(self, slot), value)
|
||||
return
|
||||
from hermes_cli.models_profile_cache import profile_slot_set
|
||||
profile_slot_set(_origin(), getattr(self, slot), value)
|
||||
|
||||
|
||||
def _fetch_caps(src: _CapsSource, timeout: float = 6.0, *, force: bool = False) -> Optional[Caps]:
|
||||
@@ -250,7 +263,7 @@ _OPENROUTER_CAPS = _CapsSource(
|
||||
_NOUS_CAPS = _CapsSource(
|
||||
"_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at",
|
||||
"_nous_caps_disk_checked", "_nous_caps_warm_started",
|
||||
lambda: nous_catalog_url(),
|
||||
lambda: nous_catalog_url(), per_profile=True,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -342,6 +342,23 @@ _BUILTIN_SKINS: Dict[str, Dict[str, Any]] = {
|
||||
|
||||
_active_skin: Optional[SkinConfig] = None
|
||||
_active_skin_name: str = "default"
|
||||
# Routed multiplex profiles: (name, skin) per home key. ``display.skin`` and ``<home>/skins/*.yaml``
|
||||
# are per profile, and the relay display name / TUI skin payload are read under each profile's
|
||||
# override — one module slot would be last-writer-wins across profiles. Unscoped keeps the module slot.
|
||||
_active_skin_by_home: Dict[str, Tuple[str, SkinConfig]] = {}
|
||||
|
||||
|
||||
def _routed_home_key() -> Optional[str]:
|
||||
from hermes_constants import get_hermes_home_override, hermes_home_key
|
||||
return None if get_hermes_home_override() is None else hermes_home_key()
|
||||
|
||||
|
||||
def _profile_config() -> dict:
|
||||
try:
|
||||
from hermes_cli.config import load_config_readonly
|
||||
return load_config_readonly() or {}
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def _skins_dir() -> Path:
|
||||
@@ -414,6 +431,14 @@ def load_skin(name: str) -> SkinConfig:
|
||||
def get_active_skin() -> SkinConfig:
|
||||
"""Currently active skin config (cached)."""
|
||||
global _active_skin
|
||||
home_key = _routed_home_key()
|
||||
if home_key is not None:
|
||||
entry = _active_skin_by_home.get(home_key)
|
||||
if entry is None:
|
||||
# Cold routed profile: its own ``display.skin`` (nobody ran init_skin_from_config for it).
|
||||
init_skin_from_config(_profile_config())
|
||||
entry = _active_skin_by_home[home_key]
|
||||
return entry[1]
|
||||
if _active_skin is None:
|
||||
_active_skin = load_skin(_active_skin_name)
|
||||
return _active_skin
|
||||
@@ -422,12 +447,21 @@ def get_active_skin() -> SkinConfig:
|
||||
def set_active_skin(name: str) -> SkinConfig:
|
||||
"""Switch the active skin. Returns the new SkinConfig."""
|
||||
global _active_skin, _active_skin_name
|
||||
skin = load_skin(name)
|
||||
home_key = _routed_home_key()
|
||||
if home_key is not None:
|
||||
_active_skin_by_home[home_key] = (name, skin)
|
||||
return skin
|
||||
_active_skin_name = name
|
||||
_active_skin = load_skin(name)
|
||||
_active_skin = skin
|
||||
return _active_skin
|
||||
|
||||
|
||||
def get_active_skin_name() -> str:
|
||||
home_key = _routed_home_key()
|
||||
if home_key is not None:
|
||||
entry = _active_skin_by_home.get(home_key)
|
||||
return entry[0] if entry else "default"
|
||||
return _active_skin_name
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,258 @@
|
||||
"""Multiplexed gateway: hermes_cli's process-wide caches must not hand profile A's value to profile B.
|
||||
|
||||
Every test builds two real profile homes (config.yaml / .env / cache files that differ), warms a
|
||||
cache under ``set_hermes_home_override(A)`` and reads under B. Only the HTTP transport is canned —
|
||||
its payload depends on the Authorization header or URL so a leaked entry is observable.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from agent.secret_scope import build_profile_secret_scope, reset_secret_scope, set_secret_scope
|
||||
from hermes_constants import get_hermes_home, reset_hermes_home_override, set_hermes_home_override
|
||||
|
||||
|
||||
class _Resp(io.BytesIO):
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
|
||||
def _json_resp(payload) -> _Resp:
|
||||
return _Resp(json.dumps(payload).encode())
|
||||
|
||||
|
||||
class _Scoped:
|
||||
"""Run a block as one profile's multiplexed turn (home override + its .env secret scope)."""
|
||||
|
||||
def __init__(self, home):
|
||||
self.home = home
|
||||
|
||||
def __enter__(self):
|
||||
self._t = set_hermes_home_override(self.home)
|
||||
self._s = set_secret_scope(build_profile_secret_scope(self.home))
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
reset_secret_scope(self._s)
|
||||
reset_hermes_home_override(self._t)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def homes(tmp_path, monkeypatch):
|
||||
a = tmp_path / "hermes"
|
||||
b = a / "profiles" / "B"
|
||||
for home in (a, b):
|
||||
(home / "cache").mkdir(parents=True)
|
||||
monkeypatch.setenv("HERMES_HOME", str(a))
|
||||
for var in ("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL", "NOUS_INFERENCE_BASE_URL"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
return a, b
|
||||
|
||||
|
||||
def test_deepinfra_catalog_is_fetched_with_each_profiles_key(homes, monkeypatch):
|
||||
a, b = homes
|
||||
(a / ".env").write_text("DEEPINFRA_API_KEY=key-A\n", encoding="utf-8")
|
||||
(b / ".env").write_text("DEEPINFRA_API_KEY=key-B\n", encoding="utf-8")
|
||||
import hermes_cli.models as models
|
||||
|
||||
monkeypatch.setattr(models, "_deepinfra_catalog_cache", {})
|
||||
monkeypatch.setattr(models, "_deepinfra_catalog_neg_cache", {})
|
||||
|
||||
def transport(req, *, timeout, **kw):
|
||||
who = req.headers.get("Authorization", "").rsplit("-", 1)[-1] or "anon"
|
||||
return _json_resp({"data": [{"id": f"di/model-{who}", "metadata": {"tags": ["chat"]}}]})
|
||||
|
||||
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
|
||||
with _Scoped(a):
|
||||
assert models._fetch_deepinfra_models() == ["di/model-A"]
|
||||
with _Scoped(b):
|
||||
assert models._fetch_deepinfra_models() == ["di/model-B"]
|
||||
|
||||
|
||||
def test_copilot_context_cache_hit_requires_same_api_key(homes, monkeypatch):
|
||||
import hermes_cli.models as models
|
||||
|
||||
monkeypatch.setattr(models, "_copilot_context_cache", {})
|
||||
monkeypatch.setattr(models, "_copilot_context_cache_time", 0.0)
|
||||
monkeypatch.setattr(models, "_github_model_catalog_cache", None)
|
||||
|
||||
def transport(req, *, timeout, **kw):
|
||||
limit = 111 if req.headers.get("Authorization", "").endswith("copilot-A") else 222
|
||||
return _json_resp({"data": [{"id": "gpt-x", "model_picker_enabled": True,
|
||||
"supported_endpoints": ["/chat/completions"],
|
||||
"capabilities": {"type": "chat", "limits": {"max_prompt_tokens": limit}}}]})
|
||||
|
||||
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
|
||||
assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111
|
||||
assert models.get_copilot_model_context("gpt-x", api_key="copilot-B") == 222
|
||||
assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111
|
||||
|
||||
|
||||
def test_nous_reasoning_caps_follow_each_profiles_portal(homes, monkeypatch):
|
||||
a, b = homes
|
||||
(a / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-a.example/v1\n", encoding="utf-8")
|
||||
(b / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-b.example/v1\n", encoding="utf-8")
|
||||
import hermes_cli.models as models
|
||||
import hermes_cli.models_reasoning_caps as caps
|
||||
|
||||
for attr, value in (("_nous_reasoning_caps_cache", None), ("_nous_reasoning_caps_failed_at", None),
|
||||
("_nous_caps_disk_checked", False), ("_nous_caps_warm_started", False)):
|
||||
monkeypatch.setattr(models, attr, value)
|
||||
|
||||
def transport(req, *, timeout, **kw):
|
||||
effort = "low" if "portal-a" in req.full_url else "high"
|
||||
return _json_resp({"data": [{"id": "nous/m", "supported_parameters": ["reasoning"],
|
||||
"reasoning": {"supported_efforts": [effort]}}]})
|
||||
|
||||
monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport)
|
||||
with _Scoped(a):
|
||||
assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["low"]
|
||||
with _Scoped(b):
|
||||
assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["high"]
|
||||
|
||||
|
||||
def test_swr_refresh_runs_as_the_profile_that_spawned_it(homes):
|
||||
a, b = homes
|
||||
import hermes_cli.models as models
|
||||
|
||||
seen: dict[str, str] = {}
|
||||
done = threading.Event()
|
||||
|
||||
def refresh():
|
||||
seen["home"] = str(get_hermes_home())
|
||||
done.set()
|
||||
return {"fp": "fp", "at": time.time(), "models": ["m"]}
|
||||
|
||||
with _Scoped(b):
|
||||
models._spawn_swr_refresh("custom:https://gw.example/v1#fp", refresh)
|
||||
assert done.wait(5)
|
||||
assert seen["home"] == str(b)
|
||||
deadline = time.monotonic() + 5
|
||||
while not (b / "provider_models_cache.json").exists() and time.monotonic() < deadline:
|
||||
time.sleep(0.05)
|
||||
assert (b / "provider_models_cache.json").exists()
|
||||
assert not (a / "provider_models_cache.json").exists()
|
||||
|
||||
|
||||
def _write_manifest(home, model_id: str, mtime: float) -> None:
|
||||
path = home / "cache" / "model_catalog.json"
|
||||
path.write_text(json.dumps({"version": 1, "providers": {"openrouter": {"models": [
|
||||
{"id": model_id, "description": "x", "default": True}]}}}), encoding="utf-8")
|
||||
os.utime(path, (mtime, mtime))
|
||||
|
||||
|
||||
def test_model_catalog_in_process_copy_is_bound_to_its_cache_file(homes, monkeypatch):
|
||||
a, b = homes
|
||||
import hermes_cli.model_catalog as mc
|
||||
|
||||
for home in (a, b):
|
||||
(home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8")
|
||||
same_mtime = time.time() - 5 # identical mtimes: only the path can tell the two files apart
|
||||
_write_manifest(a, "vendor/a-model", same_mtime)
|
||||
_write_manifest(b, "vendor/b-model", same_mtime)
|
||||
mc.reset_cache()
|
||||
with _Scoped(a):
|
||||
assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/a-model"]
|
||||
with _Scoped(b):
|
||||
assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/b-model"]
|
||||
assert mc.get_default_model_from_cache("openrouter") == "vendor/b-model"
|
||||
|
||||
|
||||
def test_openrouter_curated_list_is_per_profile(homes, monkeypatch):
|
||||
a, b = homes
|
||||
import hermes_cli.models as models
|
||||
|
||||
for home in (a, b):
|
||||
(home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8")
|
||||
now = time.time()
|
||||
for home, mid in ((a, "vendor/a-model"), (b, "vendor/b-model")):
|
||||
(home / "cache" / "openrouter_curated_catalog.json").write_text(
|
||||
json.dumps({"fetched_at": now, "curated": [[mid, "free"]]}), encoding="utf-8")
|
||||
monkeypatch.setattr(models, "_openrouter_catalog_cache", None)
|
||||
with _Scoped(a):
|
||||
assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/a-model"]
|
||||
with _Scoped(b):
|
||||
assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/b-model"]
|
||||
|
||||
|
||||
def test_banner_skills_are_the_routed_profiles(homes):
|
||||
a, b = homes
|
||||
import hermes_cli.banner as banner
|
||||
|
||||
for home, tag in ((a, "a"), (b, "b")):
|
||||
skill = home / "skills" / f"skill_{tag}"
|
||||
skill.mkdir(parents=True)
|
||||
(skill / "SKILL.md").write_text(f"---\nname: skill_{tag}\ndescription: {tag}\n---\nbody\n", encoding="utf-8")
|
||||
banner._available_skills_cache = None
|
||||
try:
|
||||
with _Scoped(a):
|
||||
assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_a"]
|
||||
with _Scoped(b):
|
||||
assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_b"]
|
||||
finally:
|
||||
banner._available_skills_cache = None
|
||||
|
||||
|
||||
def test_failed_guest_mint_only_suppresses_that_profile(homes, monkeypatch, tmp_path):
|
||||
a, b = homes
|
||||
monkeypatch.setenv("HERMES_GUEST_ONBOARDING", "1")
|
||||
monkeypatch.setenv("HERMES_SHARED_AUTH_DIR", str(tmp_path / "shared"))
|
||||
import hermes_cli.anon_auth as anon
|
||||
import hermes_cli.auth_nous as auth_nous
|
||||
|
||||
monkeypatch.setattr(anon, "_mint_failed", False)
|
||||
monkeypatch.setattr(anon, "_mint_failed_homes", set(), raising=False)
|
||||
status = {"code": 429}
|
||||
attempts: list[str] = []
|
||||
|
||||
def client(timeout_seconds, verify):
|
||||
def handler(request):
|
||||
attempts.append(str(get_hermes_home()))
|
||||
if status["code"] == 429:
|
||||
return httpx.Response(429, json={"error": "rate"})
|
||||
return httpx.Response(201, json={"token": "anon_b", "user_id": "u", "org_id": "o"})
|
||||
return httpx.Client(transport=httpx.MockTransport(handler))
|
||||
|
||||
monkeypatch.setattr(auth_nous, "_nous_http_client", client)
|
||||
with _Scoped(a), pytest.raises(Exception):
|
||||
anon.ensure_portal_identity(explicit=True, timeout_seconds=1)
|
||||
assert attempts == [str(a)]
|
||||
status["code"] = 201
|
||||
with _Scoped(b):
|
||||
assert anon.ensure_portal_identity(explicit=True, timeout_seconds=1) is not None
|
||||
assert attempts[-1] == str(b)
|
||||
|
||||
|
||||
def test_active_skin_is_per_profile_and_leaves_launch_slot_alone(homes):
|
||||
a, b = homes
|
||||
from hermes_cli import skin_engine
|
||||
|
||||
(a / "config.yaml").write_text("display:\n skin: ares\n", encoding="utf-8")
|
||||
(b / "config.yaml").write_text("display:\n skin: mono\n", encoding="utf-8")
|
||||
skin_engine._active_skin = None
|
||||
skin_engine._active_skin_name = "default"
|
||||
getattr(skin_engine, "_active_skin_by_home", {}).clear()
|
||||
try:
|
||||
with _Scoped(a):
|
||||
skin_engine.init_skin_from_config({"display": {"skin": "ares"}})
|
||||
assert skin_engine.get_active_skin().name == "ares"
|
||||
with _Scoped(b):
|
||||
assert skin_engine.get_active_skin().name == "mono" # B's own display.skin, never A's
|
||||
with _Scoped(a):
|
||||
assert skin_engine.get_active_skin().name == "ares"
|
||||
assert skin_engine.get_active_skin_name() == "default" # routed turns never touch the launch slot
|
||||
finally:
|
||||
skin_engine._active_skin = None
|
||||
skin_engine._active_skin_name = "default"
|
||||
getattr(skin_engine, "_active_skin_by_home", {}).clear()
|
||||
Reference in New Issue
Block a user