feat: local models — managed llama.cpp runtime with one-click desktop setup

Run models locally as a first-class provider. The CLI grows a managed
llama.cpp runtime (engine install, model download, server supervision);
the desktop app grows the full setup and management story on top of it.
GUI surfaces ship behind the desktop --local launch flag (hermes desktop
--local, or the flag on the packaged app); backend routes and the CLI
are always live.

Runtime (hermes_cli/local_runtime/):
- curated GGUF catalog with per-machine variant selection: hardware
  probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice
  by context window
- derived recommendation: quality-ranked picks gated by a predicted
  decode-speed floor, bandwidth-aware on unified memory; the decision
  table is pinned as a test (pick AND reason per memory class), and the
  Recommended badge explains its pick in a tooltip fed by the resolver's
  actual branch
- engine install + model download with resumable split parts, cumulative
  plan-level progress, and staged-model integrity (a split GGUF counts
  only when every part is present)
- server supervision: spawn/adopt/stop, router mode with per-model load
  progress relayed over SSE, abandoned-request cleanup

Desktop:
- Settings -> Providers -> Local models: one-click quickstart (install
  engine, download the recommended model, boot) plus per-model download/
  activate/eject, fit-ranked catalog with context pills
- model pickers (composer dropdown + Cmd+K) show staged local models,
  in-flight downloads as live progress rows, and load-into-memory bars
- local-setup campaign tip for eligible hardware; System resources
  statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends
- friendly dead-server errors, and failed agent builds retry on the next
  send instead of wedging the session

Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
This commit is contained in:
emozilla
2026-09-01 16:01:53 -04:00
parent 375ce8eee5
commit 43e67d872f
117 changed files with 16471 additions and 126 deletions
+67
View File
@@ -9187,6 +9187,14 @@ def _build_call_kwargs(
_provider_norm == "openrouter"
or base_url_host_matches(_effective_base, "openrouter.ai")
)
# The managed local llama-server honors explicit caps too: a local
# decode burns the user's own GPU at full tilt, so a caller that
# says "this is a 64-token task" must be believed — an uncapped
# local generation whose EOS never comes runs to the full context
# window. No wire-format quirks apply (llama.cpp accepts
# max_tokens), and the no-default-cap policy is unchanged: this
# only forwards caps callers explicitly set.
_is_managed_local = _is_managed_local_endpoint(_effective_base)
if (
_is_anthropic_compat_endpoint(provider, _effective_base)
or _nous_on_messages
@@ -9194,6 +9202,7 @@ def _build_call_kwargs(
or _is_moa
or _is_gemini_native
or _is_openrouter
or _is_managed_local
):
# Use auxiliary_max_tokens_param() so models that require
# max_completion_tokens (GPT-5 family, Copilot) get the right
@@ -9539,6 +9548,49 @@ def _is_streaming_rejected_error(exc: Exception) -> bool:
)
_MANAGED_LOCAL_STATE_TTL_S = 15.0
_managed_local_cache: "tuple[float, str]" = (0.0, "")
def _managed_local_netloc() -> str:
"""host:port of the managed local llama-server, or "" when none.
Read from the supervisor's state file (written at spawn, removed on
stop) with a short TTL so per-request checks don't hit the disk. The
state file is the same source provider resolution uses, so the match
is exact — no false positives on other localhost endpoints.
"""
global _managed_local_cache
now = time.monotonic()
ts, cached = _managed_local_cache
if now - ts < _MANAGED_LOCAL_STATE_TTL_S:
return cached
netloc = ""
try:
from hermes_cli.local_runtime.supervisor import state_path
raw = state_path().read_text(encoding="utf-8")
base = str((json.loads(raw) or {}).get("base_url", ""))
netloc = urlparse(base).netloc.lower()
except Exception:
netloc = ""
_managed_local_cache = (now, netloc)
return netloc
def _is_managed_local_endpoint(base_url: Optional[str]) -> bool:
"""True when *base_url* targets the llama-server this Hermes manages."""
if not base_url:
return False
managed = _managed_local_netloc()
if not managed:
return False
try:
return urlparse(str(base_url)).netloc.lower() == managed
except Exception:
return False
def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
"""Detect providers that only accept streaming (non-stream = HTTP 400).
@@ -9554,6 +9606,18 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
Beyond the known-host list, users can mark ANY custom endpoint as
stream-only via ``auxiliary.stream_only_base_urls`` in config.yaml
(list of substrings matched against the endpoint URL).
The managed local llama-server is always streamed for a different
reason: cancellation. llama-server only notices a dead client when it
writes to the socket. A non-streamed request writes once — after the
FULL generation — so an abandoned call (client timeout, retry, app
exit) keeps the GPU decoding to the end of the context window with
nobody listening; requests that queue behind a model load are the
worst case, since the client is long gone before decode even starts.
Streaming writes every few tokens, so an abandoned decode dies at the
first post-disconnect chunk (verified against llama-server b10362:
streamed disconnect cancels in <1s through the router; non-streamed
survives until the server's next incidental socket poll, if ever).
"""
_url = str(base_url or "").lower()
if not _url:
@@ -9561,6 +9625,9 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
# Tencent Copilot — "Non-stream chat request is currently not supported"
if base_url_host_matches(_url, "copilot.tencent.com"):
return True
# Managed local llama-server — streamed so abandonment cancels decode.
if _is_managed_local_endpoint(_url):
return True
try:
from hermes_cli.config import load_config
aux_cfg = (load_config() or {}).get("auxiliary", {})
+98 -1
View File
@@ -1055,6 +1055,59 @@ def should_use_direct_api_call(agent) -> bool:
_DIRECT_API_ACTIVITY_HEARTBEAT_SECONDS = 15.0
def _managed_local_load_notice(agent, api_kwargs: dict) -> "Optional[str]":
"""A live phase notice while the managed local server works before the
first token, or None when neither phase (nor the managed server) applies:
- "⏳ loading <model> into memory — N%" (weights streaming off disk;
real per-tensor percent from the router's SSE stream)
- "⚙ processing prompt — N of ~M tokens (P%)" (prefill; live counter
from /slots, denominator estimated from the request body)
A cold local model spends ~tens of seconds loading and a long-context
turn spends tens more in prefill; without this, both windows render as
the generic "no output yet (provider may be slow or overloaded)" stall
warning — alarming copy for healthy, expected phases.
"""
try:
base = str(getattr(agent, "base_url", "") or "")
if not base:
return None
import json as _json
from urllib.parse import urlparse
from hermes_cli.local_runtime.load_progress import (
get_loading_progress,
get_prefill_progress,
)
from hermes_cli.local_runtime.supervisor import state_path
state = _json.loads(state_path().read_text(encoding="utf-8"))
managed = urlparse(str(state.get("base_url", ""))).netloc.lower()
if not managed or urlparse(base).netloc.lower() != managed:
return None
model = str(api_kwargs.get("model", ""))
progress = get_loading_progress().get(model)
if progress is not None:
return (
f"⏳ loading {model} into memory — {progress['percent']}% "
"(responses start once the model is loaded)"
)
prefill = get_prefill_progress(model)
if prefill is not None:
processed = int(prefill["processed"])
total = estimate_request_context_tokens(api_kwargs)
if total and total >= processed:
pct = max(0, min(100, round(processed / total * 100)))
return f"⚙ processing prompt — {pct}%"
# Counter past the estimate (estimator undercounted): no honest
# denominator, so no percent — the UI shows label-only.
return "⚙ processing prompt"
return None
except Exception: # noqa: BLE001 — a status nicety must never break a call
return None
def _resolve_direct_stale_timeout(agent, api_kwargs: dict) -> float:
"""Stale budget for the inline non-streaming call.
@@ -5322,9 +5375,54 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
t.start()
_last_heartbeat = time.time()
_HEARTBEAT_INTERVAL = 30.0 # seconds between gateway activity touches
# Managed local server: a cold model streams weights off disk for tens
# of seconds before the first token can exist. Surface THAT immediately
# (real per-tensor percent from the router's SSE stream) instead of
# letting the wait fall through to the 30s "provider may be slow or
# overloaded" copy. Checked on a ~1s cadence only while no chunks have
# arrived; the probe is an in-memory snapshot read, not a network call.
_last_load_poll = 0.0
_load_notice_shown = False
_load_notice_misses = 0
_is_local_base = bool(agent.base_url) and is_local_endpoint(agent.base_url)
while t.is_alive():
t.join(timeout=0.3)
_hb_now = time.time()
# Cold-load window: last_chunk_time is touched at request-client
# creation and then only by REAL chunks, so "no chunk for 2s+" is
# true through a model load (nothing can stream while the child is
# still mapping weights) and false during healthy token flow —
# which is what keeps this poll off the streaming hot path. The
# probe itself is an in-memory snapshot read.
if (
_is_local_base
and _hb_now - last_chunk_time["t"] >= 2.0
and _hb_now - _last_load_poll >= 1.0
):
_last_load_poll = _hb_now
_load_notice = _managed_local_load_notice(agent, api_kwargs)
if _load_notice is not None:
agent._emit_wait_notice(_load_notice)
agent._touch_activity("local model loading")
_load_notice_shown = True
_load_notice_misses = 0
# Loading IS liveness for the heartbeat; the stale detector
# needs no help — the local floor (900s) dwarfs any load.
_last_heartbeat = _hb_now
continue
if _load_notice_shown:
# One missed sample is routine (a /slots read straddling a
# batch boundary, a 2s probe timeout under load) — clearing
# on it made the status line strobe blank once every few
# seconds mid-prefill. Only a SUSTAINED absence means the
# phase really ended.
_load_notice_misses += 1
if _load_notice_misses >= 3:
_load_notice_shown = False
_load_notice_misses = 0
agent._emit_wait_notice("")
# Periodic heartbeat: touch the agent's activity tracker so the
# gateway's inactivity monitor knows we're alive while waiting
# for stream chunks. Without this, long thinking pauses (e.g.
@@ -5333,7 +5431,6 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# activity on each chunk, but the gap between API call start
# and first chunk can exceed the gateway timeout — especially
# when the stale-stream timeout is disabled (local providers).
_hb_now = time.time()
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
_last_heartbeat = _hb_now
_waiting_secs = int(_hb_now - last_chunk_time["t"])
+67
View File
@@ -652,6 +652,40 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str
)
def _maybe_grow_local_window(agent: Any, compressor: Any,
request_tokens: int) -> Optional[int]:
"""Try growing the managed local model's context window before
compressing. Returns the new window when the ladder granted one, else
None (hold / at native / not a managed local session).
The window ladder's design order: models launch at their zero-spill
window and grow toward native max as the session needs room;
compression is the move of last resort. Cheap for every non-local
provider: one lowercase compare, no imports.
"""
provider = (getattr(agent, "provider", "") or "").strip().lower()
if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"):
return None
base_url = getattr(agent, "base_url", "") or ""
if "127.0.0.1" not in base_url and "localhost" not in base_url:
return None
try:
from hermes_cli.local_runtime.growth import maybe_grow_window
current_window = int(getattr(compressor, "context_length", 0) or 0)
if current_window <= 0:
return None
return maybe_grow_window(
getattr(agent, "model", "") or "",
base_url=base_url,
session_tokens=int(request_tokens),
current_window=current_window,
)
except Exception as exc: # noqa: BLE001 — growth must never break a turn
logger.debug("local window growth check failed: %s", exc)
return None
def _ra():
"""Lazy reference to ``run_agent`` so callers can patch
``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` /
@@ -2876,6 +2910,39 @@ def run_conversation(
and not _compression_cooldown
and _compressor.should_compress(request_pressure_tokens)
):
# Managed local runtime: try GROWING the context window before
# compressing (the window ladder's design order — compression is
# the move of last resort, once the window is at the model's
# native max or physics/speed say stop). Only fires for a
# llamacpp-flavored provider whose base_url is the server this
# process supervises; every other provider falls straight
# through to compression, exactly as before.
_grown_window = _maybe_grow_local_window(
agent, _compressor, request_pressure_tokens
)
if _grown_window:
# The server now grants a bigger window: recalibrate the
# compressor to it and skip compression this pass — the
# request that was over the OLD threshold fits the new one.
_compressor.update_model(
agent.model,
_grown_window,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown_window // 1024}K "
f"(local model; conversation continues uncompressed)"
)
# This preflight iteration never reached the provider —
# refund the consumed call/budget exactly as the compression
# path below does before ITS continue.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
continue
if _moa_prepared_request is not None:
pending_moa_prepared_request = _moa_prepared_request
compression_attempts += 1
+44 -3
View File
@@ -519,6 +519,28 @@ def _lookup_supports_vision(
return override
if not provider or not model:
return None
# Managed local runtime: the server that would receive the image is
# the authority on whether it can see (its /props reports modalities
# when a vision projector is loaded; the catalog covers staged-but-
# unloaded models). Cloud catalogs have never heard of a local GGUF,
# so without this answer every local model reads as text-only and
# images detour to a cloud auxiliary — wrong twice for a local-first
# user (broken feature, and a screenshot leaving the machine).
try:
from hermes_cli.local_runtime.capabilities import (
is_managed_provider,
managed_model_supports_vision,
)
if is_managed_provider(provider, _resolve_inference_base_url(cfg, provider) or ""):
managed = managed_model_supports_vision(model)
if managed is not None:
return managed
except Exception as exc: # pragma: no cover - defensive
logger.debug("image_routing: managed-runtime caps lookup failed for %s:%s — %s",
provider, model, exc)
caps = None
try:
from agent.models_dev import get_model_capabilities
@@ -813,12 +835,31 @@ def _file_to_data_url(path: Path) -> Optional[str]:
logger.warning("image_routing: failed to read %s — %s", path, exc)
return None
mime = _guess_mime(path, raw=raw)
if mime not in _UNIVERSALLY_SUPPORTED_MIMES:
accepted = _UNIVERSALLY_SUPPORTED_MIMES
# The managed local server decodes fewer formats than cloud providers
# (no WebP — and a WebP part fails SILENTLY: the model never sees an
# image and confabulates a description). When the active main model is
# served by the managed runtime, narrow the accepted set so those
# formats transcode to PNG here instead of vanishing server-side.
try:
from agent.auxiliary_client import _runtime_main_value
from hermes_cli.local_runtime.capabilities import (
ACCEPTED_IMAGE_MIMES,
is_managed_provider,
)
if is_managed_provider(
str(_runtime_main_value("provider") or ""),
str(_runtime_main_value("base_url") or "")):
accepted = ACCEPTED_IMAGE_MIMES
except Exception: # noqa: BLE001 — best-effort narrowing only
pass
if mime not in accepted:
transcoded = _transcode_to_png(raw)
if transcoded is None:
logger.warning(
"image_routing: %s is %s which is not accepted by all major "
"vision providers and could not be transcoded to PNG; "
"image_routing: %s is %s which is not accepted by the "
"active provider and could not be transcoded to PNG; "
"skipping this attachment.",
path, mime,
)
+50
View File
@@ -1502,6 +1502,35 @@ def fetch_endpoint_model_metadata(
model_alias = props.get("model_alias", "")
if n_ctx and model_alias and model_alias in cache:
cache[model_alias]["context_length"] = n_ctx
else:
# Router mode: bare /props 400s and telemetry is
# per-child (?model=). Enumerate children via the
# native /models (carries status) and read each
# LOADED child's granted window — the value the
# context policy actually granted, which the meter
# and compressor must follow. Unloaded children are
# skipped: probing them could trigger an autoload.
native = requests.get(base + "/models", headers=headers, timeout=5, verify=_verify)
if native.ok:
children = (native.json() or {}).get("data", [])
for child in children[:16]:
if not isinstance(child, dict):
continue
child_id = child.get("id")
status = (child.get("status") or {}).get("value")
if not child_id or child_id not in cache or status not in ("loaded", "ready"):
continue
pr = requests.get(
base + "/v1/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if not pr.ok:
pr = requests.get(
base + "/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if pr.ok:
child_ctx = (pr.json().get("default_generation_settings") or {}).get("n_ctx")
if child_ctx:
cache[child_id]["context_length"] = child_ctx
except Exception:
pass
@@ -2375,6 +2404,27 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
return int(ctx)
break
# llama.cpp: /props reports default_generation_settings.n_ctx —
# the RUNTIME window the server grants. Critically, the router
# answers this (from its preset) even for a model that is not
# currently loaded, while /v1/models reports meta=null until
# load. Without this probe, resolving a lazily-loaded model at
# session start finds no metadata and falls through to the
# name-pattern defaults, where a family catch-all (e.g. "qwen"
# = 131072) misreports a server launched at 262144.
if server_type == "llamacpp":
for props_path in (f"/props?model={model}", "/props"):
try:
resp = client.get(f"{server_url}{props_path}")
except httpx.HTTPError:
break
if resp.status_code != 200:
continue
n_ctx = (resp.json().get("default_generation_settings")
or {}).get("n_ctx")
if isinstance(n_ctx, (int, float)) and n_ctx:
return int(n_ctx)
# LM Studio / vLLM / llama.cpp / Anthropic-compat proxies:
# try /v1/models/{model}
resp = client.get(f"{server_url}/v1/models/{model}")
+291
View File
@@ -0,0 +1,291 @@
"""Idle deferral for background reviews on the managed local runtime.
The post-turn review fork replays the whole conversation on the review
runtime. On a cloud provider that costs seconds and runs concurrently
with whatever the user does next. When the review runtime IS the managed
llama-server, the same fork monopolizes the GPU the user's next prompt
needs, for minutes — and the next live turn cancels it, so an active
session tends to pay the decode cost AND lose the learning.
This module keeps the decision to learn exactly where it was (turn end,
nudge intervals, full-strength model, full transcript) and moves only
the execution moment: reviews bound for the managed local endpoint are
queued and dispatched when the machine is quiet. Everything else runs
immediately, as before.
Policy (auxiliary.background_review.defer):
auto (default) — defer exactly when the resolved review runtime
targets the managed local server.
never — old behavior everywhere.
Explicit /refine (focus set) never defers: an explicit ask runs now,
matching its bypass of the enabled gate.
Queue semantics:
- One slot per session, newest snapshot wins. A review replays the whole
conversation, so a newer snapshot strictly supersedes an older one —
coalescing is deduplication, not loss.
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn
wrapper observing the run token's cancel flag, not killed-and-forgotten.
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless
of idleness — deferral may delay learning, never lose it.
- In-memory, best-effort: dropped on process exit, the same durability
contract the immediate daemon-thread fork always had.
Idle truth comes from the supervisor's /slots (machine-level: it sees
every client of the managed server, including other Hermes profiles) and
must hold for a settle window so a review is not launched into the gap
between two quick prompts. Local in-process turn liveness is tracked via
note_turn_started/note_turn_finished from run_conversation.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.request
from typing import Any, Callable, Dict, List, Optional
logger = logging.getLogger(__name__)
# Sustained-quiet window before dispatch. Long enough that "typed two
# prompts back to back" does not look idle; short enough that walking
# away for coffee runs the queue.
_IDLE_SETTLE_S = 15.0
# Poll cadence while the queue is non-empty. The thread parks when empty.
_POLL_INTERVAL_S = 5.0
# Age at which a queued review dispatches regardless of idleness.
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
return raw if raw in ("auto", "never") else "auto"
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
try:
value = float(raw)
except (TypeError, ValueError):
return _MAX_AGE_DEFAULT_S
return value if value > 0 else _MAX_AGE_DEFAULT_S
def review_targets_managed_local(agent: Any,
task_cfg: Optional[Dict[str, Any]]) -> bool:
"""Would this review fork decode on the llama-server WE manage?
Resolves the review runtime the same way the fork itself will and
exact-matches its netloc against the supervisor state file — the
matcher that cannot false-positive on external local servers. Any
failure reads False: immediate spawn is always the safe default.
Order matters: the netloc probe (one TTL-cached state-file read)
runs FIRST, so machines with no managed server — every cloud-only
install — return False without resolving the review runtime at all.
This wrapper runs on the turn's tail; runtime resolution belongs on
that path only when a managed server actually exists.
"""
try:
from agent.auxiliary_client import (
_is_managed_local_endpoint,
_managed_local_netloc,
)
if not _managed_local_netloc():
return False
from agent.background_review import _resolve_review_runtime
runtime = _resolve_review_runtime(agent, task_cfg)
return _is_managed_local_endpoint(runtime.get("base_url"))
except Exception: # noqa: BLE001
return False
class _PendingReview:
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
self.agent = agent
self.session_key = session_key
self.kwargs = kwargs
self.enqueued_at = time.monotonic()
class ReviewIdleQueue:
"""Session-coalescing queue + idle-gated dispatcher thread."""
def __init__(self) -> None:
self._lock = threading.Lock()
self._pending: Dict[str, _PendingReview] = {}
self._wake = threading.Event()
self._thread: Optional[threading.Thread] = None
self._live_turns = 0
self._quiet_since: Optional[float] = None
# Test seams — replaced by unit tests, never in production.
self._now: Callable[[], float] = time.monotonic
self._server_idle: Callable[[], bool] = _managed_server_idle
# ── turn liveness (this process) ────────────────────────────
def note_turn_started(self) -> None:
with self._lock:
self._live_turns += 1
self._quiet_since = None
def note_turn_finished(self) -> None:
with self._lock:
self._live_turns = max(0, self._live_turns - 1)
if self._live_turns == 0:
self._quiet_since = self._now()
self._wake.set()
# ── queue ────────────────────────────────────────────────────
def enqueue(self, agent: Any, session_key: str,
kwargs: Dict[str, Any]) -> None:
"""Add (or replace — newest snapshot wins) a session's pending review."""
with self._lock:
existing = self._pending.get(session_key)
item = _PendingReview(agent, session_key, kwargs)
# Stamp through the queue's clock (test seam); keep the ORIGINAL
# enqueue time on coalesce so a busy session cannot push its
# review's age-out forever.
item.enqueued_at = (existing.enqueued_at if existing is not None
else self._now())
self._pending[session_key] = item
self._ensure_thread()
self._wake.set()
logger.info("Background review deferred (session=%s, queued=%d)",
session_key[-12:], len(self._pending))
def pending_count(self) -> int:
with self._lock:
return len(self._pending)
# ── dispatcher ───────────────────────────────────────────────
def _ensure_thread(self) -> None:
with self._lock:
if self._thread is None or not self._thread.is_alive():
self._thread = threading.Thread(
target=self._run, daemon=True, name="bg-review-idle-queue")
self._thread.start()
def _quiet_for(self) -> float:
"""Seconds this process has been turn-free (0 while a turn runs)."""
with self._lock:
if self._live_turns > 0 or self._quiet_since is None:
return 0.0
return self._now() - self._quiet_since
def _pop_dispatchable(self) -> Optional[_PendingReview]:
"""Oldest aged-out item, else any item once quiet+idle hold."""
with self._lock:
if not self._pending:
return None
items = sorted(self._pending.values(),
key=lambda p: p.enqueued_at)
aged = [p for p in items
if self._now() - p.enqueued_at
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
candidate = aged[0] if aged else None
if candidate is None:
if self._quiet_for() < _IDLE_SETTLE_S:
return None
if not self._server_idle():
return None
with self._lock:
if not self._pending:
return None
candidate = min(self._pending.values(),
key=lambda p: p.enqueued_at)
with self._lock:
return self._pending.pop(candidate.session_key, None)
def _run(self) -> None:
while True:
self._wake.wait()
with self._lock:
if not self._pending:
self._wake.clear()
continue
item = None
try:
item = self._pop_dispatchable()
if item is not None:
if not self._still_enabled(item):
logger.info(
"Deferred background review dropped: reviews "
"were disabled while it was queued (session=%s)",
item.session_key[-12:])
continue
logger.info(
"Dispatching deferred background review "
"(session=%s, waited=%.0fs, queued=%d)",
item.session_key[-12:],
self._now() - item.enqueued_at,
self.pending_count())
item.agent._spawn_background_review_now(**item.kwargs)
except Exception: # noqa: BLE001 — dispatcher must survive anything
logger.warning("Deferred review dispatch failed",
exc_info=True)
if item is None:
time.sleep(_POLL_INTERVAL_S)
@staticmethod
def _still_enabled(item: _PendingReview) -> bool:
"""Re-check the enabled gate at DISPATCH time.
The entry wrapper gates at enqueue time, but minutes may pass in
the queue — a user who sets background_review.enabled: false while
a review waits means it, and the dispatch must not resurrect it.
Fail-open like the gate itself (a broken config never silently
disables reviews)."""
try:
from agent.background_review import load_background_review_settings
enabled, _ = load_background_review_settings()
return enabled
except Exception: # noqa: BLE001
return True
def _managed_server_idle() -> bool:
"""Machine-level idle: no processing slot on any loaded model of the
managed router. Unreachable/no state file reads idle (nothing to
contend with). One /models + one /slots call per loaded model."""
try:
from hermes_cli.local_runtime.supervisor import state_path
state = json.loads(state_path().read_text(encoding="utf-8"))
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
key = str(state.get("api_key", ""))
if not base:
return True
headers = {"Authorization": f"Bearer {key}"}
req = urllib.request.Request(f"{base}/models", headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
models = json.loads(r.read())
loaded = [m["id"] for m in models.get("data", [])
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
from urllib.parse import quote
for mid in loaded:
req = urllib.request.Request(f"{base}/slots?model={quote(mid)}",
headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
slots = json.loads(r.read())
if any(s.get("is_processing") for s in slots
if isinstance(s, dict)):
return False
return True
except Exception: # noqa: BLE001
return True
# Module singleton — one queue per process, like the load-progress watcher.
QUEUE = ReviewIdleQueue()
+28
View File
@@ -1128,6 +1128,34 @@ def build_turn_context(
_compress_block_reason = _info(_preflight_tokens)[1]
except Exception:
_compress_block_reason = None
if _should_compress_now:
# Managed local runtime: growing the window beats compressing —
# the ladder's design order (same seam as the conversation
# loop's pre-API gate; see _maybe_grow_local_window there).
try:
from agent.conversation_loop import _maybe_grow_local_window
_grown = _maybe_grow_local_window(
agent, _compressor, _preflight_tokens
)
except Exception:
_grown = None
if _grown:
_compressor.update_model(
agent.model,
_grown,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown // 1024}K "
f"(local model; conversation continues uncompressed)"
)
_should_compress_now = _compressor.should_compress(
_preflight_tokens
)
if _should_compress_now:
_preflight_compressed = True
# Compression is actually running (block cleared / was never
+9
View File
@@ -16741,6 +16741,15 @@ ipcMain.on('hermes:translucency:support', event => {
event.returnValue = { glass: GLASS_SUPPORTED, translucency: TRANSLUCENCY_SUPPORTED }
})
// Launch-flag facts the renderer needs before first paint (same sendSync
// pattern as translucency). `--local` gates every local-models GUI surface;
// it arrives from `hermes desktop --local` or directly on Hermes.exe (a
// shortcut edit), and survives self-relaunches because collectRelaunchArgs
// only strips internal flags.
ipcMain.on('hermes:launch-flags', event => {
event.returnValue = { localModels: process.argv.includes('--local') }
})
ipcMain.on('hermes:translucency', (_event, payload) => {
const next = normalizeTranslucency(payload, GLASS_SUPPORTED)
const previous = translucencyState
+4
View File
@@ -10,10 +10,14 @@ import { contextBridge, ipcRenderer, webFrame, webUtils } from 'electron'
const translucencySupport = ipcRenderer.sendSync('hermes:translucency:support')
const hudWindowing = ipcRenderer.sendSync('hermes:hud:windowing')
const hudNativeDrag = hudWindowing?.nativeDrag === true
const launchFlags = ipcRenderer.sendSync('hermes:launch-flags')
contextBridge.exposeInMainWorld('hermesDesktop', {
glassSupported: translucencySupport?.glass === true,
translucencySupported: translucencySupport?.translucency === true,
// Launch-flag fact: the app was started with --local, so the renderer may
// show the local-models surfaces. Static for the window's lifetime.
localModelsEnabled: launchFlags?.localModels === true,
getConnection: profile => ipcRenderer.invoke('hermes:connection', profile),
// Registry-scoped backend resolution: { connectionId, profile } → descriptor.
getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload),
+166
View File
@@ -0,0 +1,166 @@
import type {
LocalCatalogModel,
LocalHardware,
LocalModelsStatus,
LocalRuntimeJob
} from '@/types/hermes'
import { hermesApi, profileScoped } from './client'
// The desktop surface of the managed llama.cpp runtime: status/catalog
// reads, download/install/activate jobs, and server control.
export function getLocalModelsStatus(): Promise<LocalModelsStatus> {
return hermesApi<LocalModelsStatus>({
...profileScoped(),
path: '/api/local-models/status'
})
}
export function getLocalHardware(): Promise<LocalHardware> {
return hermesApi<LocalHardware>({
...profileScoped(),
path: '/api/local-models/hardware'
})
}
export function getLocalCatalog(): Promise<{ models: LocalCatalogModel[] }> {
return hermesApi<{ models: LocalCatalogModel[] }>({
...profileScoped(),
path: '/api/local-models/catalog'
})
}
export function installLocalRuntime(backend?: string): Promise<{ backend: string; job_id: string; tag: string }> {
return hermesApi<{ backend: string; job_id: string; tag: string }>({
...profileScoped(),
body: { backend: backend ?? null },
method: 'POST',
path: '/api/local-models/runtime/install'
})
}
export interface QuickstartResponse {
display_name: string
download_bytes: number
job_id: string
model_id: string
needs_download: boolean
needs_runtime: boolean
}
export function quickstartLocalModels(modelId?: string): Promise<QuickstartResponse> {
return hermesApi<QuickstartResponse>({
...profileScoped(),
body: { model_id: modelId ?? null },
method: 'POST',
path: '/api/local-models/quickstart'
})
}
export function downloadLocalModel(modelId: string): Promise<{ already_downloaded?: boolean; job_id: null | string }> {
return hermesApi<{ already_downloaded?: boolean; job_id: null | string }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/download'
})
}
export function deleteLocalModel(modelId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
method: 'DELETE',
path: `/api/local-models/models/${encodeURIComponent(modelId)}`
})
}
export function getLocalRuntimeJob(jobId: string): Promise<LocalRuntimeJob> {
return hermesApi<LocalRuntimeJob>({
...profileScoped(),
path: `/api/local-models/jobs/${encodeURIComponent(jobId)}`
})
}
export function getLocalModelsJobs(): Promise<{ jobs: LocalRuntimeJob[] }> {
return hermesApi<{ jobs: LocalRuntimeJob[] }>({
...profileScoped(),
path: '/api/local-models/jobs'
})
}
export function activateLocalModel(modelId: string): Promise<{ job_id: string }> {
return hermesApi<{ job_id: string }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/activate'
})
}
export function ejectLocalModel(modelId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/eject'
})
}
export function setLocalServer(action: 'start' | 'stop'): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
body: { action },
method: 'POST',
path: '/api/local-models/server'
})
}
// ── Hugging Face browser + sideload ─────────────────────────────
export interface HFSearchHit {
repo: string
downloads: number
likes: number
updated: string
gated: boolean
}
export interface HFFileGroup {
label: string
paths: string[]
total_bytes: number
fit: 'fits-gpu' | 'needs-ram' | 'too-big' | 'unknown'
}
export function searchHFModels(q: string, limit = 20): Promise<{ hits: HFSearchHit[] }> {
return hermesApi<{ hits: HFSearchHit[] }>({
...profileScoped(),
path: `/api/local-models/search?q=${encodeURIComponent(q)}&limit=${limit}`
})
}
export function listHFRepoFiles(repo: string): Promise<{ files: HFFileGroup[] }> {
return hermesApi<{ files: HFFileGroup[] }>({
...profileScoped(),
path: `/api/local-models/search/files?repo=${encodeURIComponent(repo)}`
})
}
export function downloadBrowsedModel(repo: string, paths: string[]): Promise<{ already_downloaded?: boolean; job_id: null | string; model_id: string }> {
return hermesApi<{ already_downloaded?: boolean; job_id: null | string; model_id: string }>({
...profileScoped(),
body: { paths, repo },
method: 'POST',
path: '/api/local-models/download-browsed'
})
}
export function sideloadLocalModel(path: string): Promise<{ already_present?: boolean; model_id: string; ok: boolean }> {
return hermesApi<{ already_present?: boolean; model_id: string; ok: boolean }>({
...profileScoped(),
body: { path },
method: 'POST',
path: '/api/local-models/sideload'
})
}
@@ -44,6 +44,7 @@ import {
isCurrentGatewaySwitch,
registerGatewaySwitchLifecycle
} from '@/store/gateway-switch'
import { checkLocalRuntimeUpdate, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
import { notify, notifyError } from '@/store/notifications'
import {
$activeGatewayProfile,
@@ -663,6 +664,12 @@ export function useGatewayBoot({
completeDesktopBoot()
bootCompleted = true
// Rediscover local-runtime jobs (model downloads, runtime installs)
// that were running before a reload — the backend registry is the
// authority; this just resumes following it.
watchLocalRuntimeJobs()
// One-per-session engine-update pointer (enabled runtimes only).
void checkLocalRuntimeUpdate()
} catch (err) {
const mayPublishFailure =
!cancelled && (switchToken === null ? !$gatewaySwitching.get() : isCurrentGatewaySwitch(switchToken))
+18 -1
View File
@@ -12,6 +12,7 @@ import {
Archive,
BarChart3,
Bell,
Cpu,
Download,
Globe,
Info,
@@ -31,6 +32,7 @@ import { cn } from '@/lib/utils'
import { $commandPaletteOpen, openCommandPalettePage } from '@/store/command-palette'
import { confirm } from '@/store/confirm'
import { bindingsFor } from '@/store/keybinds'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { notifyError } from '@/store/notifications'
import { useRouteEnumParam } from '../hooks/use-route-enum-param'
@@ -217,7 +219,22 @@ export function SettingsView({ onClose, onConfigSaved, onMainModelChanged }: Set
id: 'pview:custom-endpoints',
label: t.settings.nav.providerCustomEndpoints,
onSelect: () => openProviderView('custom-endpoints')
}
},
// Local models ships behind the --local launch flag: no flag, no
// nav entry (the pane itself also refuses to render, so a stale
// ?pview=local deep link falls back to accounts-shaped emptiness
// rather than a hidden feature).
...($localModelsEnabled.get()
? [
{
active: activeView === 'providers' && providerView === 'local',
icon: Cpu,
id: 'pview:local',
label: t.settings.nav.providerLocalModels,
onSelect: () => openProviderView('local')
}
]
: [])
],
gapBefore: true,
icon: Zap,
@@ -0,0 +1,556 @@
import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
import { MemoryRouter, useLocation } from 'react-router'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { I18nProvider } from '@/i18n'
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
import type { LocalCatalogModel, LocalHardware, LocalModelsStatus, LocalRuntimeJob } from '@/types/hermes'
import { LocalModelsSettings } from './local-models-settings'
// Mock the API layer — the pane's contract is what it RENDERS from these
// payloads, not transport.
vi.mock('@/hermes', () => ({
activateLocalModel: vi.fn(),
deleteLocalModel: vi.fn(),
downloadBrowsedModel: vi.fn(),
downloadLocalModel: vi.fn(),
ejectLocalModel: vi.fn(),
getLocalCatalog: vi.fn(),
getLocalHardware: vi.fn(),
getLocalModelsJobs: vi.fn(),
getLocalModelsStatus: vi.fn(),
getLocalRuntimeJob: vi.fn(),
installLocalRuntime: vi.fn(),
listHFRepoFiles: vi.fn(),
quickstartLocalModels: vi.fn(),
searchHFModels: vi.fn(),
sideloadLocalModel: vi.fn()
}))
import * as hermes from '@/hermes'
const mocked = vi.mocked(hermes)
const BASE_STATUS: LocalModelsStatus = {
enabled: true,
tag: 'b10290',
configured_tag: 'b10290',
update_available: false,
runtime_installed: false,
runtime_backend: null,
server_running: false,
server_base_url: null,
active_model_id: null,
loaded_models: {},
models: [],
models_dir: 'C:/somewhere/models'
}
const BASE_HARDWARE: LocalHardware = {
uma: false,
vram_total_bytes: 32 * 2 ** 30,
vram_usable_bytes: 26 * 2 ** 30,
ram_total_bytes: 256 * 2 ** 30,
ram_available_bytes: 200 * 2 ** 30,
vram_label: '32.0 GB',
gpu_name: 'NVIDIA GeForce RTX 5090',
gpu_util_percent: 12,
vram_used_bytes: 6 * 2 ** 30
}
const FITTING_MODEL: LocalCatalogModel = {
id: 'Qwen3.6-27B-UD-Q4_K_XL',
display_name: 'Qwen3.6 27B',
description: 'Best all-round agent model; long context stays fast',
size_bytes: 17.6 * 2 ** 30,
size_label: '17.6 GB',
native_context: 262144,
native_context_label: '256K',
recommended: true,
downloaded: false,
mtp: false,
fits: true,
fit_summary: 'runs at its full 256K context',
start_window: 262144,
start_window_label: '256K',
spilled: false
}
const SPILLED_MODEL: LocalCatalogModel = {
...FITTING_MODEL,
id: 'Spilled-Model',
display_name: 'Spilled Model',
recommended: false,
fits: true,
spilled: true,
start_window: 65536,
start_window_label: '64K',
fit_summary: 'starts at 64K and grows toward 256K as you use it (larger than your GPU memory — runs slower)'
}
const REFUSED_MODEL: LocalCatalogModel = {
...FITTING_MODEL,
id: 'Huge-Model',
display_name: 'Huge Model',
recommended: false,
fits: false,
fit_summary: 'Needs more memory than this machine has',
fit_detail: 'needs ~60 GiB at the 64K floor',
start_window: undefined,
start_window_label: undefined
}
function renderPane() {
return render(
<MemoryRouter>
<I18nProvider>
<LocalModelsSettings />
</I18nProvider>
</MemoryRouter>
)
}
// The fresh-machine states these tests exercise now lead with the
// quickstart card; the full pane (runtime rows, model list, browser)
// is one 'Configure…' click away. Render and click through.
async function renderFullPane() {
const result = renderPane()
const configure = await screen.findByRole('button', { name: /configure/i })
fireEvent.click(configure)
return result
}
beforeEach(() => {
mocked.getLocalModelsStatus.mockResolvedValue(BASE_STATUS)
mocked.getLocalHardware.mockResolvedValue(BASE_HARDWARE)
mocked.getLocalCatalog.mockResolvedValue({ models: [FITTING_MODEL, SPILLED_MODEL, REFUSED_MODEL] })
mocked.getLocalModelsJobs.mockResolvedValue({ jobs: [] })
$localRuntimeJobs.set([])
})
afterEach(() => {
cleanup()
vi.clearAllMocks()
})
describe('LocalModelsSettings', () => {
it('offers the runtime install with a plain-language explanation', async () => {
await renderFullPane()
expect(await screen.findByText('Install the local runtime')).toBeTruthy()
expect(screen.getByText(/runs? entirely on this machine/i)).toBeTruthy()
expect(screen.getByRole('button', { name: /install runtime/i })).toBeTruthy()
})
it('shows every catalog model with fit pills; unaffordable ones stay visible with the reason', async () => {
await renderFullPane()
expect(await screen.findByText('Qwen3.6 27B')).toBeTruthy()
// The fitting model reads as pills, not prose: green memory pill +
// green full-context pill (start_window == native, resident on GPU).
expect(screen.getByText('Fits your GPU')).toBeTruthy()
expect(screen.getByText('Full 256K context').className).toContain('emerald')
// The refused model is NOT hidden (discoverability rule): red memory
// pill, plus the ceiling it would have had.
expect(screen.getByText('Huge Model')).toBeTruthy()
expect(screen.getByText('Too big for this machine')).toBeTruthy()
// The spilled model reads amber + ONE quiet ceiling pill — the same
// 'Up to' shape the refused row wears; no start/grow pair.
expect(screen.getByText('Spilled Model')).toBeTruthy()
expect(screen.getByText('Uses system RAM')).toBeTruthy()
expect(screen.getAllByText('Up to 256K context').length).toBe(2)
expect(screen.queryByText(/Starts at/)).toBeNull()
// Its download button is disabled; the fitting model's is enabled once
// the runtime exists (here runtime_installed=false, so both disabled —
// asserted separately below).
const buttons = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
expect(buttons.every(b => (b as HTMLButtonElement).disabled)).toBe(true)
})
it('orders the catalog by fit: resident first, then spilled, then too-big', async () => {
// Scrambled input — the pane, not the backend, owns display order.
mocked.getLocalCatalog.mockResolvedValue({ models: [REFUSED_MODEL, SPILLED_MODEL, FITTING_MODEL] })
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
// The matched element is the row-title span; the recommended row's
// includes its nested pill copy — strip it before comparing order.
const names = screen
.getAllByText(/^(Qwen3\.6 27B|Spilled Model|Huge Model)$/)
.map(el => el.textContent?.replace('Recommended', ''))
expect(names).toEqual(['Qwen3.6 27B', 'Spilled Model', 'Huge Model'])
})
it('never greens the full-context pill on a system-RAM model', async () => {
// Full native window, but earned by spilling into system RAM: the
// pill must not wear the green that would recommend exactly the
// wrong model.
const spilledFull: LocalCatalogModel = {
...FITTING_MODEL,
id: 'Spilled-Full',
display_name: 'Spilled Full',
recommended: false,
spilled: true,
fit_summary: 'runs its full 256K context, partly from system RAM'
}
mocked.getLocalCatalog.mockResolvedValue({ models: [spilledFull] })
await renderFullPane()
await screen.findByText('Spilled Full')
expect(screen.getByText('Full 256K context').className).not.toContain('emerald')
})
it('explains the Recommended pick on hover', async () => {
// The tooltip is the resolver's own reason, and it must actually OPEN:
// Tip works by asChild-cloning hover handlers onto the pill, so a Pill
// that swallows its rest props kills the tooltip silently (the pill
// still renders, nothing appears on hover).
mocked.getLocalCatalog.mockResolvedValue({
models: [{ ...FITTING_MODEL, recommended_reason: 'speed-gated-quality' }]
})
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
fireEvent.pointerMove(screen.getByText('Recommended'))
fireEvent.pointerEnter(screen.getByText('Recommended'))
await waitFor(() =>
expect(screen.getAllByText(/would respond too slowly on its memory bandwidth/).length).toBeGreaterThan(0)
)
})
it('enables downloads only once the runtime is installed', async () => {
mocked.getLocalModelsStatus.mockResolvedValue({
...BASE_STATUS,
runtime_installed: true,
runtime_backend: 'cuda'
})
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
const [fittingButton] = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
expect((fittingButton as HTMLButtonElement).disabled).toBe(false)
})
it('shows hardware facts after backfill', async () => {
await renderFullPane()
expect(await screen.findByText('NVIDIA GeForce RTX 5090')).toBeTruthy()
expect(screen.getByText(/32\.0 GB GPU memory/)).toBeTruthy()
expect(screen.getByText(/256\.0 GB RAM/)).toBeTruthy()
})
it('tracks a download job to completion and refreshes', async () => {
mocked.getLocalModelsStatus.mockResolvedValue({
...BASE_STATUS,
runtime_installed: true,
runtime_backend: 'cuda'
})
mocked.downloadLocalModel.mockResolvedValue({ job_id: 'j1' })
const running: LocalRuntimeJob = {
job_id: 'j1',
kind: 'model-download',
target: 'Qwen3.6 27B',
model_id: FITTING_MODEL.id,
status: 'running',
phase: 'downloading',
detail: 'Qwen3.6 27B — 17.6 GB',
total_bytes: 100,
done_bytes: 40,
percent: 40,
error: null
}
mocked.getLocalModelsJobs
.mockResolvedValueOnce({ jobs: [running] })
.mockResolvedValue({ jobs: [{ ...running, status: 'done', phase: 'done', done_bytes: 100, percent: 100 }] })
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
const [download] = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
download.click()
// The app-level watcher follows the job; when it settles the pane
// refreshes (status + catalog re-fetched).
await waitFor(() => {
expect(mocked.getLocalModelsJobs).toHaveBeenCalled()
expect(mocked.getLocalModelsStatus.mock.calls.length).toBeGreaterThanOrEqual(2)
})
})
it('renders progress for a download discovered from the store (survives pane remount)', async () => {
mocked.getLocalModelsStatus.mockResolvedValue({
...BASE_STATUS,
runtime_installed: true,
runtime_backend: 'cuda'
})
// A running job already in the app-level store — as after closing and
// reopening the pane mid-download.
$localRuntimeJobs.set([
{
job_id: 'j9',
kind: 'model-download',
target: 'Qwen3.6 27B',
model_id: FITTING_MODEL.id,
status: 'running',
phase: 'downloading',
detail: '',
total_bytes: 100,
done_bytes: 62,
percent: 62,
error: null
}
])
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
// The fitting row shows byte progress; the remaining download
// buttons belong to the other rows (spilled + refused).
expect(screen.getAllByText(/0\.0 GB of 0\.0 GB|of/).length).toBeGreaterThan(0)
const remaining = screen.queryAllByRole('button', { name: /download · 17\.6 GB/i })
expect(remaining.length).toBe(2)
expect(remaining.some(b => (b as HTMLButtonElement).disabled)).toBe(true)
})
it('surfaces a failed download with the backend message', async () => {
mocked.getLocalModelsStatus.mockResolvedValue({
...BASE_STATUS,
runtime_installed: true,
runtime_backend: 'cuda'
})
$localRuntimeJobs.set([
{
job_id: 'j2',
kind: 'model-download',
target: 'Qwen3.6 27B',
model_id: FITTING_MODEL.id,
status: 'error',
phase: 'verifying',
detail: '',
total_bytes: 100,
done_bytes: 100,
error: 'Downloaded file failed its integrity check and was removed — try again'
}
])
await renderFullPane()
await screen.findByText('Qwen3.6 27B')
expect(await screen.findByText(/integrity check/)).toBeTruthy()
})
})
describe('quickstart', () => {
it('leads with one button on a fresh machine and fires the quickstart job', async () => {
mocked.quickstartLocalModels.mockResolvedValue({
display_name: 'Qwen3.6 27B',
download_bytes: FITTING_MODEL.size_bytes,
job_id: 'q1',
model_id: 'qwen3.6-27b',
needs_download: true,
needs_runtime: true
})
renderPane()
// The card names the recommended model and the one-click action; the
// runtime/model machinery is NOT on screen.
expect(await screen.findByRole('button', { name: /set up for me/i })).toBeTruthy()
expect(screen.queryByText('Install the local runtime')).toBeNull()
fireEvent.click(screen.getByRole('button', { name: /set up for me/i }))
await waitFor(() => {
expect(mocked.quickstartLocalModels).toHaveBeenCalled()
})
})
it('pins the quickstart progress view while the job runs', async () => {
$localRuntimeJobs.set([
{
job_id: 'q1',
kind: 'quickstart',
target: 'Qwen3.6 27B',
model_id: 'qwen3.6-27b',
status: 'running',
phase: 'downloading',
detail: 'Qwen3.6 27B — 17.6 GB',
total_bytes: 100,
done_bytes: 30,
percent: 30,
error: null
}
])
renderPane()
expect(await screen.findByText('Qwen3.6 27B — 17.6 GB')).toBeTruthy()
// One job, one view: no Set up / Configure buttons while it runs.
expect(screen.queryByRole('button', { name: /set up for me/i })).toBeNull()
})
it('skips the card entirely once a model is staged', async () => {
mocked.getLocalModelsStatus.mockResolvedValue({
...BASE_STATUS,
runtime_installed: true,
runtime_backend: 'cuda',
models: [{ id: 'Qwen3.6-27B-UD-Q4_K_XL', size_bytes: 17 * 2 ** 30, size_label: '17.6 GB' }]
})
renderPane()
// Straight to the full pane — no quickstart hero for a working setup.
expect(await screen.findByText('Qwen3.6 27B')).toBeTruthy()
expect(screen.queryByRole('button', { name: /set up for me/i })).toBeNull()
})
})
describe('BrowseSection', () => {
it('searches HF after a pause and shows fit-priced files on demand', async () => {
vi.useFakeTimers()
try {
vi.mocked(hermes.searchHFModels).mockResolvedValue({
hits: [{ downloads: 872724, gated: false, likes: 47, repo: 'unsloth/Qwen3.8-27B-GGUF', updated: '2026-08-18' }]
})
vi.mocked(hermes.listHFRepoFiles).mockResolvedValue({
files: [
{ fit: 'fits-gpu', label: 'Q4_K_M', paths: ['Qwen3.8-27B-Q4_K_M.gguf'], total_bytes: 17 * 2 ** 30 },
{ fit: 'too-big', label: 'F16', paths: ['Qwen3.8-27B-F16.gguf'], total_bytes: 56 * 2 ** 30 }
]
})
render(
<MemoryRouter>
<I18nProvider>
<LocalModelsSettings />
</I18nProvider>
</MemoryRouter>
)
await act(async () => {
await vi.runOnlyPendingTimersAsync()
})
// Fresh machine leads with the quickstart card — enter the full pane.
fireEvent.click(screen.getByRole('button', { name: /configure/i }))
const box = screen.getByPlaceholderText(/search models/i)
fireEvent.change(box, { target: { value: 'qwen' } })
// Debounce: no call until the pause elapses.
expect(hermes.searchHFModels).not.toHaveBeenCalled()
await act(async () => {
await vi.advanceTimersByTimeAsync(400)
})
expect(hermes.searchHFModels).toHaveBeenCalledWith('qwen')
expect(screen.getByText('unsloth/Qwen3.8-27B-GGUF')).toBeTruthy()
fireEvent.click(screen.getByRole('button', { name: /show files/i }))
await act(async () => {
await vi.runOnlyPendingTimersAsync()
})
expect(screen.getByText('Q4_K_M')).toBeTruthy()
// Each tile has an explicit download button; the too-big quant's is
// disabled, the fitting one is live and starts the download.
const q4Btn = screen.getByRole('button', { name: 'Download Q4_K_M' })
const f16Btn = screen.getByRole('button', { name: 'Download F16' })
expect((f16Btn as HTMLButtonElement).disabled).toBe(true)
expect((q4Btn as HTMLButtonElement).disabled).toBe(false)
vi.mocked(hermes.downloadBrowsedModel).mockResolvedValue({ job_id: 'j1', model_id: 'Qwen3.8-27B-Q4_K_M' })
fireEvent.click(q4Btn)
await act(async () => {
await vi.runOnlyPendingTimersAsync()
})
expect(hermes.downloadBrowsedModel).toHaveBeenCalledWith('unsloth/Qwen3.8-27B-GGUF', ['Qwen3.8-27B-Q4_K_M.gguf'])
} finally {
vi.useRealTimers()
}
})
})
describe('added-by-you rows', () => {
it('staged models outside the catalog get the full action set', async () => {
vi.mocked(hermes.getLocalModelsStatus).mockResolvedValue({
...BASE_STATUS,
loaded_models: { 'Hermes-4.3-36B-Q5_K_M': 'loaded' },
models: [{ id: 'Hermes-4.3-36B-Q5_K_M', size_bytes: 25 * 2 ** 30, size_label: '25.0 GB' }],
placement: {
'Hermes-4.3-36B-Q5_K_M': {
granted_window_label: '96K',
spilled: false,
window: 98304,
window_label: '96K'
}
},
server_running: true
})
vi.mocked(hermes.getLocalCatalog).mockResolvedValue({ models: [] })
renderPane()
await screen.findByText('Hermes-4.3-36B-Q5_K_M')
// Full management surface: Use, eject, delete, live placement pill.
expect(screen.getByText(/added by you/i)).toBeTruthy()
expect(screen.getByRole('button', { name: /use/i })).toBeTruthy()
expect(screen.getByText(/96K/)).toBeTruthy()
const buttons = screen.getAllByRole('button')
expect(buttons.length).toBeGreaterThanOrEqual(3)
})
})
describe('quickstart completion navigation', () => {
it('lands on a new chat when a quickstart it watched finishes; stale done jobs on mount never navigate', async () => {
const routeProbe = vi.fn()
function Probe() {
const loc = useLocation()
routeProbe(loc.pathname)
return null
}
const doneJob: LocalRuntimeJob = {
done_bytes: 0,
detail: '',
error: null,
job_id: 'stale-done',
kind: 'quickstart',
model_id: 'qwen3.8-27b',
phase: 'done',
status: 'done',
target: 'Qwen3.8 27B',
total_bytes: null
}
// A finished quickstart already in history when the pane mounts —
// must NOT trigger navigation.
$localRuntimeJobs.set([doneJob])
render(
<MemoryRouter initialEntries={['/settings']}>
<I18nProvider>
<LocalModelsSettings />
</I18nProvider>
<Probe />
</MemoryRouter>
)
await act(async () => {})
expect(routeProbe).not.toHaveBeenCalledWith('/')
// A quickstart the pane SAW running that then completes -> navigate.
const running: LocalRuntimeJob = { ...doneJob, job_id: 'live-run', phase: 'downloading', status: 'running' }
await act(async () => {
$localRuntimeJobs.set([doneJob, running])
})
await act(async () => {
$localRuntimeJobs.set([doneJob, { ...running, phase: 'done', status: 'done' }])
})
expect(routeProbe).toHaveBeenCalledWith('/')
})
})
File diff suppressed because it is too large Load Diff
+22 -4
View File
@@ -1,4 +1,4 @@
import type { ReactNode } from 'react'
import type { ComponentProps, ReactNode } from 'react'
import { Badge } from '@/components/ui/badge'
import { Button } from '@/components/ui/button'
@@ -22,10 +22,28 @@ export function SettingsContent({ children, bare = false }: { children: ReactNod
)
}
const PILL_VARIANT = { muted: 'muted', primary: 'default', warn: 'warn' } as const
const PILL_VARIANT = {
muted: 'muted',
primary: 'default',
success: 'success',
warn: 'warn',
destructive: 'destructive'
} as const
export function Pill({ tone = 'muted', children }: { tone?: keyof typeof PILL_VARIANT; children: ReactNode }) {
return <Badge variant={PILL_VARIANT[tone]}>{children}</Badge>
// Rest props spread through to the Badge's DOM node — REQUIRED for Radix
// `asChild` composition (wrapping a Pill in `Tip` clones it with the hover
// handlers and ref as props; swallowing them left every tooltip on a Pill
// silently dead).
export function Pill({
tone = 'muted',
children,
...props
}: { tone?: keyof typeof PILL_VARIANT; children: ReactNode } & Omit<ComponentProps<typeof Badge>, 'variant'>) {
return (
<Badge variant={PILL_VARIANT[tone]} {...props}>
{children}
</Badge>
)
}
export function SectionHeading({
@@ -7,6 +7,7 @@ import {
FEATURED_ID,
FeaturedProviderRow,
FireworksProviderRow,
LocalModelsProviderRow,
OpenRouterProviderRow,
ProviderRow,
providerTitle,
@@ -21,6 +22,7 @@ import { Check, ChevronDown, ChevronRight, KeyRound, Loader2, Terminal, Trash2 }
import { normalize } from '@/lib/text'
import { cn } from '@/lib/utils'
import { confirm } from '@/store/confirm'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { notify, notifyError } from '@/store/notifications'
import { $desktopOnboarding, startManualLocalEndpoint, startManualProviderOAuth } from '@/store/onboarding'
import type { EnvVarInfo, OAuthProvider } from '@/types/hermes'
@@ -29,6 +31,7 @@ import { isKeyVar, ProviderKeyRows } from './credential-key-ui'
import { CustomEndpointsSettings } from './custom-endpoints-settings'
import { SettingsCategoryHeading, useEnvCredentials } from './env-credentials'
import { providerGroup, providerMeta, providerPriority } from './helpers'
import { LocalModelsSettings } from './local-models-settings'
import { SettingsContent, SettingsSkeleton } from './primitives'
// The embedded terminal (and thus the "run disconnect command" path) only
@@ -46,7 +49,7 @@ function GroupLabel({ children }: { children: ReactNode }) {
}
// Sub-views surfaced as a sidebar subnav: account sign-in vs raw API keys.
export const PROVIDER_VIEWS = ['accounts', 'keys', 'custom-endpoints'] as const
export const PROVIDER_VIEWS = ['accounts', 'keys', 'custom-endpoints', 'local'] as const
export type ProviderView = (typeof PROVIDER_VIEWS)[number]
@@ -117,24 +120,26 @@ function buildProviderKeyGroups(vars: Record<string, EnvVarInfo>): ProviderKeyGr
// Deliberately a near-1:1 replica of the first-run onboarding picker
// (`Picker` in desktop-onboarding-overlay): same recommended card, same
// Fireworks #2 quick-key row, same provider rows, same "Other providers"
// disclosure, same OpenRouter quick-key row, and the same bottom-right
// "I have an API key" affordance. The leaf cards are the exact shared
// components, so the two surfaces stay visually identical. Selecting a
// provider hands off to the shared onboarding overlay, which runs that
// provider's real sign-in flow; the key affordances open the API-key
// catalog below.
// always-visible Local models row, same provider rows, same "Other
// providers" disclosure (Fireworks and OpenRouter quick-key rows live
// inside it on both surfaces), and the same bottom-right "I have an API
// key" affordance. The leaf cards are the exact shared components, so
// the two surfaces stay visually identical. Selecting a provider hands
// off to the shared onboarding overlay, which runs that provider's real
// sign-in flow; the key affordances open the API-key catalog below.
function OAuthPicker({
disconnecting,
onDisconnect,
onTerminalDisconnect,
onWantApiKey,
onWantLocalModels,
providers
}: {
disconnecting: null | string
onDisconnect: (provider: OAuthProvider) => void
onTerminalDisconnect: (provider: OAuthProvider) => void
onWantApiKey: () => void
onWantLocalModels: () => void
providers: OAuthProvider[]
}) {
const { t } = useI18n()
@@ -176,8 +181,9 @@ function OAuthPicker({
{p.intro}
</p>
{featured && <FeaturedProviderRow onSelect={select} provider={featured} />}
{/* Slot #2 — always visible, matching onboarding / CANONICAL_PROVIDERS. */}
<FireworksProviderRow onClick={onWantApiKey} />
{/* Slot #2 — the no-account path, matching onboarding. Behind the
--local launch flag like every local-models surface. */}
{$localModelsEnabled.get() && <LocalModelsProviderRow onClick={onWantLocalModels} />}
{connected.length > 0 && (
<>
<GroupLabel>{p.connected}</GroupLabel>
@@ -199,6 +205,7 @@ function OAuthPicker({
{others.map(p => (
<ProviderRow key={p.id} onSelect={select} provider={p} />
))}
<FireworksProviderRow onClick={onWantApiKey} />
<OpenRouterProviderRow onClick={onWantApiKey} />
</>
)}
@@ -507,6 +514,13 @@ export function ProvidersSettings({
return <CustomEndpointsSettings onConfigSaved={onConfigSaved} onMainModelChanged={onMainModelChanged} />
}
if (view === 'local') {
// Strict --local gate: without the launch flag the pane doesn't render
// even when local models are configured — a stale ?pview=local deep link
// (or an old shortcut) lands on the accounts view instead.
return $localModelsEnabled.get() ? <LocalModelsSettings /> : null
}
return (
<SettingsContent>
<OAuthPicker
@@ -514,6 +528,7 @@ export function ProvidersSettings({
onDisconnect={provider => void handleDisconnect(provider)}
onTerminalDisconnect={provider => void handleTerminalDisconnect(provider)}
onWantApiKey={() => onViewChange('keys')}
onWantLocalModels={() => onViewChange('local')}
providers={oauthProviders}
/>
</SettingsContent>
@@ -8,6 +8,7 @@ import { useApprovalModeStatusbarItem } from '@/app/shell/approval-mode-menu'
import { ContextUsagePanel } from '@/app/shell/context-usage-panel'
import { GatewayMenuPanel } from '@/app/shell/gateway-menu-panel'
import { useContextBreakdown } from '@/app/shell/hooks/use-context-breakdown'
import { useSystemResourcesStatusbarItem } from '@/app/shell/system-resources-statusbar'
import { $paneVisible, togglePaneVisible } from '@/components/pane-shell/tree/store'
import { Codicon } from '@/components/ui/codicon'
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
@@ -268,6 +269,7 @@ export function useStatusbarItems({
const contextBar = useMemo(() => contextBarLabel(gaugeUsage), [gaugeUsage])
const approvalModeItem = useApprovalModeStatusbarItem(activeGatewayProfile, requestGateway)
const systemResourcesItem = useSystemResourcesStatusbarItem()
const gatewayMenuContent = useMemo(
() => (close: () => void) => (
@@ -546,9 +548,12 @@ export function useStatusbarItems({
},
{
detail: contextBar || undefined,
hidden: !contextUsage,
// Never self-hide: the user opted this item in (it's hidden-by-
// default), so an empty label must render as a waiting placeholder,
// not a vanished item — an enabled-but-invisible toggle reads as
// "another item took its spot".
id: 'context-usage',
label: contextUsage,
label: contextUsage || '—',
menuAlign: 'end',
menuClassName: 'w-auto border-(--ui-stroke-secondary) p-0',
menuContent: (
@@ -565,6 +570,7 @@ export function useStatusbarItems({
toggleLabel: copy.toggleSessionTimer,
variant: 'text'
},
systemResourcesItem,
{
...approvalModeItem,
hidden: gatewayState !== 'open',
@@ -598,6 +604,7 @@ export function useStatusbarItems({
gaugeUsage,
sessionStartedAt,
gatewayState,
systemResourcesItem,
terminalShowing,
turnStartedAt
]
@@ -1,8 +1,10 @@
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'
import { DropdownMenu, DropdownMenuContent } from '@/components/ui/dropdown-menu'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
import {
$modelVisibilityOpen,
$visibleModels,
@@ -10,6 +12,7 @@ import {
setModelVisibilityOpen,
setVisibleModels
} from '@/store/model-visibility'
import type { LocalRuntimeJob } from '@/types/hermes'
import { ModelCatalogMenu, type ModelMenuController } from './model-catalog-menu'
@@ -24,11 +27,23 @@ const getGlobalModelOptions = vi.fn()
vi.mock('@/hermes', () => ({
getGlobalModelOptions: (...args: unknown[]) => getGlobalModelOptions(...args),
// The menu kicks the app-level job poller on mount; echo the store so a
// poll can't wipe the jobs a test staged (the real backend is authority,
// and here the store plays that part).
getLocalModelsJobs: vi.fn(async () => {
const { $localRuntimeJobs } = await import('@/store/local-runtime-jobs')
return { jobs: [...$localRuntimeJobs.get()] }
}),
getLocalModelsStatus: vi.fn().mockResolvedValue({ loading: {} }),
setApiRequestProfile: vi.fn()
}))
beforeEach(() => {
$visibleModels.set(null)
$localRuntimeJobs.set([])
// These suites exercise the local-models rows, which ship behind --local.
$localModelsEnabled.set(true)
setModelVisibilityOpen(false)
getGlobalModelOptions.mockResolvedValue({
providers: [{ models: ['gemini-3.1-pro', 'gemini-2.5-flash'], name: 'Google', slug: 'google' }]
@@ -106,3 +121,77 @@ describe('the catalog owns model curation', () => {
expect($modelVisibilityOpen.get()).toBe(true)
})
})
describe('in-flight local downloads', () => {
const DOWNLOAD_JOB: LocalRuntimeJob = {
job_id: 'dl1',
kind: 'model-download',
target: 'Qwen3.8 Flash Next (UD-Q4_K_XL)',
model_id: 'qwen3.8-flash-next',
status: 'running',
phase: 'downloading',
detail: '',
total_bytes: 100,
done_bytes: 41,
percent: 41,
error: null
}
it('shows a downloading model as a disabled progress row in its own Local group', async () => {
// No llamacpp provider in the catalog (first-ever download).
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderMenu()
await screen.findByText(/Gemini 3\.1 Pro/i)
const row = screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
expect(row).toBeTruthy()
expect(screen.getByText('41%')).toBeTruthy()
expect(row.closest('[role="menuitem"]')?.getAttribute('aria-disabled')).toBe('true')
})
it('shows the download inside the Local provider group when it exists', async () => {
getGlobalModelOptions.mockResolvedValue({
providers: [
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
]
})
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderMenu()
await screen.findByText(/Qwen3\.6 27B/i)
expect(screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
// One Local heading — the trailing fallback group must not double up.
expect(screen.getAllByText('Local').length).toBe(1)
})
it('drops the placeholder row once the download settles', async () => {
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderMenu()
await screen.findByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
$localRuntimeJobs.set([{ ...DOWNLOAD_JOB, status: 'done', phase: 'done' }])
await waitFor(() => {
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
})
})
it('hides the local provider group and download rows without the --local flag (strict)', async () => {
$localModelsEnabled.set(false)
getGlobalModelOptions.mockResolvedValue({
providers: [
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
]
})
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderMenu()
// Staged models exist and a download is running — none of it shows.
await screen.findByText(/Gemini 3\.1 Pro/i)
expect(screen.queryByText(/Qwen3\.6 27B/i)).toBeNull()
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
expect(screen.queryByText('Local')).toBeNull()
})
})
@@ -19,12 +19,16 @@ import { HighlightMatches } from '@/components/ui/highlight-matches'
import { usePointerQuiet } from '@/components/ui/keyboard-first'
import { Skeleton } from '@/components/ui/skeleton'
import type { HermesGateway } from '@/hermes'
import { getLocalModelsStatus } from '@/hermes'
import { useI18n } from '@/i18n'
import { modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options'
import { displayModelName, modelDisplayParts } from '@/lib/model-status-label'
import { DEFAULT_REASONING_EFFORT, reasoningEffortLabel } from '@/lib/reasoning-effort'
import { normalize } from '@/lib/text'
import { useStoreSelector } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $localRuntimeJobs, runningModelDownloads, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
import {
$visibleModels,
collapseModelFamilies,
@@ -36,7 +40,7 @@ import {
} from '@/store/model-visibility'
import { $collapsedProviders, toggleCollapsedProvider } from '@/store/provider-collapse'
import { $defaultReasoningEffort } from '@/store/session'
import type { ModelOptionProvider, ModelOptionsResponse } from '@/types/hermes'
import type { LocalModelLoadProgress, ModelOptionProvider, ModelOptionsResponse } from '@/types/hermes'
import { type FastControl, ModelEditSubmenu, resolveFastControl } from './model-edit-submenu'
@@ -134,6 +138,7 @@ export function ModelCatalogMenu({
}: ModelCatalogMenuProps) {
const { t } = useI18n()
const copy = t.shell.modelMenu
const copyPicker = t.modelPicker
const closeMenu = useContext(ModelMenuCloseContext)
const [search, setSearch] = useState('')
const collapsedProviders = useStoreCollapsed()
@@ -154,6 +159,78 @@ export function ModelCatalogMenu({
const loading = modelOptions.isPending && !modelOptions.data
// Every local-models read in this menu sits behind the --local launch
// flag: no status polling, no download rows, and the llamacpp provider
// group hides even when models are staged (the flag is strict).
const localModelsEnabled = $localModelsEnabled.get()
// Live load state for the managed local server: which model is loading
// into memory right now, with a REAL percent (per-tensor callback relayed
// over the router's SSE stream). Polled only while this menu is mounted
// (it unmounts on close); errors read as "nothing loading" — remote-only
// installs have no local-models routes.
const localStatus = useQuery({
queryKey: ['local-models-loading', profile],
queryFn: () => getLocalModelsStatus(),
enabled: localModelsEnabled,
refetchInterval: 2_000,
retry: false
})
const loadingModels: Record<string, LocalModelLoadProgress> = localStatus.data?.loading ?? {}
// Models on their way into the local library (downloads + quickstart runs
// still fetching bytes) — rendered as disabled progress rows so the user
// sees the model coming instead of wondering where it went. The jobs store
// republishes every ~700ms with fresh byte counts while anything runs; a
// whole-store subscription here would re-render the entire menu per tick
// (breaking open submenus and focus — the #72163 class). Subscribe to a
// STABLE identity projection instead: it changes only when a download
// starts or ends. Each row selects its own percent scalar.
const downloadsKey = useStoreSelector($localRuntimeJobs, jobs =>
localModelsEnabled
? runningModelDownloads(jobs)
.map(job => `${job.job_id}\u0000${job.target}`)
.join('\u0001')
: ''
)
const downloads = useMemo(
() =>
downloadsKey === ''
? []
: downloadsKey.split('\u0001').map(pair => {
const [jobId, target] = pair.split('\u0000')
return { jobId, target }
}),
[downloadsKey]
)
useEffect(() => {
if (localModelsEnabled) {
watchLocalRuntimeJobs()
}
}, [localModelsEnabled])
// A finished download turns into a real selectable model: refetch the
// catalog so the placeholder row is replaced while the menu is open.
const refetchOptions = modelOptions.refetch
useEffect(() => {
let prevActive = runningModelDownloads($localRuntimeJobs.get()).length > 0
return $localRuntimeJobs.listen(next => {
const active = runningModelDownloads(next).length > 0
if (prevActive && !active) {
void refetchOptions()
}
prevActive = active
})
}, [refetchOptions])
const error = modelOptions.error
? modelOptions.error instanceof Error
? modelOptions.error.message
@@ -170,12 +247,27 @@ export function ModelCatalogMenu({
)
const pickerProviders = useMemo(
() => providers?.filter(provider => provider.slug.toLowerCase() !== 'moa') ?? [],
[providers]
() =>
providers?.filter(
provider =>
provider.slug.toLowerCase() !== 'moa' &&
// Strict --local gate: staged local models exist on disk, but
// without the flag the GUI doesn't offer them.
(localModelsEnabled || provider.slug !== LOCAL_PROVIDER_SLUG)
) ?? [],
[providers, localModelsEnabled]
)
const current = controller.current
const q = normalize(search)
// In-flight downloads render inside the Local provider group when it
// exists, else as their own trailing 'Local' group (first download —
// nothing staged yet, so the catalog has no local provider row).
const shownDownloads = q ? downloads.filter(job => (job.target || '').toLowerCase().includes(q)) : downloads
const hasLocalGroup = pickerProviders.some(provider => provider.slug === LOCAL_PROVIDER_SLUG)
// Resolve visibility HERE, against the catalog we actually fetched: an empty
// provider list would otherwise resolve to an empty key set that reads as
// "user hid everything" and blanks the menu on first open.
@@ -189,8 +281,6 @@ export function ModelCatalogMenu({
[pickerProviders, search, current.model, current.provider, shownKeys]
)
const q = normalize(search)
// Presets are searchable rows like everything else — an unfiltered preset
// sitting under zero model matches would otherwise become the "first match"
// Enter commits.
@@ -367,7 +457,7 @@ export function ModelCatalogMenu({
<DropdownMenuItem className={dropdownMenuRow} disabled>
{error}
</DropdownMenuItem>
) : groups.length === 0 && moaPresets.length === 0 ? (
) : groups.length === 0 && moaPresets.length === 0 && shownDownloads.length === 0 ? (
<DropdownMenuItem className={dropdownMenuRow} disabled>
{copy.noModels}
</DropdownMenuItem>
@@ -412,6 +502,10 @@ export function ModelCatalogMenu({
const isCurrent = activeId !== null
const name = modelDisplayParts(family.id).name
const caps = group.provider.capabilities?.[family.id]
// Managed local model loading into memory right now:
// real load percent, keyed by exact model id (remote
// providers never collide with GGUF stems).
const loadProgress = loadingModels[family.id] ?? (family.fastId ? loadingModels[family.fastId] : undefined)
// Effective settings for this row: the live choice when it's
// the active model, otherwise its remembered preset. Row
@@ -461,8 +555,28 @@ export function ModelCatalogMenu({
<HighlightMatches query={search} text={name} />
{meta ? <span className="text-(--ui-text-tertiary)"> {meta}</span> : null}
</span>
{loadProgress ? (
<span
className="ml-auto flex shrink-0 items-center gap-1.5"
title={copyPicker.loadingIntoMemory}
>
<span className="h-1 w-14 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<span
className="block h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(2, loadProgress.percent)}%` }}
/>
</span>
<span className="text-[0.62rem] tabular-nums text-(--ui-text-tertiary)">
{loadProgress.percent}%
</span>
</span>
) : null}
{isCurrent ? (
<Codicon className="ml-auto text-foreground" name="check" size="0.75rem" />
<Codicon
className={cn('text-foreground', loadProgress ? 'ml-1' : 'ml-auto')}
name="check"
size="0.75rem"
/>
) : null}
</DropdownMenuSubTrigger>
<ModelEditSubmenu
@@ -486,9 +600,22 @@ export function ModelCatalogMenu({
</DropdownMenuSub>
)
})}
{!collapsed &&
slug === LOCAL_PROVIDER_SLUG &&
shownDownloads.map(job => <DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />)}
</DropdownMenuGroup>
)
})}
{!hasLocalGroup && shownDownloads.length > 0 && (
<DropdownMenuGroup className="py-0.5" key="local-downloads">
<DropdownMenuLabel className="px-2 pb-0.5 pt-0.5 text-[0.625rem] font-semibold uppercase tracking-wider text-(--ui-text-tertiary)">
{copyPicker.localDownloadsHeading}
</DropdownMenuLabel>
{shownDownloads.map(job => (
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
))}
</DropdownMenuGroup>
)}
</div>
)}
@@ -540,6 +667,47 @@ export function ModelCatalogMenu({
/** Re-exported so callers building a footer row match the catalog's rows. */
export { dropdownMenuRow }
// The backend's provider row for staged local models (inventory.py's
// _local_runtime_row). Downloads-in-flight attach to this group.
const LOCAL_PROVIDER_SLUG = 'llamacpp'
// A model still downloading: visible so the user knows it's coming (and
// where it will land), disabled so it can't be selected early, with the
// same byte progress the Local Models pane shows. Percent is selected HERE,
// per row, so the 700ms byte ticks repaint this leaf only — the menu tree
// above subscribes to download identity, not progress.
function DownloadingModelRow({ jobId, target }: { jobId: string; target: string }) {
const { t } = useI18n()
const copy = t.modelPicker
const percent = useStoreSelector(
$localRuntimeJobs,
jobs => jobs.find(job => job.job_id === jobId)?.percent ?? null
)
return (
<DropdownMenuItem
className={cn(dropdownMenuRow, 'opacity-60')}
disabled
onSelect={event => event.preventDefault()}
textValue=""
>
<span className="min-w-0 flex-1 truncate">{target}</span>
<span className="ml-auto flex shrink-0 items-center gap-1.5" title={copy.downloading}>
<span className="h-1 w-14 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<span
className="block h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(2, percent ?? 0)}%` }}
/>
</span>
<span className="text-[0.62rem] tabular-nums text-(--ui-text-tertiary)">
{typeof percent === 'number' ? `${percent}%` : copy.downloading}
</span>
</span>
</DropdownMenuItem>
)
}
// Collapsed we show the user's chosen models (or the curated default); typing
// spans every available model so anything is reachable past the cut. A search
// is itself a narrowing action, so we do NOT cap per-provider matches.
@@ -0,0 +1,172 @@
import { useStore } from '@nanostores/react'
import { useEffect, useState } from 'react'
import type { StatusbarItem } from '@/app/shell/statusbar-controls'
import { getLocalHardware } from '@/hermes'
import { useI18n } from '@/i18n'
import { Activity } from '@/lib/icons'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $statusbarHiddenIds } from '@/store/statusbar-prefs'
import type { LocalHardware } from '@/types/hermes'
// Live host-resource readout for the bottom bar: GPU utilization + VRAM +
// RAM, fed by /api/local-models/hardware. Hidden by default (an item most
// users don't watch); the poll runs ONLY while the item is shown, so the
// hidden default costs nothing. 5s cadence — resource numbers, not a
// heartbeat.
const POLL_MS = 5_000
function gb(bytes: number | null | undefined): string {
return bytes ? `${(bytes / (1 << 30)).toFixed(0)}G` : '—'
}
function gbLong(bytes: number | null | undefined): string {
return bytes ? `${(bytes / (1 << 30)).toFixed(1)} GB` : '—'
}
function MeterRow({ label, percent, value }: { label: string; percent: number | null; value: string }) {
return (
<div className="grid gap-1">
<div className="flex items-baseline justify-between gap-2">
{/* Label yields, value never does: if anything ever narrows the row
again, a truncated label beats a clipped number — "15.2 GB" losing
its tail reads as a wrong number, not a cut one. */}
<span className="truncate text-muted-foreground">{label}</span>
<span className="shrink-0 whitespace-nowrap tabular-nums text-foreground">{value}</span>
</div>
{percent !== null && (
<div className="h-1.5 w-full overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<div
className="h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(1, Math.min(100, percent))}%` }}
/>
</div>
)}
</div>
)
}
export function useSystemResourcesStatusbarItem(): StatusbarItem {
const { t } = useI18n()
const copy = t.shell.statusbar.systemResources
const hiddenIds = useStore($statusbarHiddenIds)
// Behind the --local launch flag: without it the item is absent from the
// bar AND from the customize menu (no toggleLabel), and never polls.
const enabled = $localModelsEnabled.get()
const shown = enabled && !hiddenIds.includes('system-resources')
const [hardware, setHardware] = useState<LocalHardware | null>(null)
useEffect(() => {
if (!shown) {
return
}
let cancelled = false
let timer: number | null = null
const poll = async () => {
try {
const next = await getLocalHardware()
if (!cancelled) {
setHardware(next)
}
} catch {
if (!cancelled) {
setHardware(null)
}
}
if (!cancelled) {
timer = window.setTimeout(() => void poll(), POLL_MS)
}
}
void poll()
return () => {
cancelled = true
if (timer !== null) {
window.clearTimeout(timer)
}
}
}, [shown])
const hasGpu = Boolean(hardware?.gpu_name)
const vramPercent =
hardware?.vram_used_bytes != null && hardware.vram_total_bytes
? Math.round((hardware.vram_used_bytes / hardware.vram_total_bytes) * 100)
: null
const ramUsed = hardware ? hardware.ram_total_bytes - hardware.ram_available_bytes : null
const ramPercent = hardware?.ram_total_bytes && ramUsed != null ? Math.round((ramUsed / hardware.ram_total_bytes) * 100) : null
// Compact bar label: the numbers a local-inference user glances at.
// "GPU 34% · 18G/32G" with a GPU; "RAM 41G/256G" without.
const label = hardware
? hasGpu
? `GPU ${hardware.gpu_util_percent ?? 0}%${
hardware.vram_used_bytes != null ? ` · ${gb(hardware.vram_used_bytes)}/${gb(hardware.vram_total_bytes)}` : ''
}`
: `RAM ${gb(ramUsed)}/${gb(hardware.ram_total_bytes)}`
: copy.loading
return {
detail: undefined,
hidden: !enabled,
icon: <Activity className="size-3" />,
id: 'system-resources',
label,
menuAlign: 'end',
menuClassName: 'w-64 p-0',
menuContent: (
<div
className="grid grid-cols-[minmax(0,1fr)] gap-3 p-3 text-[0.75rem]"
data-slot="system-resources-panel"
>
{/* min-w-0 everywhere a flex/grid child must shrink: grid items
default min-width:auto, so a long GPU name's nowrap min-content
props the track open past the w-64 box and overflow-x:hidden
shears off every right-aligned value. With the track clamped,
`truncate` can finally act. */}
<div className="flex min-w-0 items-baseline justify-between gap-2">
<p className="shrink-0 font-medium text-foreground">{copy.title}</p>
{hardware?.gpu_name && (
<span className="min-w-0 truncate text-[0.6875rem] text-muted-foreground">{hardware.gpu_name}</span>
)}
</div>
{hasGpu && (
<MeterRow
label={copy.gpuUtilization}
percent={hardware?.gpu_util_percent ?? null}
value={`${hardware?.gpu_util_percent ?? 0}%`}
/>
)}
{hasGpu && (
<MeterRow
label={copy.gpuMemory}
percent={vramPercent}
value={`${gbLong(hardware?.vram_used_bytes)} / ${gbLong(hardware?.vram_total_bytes)}`}
/>
)}
<MeterRow
label={copy.ram}
percent={ramPercent}
value={`${gbLong(ramUsed)} / ${gbLong(hardware?.ram_total_bytes)}`}
/>
{hardware?.uma && <p className="text-[0.6875rem] text-muted-foreground">{copy.unifiedNote}</p>}
</div>
),
toggleLabel: enabled ? copy.toggle : undefined,
variant: 'menu'
}
}
@@ -11,13 +11,17 @@ import { SCAFFOLD_LABEL_CLASS } from '@/components/chat/scaffold-row'
import { Codicon } from '@/components/ui/codicon'
import { Loader } from '@/components/ui/loader'
import { StatusPulse } from '@/components/ui/status-pulse'
import { getLocalModelsStatus } from '@/hermes'
import { useI18n } from '@/i18n'
import { cn } from '@/lib/utils'
import { $backgroundResume } from '@/store/background-delegation'
import { sessionCompacting } from '@/store/compaction'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { sessionAwaitingInput } from '@/store/prompts'
import { sessionProviderWait } from '@/store/provider-wait'
import { parseModelLoadWait, sessionProviderWait } from '@/store/provider-wait'
import { $currentModel } from '@/store/session'
import { type DraftingTool, sessionDraftingTool } from '@/store/tool-drafting'
import type { LocalModelLoadProgress } from '@/types/hermes'
// A status line is scaffolding like any other — "Editing" while the model
// drafts a call is the same kind of line as "Explored 3 files" once it has run,
@@ -51,6 +55,100 @@ const HintText: FC<{ children: ReactNode }> = ({ children }) => (
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shimmer min-w-0 flex-1 truncate')}>{children}</span>
)
/** Renderer-side load synthesis: poll the local-models status while a turn
* is busy with NO progress frame from the backend. The backend's wait loop
* only narrates the MAIN chat request — a model load triggered while the
* gateway is still initializing, or one consumed by a parallel auxiliary
* call (title generation autoloads the same model), never gets a frame,
* and the load looked like nothing was happening. The status route reads
* the same SSE snapshot, so this bar carries the identical percent. */
function useLocalModelLoad(active: boolean): LocalModelLoadProgress & { model: string } | null {
const model = useStore($currentModel)
const [progress, setProgress] = useState<(LocalModelLoadProgress & { model: string }) | null>(null)
// Behind the --local launch flag: without it, no status polling and no
// load bar (the local server can't be the current provider anyway).
const enabled = $localModelsEnabled.get()
useEffect(() => {
if (!enabled || !active || !model) {
setProgress(null)
return
}
let cancelled = false
let timer: number | undefined
const tick = async () => {
try {
const status = await getLocalModelsStatus()
const entry = status.loading?.[model]
if (!cancelled) {
setProgress(entry ? { ...entry, model } : null)
}
} catch {
if (!cancelled) {
setProgress(null)
}
}
if (!cancelled) {
timer = window.setTimeout(() => void tick(), 1_500)
}
}
void tick()
return () => {
cancelled = true
if (timer !== undefined) {
window.clearTimeout(timer)
}
}
}, [enabled, active, model])
return progress
}
/** Wait hint with a real progress bar for managed-local model loads and
* prompt processing. The percents come from llama-server itself (per-tensor
* load callback / live prefill counter, via the gateway's wait frames), so a
* determinate bar is honest — a 40s cold load or a long prefill reads as
* visible progress instead of an alarming stall. */
const WaitHint: FC<{ hint: string }> = ({ hint }) => {
const { t } = useI18n()
const load = parseModelLoadWait(hint)
if (!load) {
return <HintText>{hint}</HintText>
}
const label =
load.kind === 'load' ? t.assistant.thread.loadingLocalModel(load.model) : t.assistant.thread.processingPrompt
return <ProgressHint label={label} percent={load.percent} />
}
const ProgressHint: FC<{ label: string; percent: null | number }> = ({ label, percent }) => (
<span className="flex min-w-0 flex-1 items-center gap-2">
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shimmer min-w-0 shrink truncate')}>{label}</span>
{percent !== null && (
<>
<span className="h-1 w-24 shrink-0 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<span
className="block h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(2, percent)}%` }}
/>
</span>
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shrink-0 tabular-nums')}>{percent}%</span>
</>
)}
</span>
)
/** These indicators render inside whichever transcript mounted them, so every
* session-scoped signal comes from that surface's view — a tile must never
* show the primary chat's compaction, prompt-wait, or turn timer. */
@@ -147,6 +245,10 @@ export const ResponseLoadingIndicator: FC = () => {
const { compacting, drafting, providerWait, turnStartedAt } = useThreadSessionStatus()
const elapsed = useElapsedSeconds(true, undefined, turnStartedAt)
const hint = useStatusHint(compacting, drafting, providerWait)
// Renderer-synthesized load bar: covers loads the backend's wait loop
// can't narrate (gateway still initializing, or an auxiliary call — not
// the main request — triggered the autoload). A real wait frame wins.
const localLoad = useLocalModelLoad(!hint)
return (
<StatusRow data-slot="aui_response-loading" label={hint || t.assistant.thread.loadingResponse}>
@@ -155,7 +257,11 @@ export const ResponseLoadingIndicator: FC = () => {
className="dither inline-block size-3 rounded-[2px] text-midground/80"
kind="opacity"
/>
{hint && <HintText>{hint}</HintText>}
{hint ? (
<WaitHint hint={hint} />
) : localLoad ? (
<ProgressHint label={t.assistant.thread.loadingLocalModel(localLoad.model)} percent={localLoad.percent} />
) : null}
<ActivityTimerText seconds={elapsed} />
</StatusRow>
)
@@ -207,6 +313,7 @@ export const BackgroundResumeNotice: FC = () => {
// so that per-token updates re-render only this leaf, not the whole
// AssistantMessage subtree.
export const TurnActivityIndicator: FC = () => {
const { t } = useI18n()
const activity = useAuiState(s => activitySignature(s.message.content))
// Timestamp of the last visible progress, held from the moment the quiet
@@ -227,6 +334,10 @@ export const TurnActivityIndicator: FC = () => {
// turn of a fresh chat — so the row can't wait for the store to catch up.
const messageRunning = useAuiState(s => s.message.status?.type === 'running')
// Renderer-synthesized load bar (see ResponseLoadingIndicator).
const working = busy || messageRunning
const localLoad = useLocalModelLoad(working && !hint && !toolNarrating)
useEffect(() => {
setQuietSince(undefined)
const seenAt = Date.now()
@@ -240,8 +351,10 @@ export const TurnActivityIndicator: FC = () => {
// TURN_QUIET_S first, or a run of quick calls would strobe a row between
// each one. The two exemptions are waits already accounted for elsewhere: a
// question the user is answering, and a tool call carrying its own timer.
const working = busy || messageRunning
const active = working && !awaitingInput && !toolNarrating && (Boolean(hint) || quietSince !== undefined)
// A live local-model load is a named wait too — it must not wait out the
// quiet window (the load IS the story from second one).
const active =
working && !awaitingInput && !toolNarrating && (Boolean(hint) || localLoad !== null || quietSince !== undefined)
// Compaction owns the whole turn, so it keeps counting from the turn's start;
// anything else counts from the moment the turn last produced something — the
@@ -263,7 +376,11 @@ export const TurnActivityIndicator: FC = () => {
className="dither inline-block size-3 rounded-[2px] text-midground/80"
kind="opacity"
/>
{hint && <HintText>{hint}</HintText>}
{hint ? (
<WaitHint hint={hint} />
) : localLoad ? (
<ProgressHint label={t.assistant.thread.loadingLocalModel(localLoad.model)} percent={localLoad.percent} />
) : null}
<ActivityTimerText seconds={elapsed} />
</StatusRow>
)
@@ -0,0 +1,151 @@
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
import { cleanup, render, screen, waitFor } from '@testing-library/react'
import type { ReactElement } from 'react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { I18nProvider } from '@/i18n'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
import { stubMenuDomApis, stubResizeObserver } from '@/test/jsdom'
import type { LocalRuntimeJob, ModelOptionsResponse } from '@/types/hermes'
import { ModelPickerDialog } from './model-picker'
vi.mock('@/hermes', () => ({
getLocalModelsStatus: vi.fn().mockResolvedValue({ loading: {} })
}))
vi.mock('@/lib/model-options', async importOriginal => ({
...(await importOriginal<Record<string, unknown>>()),
requestModelOptions: vi.fn()
}))
import { requestModelOptions } from '@/lib/model-options'
stubResizeObserver()
stubMenuDomApis()
const OPTIONS: ModelOptionsResponse = {
model: 'Qwen3.6-27B-UD-Q4_K_XL',
provider: 'llamacpp',
providers: [
{
slug: 'llamacpp',
name: 'Local',
models: ['Qwen3.6-27B-UD-Q4_K_XL'],
is_current: true,
authenticated: true
},
{
slug: 'nous',
name: 'Nous',
models: ['Hermes-4.5'],
authenticated: true
}
]
}
const DOWNLOAD_JOB: LocalRuntimeJob = {
job_id: 'dl1',
kind: 'model-download',
target: 'Qwen3.8 Flash Next (UD-Q4_K_XL)',
model_id: 'qwen3.8-flash-next',
status: 'running',
phase: 'downloading',
detail: '',
total_bytes: 100,
done_bytes: 41,
percent: 41,
error: null
}
function renderPicker(ui?: Partial<Parameters<typeof ModelPickerDialog>[0]>) {
const client = new QueryClient({ defaultOptions: { queries: { retry: false } } })
const element: ReactElement = (
<QueryClientProvider client={client}>
<I18nProvider>
<ModelPickerDialog
currentModel="Qwen3.6-27B-UD-Q4_K_XL"
currentProvider="llamacpp"
onOpenChange={() => undefined}
onSelect={() => undefined}
open
{...ui}
/>
</I18nProvider>
</QueryClientProvider>
)
return render(element)
}
beforeEach(() => {
vi.mocked(requestModelOptions).mockResolvedValue(OPTIONS)
$localRuntimeJobs.set([])
// These suites exercise the local-models rows, which ship behind --local.
$localModelsEnabled.set(true)
})
afterEach(() => {
cleanup()
vi.clearAllMocks()
})
describe('ModelPickerDialog download rows', () => {
it('shows an in-flight download as a disabled progress row in the Local group', async () => {
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderPicker()
expect(await screen.findByText('Qwen3.6-27B-UD-Q4_K_XL')).toBeTruthy()
const row = screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
expect(row).toBeTruthy()
expect(screen.getByText('41%')).toBeTruthy()
// Disabled: cmdk marks the item unselectable.
const item = row.closest('[cmdk-item]')
expect(item?.getAttribute('aria-disabled')).toBe('true')
})
it('shows a first-ever download under its own Local group when no local provider exists yet', async () => {
$localRuntimeJobs.set([DOWNLOAD_JOB])
vi.mocked(requestModelOptions).mockResolvedValue({
providers: [OPTIONS.providers![1]]
})
renderPicker()
expect(await screen.findByText('Hermes-4.5')).toBeTruthy()
expect(screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
expect(screen.getByText('41%')).toBeTruthy()
})
it('quickstart shows while downloading but not during later phases', async () => {
const quickstart: LocalRuntimeJob = { ...DOWNLOAD_JOB, job_id: 'q1', kind: 'quickstart', phase: 'downloading' }
$localRuntimeJobs.set([quickstart])
renderPicker()
expect(await screen.findByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
// The model is staged once quickstart moves on to activating it — the
// placeholder row must leave rather than sit beside the real model.
$localRuntimeJobs.set([{ ...quickstart, phase: 'starting-server' }])
await waitFor(() => {
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
})
})
it('refetches the model options when a download it saw running completes', async () => {
$localRuntimeJobs.set([DOWNLOAD_JOB])
renderPicker()
await screen.findByText('Qwen3.6-27B-UD-Q4_K_XL')
expect(vi.mocked(requestModelOptions).mock.calls.length).toBe(1)
$localRuntimeJobs.set([{ ...DOWNLOAD_JOB, status: 'done', phase: 'done' }])
await waitFor(() => {
expect(vi.mocked(requestModelOptions).mock.calls.length).toBe(2)
})
})
})
+169 -4
View File
@@ -1,12 +1,16 @@
import { useQuery } from '@tanstack/react-query'
import { useState } from 'react'
import { useEffect, useMemo, useState } from 'react'
import { getLocalModelsStatus } from '@/hermes'
import { useI18n } from '@/i18n'
import { modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options'
import { modelSearchText } from '@/lib/model-search-text'
import { currentPickerSelection } from '@/lib/model-status-label'
import { normalize } from '@/lib/text'
import type { ModelOptionProvider, ModelPricing } from '@/types/hermes'
import { useStoreSelector } from '@/lib/use-session-slice'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $localRuntimeJobs, runningModelDownloads, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
import type { LocalModelLoadProgress, ModelOptionProvider, ModelPricing } from '@/types/hermes'
import type { HermesGateway } from '../hermes'
import { cn } from '../lib/utils'
@@ -67,6 +71,81 @@ export function ModelPickerDialog({
enabled: open
})
// Live load state for the managed local server: which model is loading
// into memory right now, with a REAL percent (per-tensor callback relayed
// over the router's SSE stream). Polled only while the picker is open —
// 2s idle cadence is enough for a bar under a ~40s load. Errors read as
// "nothing loading" (remote-only installs have no local-models routes).
// Every local-models read here sits behind the --local launch flag (strict:
// the llamacpp provider group hides even with staged models on disk).
const localModelsEnabled = $localModelsEnabled.get()
const localStatus = useQuery({
queryKey: ['local-models-loading', profile],
queryFn: () => getLocalModelsStatus(),
enabled: open && localModelsEnabled,
refetchInterval: 2_000,
retry: false
})
const loadingModels: Record<string, LocalModelLoadProgress> = localStatus.data?.loading ?? {}
// Models on their way into the local library right now (downloads +
// quickstart runs), rendered as grayed progress rows. The jobs store
// republishes every ~700ms with fresh byte counts while anything runs —
// and this dialog stays MOUNTED app-wide when closed — so subscribe only
// to download identity (changes when a download starts/ends, and never
// while closed); each row selects its own percent scalar (#72163 class).
const downloadsKey = useStoreSelector($localRuntimeJobs, jobs =>
open && localModelsEnabled
? runningModelDownloads(jobs)
.map(job => `${job.job_id}\u0000${job.target}`)
.join('\u0001')
: ''
)
const downloads = useMemo(
() =>
downloadsKey === ''
? []
: downloadsKey.split('\u0001').map(pair => {
const [jobId, target] = pair.split('\u0000')
return { jobId, target }
}),
[downloadsKey]
)
// Rediscover in-flight work on open: the poller idles when nothing was
// running, and a download can start from any surface.
useEffect(() => {
if (open && localModelsEnabled) {
watchLocalRuntimeJobs()
}
}, [open, localModelsEnabled])
// A finished download turns into a real selectable model — refetch the
// options so the placeholder row is replaced while the picker is open.
const refetchOptions = modelOptions.refetch
useEffect(() => {
if (!open) {
return
}
let prevActive = runningModelDownloads($localRuntimeJobs.get()).length > 0
return $localRuntimeJobs.listen(next => {
const active = runningModelDownloads(next).length > 0
if (prevActive && !active) {
void refetchOptions()
}
prevActive = active
})
}, [open, refetchOptions])
const providers = modelOptions.data?.providers ?? []
const { model: optionsModel, provider: optionsProvider } = currentPickerSelection(
@@ -117,8 +196,10 @@ export function ModelPickerDialog({
<ModelResults
currentModel={optionsModel || currentModel}
currentProvider={optionsProvider || currentProvider}
downloads={downloads}
error={error}
loading={loading}
loadingModels={loadingModels}
onSelectModel={selectModel}
providers={providers}
search={search}
@@ -145,6 +226,8 @@ function ModelResults({
providers,
currentModel,
currentProvider,
downloads,
loadingModels,
onSelectModel,
search
}: {
@@ -153,6 +236,8 @@ function ModelResults({
providers: ModelOptionProvider[]
currentModel: string
currentProvider: string
downloads: { jobId: string; target: string }[]
loadingModels: Record<string, LocalModelLoadProgress>
onSelectModel: (provider: ModelOptionProvider, model: string) => void
search: string
}) {
@@ -188,15 +273,29 @@ function ModelResults({
// Only configured providers (those with curated models) are selectable
// here. Switching to a NOT-yet-configured provider goes through the
// "Add provider" footer button, which opens the full onboarding selector.
const configured = providers.filter(p => (p.models ?? []).length > 0)
// The local provider sits behind the --local launch flag (strict: staged
// models on disk don't show without it). Module-level read — a launch flag
// can't change mid-session.
const localModelsShown = $localModelsEnabled.get()
const configured = providers.filter(
p => (p.models ?? []).length > 0 && (localModelsShown || p.slug !== LOCAL_PROVIDER_SLUG)
)
// In-flight local downloads render as disabled progress rows: inside the
// Local group when it exists, else as their own group (first download —
// nothing staged yet, so the backend reports no Local provider at all).
const visibleDownloads = downloads.filter(job => !q || (job.target || '').toLowerCase().includes(q))
const hasLocalGroup = configured.some(p => p.slug === LOCAL_PROVIDER_SLUG)
return (
<>
{configured.map(provider => {
// Preserve the backend's curated order — filter in place, no re-sort.
const models = (provider.models ?? []).filter(m => matches(provider, m))
const groupDownloads = provider.slug === LOCAL_PROVIDER_SLUG ? visibleDownloads : []
if (models.length === 0) {
if (models.length === 0 && groupDownloads.length === 0) {
return null
}
@@ -215,6 +314,10 @@ function ModelResults({
const isCurrent = model === currentModel && provider.slug === currentProvider
const price = provider.pricing?.[model]
const locked = unavailable.has(model)
// Managed local model loading into memory right now: show the
// real load percent inline (keyed by exact model id — remote
// providers never match).
const loadProgress = loadingModels[model]
return (
<CommandItem
@@ -236,6 +339,19 @@ function ModelResults({
<span className="min-w-0 flex-1 truncate">
<HighlightMatches query={search} text={model} />
</span>
{loadProgress && (
<span className="flex shrink-0 items-center gap-1.5" title={copy.loadingIntoMemory}>
<span className="h-1 w-16 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<span
className="block h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(2, loadProgress.percent)}%` }}
/>
</span>
<span className="text-[0.62rem] tabular-nums text-muted-foreground">
{loadProgress.percent}%
</span>
</span>
)}
{locked && (
<span className="shrink-0 text-[0.62rem] uppercase tracking-wide opacity-80">{copy.pro}</span>
)}
@@ -243,6 +359,9 @@ function ModelResults({
</CommandItem>
)
})}
{groupDownloads.map(job => (
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
))}
{unavailable.size > 0 && (
<div className="px-6 pb-2 pt-1 text-[0.62rem] leading-relaxed text-muted-foreground">
{copy.proNeedsSubscription}
@@ -251,10 +370,56 @@ function ModelResults({
</CommandGroup>
)
})}
{!hasLocalGroup && visibleDownloads.length > 0 && (
<CommandGroup heading={copy.localDownloadsHeading} key="local-downloads">
{visibleDownloads.map(job => (
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
))}
</CommandGroup>
)}
</>
)
}
// The backend's provider row for staged local models (inventory.py's
// _local_runtime_row). Downloads-in-flight attach to this group.
const LOCAL_PROVIDER_SLUG = 'llamacpp'
// A model still downloading: visible so the user knows it's coming (and
// where it will land), disabled so it can't be selected early, with the
// same byte progress the settings pane shows. Percent is selected here, per
// row, so the poller's 700ms byte ticks repaint this leaf only.
function DownloadingModelRow({ jobId, target }: { jobId: string; target: string }) {
const { t } = useI18n()
const copy = t.modelPicker
const percent = useStoreSelector(
$localRuntimeJobs,
jobs => jobs.find(job => job.job_id === jobId)?.percent ?? null
)
return (
<CommandItem
className="flex items-center gap-2 pl-6 font-mono opacity-60"
disabled
value={`downloading:${jobId}`}
>
<span className="min-w-0 flex-1 truncate">{target}</span>
<span className="flex shrink-0 items-center gap-1.5" title={copy.downloading}>
<span className="h-1 w-16 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
<span
className="block h-full rounded-full bg-primary transition-[width] duration-500"
style={{ width: `${Math.max(2, percent ?? 0)}%` }}
/>
</span>
<span className="text-[0.62rem] tabular-nums text-muted-foreground">
{typeof percent === 'number' ? `${percent}%` : copy.downloading}
</span>
</span>
</CommandItem>
)
}
// Compact In/Out $/Mtok price tag, mirroring the CLI picker's price columns.
// Renders nothing when pricing is unavailable for the model.
function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boolean }) {
@@ -11,6 +11,7 @@ import { Check, ChevronDown, ChevronLeft, KeyRound, Loader2 } from '@/lib/icons'
import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors'
import { cn } from '@/lib/utils'
import { $desktopBoot, type DesktopBootState } from '@/store/boot'
import { $localModelsEnabled } from '@/store/local-models-flag'
import {
$desktopOnboarding,
clearPendingProviderOAuth,
@@ -32,6 +33,7 @@ import { DocsLink, FlowPanel, Status } from './flow'
import {
FeaturedProviderRow,
FireworksProviderRow,
LocalModelsProviderRow,
OpenRouterProviderRow,
ProviderRow,
sortProviders
@@ -41,6 +43,7 @@ export {
FeaturedProviderRow,
FireworksProviderRow,
KeyProviderRow,
LocalModelsProviderRow,
OpenRouterProviderRow,
ProviderRow,
providerTitle,
@@ -478,10 +481,29 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
const collapsible = Boolean(featured)
const showRest = !collapsible || showAll
// "Run models locally" leaves the picker for Settings -> Providers ->
// Local Models, where install/download live. First-run: persist the skip
// (same contract as ChooseLaterLink) so the blocking overlay never
// re-nags; manual mode just closes. window.location keeps this picker
// router-independent (it renders outside the route tree on first run).
const openLocalModels = () => {
if (manual) {
closeManualOnboarding()
} else {
dismissFirstRunOnboarding()
}
window.location.hash = '#/settings?tab=providers&pview=local'
}
return (
<div className="grid gap-2">
<div className="grid max-h-[60dvh] gap-2 overflow-y-auto p-1">
{featured ? <FeaturedProviderRow onSelect={select} provider={featured} /> : null}
{/* The no-account path: everything runs on this machine. Shipped
behind the --local launch flag. (Fireworks moved into the
expanded list on main.) */}
{$localModelsEnabled.get() ? <LocalModelsProviderRow onClick={openLocalModels} /> : null}
{showRest ? (
<>
{/* Fireworks leads the expanded list, matching CANONICAL_PROVIDERS
@@ -95,6 +95,14 @@ export function FireworksProviderRow({ onClick }: { onClick: () => void }) {
return <KeyProviderRow onClick={onClick} pitch={t.onboarding.fireworksPitch} title="Fireworks AI" />
}
/** Onboarding row for the managed local runtime: no account, no key — the
* destination is the Local Models pane where install/download live. */
export function LocalModelsProviderRow({ onClick }: { onClick: () => void }) {
const { t } = useI18n()
return <KeyProviderRow onClick={onClick} pitch={t.onboarding.localModelsPitch} title={t.onboarding.localModelsTitle} />
}
export function OpenRouterProviderRow({ onClick }: { onClick: () => void }) {
const { t } = useI18n()
@@ -98,6 +98,7 @@ export function TipHost() {
return (
<TipBubble
action={tip.action}
anchor={anchor}
keybind={tip.keybind}
onClose={retireActiveTip}
@@ -0,0 +1,144 @@
/**
* The campaign offer against the live stores: eligibility fetch, the showTip
* wiring (button included), retirement, the reshow clock, and the cursor
* guard that keeps a campaign showing from restarting the rotation's walk.
*/
import { cleanup } from '@testing-library/react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const getLocalModelsStatus = vi.fn()
const getLocalCatalog = vi.fn()
vi.mock('@/hermes', () => ({
getLocalCatalog: (...args: unknown[]) => getLocalCatalog(...args),
getLocalModelsStatus: (...args: unknown[]) => getLocalModelsStatus(...args)
}))
import { en } from '@/i18n/en'
import { LOCAL_SETUP_TIP_ID } from '@/lib/tips/local-cta'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $connection } from '@/store/session'
import { $activeTip, $lastTipId, $retiredTips, $tipShownAt } from '@/store/tips'
import { offerLocalSetupTip, resetLocalSetupOfferCache } from './local-setup-offer'
function primeEligibleBackend() {
getLocalModelsStatus.mockResolvedValue({ models: [], runtime_installed: false })
getLocalCatalog.mockResolvedValue({ models: [{ fits: true, id: 'qwen3.8-27b' }] })
}
async function flushFetch() {
await Promise.resolve()
await Promise.resolve()
await Promise.resolve()
}
describe('offerLocalSetupTip', () => {
beforeEach(() => {
resetLocalSetupOfferCache()
// The campaign ships behind --local like every local-models surface.
$localModelsEnabled.set(true)
$activeTip.set(null)
$retiredTips.set([])
$tipShownAt.set({})
$lastTipId.set(null)
$connection.set({ mode: 'local' } as never)
getLocalModelsStatus.mockReset()
getLocalCatalog.mockReset()
})
afterEach(() => {
cleanup()
})
it('holds the first quiet moment while the read flies, then shows on the next', async () => {
primeEligibleBackend()
const openLocalModels = vi.fn()
// First offer: fetch in flight — the moment is HELD (true, so the
// rotation's walk cannot take it and arm the cooldown ahead of the
// campaign), but nothing is on screen yet.
expect(offerLocalSetupTip(en.tips, openLocalModels)).toBe(true)
expect($activeTip.get()).toBeNull()
await flushFetch()
// Second offer: cached yes — bubble goes up with the CTA wired.
expect(offerLocalSetupTip(en.tips, openLocalModels)).toBe(true)
const tip = $activeTip.get()
expect(tip?.tipId).toBe(LOCAL_SETUP_TIP_ID)
expect(tip?.action?.label).toBe(en.tips.items['local-setup'].action)
tip?.action?.onSelect()
expect(openLocalModels).toHaveBeenCalledTimes(1)
// The CTA closes the bubble on its way to the pane.
expect($activeTip.get()).toBeNull()
})
it('never restarts the rotation walk: the campaign id stays out of the cursor', async () => {
primeEligibleBackend()
$lastTipId.set('cron')
offerLocalSetupTip(en.tips, vi.fn())
await flushFetch()
offerLocalSetupTip(en.tips, vi.fn())
expect($activeTip.get()?.tipId).toBe(LOCAL_SETUP_TIP_ID)
expect($lastTipId.get()).toBe('cron')
})
it('stays quiet on an ineligible machine without refetching', async () => {
getLocalModelsStatus.mockResolvedValue({ models: [{ id: 'staged' }], runtime_installed: true })
getLocalCatalog.mockResolvedValue({ models: [{ fits: true, id: 'qwen3.8-27b' }] })
offerLocalSetupTip(en.tips, vi.fn())
await flushFetch()
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
expect($activeTip.get()).toBeNull()
expect(getLocalModelsStatus).toHaveBeenCalledTimes(1)
})
it('honors the ✕ forever and the ignored-bubble clock for a week', async () => {
primeEligibleBackend()
$retiredTips.set([LOCAL_SETUP_TIP_ID])
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
expect(getLocalModelsStatus).not.toHaveBeenCalled()
$retiredTips.set([])
$tipShownAt.set({ [LOCAL_SETUP_TIP_ID]: Date.now() - 60_000 })
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
expect(getLocalModelsStatus).not.toHaveBeenCalled()
})
it('asks nothing of a remote backend', () => {
$connection.set({ mode: 'remote' } as never)
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
expect(getLocalModelsStatus).not.toHaveBeenCalled()
})
it('never runs without the --local launch flag (strict), even on an eligible machine', () => {
$localModelsEnabled.set(false)
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
// Declined before any read: no fetch, no held moment, no cooldown spent.
expect(getLocalModelsStatus).not.toHaveBeenCalled()
expect($activeTip.get()).toBeNull()
})
it('a failed read stands down for the session instead of retrying', async () => {
getLocalModelsStatus.mockRejectedValue(new Error('backend gone'))
getLocalCatalog.mockRejectedValue(new Error('backend gone'))
offerLocalSetupTip(en.tips, vi.fn())
await flushFetch()
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
expect(getLocalModelsStatus).toHaveBeenCalledTimes(1)
})
})
@@ -0,0 +1,126 @@
/**
* The local-setup campaign: one bubble on the model pill for machines that
* could run local models and haven't set them up.
*
* Not a rotation tip — a campaign the rotation CONSULTS first at each quiet
* due moment (use-tip-rotation.ts): conditional (most machines qualify or
* don't, permanently), actionable (it carries the one button a tip may
* have), and perishable (setting up local models — or the ✕ — ends it).
* A live "your GPU can run this, free and private" outranks the walk's
* "the model name is a button" whenever both are true, and an ignored
* bubble may return in a week rather than walking on forever.
*
* Eligibility is fetched, not assumed: the backend's own fit check (the
* same catalog `fits` the Local Models pane prices its hero with) decides
* whether this machine qualifies. Reads are lazy — nothing polls for a
* bubble. The first quiet due moment kicks one status+catalog read and
* holds the turn (no walk tip may spend the cooldown ahead of a pending
* campaign); the cached answer serves every later one. Completing
* setup flips the next read to ineligible, so the campaign retires itself
* without bookkeeping — and the cache dies with a connection change,
* because eligibility is a fact about the backend's machine.
*/
import { getLocalCatalog, getLocalModelsStatus } from '@/hermes'
import type { Translations } from '@/i18n/types'
import { LOCAL_SETUP_TIP_ID, localSetupDue, localSetupEligible } from '@/lib/tips/local-cta'
import { $localModelsEnabled } from '@/store/local-models-flag'
import { $connection } from '@/store/session'
import { $retiredTips, $tipShownAt, dismissTip, showTip } from '@/store/tips'
/** The pill the bubble points at — the same handle the rotation's
* model-switch tip uses, so the two can never drift to different anchors. */
const MODEL_PILL_TARGETS = ['[data-tour="model-pill"]'] as const
let eligibilityCache: { eligible: boolean } | null = null
let eligibilityInFlight = false
let boundToConnection = false
/** Reset the session cache — tests only. */
export function resetLocalSetupOfferCache(): void {
eligibilityCache = null
eligibilityInFlight = false
}
/**
* Offer the campaign the current quiet moment. True = it put its bubble up
* and the moment is spent; false = the rotation's walk may have it.
*/
export function offerLocalSetupTip(copy: Translations['tips'], openLocalModels: () => void): boolean {
// Local models ship behind the --local launch flag; without it there is
// no Local Models pane for the button to open, so the campaign never runs
// (and never spends a status/catalog read).
if (!$localModelsEnabled.get()) {
return false
}
if ($retiredTips.get().includes(LOCAL_SETUP_TIP_ID)) {
return false
}
if (!localSetupDue(Date.now(), $tipShownAt.get()[LOCAL_SETUP_TIP_ID])) {
return false
}
// Local backends only: on a remote connection (cloud resolves to remote)
// the models would run on the far machine, and "stays on your computer"
// would be promising someone else's computer. Checked before the cache so
// a re-home mid-session can't serve a stale yes.
if (($connection.get()?.mode ?? null) !== 'local') {
return false
}
if (!boundToConnection) {
boundToConnection = true
$connection.listen(() => resetLocalSetupOfferCache())
}
if (!eligibilityCache) {
if (!eligibilityInFlight) {
eligibilityInFlight = true
void Promise.all([getLocalModelsStatus(), getLocalCatalog()])
.then(([status, catalog]) => {
eligibilityCache = {
eligible: localSetupEligible($connection.get()?.mode ?? null, status, catalog.models)
}
})
.catch(() => {
// No backend answer, no campaign this session. The next launch —
// or the next connection — asks again.
eligibilityCache = { eligible: false }
})
.finally(() => {
eligibilityInFlight = false
})
}
// Hold the moment while the read flies: nothing shows and no cooldown
// arms, so the next tick answers from the cache. Handing this moment to
// the rotation instead would put a walk tip up first and park the
// campaign behind the six-hour cooldown — the exact inversion of the
// priority. Costs an ineligible machine one 30s tick, once per session.
return true
}
if (!eligibilityCache.eligible) {
return false
}
showTip({
action: {
label: copy.items['local-setup'].action,
onSelect: () => {
dismissTip()
openLocalModels()
}
},
side: 'top',
targets: MODEL_PILL_TARGETS,
text: copy.items['local-setup'].text,
tipId: LOCAL_SETUP_TIP_ID,
title: copy.items['local-setup'].title
})
return true
}
@@ -21,8 +21,11 @@ import { useI18n } from '@/i18n'
import { iconSize, X } from '@/lib/icons'
import { useKeybindHint } from '@/lib/keybinds/use-keybind-hint'
import type { TipSide } from '@/lib/tips/catalog'
import type { ActiveTip } from '@/store/tips'
export interface TipBubbleProps {
/** A call to action rendered as the bubble's one button. See ActiveTip. */
action?: ActiveTip['action']
/** The element the arrow points at. */
anchor: HTMLElement
/** Keybind action id; its live combo prints under the text. */
@@ -34,7 +37,7 @@ export interface TipBubbleProps {
title?: string
}
export function TipBubble({ anchor, keybind, onClose, side, text, title }: TipBubbleProps) {
export function TipBubble({ action, anchor, keybind, onClose, side, text, title }: TipBubbleProps) {
const { t } = useI18n()
const combo = useKeybindHint(keybind ?? '')
const anchorRef = useRef<HTMLElement | null>(anchor)
@@ -81,6 +84,19 @@ export function TipBubble({ anchor, keybind, onClose, side, text, title }: TipBu
{text}
</p>
{combo && <KbdCombo className="mt-2" combo={combo} size="sm" variant="inverted" />}
{action && (
// The CTA: still not a focus trap — the button is tabbable when
// reached but nothing steals the caret to get there. Inverted
// fill against the accent surface, same currentColor discipline
// as the rest of the bubble.
<button
className="mt-2.5 inline-flex cursor-pointer items-center rounded-md bg-current/15 px-2.5 py-1 text-[length:var(--conversation-caption-font-size)] font-semibold transition-colors hover:bg-current/25"
onClick={action.onSelect}
type="button"
>
{action.label}
</button>
)}
</div>
<button
aria-label={t.tips.close}
@@ -20,7 +20,9 @@
*/
import { useEffect } from 'react'
import { useNavigate } from 'react-router'
import { SETTINGS_ROUTE } from '@/app/routes'
import type { Translations } from '@/i18n/types'
import { resolveTipAnchor } from '@/lib/tips/anchor'
import { TIP_CATALOG } from '@/lib/tips/catalog'
@@ -28,6 +30,8 @@ import { nextTip } from '@/lib/tips/rotation'
import { $awaitingResponse, $busy } from '@/store/session'
import { $activeTip, $lastTipId, $nextTipAt, $retiredTips, $tipsEnabled, showTip } from '@/store/tips'
import { offerLocalSetupTip } from './local-setup-offer'
const TICK_MS = 30_000
/** Nothing in the first stretch of a launch, however long the cooldown says
* it's been: you opened the app to do a thing, and the tip can wait until
@@ -60,6 +64,8 @@ function appIsQuiet(lastTypedAt: number): boolean {
/** Drive the ambient rotation for as long as the host is mounted. */
export function useTipRotation(copy: Translations['tips']) {
const navigate = useNavigate()
useEffect(() => {
let lastTypedAt = 0
let settledAt = Date.now() + SETTLE_MIN_MS + Math.random() * SETTLE_SPREAD_MS
@@ -83,6 +89,18 @@ export function useTipRotation(copy: Translations['tips']) {
return
}
// Campaigns outrank the walk: a conditional, actionable tip that is
// live right now (the local-setup CTA) says something about THIS
// machine, which beats the catalog's standing introduction. It shares
// the cooldown, so taking the moment still costs it the usual hours.
if (
offerLocalSetupTip(copy, () => {
navigate(`${SETTINGS_ROUTE}?tab=providers&pview=local`)
})
) {
return
}
// Only tips with something on screen to point at are candidates, so the
// rotation never burns a turn on a pane the user isn't showing.
const onScreen = TIP_CATALOG.filter(tip => resolveTipAnchor(document, tip.targets))
@@ -128,5 +146,5 @@ export function useTipRotation(copy: Translations['tips']) {
window.clearInterval(timer)
window.removeEventListener('keydown', noteTyping, true)
}
}, [copy])
}, [copy, navigate])
}
+1
View File
@@ -13,6 +13,7 @@ const badgeVariants = cva(
variant: {
default: 'bg-primary/10 text-primary',
muted: 'bg-muted text-muted-foreground',
success: 'bg-emerald-500/10 text-emerald-600 dark:text-emerald-300',
warn: 'bg-amber-500/10 text-amber-600 dark:text-amber-300',
destructive: 'bg-destructive/10 text-destructive',
outline: 'border border-(--ui-stroke-secondary) text-muted-foreground',
+3
View File
@@ -301,6 +301,9 @@ declare global {
glassSupported?: boolean
/** Main-process fact: this OS can do any translucency at all (not Linux). */
translucencySupported?: boolean
/** Launch flag: the app was started with --local, enabling the
* local-models GUI surfaces. Absent/false = every local surface hides. */
localModelsEnabled?: boolean
setTranslucency?: (payload: TranslucencyState) => void
setKeepAwake?: (on: boolean) => void
setDisableF12?: (blocked: boolean) => void
+1
View File
@@ -18,6 +18,7 @@ export {
export type { ProfileScope } from './api/client'
export * from './api/config'
export * from './api/cron'
export * from './api/local-models'
export * from './api/mcp'
export * from './api/messaging'
export * from './api/models'
+5
View File
@@ -2890,6 +2890,11 @@ export const ar = defineLocale({
title: 'بدّل النموذج أثناء المحادثة',
text: 'اسم النموذج زر. غيّره كلما تغيّرت طبيعة العمل.'
},
'local-setup': {
title: 'هذا الجهاز يمكنه تشغيل النماذج محليًا',
text: 'عتادك قادر على تشغيل نموذج محلي. تبقى محادثاتك على جهازك ولا تكلف شيئًا.',
action: 'إعداد الآن'
},
'right-pane': {
title: 'لوحة العمل',
text: 'الملفات والطرفية والمراجعة والمتصفح المدمج تتشارك اللوحة الجانبية.'
+137
View File
@@ -395,6 +395,7 @@ export const en: Translations = {
providerAccounts: 'Accounts',
providerApiKeys: 'API keys',
providerCustomEndpoints: 'Custom Endpoints',
providerLocalModels: 'Local Models',
gateway: 'Gateways',
apiKeys: 'Tools & Keys',
keybinds: 'Keyboard Shortcuts',
@@ -1117,6 +1118,121 @@ export const en: Translations = {
curator: { label: 'Curator', hint: 'Skill-usage review' }
}
},
localModels: {
title: 'Local Models',
runtimeTitle: 'Local runtime',
runtimeReady: backend => `Ready · ${backend}`,
serverRunning: 'Running',
runtimeInstalled: 'llama.cpp runtime installed',
runtimeInstalledDetail: (tag, backend) =>
`Build ${tag}, ${backend} backend. Hermes starts and manages the server for you.`,
installTitle: 'Install the local runtime',
installDetail:
'Downloads the llama.cpp inference engine (a few hundred MB). Models you download run entirely on this machine — no account, nothing leaves your computer.',
installAction: 'Install runtime',
installing: 'Installing runtime…',
installFailed: 'Runtime install failed',
hardwareTitle: 'This machine',
hardwareLoading: 'Checking your hardware…',
vram: label => `${label} GPU memory`,
ram: label => `${label} RAM`,
unifiedMemory: 'Unified memory',
modelsTitle: 'Models',
recommended: 'Recommended',
/* The Recommended badge's tooltip, keyed by the resolver branch that
made the pick. Qualitative on purpose: predictions order candidates,
they are not promises to print. */
recommendedReason: {
'best-quality-resident':
'The highest-quality model that runs entirely on your GPU at full speed. Picks weigh quality against predicted speed on this hardware.',
'speed-gated-quality':
'A higher-quality model fits this machine but would respond too slowly on its memory bandwidth — this is the best model that stays fast.',
'fastest-resident':
'No model reaches full speed on this hardware; this one comes closest while running entirely in GPU memory.',
'least-painful-spilled':
'No model fits entirely in GPU memory here — this one runs best from system RAM.'
} as Record<string, string>,
downloaded: 'Downloaded',
downloadAction: size => `Download · ${size}`,
downloadProgress: (done, total) => `Downloading ${done} of ${total}`,
downloadDoneToast: model => `${model} is ready.`,
installDoneToast: 'Local runtime installed and ready.',
quickstartTitle: 'Run a model on this machine',
quickstartDetail: (model, size) =>
`One click sets everything up: the local engine, ${model} (${size} download), and your default for new chats. Nothing leaves this computer.`,
quickstartDetailReady: model => `One click makes ${model} your default for new chats. Everything runs on this machine.`,
quickstartAction: 'Set up for me',
quickstartConfigure: 'Configure…',
quickstartDoneToast: model => `${model} is set up — new chats run on this machine.`,
quickstartFailed: 'Local model setup failed',
quickstartStageEngine: 'Engine',
quickstartStageModel: 'Model',
quickstartStageFinish: 'Finish',
useAction: 'Use',
activePill: 'Default',
updateTitle: 'Engine update available',
updateDetail: (next, current) => `A newer llama.cpp build (${next}) is ready to install — you're on ${current}. Models keep working during the download.`,
updateAction: 'Update engine',
updating: 'Updating engine…',
upToDateTitle: 'Engine up to date',
upToDateDetail: (tag, backend) => `Running llama.cpp ${tag} (${backend}) — the latest build Hermes ships.`,
updateToast: next => `A newer local engine build (${next}) is available. Update from Settings → Local Models.`,
activeDetail: 'New chats use this model — it loads when you send your first message',
activeNotLoaded: 'Loads on your first message',
loadedPill: 'In memory',
placementResident: 'all on GPU',
placementSpilled: 'partly in RAM',
placementResidentTip: 'Running entirely in GPU memory at this context window — full speed.',
placementSpilledTip:
'Part of this model runs from system RAM — it works, but slower. A more compact build or a smaller context would fit fully.',
loadingPill: 'Loading…',
ejectTip: 'Free GPU memory (loads again on the next message)',
ejected: 'Model unloaded — GPU memory freed.',
ejectFailed: 'Could not unload the model',
stopServer: 'Turn off',
startServer: 'Turn on',
runtimeRunningDetail: 'The local server is running. Turning it off frees all GPU memory and stops new chats from using local models until you turn it back on.',
serverStopped: 'Local server stopped — GPU memory freed.',
serverStarted: 'Local server running.',
serverStopFailed: 'Could not stop the local server',
serverStartFailed: 'Could not start the local server',
activating: 'Starting…',
activateFailed: model => `Could not switch to ${model}`,
activateDoneToast: model => `New chats use ${model}.`,
downloadFailed: model => `Download of ${model} failed`,
pillFitsGpu: 'Fits your GPU',
pillUsesRam: 'Uses system RAM',
pillTooBig: 'Too big for this machine',
browseTitle: 'Find more models',
browseHint: 'Search all of Hugging Face. Models you download here are sized to your machine automatically, but not tested by us.',
browsePlaceholder: 'Search models by name or author…',
browseSearching: 'Searching Hugging Face',
browseListing: 'Reading model files',
browseShowFiles: 'Show files',
browseRefresh: 'Refresh',
browseDownloads: 'downloads',
browseLikes: 'likes',
browseGated: 'requires Hugging Face sign-in',
browseNoGguf: 'No compatible model files found.',
browseFitUnknown: 'Fit unknown',
browseAlreadyDownloaded: 'Already downloaded.',
addedByYou: 'Added by you',
browseDownloadStarted: 'Downloading {name}',
browseDownloadAria: 'Download {name}',
sideloadButton: 'Add model file',
sideloadTitle: 'Choose a GGUF model file',
sideloadDone: 'Added {name}.',
sideloadAlreadyPresent: 'Already in your library.',
pillFullContext: max => `Full ${max} context`,
pillFullContextTip: "Runs at the model's complete context window from the start",
pillUpTo: max => `Up to ${max} context`,
pillGrowsTip: 'Grows automatically as your conversation needs more room',
pillVision: 'Sees images',
deleteAction: 'Delete model',
deleteConfirm: model => `Delete ${model} from disk?`,
deleted: model => `${model} deleted.`,
deleteFailed: 'Delete failed'
},
providers: {
connectAccount: 'Connect an account',
haveApiKey: 'Have an API key instead?',
@@ -2762,6 +2878,8 @@ export const en: Translations = {
connected: 'Connected',
featuredPitch: 'One subscription, 300+ frontier models — the recommended way to run Hermes',
fireworksPitch: 'Direct model API — Fireworks-hosted frontier models',
localModelsTitle: 'Run models locally',
localModelsPitch: 'No account needed — download a model and run it on this machine',
openRouterPitch: 'One key, hundreds of models — a solid default',
apiKeyOptions: {
fireworks: {
@@ -2835,6 +2953,9 @@ export const en: Translations = {
noModels: 'No models found.',
addProvider: 'Add provider',
loadFailed: 'Could not load models',
loadingIntoMemory: 'Loading into memory',
downloading: 'Downloading',
localDownloadsHeading: 'Local',
noAuthenticatedProviders: 'No authenticated providers.',
pro: 'Pro',
proNeedsSubscription: 'Pro models need a paid Nous subscription.',
@@ -2961,6 +3082,15 @@ export const en: Translations = {
openStarmap: 'Open memory graph',
turnRunning: 'Running',
contextUsage: 'Context usage',
systemResources: {
title: 'System Resources',
loading: 'Resources…',
gpuUtilization: 'GPU utilization',
gpuMemory: 'GPU memory',
ram: 'RAM',
unifiedNote: 'Unified memory — the GPU and system share this pool.',
toggle: 'System resources'
},
contextUsagePanel: {
categories: {
conversation: 'Conversation',
@@ -3217,6 +3347,8 @@ export const en: Translations = {
loadingSession: 'Loading session',
showEarlier: 'Show earlier messages',
loadingResponse: 'Hermes is loading a response',
loadingLocalModel: model => `Loading ${model} into memory`,
processingPrompt: 'Processing prompt',
resumeWhenBackgroundDone: count =>
count === 1
? 'Will resume when the background task finishes'
@@ -3545,6 +3677,11 @@ export const en: Translations = {
title: 'Switch models mid-thread',
text: 'The model name is a button. Change it whenever the work changes shape.'
},
'local-setup': {
title: 'This machine can run models locally',
text: 'Your hardware can serve a local model. Chats stay on your computer and cost nothing.',
action: 'Set it up'
},
'right-pane': {
title: 'The working pane',
text: 'Files, terminal, review and the in-app browser share the right side.'
+119
View File
@@ -278,6 +278,7 @@ export const ja = defineLocale({
providerAccounts: 'アカウント',
providerApiKeys: 'API キー',
providerCustomEndpoints: 'カスタムエンドポイント',
providerLocalModels: 'ローカルモデル',
gateway: 'ゲートウェイ',
apiKeys: 'ツールとキー',
keybinds: 'キーボードショートカット',
@@ -1023,6 +1024,106 @@ export const ja = defineLocale({
curator: { label: 'キュレーター', hint: 'スキル使用レビュー' }
}
},
localModels: {
title: 'ローカルモデル',
runtimeTitle: 'ローカルランタイム',
runtimeReady: backend => `準備完了 · ${backend}`,
serverRunning: '実行中',
runtimeInstalled: 'llama.cpp ランタイムをインストール済み',
runtimeInstalledDetail: (tag, backend) =>
`ビルド ${tag}、${backend} バックエンド。サーバーは Hermes が起動・管理します。`,
installTitle: 'ローカルランタイムをインストール',
installDetail:
'llama.cpp 推論エンジン(数百 MB)をダウンロードします。ダウンロードしたモデルはすべてこのマシン上で動作します——アカウント不要、データが外部に送られることはありません。',
installAction: 'ランタイムをインストール',
installing: 'ランタイムをインストール中…',
installFailed: 'ランタイムのインストールに失敗しました',
hardwareTitle: 'このマシン',
hardwareLoading: 'ハードウェアを確認中…',
vram: label => `GPU メモリ ${label}`,
ram: label => `RAM ${label}`,
unifiedMemory: 'ユニファイドメモリ',
modelsTitle: 'モデル',
recommended: 'おすすめ',
recommendedReason: {
'best-quality-resident':
'GPU に完全に載り、フルスピードで動くモデルの中で最高品質です。おすすめは品質とこのハードウェアでの予測速度を両立させて選ばれます。',
'speed-gated-quality':
'より高品質なモデルもこのマシンに載りますが、メモリ帯域の制約で応答が遅くなります — これは速度を保てる最良のモデルです。',
'fastest-resident':
'このハードウェアでフルスピードに達するモデルはありません。GPU メモリ内で動くものの中で最速です。',
'least-painful-spilled':
'GPU メモリに完全に収まるモデルはありません — システム RAM からの実行で最も快適なモデルです。'
} as Record<string, string>,
downloaded: 'ダウンロード済み',
downloadAction: size => `ダウンロード · ${size}`,
downloadProgress: (done, total) => `ダウンロード中 ${done} / ${total}`,
downloadDoneToast: model => `${model} の準備ができました。`,
installDoneToast: 'ローカルランタイムのインストールが完了しました。',
useAction: '使用する',
activePill: 'デフォルト',
updateTitle: 'エンジンの更新があります',
updateDetail: (next, current) => `新しい llama.cpp ビルド(${next})をインストールできます——現在は ${current} です。ダウンロード中もモデルは引き続き使えます。`,
updateAction: 'エンジンを更新',
updating: 'エンジンを更新中…',
upToDateTitle: 'エンジンは最新です',
upToDateDetail: (tag, backend) => `llama.cpp ${tag}(${backend})で動作中——Hermes が提供する最新ビルドです。`,
updateToast: next => `ローカルエンジンの新しいビルド(${next})があります。設定 → ローカルモデル から更新できます。`,
activeDetail: '新しいチャットはこのモデルを使用——最初のメッセージ送信時に読み込みます',
activeNotLoaded: '最初のメッセージで読み込みます',
loadedPill: '読み込み済み',
placementResident: 'すべて GPU 上',
placementSpilled: '一部 RAM 上',
placementResidentTip: 'このコンテキストウィンドウで GPU メモリ内で完全に動作しています — フルスピード。',
placementSpilledTip: 'モデルの一部がシステム RAM から動作しています — 動作しますが遅くなります。よりコンパクトなビルドか小さいコンテキストなら完全に収まります。',
loadingPill: '読み込み中…',
ejectTip: 'GPU メモリを解放(必要時に再読み込み)',
ejected: 'モデルをアンロードしました——GPU メモリを解放しました。',
ejectFailed: 'モデルをアンロードできませんでした',
stopServer: 'オフにする',
startServer: 'オンにする',
runtimeRunningDetail: 'ローカルサーバーが実行中です。オフにすると GPU メモリを全て解放し、再度オンにするまで新しいチャットはローカルモデルを使用しません。',
serverStopped: 'ローカルサーバーを停止しました——GPU メモリを解放しました。',
serverStarted: 'ローカルサーバー実行中。',
serverStopFailed: 'ローカルサーバーを停止できませんでした',
serverStartFailed: 'ローカルサーバーを起動できませんでした',
activating: '起動中…',
activateFailed: model => `${model} への切り替えに失敗しました`,
activateDoneToast: model => `新しいチャットは ${model} を使用します。`,
downloadFailed: model => `${model} のダウンロードに失敗しました`,
pillFitsGpu: 'GPU に完全に収まります',
pillUsesRam: 'システム RAM を使用',
pillTooBig: 'このマシンには大きすぎます',
browseTitle: 'さらにモデルを探す',
browseHint: 'Hugging Face 全体を検索できます。ここでダウンロードしたモデルは自動でマシンに合わせて動作しますが、当方でのテストは行われていません。',
browsePlaceholder: 'モデル名または作者で検索…',
browseSearching: 'Hugging Face を検索中',
browseListing: 'モデルファイルを読み込み中',
browseShowFiles: 'ファイルを表示',
browseRefresh: '更新',
browseDownloads: 'ダウンロード',
browseLikes: 'いいね',
browseGated: 'Hugging Face へのサインインが必要',
browseNoGguf: '互換性のあるモデルファイルが見つかりません。',
browseFitUnknown: '適合状況は不明',
browseAlreadyDownloaded: 'ダウンロード済みです。',
addedByYou: 'あなたが追加',
browseDownloadStarted: '{name} をダウンロード中',
browseDownloadAria: '{name} をダウンロード',
sideloadButton: 'モデルファイルを追加',
sideloadTitle: 'GGUF モデルファイルを選択',
sideloadDone: '{name} を追加しました。',
sideloadAlreadyPresent: '既にライブラリにあります。',
pillFullContext: max => `フル ${max} コンテキスト`,
pillFullContextTip: '最初からモデルの完全なコンテキストウィンドウで動作します',
pillUpTo: max => `最大 ${max} コンテキスト`,
pillGrowsTip: '会話が必要とするにつれて自動的に拡張します',
pillVision: '画像対応',
deleteAction: 'モデルを削除',
deleteConfirm: model => `${model} をディスクから削除しますか?`,
deleted: model => `${model} を削除しました。`,
deleteFailed: '削除に失敗しました'
},
providers: {
connectAccount: 'アカウントを接続',
haveApiKey: 'API キーをお持ちですか?',
@@ -2406,6 +2507,8 @@ export const ja = defineLocale({
connected: '接続済み',
featuredPitch: '1 つのサブスクリプションで 300 以上の最先端モデル — Hermes を実行するための推奨方法',
fireworksPitch: '直接モデル API — Fireworks がホストする最先端モデル',
localModelsTitle: 'モデルをローカルで実行',
localModelsPitch: 'アカウント不要——モデルをダウンロードしてこのマシンで実行',
openRouterPitch: '1 つのキーで数百のモデル — 堅実なデフォルト',
apiKeyOptions: {
fireworks: {
@@ -2479,6 +2582,8 @@ export const ja = defineLocale({
noModels: 'モデルが見つかりません。',
addProvider: 'プロバイダーを追加',
loadFailed: 'モデルを読み込めませんでした',
downloading: 'ダウンロード中',
localDownloadsHeading: 'ローカル',
noAuthenticatedProviders: '認証済みプロバイダーがありません。',
pro: 'Pro',
proNeedsSubscription: 'Pro モデルには有料の Nous サブスクリプションが必要です。',
@@ -2591,6 +2696,15 @@ export const ja = defineLocale({
openStarmap: 'メモリグラフを開く',
turnRunning: '実行中',
contextUsage: 'コンテキスト使用状況',
systemResources: {
title: 'システムリソース',
loading: 'リソース…',
gpuUtilization: 'GPU 使用率',
gpuMemory: 'GPU メモリ',
ram: 'RAM',
unifiedNote: 'ユニファイドメモリ——GPU とシステムがこのプールを共有します。',
toggle: 'システムリソース'
},
contextUsagePanel: {
categories: {
conversation: '会話',
@@ -3160,6 +3274,11 @@ export const ja = defineLocale({
title: '会話の途中でモデルを変更',
text: 'モデル名はボタンです。作業の性質が変わったら切り替えてください。'
},
'local-setup': {
title: 'このマシンはローカルでモデルを実行できます',
text: 'お使いのハードウェアでローカルモデルを動かせます。会話はこのコンピュータから出ず、料金もかかりません。',
action: 'セットアップ'
},
'right-pane': {
title: '作業用ペイン',
text: 'ファイル、ターミナル、レビュー、アプリ内ブラウザはサイドペインにまとまっています。'
+124 -2
View File
@@ -338,6 +338,7 @@ export interface Translations {
providerAccounts: string
providerApiKeys: string
providerCustomEndpoints: string
providerLocalModels: string
gateway: string
apiKeys: string
keybinds: string
@@ -966,6 +967,107 @@ export interface Translations {
notInCatalog: string
tasks: Record<string, AuxTaskCopy>
}
localModels: {
title: string
runtimeTitle: string
runtimeReady: (backend: string) => string
serverRunning: string
runtimeInstalled: string
runtimeInstalledDetail: (tag: string, backend: string) => string
installTitle: string
installDetail: string
installAction: string
installing: string
installFailed: string
hardwareTitle: string
hardwareLoading: string
vram: (label: string) => string
ram: (label: string) => string
unifiedMemory: string
modelsTitle: string
recommended: string
/** Recommended-badge tooltip by resolver branch; unknown keys (newer
* backend) simply show no tooltip. */
recommendedReason: Record<string, string>
downloaded: string
downloadAction: (size: string) => string
downloadProgress: (done: string, total: string) => string
downloadDoneToast: (model: string) => string
installDoneToast: string
quickstartTitle: string
quickstartDetail: (model: string, size: string) => string
quickstartDetailReady: (model: string) => string
quickstartAction: string
quickstartConfigure: string
quickstartDoneToast: (model: string) => string
quickstartFailed: string
quickstartStageEngine: string
quickstartStageModel: string
quickstartStageFinish: string
useAction: string
activePill: string
updateTitle: string
updateDetail: (next: string, current: string) => string
updateAction: string
updating: string
upToDateTitle: string
upToDateDetail: (tag: string, backend: string) => string
updateToast: (next: string) => string
activeDetail: string
activeNotLoaded: string
loadedPill: string
placementResident: string
placementSpilled: string
placementResidentTip: string
placementSpilledTip: string
loadingPill: string
ejectTip: string
ejected: string
ejectFailed: string
stopServer: string
startServer: string
runtimeRunningDetail: string
serverStopped: string
serverStarted: string
serverStopFailed: string
serverStartFailed: string
activating: string
activateFailed: (model: string) => string
activateDoneToast: (model: string) => string
downloadFailed: (model: string) => string
pillFitsGpu: string
pillUsesRam: string
pillTooBig: string
browseTitle: string
browseHint: string
browsePlaceholder: string
browseSearching: string
browseListing: string
browseShowFiles: string
browseRefresh: string
browseDownloads: string
browseLikes: string
browseGated: string
browseNoGguf: string
browseFitUnknown: string
browseAlreadyDownloaded: string
addedByYou: string
browseDownloadStarted: string
browseDownloadAria: string
sideloadButton: string
sideloadTitle: string
sideloadDone: string
sideloadAlreadyPresent: string
pillFullContext: (max: string) => string
pillFullContextTip: string
pillUpTo: (max: string) => string
pillGrowsTip: string
pillVision: string
deleteAction: string
deleteConfirm: (model: string) => string
deleted: (model: string) => string
deleteFailed: string
}
providers: {
connectAccount: string
haveApiKey: string
@@ -2350,6 +2452,8 @@ export interface Translations {
connected: string
featuredPitch: string
fireworksPitch: string
localModelsTitle: string
localModelsPitch: string
openRouterPitch: string
apiKeyOptions: Record<string, { short: string; description: string }>
backToSignIn: string
@@ -2400,6 +2504,9 @@ export interface Translations {
noModels: string
addProvider: string
loadFailed: string
loadingIntoMemory: string
downloading: string
localDownloadsHeading: string
noAuthenticatedProviders: string
pro: string
proNeedsSubscription: string
@@ -2526,6 +2633,15 @@ export interface Translations {
openStarmap: string
turnRunning: string
contextUsage: string
systemResources: {
title: string
loading: string
gpuUtilization: string
gpuMemory: string
ram: string
unifiedNote: string
toggle: string
}
contextUsagePanel: {
categories: {
conversation: string
@@ -2777,6 +2893,8 @@ export interface Translations {
loadingSession: string
showEarlier: string
loadingResponse: string
loadingLocalModel: (model: string) => string
processingPrompt: string
resumeWhenBackgroundDone: (count: number) => string
thinking: string
thought: string
@@ -3027,8 +3145,12 @@ export interface Translations {
tips: {
close: string
/** Keyed by `TipId`, so a new tip without copy is a type error. */
items: Record<TipId, { title: string; text: string }>
/** Keyed by `TipId`, so a new tip without copy is a type error. Plus the
* campaign tips, which live outside the rotation's catalog: they carry
* a button, and `action` is its label. */
items: Record<TipId, { title: string; text: string }> & {
'local-setup': { title: string; text: string; action: string }
}
}
errors: {
+114
View File
@@ -269,6 +269,7 @@ export const zhHant = defineLocale({
providerAccounts: '帳號',
providerApiKeys: 'API 金鑰',
providerCustomEndpoints: '自訂端點',
providerLocalModels: '本地模型',
gateway: '閘道',
apiKeys: '工具與金鑰',
keybinds: '鍵盤快捷鍵',
@@ -986,6 +987,101 @@ export const zhHant = defineLocale({
curator: { label: '策展器', hint: '技能使用審查' }
}
},
localModels: {
title: '本地模型',
runtimeTitle: '本地執行環境',
runtimeReady: backend => `就緒 · ${backend}`,
serverRunning: '執行中',
runtimeInstalled: '已安裝 llama.cpp 執行環境',
runtimeInstalledDetail: (tag, backend) => `組建 ${tag},${backend} 後端。Hermes 會為您啟動並管理伺服器。`,
installTitle: '安裝本地執行環境',
installDetail:
'下載 llama.cpp 推理引擎(數百 MB)。下載的模型完全在本機執行——無需帳號,資料不會離開您的電腦。',
installAction: '安裝執行環境',
installing: '正在安裝執行環境…',
installFailed: '執行環境安裝失敗',
hardwareTitle: '本機配置',
hardwareLoading: '正在檢測硬體…',
vram: label => `${label} 顯示記憶體`,
ram: label => `${label} 記憶體`,
unifiedMemory: '統一記憶體',
modelsTitle: '模型',
recommended: '推薦',
recommendedReason: {
'best-quality-resident': '在完全駐留 GPU 且保持全速的模型中品質最高。推薦會在品質與該硬體的預計速度之間權衡。',
'speed-gated-quality': '有更高品質的模型可以裝入這台機器,但受記憶體頻寬限制回應會太慢——這是保持流暢的最佳模型。',
'fastest-resident': '沒有模型能在該硬體上達到全速;這是完全駐留 GPU 記憶體中最快的一個。',
'least-painful-spilled': '沒有模型能完全裝入 GPU 記憶體——這是從系統記憶體執行表現最好的一個。'
} as Record<string, string>,
downloaded: '已下載',
downloadAction: size => `下載 · ${size}`,
downloadProgress: (done, total) => `正在下載 ${done} / ${total}`,
downloadDoneToast: model => `${model} 已就緒。`,
installDoneToast: '本地執行環境已安裝就緒。',
useAction: '使用',
activePill: '預設',
updateTitle: '引擎有可用更新',
updateDetail: (next, current) => `新的 llama.cpp 組建(${next})可以安裝——目前為 ${current}。下載期間模型仍可正常使用。`,
updateAction: '更新引擎',
updating: '正在更新引擎…',
upToDateTitle: '引擎已是最新',
upToDateDetail: (tag, backend) => `正在執行 llama.cpp ${tag}(${backend})——Hermes 提供的最新組建。`,
updateToast: next => `本地引擎有新組建(${next})。可在 設定 → 本地模型 中更新。`,
activeDetail: '新對話使用此模型——傳送首條訊息時載入',
activeNotLoaded: '首條訊息時載入',
loadedPill: '已載入',
placementResident: '全部在 GPU',
placementSpilled: '部分在記憶體',
placementResidentTip: '完全在 GPU 記憶體中以此上下文視窗執行——全速。',
placementSpilledTip: '模型的一部分從系統記憶體執行——可用但較慢。更緊湊的版本或更小的上下文可以完全放入顯示記憶體。',
loadingPill: '載入中…',
ejectTip: '釋放顯示記憶體(需要時重新載入)',
ejected: '模型已卸載——顯示記憶體已釋放。',
ejectFailed: '無法卸載模型',
stopServer: '關閉',
startServer: '開啟',
runtimeRunningDetail: '本地伺服器執行中。關閉後將釋放全部顯示記憶體,新對話將不再使用本地模型,直到您重新開啟。',
serverStopped: '本地伺服器已停止——顯示記憶體已釋放。',
serverStarted: '本地伺服器執行中。',
serverStopFailed: '無法停止本地伺服器',
serverStartFailed: '無法啟動本地伺服器',
activating: '啟動中…',
activateFailed: model => `無法切換到 ${model}`,
activateDoneToast: model => `新對話將使用 ${model}。`,
downloadFailed: model => `${model} 下載失敗`,
pillFitsGpu: '完全在 GPU 上執行',
pillUsesRam: '使用系統記憶體',
pillTooBig: '超出本機記憶體',
browseTitle: '發現更多模型',
browseHint: '搜尋整個 Hugging Face。在這裡下載的模型會自動適配你的機器,但未經我們測試。',
browsePlaceholder: '按名稱或作者搜尋模型…',
browseSearching: '正在搜尋 Hugging Face',
browseListing: '正在讀取模型檔案',
browseShowFiles: '查看檔案',
browseRefresh: '重新整理',
browseDownloads: '次下載',
browseLikes: '個讚',
browseGated: '需要登入 Hugging Face',
browseNoGguf: '未找到相容的模型檔案。',
browseFitUnknown: '適配情況未知',
browseAlreadyDownloaded: '已下載。',
addedByYou: '由你新增',
browseDownloadStarted: '正在下載 {name}',
browseDownloadAria: '下載 {name}',
sideloadButton: '新增模型檔案',
sideloadTitle: '選擇 GGUF 模型檔案',
sideloadDone: '已新增 {name}。',
sideloadAlreadyPresent: '已在你的庫中。',
pillFullContext: max => `完整 ${max} 上下文`,
pillFullContextTip: '從一開始就以模型的完整上下文視窗執行',
pillUpTo: max => `最高 ${max} 上下文`,
pillGrowsTip: '隨著對話需要更多空間自動增長',
pillVision: '識圖',
deleteAction: '刪除模型',
deleteConfirm: model => `從磁碟刪除 ${model}?`,
deleted: model => `已刪除 ${model}。`,
deleteFailed: '刪除失敗'
},
providers: {
connectAccount: '連結帳號',
haveApiKey: '改用 API 金鑰?',
@@ -2323,6 +2419,8 @@ export const zhHant = defineLocale({
connected: '已連線',
featuredPitch: '一個訂閱,300+ 前沿模型 — 執行 Hermes 的建議方式',
fireworksPitch: '直接模型 API — Fireworks 託管的前沿模型',
localModelsTitle: '本地執行模型',
localModelsPitch: '無需帳號——下載模型,在本機執行',
openRouterPitch: '一個金鑰,數百個模型 — 穩定的預設選擇',
apiKeyOptions: {
fireworks: { short: '直接模型 API', description: '直接存取 Fireworks AI 託管的模型。' },
@@ -2387,6 +2485,8 @@ export const zhHant = defineLocale({
noModels: '找不到模型。',
addProvider: '新增提供方',
loadFailed: '無法載入模型',
downloading: '下載中',
localDownloadsHeading: '本地',
noAuthenticatedProviders: '沒有已驗證的提供方。',
pro: 'Pro',
proNeedsSubscription: 'Pro 模型需要付費 Nous 訂閱。',
@@ -2499,6 +2599,15 @@ export const zhHant = defineLocale({
openStarmap: '開啟記憶圖譜',
turnRunning: '執行中',
contextUsage: '上下文使用量',
systemResources: {
title: '系統資源',
loading: '資源…',
gpuUtilization: 'GPU 使用率',
gpuMemory: '顯示記憶體',
ram: '記憶體',
unifiedNote: '統一記憶體——GPU 與系統共享此記憶體池。',
toggle: '系統資源'
},
contextUsagePanel: {
categories: {
conversation: '對話',
@@ -3038,6 +3147,11 @@ export const zhHant = defineLocale({
title: '對話中隨時換模型',
text: '模型名稱就是按鈕。工作性質變了就換一個。'
},
'local-setup': {
title: '這台電腦可以本地執行模型',
text: '你的硬體可以執行本地模型。對話不離開你的電腦,而且完全免費。',
action: '立即設定'
},
'right-pane': {
title: '工作面板',
text: '檔案、終端機、審閱與內建瀏覽器都在側邊面板裡。'
+127
View File
@@ -382,6 +382,7 @@ export const zh: Translations = {
providerAccounts: '账号',
providerApiKeys: 'API 密钥',
providerCustomEndpoints: '自定义端点',
providerLocalModels: '本地模型',
gateway: '网关',
apiKeys: '工具与密钥',
keybinds: '键盘快捷键',
@@ -1311,6 +1312,111 @@ export const zh: Translations = {
curator: { label: '维护器', hint: '技能使用审查' }
}
},
localModels: {
title: '本地模型',
runtimeTitle: '本地运行时',
runtimeReady: backend => `就绪 · ${backend}`,
serverRunning: '运行中',
runtimeInstalled: '已安装 llama.cpp 运行时',
runtimeInstalledDetail: (tag, backend) => `构建 ${tag},${backend} 后端。Hermes 会为您启动并管理服务器。`,
installTitle: '安装本地运行时',
installDetail: '下载 llama.cpp 推理引擎(几百 MB)。下载的模型完全在本机运行——无需账号,数据不会离开您的电脑。',
installAction: '安装运行时',
installing: '正在安装运行时…',
installFailed: '运行时安装失败',
quickstartTitle: '在本机运行模型',
quickstartDetail: (model, size) =>
`一键完成所有设置:本地引擎、${model}(需下载 ${size}),并设为新会话的默认模型。数据不会离开这台电脑。`,
quickstartDetailReady: model => `一键将 ${model} 设为新会话的默认模型。所有内容都在本机运行。`,
quickstartAction: '为我设置',
quickstartConfigure: '自定义…',
quickstartDoneToast: model => `${model} 已就绪——新会话将在本机运行。`,
quickstartFailed: '本地模型设置失败',
quickstartStageEngine: '引擎',
quickstartStageModel: '模型',
quickstartStageFinish: '完成',
hardwareTitle: '本机配置',
hardwareLoading: '正在检测硬件…',
vram: label => `${label} 显存`,
ram: label => `${label} 内存`,
unifiedMemory: '统一内存',
modelsTitle: '模型',
recommended: '推荐',
recommendedReason: {
'best-quality-resident': '在完全驻留 GPU 且保持全速的模型中质量最高。推荐会在质量与该硬件的预计速度之间权衡。',
'speed-gated-quality': '有更高质量的模型可以装入这台机器,但受内存带宽限制响应会太慢——这是保持流畅的最佳模型。',
'fastest-resident': '没有模型能在该硬件上达到全速;这是完全驻留 GPU 内存中最快的一个。',
'least-painful-spilled': '没有模型能完全装入 GPU 内存——这是从系统内存运行表现最好的一个。'
} as Record<string, string>,
downloaded: '已下载',
downloadAction: size => `下载 · ${size}`,
downloadProgress: (done, total) => `正在下载 ${done} / ${total}`,
downloadDoneToast: model => `${model} 已就绪。`,
installDoneToast: '本地运行时已安装就绪。',
useAction: '使用',
activePill: '默认',
updateTitle: '引擎有可用更新',
updateDetail: (next, current) => `新的 llama.cpp 构建(${next})可以安装——当前为 ${current}。下载期间模型仍可正常使用。`,
updateAction: '更新引擎',
updating: '正在更新引擎…',
upToDateTitle: '引擎已是最新',
upToDateDetail: (tag, backend) => `正在运行 llama.cpp ${tag}(${backend})——Hermes 提供的最新构建。`,
updateToast: next => `本地引擎有新构建(${next})。可在 设置 → 本地模型 中更新。`,
activeDetail: '新对话使用此模型——发送首条消息时加载',
activeNotLoaded: '首条消息时加载',
loadedPill: '已加载',
placementResident: '全部在 GPU',
placementSpilled: '部分在内存',
placementResidentTip: '完全在 GPU 显存中以此上下文窗口运行——全速。',
placementSpilledTip: '模型的一部分从系统内存运行——可用但较慢。更紧凑的版本或更小的上下文可以完全放入显存。',
loadingPill: '加载中…',
ejectTip: '释放显存(需要时重新加载)',
ejected: '模型已卸载——显存已释放。',
ejectFailed: '无法卸载模型',
stopServer: '关闭',
startServer: '开启',
runtimeRunningDetail: '本地服务器正在运行。关闭后将释放全部显存,新对话将不再使用本地模型,直到您重新开启。',
serverStopped: '本地服务器已停止——显存已释放。',
serverStarted: '本地服务器运行中。',
serverStopFailed: '无法停止本地服务器',
serverStartFailed: '无法启动本地服务器',
activating: '启动中…',
activateFailed: model => `无法切换到 ${model}`,
activateDoneToast: model => `新对话将使用 ${model}。`,
downloadFailed: model => `${model} 下载失败`,
pillFitsGpu: '完全在 GPU 上运行',
pillUsesRam: '使用系统内存',
pillTooBig: '超出本机内存',
browseTitle: '发现更多模型',
browseHint: '搜索整个 Hugging Face。在这里下载的模型会自动适配你的机器,但未经我们测试。',
browsePlaceholder: '按名称或作者搜索模型…',
browseSearching: '正在搜索 Hugging Face',
browseListing: '正在读取模型文件',
browseShowFiles: '查看文件',
browseRefresh: '刷新',
browseDownloads: '次下载',
browseLikes: '个赞',
browseGated: '需要登录 Hugging Face',
browseNoGguf: '未找到兼容的模型文件。',
browseFitUnknown: '适配情况未知',
browseAlreadyDownloaded: '已下载。',
addedByYou: '由你添加',
browseDownloadStarted: '正在下载 {name}',
browseDownloadAria: '下载 {name}',
sideloadButton: '添加模型文件',
sideloadTitle: '选择 GGUF 模型文件',
sideloadDone: '已添加 {name}。',
sideloadAlreadyPresent: '已在你的库中。',
pillFullContext: max => `完整 ${max} 上下文`,
pillFullContextTip: '从一开始就以模型的完整上下文窗口运行',
pillUpTo: max => `最高 ${max} 上下文`,
pillGrowsTip: '随着对话需要更多空间自动增长',
pillVision: '识图',
deleteAction: '删除模型',
deleteConfirm: model => `从磁盘删除 ${model}?`,
deleted: model => `已删除 ${model}。`,
deleteFailed: '删除失败'
},
providers: {
connectAccount: '连接账号',
haveApiKey: '改用 API 密钥?',
@@ -2933,6 +3039,8 @@ export const zh: Translations = {
connected: '已连接',
featuredPitch: '一个订阅,300+ 前沿模型 — 运行 Hermes 的推荐方式',
fireworksPitch: '直接模型 API — Fireworks 托管的前沿模型',
localModelsTitle: '本地运行模型',
localModelsPitch: '无需账号——下载模型,在本机运行',
openRouterPitch: '一个密钥,数百个模型 — 稳妥的默认选择',
apiKeyOptions: {
fireworks: { short: '直接模型 API', description: '直接访问 Fireworks AI 托管的模型。' },
@@ -2998,6 +3106,9 @@ export const zh: Translations = {
noModels: '未找到模型。',
addProvider: '添加提供方',
loadFailed: '无法加载模型',
loadingIntoMemory: '正在载入内存',
downloading: '下载中',
localDownloadsHeading: '本地',
noAuthenticatedProviders: '没有已认证的提供方。',
pro: 'Pro',
proNeedsSubscription: 'Pro 模型需要付费 Nous 订阅。',
@@ -3124,6 +3235,15 @@ export const zh: Translations = {
openStarmap: '打开记忆图谱',
turnRunning: '运行中',
contextUsage: '上下文用量',
systemResources: {
title: '系统资源',
loading: '资源…',
gpuUtilization: 'GPU 利用率',
gpuMemory: '显存',
ram: '内存',
unifiedNote: '统一内存——GPU 与系统共享此内存池。',
toggle: '系统资源'
},
contextUsagePanel: {
categories: {
conversation: '对话',
@@ -3377,6 +3497,8 @@ export const zh: Translations = {
loadingSession: '正在加载会话',
showEarlier: '显示更早的消息',
loadingResponse: 'Hermes 正在加载回复',
loadingLocalModel: model => `正在将 ${model} 载入内存`,
processingPrompt: '正在处理提示词',
resumeWhenBackgroundDone: count =>
count === 1 ? '后台任务完成后将自动继续' : `${count} 个后台任务完成后将自动继续`,
thinking: '思考中',
@@ -3688,6 +3810,11 @@ export const zh: Translations = {
title: '对话中随时换模型',
text: '模型名称就是按钮。工作性质变了就换一个。'
},
'local-setup': {
title: '这台电脑可以本地运行模型',
text: '你的硬件可以运行本地模型。对话不离开你的电脑,而且完全免费。',
action: '立即设置'
},
'right-pane': {
title: '工作面板',
text: '文件、终端、审阅和内置浏览器都在侧边面板里。'
+2
View File
@@ -40,6 +40,7 @@ import {
IconEar as Ear,
IconEarOff as EarOff,
IconEgg as Egg,
IconPlayerEjectFilled as Eject,
IconExternalLink as ExternalLink,
IconEye as Eye,
IconEyeOff as EyeOff,
@@ -170,6 +171,7 @@ export {
Ear,
EarOff,
Egg,
Eject,
ExternalLink,
Eye,
EyeOff,
@@ -1,6 +1,6 @@
import { describe, expect, it } from 'vitest'
import { currentPickerSelection, displayModelName, formatModelStatusLabel } from './model-status-label'
import { currentPickerSelection, displayModelName, formatModelStatusLabel, modelDisplayParts } from './model-status-label'
import { reasoningEffortLabel } from './reasoning-effort'
describe('model-status-label', () => {
@@ -16,6 +16,18 @@ describe('model-status-label', () => {
expect(displayModelName('anthropic/claude-haiku-4-5-20251001')).toBe('Haiku 4 5')
})
it('renders local GGUF ids as a clean name with a quant tag', () => {
expect(modelDisplayParts('Qwen3.6-27B-UD-Q4_K_XL')).toEqual({ name: 'Qwen3.6 27B', tag: 'Q4' })
expect(modelDisplayParts('Nemotron-3-Nano-30B-A3B-UD-Q4_K_XL')).toEqual({
name: 'Nemotron 3 Nano 30B A3B',
tag: 'Q4'
})
expect(modelDisplayParts('Qwen3-4B-Instruct-2507-UD-Q8_K_XL')).toEqual({ name: 'Qwen3 4B', tag: 'Q8' })
expect(modelDisplayParts('some-model-Q6_K')).toEqual({ name: 'Some Model', tag: 'Q6' })
// Cloud ids keep their existing behavior.
expect(modelDisplayParts('anthropic/claude-opus-4.8-fast').tag).toBe('Fast')
})
it('maps reasoning effort to compact labels', () => {
expect(reasoningEffortLabel('high')).toBe('High')
expect(reasoningEffortLabel('xhigh')).toBe('XHigh')
+19 -5
View File
@@ -75,12 +75,26 @@ export function modelDisplayParts(model: string): { name: string; tag: string }
let base = modelBaseId(model)
let tag = ''
for (const [pattern, label] of VARIANT_TAGS) {
if (pattern.test(base)) {
tag = label
base = base.replace(pattern, '')
// Local GGUF ids carry a quant suffix (`…-UD-Q4_K_XL`, `…-Q8_0`). Render it
// as a quiet tag — "Qwen3.6 27B · Q4" — never as part of the name. Without
// this the composer pill reads raw quant soup ("Qwen3.6 27B UD Q4 K XL").
const quant = base.match(/-(?:UD-)?(Q\d(?:_[A-Z0-9]+)*|IQ\d(?:_[A-Z0-9]+)*|F16|BF16)$/i)
break
if (quant) {
tag = quant[1].split('_')[0].toUpperCase()
base = base.slice(0, -quant[0].length)
// Instruct/chat markers are noise once the quant confirmed a local build.
base = base.replace(/-(?:Instruct|Chat)(?:-\d{4})?$/i, '')
}
if (!tag) {
for (const [pattern, label] of VARIANT_TAGS) {
if (pattern.test(base)) {
tag = label
base = base.replace(pattern, '')
break
}
}
}
@@ -0,0 +1,86 @@
/**
* The local-setup campaign's policy, tested as the pure decisions they are:
* who qualifies (eligibility), when the bubble may return (the clock), and
* that the campaign cannot corrupt the rotation's cursor.
*/
import { describe, expect, it } from 'vitest'
import { TIP_CATALOG } from '@/lib/tips/catalog'
import { LOCAL_SETUP_RESHOW_MS, LOCAL_SETUP_TIP_ID, localSetupDue, localSetupEligible } from '@/lib/tips/local-cta'
import type { LocalCatalogModel, LocalModelsStatus } from '@/types/hermes'
function status(overrides: Partial<LocalModelsStatus> = {}): LocalModelsStatus {
return {
active_model_id: null,
loading: {},
models: [],
models_dir: '',
placement: null,
runtime_installed: false,
server_running: false,
...overrides
} as LocalModelsStatus
}
function fittingModel(): LocalCatalogModel {
return { fits: true, id: 'qwen3.8-27b' } as LocalCatalogModel
}
describe('localSetupEligible', () => {
it('offers setup to a local backend with a fitting model and nothing staged', () => {
expect(localSetupEligible('local', status(), [fittingModel()])).toBe(true)
})
it('never promises local privacy on a remote or cloud backend', () => {
for (const mode of ['remote', 'cloud', 'ssh', null]) {
expect(localSetupEligible(mode, status(), [fittingModel()])).toBe(false)
}
})
it('stays quiet when no catalog model fits the machine', () => {
expect(localSetupEligible('local', status(), [{ fits: false } as LocalCatalogModel])).toBe(false)
expect(localSetupEligible('local', status(), [])).toBe(false)
})
it('retires itself once the machine is set up', () => {
const setUp = status({
models: [{ id: 'qwen3.8-27b' }] as LocalModelsStatus['models'],
runtime_installed: true
})
expect(localSetupEligible('local', setUp, [fittingModel()])).toBe(false)
})
it('still offers when the runtime exists but no model is staged', () => {
expect(localSetupEligible('local', status({ runtime_installed: true }), [fittingModel()])).toBe(true)
})
it('answers false while state is still loading', () => {
expect(localSetupEligible('local', null, [fittingModel()])).toBe(false)
expect(localSetupEligible('local', status(), null)).toBe(false)
})
})
describe('localSetupDue', () => {
it('is due when never shown', () => {
expect(localSetupDue(Date.now(), undefined)).toBe(true)
})
it('holds for a week after an ignored showing, then returns', () => {
const now = Date.now()
expect(localSetupDue(now, now - LOCAL_SETUP_RESHOW_MS + 1000)).toBe(false)
expect(localSetupDue(now, now - LOCAL_SETUP_RESHOW_MS)).toBe(true)
})
})
describe('campaign identity', () => {
it('lives outside the rotation catalog — the walk must never land on it', () => {
// TipId already excludes the campaign id at the type level; this guards
// the runtime data against someone re-adding it as a catalog entry.
const ids: readonly string[] = TIP_CATALOG.map(def => def.id)
expect(ids).not.toContain(LOCAL_SETUP_TIP_ID)
})
})
+55
View File
@@ -0,0 +1,55 @@
/**
* The local-setup call to action — eligibility as data in, data out.
*
* The one campaign tip the app currently runs: a machine that could serve a
* local model, on a backend that has none staged, gets one bubble on the
* model pill saying so, with the button that starts the already-built
* one-click setup. Everything here is a pure decision over fetched state so
* the policy is testable without a DOM or a backend.
*
* Why this is not a rotation tip: the rotation teaches the app the user
* already has, on a weeks-long walk. This tip is conditional (most machines
* either qualify or don't, permanently), actionable (it carries a button),
* and perishable (setting up local models — or ✕ — ends it forever). It
* outranks the walk when live because "your GPU can do this" beats "the
* model name is a button" every time both are true.
*/
import type { LocalCatalogModel, LocalModelsStatus } from '@/types/hermes'
/** Retirement/shown-at ledger id. Not a `TipId` — the rotation never walks it. */
export const LOCAL_SETUP_TIP_ID = 'local-setup'
/** A CTA ignored (timed out) may return, but on a much longer clock than the
* rotation's: it is the same message twice, not a tour moving on. */
export const LOCAL_SETUP_RESHOW_MS = 7 * 24 * 60 * 60_000
/**
* Machines qualify when the catalog has a model that FITS (the backend's
* physics check, same answer the pane's hero uses) and nothing is servable
* yet — no runtime or no staged models. A set-up machine never qualifies, so
* completing setup retires this tip without any bookkeeping.
*
* `connectionMode` must be 'local': on a remote backend (cloud resolves to
* remote) the models would run on the far machine, and a bubble promising
* "stays on your computer" would be promising someone else's computer.
*/
export function localSetupEligible(
connectionMode: null | string,
status: LocalModelsStatus | null,
catalog: readonly LocalCatalogModel[] | null
): boolean {
if (connectionMode !== 'local' || !status || !catalog) {
return false
}
const needsSetup = !status.runtime_installed || status.models.length === 0
return needsSetup && catalog.some(model => model.fits)
}
/** Due = never shown, or shown long enough ago that repeating it reads as a
* reminder rather than a nag. Retirement is the caller's ledger, not ours. */
export function localSetupDue(now: number, shownAt: number | undefined): boolean {
return shownAt === undefined || now - shownAt >= LOCAL_SETUP_RESHOW_MS
}
@@ -0,0 +1,40 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
describe('$localModelsEnabled', () => {
beforeEach(() => {
vi.resetModules()
})
it('reads true when the preload bridge reports the --local launch flag', async () => {
Object.defineProperty(window, 'hermesDesktop', {
configurable: true,
value: { localModelsEnabled: true }
})
const { $localModelsEnabled } = await import('./local-models-flag')
expect($localModelsEnabled.get()).toBe(true)
})
it('defaults to false when the bridge omits the flag (older preload, web)', async () => {
Object.defineProperty(window, 'hermesDesktop', {
configurable: true,
value: {}
})
const { $localModelsEnabled } = await import('./local-models-flag')
expect($localModelsEnabled.get()).toBe(false)
})
it('defaults to false with no bridge at all', async () => {
Object.defineProperty(window, 'hermesDesktop', {
configurable: true,
value: undefined
})
const { $localModelsEnabled } = await import('./local-models-flag')
expect($localModelsEnabled.get()).toBe(false)
})
})
@@ -0,0 +1,16 @@
import { atom } from 'nanostores'
/**
* Launch-flag gate for every local-models surface in the GUI.
*
* Local models ship on main behind `--local` (either `hermes desktop --local`
* or the flag on Hermes.exe itself). The flag is strict: without it the GUI
* shows no local-models surface at all, even on a machine where local models
* are configured and running — the backend routes stay live, only the
* desktop's presentation is gated. Read once from the preload bridge at
* module load; a launch flag can't change mid-session, so nothing rewrites
* it outside tests.
*/
export const $localModelsEnabled = atom<boolean>(
typeof window !== 'undefined' && window.hermesDesktop?.localModelsEnabled === true
)
@@ -0,0 +1,169 @@
import { atom } from 'nanostores'
import { getLocalModelsJobs, getLocalModelsStatus } from '@/hermes'
import { translateNow } from '@/i18n'
import { notify, notifyError } from '@/store/notifications'
import type { LocalRuntimeJob } from '@/types/hermes'
// App-level tracker for local-runtime jobs (runtime installs, model
// downloads). The AUTHORITY is the backend job registry — this store is a
// cache of it (desktop guide: server truth is cached, not owned). Living at
// the store layer, not in the settings pane, is what makes a download
// survive the pane unmounting: anything can start a job, the poller follows
// it to completion, and completion/failure notify app-wide exactly once.
export const $localRuntimeJobs = atom<readonly LocalRuntimeJob[]>([])
const POLL_ACTIVE_MS = 700
let timer: null | number = null
let polling = false
// Jobs we've already toasted for, so a poll race can't double-notify.
const settledNotified = new Set<string>()
function jobsEqual(a: readonly LocalRuntimeJob[], b: readonly LocalRuntimeJob[]) {
if (a.length !== b.length) {
return false
}
return a.every((job, i) => {
const other = b[i]
return (
job.job_id === other.job_id &&
job.status === other.status &&
job.phase === other.phase &&
job.done_bytes === other.done_bytes
)
})
}
function notifySettled(previous: readonly LocalRuntimeJob[], next: readonly LocalRuntimeJob[]) {
const wasRunning = new Set(previous.filter(j => j.status === 'running').map(j => j.job_id))
for (const job of next) {
if (job.status === 'running' || !wasRunning.has(job.job_id) || settledNotified.has(job.job_id)) {
continue
}
settledNotified.add(job.job_id)
if (job.status === 'done') {
notify({
durationMs: 6_000,
kind: 'success',
title: translateNow('settings.localModels.title'),
message:
job.kind === 'model-download'
? translateNow('settings.localModels.downloadDoneToast', job.target)
: job.kind === 'model-activate'
? translateNow('settings.localModels.activateDoneToast', job.target)
: job.kind === 'quickstart'
? translateNow('settings.localModels.quickstartDoneToast', job.target)
: translateNow('settings.localModels.installDoneToast')
})
} else {
notifyError(
new Error(job.error ?? job.detail ?? 'failed'),
job.kind === 'model-download'
? translateNow('settings.localModels.downloadFailed', job.target)
: job.kind === 'model-activate'
? translateNow('settings.localModels.activateFailed', job.target)
: job.kind === 'quickstart'
? translateNow('settings.localModels.quickstartFailed')
: translateNow('settings.localModels.installFailed')
)
}
}
}
async function poll() {
try {
const { jobs } = await getLocalModelsJobs()
const previous = $localRuntimeJobs.get()
if (!jobsEqual(previous, jobs)) {
notifySettled(previous, jobs)
$localRuntimeJobs.set(jobs)
}
} catch {
// Backend unreachable — keep the last snapshot; the next poll retries.
}
const anyRunning = $localRuntimeJobs.get().some(j => j.status === 'running')
if (anyRunning) {
timer = window.setTimeout(() => void poll(), POLL_ACTIVE_MS)
} else {
polling = false
timer = null
}
}
// Idempotent kick: start (or keep) the poll loop while work is in flight.
// Call after starting a job AND on app boot (to rediscover work started
// before a reload).
export function watchLocalRuntimeJobs() {
if (polling) {
return
}
polling = true
if (timer !== null) {
window.clearTimeout(timer)
}
void poll()
}
// Selector: the running download job for a catalog model id, if any.
export function runningDownloadFor(jobs: readonly LocalRuntimeJob[], modelId: string): LocalRuntimeJob | null {
return jobs.find(j => j.kind === 'model-download' && j.status === 'running' && j.model_id === modelId) ?? null
}
// Selector: every model on its way to the library right now — plain
// downloads plus quickstart runs while they are still fetching bytes
// (later quickstart phases mean the model is staged and activating).
// The model picker renders these as disabled progress rows.
const DOWNLOAD_PHASES = new Set(['starting', 'installing-runtime', 'downloading'])
export function runningModelDownloads(jobs: readonly LocalRuntimeJob[]): LocalRuntimeJob[] {
return jobs.filter(
j =>
j.status === 'running' &&
(j.kind === 'model-download' || (j.kind === 'quickstart' && DOWNLOAD_PHASES.has(j.phase)))
)
}
export function runningRuntimeInstall(jobs: readonly LocalRuntimeJob[]): LocalRuntimeJob | null {
return jobs.find(j => j.kind === 'runtime-install' && j.status === 'running') ?? null
}
// One engine-update toast per app session: checked at boot (after the
// gateway is ready), only when the user runs the local engine. The
// download itself is always a button click in Local Models — this is a
// pointer, not an installer.
let updateNotified = false
export async function checkLocalRuntimeUpdate() {
if (updateNotified) {
return
}
try {
const status = await getLocalModelsStatus()
if (status.enabled && status.update_available) {
updateNotified = true
notify({
durationMs: 10_000,
kind: 'info',
title: translateNow('settings.localModels.title'),
message: translateNow('settings.localModels.updateToast', status.configured_tag)
})
}
} catch {
// Backend without the endpoint (older runtime) or transient failure —
// silently skip; the pane still shows the update row when opened.
}
}
@@ -0,0 +1,62 @@
import { describe, expect, it } from 'vitest'
import { parseModelLoadWait, providerWaitText } from './provider-wait'
// The load-notice string is minted by the backend
// (agent/chat_completion_helpers._managed_local_load_notice) and parsed
// here — these tests pin the desktop side of that cross-language contract
// (the backend pins its side in tests/hermes_cli/test_load_progress.py).
describe('providerWaitText', () => {
it('accepts the managed-local load frame', () => {
const frame = '⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 42% (responses start once the model is loaded)'
expect(providerWaitText(frame)).toBe(frame)
})
it('still accepts classic wait frames and rejects spinner noise', () => {
expect(providerWaitText('⏳ waiting on local-model — 30s with no output yet')).not.toBe('')
expect(providerWaitText('◉_◉ cogitating...')).toBe('')
})
})
describe('parseModelLoadWait', () => {
it('extracts model and percent from a load frame', () => {
expect(
parseModelLoadWait('⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 42% (responses start once the model is loaded)')
).toEqual({ kind: 'load', model: 'Qwen3.6-35B-A3B-UD-Q4_K_M', percent: 42 })
})
it('extracts the percent from a prefill frame', () => {
expect(parseModelLoadWait('⚙ processing prompt — 31%')).toEqual({
kind: 'prefill',
model: '',
percent: 31
})
})
it('parses a percentless prefill frame with a null percent (no fake bar)', () => {
expect(parseModelLoadWait('⚙ processing prompt')).toEqual({
kind: 'prefill',
model: '',
percent: null
})
})
it('returns null for every other wait frame', () => {
expect(parseModelLoadWait('⏳ waiting on qwen — 30s with no output yet')).toBeNull()
expect(parseModelLoadWait('⚠ no output from provider for 900s — reconnecting...')).toBeNull()
expect(parseModelLoadWait('')).toBeNull()
})
it('clamps out-of-range percents', () => {
expect(parseModelLoadWait('⏳ loading m into memory — 999%')?.percent).toBe(100)
})
})
describe('providerWaitText accepts prefill frames', () => {
it('passes the ⚙ processing-prompt frame through', () => {
const frame = '⚙ processing prompt — 31%'
expect(providerWaitText(frame)).toBe(frame)
})
})
+38 -1
View File
@@ -45,5 +45,42 @@ export function clearAllProviderWaits(): void {
export function providerWaitText(text: string): string {
const value = text.trim()
return /^(?:⏳|⚠|↻)\s*(?:waiting on|no (?:output|response)|model returned)/i.test(value) ? value : ''
return /^(?:⏳|⚠|↻|⚙)\s*(?:waiting on|loading|processing prompt|no (?:output|response)|model returned)/i.test(value)
? value
: ''
}
/** Parse a managed-local progress frame into bar-renderable parts, or null
* for every other wait frame. Two shapes, both minted by the backend's
* _managed_local_load_notice (the percents are real — per-tensor load
* callback / live prefill counter — so a determinate bar is honest):
* "⏳ loading <model> into memory — 43% …" -> kind: 'load'
* "⚙ processing prompt — 31%" -> kind: 'prefill'
* A percentless prefill frame ("⚙ processing prompt") parses with
* percent: null and renders as label-only, no fake bar. */
export function parseModelLoadWait(
text: string
): null | { kind: 'load' | 'prefill'; model: string; percent: null | number } {
const value = text.trim()
const load = /^⏳\s*loading\s+(.+?)\s+into memory\s+—\s+(\d{1,3})%/i.exec(value)
if (load) {
return {
kind: 'load',
model: load[1],
percent: Math.max(0, Math.min(100, Number(load[2])))
}
}
const prefill = /^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?/i.exec(value)
if (prefill) {
return {
kind: 'prefill',
model: '',
percent: prefill[1] === undefined ? null : Math.max(0, Math.min(100, Number(prefill[1])))
}
}
return null
}
@@ -25,6 +25,7 @@ export const STATUSBAR_HIDDEN_BY_DEFAULT: readonly string[] = [
'cron',
'running-timer',
'session-timer',
'system-resources',
'terminal',
'webhooks'
]
+31 -2
View File
@@ -26,7 +26,7 @@
import { atom } from 'nanostores'
import { Codecs, persistentAtom } from '@/lib/persisted'
import type { TipSide } from '@/lib/tips/catalog'
import { TIP_CATALOG, type TipSide } from '@/lib/tips/catalog'
import { mirrorDisplayToggle } from '@/store/display-toggles'
/** Hours, not minutes. The catalog is ten tips and it should take weeks. */
@@ -34,6 +34,11 @@ const COOLDOWN_MS = 6 * 60 * 60_000
/** A tip as the bubble needs it: resolved copy, resolved anchor. */
export interface ActiveTip {
/** A call to action: one button under the text. What separates a campaign
* tip from the rotation's — the rotation teaches, this one offers to DO
* the thing, and the button is the only path (an ambient bubble must
* never make its whole face clickable). Clicking closes the tip. */
action?: { label: string; onSelect: () => void }
/** Keybind action id whose live combo the bubble prints. */
keybind?: string
side: TipSide
@@ -65,6 +70,23 @@ export const $activeTip = atom<ActiveTip | null>(null)
// model's schema entirely rather than staying on offer and being dropped.
mirrorDisplayToggle('display.in_app_tips', ENABLED_KEY, $tipsEnabled)
/** When each campaign tip (id outside the rotation catalog) last showed.
* Campaign tips re-offer on their own long clock instead of walking on;
* `$retiredTips` still owns the hard ✕. */
export const $tipShownAt = persistentAtom<Record<string, number>>(
'hermes.desktop.tips.shownAt.v1',
{},
Codecs.json(value => {
if (!value || typeof value !== 'object' || Array.isArray(value)) {
return {}
}
return Object.fromEntries(
Object.entries(value).filter((entry): entry is [string, number] => typeof entry[1] === 'number')
)
})
)
export function setTipsEnabled(enabled: boolean): void {
if (!enabled) {
// Including whichever one is up: the switch is answering a bubble on
@@ -85,7 +107,14 @@ export function resetTips(): void {
/** Put a tip on screen, replacing whatever was there. */
export function showTip(tip: ActiveTip): void {
if (tip.tipId) {
$lastTipId.set(tip.tipId)
// The cursor belongs to the rotation's walk. A campaign tip (an id the
// catalog doesn't hold) records when it showed but must not move the
// cursor — nextTip treats an unknown id as "start over at the top".
if (TIP_CATALOG.some(def => def.id === tip.tipId)) {
$lastTipId.set(tip.tipId)
}
$tipShownAt.set({ ...$tipShownAt.get(), [tip.tipId]: Date.now() })
}
// Any tip starts the cooldown, an agent's included: whoever just pointed at
+91
View File
@@ -1237,6 +1237,97 @@ export interface StatusResponse {
version: string
}
// ── Managed local runtime (llama.cpp) ──────────────────────────
export interface LocalModelPlacement {
window?: number
window_label?: string
spilled?: boolean
granted_window?: number
granted_window_label?: string
}
export interface LocalModelLoadProgress {
stage: string
value: number
percent: number
}
export interface LocalModelsStatus {
enabled: boolean
tag: string
configured_tag: string
update_available: boolean
runtime_installed: boolean
runtime_backend: string | null
server_running: boolean
server_base_url: string | null
active_model_id: string | null
loaded_models: Record<string, string>
/** Models loading into memory right now: real per-tensor load percent. */
loading?: Record<string, LocalModelLoadProgress>
placement?: Record<string, LocalModelPlacement>
models: { id: string; size_bytes: number; size_label: string }[]
models_dir: string
}
export interface LocalHardware {
uma: boolean
vram_total_bytes: number
vram_usable_bytes: number
ram_total_bytes: number
ram_available_bytes: number
vram_label: string
gpu_name: string | null
gpu_util_percent: number | null
vram_used_bytes: number | null
}
export interface LocalCatalogModel {
id: string
display_name: string
description: string
size_bytes: number
size_label: string
native_context: number
native_context_label: string
recommended: boolean
/** Why the resolver picked this entry (recommended rows only):
* best-quality-resident | speed-gated-quality | fastest-resident |
* least-painful-spilled. Renders as the Recommended badge's tooltip. */
recommended_reason?: string | null
downloaded: boolean
downloaded_model_id?: string | null
downloaded_quant?: string | null
mtp: boolean
vision?: boolean
fits: boolean
fit_summary: string
fit_detail?: string
model_id?: string
quant?: string
quant_reason?: string
quant_validated?: boolean
variant_count?: number
start_window?: number
start_window_label?: string
spilled?: boolean
}
export interface LocalRuntimeJob {
job_id: string
kind: 'model-activate' | 'model-download' | 'quickstart' | 'runtime-install'
target: string
model_id: string | null
status: 'running' | 'done' | 'error'
phase: string
detail: string
total_bytes: number | null
done_bytes: number
percent?: number
error: string | null
}
export interface ActionResponse {
name: string
ok: boolean
+91 -44
View File
@@ -107,6 +107,34 @@ _EXCLUDED_DIRS = {
".ruff_cache",
}
# Hermes-managed runtime downloads that only exist at the top of a profile
# home: local GGUF models, llama.cpp runtime binaries, and the managed Node
# installation. All of them are re-downloaded on demand (model catalog,
# runtime bootstrap, node installer) and routinely reach tens to hundreds of
# GB, so zipping them turns a backup into an hours-long compress of
# incompressible weights (the "backup stuck at N files" symptom). Matched
# ONLY at the root of HERMES_HOME and at ``profiles/<name>/`` — a deeper
# directory that happens to share one of these names (a skill's ``models/``,
# a user checkout) is user data and stays in the backup.
_EXCLUDED_ROOT_DIRS = {
"models",
"runtimes",
"node",
}
def _in_excluded_root_dir(rel_path: Path) -> bool:
"""True when *rel_path* (relative to HERMES_HOME) is, or sits inside, a
Hermes-managed runtime tree at the top of a profile home."""
parts = rel_path.parts
if not parts:
return False
if parts[0] in _EXCLUDED_ROOT_DIRS:
return True
# Named profiles are profile homes too: profiles/<name>/models etc.
return len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS
# File-name suffixes to skip
_EXCLUDED_SUFFIXES = (
".pyc",
@@ -128,6 +156,16 @@ _EXCLUDED_NAMES = {
"cron.pid",
}
# File-name prefixes to skip. The desktop updater's pre-flight drops
# ``state.db.pre-update-emergency-<timestamp>.bak`` at the HERMES_HOME root
# (apps/desktop/electron/main.ts preflightStateDb) — a backup artifact in
# the same class as ``backups/`` and ``state-snapshots/``, so a full backup
# must not re-ship it. Matched by prefix because the name carries a
# timestamp; a plain ``.bak`` suffix rule would drop user files.
_EXCLUDED_PREFIXES = (
"state.db.pre-update-emergency-",
)
# File names that ``hermes import`` must never overwrite, matched by basename so
# they're caught for the root profile (``gateway_state.json``) and for named
# profiles alike (``profiles/<name>/gateway_state.json``).
@@ -335,6 +373,9 @@ def _should_exclude(rel_path: Path) -> bool:
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
parts = rel_path.parts
if _in_excluded_root_dir(rel_path):
return True
for part in parts:
if part not in _EXCLUDED_DIRS:
continue
@@ -350,6 +391,9 @@ def _should_exclude(rel_path: Path) -> bool:
if name in _EXCLUDED_NAMES:
return True
if name.startswith(_EXCLUDED_PREFIXES):
return True
if name.endswith(_EXCLUDED_SUFFIXES):
return True
@@ -372,6 +416,48 @@ def _should_skip_backup_file(abs_path: Path, rel_path: Path, out_path: Path) ->
return False
def _iter_backup_files(
hermes_root: Path,
out_path: Path,
skipped_dirs: Optional[set] = None,
):
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
The one owner of the backup walk policy: directory pruning (so os.walk
never descends a multi-GB excluded tree), the root-only ``hermes-agent``
carve-out, profile-home-root runtime trees, and the per-file exclusion
rules — shared by the manual ``hermes backup`` path and the automatic
pre-update/pre-migration path so the two can never drift.
``skipped_dirs``, when given, collects pruned directories (root-relative,
as strings) for the end-of-run summary.
"""
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
rel_dir = Path(dirpath).relative_to(hermes_root)
# ``hermes-agent`` is only pruned at the root level; nested dirs
# with the same name (e.g. in skills/) must be preserved. Managed
# runtime trees (models/, runtimes/, node/) are pruned only at a
# profile-home root — see _EXCLUDED_ROOT_DIRS.
is_root = rel_dir == Path(".")
orig_dirnames = dirnames[:]
dirnames[:] = [
d for d in dirnames
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
and not _in_excluded_root_dir(rel_dir / d)
]
if skipped_dirs is not None:
for removed in set(orig_dirnames) - set(dirnames):
skipped_dirs.add(str(rel_dir / removed))
for fname in filenames:
rel = rel_dir / fname
fpath = hermes_root / rel
if _should_skip_backup_file(fpath, rel, out_path):
continue
yield fpath, rel
# ---------------------------------------------------------------------------
# SQLite safe copy
# ---------------------------------------------------------------------------
@@ -869,33 +955,10 @@ def _run_backup_locked(args, hermes_root: Path) -> None:
scan_started = time.monotonic()
logger.info("backup phase=scan status=started")
print(f"Scanning {display_hermes_home()} ...")
files_to_add: list[tuple[Path, Path]] = [] # (absolute, relative)
skipped_dirs = set()
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
dp = Path(dirpath)
rel_dir = dp.relative_to(hermes_root)
# Prune excluded directories in-place so os.walk doesn't descend
# ``hermes-agent`` is only pruned at the root level; nested dirs
# with the same name (e.g. in skills/) must be preserved.
is_root = rel_dir == Path(".")
orig_dirnames = dirnames[:]
dirnames[:] = [
d for d in dirnames
if d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root)
]
for removed in set(orig_dirnames) - set(dirnames):
skipped_dirs.add(str(rel_dir / removed))
for fname in filenames:
fpath = dp / fname
rel = fpath.relative_to(hermes_root)
if _should_skip_backup_file(fpath, rel, out_path):
continue
files_to_add.append((fpath, rel))
skipped_dirs: set = set()
files_to_add: list[tuple[Path, Path]] = list(
_iter_backup_files(hermes_root, out_path, skipped_dirs)
)
# External memory-provider state (e.g. ~/.honcho, ~/.hindsight) lives
# outside HERMES_HOME, so the walk above never sees it. Ask the active
@@ -2328,24 +2391,8 @@ def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional
"""
scan_started = time.monotonic()
logger.info("automatic backup phase=scan status=started")
files_to_add: list[tuple[Path, Path]] = []
try:
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
dp = Path(dirpath)
# Prune excluded directories in-place so os.walk doesn't descend
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
for fname in filenames:
fpath = dp / fname
try:
rel = fpath.relative_to(hermes_root)
except ValueError:
continue
if _should_skip_backup_file(fpath, rel, out_path):
continue
files_to_add.append((fpath, rel))
files_to_add = list(_iter_backup_files(hermes_root, out_path))
except OSError as exc:
logger.warning("Full-zip backup: walk failed: %s", exc)
return None
+1
View File
@@ -3013,6 +3013,7 @@ class CLICommandsMixin:
review_memory=True,
review_skills=review_skills,
focus=focus or None,
explicit=True,
)
except Exception as exc:
_cprint(f" /refine failed to start: {exc}")
+23
View File
@@ -4012,6 +4012,29 @@ DEFAULT_CONFIG = {
"region": "global",
},
# Managed llama.cpp local runtime (see docs: user-guide/local-models).
# Hermes downloads official llama.cpp release binaries, then spawns and
# supervises one llama-server in router mode. Context sizing is policy,
# not preference: there are deliberately no context/VRAM knobs here.
"local_runtime": {
# Master switch for the managed runtime. Off = detection-only
# (Hermes still finds an external llama-server you run yourself).
"enabled": False,
# Pinned llama.cpp release tag (rolling bNNNN). Bumped by Hermes
# releases after the validation suite re-runs, not tracked live.
"tag": "b10679",
# Inference backend: auto = CUDA on NVIDIA, Metal on macOS, Vulkan on
# other GPUs, else CPU. Explicit values: cuda|metal|vulkan|hip|cpu.
"backend": "auto",
# Router process: how many models may be resident at once.
"models_max": 4,
# Port for the managed server. 0 = pick a free port at spawn.
"port": 0,
# Extra ports detection probes for an external llama-server, in
# addition to the default 8080.
"detect_ports": [],
},
# Config schema version - bump this when adding new required fields
"_config_version": 39,
}
+93 -3
View File
@@ -206,6 +206,31 @@ def build_models_payload(
excluded_providers=ctx.excluded_providers or [],
)
# Managed local runtime: staged GGUFs are selectable like any provider's
# models. list_authenticated_providers can't know about them (no
# credential, no custom_providers entry — the credential is
# reachability), so inject the row here where every picker surface
# inherits it. Present whenever models are staged; picking one routes
# through the llamacpp alias -> managed/detected server resolution.
local_row = _local_runtime_row(ctx)
if local_row is not None:
rows = [r for r in rows if str(r.get("slug", "")).lower() != "llamacpp"]
rows.append(local_row)
# A live session on the managed server reports provider "custom"
# (the resolution seam's generic label for a raw base_url), which
# would otherwise materialize a duplicate "Custom endpoint" row
# carrying the same staged models and stealing the checkmark. The
# Local row owns the managed server's identity — drop custom rows
# that point at the managed endpoint.
if local_row.get("is_current"):
def _is_managed_custom(row: dict) -> bool:
if str(row.get("slug", "")).lower() != "custom":
return False
models = {str(m) for m in (row.get("models") or [])}
return bool(models) and models <= set(local_row["models"])
rows = [r for r in rows if not _is_managed_custom(r)]
moa_row = _moa_provider_row(ctx.current_provider)
if moa_row is not None:
rows = [moa_row] + [r for r in rows if str(r.get("slug", "")).lower() != "moa"]
@@ -217,9 +242,15 @@ def build_models_payload(
# has lost its credential, list_authenticated_providers() omits it;
# keep that one row visible so the UI can show the saved selection and
# a re-auth affordance instead of appearing to jump to another provider.
rows = list(rows) + _append_unconfigured_rows(
rows, ctx, current_only=True
)
# Exception: a "custom" current whose endpoint is the managed local
# server is already represented (with the checkmark) by the Local row
# — the skeleton would resurrect the duplicate the dedup above removed.
_local_owns_current = bool(local_row and local_row.get("is_current")
and (ctx.current_provider or "").lower() == "custom")
if not _local_owns_current:
rows = list(rows) + _append_unconfigured_rows(
rows, ctx, current_only=True
)
# --- Deduplicate: remove models from aggregators that overlap with
# user-defined providers. When a local proxy (e.g. litellm-proxy)
@@ -735,6 +766,15 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
if current_slug and slug == current_slug:
kept.append(row)
continue
if row.get("source") == "local-runtime":
# Managed local models are explicit configuration by existence:
# the user downloaded gigabytes into the machine-scoped models
# dir. There is deliberately no config credential to find
# (credential is reachability), so without this clause the row
# only survives on the profile where Use was last clicked —
# every other profile loses local models from its picker.
kept.append(row)
continue
if slug == "moa":
# MoA is a virtual routing mode, not an independently configured
# provider. Hide it from explicit-only pickers unless it is the
@@ -986,6 +1026,56 @@ def _apply_pricing(
row["unavailable_models"] = []
def _local_runtime_row(ctx: "ConfigContext") -> dict | None:
"""Build the ``llamacpp`` provider row from staged local models.
Present whenever GGUFs are staged in the managed models directory —
downloaded models must be selectable even before the server is running
(selection starts it via the runtime_provider seam / activate flow).
Returns ``None`` when nothing is staged.
"""
try:
from hermes_cli.local_runtime.bootstrap import staged_model_ids
staged = staged_model_ids()
if not staged:
return None
current = (ctx.current_provider or "").strip().lower() in (
"llamacpp", "llama.cpp", "llama-cpp")
if not current:
# A LIVE session on the managed server reports provider "custom"
# (the resolution seam's label) with the managed base_url. Match
# on the endpoint so the picker still marks this row current —
# otherwise the session the user is chatting in shows no
# selection.
try:
from hermes_cli.local_runtime.endpoint import _state_endpoint
managed = _state_endpoint()
current = bool(
managed
and (ctx.current_base_url or "").strip().rstrip("/")
== managed["base_url"].rstrip("/"))
except Exception:
current = False
return {
"slug": "llamacpp",
# Bare "Local" everywhere user-facing: the engine name is an
# implementation detail (the pane brands this "Local models").
"name": "Local",
"is_current": current,
"is_user_defined": False,
"models": staged,
"total_models": len(staged),
"source": "local-runtime",
"authenticated": True, # the credential is reachability
"auth_type": "local",
"warning": None,
}
except Exception:
return None
def _moa_provider_row(current_provider: str = "") -> dict | None:
"""Build the virtual ``moa`` provider row for model pickers.
+55
View File
@@ -0,0 +1,55 @@
"""Managed llama.cpp runtime.
Hermes downloads, verifies, supervises, and updates one llama-server, and
decides per machine which model build and context window to run. Key
modules:
- ``binaries`` — resolve/download/verify official llama.cpp release zips
into ``$HERMES_HOME/runtimes/llamacpp/<tag>/``.
- ``supervisor``— spawn and supervise one llama-server in router mode;
readiness is a touch generation, never health-200 alone.
- ``detect`` — find an already-running llama-server (external or ours).
- ``estimator`` / ``context_policy`` / ``growth`` — price context memory
per architecture and run the window ladder (zero-spill start, grow
toward native max, compress only at the top).
- ``catalog`` / ``presets`` — the curated model list and the per-model
launch flags that carry policy decisions to the router.
Everything is driven by the ``local_runtime`` section of config.yaml.
"""
from hermes_cli.local_runtime.binaries import ( # noqa: F401
BinaryResolutionError,
ensure_runtime_installed,
resolve_assets,
select_backend,
)
from hermes_cli.local_runtime.bootstrap import ( # noqa: F401
ensure_local_runtime,
shutdown_local_runtime,
)
from hermes_cli.local_runtime.context_policy import ( # noqa: F401
FLOOR,
growth_decision,
initial_window,
ladder,
launch_args,
)
from hermes_cli.local_runtime.growth import ( # noqa: F401
clear_window_override,
load_window_overrides,
maybe_grow_window,
save_window_override,
)
from hermes_cli.local_runtime.detect import detect_server # noqa: F401
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint # noqa: F401
from hermes_cli.local_runtime.estimator import ( # noqa: F401
HardwareBudget,
ctx_bytes,
physics_check,
profile_from_gguf,
)
from hermes_cli.local_runtime.gguf import read_gguf_header # noqa: F401
from hermes_cli.local_runtime.hardware import probe_budget # noqa: F401
from hermes_cli.local_runtime.presets import generate_presets # noqa: F401
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor # noqa: F401
+351
View File
@@ -0,0 +1,351 @@
"""Binary acquisition for the managed llama.cpp runtime.
llama.cpp publishes per-tag assets (rolling ``bNNNN`` tags, no semver).
Backends are dlopen'd plugins, so a runtime = CPU/base zip + backend zip
extracted into one directory, plus the cudart runtime zip on Windows CUDA
(end users have no CUDA toolkit). We pin the tag in config, sha256-verify
every download, and keep the previous tag for rollback (N-1).
Layout: ``$HERMES_HOME/runtimes/llamacpp/<tag>/<backend>/<binaries>``
with a ``manifest.json`` recording zips, sha256s, and the verified
llama-server version string.
"""
from __future__ import annotations
import hashlib
import json
import logging
import platform
import shutil
import subprocess
import urllib.request
import zipfile
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable
from hermes_constants import get_hermes_home
logger = logging.getLogger(__name__)
RELEASE_URL = "https://github.com/ggml-org/llama.cpp/releases/download/{tag}/{asset}"
# Windows CUDA zips ship per CUDA major; the runtime zip must be paired with
# its cudart zip so end users need no toolkit. 13.3 verified on 13.1 and
# 13.2 drivers.
_WIN_CUDA_VERSION = "13.3"
# arm64 Windows CUDA prebuilts landed upstream (~b1036x) on CUDA 13.4 —
# verified against live asset lists (b10362, b10630, b10679). Tags at or before
# b10290 don't have them; resolution succeeds and the download 404s
# honestly on such tags, which only arises if a user pins backward.
_WIN_CUDA_VERSION_ARM64 = "13.4"
# Fallback when the config section is missing entirely (deep-merge normally
# guarantees the key). Single source: DEFAULT_CONFIG owns the shipped tag.
def default_tag() -> str:
from hermes_cli.config_defaults import DEFAULT_CONFIG
return DEFAULT_CONFIG["local_runtime"]["tag"]
class BinaryResolutionError(RuntimeError):
"""No usable asset combination for this platform/backend."""
@dataclass
class AssetPlan:
"""The exact zips one runtime install needs, in extraction order."""
tag: str
backend: str # cuda | metal | vulkan | hip | cpu
assets: list[str] = field(default_factory=list)
@property
def install_dir(self) -> Path:
return runtimes_root() / self.tag / self.backend
def runtimes_root() -> Path:
"""Machine-scoped, deliberately NOT profile-scoped. Engine binaries,
presets, and server state describe this machine's hardware and its one
managed server (stable port) — a second profile re-downloading the
engine or fighting over the port would be the bug. Profile-scoped
things (which model is the default, enabled) live in each profile's
config.yaml as ever."""
from hermes_constants import get_default_hermes_root
return get_default_hermes_root() / "runtimes" / "llamacpp"
def installed_tags() -> list[str]:
"""Tags with a verified install (manifest carries verified_version),
newest first by release number. The boot ladder and the update check
both read installed-ness from here — one resolver, every caller."""
root = runtimes_root()
if not root.exists():
return []
found: list[str] = []
for entry in root.iterdir():
if not entry.is_dir() or entry.name == "downloads":
continue
for manifest in entry.glob("*/manifest.json"):
try:
if json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
found.append(entry.name)
break
except (json.JSONDecodeError, OSError):
continue
def _release_number(tag: str) -> int:
digits = "".join(ch for ch in tag if ch.isdigit())
return int(digits) if digits else 0
return sorted(set(found), key=_release_number, reverse=True)
def _host_os_arch() -> tuple[str, str]:
"""(os, arch) normalized to release-asset vocabulary.
PITFALL: PROCESSOR_ARCHITECTURE lies under x64 emulation on
ARM64 Windows. platform.machine() reads the same env on some Pythons, so
on Windows prefer PROCESSOR_IDENTIFIER's text when present.
"""
system = platform.system().lower()
os_name = {"windows": "win", "darwin": "macos", "linux": "ubuntu"}.get(system, system)
machine = platform.machine().lower()
arch = "arm64" if machine in ("arm64", "aarch64") else "x64"
if os_name == "win":
import os as _os
ident = _os.environ.get("PROCESSOR_IDENTIFIER", "")
if "armv8" in ident.lower() or "arm " in ident.lower():
arch = "arm64"
return os_name, arch
def select_backend(gpu_vendor: str | None, os_name: str | None = None) -> str:
"""Backend choice per design: CUDA if NVIDIA, Metal on macOS, Vulkan if
a non-NVIDIA GPU is present, else CPU. ``--list-devices`` validates the
choice post-install; the supervisor's touch generation is ground truth."""
if os_name is None:
os_name, _ = _host_os_arch()
if os_name == "macos":
return "metal"
vendor = (gpu_vendor or "").lower()
if "nvidia" in vendor:
return "cuda"
if vendor in ("amd", "intel") or "radeon" in vendor or "arc" in vendor:
return "vulkan"
return "cpu"
def resolve_assets(tag: str, backend: str, os_name: str | None = None,
arch: str | None = None) -> AssetPlan:
"""Compose the asset list for (tag, backend, platform).
Raises BinaryResolutionError for combinations the release does not ship
(a platform/backend pair upstream publishes no artifact for). Callers
fall back down the backend ladder: cuda -> vulkan -> cpu.
"""
host_os, host_arch = _host_os_arch()
os_name = os_name or host_os
arch = arch or host_arch
plan = AssetPlan(tag=tag, backend=backend)
if os_name == "macos":
# macOS tarballs are unified (Metal built in).
plan.assets = [f"llama-{tag}-bin-macos-{arch}.tar.gz"]
return plan
if os_name == "ubuntu":
if backend == "cuda":
# No prebuilt Linux CUDA zips at current tags — Linux CUDA users
# build from source or use vulkan; resolver is honest about it.
raise BinaryResolutionError(
f"no prebuilt linux CUDA asset at {tag}; use vulkan/cpu or a source build")
suffix = {"vulkan": f"vulkan-{arch}", "hip": f"rocm-7.2-{arch}",
"cpu": arch}.get(backend)
if suffix is None:
raise BinaryResolutionError(f"unsupported linux backend {backend}")
plan.assets = [f"llama-{tag}-bin-ubuntu-{suffix}.tar.gz"]
return plan
if os_name == "win":
if backend == "cuda":
cuda_ver = _WIN_CUDA_VERSION_ARM64 if arch == "arm64" else _WIN_CUDA_VERSION
plan.assets = [
f"llama-{tag}-bin-win-cuda-{cuda_ver}-{arch}.zip",
f"cudart-llama-bin-win-cuda-{cuda_ver}-{arch}.zip",
]
elif backend == "vulkan":
if arch == "arm64":
raise BinaryResolutionError(f"no win-vulkan-arm64 asset at {tag}")
plan.assets = [f"llama-{tag}-bin-win-vulkan-x64.zip"]
elif backend == "hip":
plan.assets = [f"llama-{tag}-bin-win-hip-radeon-x64.zip"]
elif backend == "cpu":
plan.assets = [f"llama-{tag}-bin-win-cpu-{arch}.zip"]
else:
raise BinaryResolutionError(f"unsupported windows backend {backend}")
return plan
raise BinaryResolutionError(f"unsupported platform {os_name}-{arch}")
def _sha256(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(1 << 22), b""):
h.update(chunk)
return h.hexdigest()
def _download(url: str, dest: Path,
progress: "Callable[[int, int], None] | None" = None) -> None:
"""Stream url -> dest. ``progress(done_bytes, total_bytes)`` ticks per
chunk (total 0 when the server sends no Content-Length) — a several-
hundred-MB archive on a slow line must never look hung."""
logger.info("downloading %s", url)
tmp = dest.with_suffix(dest.suffix + ".part")
with urllib.request.urlopen(url, timeout=120) as r, open(tmp, "wb") as f:
total = int(r.headers.get("Content-Length") or 0)
done = 0
while True:
chunk = r.read(1 << 20)
if not chunk:
break
f.write(chunk)
done += len(chunk)
if progress is not None:
progress(done, total)
tmp.replace(dest)
def _extract(archive: Path, dest: Path,
progress: "Callable[[int, int], None] | None" = None) -> None:
"""Extract member by member so ``progress(done, total)`` can tick in
uncompressed bytes — big archives take real time on laptop disks."""
if archive.name.endswith(".zip"):
with zipfile.ZipFile(archive) as z:
members = z.infolist()
total = sum(m.file_size for m in members)
done = 0
for m in members:
z.extract(m, dest)
done += m.file_size
if progress is not None:
progress(done, total)
else:
import tarfile
with tarfile.open(archive) as t:
members = t.getmembers()
total = sum(m.size for m in members)
done = 0
for m in members:
t.extract(m, dest, filter="data")
done += m.size
if progress is not None:
progress(done, total)
def server_binary(install_dir: Path) -> Path:
"""Locate llama-server within an extracted runtime (zips differ in
whether they nest a build/bin directory)."""
names = ("llama-server.exe", "llama-server")
for name in names:
direct = install_dir / name
if direct.exists():
return direct
for name in names:
hits = sorted(install_dir.rglob(name))
if hits:
return hits[0]
raise BinaryResolutionError(f"llama-server not found under {install_dir}")
def verify_install(install_dir: Path, tag: str) -> str:
"""Run --version; require the tag's build number in the output.
(The binary prints the tag WITHOUT the 'b' prefix.)"""
exe = server_binary(install_dir)
out = subprocess.run([str(exe), "--version"], capture_output=True,
text=True, encoding="utf-8", errors="replace",
timeout=60, cwd=str(exe.parent))
text = (out.stdout + out.stderr).strip()
if tag.lstrip("b") not in text:
raise BinaryResolutionError(
f"version check failed for {exe}: expected {tag}, got: {text[:120]}")
return text.splitlines()[0] if text else ""
def prune_old_tags(keep: list[str]) -> None:
"""Retain only the tags in ``keep`` (current + previous — N-1 rollback).
The shared ``downloads/`` archive cache is not a tag and always survives."""
root = runtimes_root()
if not root.exists():
return
for entry in root.iterdir():
if entry.is_dir() and entry.name != "downloads" and entry.name not in keep:
shutil.rmtree(entry, ignore_errors=True)
logger.info("pruned old runtime %s", entry.name)
def ensure_runtime_installed(tag: str, backend: str,
expected_sha256: dict[str, str] | None = None,
progress: "Callable[[str, int, int, str], None] | None" = None) -> Path:
"""Idempotent: resolve, download, verify, extract, version-check.
``expected_sha256`` maps asset name -> hash when the catalog pins them;
without pins the computed hash is recorded in the manifest (trust on
first download, verified on every reinstall).
``progress(stage, done_bytes, total_bytes, label)`` ticks through the
slow parts — stage is "download" | "extract" | "verify", label is the
asset counter ("1/2") when the plan has several archives.
Returns the install directory containing llama-server.
"""
plan = resolve_assets(tag, backend)
install_dir = plan.install_dir
manifest_path = install_dir / "manifest.json"
if manifest_path.exists():
try:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
if manifest.get("verified_version"):
return install_dir
except (json.JSONDecodeError, OSError):
pass # damaged manifest -> reinstall
install_dir.mkdir(parents=True, exist_ok=True)
downloads = runtimes_root() / "downloads"
downloads.mkdir(parents=True, exist_ok=True)
recorded: dict[str, str] = {}
n_assets = len(plan.assets)
for i, asset in enumerate(plan.assets, 1):
label = f"{i}/{n_assets}" if n_assets > 1 else ""
archive = downloads / asset
if not archive.exists():
_download(RELEASE_URL.format(tag=tag, asset=asset), archive,
progress=(lambda d, t, _l=label: progress("download", d, t, _l))
if progress is not None else None)
if progress is not None:
progress("verify", 0, 0, label)
digest = _sha256(archive)
expected = (expected_sha256 or {}).get(asset)
if expected and digest != expected:
archive.unlink(missing_ok=True)
raise BinaryResolutionError(
f"sha256 mismatch for {asset}: expected {expected}, got {digest}")
recorded[asset] = digest
_extract(archive, install_dir,
progress=(lambda d, t, _l=label: progress("extract", d, t, _l))
if progress is not None else None)
if progress is not None:
progress("verify", 0, 0, "")
version = verify_install(install_dir, tag)
manifest_path.write_text(json.dumps({
"tag": tag, "backend": plan.backend, "assets": recorded,
"verified_version": version,
}, indent=2), encoding="utf-8")
logger.info("installed llama.cpp %s (%s): %s", tag, backend, version)
return install_dir
+337
View File
@@ -0,0 +1,337 @@
"""Bootstrap for the managed runtime: config -> installed binaries ->
running supervised server.
One public call, ``ensure_local_runtime(config)``, safe to call at any
session start:
- disabled or already-running (state file answers /health) -> no-op
- enabled -> install binaries if missing (idempotent), spawn supervisor
Kept import-light: callers gate on config before importing this module so
sessions with local_runtime disabled never pay the import.
"""
from __future__ import annotations
import logging
import os
import subprocess
import time
from pathlib import Path
from hermes_constants import get_hermes_home # noqa: F401 — config paths
from hermes_cli.local_runtime.binaries import runtimes_root
logger = logging.getLogger(__name__)
_SUPERVISOR = None # process-wide singleton; one router per Hermes process
def _detect_gpu_vendor() -> str | None:
"""Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi
(resolved by the hardware probe's PATH-independent ladder — a stripped
service PATH must not demote an NVIDIA box to vulkan/cpu); anything
else defers to select_backend's fallback ladder."""
from hermes_cli.local_runtime.hardware import _nvidia_smi_path
smi = _nvidia_smi_path()
if smi is None:
return None
try:
out = subprocess.run(
[smi, "--query-gpu=name", "--format=csv,noheader"],
capture_output=True, text=True, timeout=10)
if out.returncode == 0 and out.stdout.strip():
return "nvidia " + out.stdout.strip().splitlines()[0]
except (OSError, subprocess.TimeoutExpired):
pass
return None
def models_dir() -> Path:
"""Machine-scoped, deliberately NOT profile-scoped: a 20 GB GGUF is a
machine asset, and every profile shares the one managed server that
serves it. See runtimes_root() for the same rule on the engine."""
from hermes_constants import get_default_hermes_root
return get_default_hermes_root() / "models"
def assets_dir() -> Path:
"""Non-model companion files (mmproj vision projectors, spec-decode
draft models). A subdirectory so the router's model listing — and our
staged_models() — never mistakes an asset for a servable model."""
return models_dir() / "assets"
def staged_models() -> "list[Path]":
"""Servable staged models: single-file GGUFs count when present; a
split GGUF counts once, by its first part, and only when EVERY part
is on disk — a mid-download split is not servable and must not
surface anywhere as a model. Continuation parts and assets/ never
count."""
import re
part = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$")
files = sorted(models_dir().glob("*.gguf"))
names = {p.name for p in files}
out = []
for p in files:
m = part.search(p.name)
if m is None:
out.append(p)
continue
if m.group(1) != "00001":
continue
stem = p.name[: m.start()]
total = int(m.group(2))
if all(f"{stem}-{i:05d}-of-{m.group(2)}.gguf" in names
for i in range(2, total + 1)):
out.append(p)
return out
def staged_model_ids() -> "list[str]":
import re
return [re.sub(r"-\d{5}-of-\d{5}$", "", p.stem) for p in staged_models()]
def _presets_stale() -> bool:
"""True when a staged model has no section in the preset INI — it
would autoload with stock fit instead of a policy decision."""
try:
from hermes_cli.local_runtime.presets import read_preset_decisions
known = set(read_preset_decisions())
return any(mid not in known for mid in staged_model_ids())
except Exception: # noqa: BLE001
return False
def _stop_state_server(state: dict) -> None:
"""Best-effort stop of the server the state file points at (an
incumbent this process doesn't supervise). The state pid is ours by
contract — the file only ever describes the managed server."""
from hermes_cli.local_runtime.endpoint import _pid_alive
pid = state.get("pid")
try:
pid = int(pid)
except (TypeError, ValueError):
return
if pid <= 0:
return
try:
import signal
os.kill(pid, signal.SIGTERM)
except (OSError, ValueError):
return
# Give it a moment to release the port and the GPU. Liveness via
# psutil — on Windows os.kill(pid, 0) TERMINATES the process, it is
# not a probe (the endpoint.py pitfall note; #local-models review).
for _ in range(50):
if not _pid_alive(pid):
return
time.sleep(0.1)
def refresh_local_runtime() -> bool:
"""Restart the managed server so it rescans the models directory.
The router's model list is SPAWN-ONLY: a GGUF added after start is
invisible to GET /models and 400s on completion, so anything that
changes the staged set while the server runs must bounce it. Covers
both ownership shapes: a supervised server restarts in-process; an
ADOPTED server (started by a previous backend session — the normal
shape after any restart) is stopped via its state-file pid and
replaced with a supervised boot. Without the adopted branch, every
download/delete in a post-restart session silently no-ops the bounce
and the router serves a stale catalog. Returns False when there is
nothing to refresh (no server anywhere; next boot scans fresh).
"""
global _SUPERVISOR
try:
from hermes_cli.config import load_config
if _SUPERVISOR is None:
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is None:
return False
logger.info("bouncing adopted llama-server (pid=%s) to rescan models",
state.get("pid"))
_stop_state_server(state)
else:
shutdown_local_runtime()
return ensure_local_runtime(load_config(), force=True) is not None
except Exception as exc: # noqa: BLE001
logger.warning("local runtime refresh failed: %s", exc)
return False
def ensure_local_runtime(config: dict, force: bool = False) -> "object | None":
"""Idempotent boot of the managed runtime. Returns the supervisor (or
None when disabled/unavailable). Never raises into a session start —
failures log and return None; chat falls back to configured providers.
``force=True`` skips the enabled gate — used by the explicit "Use this
model" action, where the click IS the opt-in (the caller records it in
config so future boots auto-start).
"""
global _SUPERVISOR
section = (config or {}).get("local_runtime") or {}
if not force and not section.get("enabled"):
return None
if _SUPERVISOR is not None:
return _SUPERVISOR
# Residency: no staged models means nothing to serve — don't boot an
# empty server. The walked-away story handled with zero configuration
# (delete your last model and boots stop); Use force-boots as ever.
if not force and not staged_models():
logger.info("local runtime enabled but no models staged; not booting")
return None
# Another Hermes process may already be supervising — reuse via state,
# but ONLY while its launch policy still covers every staged model. A
# server whose preset file predates a download serves the new model
# with no policy at all (--models-autoload + stock fit: f16 KV at max
# context, no placement — the silent-demotion busy-wait on WDDM). A
# stale incumbent gets stopped and replaced by a fresh boot with
# regenerated presets; sessions ride through exactly like any other
# supervised restart (stable port + persisted key).
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is not None:
if not _presets_stale():
logger.info("managed llama-server already running (another process)")
return None
logger.info("running server's presets predate the staged models; "
"replacing it so every model launches with a policy")
_stop_state_server(state)
try:
from hermes_cli.local_runtime.binaries import (
ensure_runtime_installed,
select_backend,
)
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import generate_presets
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
backend = section.get("backend", "auto")
if backend == "auto":
backend = select_backend(_detect_gpu_vendor())
# Boot ladder: serve what is INSTALLED, never download here. The
# configured tag (config root-of-trust; deep-merge supplies the
# Hermes-release default when unpinned) is preferred; when it isn't
# installed yet, the newest installed tag serves and the status
# endpoint reports the pending update — the download is a deliberate
# button click in the pane, not a boot-path surprise (a multi-minute
# inline download here is exactly how the onboarding bounce returns).
from hermes_cli.local_runtime.binaries import default_tag, installed_tags
tag = section.get("tag") or default_tag()
have = installed_tags()
if tag not in have:
if not have:
logger.info("local runtime enabled but no build installed; "
"install happens in the Local Models pane")
return None
logger.info("configured tag %s not installed; serving %s "
"(update is a click in Local Models)", tag, have[0])
tag = have[0]
install_dir = ensure_runtime_installed(tag, backend)
mdir = models_dir()
mdir.mkdir(parents=True, exist_ok=True)
# Context policy: one launch decision per staged model, carried to
# the router via the preset INI. Priced against CAPACITY, not live
# free VRAM: this runs while the outgoing server instance may still
# hold the card (restart, refresh after a download), and its memory
# is freed before the new instance loads anything. Pricing against
# live-free here once pinned a fitting model's weights to CPU
# because the probe saw the predecessor's VRAM as gone.
preset_path = runtimes_root() / "presets.ini"
try:
entries = generate_presets(mdir, probe_budget(planning=True), preset_path)
for entry in entries:
if entry.refusal:
logger.warning("model refused by physics check: %s", entry.refusal)
except Exception as exc: # noqa: BLE001 — policy failure must not block serving
# Degradation ladder: a STALE policy still beats no policy —
# stock fit (f16 KV at max context, no placement) is the
# silent-busy-wait failure on Windows. Keep serving with the
# previous INI when one exists; only a first boot with no INI
# at all falls to stock fit.
if preset_path.exists():
logger.error("preset generation failed (%s); serving with the "
"PREVIOUS launch policies — models staged since "
"the last successful generation run unpoliced "
"until this is fixed", exc)
else:
logger.error("preset generation failed (%s) and no previous "
"policy file exists; router runs stock fit", exc)
preset_path = None
sup = LlamaServerSupervisor(
install_dir, mdir,
models_max=int(section.get("models_max", 4)),
port=int(section.get("port", 0)) or None,
preset_path=preset_path,
)
try:
sup.start()
except Exception:
# start() can fail after the router process exists (health
# timeout, spawn error): leaving it running unsupervised
# strands its VRAM behind a port nothing will clean up.
try:
sup.stop()
except Exception: # noqa: BLE001 — cleanup is best-effort
pass
raise
_SUPERVISOR = sup
logger.info("managed llama-server up at %s (backend=%s tag=%s)",
sup.base_url, backend, tag)
_start_idle_sweeper(sup)
return sup
except Exception as exc: # noqa: BLE001 — never break session start
logger.warning("managed local runtime unavailable: %s", exc)
return None
def shutdown_local_runtime() -> None:
global _SUPERVISOR
if _SUPERVISOR is not None:
_SUPERVISOR.stop()
_SUPERVISOR = None
def get_supervisor():
"""The process-local supervisor, or None (server may still be running
under another process — check the state file)."""
return _SUPERVISOR
def _start_idle_sweeper(sup) -> None:
"""Idle-residency loop: every couple of minutes, unload non-primary
models idle past the supervisor's threshold. Daemon thread tied to the
supervisor's lifetime — exits when the server stops."""
import threading
def _loop():
while sup.proc is not None and sup.proc.poll() is None:
time.sleep(120)
try:
sup.sweep_idle()
except Exception as exc: # noqa: BLE001
logger.debug("idle sweep skipped: %s", exc)
threading.Thread(target=_loop, daemon=True,
name="local-runtime-idle-sweep").start()
+118
View File
@@ -0,0 +1,118 @@
"""Capability answers for models served by the managed runtime.
Capability lookups (vision, and whatever comes next) consult cloud-shaped
catalogs that have never heard of a local GGUF, so a vision-capable local
model reads as text-only and images detour to an auxiliary cloud model —
the wrong behavior twice over for a local-first user (broken feature, and
a screenshot silently leaving the machine).
The managed runtime can answer from ground truth instead, best source
first:
1. The RUNNING child's /props: llama-server reports a ``modalities`` block
when a vision projector is loaded. The server that will receive the
image says whether it can see — no inference, no catalog.
2. The catalog entry's declared capability (the ``vision`` tag + mmproj
asset) for staged-but-unloaded models: what the model WILL support once
its projector loads beside it.
3. None — not one of ours, or nothing known; the caller falls through to
its other sources.
"""
from __future__ import annotations
import json
import logging
import urllib.request
logger = logging.getLogger(__name__)
_LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
# Image formats the managed server's decoder actually handles. llama.cpp
# decodes with stb_image: PNG/JPEG/GIF/BMP yes, WebP NO — and a WebP part
# fails SILENTLY (no HTTP error, no log line; the model just never sees an
# image and confabulates a description). Anything outside this set must be
# transcoded before the request. Measured against the live server: the
# same red square answered 'Red' as PNG and 'Unseen' as WebP.
ACCEPTED_IMAGE_MIMES = frozenset({"image/png", "image/jpeg"})
def is_managed_provider(provider: str, base_url: str = "") -> bool:
"""True when this provider/base_url pair points at the managed server.
``custom`` only counts when the base_url IS the managed endpoint —
background lookups must never claim someone else's custom server."""
p = (provider or "").strip().lower()
if p in _LLAMACPP_ALIASES:
return True
if p == "custom" and base_url:
try:
from hermes_cli.local_runtime.growth import is_managed_endpoint
return is_managed_endpoint(base_url)
except Exception: # noqa: BLE001
return False
return False
def _props_modalities(model_id: str) -> "bool | None":
"""Ask the running server whether this loaded child sees images.
None when the server is down, the model isn't loaded, or the build
doesn't report modalities."""
try:
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is None:
return None
base = state["base_url"].rsplit("/v1", 1)[0]
req = urllib.request.Request(
f"{base}/props?model={model_id}",
headers={"Authorization": f"Bearer {state.get('api_key', '')}"})
with urllib.request.urlopen(req, timeout=3) as r:
props = json.load(r)
modalities = props.get("modalities")
if isinstance(modalities, dict) and "vision" in modalities:
return bool(modalities["vision"])
return None
except Exception: # noqa: BLE001
return None
def managed_model_supports_vision(model_id: str) -> "bool | None":
"""Ground-truth vision capability for a staged model, or None when the
model isn't ours / nothing is known (caller keeps falling through)."""
if not model_id:
return None
# Only answer for models actually staged with us.
try:
from hermes_cli.local_runtime.bootstrap import staged_model_ids
if model_id not in staged_model_ids():
return None
except Exception: # noqa: BLE001
return None
live = _props_modalities(model_id)
if live is not None:
return live
# Staged but not loaded (or an older server build): the catalog knows
# whether this model ships a vision projector.
try:
from hermes_cli.local_runtime.bootstrap import assets_dir
from hermes_cli.local_runtime.catalog import find_entry_for_model
hit = find_entry_for_model(model_id)
if hit is None:
return None
entry = hit[0]
if entry.mmproj is None:
return False
# Capability requires the projector to actually be on disk — a
# model downloaded before its mmproj (partial delete, old layout)
# genuinely cannot see.
return (assets_dir() / entry.mmproj.local_name).exists()
except Exception: # noqa: BLE001
return None
+174
View File
@@ -0,0 +1,174 @@
{
"schema_version": 1,
"models": [
{
"id": "qwen3.8-27b",
"display_name": "Qwen3.8 27B",
"description": "Best all-round agent model; sees images; long context stays fast",
"repo": "unsloth/Qwen3.8-27B-GGUF",
"variants": [
{
"quant": "UD-Q4_K_M",
"files": [
{
"path": "Qwen3.8-27B-UD-Q4_K_M.gguf",
"size_bytes": 16464440224
}
]
}
],
"n_ctx_train": 262144,
"full_layers": 16,
"recurrent_layers": 48,
"per_layer_f16": 4096,
"n_vocab": 248320,
"mmproj": {
"path": "mmproj-BF16.gguf",
"size_bytes": 931146432,
"local": "mmproj-Qwen3.8-27B-BF16.gguf"
},
"mtp": true,
"mtp_draft_depth": 2,
"sampling": {
"temp": "1.0",
"top-p": "0.95",
"top-k": "20",
"min-p": "0.0"
},
"quality": 90,
"decode_fraction": 1.0
},
{
"id": "qwen3.8-flash-next",
"display_name": "Qwen3.8 Flash Next",
"description": "Frontier-scale model; needs a very large GPU to run well",
"repo": "unsloth/Qwen3.8-Flash-Next-GGUF",
"variants": [
{
"quant": "UD-Q4_K_XL",
"files": [
{
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf",
"size_bytes": 10946624
},
{
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf",
"size_bytes": 49859583136
},
{
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf",
"size_bytes": 49376141504
},
{
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf",
"size_bytes": 12087983520
}
]
}
],
"n_ctx_train": 262144,
"full_layers": 12,
"recurrent_layers": 36,
"per_layer_f16": 2048,
"moe": true,
"n_vocab": 248320,
"mmproj": {
"path": "mmproj-BF16.gguf",
"size_bytes": 907542944,
"local": "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
},
"min_engine": "b10678",
"quality": 95,
"decode_fraction": 0.08
},
{
"id": "qwen3.6-35b-a3b",
"display_name": "Qwen3.6 35B-A3B",
"description": "Bigger mixture-of-experts with multi-token prediction; sees images",
"repo": "unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
"variants": [
{
"quant": "UD-Q4_K_M",
"files": [
{
"path": "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf",
"size_bytes": 22663387424
}
],
"validated": true
}
],
"n_ctx_train": 262144,
"full_layers": 10,
"recurrent_layers": 30,
"per_layer_f16": 2048,
"moe": true,
"mtp": true,
"n_vocab": 248320,
"mtp_draft_depth": 2,
"mmproj": {
"path": "mmproj-BF16.gguf",
"size_bytes": 902822528,
"local": "mmproj-Qwen3.6-35B-A3B-BF16.gguf"
},
"sampling": {
"temp": "1.0",
"top-p": "0.95",
"top-k": "20",
"min-p": "0.0"
},
"quality": 80,
"decode_fraction": 0.15
},
{
"id": "deepseek-v4-flash",
"display_name": "DeepSeek V4 Flash",
"description": "Frontier-class model for machines with 128GB+ memory",
"repo": "unsloth/DeepSeek-V4-Flash-0731-GGUF",
"variants": [
{
"quant": "UD-Q4_K_XL",
"files": [
{
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00001-of-00005.gguf",
"size_bytes": 5257408
},
{
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00002-of-00005.gguf",
"size_bytes": 48935523072
},
{
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00003-of-00005.gguf",
"size_bytes": 48980787136
},
{
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00004-of-00005.gguf",
"size_bytes": 49999168416
},
{
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00005-of-00005.gguf",
"size_bytes": 7174505088
}
]
}
],
"n_ctx_train": 1048576,
"full_layers": 43,
"recurrent_layers": 0,
"per_layer_f16": 1152,
"moe": true,
"n_vocab": 163840,
"draft": {
"path": "dspark-DeepSeek-V4-Flash-0731-Q8_0.gguf",
"size_bytes": 10896057440
},
"sampling": {
"temp": "1.0",
"top-p": "0.95",
"min-p": "0.01"
},
"quality": 85,
"decode_fraction": 0.1
}
]
}
+464
View File
@@ -0,0 +1,464 @@
"""Curated starter catalog for the managed local runtime.
Small and honest: every entry carries the estimator inputs (measured on
real GGUFs) so the picker can price a model BEFORE the user downloads
gigabytes. Once a file is on disk, profile_from_gguf() is the authority
and the catalog numbers are only used for the download decision. Entries
whose base config is gated upstream carry a same-family conservative
prior (commented) — the GGUF header corrects it at load time.
Each model ships ONE build, Q4-class (UD-Q4_K_M where the repo has it,
UD-Q4_K_XL elsewhere). Q4 is the quant class current engines optimize
for and the sweet spot of the size/quality curve, so there is no quant
ladder: headroom buys a bigger context window, never a bigger quant,
and every machine runs the same well-tested build. Below Q4 the quality
loss is too severe to ship as someone's first local-AI experience; the
fit policy prices the build honestly (zero-spill, spilled, or refused by
the physics check).
Validation lifecycle: builds proven end-to-end on real hardware are
marked validated. Day-0 entries ship before that proof (they simply lack
the validated flag) — ensure_model_ready's touch generation still gates
every first load at runtime.
Multi-file models: variants may carry split-GGUF parts (llama-server loads
from the first part; all parts download together). Entries may carry an
mmproj (vision projector) and a speculative-decode draft model — both
download alongside the weights. MTP-integrated models run spec decode
wherever they load; a separate draft model attaches only when the launch
decision spills, where its speedup is largest.
File sizes come from HF LFS metadata and feed the estimator, the fit
pills, and download progress. There is no download-time integrity check
by design: a corrupt or truncated file surfaces as a llama.cpp
load error at first use, and the reachability test catches upstream
re-uploads by size drift before users do.
This is deliberately not a live registry feed: entries are reviewed like a
version bump (the same policy governs vendor recipe ingestion — parsed
data, never executed commands).
Vendor recipes overlay: a per-SKU recipes repo may SUPPLEMENT these
entries where applicable — vendor SKUs only, never the base layer for
other platforms. A recipe may enrich identity (GGUF/quant/sha), perf
hints (-b/-ub, spec-decode), and sampling defaults; it never carries
context/slots/placement/serving flags (the fit policy owns those).
Resolution: exact SKU -> GPU-class bucket -> fit-only. Snapshot-synced,
reviewed like a tag bump.
"""
from __future__ import annotations
import json
import logging
import re
import threading
import time
import urllib.request
from dataclasses import dataclass, field
from pathlib import PurePosixPath
from hermes_cli.local_runtime.context_policy import (
FLOOR,
RUNTIME_OVERHEAD_BYTES,
TARGET_WINDOW,
ub_logits_bytes,
)
from hermes_cli.local_runtime.estimator import (
HardwareBudget,
LayerKind,
ModelProfile,
ctx_bytes,
)
logger = logging.getLogger(__name__)
_GIB = 1 << 30
_PART_SUFFIX = re.compile(r"-\d{5}-of-\d{5}$")
@dataclass(frozen=True)
class AssetFile:
"""One downloadable file: repo-relative path and exact bytes (the size
feeds the estimator and the download progress bar; there is no
download-time integrity check by design — a corrupt file surfaces as a
llama.cpp load error). ``local`` overrides the on-disk name (repos
reuse generic names like mmproj-BF16.gguf across models). Non-model
extras live under the models dir's assets/ subdirectory so the router
never lists them."""
path: str # repo-relative (may include a subdir)
size_bytes: int
local: str | None = None
@property
def local_name(self) -> str:
return self.local or PurePosixPath(self.path).name
@dataclass(frozen=True)
class QuantVariant:
"""One downloadable build of a model. Split GGUFs list every part in
files; the model loads from the first part."""
quant: str # e.g. "UD-Q4_K_M"
files: tuple # AssetFile, first = the load target
validated: bool = False # proven end-to-end on real hardware
@property
def model_id(self) -> str:
stem = PurePosixPath(self.files[0].path).name.removesuffix(".gguf")
return _PART_SUFFIX.sub("", stem)
@property
def size_bytes(self) -> int:
return sum(f.size_bytes for f in self.files)
@property
def weights_bytes(self) -> int:
"""Pre-download weights estimate: GGUF bytes ≈ tensor bytes + a
small header (<2%) — a safe, slightly conservative stand-in until
profile_from_gguf reads the real table."""
return self.size_bytes
@dataclass(frozen=True)
class CatalogEntry:
id: str # stable family id (variant-independent)
display_name: str
description: str # one line, plain language
repo: str # HF repo
variants: tuple # QuantVariant (exactly one, Q4-class)
# Estimator inputs (measured or config-derived; quant changes weights,
# never KV). Entries with gated upstream configs carry a conservative
# same-family prior — the GGUF header is the authority after download.
n_ctx_train: int
full_layers: int
recurrent_layers: int
per_layer_f16: int # KV bytes/token per full-attention layer
swa_layers: int = 0
swa_window: int = 0
moe: bool = False
mtp: bool = False # ships MTP heads (spec decode when loaded)
# Speculative draft depth for MTP models. Per-model and measured:
# deeper drafting pays only while draft acceptance holds, and the
# break-even depth differs by model.
mtp_draft_depth: int = 3
# Vocab size prices the GPU logits buffers (ubatch x vocab x fp32,
# doubled under MTP backend sampling) — a multi-GiB term at large
# vocab sizes that a weights-only fit would miss.
n_vocab: int = 0
mmproj: "AssetFile | None" = None # vision projector, downloads with model
draft: "AssetFile | None" = None # spec-decode draft model (e.g. DSpark)
sampling: dict = field(default_factory=dict) # INI long-form launch defaults
# Oldest llama.cpp release tag that can load this model (day-0
# architectures need the release where their support landed). Empty
# means any installed engine. The pane gates download/activate on it.
min_engine: str = ""
# Editorial quality ordering (higher = smarter), authored once,
# globally, at catalog-authoring time — Artificial Analysis-informed
# where they cover the model (scripts/aa_quality_sync.py proposes,
# the commit decides), editorial elsewhere. Ranks entries for the
# per-machine recommendation; never displayed as a score (it grades
# the full-precision model, not our Q4 build).
quality: int = 0
# Fraction of the build's bytes read per decoded token: 1.0 for dense
# models (every weight streams every token), the active slice for MoE
# (attention + shared + routed experts over total). With memory
# bandwidth this predicts decode speed — the physics half of the
# recommendation.
decode_fraction: float = 1.0
def profile(self, variant: QuantVariant) -> ModelProfile:
layers = ([(LayerKind.FULL, self.per_layer_f16)] * self.full_layers
+ [(LayerKind.SWA, self.per_layer_f16)] * self.swa_layers
+ [(LayerKind.RECURRENT, 0)] * self.recurrent_layers)
return ModelProfile(
name=variant.model_id, weights_bytes=variant.weights_bytes,
embd_table_bytes=0, n_ctx_train=self.n_ctx_train,
layers=layers, swa_window=self.swa_window, moe=self.moe,
n_vocab=self.n_vocab,
kv_scale=1.2 if self.mtp else 1.0)
def download_files(self, variant: QuantVariant) -> tuple:
"""Everything a download job fetches for this variant, in order."""
extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
return tuple(variant.files) + extras
def download_bytes(self, variant: QuantVariant) -> int:
return sum(f.size_bytes for f in self.download_files(variant))
@dataclass(frozen=True)
class VariantChoice:
"""Selection result: which build this machine should download and why.
reason_key is a UI-copy discriminator, not display text."""
variant: QuantVariant
zero_spill: bool
reason_key: str # "best-large-window" | "best-fits" | "smallest-fits-spilled"
def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice | None:
"""Fit the entry's one build (Q4-class) to this machine.
Every entry ships exactly one variant (see the module docstring for
why there is no quant ladder); headroom buys a bigger window, never
a bigger quant. The fit shapes:
- "best-large-window": zero-spills at TARGET_WINDOW
- "best-fits": zero-spills at the 64K floor
- "smallest-fits-spilled": weights spill to host RAM, priced honestly
- None: even spilled, physics refuses (the machine can't run it)
"""
overhead = (RUNTIME_OVERHEAD_BYTES
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
+ ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
native = entry.n_ctx_train or FLOOR
variant = entry.variants[-1]
profile = entry.profile(variant)
need = variant.weights_bytes + overhead
if (need + ctx_bytes(profile, min(TARGET_WINDOW, native))
<= budget.usable_vram_bytes):
return VariantChoice(variant=variant, zero_spill=True,
reason_key="best-large-window")
floor_kv = ctx_bytes(profile, min(FLOOR, native))
if need + floor_kv <= budget.usable_vram_bytes:
return VariantChoice(variant=variant, zero_spill=True,
reason_key="best-fits")
if need + floor_kv <= budget.usable_vram_bytes + budget.ram_available_bytes:
return VariantChoice(variant=variant, zero_spill=False,
reason_key="smallest-fits-spilled")
return None
# ── recommendation: best quality that fits and isn't miserably slow ──
#
# Two axes, each living where it belongs. QUALITY is a judgment made once,
# globally, at authoring time (entry.quality — AA-informed, editorially
# owned). SPEED is physics computed per machine: decode is memory-bound,
# so predicted tok/s ≈ bandwidth / bytes-read-per-token, and the bytes per
# token are the build's size scaled by its decode fraction (dense reads
# everything; MoE reads the active slice). The pick: highest quality among
# entries that run resident and clear a pleasant speed floor; else the
# fastest resident entry; else the least-painful spilled one.
#
# The bandwidth axis is the `uma` flag for now: every discrete card that
# matters is 900+ GB/s GDDR while the unified-memory class measures ~1/5th
# of that, so the flag IS the high/low split. A measured per-machine
# bandwidth (one cached memcpy probe) can replace these class constants
# without touching the rule; predictions order candidates and gate the
# floor — they are not display values.
_DISCRETE_BANDWIDTH_GB_S = 1000.0 # representative GDDR6X/GDDR7 class
_UMA_BANDWIDTH_GB_S = 210.0 # measured on unified-memory NVIDIA
_HOST_BANDWIDTH_GB_S = 80.0 # spilled weights stream over host DRAM
# The one editorial constant in the tree: below this predicted decode
# speed a model stops feeling pleasant for agentic use (roughly reading
# speed with headroom for tool-call bursts). Distinct from the growth
# policy's 6 tok/s compress floor, which marks unusable, not unpleasant.
PLEASANT_FLOOR_TOK_S = 20.0
def predicted_decode_tok_s(entry: CatalogEntry, variant: QuantVariant,
budget: HardwareBudget, *,
spilled: bool = False) -> float:
"""Memory-bound decode prediction for ordering and floor-gating."""
bandwidth = (_HOST_BANDWIDTH_GB_S if spilled
else _UMA_BANDWIDTH_GB_S if budget.uma
else _DISCRETE_BANDWIDTH_GB_S)
bytes_per_token = max(1.0, variant.size_bytes * entry.decode_fraction)
return bandwidth * 1e9 / bytes_per_token
def recommended_entry(budget: HardwareBudget,
entries: "tuple[CatalogEntry, ...] | None" = None
) -> "tuple[CatalogEntry, str] | None":
"""The catalog's default pick for THIS machine, with its reason.
Callers pass pre-filtered entries when some are ineligible for
reasons the catalog can't know (engine too old); default is the full
catalog. Returns (entry, reason) — the reason is a key the UI turns
into the Recommended badge's tooltip, so the rationale shown to the
user is the branch that actually fired, never a parallel explanation
that can drift:
best-quality-resident quality won among resident entries that
clear the pleasant floor
speed-gated-quality same, but the floor eliminated a HIGHER
quality candidate — the exact 'why not the
big model?' a unified-memory owner asks
fastest-resident nothing resident clears the floor; the
quickest resident entry wins
least-painful-spilled nothing runs resident; fastest from host
memory (MoE by construction)
Returns None only when nothing fits at all.
"""
pool = CATALOG if entries is None else entries
fitting: list[tuple[CatalogEntry, VariantChoice]] = []
for entry in pool:
choice = select_variant(entry, budget)
if choice is not None:
fitting.append((entry, choice))
if not fitting:
return None
resident = [(e, c) for e, c in fitting if c.zero_spill]
pleasant = [
(e, c) for e, c in resident
if predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
]
if pleasant:
pick = max(pleasant, key=lambda t: (t[0].quality, -t[1].variant.size_bytes))[0]
floor_gated = any(e.quality > pick.quality for e, _ in resident)
return (pick, "speed-gated-quality" if floor_gated
else "best-quality-resident")
if resident:
pick = max(resident,
key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget))[0]
return (pick, "fastest-resident")
# Everything spills: take the least painful — fastest predicted decode
# from host memory (MoE wins here by construction; a dense spill
# streams every weight over the host bus).
pick = max(fitting,
key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget,
spilled=True))[0]
return (pick, "least-painful-spilled")
def recommended_id(budget: HardwareBudget,
entries: "tuple[CatalogEntry, ...] | None" = None) -> str | None:
picked = recommended_entry(budget, entries)
return picked[0].id if picked is not None else None
# ── catalog data: packaged JSON, refreshed from GitHub in memory ─
#
# The catalog DATA lives in catalog.json (checked in beside this module
# and shipped as package data); this module keeps all policy. At import
# we load the packaged copy — no network on the import path. A TTL-gated
# background refresh fetches the same file from the repo's main branch
# and swaps it in memory only: nothing on disk changes, so a git
# checkout never sees a dirty tracked file and the packaged copy remains
# the offline truth. A reverted commit on main heals every install on
# its next fetch, and day-0 entries reach users without an app release.
_CATALOG_URL = ("https://raw.githubusercontent.com/NousResearch/hermes-agent"
"/main/hermes_cli/local_runtime/catalog.json")
_SCHEMA_VERSION = 1
_REFRESH_TTL_S = 6 * 3600
_refresh_lock = threading.Lock()
_last_refresh_attempt = 0.0
def _asset_from(d: "dict | None") -> "AssetFile | None":
if not d:
return None
return AssetFile(path=d["path"], size_bytes=int(d["size_bytes"]),
local=d.get("local"))
def _load_catalog(doc: dict) -> "tuple[CatalogEntry, ...]":
"""Parse a catalog document into entries. Unknown fields are ignored
(newer catalogs stay readable by older apps); a major schema bump is
the signal that they wouldn't be, and the caller skips the document."""
if int(doc.get("schema_version", 0)) != _SCHEMA_VERSION:
raise ValueError(f"catalog schema {doc.get('schema_version')!r} "
f"(this build reads {_SCHEMA_VERSION})")
entries = []
for m in doc["models"]:
variants = tuple(
QuantVariant(quant=v["quant"],
files=tuple(_asset_from(f) for f in v["files"]),
validated=bool(v.get("validated")))
for v in m["variants"])
entries.append(CatalogEntry(
id=m["id"], display_name=m["display_name"],
description=m["description"], repo=m["repo"], variants=variants,
n_ctx_train=int(m["n_ctx_train"]),
full_layers=int(m["full_layers"]),
recurrent_layers=int(m["recurrent_layers"]),
per_layer_f16=int(m["per_layer_f16"]),
swa_layers=int(m.get("swa_layers", 0)),
swa_window=int(m.get("swa_window", 0)),
moe=bool(m.get("moe")), mtp=bool(m.get("mtp")),
mtp_draft_depth=int(m.get("mtp_draft_depth", 3)),
n_vocab=int(m.get("n_vocab", 0)),
mmproj=_asset_from(m.get("mmproj")),
draft=_asset_from(m.get("draft")),
sampling=dict(m.get("sampling", {})),
min_engine=str(m.get("min_engine", "")),
quality=int(m.get("quality", 0)),
decode_fraction=float(m.get("decode_fraction", 1.0)),
))
return tuple(entries)
def _packaged_catalog() -> "tuple[CatalogEntry, ...]":
from importlib.resources import files
raw = files("hermes_cli.local_runtime").joinpath("catalog.json").read_text(
encoding="utf-8")
return _load_catalog(json.loads(raw))
CATALOG: "tuple[CatalogEntry, ...]" = _packaged_catalog()
def refresh_catalog(force: bool = False) -> bool:
"""Fetch the current catalog from the repo and swap it in memory.
Best-effort by design: any failure (offline, GitHub down, unreadable
schema) leaves the running catalog untouched and retries after the
TTL. Returns True when a fetched document replaced the catalog."""
global CATALOG, _last_refresh_attempt
now = time.monotonic()
with _refresh_lock:
if not force and now - _last_refresh_attempt < _REFRESH_TTL_S:
return False
_last_refresh_attempt = now
try:
req = urllib.request.Request(
_CATALOG_URL, headers={"User-Agent": "hermes-local-runtime"})
with urllib.request.urlopen(req, timeout=10) as r:
fetched = _load_catalog(json.load(r))
except Exception as exc: # noqa: BLE001
logger.debug("catalog refresh skipped: %s", exc)
return False
if fetched != CATALOG:
logger.info("catalog refreshed from repo (%d models)", len(fetched))
CATALOG = fetched
return True
def refresh_catalog_soon() -> None:
"""TTL-gated background refresh; returns immediately. The caller's
current request serves the catalog it already has — the refresh
lands for the next one."""
if time.monotonic() - _last_refresh_attempt < _REFRESH_TTL_S:
return
threading.Thread(target=refresh_catalog, daemon=True,
name="catalog-refresh").start()
def catalog_by_id() -> dict[str, CatalogEntry]:
return {entry.id: entry for entry in CATALOG}
def find_variant(entry_id: str, model_id: str) -> QuantVariant | None:
entry = catalog_by_id().get(entry_id)
if entry is None:
return None
return next((v for v in entry.variants if v.model_id == model_id), None)
def find_entry_for_model(model_id: str) -> "tuple[CatalogEntry, QuantVariant] | None":
"""Locate the entry + variant that owns a staged model id."""
for entry in CATALOG:
for variant in entry.variants:
if variant.model_id == model_id:
return entry, variant
return None
+286
View File
@@ -0,0 +1,286 @@
"""Context policy — the window ladder for managed local models.
One contract: any model runs at any window up to its native max; hardware
and session depth only change tokens/s. Constants, not knobs — nothing in
this module reads config.
The policy encodes behavior measured on real hardware (llama.cpp,
discrete NVIDIA GPUs on Windows/WDDM, and unified-memory devices):
- Windows never over-allocates VRAM ahead of need. On WDDM, allocating
past residency slows decode roughly 9x even at identical conversation
depth — the driver silently demotes pages instead of failing. Every
window grant therefore re-fits against live memory at grant time.
- Models launch at the largest window that fits entirely in GPU memory
(zero-spill) and grow toward their native max as the session needs
room, at request boundaries only.
- Growth re-prefills the conversation into the larger window. Measured
cost is comparable to save/restore on discrete GPUs, and recurrent or
hybrid-attention models cannot rewind mid-sequence anyway, so
re-prefill is the only mechanism that works for every architecture.
- Every recommended model gets at least a 64K window. When weights alone
exceed VRAM, the fit deliberately spills weights to host RAM to
protect that floor (measured: an explicit context size makes the fit
spill weights and hold the window rather than shrink it).
- Below ~6 tok/s decode, growth stops and compression becomes the
default; deeper context is an explicit per-session choice. The deepest
measured host-spilled configuration bottomed out near this rate.
- Spilled mixture-of-experts configs pin expert/FFN weights to host so
attention and KV stay GPU-resident — measured ~1.75x faster than
spilling layers naively at the same host byte count.
- Speculative decoding (MTP) defaults on only for spilled configs, where
its speedup is largest (measured 1.43x spilled vs 1.35x resident).
"""
from __future__ import annotations
from dataclasses import dataclass, field
from hermes_cli.local_runtime.estimator import (
HardwareBudget,
ModelProfile,
PhysicsRefusal,
ctx_bytes,
physics_check,
)
FLOOR = 64 * 1024 # = target; one internal constant
_LADDER_GROWTH = 1.5
_GROW_AT_OCCUPANCY = 0.85 # of the current window, at turn boundary
SPEED_FLOOR_TOK_S = 6.0 # deepest measured spill bottomed near this
_EARLY_COST_CTX_FRACTION = 0.15 # bounded early cost when weights spill
# TARGET_WINDOW: the smallest ladder rung at which compression becomes the
# exception rather than the routine. Measured over 161 real agentic
# sessions: 66% complete uncompressed in 64K, 82% in 96K, 91% in 144K —
# and the marginal gain past 144K (+6 points for 216K) falls below the
# quality cost of stepping down another quant. Quant selection prefers
# the best build that reaches this; the FLOOR remains the guarantee.
TARGET_WINDOW = 144 * 1024
# What a load really costs beyond weights + KV: CUDA contexts and compute
# buffers at the DEFAULT microbatch (-ub 512, no MTP). Measured on a
# 32 GiB card: a model estimated at 29.3 GiB (weights+KV) loaded at
# ~31.2 GiB resident and the server's own fit still shaved a layer to
# CPU. Microbatch/MTP logits buffers are priced separately per model
# (ub_logits_bytes — they scale with the model's vocab and doubled once
# packed a card 3.9 GiB past this constant). Callers add mmproj bytes on
# top.
RUNTIME_OVERHEAD_BYTES = int(1.5 * (1 << 30))
def ladder(native: int) -> list[int]:
"""64K -> 96K -> 128K -> ... -> native (native always the last rung)."""
rungs: list[int] = []
step = float(FLOOR)
while step < native:
rungs.append(int(step))
step *= _LADDER_GROWTH
rungs.append(native)
return rungs
@dataclass
class WindowDecision:
window: int
spill_bytes: int # weights displaced to host at this window
kv_on_gpu: bool
reasons: list[str] = field(default_factory=list)
@property
def spilled(self) -> bool:
return self.spill_bytes > 0
def initial_window(profile: ModelProfile, budget: HardwareBudget,
*, flash_attention: bool = True,
overhead_bytes: int = 0) -> WindowDecision | PhysicsRefusal:
"""The launch decision: largest cheap rung, never below the floor.
Zero-spill rung: weights + ctx + overhead fit usable VRAM entirely.
Bounded-early-cost rung: weights already exceed VRAM; take the largest
rung whose ctx stays <= ~15% of usable VRAM.
Floor everywhere, capped at native.
``overhead_bytes``: runtime cost beyond weights+KV (RUNTIME_OVERHEAD
plus the vision projector when one loads). Zero keeps this function
pure physics for decision-table tests; production callers pass it.
"""
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
if refusal:
return refusal
native = profile.n_ctx_train or FLOOR
rungs = ladder(native)
reasons: list[str] = []
best_zero_spill: int | None = None
for rung in rungs:
need = (profile.weights_bytes + overhead_bytes
+ ctx_bytes(profile, rung, flash_attention=flash_attention))
if need <= budget.usable_vram_bytes:
best_zero_spill = rung
else:
break
if best_zero_spill is not None and best_zero_spill >= min(FLOOR, native):
window = best_zero_spill
reasons.append(f"largest zero-spill rung ({window // 1024}K)")
else:
# Weights spill from turn one (steep-curve model on a small card) —
# hold the floor, bound the early ctx cost.
cap = int(budget.usable_vram_bytes * _EARLY_COST_CTX_FRACTION)
window = min(FLOOR, native)
for rung in rungs:
if rung < window:
continue
if ctx_bytes(profile, rung, flash_attention=flash_attention) <= cap:
window = rung
else:
break
reasons.append(f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)")
kv = ctx_bytes(profile, window, flash_attention=flash_attention)
spill = max(0, profile.weights_bytes + kv - budget.usable_vram_bytes)
return WindowDecision(window=window, spill_bytes=spill,
kv_on_gpu=kv <= budget.usable_vram_bytes,
reasons=reasons)
@dataclass
class GrowthDecision:
action: str # "grow" | "hold" | "compress-default"
next_window: int | None = None
reason: str = ""
def growth_decision(profile: ModelProfile, budget: HardwareBudget, *,
current_window: int, session_tokens: int,
measured_decode_tok_s: float | None,
server_idle: bool,
flash_attention: bool = True,
occupancy_confirmed: bool = False) -> GrowthDecision:
"""One growth evaluation, END-OF-TURN ONLY (caller guarantees the turn
boundary; recurrent state cannot rewind mid-sequence).
Gate ordering:
1. occupancy (~85%) — nothing to do before the edge;
2. native cap — the contract tops out at trained context;
3. idleness — growth re-grants only on an otherwise-idle
server (concurrency design);
4. speed floor — below it, compression becomes the default and deeper
is an explicit user choice;
5. re-fit against LIVE free memory (the rung must fit residency
NOW, not at launch time — over-allocation is the slow path).
``occupancy_confirmed``: the caller has independently established that
the session is at its window's edge (the agent's compression gate fired
on its own threshold). Skips gate 1 so two separately-derived edge
definitions can't deadlock into compress-before-grow.
"""
if not occupancy_confirmed and session_tokens < current_window * _GROW_AT_OCCUPANCY:
return GrowthDecision("hold", reason="session below growth occupancy")
native = profile.n_ctx_train or current_window
if current_window >= native:
return GrowthDecision("compress-default",
reason="at native window; compression is the only move")
if not server_idle:
return GrowthDecision("hold", reason="server busy; re-grant deferred to idle")
if measured_decode_tok_s is not None and measured_decode_tok_s < SPEED_FLOOR_TOK_S:
return GrowthDecision(
"compress-default",
reason=(f"decode {measured_decode_tok_s:.1f} tok/s below the "
f"~{SPEED_FLOOR_TOK_S:.0f} tok/s floor; growth is now an "
"explicit per-session choice"))
next_rung = next((r for r in ladder(native) if r > current_window), native)
# Re-fit against live free memory: allocation beyond residency is the
# slow path, so a rung that no longer fits doesn't get granted.
kv = ctx_bytes(profile, next_rung, flash_attention=flash_attention)
total_need = profile.weights_bytes + kv
if total_need > budget.usable_vram_bytes + budget.ram_available_bytes:
return GrowthDecision("compress-default",
reason="next rung exceeds physics; compression instead")
return GrowthDecision("grow", next_window=next_rung,
reason=f"rung {current_window // 1024}K -> {next_rung // 1024}K")
def spill_overrides(profile: ModelProfile) -> list[str]:
"""-ot placement for spilled configs: expert/FFN weights to host so
attention + KV stay GPU-resident. MoE gets the expert pattern;
hybrids push recurrent-layer FFNs (their n_head_kv==0 layers carry no
KV worth protecting)."""
if profile.moe:
return ["-ot", r"blk\.\d+\.ffn_.*_exps\.weight=CPU"]
if profile.recurrent_layer_count:
return ["-ot", r"blk\.\d+\.ffn_.*\.weight=CPU"]
return [] # dense: fit's back-to-front layer cut is the only axis
def launch_args(profile: ModelProfile, decision: WindowDecision, *,
flash_attention: bool = True,
mtp_capable: bool = False,
mtp_draft_depth: int = 3,
uma: bool = False,
mtp_prefill: bool = False) -> list[str]:
"""Per-model launch flags from a window decision. Explicit -c puts fit
into spill-weights-and-hold-ctx; q8 KV cache wherever flash attention
exists; -ot placement on spilled configs — DISCRETE cards only.
``uma``: on unified memory there is no bus to protect tensors from —
"CPU" and "GPU" are the same silicon, and pinning FFN weights to the
host path just forces CPU compute (measured well over 2x slower than
letting the allocator place everything). The discrete
~1.75x win the -ot pattern encodes does not transfer; a spilled UMA
config runs unpinned.
MTP and the large prefill microbatch both win, and whether they may
STACK is a fit question, not a rule: backend sampling keeps a
ubatch x vocab x fp32 logits buffer on the GPU and MTP's draft
context doubles it, so the stacked posture costs a few GiB extra at
large vocab. Where it fits, it measures best on both axes (Qwen3.8
Q4 on a 32 GiB card: 93.3 tok/s decode vs 89.5 at ub512, prefill
slightly better too); where it doesn't, ub512 keeps the decode win
without packing the card. ``mtp_prefill`` is that fit verdict —
presets decide it against the priced margin, and ub_logits_bytes()
prices the same choice so the flag and its cost travel together."""
args = ["-c", str(decision.window)]
if mtp_capable:
args += ["--spec-type", "draft-mtp",
"--spec-draft-n-max", str(mtp_draft_depth),
"--backend-sampling", "--spec-draft-backend-sampling"]
if mtp_prefill:
args += ["-b", "4096", "-ub", "2048"]
else:
args += ["-b", "2048", "-ub", "2048"]
if flash_attention:
args += ["-ctk", "q8_0", "-ctv", "q8_0", "-fa", "on"]
if decision.spilled and not uma:
args += spill_overrides(profile)
return args
def ub_logits_bytes(n_vocab: int, *, mtp_capable: bool,
mtp_prefill: bool = False) -> int:
"""GPU logits/compute-buffer cost of the microbatch posture chosen by
launch_args, priced from the model's own vocab and calibrated against
measured server RSS (Qwen3.8 Q4, both postures, three windows):
stacked (MTP + ub2048): ubatch x vocab x fp32 x 1.5 (~2.9 GiB at
248K vocab; fitted 2.5, rounded up)
decode (MTP + ub512): ubatch x vocab x fp32 x 2 (~1.0 GiB)
plain (ub2048): ubatch x vocab x fp32 (~1.9 GiB)
Callers add this to RUNTIME_OVERHEAD per model — the flag and its
price travel together or the fit lies."""
v = max(0, int(n_vocab))
if mtp_capable and mtp_prefill:
return int(2048 * v * 4 * 1.5)
if mtp_capable:
return 512 * v * 4 * 2
return 2048 * v * 4
+80
View File
@@ -0,0 +1,80 @@
"""Detection of running llama-server instances.
Probes well-known local roots and fingerprints genuine llama-server via
/props (build_info + model fields — Ollama and LM Studio answer /v1/models
but not /props). The credential is reachability; detection never needs a
key, but honors one if the probed server requires it (401 -> detected,
auth_required=True).
"""
from __future__ import annotations
import json
import urllib.error
import urllib.request
from dataclasses import dataclass
# Always 127.0.0.1 — resolving localhost costs ~2s/request on Windows.
DEFAULT_PROBE_PORTS = (8080,) # llama-server default; managed port comes from config
@dataclass
class DetectedServer:
base_url: str # OpenAI-compatible /v1 root
build_info: str # e.g. "b10290-c8e03ce81"
model_path: str # currently loaded model (may be empty in router mode)
n_ctx: int | None
router_mode: bool # GET /models answered -> router management available
auth_required: bool
def _get(url: str, timeout_s: int = 3) -> tuple[int, dict | None]:
try:
with urllib.request.urlopen(url, timeout=timeout_s) as r:
raw = r.read()
return r.status, (json.loads(raw) if raw else None)
except urllib.error.HTTPError as exc:
return exc.code, None
except (urllib.error.URLError, OSError, TimeoutError, json.JSONDecodeError):
return 0, None
def probe_port(port: int) -> DetectedServer | None:
"""One port: /props fingerprint, then /models for router capability."""
root = f"http://127.0.0.1:{port}"
status, props = _get(f"{root}/props")
if status == 401:
return DetectedServer(base_url=f"{root}/v1", build_info="", model_path="",
n_ctx=None, router_mode=False, auth_required=True)
if status != 200 or not isinstance(props, dict):
return None
build = str(props.get("build_info", ""))
if not build:
return None # answers /props but isn't llama-server
n_ctx = None
dgs = props.get("default_generation_settings")
if isinstance(dgs, dict):
n_ctx = dgs.get("n_ctx")
models_status, models = _get(f"{root}/models")
return DetectedServer(
base_url=f"{root}/v1",
build_info=build,
model_path=str(props.get("model_path", "")),
n_ctx=n_ctx,
router_mode=(models_status == 200 and isinstance(models, dict)
and "data" in models),
auth_required=False,
)
def detect_server(extra_ports: tuple[int, ...] = ()) -> DetectedServer | None:
"""First hit across default + extra ports (managed port, config port)."""
seen = set()
for port in (*DEFAULT_PROBE_PORTS, *extra_ports):
if port in seen:
continue
seen.add(port)
hit = probe_port(port)
if hit:
return hit
return None
+194
View File
@@ -0,0 +1,194 @@
"""Endpoint resolution for llamacpp-alias requests (provider integration).
The seam between the existing provider mechanism and the managed runtime:
``provider: llamacpp`` with no explicit base_url resolves, in order, to
1. the managed server this Hermes is supervising (state file written by
LlamaServerSupervisor.start, removed on stop, staleness-checked), or
2. a detected external llama-server.
Returns None when neither exists — the caller falls through to the normal
custom-provider path and its own error reporting.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.error
import urllib.request
LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
logger = logging.getLogger(__name__)
def _pid_alive(pid: int) -> bool:
"""Liveness for the state file's supervisor-child pid.
psutil when available; otherwise fall back to True (optimistic) — on
Windows ``os.kill(pid, 0)`` TERMINATES the process, so it must never be
used as a probe (windows-git-bash interop pitfall).
"""
if not pid or pid < 0:
return False
try:
import psutil # type: ignore
return psutil.pid_exists(pid)
except Exception: # noqa: BLE001
return True
def _state_endpoint() -> dict | None:
from hermes_cli.local_runtime.supervisor import state_path
path = state_path()
if not path.exists():
return None
try:
state = json.loads(path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError):
return None
base_url = state.get("base_url", "")
if not base_url:
return None
endpoint = {"base_url": base_url, "api_key": state.get("api_key", "")}
# Ownership proof: the stable port means a SECOND install (different
# HERMES_HOME — a scratch profile, say) can own 127.0.0.1:18434 with a
# different api key while this install's state file still points there.
# /health is a public route, so it answers 200 for ANYONE's server —
# trusting it alone sent every chat request and the load-progress
# watcher at a server that 401s our key, silently. The recorded
# supervisor pid is the tiebreaker: health-200 from a server whose
# recorded child is DEAD is someone else's server, never a starting one.
pid_ok = _pid_alive(int(state.get("pid") or 0))
# Healthy server: done (when it's ours).
try:
health = base_url.rsplit("/v1", 1)[0] + "/health"
with urllib.request.urlopen(health, timeout=3) as r:
if r.status == 200:
return endpoint if pid_ok else None
except (urllib.error.URLError, OSError, TimeoutError):
pass
# Not healthy YET: a live supervisor child is a STARTING server (state
# is written at spawn; llama-server takes seconds to listen). Resolve
# optimistically so readiness probes racing the boot see a configured
# provider, not missing credentials. A dead pid is a crashed-without-
# cleanup leftover — ignore it so requests don't blackhole.
if pid_ok:
return endpoint
return None
def resolve_llamacpp_endpoint(config: dict | None = None,
wait_for_boot_s: float = 8.0) -> dict | None:
"""Managed-first, detection-second endpoint for llamacpp aliases.
Returns {"base_url", "api_key"} or None. api_key is empty for keyless
external servers (callers substitute the SDK placeholder).
Boot-race rung: on a fresh backend start there is NO state file yet —
the lifespan boot thread is still spawning the server (config load +
preset generation + spawn ≈ 1-3 s) while the desktop's readiness probe
fires the moment the WebSocket connects. When the runtime is enabled
and installed, a missing endpoint means BOOTING, not unconfigured:
poll briefly for the state file instead of failing the probe (twice
observed as 'no usable credentials' → onboarding on restart).
"""
managed = _state_endpoint()
if managed:
return managed
from hermes_cli.local_runtime.detect import detect_server
extra = ()
if config:
ports = (config.get("local_runtime") or {}).get("detect_ports") or []
extra = tuple(int(p) for p in ports)
hit = detect_server(extra_ports=extra)
if hit and not hit.auth_required:
return {"base_url": hit.base_url, "api_key": ""}
if wait_for_boot_s > 0 and _boot_in_flight(config):
_kick_managed_boot(config)
deadline = time.monotonic() + wait_for_boot_s
while time.monotonic() < deadline:
time.sleep(0.25)
managed = _state_endpoint()
if managed:
return managed
return None
_KICK_LOCK = threading.Lock()
def _kick_managed_boot(config: dict | None) -> None:
"""Actively start the managed server when resolution finds it missing.
The wait loop above assumes some OTHER thread is bringing the server
up — true only at backend start (the lifespan boot thread). A router
that dies LATER leaves no boot in flight: the backend process was
killed with the router as part of its tree, or another install took
the stable port and the ownership guard rightly refused it. In those
states the wait just expired and agent init failed with 'no provider
configured', even though the fix is the same idempotent ensure call
the lifespan makes. Kick it here, off-thread (the resolver's wait
stays bounded; ensure's own state checks make a concurrent lifespan
boot harmless) and non-reentrant (racing resolutions kick once).
"""
if not _KICK_LOCK.acquire(blocking=False):
return # a kick is already in flight
def _boot() -> None:
try:
cfg = config
if cfg is None:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
ensure_local_runtime(cfg)
except Exception: # noqa: BLE001 — best-effort; resolution falls back
logger.warning("on-demand managed-server boot failed", exc_info=True)
finally:
_KICK_LOCK.release()
threading.Thread(target=_boot, daemon=True,
name="lr-on-demand-boot").start()
def _boot_in_flight(config: dict | None) -> bool:
"""True when the managed runtime is enabled and installed — the state
a lifespan boot thread is (or is about to be) bringing up.
Installed-ness is a verified-manifest scan under runtimes_root(), NOT a
server_binary() call — that helper requires an install_dir argument, and
calling it bare made this gate throw-and-return-False forever, silently
disabling the boot wait (the regression
test had monkeypatched this function instead of exercising it).
"""
try:
if config is None:
from hermes_cli.config import load_config
config = load_config()
if not ((config or {}).get("local_runtime") or {}).get("enabled"):
return False
import json as _json
from hermes_cli.local_runtime.binaries import runtimes_root
for manifest in runtimes_root().glob("*/*/manifest.json"):
try:
if _json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
return True
except (ValueError, OSError):
continue
return False
except Exception: # noqa: BLE001
return False
+179
View File
@@ -0,0 +1,179 @@
"""Per-layer context-memory estimator + physics check.
The whole-model dense formula misprices 1M-context hybrids by ~100x; the
per-layer walk fixes that, and every column is measured on real GGUFs:
- full-attention layer: linear in T (B1: 144.0 KiB/tok on Qwen3-4B
f16 — formula-exact)
- SWA layer: capped at the sliding window
- recurrent layer (n_head_kv == 0): constant (state is ~context-free)
- q8_0 KV = exactly 34/64 of f16 (holds on CUDA and CPU)
- weights: exact from the tensor table (within 0.01% of the loader)
The estimator is ADVISORY: fit's allocation is authoritative at launch and
the touch generation is ground truth after it. Unknown shapes round UP
(never underestimate memory).
"""
from __future__ import annotations
from dataclasses import dataclass
from enum import Enum
from hermes_cli.local_runtime.gguf import GGUFHeader
# q8_0: 34-byte blocks of 32 f16-equivalent elements (exact).
_Q8_BYTES_PER_ELEM = 34 / 32
_F16_BYTES_PER_ELEM = 2.0
# Architectures with a known SWA layer pattern: arch -> fraction of layers
# that are sliding-window. Unknown SWA archs conservatively treat every
# layer as full attention (overestimate; safe direction).
_SWA_LAYER_FRACTION = {"gemma3": 5 / 6, "gemma2": 1 / 2}
# Per-recurrent-layer state allowance (bytes/seq). Deliberately generous —
# Measured: an entire hybrid slot state is ~99 MB including 8K tokens of
# full-attn KV, so tens of MiB total is the right order; unknown SSM shapes
# must never underestimate.
_RECURRENT_STATE_PER_LAYER = 4 << 20
class LayerKind(Enum):
FULL = "full"
SWA = "swa"
RECURRENT = "recurrent"
@dataclass
class ModelProfile:
"""Everything the policy needs, decoupled from GGUF parsing so the
decision-table tests can construct profiles directly (design's
verification plan)."""
name: str
weights_bytes: int
embd_table_bytes: int
n_ctx_train: int
layers: list[tuple[LayerKind, int]] # (kind, kv_bytes_per_token_f16);
# SWA/recurrent reuse the same
# per-token figure, capped/ignored
swa_window: int = 0
moe: bool = False
architecture: str = ""
n_vocab: int = 0 # prices logits buffers (ubatch x vocab)
# Context-cost multiplier. MTP spec decode keeps a small draft
# context beside the main one. Calibrated against four measured
# server-RSS points on Qwen3.8 Q4 (128K/221K/256K, both postures):
# the draft adds ~17% to per-token KV; 1.2 rounds up so the error
# stays on the safe side (+250 MiB at 256K, never negative).
kv_scale: float = 1.0
@property
def per_token_kv_f16(self) -> int:
"""Uncapped per-token KV cost (full + SWA share)."""
return sum(b for kind, b in self.layers if kind != LayerKind.RECURRENT)
@property
def recurrent_layer_count(self) -> int:
return sum(1 for kind, _ in self.layers if kind == LayerKind.RECURRENT)
@dataclass
class HardwareBudget:
"""Memory the physics check may budget against.
Budget-source rule: discrete cards may trust the device query
(measured honest); unified-memory devices must budget from OS free
physical memory minus headroom — their device queries have been
observed off by 3x. Callers construct
this accordingly; the estimator just consumes it.
"""
usable_vram_bytes: int # live free (discrete) / derived (UMA)
total_device_bytes: int
ram_available_bytes: int
uma: bool = False
def profile_from_gguf(header: GGUFHeader) -> ModelProfile:
kv_heads = header.head_counts_kv()
dk, dv = header.head_dim_k, header.head_dim_v
swa_fraction = _SWA_LAYER_FRACTION.get(header.architecture, 0.0)
has_swa = header.sliding_window > 0 and swa_fraction > 0
layers: list[tuple[LayerKind, int]] = []
n_attn_seen = 0
n_attn_total = sum(1 for h in kv_heads if h > 0)
n_swa = round(n_attn_total * swa_fraction) if has_swa else 0
for heads in kv_heads:
if heads == 0:
layers.append((LayerKind.RECURRENT, 0))
continue
per_token = round(heads * (dk + dv) * _F16_BYTES_PER_ELEM)
# Distribute the SWA share across the first n_swa attention layers;
# only the full/SWA SPLIT matters to the totals, not which indexes.
kind = LayerKind.SWA if n_attn_seen < n_swa else LayerKind.FULL
layers.append((kind, per_token))
n_attn_seen += 1
return ModelProfile(
name=header.path,
weights_bytes=header.tensor_bytes,
embd_table_bytes=header.embd_table_bytes,
n_ctx_train=header.n_ctx_train,
layers=layers,
swa_window=header.sliding_window,
moe=header.expert_count > 0,
architecture=header.architecture,
n_vocab=header.n_vocab,
)
def kv_dtype_factor(flash_attention: bool) -> float:
"""q8_0 with FA (every backend we ship); f16 on exotic non-FA fallbacks
— the 64K guarantee stands either way, the physics check just prices
the doubled KV (design: KV dtype is behavior, not config)."""
return (_Q8_BYTES_PER_ELEM / _F16_BYTES_PER_ELEM) if flash_attention else 1.0
def ctx_bytes(profile: ModelProfile, window: int, *,
flash_attention: bool = True) -> int:
"""Context memory for one window: full layers linear in T, SWA layers
capped at the sliding window, recurrent layers constant. Scaled by
profile.kv_scale (MTP draft context)."""
factor = kv_dtype_factor(flash_attention)
total = 0.0
for kind, per_token_f16 in profile.layers:
if kind == LayerKind.RECURRENT:
total += _RECURRENT_STATE_PER_LAYER
elif kind == LayerKind.SWA:
total += per_token_f16 * factor * min(window, profile.swa_window)
else:
total += per_token_f16 * factor * window
return int(total * profile.kv_scale)
@dataclass
class PhysicsRefusal:
"""The only true refusal: weights + floor-KV + state exceed VRAM + RAM.
The remedy is a smaller quant, never a smaller window."""
needed_bytes: int
available_bytes: int
message: str
def physics_check(profile: ModelProfile, budget: HardwareBudget,
floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
needed = (profile.weights_bytes
+ ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
flash_attention=flash_attention))
available = budget.usable_vram_bytes + budget.ram_available_bytes
if needed > available:
gib = 1 << 30
return PhysicsRefusal(
needed_bytes=needed, available_bytes=available,
message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
"of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
return None
+220
View File
@@ -0,0 +1,220 @@
"""GGUF metadata + tensor-table reader (stdlib only).
Feeds the per-layer context estimator: architecture, layer count, per-layer
KV head counts (0 = recurrent layer — the hybrid discriminator), head dims,
sliding-window config, trained context, and exact weight bytes summed from
the tensor table (validated to within 0.01% of the loader's buffer).
Reads the header only (metadata + tensor infos); never touches tensor data,
so it is fast enough to run at picker time on multi-GB files.
"""
from __future__ import annotations
import struct
from dataclasses import dataclass, field
from pathlib import Path
_GGUF_MAGIC = b"GGUF"
# ggml tensor type sizes: type_id -> (block_bytes, block_elems).
# IQ-family sizes verified against ggml-common.h.
_GGML_TYPE_SIZES = {
0: (4, 1), 1: (2, 1), 2: (18, 32), 3: (20, 32), 6: (22, 32), 7: (24, 32),
8: (34, 32), 9: (36, 32), 10: (84, 256), 11: (110, 256), 12: (144, 256),
13: (176, 256), 14: (210, 256), 15: (292, 256), 16: (66, 256),
17: (74, 256), 18: (98, 256), 19: (50, 256), 20: (18, 32),
21: (110, 256), 22: (82, 256), 23: (136, 256), 24: (1, 1), 25: (2, 1),
26: (4, 1), 27: (8, 1), 28: (8, 1), 29: (56, 256), 30: (2, 1),
}
# GGUF metadata value types.
_V_UINT8, _V_INT8, _V_UINT16, _V_INT16 = 0, 1, 2, 3
_V_UINT32, _V_INT32, _V_FLOAT32, _V_BOOL = 4, 5, 6, 7
_V_STRING, _V_ARRAY, _V_UINT64, _V_INT64, _V_FLOAT64 = 8, 9, 10, 11, 12
_SCALAR_FMT = {
_V_UINT8: "<B", _V_INT8: "<b", _V_UINT16: "<H", _V_INT16: "<h",
_V_UINT32: "<I", _V_INT32: "<i", _V_FLOAT32: "<f", _V_BOOL: "<?",
_V_UINT64: "<Q", _V_INT64: "<q", _V_FLOAT64: "<d",
}
@dataclass
class GGUFHeader:
path: str
version: int
metadata: dict = field(default_factory=dict)
n_tensors: int = 0
tensor_bytes: int = 0 # exact sum over the tensor table
embd_table_bytes: int = 0 # token_embd.weight (duplicated host-side
# when fully offloaded)
# ── typed accessors ──────────────────────────────────────
@property
def architecture(self) -> str:
return str(self.metadata.get("general.architecture", ""))
def _arch_key(self, suffix: str):
return self.metadata.get(f"{self.architecture}.{suffix}")
@property
def n_layer(self) -> int:
return int(self._arch_key("block_count") or 0)
@property
def n_vocab(self) -> int:
"""Vocabulary size: prices the GPU logits buffers (they scale
ubatch x vocab). vocab_size metadata when present, else the
tokenizer list length."""
v = self._arch_key("vocab_size")
if v:
return int(v)
toks = self.metadata.get("tokenizer.ggml.tokens")
return len(toks) if isinstance(toks, list) else 0
@property
def n_ctx_train(self) -> int:
return int(self._arch_key("context_length") or 0)
@property
def sampling_defaults(self) -> dict:
"""Upstream's recommended sampling, when the file carries it.
Model publishers bake general.sampling.* keys into the GGUF
(llama-server reads them as that model's default generation
settings), so the file itself is the source of truth for how its
publisher wants it run — it arrives with the download and updates
with every re-upload, no catalog required. Returned as preset INI
keys; empty when the file carries none.
"""
ini_key = {"temp": "temp", "temperature": "temp", "top_p": "top-p",
"top_k": "top-k", "min_p": "min-p",
"repeat_penalty": "repeat-penalty",
"presence_penalty": "presence-penalty"}
out = {}
for key, value in self.metadata.items():
if not key.startswith("general.sampling."):
continue
name = ini_key.get(key.rsplit(".", 1)[-1])
if name is not None and isinstance(value, (int, float)):
num = round(float(value), 4)
out[name] = str(int(num)) if num == int(num) else str(num)
return out
@property
def n_embd(self) -> int:
return int(self._arch_key("embedding_length") or 0)
@property
def n_head(self) -> int:
v = self._arch_key("attention.head_count")
if isinstance(v, list):
return int(max(v))
return int(v or 0)
@property
def full_attention_interval(self) -> int:
"""GDN-hybrid discriminator (qwen35 family): every Nth layer is full
attention, the rest are linear/recurrent. 0 = not present."""
return int(self._arch_key("full_attention_interval") or 0)
def head_counts_kv(self) -> list[int]:
"""Per-layer KV head counts; 0 marks a recurrent/linear layer (the
n_head_kv == 0 discriminator).
Three GGUF shapes, each verified against real files:
- per-layer array (nemotron_h_moe): use as-is;
- scalar + full_attention_interval (qwen35): the scalar applies to
every INTERVAL-th layer (1-indexed: layers where (i+1) % N == 0),
zero elsewhere — pricing all layers as attention was a 4x
overestimate on Qwen3.6-27B;
- plain scalar (dense): broadcast to every layer.
"""
v = self._arch_key("attention.head_count_kv")
if isinstance(v, list):
return [int(x) for x in v]
scalar = int(v or 0)
interval = self.full_attention_interval
if interval > 1:
return [scalar if (i + 1) % interval == 0 else 0
for i in range(self.n_layer)]
return [scalar] * self.n_layer
@property
def head_dim_k(self) -> int:
v = self._arch_key("attention.key_length")
if v:
return int(v)
return self.n_embd // self.n_head if self.n_head else 0
@property
def head_dim_v(self) -> int:
v = self._arch_key("attention.value_length")
if v:
return int(v)
return self.head_dim_k
@property
def sliding_window(self) -> int:
return int(self._arch_key("attention.sliding_window") or 0)
@property
def expert_count(self) -> int:
return int(self._arch_key("expert_count") or 0)
def read_gguf_header(path: str | Path) -> GGUFHeader:
path = Path(path)
def read_str(f) -> str:
(n,) = struct.unpack("<Q", f.read(8))
return f.read(n).decode("utf-8", errors="replace")
def read_value(f, vtype: int):
if vtype == _V_STRING:
return read_str(f)
if vtype == _V_ARRAY:
(etype,) = struct.unpack("<I", f.read(4))
(n,) = struct.unpack("<Q", f.read(8))
return [read_value(f, etype) for _ in range(n)]
fmt = _SCALAR_FMT[vtype]
(value,) = struct.unpack(fmt, f.read(struct.calcsize(fmt)))
return value
with open(path, "rb") as f:
if f.read(4) != _GGUF_MAGIC:
raise ValueError(f"not a GGUF file: {path}")
(version,) = struct.unpack("<I", f.read(4))
n_tensors, n_kv = struct.unpack("<QQ", f.read(16))
metadata: dict = {}
for _ in range(n_kv):
key = read_str(f)
(vtype,) = struct.unpack("<I", f.read(4))
metadata[key] = read_value(f, vtype)
tensor_bytes = 0
embd_bytes = 0
for _ in range(n_tensors):
name = read_str(f)
(n_dims,) = struct.unpack("<I", f.read(4))
dims = struct.unpack(f"<{n_dims}Q", f.read(8 * n_dims))
(ttype,) = struct.unpack("<I", f.read(4))
f.read(8) # offset
size = _GGML_TYPE_SIZES.get(ttype)
if size is None:
raise ValueError(f"unknown ggml tensor type {ttype} in {path}")
block_bytes, block_elems = size
elems = 1
for d in dims:
elems *= d
nbytes = (elems // block_elems) * block_bytes
tensor_bytes += nbytes
if name == "token_embd.weight":
embd_bytes = nbytes
return GGUFHeader(path=str(path), version=version, metadata=metadata,
n_tensors=n_tensors, tensor_bytes=tensor_bytes,
embd_table_bytes=embd_bytes)
+143
View File
@@ -0,0 +1,143 @@
"""In-session context growth for the managed llama.cpp runtime.
The live half of the window ladder (context_policy.growth_decision): when a
session reaches the edge of its granted window, Hermes grows the window
toward the model's native max INSTEAD of compressing. Compression becomes
what the design says it is — the move of last resort, once the window is at
native (or the speed floor / physics say stop).
Mechanism: growth is re-prefill. A per-model window
override is persisted, presets regenerate with the bigger window, the
supervised server bounces, and the next request autoloads the model at the
new window and re-prefills the conversation. Nothing about the Hermes
conversation mutates — no prompt-cache or role-alternation risk; the whole
operation is server-side.
Scope guard: only a server THIS process supervises grows. Detected external
servers and other-process supervisors keep their own policies.
"""
from __future__ import annotations
import json
import logging
logger = logging.getLogger(__name__)
def window_overrides_path():
from hermes_cli.local_runtime.binaries import runtimes_root
return runtimes_root() / "window_overrides.json"
def load_window_overrides() -> dict:
"""model_id -> granted window (int). Empty on any read problem."""
try:
with open(window_overrides_path(), encoding="utf-8") as fh:
data = json.load(fh)
return {str(k): int(v) for k, v in data.items()}
except Exception: # noqa: BLE001
return {}
def save_window_override(model_id: str, window: int) -> None:
overrides = load_window_overrides()
overrides[model_id] = int(window)
path = window_overrides_path()
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(overrides, indent=1), encoding="utf-8")
def clear_window_override(model_id: str) -> None:
"""Drop a model's growth state (delete/re-download paths)."""
overrides = load_window_overrides()
if model_id in overrides:
del overrides[model_id]
window_overrides_path().write_text(
json.dumps(overrides, indent=1), encoding="utf-8")
def is_managed_endpoint(base_url: str) -> bool:
"""True when base_url is the server this process's state file points at."""
try:
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is None:
return False
return (base_url or "").rstrip("/") == str(
state.get("base_url", "")).rstrip("/")
except Exception: # noqa: BLE001
return False
def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
current_window: int,
measured_decode_tok_s: float | None = None) -> int | None:
"""One growth evaluation + execution. Returns the NEW window when the
ladder granted a bigger one, else None (hold / compress / not ours).
The caller sits at a request boundary by construction (the pre-API
compression gate), so re-prefill growth is safe at any call: the next
request rebuilds server state from scratch in the larger window —
nothing rewinds.
"""
from hermes_cli.local_runtime.bootstrap import (
get_supervisor,
refresh_local_runtime,
staged_models,
)
from hermes_cli.local_runtime.context_policy import growth_decision
from hermes_cli.local_runtime.estimator import profile_from_gguf
from hermes_cli.local_runtime.gguf import read_gguf_header
from hermes_cli.local_runtime.hardware import probe_budget
sup = get_supervisor()
if sup is None or not is_managed_endpoint(base_url):
return None
gguf = next((p for p in staged_models()
if p.stem.startswith(model_id) or model_id in p.stem), None)
if gguf is None:
return None
try:
profile = profile_from_gguf(read_gguf_header(gguf))
except (ValueError, OSError) as exc:
logger.debug("growth skip %s: unreadable gguf (%s)", model_id, exc)
return None
try:
server_idle = sup.is_idle(model_id)
except Exception: # noqa: BLE001
server_idle = False
decision = growth_decision(
# Capacity budget, not live-free: growth executes via a server
# bounce, so the grown instance loads onto a freed card. Live-free
# here is distorted by the very model being grown — it reads its
# own residency as unavailable and vetoes rungs that fit.
profile, probe_budget(planning=True),
current_window=current_window,
session_tokens=session_tokens,
measured_decode_tok_s=measured_decode_tok_s,
server_idle=server_idle,
# The caller IS the occupancy signal: this runs from the agent's
# compression gate, which fired on its own threshold. Two
# separately-derived edges must not deadlock into
# compress-before-grow.
occupancy_confirmed=True,
)
if decision.action != "grow" or not decision.next_window:
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
return None
logger.info("context growth %s: %s", model_id, decision.reason)
save_window_override(model_id, decision.next_window)
if not refresh_local_runtime():
# The override still lands at the next boot; report no growth NOW
# so the caller compresses instead of overflowing a stale window.
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
return None
return decision.next_window
+379
View File
@@ -0,0 +1,379 @@
"""Live hardware budget probe.
Budget-source rule: discrete cards may trust the device query (measured
honest within rounding); unified-memory devices must budget from OS free
physical memory minus headroom — their device queries have been observed
off by 3x in both directions. The probe classifies the device and
constructs the right HardwareBudget for the estimator.
Vendor probe quirk (WDDM carve-out): on unified-memory NVIDIA devices
under Windows, nvidia-smi answers from the legacy dedicated-VRAM
carve-out — a fraction of the pool the CUDA allocator actually
addresses uniformly at full bandwidth. The CUDA driver API is
the tiebreaker: cuDeviceGetAttribute(INTEGRATED) is the vendor's own
declaration and always wins — 1 budgets unified, 0 stays discrete no
matter what any other number says. Only when the driver API is
unreachable does the engine's --list-devices view apply, and then only
behind two independent conditions no discrete card can meet.
Every probe here must work under a stripped PATH — gateway and service
sessions don't inherit the interactive environment. nvcuda/libcuda load
through the system loader (PATH plays no part), so classification never
depends on PATH; nvidia-smi resolves through an explicit candidate
ladder (PATH first, then the driver's known install locations) and its
absence only softens the live number, never the verdict.
"""
from __future__ import annotations
import logging
import os
import re
import shutil
import subprocess
import sys
import time
from pathlib import Path
from hermes_cli.local_runtime.estimator import HardwareBudget
logger = logging.getLogger(__name__)
_GIB = 1 << 30
# Reserve carved off the card before any grant: the desktop's own
# co-residents (compositor, browser, Electron) measure ~2-2.5 GiB on a
# working machine, and a window granted into that space demotes silently
# under WDDM. 7% covers big cards; the 2 GiB floor is what the margin's
# old 512 MiB floor failed to cover in practice (a 221K grant measured
# 31.9/32.6 GiB with the desktop running — 'fits' by the math, demoted
# in reality). Small cards give up window to this; spill mode is their
# path to big models regardless.
_MARGIN_FLOOR = 2 << 30
_MARGIN_FRACTION = 0.09
# UMA headroom: on unified-memory machines (Apple Silicon, unified-memory
# NVIDIA) the model shares physical memory with the OS and every app, so
# budget from RAM minus this fraction.
_UMA_HEADROOM_FRACTION = 0.20
# Engine-fallback gates for the unified-pool quirk — BOTH must hold, and
# no discrete card can meet either: (1) the allocator's pool exceeds the
# smi report by well past rounding/ECC slack (discrete cards agree within
# ~2%; carve-out disagreement runs to whole multiples), and (2) the pool is
# system-RAM-sized — a workstation card in a RAM-matched box fails (1)
# because its smi and allocator AGREE, and a big discrete card in a
# bigger box fails (2). The driver's INTEGRATED attribute, when
# readable, bypasses both gates in whichever direction it points.
_POOL_DISAGREEMENT_FACTOR = 1.5
_POOL_RAM_FRACTION = 0.75
# cuDeviceGetAttribute enum: device is integrated with host memory.
_CU_DEVICE_ATTRIBUTE_INTEGRATED = 18
# One probe per process once a device answers (silicon doesn't change);
# a miss retries after this long so a runtime installed mid-session gets
# picked up by the engine fallback.
_POOL_NEGATIVE_TTL_S = 60.0
_pool_probe_cache: tuple[float, "tuple[int, bool | None] | None"] | None = None
# ' CUDA0: NVIDIA Example Device (1234-core Example GPU) (46464 MiB, 46284 MiB free)'
# — greedy .* pins the LAST parenthesized group, so device names carrying
# their own parentheses parse correctly.
_DEVICE_LINE_RE = re.compile(r"CUDA\d+:.*\((\d+)\s*MiB,\s*\d+\s*MiB free\)\s*$")
def _ram_bytes() -> tuple[int, int]:
"""(total, available) physical memory, cross-platform stdlib."""
try:
import ctypes
class MEMORYSTATUSEX(ctypes.Structure):
_fields_ = [("dwLength", ctypes.c_ulong),
("dwMemoryLoad", ctypes.c_ulong),
("ullTotalPhys", ctypes.c_ulonglong),
("ullAvailPhys", ctypes.c_ulonglong),
("ullTotalPageFile", ctypes.c_ulonglong),
("ullAvailPageFile", ctypes.c_ulonglong),
("ullTotalVirtual", ctypes.c_ulonglong),
("ullAvailVirtual", ctypes.c_ulonglong),
("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
stat = MEMORYSTATUSEX()
stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat))
return stat.ullTotalPhys, stat.ullAvailPhys
except (AttributeError, OSError):
pass
if sys.platform == "darwin":
# macOS getconf has no _PHYS_PAGES/_AVPHYS_PAGES (exit 64, "no such
# configuration parameter") — the POSIX branch below returns (0, 0)
# and every model reads unavailable. sysctl is the platform truth.
try:
total = int(subprocess.run(
["/usr/sbin/sysctl", "-n", "hw.memsize"],
capture_output=True, text=True, timeout=5).stdout.strip() or 0)
if total <= 0:
return 0, 0
avail = total // 2 # conservative fallback
try:
out = subprocess.run(["/usr/bin/vm_stat"], capture_output=True,
text=True, timeout=5).stdout
page_m = re.search(r"page size of (\d+)", out)
page = int(page_m.group(1)) if page_m else 16384
pages = 0
# free + inactive + purgeable ≈ reclaimable-on-demand; the
# speculative pool is dropped by the OS under pressure too.
for key in ("Pages free", "Pages inactive", "Pages purgeable",
"Pages speculative"):
m = re.search(rf"{key}:\s+(\d+)\.", out)
if m:
pages += int(m.group(1))
if pages > 0:
avail = pages * page
except (OSError, ValueError):
pass
return total, avail
except (OSError, ValueError):
return 0, 0
# POSIX
try:
page = int(subprocess.run(["getconf", "PAGE_SIZE"], capture_output=True,
text=True, timeout=5).stdout or 4096)
total = int(subprocess.run(["getconf", "_PHYS_PAGES"], capture_output=True,
text=True, timeout=5).stdout or 0) * page
avail = total // 2 # conservative when _AVPHYS is unavailable
try:
avail = int(subprocess.run(["getconf", "_AVPHYS_PAGES"],
capture_output=True, text=True,
timeout=5).stdout or 0) * page or avail
except (OSError, ValueError):
pass
return total, avail
except (OSError, ValueError):
return 0, 0
# nvidia-smi lives at a fixed path under the driver install; PATH presence
# varies by session type (services and gateways often run with a minimal
# environment) and by driver generation (legacy NVSMI dir was never on
# PATH). Resolution result is cached: the driver doesn't move mid-process.
_smi_path_cache: "tuple[str | None] | None" = None
def _nvidia_smi_path() -> str | None:
"""Absolute path to nvidia-smi, or None. PATH first (respects user
overrides), then the driver's known install locations on Windows;
on Linux/WSL the PATH lookup is the whole ladder."""
global _smi_path_cache
if _smi_path_cache is not None:
return _smi_path_cache[0]
found = shutil.which("nvidia-smi")
if found is None and os.name == "nt":
windir = os.environ.get("SystemRoot", r"C:\Windows")
for candidate in (
# DCH drivers (every modern install) place it in System32.
Path(windir) / "System32" / "nvidia-smi.exe",
# Legacy standalone drivers used NVSMI, never on PATH.
Path(os.environ.get("ProgramFiles", r"C:\Program Files"))
/ "NVIDIA Corporation" / "NVSMI" / "nvidia-smi.exe",
):
if candidate.exists():
found = str(candidate)
break
_smi_path_cache = (found,)
return found
def _nvidia_vram() -> tuple[int, int] | None:
"""(total, free) MiB->bytes from nvidia-smi, or None."""
exe = _nvidia_smi_path()
if exe is None:
return None
try:
out = subprocess.run(
[exe, "--query-gpu=memory.total,memory.free",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
if out.returncode != 0 or not out.stdout.strip():
return None
total_mib, free_mib = (int(x) for x in out.stdout.strip().splitlines()[0].split(","))
return total_mib << 20, free_mib << 20
except (OSError, ValueError, subprocess.TimeoutExpired):
return None
def _cuda_driver_pool() -> "tuple[int, bool | None] | None":
"""(allocator_total_bytes, integrated_or_None) from the CUDA driver
API, or None when unreachable. ctypes against the driver's own DLL/SO
— no toolkit, no subprocess, ~ms. INTEGRATED is the vendor's own
unified-memory declaration; total is the pool the allocator will
actually hand out (on carve-out devices, several times what
nvidia-smi reports)."""
import ctypes
for name in ("nvcuda.dll", "libcuda.so.1", "libcuda.so"):
try:
cuda = ctypes.CDLL(name)
break
except OSError:
continue
else:
return None
try:
if cuda.cuInit(0) != 0:
return None
dev = ctypes.c_int()
if cuda.cuDeviceGet(ctypes.byref(dev), 0) != 0:
return None
total = ctypes.c_size_t()
getter = getattr(cuda, "cuDeviceTotalMem_v2", None) or cuda.cuDeviceTotalMem
if getter(ctypes.byref(total), dev) != 0 or total.value <= 0:
return None
integrated: bool | None = None
attr = ctypes.c_int()
if cuda.cuDeviceGetAttribute(
ctypes.byref(attr), _CU_DEVICE_ATTRIBUTE_INTEGRATED, dev) == 0:
integrated = bool(attr.value)
return total.value, integrated
except (OSError, AttributeError):
return None
def _engine_device_pool() -> "tuple[int, bool | None] | None":
"""(engine_total_bytes, None) from the installed runtime's own
--list-devices, or None. The fallback truth source when the driver
API is unreachable: asks the exact binary that will do the
allocating. Carries no integrated verdict — callers must gate it."""
try:
from hermes_cli.local_runtime.binaries import (
installed_tags,
runtimes_root,
server_binary,
)
tags = installed_tags()
if not tags:
return None
tag_dir = runtimes_root() / tags[0]
backend_dirs = [d for d in tag_dir.iterdir() if d.is_dir()]
if not backend_dirs:
return None
exe = server_binary(backend_dirs[0])
out = subprocess.run([str(exe), "--list-devices"], capture_output=True,
text=True, timeout=30, cwd=str(exe.parent))
if out.returncode != 0:
return None
for line in (out.stdout + out.stderr).splitlines():
m = _DEVICE_LINE_RE.search(line)
if m:
return int(m.group(1)) << 20, None
return None
except Exception: # noqa: BLE001 — a probe miss must never block budgeting
return None
def _device_pool_view() -> "tuple[int, bool | None] | None":
"""Best available allocator-side view, cached: a hit is permanent for
the process, a miss retries after a short TTL (the engine binary can
appear mid-session via a pane install)."""
global _pool_probe_cache
now = time.monotonic()
if _pool_probe_cache is not None:
stamp, view = _pool_probe_cache
if view is not None or now - stamp < _POOL_NEGATIVE_TTL_S:
return view
view = _cuda_driver_pool() or _engine_device_pool()
_pool_probe_cache = (now, view)
return view
def _unified_pool_bytes(smi_total: int, ram_total: int) -> int | None:
"""The real pool size when this NVIDIA device is unified memory behind
a WDDM carve-out, else None (trust nvidia-smi as ever).
The driver's INTEGRATED attribute decides when readable — in BOTH
directions (0 pins discrete even if the numbers look weird; a driver
that declares integrated is believed even at modest pool sizes). Only
an attribute-less view (engine fallback) needs the two numeric gates;
both must hold and no discrete card meets either.
"""
view = _device_pool_view()
if view is None:
return None
pool, integrated = view
if integrated is False:
return None
if integrated is True:
return pool
if (smi_total > 0 and pool >= int(smi_total * _POOL_DISAGREEMENT_FACTOR)
and ram_total > 0 and pool >= int(ram_total * _POOL_RAM_FRACTION)):
return pool
return None
def probe_budget(*, planning: bool = False) -> HardwareBudget:
"""Construct the budget per the source rules above.
``planning=False`` (default): LIVE budget — free VRAM right now. The
right input for launch-time fit decisions and growth re-grants.
``planning=True``: CAPACITY budget — what this machine can run once
the runtime manages placement (total device memory minus the margin).
The right input for catalog pricing and quant selection: pricing
against live-free while a model is already loaded made every row read
'larger than your GPU memory' and degraded quant picks to Q2 on a
32 GiB card. The managed server
unloads/relaunches models itself, so at load time the capacity is
genuinely available.
"""
ram_total, ram_avail = _ram_bytes()
vram = _nvidia_vram()
# Unified-memory NVIDIA: the CUDA allocator pool is the real
# capacity. Classification comes from the driver API/engine — it
# must not require nvidia-smi (stripped-PATH sessions lose smi but
# nvcuda loads via the system loader regardless). Crossing the
# carve-out costs nothing (effective bandwidth is flat through the
# boundary; smi's used/total merely saturate at it) — the carve-out
# is an OS accounting knob, not a GPU limit. Deliberately NOT
# clamped to OS RAM: carved-out memory is invisible to
# GlobalMemoryStatusEx (the OS reports correspondingly less total
# RAM), so a RAM clamp would throw away exactly the carved capacity.
unified = _unified_pool_bytes(vram[0] if vram else 0, ram_total)
if unified is not None:
logger.info(
"unified-memory NVIDIA device: allocator pool %.1f GiB "
"(nvidia-smi carve-out: %s); budgeting from the pool",
unified / _GIB,
f"{vram[0] / _GIB:.1f} GiB" if vram else "unavailable")
if planning:
base = unified
else:
# Live: dedicated-free plus what the OS can still give. smi's
# free saturates at the carve-out so this under-counts a bit —
# the safe direction (the pool edge is a measured soft cliff:
# decode collapses ~3.5x when concurrent demand hits it).
# Without smi, OS-available alone is the honest floor.
live = (vram[1] + ram_avail) if vram else ram_avail
base = min(unified, live)
usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
return HardwareBudget(usable_vram_bytes=usable,
total_device_bytes=unified,
ram_available_bytes=0, uma=True)
if vram is None:
# No NVIDIA device visible: Metal/Vulkan/CPU paths budget from RAM
# as UMA (Apple Silicon) — conservative for discrete AMD until a
# vendor probe lands (E3 hardware).
base = ram_total if planning else ram_avail
usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
return HardwareBudget(usable_vram_bytes=usable,
total_device_bytes=ram_total,
ram_available_bytes=0, uma=True)
total, free = vram
margin = max(_MARGIN_FLOOR, int(total * _MARGIN_FRACTION))
base = total if planning else free
return HardwareBudget(usable_vram_bytes=max(0, base - margin),
total_device_bytes=total,
ram_available_bytes=ram_avail if not planning else ram_total,
uma=False)
+161
View File
@@ -0,0 +1,161 @@
"""Browse Hugging Face for GGUF models the user can run.
The curated catalog is the front page; this module is the firehose behind
it — day-0 models not yet in the catalog, community quants,
anything. Three rules keep it safe and honest:
1. Acquisition only. Nothing here serves a model: a browsed download
lands in the machine-scoped models dir and from that moment the
normal machinery owns it — staleness bounce, preset generation from
the real GGUF header, fit policy, placement pills.
2. The fit verdict shown BEFORE download is a rough cut priced from file
size alone (weights dominate; KV/overhead use conservative fill-ins).
After download the GGUF header is the authority, as everywhere.
3. HF is queried directly with short timeouts and a small in-process
cache. No third-party proxy service; if HF rate limits ever bite at
fleet scale, revisit with a caching proxy then.
"""
from __future__ import annotations
import json
import logging
import re
import time
import urllib.parse
import urllib.request
from dataclasses import dataclass, field
logger = logging.getLogger(__name__)
_HF = "https://huggingface.co"
_TIMEOUT_S = 15
# Rough-fit fill-ins for pre-download pricing: a mid-size model's 64K-floor
# KV plus runtime overhead. Deliberately round numbers — the verdict bands
# are coarse (fits GPU / needs RAM / too big), not window grants.
_ROUGH_KV_AND_OVERHEAD = 4 << 30
# Tiny TTL cache: the pane fires a search per keystroke pause and re-opens
# repos the user flips between. Process-local, size-capped, no invalidation
# subtleties — upstream truth changes slowly at this granularity.
_CACHE: dict[str, tuple[float, object]] = {}
_CACHE_TTL_S = 300
_CACHE_MAX = 128
def _get_json(url: str) -> object:
now = time.monotonic()
hit = _CACHE.get(url)
if hit and now - hit[0] < _CACHE_TTL_S:
return hit[1]
req = urllib.request.Request(url, headers={"User-Agent": "hermes-local-models"})
with urllib.request.urlopen(req, timeout=_TIMEOUT_S) as r:
data = json.load(r)
if len(_CACHE) >= _CACHE_MAX:
_CACHE.pop(min(_CACHE, key=lambda k: _CACHE[k][0]))
_CACHE[url] = (now, data)
return data
@dataclass(frozen=True)
class HFModelHit:
repo: str # e.g. "unsloth/Qwen3.8-27B-GGUF"
downloads: int
likes: int
updated: str # ISO date from HF
gated: bool
@dataclass(frozen=True)
class HFFileGroup:
"""One downloadable quant: a single GGUF or all parts of a split one."""
label: str # e.g. "Q4_K_M" or the file stem
paths: tuple[str, ...] # repo-relative, split parts in order
total_bytes: int
fit: str = "unknown" # fits-gpu | needs-ram | too-big | unknown
_QUANT_RE = re.compile(
r"(?:IQ|Q)\d[_A-Z0-9]*|F16|BF16|F32", re.IGNORECASE)
_SPLIT_RE = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$", re.IGNORECASE)
def search_models(query: str, limit: int = 20) -> list[HFModelHit]:
"""Full-text search over HF models that ship GGUF files, most
downloaded first (the closest public signal to 'trending')."""
q = urllib.parse.quote(query.strip())
url = (f"{_HF}/api/models?search={q}&filter=gguf&sort=downloads"
f"&direction=-1&limit={max(1, min(int(limit), 50))}")
out: list[HFModelHit] = []
for m in _get_json(url):
out.append(HFModelHit(
repo=str(m.get("id", "")),
downloads=int(m.get("downloads") or 0),
likes=int(m.get("likes") or 0),
updated=str(m.get("lastModified") or ""),
gated=bool(m.get("gated")),
))
return out
def _quant_label(filename: str) -> str:
m = _QUANT_RE.search(filename)
return m.group(0).upper() if m else filename
def repo_files(repo: str) -> list[HFFileGroup]:
"""The servable GGUFs in a repo, grouped: split parts collapse into one
entry (first part is what llama.cpp loads), mmproj/draft companions are
excluded (they aren't standalone models). Largest quant first."""
url = f"{_HF}/api/models/{urllib.parse.quote(repo)}/tree/main?recursive=true"
files = _get_json(url)
singles: list[tuple[str, int]] = []
splits: dict[str, list[tuple[int, str, int]]] = {}
for f in files:
path = str(f.get("path", ""))
if not path.lower().endswith(".gguf"):
continue
name = path.rsplit("/", 1)[-1].lower()
if name.startswith("mmproj") or name.startswith("dspark") or "draft" in name:
continue
size = int(f.get("size") or 0)
m = _SPLIT_RE.search(path)
if m:
stem = path[: m.start()]
splits.setdefault(stem, []).append((int(m.group(1)), path, size))
else:
singles.append((path, size))
groups: list[HFFileGroup] = []
for path, size in singles:
groups.append(HFFileGroup(label=_quant_label(path), paths=(path,),
total_bytes=size))
for stem, parts in splits.items():
parts.sort()
groups.append(HFFileGroup(
label=_quant_label(stem),
paths=tuple(p for _, p, _ in parts),
total_bytes=sum(s for _, _, s in parts)))
groups.sort(key=lambda g: g.total_bytes, reverse=True)
return groups
def rough_fit(total_bytes: int, budget) -> str:
"""Coarse pre-download verdict from file size alone. The GGUF header
refines this after download; bands match the catalog pills' language.
File size ≈ in-memory weights for GGUF (mmap'd as-is)."""
need = total_bytes + _ROUGH_KV_AND_OVERHEAD
if need <= budget.usable_vram_bytes:
return "fits-gpu"
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
return "needs-ram"
return "too-big"
def priced_repo_files(repo: str, budget) -> list[HFFileGroup]:
from dataclasses import replace
return [replace(g, fit=rough_fit(g.total_bytes, budget))
for g in repo_files(repo)]
+198
View File
@@ -0,0 +1,198 @@
"""Live model-load progress from the managed llama-server router.
llama-server's child processes emit per-tensor load progress
({stages, current, value}, throttled upstream to ~200ms) which the
router relays ONLY over its /models/sse stream — GET /models carries
just the coarse status string. This module owns one lazy background
watcher on that stream and keeps an in-memory snapshot other code can
poll cheaply:
get_loading_progress() -> {model_id: {"stage", "value", "percent"}}
"percent" is a composite across stages so a bar doesn't sprint 0->100
once per stage: the text model dominates load time (its weights dwarf
the mmproj/spec extras), so it gets the lion's share of the range and
the extras split the remainder.
The watcher starts on first call, reconnects with backoff (the router
bounces on model download/eject), and never raises into callers — no
router, no state file, or no SSE support (older engines) all read as
"nothing loading". Safe from any process on the machine: the endpoint
comes from the supervisor's machine-scoped state file.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.request
logger = logging.getLogger(__name__)
_TEXT_STAGE_SHARE = 0.85 # composite range share for the text model
_RECONNECT_DELAY_S = 3.0
_STALE_ENTRY_TTL_S = 120.0 # a loading entry with no events this long is dead
_lock = threading.Lock()
_watcher: threading.Thread | None = None
_snapshot: dict[str, dict] = {}
def _composite_percent(stages: list[str], current: str, value: float) -> int:
"""Map (stage, in-stage value) onto one 0-100 range, text-heavy."""
if not stages or current not in stages or len(stages) == 1:
return max(0, min(100, round(value * 100)))
extras = [s for s in stages if s != "text_model"]
extra_share = (1.0 - _TEXT_STAGE_SHARE) / len(extras) if extras else 0.0
offset = 0.0
for stage in stages:
share = _TEXT_STAGE_SHARE if stage == "text_model" else extra_share
if stage == current:
return max(0, min(100, round((offset + share * value) * 100)))
offset += share
return max(0, min(100, round(value * 100)))
def _endpoint() -> "tuple[str, str] | None":
"""(base_root, api_key) of the managed router, or None.
Resolved through the endpoint module's ownership-guarded reader, not
a raw state-file read: on the shared stable port, a foreign install's
server answers /health for anyone, and a raw read would attach this
watcher to someone else's SSE stream (or spin on 401s against it).
The guard's dead-pid check is the ownership proof."""
try:
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is None:
return None
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
return (base, str(state.get("api_key", ""))) if base else None
except Exception: # noqa: BLE001
return None
def _apply_event(model: str, event: str, data: dict) -> None:
with _lock:
status = str(data.get("status", ""))
if event in ("status_change", "model_status") and status == "loading":
progress = data.get("progress") or {}
stages = [str(s) for s in (progress.get("stages") or [])]
current = str(progress.get("current", ""))
value = progress.get("value")
entry = _snapshot.setdefault(model, {"stage": "", "value": 0.0,
"percent": 0, "ts": 0.0})
entry["ts"] = time.monotonic()
if current and isinstance(value, (int, float)):
entry["stage"] = current
entry["value"] = float(value)
entry["percent"] = _composite_percent(stages, current, float(value))
elif event in ("status_change", "model_status", "model_remove"):
# Any terminal status (loaded/unloaded/failed) ends the load.
if status != "loading":
_snapshot.pop(model, None)
def _watch() -> None:
while True:
endpoint = _endpoint()
if endpoint is None:
with _lock:
_snapshot.clear()
time.sleep(_RECONNECT_DELAY_S)
continue
base, key = endpoint
try:
req = urllib.request.Request(
f"{base}/models/sse",
headers={"Authorization": f"Bearer {key}",
"Accept": "text/event-stream"})
with urllib.request.urlopen(req, timeout=60) as r:
buf = b""
while True:
chunk = r.read1(4096) if hasattr(r, "read1") else r.read(4096)
if not chunk:
break
buf += chunk
while b"\n" in buf:
line, buf = buf.split(b"\n", 1)
text = line.decode("utf-8", "replace").strip()
if not text.startswith("data:"):
continue
try:
msg = json.loads(text[5:].strip())
_apply_event(str(msg.get("model", "")),
str(msg.get("event", "")),
msg.get("data") or {})
except (json.JSONDecodeError, TypeError):
continue
except Exception as exc: # noqa: BLE001 — watcher must never die loud
logger.debug("load-progress SSE reconnecting: %s", exc)
# Stream ended (router bounce, timeout, error): loading entries from
# the dead connection are unverifiable — drop rather than freeze.
with _lock:
_snapshot.clear()
time.sleep(_RECONNECT_DELAY_S)
def _ensure_watcher() -> None:
global _watcher
with _lock:
if _watcher is None or not _watcher.is_alive():
_watcher = threading.Thread(target=_watch, daemon=True,
name="llamacpp-load-progress")
_watcher.start()
def get_loading_progress() -> dict[str, dict]:
"""{model_id: {"stage", "value", "percent"}} for models loading right
now. Empty when nothing is loading (or nothing is knowable)."""
_ensure_watcher()
now = time.monotonic()
with _lock:
return {m: {"stage": e["stage"], "value": e["value"],
"percent": e["percent"]}
for m, e in _snapshot.items()
if now - e["ts"] < _STALE_ENTRY_TTL_S}
def get_prefill_progress(model: str) -> "dict | None":
"""{"processed": tokens} while the managed server is prompt-processing
for ``model``, or None (idle, decoding, unreachable, or foreign server).
llama-server's /slots reports ``n_prompt_tokens_processed`` climbing in
real time during prefill, but exposes no total — callers supply their
own denominator (the request's estimated token count). Busiest
processing slot wins when several are active: a parallel small request
(title generation) freezes its counter during decode while a live
prefill keeps climbing past it. One authenticated HTTP call per poll;
every failure reads as "no prefill" — this is garnish, never load-
bearing.
"""
ep = _endpoint()
if ep is None:
return None
base, key = ep
try:
from urllib.parse import quote
req = urllib.request.Request(
f"{base}/slots?model={quote(model)}",
headers={"Authorization": f"Bearer {key}"})
with urllib.request.urlopen(req, timeout=2) as r:
slots = json.loads(r.read())
except Exception: # noqa: BLE001
return None
best = 0
for slot in slots if isinstance(slots, list) else []:
if not slot.get("is_processing"):
continue
try:
processed = int(slot.get("n_prompt_tokens_processed") or 0)
except (TypeError, ValueError):
continue
best = max(best, processed)
return {"processed": best} if best > 0 else None
+265
View File
@@ -0,0 +1,265 @@
"""Per-model preset generation (--models-preset INI) — the router-side
carrier for context-policy launch decisions.
The INI shape is what the router itself generates per child: a
[model-id] section whose keys are long-form
llama-server flag names without the leading dashes.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from pathlib import Path
from hermes_cli.local_runtime.context_policy import (
RUNTIME_OVERHEAD_BYTES,
WindowDecision,
initial_window,
launch_args,
ub_logits_bytes,
)
from hermes_cli.local_runtime.estimator import (
HardwareBudget,
PhysicsRefusal,
profile_from_gguf,
)
from hermes_cli.local_runtime.gguf import read_gguf_header
logger = logging.getLogger(__name__)
# args list -> INI keys. Flags the policy owns; everything else stays out
# of the preset (recipe sampling defaults merge in a later pass).
_FLAG_TO_KEY = {
"-c": "ctx-size",
"-b": "batch-size",
"-ub": "ubatch-size",
"-ctk": "cache-type-k",
"-ctv": "cache-type-v",
"-fa": "flash-attn",
"-ot": "override-tensor",
"--spec-type": "spec-type",
"--spec-draft-n-max": "spec-draft-n-max",
}
@dataclass
class PresetEntry:
model_id: str
window: int
spilled: bool
refusal: str | None = None
keys: dict[str, str] | None = None
def _args_to_keys(args: list[str]) -> dict[str, str]:
keys: dict[str, str] = {}
i = 0
while i < len(args):
flag = args[i]
key = _FLAG_TO_KEY.get(flag)
if key is None:
i += 1
continue
keys[key] = args[i + 1]
i += 2
return keys
def generate_presets(models_dir: Path, budget: HardwareBudget,
preset_path: Path,
mtp_capable: set[str] | None = None) -> list[PresetEntry]:
"""Walk the staged models, run the launch decision per model, and
write one INI. Refused models get no section (the router simply won't
have policy for them; the picker surfaces the refusal + smaller-quant
suggestion from the returned entries).
Catalog-declared companions merge in here: sampling defaults (policy
keys always win), the vision projector when present, and a spec-decode
draft model iff the decision spilled — the rule: speculative
decode is a spill amplifier, so a resident draft accelerates a spilled
main model; a zero-spill model doesn't pay the draft's memory."""
from hermes_cli.local_runtime.bootstrap import assets_dir
from hermes_cli.local_runtime.catalog import find_entry_for_model
entries: list[PresetEntry] = []
sections: list[str] = []
for gguf in _staged_in(models_dir):
model_id = _strip_part(gguf.stem)
try:
header = read_gguf_header(gguf)
profile = profile_from_gguf(header)
except (ValueError, OSError) as exc:
logger.warning("preset skip %s: %s", gguf.name, exc)
continue
# Overhead beyond weights+KV: runtime buffers, the vision projector
# when this model ships one, and the logits buffers of whichever
# microbatch/MTP posture launch_args will choose — flag and price
# decided together, from the same facts.
hit = find_entry_for_model(model_id)
entry = hit[0] if hit is not None else None
is_mtp = (entry.mtp if entry is not None
else model_id in (mtp_capable or set()))
if is_mtp and profile.kv_scale == 1.0:
# Header-derived profiles don't know about MTP's draft
# context; apply the calibrated KV multiplier here so the
# launch fit prices what the server will actually allocate.
import dataclasses
profile = dataclasses.replace(profile, kv_scale=1.2)
mmproj_bytes = 0
if entry is not None and entry.mmproj is not None:
mmproj_path = assets_dir() / entry.mmproj.local_name
if mmproj_path.exists():
mmproj_bytes = entry.mmproj.size_bytes
# MTP posture ladder — window first, prefill second: price the
# launch under both postures and keep whichever grants the larger
# window (the stacked posture's bigger compute buffer buys ~3x
# short-prompt prefill but costs ~2 GiB that would otherwise be
# window; measured at 256K the ub512 posture still prefills at
# 2.7K tok/s, so window wins ties only one way: never trade
# context away for prefill). Same window -> stacked.
mtp_prefill = False
logits_bytes = ub_logits_bytes(profile.n_vocab, mtp_capable=is_mtp)
if is_mtp:
stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True,
mtp_prefill=True)
stacked_probe = initial_window(
profile, budget,
overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
+ stacked_logits))
plain_probe = initial_window(
profile, budget,
overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
+ logits_bytes))
if (not isinstance(stacked_probe, PhysicsRefusal)
and not stacked_probe.spilled
and (isinstance(plain_probe, PhysicsRefusal)
or stacked_probe.window >= plain_probe.window)):
mtp_prefill = True
logits_bytes = stacked_logits
decision = initial_window(
profile, budget,
overhead_bytes=RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
if isinstance(decision, PhysicsRefusal):
entries.append(PresetEntry(model_id=model_id, window=0,
spilled=False, refusal=decision.message))
continue
# Session growth (growth.py): a persisted override lifts the launch
# window to where the ladder last grew it — capped at native, and
# only when physics still clears the bigger window on THIS boot's
# budget (a smaller-VRAM day re-fits honestly back down).
try:
from hermes_cli.local_runtime.estimator import ctx_bytes
from hermes_cli.local_runtime.growth import load_window_overrides
override = load_window_overrides().get(model_id)
native = profile.n_ctx_train or decision.window
if override and override > decision.window:
target = min(int(override), native)
kv = ctx_bytes(profile, target)
need = (profile.weights_bytes + kv
+ RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
spill = max(0, need - budget.usable_vram_bytes)
decision = WindowDecision(
window=target, spill_bytes=spill,
kv_on_gpu=kv <= budget.usable_vram_bytes,
reasons=[f"grown window restored ({target // 1024}K)"])
except Exception as exc: # noqa: BLE001 — overrides are advisory
logger.debug("window override skipped for %s: %s", model_id, exc)
# (entry and is_mtp resolved above, where the overhead was priced —
# the launch flags below MUST match that pricing.)
args = launch_args(profile, decision, mtp_capable=is_mtp,
mtp_draft_depth=(entry.mtp_draft_depth
if entry is not None else 3),
uma=budget.uma, mtp_prefill=mtp_prefill)
keys = _args_to_keys(args)
if entry is not None and is_mtp:
# Integrated-MTP targets sample on the backend, and so does
# the draft (pairing validated against the vendor's published
# llama.cpp recipes for these models).
keys["backend-sampling"] = "on"
keys["spec-draft-backend-sampling"] = "on"
# Sampling deference ladder, under the policy keys (policy wins
# on clash). The GGUF's own general.sampling.* metadata is the
# publisher's recommendation — it arrives with the file, updates
# with every re-upload, and covers models the catalog has never
# heard of. Catalog sampling applies only where the file is
# silent; a model carrying neither runs llama.cpp defaults.
for k, v in header.sampling_defaults.items():
keys.setdefault(k, v)
if entry is not None:
for k, v in (entry.sampling or {}).items():
keys.setdefault(k, v)
if entry.mmproj is not None:
mmproj_path = assets_dir() / entry.mmproj.local_name
if mmproj_path.exists():
keys["mmproj"] = str(mmproj_path)
if entry.draft is not None and decision.spilled:
draft_path = assets_dir() / entry.draft.local_name
if draft_path.exists():
keys["model-draft"] = str(draft_path)
keys["spec-type"] = "draft-dspark"
# Unsloth's measured cliff: acceptance 83% at 2-3
# drafts, collapses at 4.
keys["spec-draft-n-max"] = "3"
entries.append(PresetEntry(model_id=model_id, window=decision.window,
spilled=decision.spilled, keys=keys))
body = "\n".join(f"{k} = {v}" for k, v in keys.items())
sections.append(f"[{model_id}]\n{body}\n")
preset_path.parent.mkdir(parents=True, exist_ok=True)
preset_path.write_text("\n".join(sections), encoding="utf-8")
logger.info("wrote %d preset sections to %s", len(sections), preset_path)
return entries
def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEntry]:
"""The launch decisions the running server was actually given, read
back from the preset INI (the INI is the record — it's what spawned
the children). Missing/unparseable file returns {}."""
import configparser
if preset_path is None:
from hermes_cli.local_runtime.binaries import runtimes_root
preset_path = runtimes_root() / "presets.ini"
out: dict[str, PresetEntry] = {}
try:
parser = configparser.ConfigParser()
parser.read(preset_path, encoding="utf-8")
for section in parser.sections():
window = parser.getint(section, "ctx-size", fallback=0)
spilled = parser.has_option(section, "override-tensor")
out[section] = PresetEntry(model_id=section, window=window,
spilled=spilled)
except Exception as exc: # noqa: BLE001
logger.debug("preset read-back failed: %s", exc)
return out
def _strip_part(stem: str) -> str:
import re
return re.sub(r"-\d{5}-of-\d{5}$", "", stem)
def _staged_in(models_dir: Path) -> "list[Path]":
"""Servable models in an arbitrary directory (split first-parts only) —
the validation harness points at non-default dirs."""
import re
part = re.compile(r"-(\d{5})-of-\d{5}\.gguf$")
out = []
for p in sorted(models_dir.glob("*.gguf")):
m = part.search(p.name)
if m and m.group(1) != "00001":
continue
out.append(p)
return out
+500
View File
@@ -0,0 +1,500 @@
"""Supervision of one llama-server in router mode.
The router process is ours (restart with backoff on crash); router children
are its problem — child failures surface via GET /models exit_code, never
auto-retried here.
Readiness rules (each learned the hard way on real hardware):
- health-200 is NOT readiness; every readiness claim requires a touch
generation (temp-0, expected token, generous budget, reasoning_content
scanned).
- Always dial 127.0.0.1 — resolving localhost adds ~2s per request on
Windows via IPv6 fallback.
- /metrics is opt-in (--metrics) and carries no KV-usage metric;
idleness = requests_processing == 0 and no slot is_processing.
- The router's LRU eviction has no pin for the primary model: until an
upstream pin exists, keep_primary_loaded re-touches the primary after
any other model load.
"""
from __future__ import annotations
import json
import logging
import secrets
import socket
import subprocess
import threading
import time
import urllib.error
import urllib.request
from pathlib import Path
from hermes_cli.local_runtime.binaries import server_binary, runtimes_root
logger = logging.getLogger(__name__)
TOUCH_PROMPT = "Reply with exactly one word: the capital of France."
TOUCH_EXPECT = "paris"
_RESTART_BACKOFF_S = (1, 5, 15, 60)
def state_path() -> Path:
"""Endpoint state for other Hermes processes (provider resolution reads
this to route llamacpp-alias requests at the managed server)."""
return runtimes_root() / "server.json"
def _free_port() -> int:
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
return s.getsockname()[1]
# Default port for the managed server, chosen once and reused across
# restarts. Sessions persist the resolved base_url; an ephemeral port
# would strand every resumed session on a dead endpoint after each
# restart. Deliberately NOT 8080 so we never collide with a user's own
# llama-server/Ollama-adjacent stack.
_DEFAULT_PORT = 18434
def _stable_port() -> int:
"""The stable default port, falling back to an ephemeral one only when
something else already listens there (and it isn't a leftover managed
server, which stop() would have cleaned up)."""
try:
with socket.socket() as s:
s.bind(("127.0.0.1", _DEFAULT_PORT))
return _DEFAULT_PORT
except OSError:
logger.warning(
"port %d busy; managed llama-server falling back to an ephemeral "
"port — existing sessions may need a model re-pick", _DEFAULT_PORT)
return _free_port()
def _stable_api_key() -> str:
"""One key for the life of the install, persisted beside the runtimes.
Endpoint identity must survive restarts as a UNIT — sessions persist the
resolved base_url + api_key, so a per-boot key strands every resumed
session on HTTP 401 exactly the way a per-boot port would strand them
on connection errors. Rotating it buys nothing: the key exists to stop
other loopback processes free-riding, and it lives on the same disk as
the state file that would leak it. Delete the file to rotate manually.
"""
key_path = runtimes_root() / ".api_key"
try:
existing = key_path.read_text(encoding="utf-8").strip()
if len(existing) >= 16:
return existing
except OSError:
pass
key = secrets.token_urlsafe(24)
try:
key_path.parent.mkdir(parents=True, exist_ok=True)
key_path.write_text(key, encoding="utf-8")
except OSError as exc:
logger.warning("could not persist api key (%s); sessions will need "
"a re-pick after restart", exc)
return key
class LlamaServerSupervisor:
"""Own one llama-server router process for the life of a Hermes session.
Usage::
sup = LlamaServerSupervisor(install_dir, models_dir)
sup.start() # spawn + wait healthy
sup.ensure_model_ready(name) # load + touch-generate
... sup.base_url is the /v1 endpoint, sup.api_key its key ...
sup.stop()
"""
def __init__(self, install_dir: Path, models_dir: Path, *,
models_max: int = 4, port: int | None = None,
extra_args: list[str] | None = None,
log_path: Path | None = None,
preset_path: Path | None = None):
self.install_dir = Path(install_dir)
self.models_dir = Path(models_dir)
self.models_max = models_max
self.port = port or _stable_port()
self.api_key = _stable_api_key()
self.extra_args = list(extra_args or [])
self.log_path = log_path or (self.models_dir.parent / "logs" / "llama-server.log")
self.preset_path = preset_path
self.proc: subprocess.Popen | None = None
self.primary_model: str | None = None
self._restarts = 0
self._stopping = False
self._watchdog: threading.Thread | None = None
self._log_handle = None
self._idle_since: dict[str, float] = {}
# ── endpoints ────────────────────────────────────────────
@property
def base_url(self) -> str:
return f"http://127.0.0.1:{self.port}/v1"
def _url(self, route: str) -> str:
return f"http://127.0.0.1:{self.port}{route}"
def _request(self, route: str, body: dict | None = None, timeout_s: int = 30) -> dict:
req = urllib.request.Request(
self._url(route),
data=json.dumps(body).encode() if body is not None else None,
headers={"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}"},
)
with urllib.request.urlopen(req, timeout=timeout_s) as r:
raw = r.read()
return json.loads(raw) if raw else {}
# ── lifecycle ────────────────────────────────────────────
def _spawn(self) -> None:
exe = server_binary(self.install_dir)
cmd = [
str(exe),
"--host", "127.0.0.1",
"--port", str(self.port),
"--api-key", self.api_key,
"--models-dir", str(self.models_dir),
"--models-max", str(self.models_max),
# The residency contract at the layer that sees every message:
# a chat request to a staged-but-unloaded model loads it (slow
# first token) instead of failing with 'model not found' —
# without this flag, chat after an eject is a bare 400/404.
"--models-autoload",
"--metrics", # opt-in flag; supervisor telemetry needs it
"--slots", # /slots endpoint is also opt-in; is_idle reads it
"--no-webui",
"--jinja",
# Direct I/O on model load: bypasses the page cache, so a
# multi-GB load doesn't evict half the OS cache — measured
# faster loads on NVMe, and our router bounces (download/
# delete/activate) reload models often enough to care.
"-dio",
]
if self.preset_path and self.preset_path.exists():
cmd += ["--models-preset", str(self.preset_path)]
cmd += [
*self.extra_args,
]
self.log_path.parent.mkdir(parents=True, exist_ok=True)
if self._log_handle is not None:
# The crash-restart loop calls _spawn repeatedly; without
# closing the prior handle each restart leaks one fd.
try:
self._log_handle.close()
except Exception: # noqa: BLE001 — best-effort
pass
self._log_handle = open(self.log_path, "a", encoding="utf-8", errors="replace")
self._log_handle.write(f"\n# spawn: {cmd}\n")
self._log_handle.flush()
# list-args, never a shell: spaced paths (user homes) must survive.
self.proc = subprocess.Popen(cmd, stdout=self._log_handle,
stderr=subprocess.STDOUT, cwd=str(exe.parent))
logger.info("llama-server router spawned pid=%s port=%s", self.proc.pid, self.port)
# State goes down at SPAWN, not after health: endpoint resolution
# treats a live-pid-but-not-yet-healthy server as "starting" rather
# than "unconfigured", so a readiness probe racing the boot doesn't
# throw the app back to onboarding (observed on first restart test).
self._write_state()
def start(self, timeout_s: int = 120) -> None:
self._stopping = False
self._spawn()
self._wait_health(timeout_s)
self._write_state()
self._watchdog = threading.Thread(target=self._watch, daemon=True,
name="llamacpp-supervisor")
self._watchdog.start()
def _write_state(self) -> None:
path = state_path()
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps({
"base_url": self.base_url,
"api_key": self.api_key,
"pid": self.proc.pid if self.proc else None,
}), encoding="utf-8")
def _wait_health(self, timeout_s: int) -> None:
deadline = time.monotonic() + timeout_s
while time.monotonic() < deadline:
if self.proc and self.proc.poll() is not None:
raise RuntimeError(
f"llama-server exited rc={self.proc.returncode} during startup "
f"(log: {self.log_path})")
try:
with urllib.request.urlopen(self._url("/health"), timeout=3) as r:
if r.status == 200:
return
except (urllib.error.URLError, OSError, TimeoutError):
pass
time.sleep(1)
raise TimeoutError(f"llama-server not healthy after {timeout_s}s (log: {self.log_path})")
def _watch(self) -> None:
"""Restart the router (not its children) on crash, with backoff."""
while not self._stopping:
proc = self.proc
if proc is None:
return
rc = proc.poll()
if rc is None:
time.sleep(2)
continue
if self._stopping:
return
backoff = _RESTART_BACKOFF_S[min(self._restarts, len(_RESTART_BACKOFF_S) - 1)]
logger.warning("llama-server exited rc=%s; restart #%s in %ss",
rc, self._restarts + 1, backoff)
time.sleep(backoff)
self._restarts += 1
try:
self._reap_orphaned_children()
self._spawn()
self._wait_health(120)
if self.primary_model:
self.ensure_model_ready(self.primary_model)
except Exception as exc: # noqa: BLE001
logger.error("llama-server restart failed: %s", exc)
def stop(self) -> None:
self._stopping = True
state_path().unlink(missing_ok=True)
if self.proc and self.proc.poll() is None:
self._terminate_tree(self.proc)
if self._log_handle:
self._log_handle.close()
self._log_handle = None
@staticmethod
def _terminate_tree(proc: subprocess.Popen) -> None:
"""Terminate the router AND its model children.
The router spawns one child llama-server per loaded model, each
holding gigabytes of VRAM. Terminating only the router (on
Windows, TerminateProcess — no signal handlers, no cleanup pass)
orphans those children: the port goes quiet but the weights stay
resident, and the next spawn re-loads models alongside a ghost
still holding the memory. Enumerate children FIRST (the parent
must be alive to walk them), then terminate parent and children
together, escalating to kill for stragglers.
"""
children: list = []
try:
import psutil
children = psutil.Process(proc.pid).children(recursive=True)
except Exception: # noqa: BLE001 — no psutil view; still stop the router
children = []
proc.terminate()
for child in children:
try:
child.terminate()
except Exception: # noqa: BLE001
pass
try:
proc.wait(timeout=15)
except subprocess.TimeoutExpired:
proc.kill()
for child in children:
try:
if child.is_running():
child.kill()
except Exception: # noqa: BLE001
pass
def _reap_orphaned_children(self) -> None:
"""Kill model children orphaned by a router crash, before respawn.
A crashed router can't clean up its children, and a dead parent
can't be walked — so match by identity instead: any process
running OUR llama-server binary whose parent is gone is an
orphan of a previous router. Their VRAM must come back before
the new router loads models next to the ghosts. External
llama-servers (different binary path) never match.
"""
try:
import psutil
exe = str(server_binary(self.install_dir))
except Exception: # noqa: BLE001
return
for p in psutil.process_iter(["exe", "ppid"]):
try:
if p.info.get("exe") != exe:
continue
if self.proc is not None and p.pid == self.proc.pid:
continue
ppid = p.info.get("ppid") or 0
if ppid and psutil.pid_exists(ppid):
continue
logger.warning("reaping orphaned llama-server child pid=%s", p.pid)
p.kill()
except (psutil.NoSuchProcess, psutil.AccessDenied):
continue
# ── model management (router endpoints) ──────────────────
def models(self) -> dict:
"""{model_id: status_value} from GET /models."""
data = self._request("/models")
return {m["id"]: m.get("status", {}).get("value", "unknown")
for m in data.get("data", [])}
def model_failures(self) -> dict:
"""{model_id: exit_code} for children that died — surfaced to the
UI, never auto-retried (design: router children are its problem)."""
data = self._request("/models")
out = {}
for m in data.get("data", []):
status = m.get("status", {})
if status.get("value") == "failed" or status.get("exit_code"):
out[m["id"]] = status.get("exit_code")
return out
def load_model(self, model_id: str, timeout_s: int = 600) -> None:
self._request("/models/load", {"model": model_id}, timeout_s=timeout_s)
def unload_model(self, model_id: str) -> None:
"""Free the child's VRAM now. Route existence verified empirically
on b10290 (POST /models/unload; bogus name -> 400 'model is not
found'). Momentary action: never touches primary_model — the
declaration is durable, an eject is not (residency design).
Settle before returning: for a few seconds after unload returns,
the router still routes to the dying child and answers chat with
500 'proxy error: Could not establish connection' (probed on
b10362). Waiting for the model to report unloaded means the next
message autoloads cleanly instead of racing the teardown.
"""
self._request("/models/unload", {"model": model_id}, timeout_s=120)
deadline = time.monotonic() + 15
while time.monotonic() < deadline:
try:
if self.models().get(model_id) not in ("loaded", "ready", "unloading"):
return
except Exception: # noqa: BLE001
return
time.sleep(0.3)
# ── idle residency (non-primary models) ──────────────────
# A model that has gone quiet gets its VRAM back after this long. A
# constant, not a knob: long enough that an active conversation never
# trips it, short enough that a wandered-off session frees ~20 GiB
# within the hour. No exemptions (residency v2): demand reloads
# anything the user comes back to.
IDLE_UNLOAD_S = 15 * 60
def sweep_idle(self, now: float | None = None) -> list[str]:
"""Unload models idle past IDLE_UNLOAD_S. Returns the model ids
unloaded. Idle means no busy slots and no queued work, tracked
per model across calls; a model seen busy resets its clock."""
now = time.monotonic() if now is None else now
unloaded: list[str] = []
try:
statuses = self.models()
except Exception: # noqa: BLE001
return unloaded
for model_id, status in statuses.items():
if status not in ("loaded", "ready"):
self._idle_since.pop(model_id, None)
continue
if not self.is_idle(model_id):
self._idle_since.pop(model_id, None)
continue
first_idle = self._idle_since.setdefault(model_id, now)
if now - first_idle >= self.IDLE_UNLOAD_S:
try:
self.unload_model(model_id)
self._idle_since.pop(model_id, None)
unloaded.append(model_id)
logger.info("idle-unloaded %s (idle %ds)", model_id,
int(now - first_idle))
except Exception as exc: # noqa: BLE001
logger.warning("idle unload of %s failed: %s", model_id, exc)
return unloaded
def touch_generate(self, model_id: str, timeout_s: int = 300) -> bool:
"""The readiness proof. Generous budget + reasoning_content scan —
small token budgets false-fail reasoning models, which spend their
first tokens thinking."""
try:
resp = self._request("/v1/chat/completions", {
"model": model_id,
"messages": [{"role": "user", "content": TOUCH_PROMPT}],
"max_tokens": 512, "temperature": 0,
}, timeout_s=timeout_s)
msg = resp["choices"][0]["message"]
blob = (msg.get("content") or "") + " " + (msg.get("reasoning_content") or "")
return TOUCH_EXPECT in blob.lower()
except Exception as exc: # noqa: BLE001
logger.warning("touch generation failed for %s: %s", model_id, exc)
return False
def ensure_model_ready(self, model_id: str, timeout_s: int = 600) -> bool:
"""Load if needed, then prove readiness with a touch generation."""
status = self.models().get(model_id)
if status is None:
raise KeyError(f"model {model_id} not present in models dir")
if status not in ("loaded", "ready"):
self.load_model(model_id, timeout_s=timeout_s)
return self.touch_generate(model_id)
def actual_n_ctx(self, model_id: str) -> int | None:
"""/props reconciliation: the granted window as the child reports
it — the compressor's budget and the picker's 'running at 87K of
262K' both read THIS value, never the request (design step 4)."""
try:
props = self._request(f"/props?model={model_id}")
return props.get("default_generation_settings", {}).get("n_ctx")
except Exception: # noqa: BLE001
return None
def keep_primary_loaded(self) -> None:
"""The router's LRU eviction has no pin, so after any other load
re-touch the primary to keep it most-recently-used. Best-effort
under bursty multi-model load — replaced when an upstream pin
exists."""
if self.primary_model and self.models().get(self.primary_model) in (
"loaded", "ready"):
self.touch_generate(self.primary_model, timeout_s=60)
# ── telemetry ────────────────────────────────────────────
def is_idle(self, model_id: str | None = None) -> bool:
"""No processing requests and no busy slots. Router quirk: /slots
and /metrics are per-child and require ?model= (bare calls 400),
and no KV-usage metric exists. With ``model_id`` checks that one
child; without, every loaded child."""
try:
if model_id is not None:
loaded = [model_id]
else:
loaded = [m for m, status in self.models().items()
if status in ("loaded", "ready")]
for mid in loaded:
slots = self._request(f"/slots?model={mid}")
if any(s.get("is_processing") for s in slots):
return False
req = urllib.request.Request(
self._url(f"/metrics?model={mid}"),
headers={"Authorization": f"Bearer {self.api_key}"})
with urllib.request.urlopen(req, timeout=10) as r:
text = r.read().decode()
for line in text.splitlines():
if line.startswith("llamacpp:requests_processing"):
if float(line.split()[-1]) != 0.0:
return False
return True
except Exception: # noqa: BLE001
return False
+6 -1
View File
@@ -8658,7 +8658,10 @@ def cmd_gui(args: argparse.Namespace):
if source_mode:
print("→ Launching Hermes Desktop from source build...")
launch_result = subprocess.run([npm, "exec", "--", "electron", "."], cwd=desktop_dir, env=env, check=False)
electron_argv = [npm, "exec", "--", "electron", "."]
if getattr(args, "local", False):
electron_argv.append("--local")
launch_result = subprocess.run(electron_argv, cwd=desktop_dir, env=env, check=False)
sys.exit(launch_result.returncode)
if packaged_executable is None:
@@ -8677,6 +8680,8 @@ def cmd_gui(args: argparse.Namespace):
launch_command.append("--disable-setuid-sandbox")
launch_command.extend(config_electron_flags)
if getattr(args, "local", False):
launch_command.append("--local")
print(f"→ Launching packaged Hermes Desktop: {' '.join(launch_command)}")
launch_result = subprocess.run(launch_command, cwd=desktop_dir, env=env, check=False)
sys.exit(launch_result.returncode)
+23
View File
@@ -1023,6 +1023,29 @@ def resolve_provider_full(
if custom_pdef is not None:
return custom_pdef
# 2c. Managed local runtime: the llamacpp aliases are a real provider
# whenever the managed server (or a detected external one) resolves —
# no credential and no providers: entry required, the credential is
# reachability. Without this rung the model-switch path rejected the
# very provider the Local Models 'Use' flow writes to config
# ("Unknown provider 'llamacpp'" from the desktop dropdown).
if raw in ("llamacpp", "llama.cpp", "llama-cpp"):
try:
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0)
except Exception:
endpoint = None
if endpoint:
return ProviderDef(
id="llamacpp",
name="Local",
transport="openai_chat",
api_key_env_vars=(),
base_url=endpoint["base_url"],
source="local-runtime",
)
# 3. Try models.dev directly (for providers not in our ALIASES)
try:
from agent.models_dev import get_provider_info as _mdev_provider
+47
View File
@@ -1228,6 +1228,53 @@ def _resolve_named_custom_runtime(
# `provider: ollama` with a LAN/WireGuard `base_url` doesn't silently
# fall through to OpenRouter.
requested_norm = (requested_provider or "").strip().lower()
# Managed llama.cpp runtime: a llamacpp-flavored alias with no explicit
# base_url resolves to the supervised server (or a detected external
# one) before the generic custom fallthrough. Explicit base_url always
# wins — a user pointing at a specific server means that server.
if requested_norm in ("llamacpp", "llama.cpp", "llama-cpp") and not explicit_base_url:
try:
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
endpoint = resolve_llamacpp_endpoint()
except Exception: # noqa: BLE001 — resolution is best-effort
endpoint = None
if endpoint:
return {
"provider": "custom",
"api_mode": "chat_completions",
"base_url": endpoint["base_url"],
"api_key": (explicit_api_key or "").strip()
or endpoint["api_key"] or "no-key-required",
"source": "local-runtime",
"requested_provider": requested_provider,
}
# No server to serve this model. Say so and stop — falling through
# to the generic custom path sends the request to whatever provider
# picks it up (OpenRouter with a placeholder key), and the user's
# "local server is off" surfaces as that provider's baffling
# "401 Invalid API key". The switch's own state picks the message:
# the user who turned the server off gets pointed at the switch,
# anyone else at the setup pane.
try:
from hermes_cli.config import load_config as _load_cfg
_lr_enabled = bool((_load_cfg().get("local_runtime") or {}).get("enabled"))
except Exception: # noqa: BLE001
_lr_enabled = False
if _lr_enabled:
raise ValueError(
"The local model server isn't running. It may still be "
"starting — try again in a moment, or check Settings → "
"Providers → Local models."
)
raise ValueError(
"The local model server is turned off. Turn it back on in "
"Settings → Providers → Local models, or switch to another "
"model."
)
if requested_norm and requested_norm != "custom":
try:
from hermes_cli.auth import resolve_provider as _resolve_provider
+5
View File
@@ -55,6 +55,11 @@ def build_gui_parser(subparsers, *, cmd_gui: Callable) -> None:
action="store_true",
help="Skip npm install/package and launch the existing unpacked app from apps/desktop/release",
)
gui_parser.add_argument(
"--local",
action="store_true",
help="Show the local-models UI in the desktop app (models pane, quickstart, picker rows)",
)
gui_parser.add_argument(
"--force-build",
action="store_true",
File diff suppressed because it is too large Load Diff
+34
View File
@@ -511,6 +511,28 @@ async def _lifespan(app: "FastAPI"):
# sweeping stale sessions on schedule, independent of list requests.
auto_archive_task = asyncio.create_task(_auto_archive_ticker_loop())
# Managed local runtime: when the user opted in (local_runtime.enabled,
# set by the Local Models 'Use' action), bring the llama-server back up
# so a restart doesn't strand a llamacpp main model without a backend.
# Off-thread and best-effort: binary check + spawn + health poll must
# not delay the server socket, and failure falls back to configured
# cloud providers exactly like a cold start.
def _boot_local_runtime():
try:
from hermes_cli.config import load_config
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
# Server only — models load on first inference, always (residency
# design: downloaded = available; demand loads; idleness
# evicts). An empty router holds no VRAM; warming a model at
# boot would reload gigabytes nobody asked for yet.
ensure_local_runtime(load_config())
except Exception as exc: # noqa: BLE001
logging.getLogger(__name__).warning("local runtime boot failed: %s", exc)
threading.Thread(target=_boot_local_runtime, daemon=True,
name="local-runtime-boot").start()
try:
yield
finally:
@@ -523,6 +545,14 @@ async def _lifespan(app: "FastAPI"):
selftest_task.cancel()
auto_archive_task.cancel()
await PTY_REGISTRY.close_all()
# Stop the managed llama-server with its parent — a supervisor-less
# orphan would keep VRAM pinned after the app closes.
try:
from hermes_cli.local_runtime.bootstrap import shutdown_local_runtime
shutdown_local_runtime()
except Exception: # noqa: BLE001
pass
if os.getenv("HERMES_DESKTOP") == "1":
_terminate_desktop_managed_gateway()
@@ -3289,6 +3319,10 @@ def _git_path(path: str) -> str:
from hermes_cli.web_routers import git as _git_routes # noqa: E402
app.include_router(_git_routes.router)
from hermes_cli.web_routers import local_models as _local_models_routes # noqa: E402
app.include_router(_local_models_routes.router)
from hermes_cli.web_routers.git import ( # noqa: E402,F401 — legacy re-exports; tests call these via web_server.<name>
git_status_route,
git_worktrees_route,
+1 -1
View File
@@ -587,7 +587,7 @@ py-modules = [
include = ["agent", "agent.*", "tools", "tools.*", "hermes_cli", "hermes_cli.*", "gateway", "gateway.*", "tui_gateway", "tui_gateway.*", "cron", "cron.*", "acp_adapter", "plugins", "plugins.*", "providers", "providers.*"]
[tool.setuptools.package-data]
hermes_cli = ["observability/schemas/*.json", "data/*.json"]
hermes_cli = ["observability/schemas/*.json", "data/*.json", "local_runtime/*.json"]
# gateway/assets/ ships status_phrases.yaml and the Telegram BotFather
# screenshot. Without this, sealed venvs (uv2nix) silently lose both —
# status phrases fall back to the tiny hardcoded set and the Telegram
+135 -22
View File
@@ -1976,6 +1976,64 @@ class AIAgent:
review_memory: bool = False,
review_skills: bool = False,
focus: Optional[str] = None,
explicit: bool = False,
) -> None:
"""Post-turn review entry point: decide WHEN, then spawn.
The decision to review (nudge intervals, enabled gate) already
happened at the call site. This wrapper adds one policy: a review
whose runtime resolves to the MANAGED LOCAL llama-server is queued
for machine idle instead of spawned into the user's GPU mid-session
(auxiliary.background_review.defer: auto|never). Everything else —
cloud runtimes, external local servers, explicit /refine — spawns
immediately, exactly as before.
``explicit`` marks a user-initiated review (/refine, with or
without focus text): never deferred. It does NOT touch the
delegate/enabled gates below — those stay keyed on ``focus`` so a
bare /refine keeps its historical gating behavior.
"""
# Delegation-subagent and enabled gates run here at enqueue/spawn
# time; the idle dispatcher re-checks the enabled gate again at
# dispatch time so a review queued for minutes cannot be
# resurrected after the user disables reviews.
if focus is None and getattr(self, "_delegate_depth", 0) > 0:
return
task_cfg = None
if focus is None:
from agent.background_review import load_background_review_settings
enabled, task_cfg = load_background_review_settings()
if not enabled:
return
kwargs = dict(
messages_snapshot=messages_snapshot,
review_memory=review_memory,
review_skills=review_skills,
focus=focus,
task_cfg=task_cfg,
)
if focus is None and not explicit:
from agent.review_idle_queue import (
QUEUE,
defer_mode,
review_targets_managed_local,
)
if (defer_mode(task_cfg) == "auto"
and review_targets_managed_local(self, task_cfg)):
session_key = str(getattr(self, "session_id", None) or id(self))
QUEUE.enqueue(self, session_key, kwargs)
return
self._spawn_background_review_now(**kwargs)
def _spawn_background_review_now(
self,
messages_snapshot: List[Dict],
review_memory: bool = False,
review_skills: bool = False,
focus: Optional[str] = None,
task_cfg: Optional[Dict[str, Any]] = None,
_requeue_attempts: int = 0,
) -> None:
"""Spawn the background memory/skill review thread.
@@ -1988,28 +2046,17 @@ class AIAgent:
``focus`` is optional user-supplied steering (from ``/refine``)
appended to the review prompt — e.g. "save the deploy workflow as a
skill". The automatic post-turn triggers never set it.
``task_cfg`` is the pre-loaded ``auxiliary.background_review``
block from the entry wrapper (None on direct calls, e.g. /refine —
the spawn path reads config itself then).
A deferred review preempted by a live turn is REQUEUED (bounded by
``_requeue_attempts``) instead of lost: on the managed local
runtime a review takes minutes, so cancel-and-forget — harmless on
cloud, where reviews finish in seconds — would silently discard
most learning on an active session.
"""
# A delegation subagent (``_delegate_depth > 0``) must not run the
# automatic post-turn review. Subagents are ephemeral workers already
# barred from writing shared MEMORY.md (``DELEGATE_BLOCKED_TOOLS``) and
# are spawned with ``skip_memory=True``, so a review here has little to
# persist — yet it inherits the subagent's (often premium) delegation
# model and replays the whole conversation at premium rates, silently
# inflating token cost (#85859). An explicit ``/refine`` (``focus`` set)
# is a deliberate user request and still runs.
if focus is None and getattr(self, "_delegate_depth", 0) > 0:
return
# Explicit off-switch for automatic post-turn forks
# (``auxiliary.background_review.enabled: false``). Manual ``/refine``
# still works — same contract as zeroing the nudge intervals (#87250).
# Load the task block once here and pass it into the spawn path so
# aux routing does not re-read config.
task_cfg = None
if focus is None:
from agent.background_review import load_background_review_settings
enabled, task_cfg = load_background_review_settings()
if not enabled:
return
from agent.background_review import (
finish_background_review_run,
prepare_background_review_run,
@@ -2030,10 +2077,25 @@ class AIAgent:
task_cfg=task_cfg,
review_run=review_run,
)
def _target_with_requeue() -> None:
target()
self._maybe_requeue_preempted_review(
review_run,
dict(
messages_snapshot=messages_snapshot,
review_memory=review_memory,
review_skills=review_skills,
focus=focus,
task_cfg=task_cfg,
_requeue_attempts=_requeue_attempts + 1,
),
)
# Carry the active profile into the review thread so MEMORY.md /
# skill review writes land in the right profile (#54937).
t = threading.Thread(
target=propagate_context_to_thread(target),
target=propagate_context_to_thread(_target_with_requeue),
daemon=True,
name="bg-review",
)
@@ -2042,6 +2104,42 @@ class AIAgent:
finish_background_review_run(self, review_run)
raise
_REVIEW_REQUEUE_MAX_ATTEMPTS = 3
def _maybe_requeue_preempted_review(self, review_run, kwargs) -> None:
"""Requeue a deferred-mode review that a live turn cancelled.
Only fires for automatic reviews whose runtime targets the managed
local server (the deferred population); bounded attempts prevent a
busy box from cycling one review forever — past the cap it is
dropped exactly like the pre-deferral behavior dropped every
cancelled review.
"""
try:
if not review_run.cancel_requested.is_set():
return # ran to completion (or never admitted for other reasons)
if kwargs.get("focus") is not None:
return
if kwargs.get("_requeue_attempts", 0) > self._REVIEW_REQUEUE_MAX_ATTEMPTS:
logger.info("Preempted background review dropped after %d requeues",
self._REVIEW_REQUEUE_MAX_ATTEMPTS)
return
from agent.review_idle_queue import (
QUEUE,
defer_mode,
review_targets_managed_local,
)
task_cfg = kwargs.get("task_cfg")
if (defer_mode(task_cfg) != "auto"
or not review_targets_managed_local(self, task_cfg)):
return
session_key = str(getattr(self, "session_id", None) or id(self))
# kwargs carries the incremented _requeue_attempts through the
# queue so the cap survives the round trip.
QUEUE.enqueue(self, session_key, dict(kwargs))
except Exception: # noqa: BLE001 — requeue is best-effort
logger.debug("Preempted-review requeue failed", exc_info=True)
def _build_memory_write_metadata(
self,
*,
@@ -9095,6 +9193,13 @@ class AIAgent:
cancel_background_review_for_live_turn(self)
# Turn liveness for the deferred-review idle queue: a queued review
# must not dispatch into the settle gap between two quick prompts.
# Marked inside the try below so the balancing note_turn_finished in
# its finally covers every exit; the actual start-mark happens as the
# first statement of the try.
from agent.review_idle_queue import QUEUE as _review_queue
from agent.aux_accounting import (
reset_accounting_context,
set_accounting_context,
@@ -9180,6 +9285,7 @@ class AIAgent:
_clear_if_owned()
try:
_review_queue.note_turn_started()
# Serialize the full load -> run -> flush region across Hermes
# processes. Gateway's asyncio lease closes alias routing inside one
# process; this durable lease covers Desktop, CLI resume, gateway,
@@ -9726,6 +9832,13 @@ class AIAgent:
reset_conversation_context(token)
if affinity_token is not None:
reset_affinity_scope(affinity_token)
# Balance the note_turn_started above — every exit path
# lands here, so the idle queue's live-turn count cannot
# leak upward and starve deferred reviews.
try:
_review_queue.note_turn_finished()
except Exception:
pass
def chat(self, message: str, stream_callback: Optional[callable] = None) -> str:
"""
+94
View File
@@ -0,0 +1,94 @@
"""Propose catalog quality updates from Artificial Analysis.
Authoring-time helper — NEVER called at runtime (their terms forbid
client-side keys, the fleet would burn the rate limit, and a
recommendation must not change because a third-party endpoint
hiccuped). Run it when adding a model or refreshing the ordering;
review the printed diff and edit catalog.json yourself. The script
proposes, the commit decides.
The catalog's `quality` stays OUR field: AA-informed where they cover a
model, editorially set where they don't (day-0 releases lag their evals;
some entries never appear). AA's Intelligence Index grades the
full-precision cloud model, not our Q4 build — fine for ordering, never
for display.
Usage:
export AA_API_KEY=... # from https://artificialanalysis.ai (free tier)
python scripts/aa_quality_sync.py
Attribution: scores by Artificial Analysis (https://artificialanalysis.ai).
"""
from __future__ import annotations
import json
import os
import sys
import urllib.request
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
CATALOG_PATH = REPO_ROOT / "hermes_cli" / "local_runtime" / "catalog.json"
AA_URL = "https://artificialanalysis.ai/api/v2/data/llms/models"
# Catalog entry id -> AA slug. Hand-maintained: AA's naming rarely matches
# HF repo names, and a wrong match silently mis-ranks a model. An entry
# absent here (or mapped to None) is editorial-only and never overwritten.
AA_SLUG_BY_ENTRY = {
"qwen3.8-27b": "qwen3-8-27b",
"qwen3.8-flash-next": "qwen3-8-flash-next",
"qwen3.6-35b-a3b": "qwen3-6-35b-a3b",
"deepseek-v4-flash": "deepseek-v4-flash",
}
def fetch_aa_models(api_key: str) -> dict[str, dict]:
req = urllib.request.Request(AA_URL, headers={"x-api-key": api_key})
with urllib.request.urlopen(req, timeout=30) as r:
doc = json.load(r)
return {m["slug"]: m for m in doc.get("data", [])}
def main() -> int:
api_key = os.environ.get("AA_API_KEY", "").strip()
if not api_key:
print("AA_API_KEY not set — create a free key at "
"https://artificialanalysis.ai and export it.", file=sys.stderr)
return 2
catalog = json.loads(CATALOG_PATH.read_text(encoding="utf-8"))
aa = fetch_aa_models(api_key)
print(f"{'entry':24s} {'catalog q':>9s} {'AA index':>9s} note")
print("-" * 70)
for model in catalog["models"]:
entry_id = model["id"]
current = model.get("quality", 0)
slug = AA_SLUG_BY_ENTRY.get(entry_id)
if not slug:
print(f"{entry_id:24s} {current:>9d} {'—':>9s} editorial only (no AA mapping)")
continue
hit = aa.get(slug)
if hit is None:
print(f"{entry_id:24s} {current:>9d} {'—':>9s} not in AA data (slug {slug!r})")
continue
index = (hit.get("evaluations") or {}).get(
"artificial_analysis_intelligence_index")
if index is None:
print(f"{entry_id:24s} {current:>9d} {'—':>9s} AA row lacks the index")
continue
proposed = round(float(index))
marker = "" if proposed == current else " <-- proposes change"
print(f"{entry_id:24s} {current:>9d} {proposed:>9d}{marker}")
print("\nReview against the decision table before editing: a quality "
"change that flips cells in tests/hermes_cli/"
"test_local_recommendation.py is the actual decision being made.")
print("Attribution: scores by Artificial Analysis "
"(https://artificialanalysis.ai).")
return 0
if __name__ == "__main__":
sys.exit(main())
+347
View File
@@ -0,0 +1,347 @@
"""Deferred background review on the managed local runtime.
Behavior contracts for agent/review_idle_queue.py and the decision
wrapper in run_agent.AIAgent._spawn_background_review:
- defer: auto + review runtime == managed local -> queued, not spawned
- defer: never, or non-managed runtime, or /refine -> immediate spawn
- queue coalesces per session (newest snapshot wins, age preserved)
- dispatch requires sustained process-quiet AND server idle
- aged-out items dispatch regardless of idleness (delay, never lose)
- preempted deferred reviews requeue with a bounded attempt cap
"""
import threading
import time
import types
import pytest
from agent.review_idle_queue import (
ReviewIdleQueue,
_IDLE_SETTLE_S,
defer_max_age_s,
defer_mode,
)
# ── config parsing ───────────────────────────────────────────────
def test_defer_mode_values():
assert defer_mode(None) == "auto"
assert defer_mode({}) == "auto"
assert defer_mode({"defer": "auto"}) == "auto"
assert defer_mode({"defer": "never"}) == "never"
assert defer_mode({"defer": "NEVER"}) == "never"
# Unknown values fall back to auto (the safe, documented default).
assert defer_mode({"defer": "sometimes"}) == "auto"
assert defer_mode({"defer": 3}) == "auto"
def test_defer_max_age_parsing():
assert defer_max_age_s(None) == 30 * 60
assert defer_max_age_s({"defer_max_age_s": 120}) == 120.0
assert defer_max_age_s({"defer_max_age_s": "600"}) == 600.0
# Nonsense and non-positive fall back to the default.
assert defer_max_age_s({"defer_max_age_s": "soon"}) == 30 * 60
assert defer_max_age_s({"defer_max_age_s": 0}) == 30 * 60
assert defer_max_age_s({"defer_max_age_s": -5}) == 30 * 60
# ── queue harness ────────────────────────────────────────────────
class _FakeAgent:
def __init__(self):
self.spawned = []
self.session_id = "sess-x"
def _spawn_background_review_now(self, **kwargs):
self.spawned.append(kwargs)
def _make_queue(now=None, server_idle=True):
q = ReviewIdleQueue()
clock = {"t": 0.0}
if now is None:
q._now = lambda: clock["t"]
else:
q._now = now
q._server_idle = lambda: server_idle
# Never start the real dispatcher thread in unit tests.
q._ensure_thread = lambda: None
return q, clock
def test_enqueue_coalesces_per_session_newest_wins_oldest_age():
q, clock = _make_queue()
agent = _FakeAgent()
clock["t"] = 100.0
q.enqueue(agent, "s1", {"messages_snapshot": ["old"], "task_cfg": {}})
clock["t"] = 200.0
q.enqueue(agent, "s1", {"messages_snapshot": ["new"], "task_cfg": {}})
q.enqueue(agent, "s2", {"messages_snapshot": ["other"], "task_cfg": {}})
assert q.pending_count() == 2
with q._lock:
item = q._pending["s1"]
# Newest snapshot won, but the age clock kept the ORIGINAL enqueue
# time so a busy session cannot push its own age-out forever.
assert item.kwargs["messages_snapshot"] == ["new"]
assert item.enqueued_at == 100.0
def test_dispatch_waits_for_sustained_quiet():
q, clock = _make_queue()
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {}})
# A live turn: nothing dispatches.
q.note_turn_started()
assert q._pop_dispatchable() is None
# Turn finished, but the settle window hasn't elapsed.
q.note_turn_finished()
assert q._pop_dispatchable() is None
# Quiet long enough -> dispatchable.
clock["t"] += _IDLE_SETTLE_S + 1
item = q._pop_dispatchable()
assert item is not None and item.session_key == "s1"
assert q.pending_count() == 0
def test_dispatch_blocked_by_busy_server():
q, clock = _make_queue(server_idle=False)
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {}})
q.note_turn_started()
q.note_turn_finished()
clock["t"] += _IDLE_SETTLE_S + 1
# Process is quiet but the managed server has a processing slot
# (another profile's session, a live prefill): hold.
assert q._pop_dispatchable() is None
assert q.pending_count() == 1
def test_aged_out_item_dispatches_despite_busy_server():
q, clock = _make_queue(server_idle=False)
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {"defer_max_age_s": 60}})
q.note_turn_started() # never goes quiet
clock["t"] += 61
item = q._pop_dispatchable()
assert item is not None
assert item.session_key == "s1"
def test_new_turn_resets_the_quiet_clock():
q, clock = _make_queue()
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {}})
q.note_turn_started()
q.note_turn_finished()
clock["t"] += _IDLE_SETTLE_S - 2
# A new prompt arrives just before the settle window closes.
q.note_turn_started()
clock["t"] += 30
assert q._pop_dispatchable() is None # still live
q.note_turn_finished()
assert q._pop_dispatchable() is None # settle restarts
clock["t"] += _IDLE_SETTLE_S + 1
assert q._pop_dispatchable() is not None
def test_nested_turns_require_all_to_finish():
q, clock = _make_queue()
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {}})
q.note_turn_started()
q.note_turn_started()
q.note_turn_finished()
clock["t"] += _IDLE_SETTLE_S + 1
assert q._pop_dispatchable() is None # one turn still live
q.note_turn_finished()
clock["t"] += _IDLE_SETTLE_S + 1
assert q._pop_dispatchable() is not None
# ── the decision wrapper ─────────────────────────────────────────
def _wrapper_agent(monkeypatch, defer="auto", managed=True):
"""A minimal object wearing the real _spawn_background_review."""
import run_agent
from agent import review_idle_queue as riq
agent = _FakeAgent()
agent._delegate_depth = 0
calls = {"enqueued": [], "spawned": []}
monkeypatch.setattr(
"agent.background_review.load_background_review_settings",
lambda: (True, {"defer": defer}),
)
monkeypatch.setattr(
riq, "review_targets_managed_local", lambda a, cfg: managed
)
monkeypatch.setattr(
riq.QUEUE, "enqueue",
lambda a, key, kw: calls["enqueued"].append((key, kw)),
)
agent._spawn_background_review_now = (
lambda **kw: calls["spawned"].append(kw)
)
bound = types.MethodType(run_agent.AIAgent._spawn_background_review, agent)
return bound, calls
def test_wrapper_defers_managed_local_auto(monkeypatch):
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
spawn([{"role": "user", "content": "hi"}], review_memory=True)
assert len(calls["enqueued"]) == 1
assert calls["spawned"] == []
key, kwargs = calls["enqueued"][0]
assert key == "sess-x"
assert kwargs["review_memory"] is True
def test_wrapper_spawns_immediately_for_non_managed(monkeypatch):
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=False)
spawn([{"role": "user", "content": "hi"}], review_skills=True)
assert calls["enqueued"] == []
assert len(calls["spawned"]) == 1
def test_wrapper_defer_never_is_old_behavior(monkeypatch):
spawn, calls = _wrapper_agent(monkeypatch, defer="never", managed=True)
spawn([{"role": "user", "content": "hi"}], review_memory=True)
assert calls["enqueued"] == []
assert len(calls["spawned"]) == 1
def test_wrapper_refine_bypasses_queue(monkeypatch):
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
spawn([{"role": "user", "content": "hi"}], review_memory=True,
focus="save the deploy workflow")
assert calls["enqueued"] == []
assert len(calls["spawned"]) == 1
assert calls["spawned"][0]["focus"] == "save the deploy workflow"
def test_wrapper_bare_refine_bypasses_queue(monkeypatch):
"""/refine with no focus text is still explicit: never deferred."""
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
spawn([{"role": "user", "content": "hi"}], review_memory=True,
focus=None, explicit=True)
assert calls["enqueued"] == []
assert len(calls["spawned"]) == 1
def test_wrapper_cloud_fast_path_skips_runtime_resolution(monkeypatch):
"""No managed server on the machine -> the classifier answers from the
TTL-cached netloc probe alone, without resolving the review runtime.
Guards the cloud-only turn tail from growing new work."""
from agent import review_idle_queue as riq
resolved = {"count": 0}
def _explode(agent, cfg):
resolved["count"] += 1
raise AssertionError("runtime resolution must not run")
monkeypatch.setattr(
"agent.auxiliary_client._managed_local_netloc", lambda: "")
monkeypatch.setattr(
"agent.background_review._resolve_review_runtime", _explode)
assert riq.review_targets_managed_local(object(), {}) is False
assert resolved["count"] == 0
def test_dispatcher_rechecks_enabled_gate(monkeypatch):
"""A review disabled while queued must not be resurrected at dispatch."""
from agent import review_idle_queue as riq
q, clock = _make_queue()
agent = _FakeAgent()
q.enqueue(agent, "s1", {"task_cfg": {}})
monkeypatch.setattr(
"agent.background_review.load_background_review_settings",
lambda: (False, {}),
)
item = None
clock["t"] += _IDLE_SETTLE_S + 1
q.note_turn_started()
q.note_turn_finished()
clock["t"] += _IDLE_SETTLE_S + 1
item = q._pop_dispatchable()
assert item is not None
assert q._still_enabled(item) is False
# ── requeue on preemption ────────────────────────────────────────
class _Run:
def __init__(self, cancelled):
self.cancel_requested = threading.Event()
if cancelled:
self.cancel_requested.set()
def _requeue_agent(monkeypatch, managed=True):
import run_agent
from agent import review_idle_queue as riq
agent = _FakeAgent()
calls = {"enqueued": []}
monkeypatch.setattr(
riq, "review_targets_managed_local", lambda a, cfg: managed
)
monkeypatch.setattr(
riq.QUEUE, "enqueue",
lambda a, key, kw: calls["enqueued"].append(kw),
)
agent._REVIEW_REQUEUE_MAX_ATTEMPTS = (
run_agent.AIAgent._REVIEW_REQUEUE_MAX_ATTEMPTS
)
bound = types.MethodType(
run_agent.AIAgent._maybe_requeue_preempted_review, agent
)
return bound, calls
def test_preempted_review_requeues(monkeypatch):
requeue, calls = _requeue_agent(monkeypatch)
requeue(_Run(cancelled=True),
{"task_cfg": {"defer": "auto"}, "focus": None,
"_requeue_attempts": 1})
assert len(calls["enqueued"]) == 1
# The attempt counter rides along so the cap survives the round trip.
assert calls["enqueued"][0]["_requeue_attempts"] == 1
def test_completed_review_does_not_requeue(monkeypatch):
requeue, calls = _requeue_agent(monkeypatch)
requeue(_Run(cancelled=False),
{"task_cfg": {"defer": "auto"}, "focus": None,
"_requeue_attempts": 1})
assert calls["enqueued"] == []
def test_requeue_attempt_cap(monkeypatch):
requeue, calls = _requeue_agent(monkeypatch)
requeue(_Run(cancelled=True),
{"task_cfg": {"defer": "auto"}, "focus": None,
"_requeue_attempts": 4})
assert calls["enqueued"] == []
def test_requeue_skips_non_managed(monkeypatch):
requeue, calls = _requeue_agent(monkeypatch, managed=False)
requeue(_Run(cancelled=True),
{"task_cfg": {"defer": "auto"}, "focus": None,
"_requeue_attempts": 1})
assert calls["enqueued"] == []
+88
View File
@@ -158,6 +158,94 @@ class TestShouldExclude:
# The .db itself is still included (and safe-copied separately)
assert not _should_exclude(Path("state.db"))
def test_excludes_managed_runtime_trees_at_root(self):
"""models/, runtimes/, and node/ at a profile-home root hold
re-downloadable GGUF weights and runtime binaries that reach
hundreds of GB — zipping them is the 20-minute-hang symptom."""
from hermes_cli.backup import _should_exclude
assert _should_exclude(Path("models/Qwen3.6-27B-Q4_K_M.gguf"))
assert _should_exclude(Path("models/assets/mmproj.gguf"))
assert _should_exclude(Path("runtimes/llamacpp/b10362/cuda/ggml-cuda.dll"))
assert _should_exclude(Path("node/node.exe"))
# Named profiles download their own copies.
assert _should_exclude(Path("profiles/clean/models/big.gguf"))
assert _should_exclude(Path("profiles/clean/runtimes/llamacpp/x.dll"))
def test_keeps_nested_dirs_named_like_runtime_trees(self):
"""A deeper directory that happens to be called models/ or node/ is
user data (a skill's assets, project files) and must survive."""
from hermes_cli.backup import _should_exclude
assert not _should_exclude(Path("skills/mlops/models/notes.md"))
assert not _should_exclude(Path("scratch/node/index.js"))
assert not _should_exclude(Path("profiles/clean/skills/x/models/a.txt"))
def test_excludes_desktop_emergency_state_db_baks(self):
"""The desktop updater's pre-flight drops timestamped
state.db.pre-update-emergency-*.bak files at the HERMES_HOME root —
backup artifacts in the same class as backups/, so a full backup
must not re-ship them."""
from hermes_cli.backup import _should_exclude
assert _should_exclude(
Path("state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
)
assert _should_exclude(
Path("profiles/coder/state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
)
# Other .bak files are user data and stay.
assert not _should_exclude(Path("config.yaml.bak"))
# ---------------------------------------------------------------------------
# _iter_backup_files tests
# ---------------------------------------------------------------------------
class TestIterBackupFiles:
def test_manual_and_automatic_paths_share_one_walk(self, tmp_path):
"""Both backup entry points must select the identical file set.
Before the walks were unified, the automatic pre-update path pruned
``hermes-agent`` at ANY depth, silently dropping nested skill dirs
like ``skills/autonomous-ai-agents/hermes-agent/`` that the manual
path preserved. One shared iterator makes that drift impossible;
this test pins the contract."""
from hermes_cli.backup import _iter_backup_files
root = tmp_path / ".hermes"
root.mkdir()
_make_hermes_tree(root)
# The case the old automatic walk got wrong: a nested dir named
# hermes-agent holding real skill content.
nested = root / "skills" / "autonomous-ai-agents" / "hermes-agent"
nested.mkdir(parents=True)
(nested / "SKILL.md").write_text("# nested skill\n")
# A root-level managed runtime tree that both paths must prune.
(root / "models").mkdir()
(root / "models" / "big.gguf").write_bytes(b"\x00" * 64)
out_path = tmp_path / "out.zip"
selected = {str(rel) for _, rel in _iter_backup_files(root, out_path)}
rel_nested = str(Path("skills/autonomous-ai-agents/hermes-agent/SKILL.md"))
assert rel_nested in selected
assert str(Path("models/big.gguf")) not in selected
assert not any(s.startswith("hermes-agent") for s in selected)
def test_skipped_dirs_collected_for_summary(self, tmp_path):
from hermes_cli.backup import _iter_backup_files
root = tmp_path / ".hermes"
root.mkdir()
_make_hermes_tree(root)
(root / "models").mkdir()
(root / "models" / "big.gguf").write_bytes(b"\x00")
skipped: set = set()
list(_iter_backup_files(root, tmp_path / "out.zip", skipped))
assert "models" in skipped
assert "hermes-agent" in skipped
# ---------------------------------------------------------------------------
# Backup tests
@@ -0,0 +1,126 @@
"""Every staged model must launch with a policy decision, never stock fit.
The managed server autoloads any GGUF in its models dir; a model missing
from the preset INI loads with llama-server defaults (f16 KV at max
context, no placement) — on Windows/WDDM that silently demotes VRAM and
decodes at a crawl. Boot must therefore refuse to adopt a running server
whose presets predate the staged set."""
from __future__ import annotations
import pytest
@pytest.fixture
def hermes_home(tmp_path, monkeypatch):
home = tmp_path / ".hermes"
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
return home
def _stage(home, name):
mdir = home / "models"
mdir.mkdir(parents=True, exist_ok=True)
(mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 32)
def _write_presets(home, *model_ids):
pdir = home / "runtimes" / "llamacpp"
pdir.mkdir(parents=True, exist_ok=True)
body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
(pdir / "presets.ini").write_text(body, encoding="utf-8")
def test_presets_stale_when_a_staged_model_has_no_section(hermes_home):
from hermes_cli.local_runtime.bootstrap import _presets_stale
_stage(hermes_home, "model-a")
_stage(hermes_home, "model-b")
_write_presets(hermes_home, "model-a")
assert _presets_stale() is True
def test_presets_current_when_every_staged_model_is_covered(hermes_home):
from hermes_cli.local_runtime.bootstrap import _presets_stale
_stage(hermes_home, "model-a")
_write_presets(hermes_home, "model-a")
assert _presets_stale() is False
def test_no_models_is_never_stale(hermes_home):
from hermes_cli.local_runtime.bootstrap import _presets_stale
_write_presets(hermes_home, "model-a")
assert _presets_stale() is False
def test_boot_replaces_incumbent_with_stale_presets(hermes_home, monkeypatch):
"""ensure_local_runtime must not adopt a running server whose presets
miss a staged model — it stops it and boots fresh (boot itself is
stubbed; the contract under test is the adopt/replace decision)."""
import hermes_cli.local_runtime.bootstrap as boot
_stage(hermes_home, "model-a")
_stage(hermes_home, "model-b")
_write_presets(hermes_home, "model-a")
stopped = {}
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint._state_endpoint",
lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 12345})
monkeypatch.setattr(boot, "_stop_state_server",
lambda state: stopped.setdefault("pid", state["pid"]))
sentinel = object()
def fake_boot(*a, **k):
raise _BootReached()
class _BootReached(Exception):
pass
# Fail fast once boot proper begins — reaching it IS the assertion.
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.ensure_runtime_installed", fake_boot)
result = boot.ensure_local_runtime({"local_runtime": {"enabled": True}})
assert stopped.get("pid") == 12345, "stale incumbent was not stopped"
# Boot proceeded past adoption (our fake raised inside the try block,
# which ensure_local_runtime swallows into a None return).
assert result is None or result is sentinel
def test_refresh_bounces_an_adopted_server(hermes_home, monkeypatch):
"""refresh_local_runtime with no in-process supervisor but a running
state-file server (the post-restart shape) must stop that server and
boot fresh — NOT silently no-op. Regression: the no-op meant every
download/delete after a backend restart left the router serving a
stale model catalog, and picking the new model failed with
'not found in this provider's model listing'."""
import hermes_cli.local_runtime.bootstrap as boot
stopped = {}
monkeypatch.setattr(boot, "_SUPERVISOR", None)
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint._state_endpoint",
lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 4242})
monkeypatch.setattr(boot, "_stop_state_server",
lambda state: stopped.setdefault("pid", state["pid"]))
booted = {}
monkeypatch.setattr(boot, "ensure_local_runtime",
lambda cfg, force=False: booted.setdefault("force", force) or object())
assert boot.refresh_local_runtime() is True
assert stopped.get("pid") == 4242, "adopted server was not stopped"
assert booted.get("force") is True, "fresh boot did not follow the stop"
def test_refresh_no_server_anywhere_is_a_noop(hermes_home, monkeypatch):
import hermes_cli.local_runtime.bootstrap as boot
monkeypatch.setattr(boot, "_SUPERVISOR", None)
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint._state_endpoint", lambda: None)
assert boot.refresh_local_runtime() is False
+51
View File
@@ -0,0 +1,51 @@
"""Launch and growth decisions must price against CAPACITY, not live-free
VRAM. Both execute through a server bounce — the outgoing instance's memory
is freed before the new one loads — so a probe that reads the predecessor's
(or the grown model's own) residency as 'gone' vetoes configurations that
genuinely fit. Symptom when this regresses: a model the pane promised
'144K on GPU' launches with its weights pinned to CPU and single-digit
tokens/s while the card sits 60% empty."""
from __future__ import annotations
import ast
import inspect
def _planning_probe_calls(source: str) -> list[bool]:
"""Every probe_budget(...) call's planning= value in the source."""
tree = ast.parse(source)
out = []
for node in ast.walk(tree):
if (isinstance(node, ast.Call)
and getattr(node.func, "id", getattr(node.func, "attr", ""))
== "probe_budget"):
planning = any(
kw.arg == "planning"
and isinstance(kw.value, ast.Constant)
and kw.value.value is True
for kw in node.keywords)
out.append(planning)
return out
def test_bootstrap_presets_price_against_capacity():
import hermes_cli.local_runtime.bootstrap as bootstrap
calls = _planning_probe_calls(inspect.getsource(bootstrap))
assert calls, "bootstrap no longer probes a budget? update this test"
assert all(calls), (
"bootstrap prices launch decisions against live-free VRAM; a "
"restart/refresh probes while the outgoing server still holds the "
"card, pinning fitting models to CPU")
def test_growth_refit_prices_against_capacity():
import hermes_cli.local_runtime.growth as growth
calls = _planning_probe_calls(inspect.getsource(growth))
assert calls, "growth no longer probes a budget? update this test"
assert all(calls), (
"growth re-fits against live-free VRAM; the grown model's own "
"residency reads as unavailable and vetoes rungs that fit the "
"post-bounce card")
+118
View File
@@ -0,0 +1,118 @@
"""The pulled catalog: packaged JSON is the offline truth, a GitHub fetch
swaps entries in memory only, and min_engine gates day-0 models.
Nothing here touches disk beyond the packaged file — the design constraint
is that a git checkout must never see a dirty tracked catalog.json."""
from __future__ import annotations
import dataclasses
import io
import json
import urllib.request
import pytest
import hermes_cli.local_runtime.catalog as cat
@pytest.fixture(autouse=True)
def _reset_refresh_state(monkeypatch):
"""Each test starts outside the TTL window with the packaged catalog."""
monkeypatch.setattr(cat, "_last_refresh_attempt", 0.0)
packaged = cat._packaged_catalog()
monkeypatch.setattr(cat, "CATALOG", packaged)
yield
def _doc_from(entries):
"""A fetchable catalog document built by mutating the packaged JSON."""
from importlib.resources import files
doc = json.loads(files("hermes_cli.local_runtime")
.joinpath("catalog.json").read_text(encoding="utf-8"))
doc["models"] = entries(doc["models"])
return doc
def _fetch_returns(monkeypatch, doc):
body = json.dumps(doc).encode()
class R(io.BytesIO):
def __enter__(self):
return self
def __exit__(self, *a):
return False
monkeypatch.setattr(urllib.request, "urlopen",
lambda *a, **k: R(body))
def test_packaged_json_round_trips_the_catalog():
"""The packaged JSON must produce a complete, selection-ready catalog:
every entry carries estimator inputs and at least one variant, and the
known invariants (best-first ordering, Q4 floor) hold — the same
contract the literals obeyed."""
assert len(cat.CATALOG) >= 4
for e in cat.CATALOG:
assert e.variants and e.n_ctx_train > 0 and e.per_layer_f16 >= 0
sizes = [v.size_bytes for v in e.variants]
assert sizes == sorted(sizes, reverse=True), f"{e.id} not best-first"
def test_refresh_swaps_in_memory_only(monkeypatch, tmp_path):
"""A fetched catalog replaces CATALOG in memory; the packaged file on
disk is untouched (checkout stays clean)."""
from importlib.resources import files
packaged_path = files("hermes_cli.local_runtime").joinpath("catalog.json")
before = packaged_path.read_text(encoding="utf-8")
def add_day0(models):
day0 = dict(models[0])
day0.update(id="day0-model", display_name="Day 0",
description="new", min_engine="b99999")
return models + [day0]
_fetch_returns(monkeypatch, _doc_from(add_day0))
assert cat.refresh_catalog(force=True) is True
assert "day0-model" in {e.id for e in cat.CATALOG}
assert cat.catalog_by_id()["day0-model"].min_engine == "b99999"
assert packaged_path.read_text(encoding="utf-8") == before
def test_refresh_failure_keeps_current_catalog(monkeypatch):
def boom(*a, **k):
raise OSError("offline")
monkeypatch.setattr(urllib.request, "urlopen", boom)
ids_before = [e.id for e in cat.CATALOG]
assert cat.refresh_catalog(force=True) is False
assert [e.id for e in cat.CATALOG] == ids_before
def test_refresh_rejects_wrong_schema(monkeypatch):
doc = _doc_from(lambda m: m)
doc["schema_version"] = 2
_fetch_returns(monkeypatch, doc)
ids_before = [e.id for e in cat.CATALOG]
assert cat.refresh_catalog(force=True) is False
assert [e.id for e in cat.CATALOG] == ids_before
def test_loader_ignores_unknown_fields():
doc = _doc_from(lambda m: m)
doc["models"][0]["future_field"] = {"anything": True}
entries = cat._load_catalog(doc)
assert entries[0].id == doc["models"][0]["id"]
def test_min_engine_gate(monkeypatch):
from hermes_cli.web_routers.local_models import _engine_too_old
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_tags",
lambda: ["b10362"])
assert _engine_too_old("") is False, "no requirement, no gate"
assert _engine_too_old("b10000") is False, "installed engine suffices"
assert _engine_too_old("b10363") is True, "newer requirement gates"
@@ -0,0 +1,59 @@
"""Catalog reachability: every entry's repo and files must exist upstream.
Existence is the contract; SIZES are advisory (they feed the estimator and
progress bars, and downloads deliberately tolerate a stale size when
upstream re-uploads — completeness is judged against the server's own
declared length, never the catalog). Size drift prints as a warning so a
catalog refresh can be batched deliberately; only a MISSING file or repo
fails.
Network-marked (skipped in hermetic CI unless explicitly enabled) — this is
the test that catches wrong repo names (the Nemotron 401) and moved files.
Run before any catalog commit:
HERMES_TEST_NETWORK=1 scripts/run_tests.sh tests/hermes_cli/test_catalog_reachability.py
"""
from __future__ import annotations
import json
import os
import urllib.request
import pytest
pytestmark = pytest.mark.skipif(
not os.environ.get("HERMES_TEST_NETWORK"),
reason="network test; set HERMES_TEST_NETWORK=1 to run",
)
def test_every_catalog_file_resolves():
from hermes_cli.local_runtime.catalog import CATALOG
problems = []
drift = []
for entry in CATALOG:
url = f"https://huggingface.co/api/models/{entry.repo}/tree/main?recursive=true"
try:
with urllib.request.urlopen(url, timeout=30) as r:
files = {f["path"]: f.get("size") for f in json.load(r)}
except Exception as exc: # noqa: BLE001
problems.append(f"{entry.id}: repo {entry.repo} unreachable ({exc})")
continue
for variant in entry.variants:
for asset in entry.download_files(variant):
if asset.path not in files:
problems.append(
f"{entry.id}/{variant.quant}: {asset.path} not in {entry.repo}")
continue
live_size = files[asset.path]
if live_size and live_size != asset.size_bytes:
drift.append(
f"{entry.id}/{variant.quant}: size drift on {asset.path} — "
f"catalog {asset.size_bytes} vs live {live_size}")
if drift:
print("\nADVISORY size drift (downloads tolerate this; refresh when convenient):")
print("\n".join(drift))
assert not problems, "\n".join(problems)
+182
View File
@@ -0,0 +1,182 @@
"""Variant-selection contracts: fit the catalog's single Q4-class build
to a machine and price it honestly. Pure decision-table tests over
synthetic budgets."""
from __future__ import annotations
import pytest
from hermes_cli.local_runtime.catalog import (
CATALOG,
catalog_by_id,
find_entry_for_model,
select_variant,
)
from hermes_cli.local_runtime.estimator import HardwareBudget
GIB = 1 << 30
def budget(vram_gib: float, ram_gib: float = 64) -> HardwareBudget:
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
total_device_bytes=int(vram_gib * GIB),
ram_available_bytes=int(ram_gib * GIB))
def test_every_entry_ships_exactly_one_q4_build():
"""No quant ladder: one Q4-class build per entry (K_M where the repo
ships it, XL elsewhere) — the quant class current engines optimize
for. Nothing below Q4 ever ships. Validation status is explicit per
variant in catalog.json; unvalidated builds are permitted (day-0
entries) and surface as unbadged rows in the pane."""
for entry in CATALOG:
assert len(entry.variants) == 1, (
f"{entry.id}: {len(entry.variants)} variants — expected exactly one")
build = entry.variants[0]
assert build.quant.startswith(("UD-Q4", "Q4")), (
f"{entry.id}: ships {build.quant}, not a Q4-class build")
for asset in entry.download_files(build):
assert asset.size_bytes > 0, f"{entry.id}: no size on {asset.path}"
def test_split_variants_have_coherent_parts():
"""Multi-file variants: same model_id from every part, exact sizes,
first file is the load target."""
entry = catalog_by_id()["deepseek-v4-flash"]
for v in entry.variants:
assert len(v.files) >= 2, "deepseek ships split GGUFs"
assert "00001-of" in v.files[0].path, "first part must be the load target"
assert v.size_bytes == sum(f.size_bytes for f in v.files)
assert entry.draft is not None, "DSpark draft rides along"
def test_selection_is_the_q4_build_even_with_headroom():
"""The selector picks the Q4 build even when bigger quants would fit
with room to spare — headroom buys window, not quant. Larger builds
stay one tile click away in the pane."""
entry = catalog_by_id()["qwen3.8-27b"]
choice = select_variant(entry, budget(60))
assert choice is not None
assert choice.zero_spill
assert choice.variant.quant == entry.variants[-1].quant # the Q4 rung
assert choice.reason_key == "best-large-window"
def test_selected_build_constant_and_fit_shape_monotone_in_vram():
"""More VRAM never changes the selected build (always the Q4 rung);
what improves is the fit shape: spilled -> floor -> target window."""
entry = catalog_by_id()["qwen3.8-27b"]
quants = set()
shapes = []
rank = {"smallest-fits-spilled": 0, "best-fits": 1, "best-large-window": 2}
for vram in (8, 12, 16, 24, 32, 48):
choice = select_variant(entry, budget(vram))
assert choice is not None
quants.add(choice.variant.quant)
shapes.append(rank[choice.reason_key])
assert quants == {entry.variants[-1].quant}, f"selection not constant: {quants}"
assert shapes == sorted(shapes), f"fit shape not monotone in VRAM: {shapes}"
def test_small_card_gets_q4_spilled_never_below():
"""8 GiB card + 27B: nothing zero-spills. The floor holds — the
selector offers Q4 spilled (priced honestly), never a sub-Q4 build."""
entry = catalog_by_id()["qwen3.8-27b"]
choice = select_variant(entry, budget(8))
assert choice is not None
assert not choice.zero_spill
assert choice.reason_key == "smallest-fits-spilled"
assert choice.variant.quant == "UD-Q4_K_M"
def test_frontier_model_refused_on_consumer_card_offered_on_big_ram():
"""DeepSeek V4 Flash (161 GB at Q4): refused outright on a 32 GiB-RAM
desktop; offered spilled on a 192 GiB-RAM workstation. The catalog
carries frontier hardware honestly instead of hiding the model."""
entry = catalog_by_id()["deepseek-v4-flash"]
assert select_variant(entry, budget(32, ram_gib=32)) is None
big = select_variant(entry, budget(32, ram_gib=192))
assert big is not None and not big.zero_spill
def test_selection_accounts_for_kv_not_just_weights():
"""The zero-spill check prices weights + KV, not weights alone: give a
machine exactly enough VRAM for the build's weights and the fit must
come back spilled, not zero-spill."""
entry = catalog_by_id()["qwen3.8-27b"]
build = entry.variants[0]
exactly_weights = HardwareBudget(
usable_vram_bytes=build.size_bytes + (100 << 20),
total_device_bytes=build.size_bytes + (100 << 20),
ram_available_bytes=64 * GIB)
choice = select_variant(entry, exactly_weights)
assert choice is not None
assert not choice.zero_spill, "KV cost ignored — weights alone can't zero-spill"
def test_floor_fallback_when_target_window_does_not_fit():
"""Cards where nothing clears the target keep the old rule: highest
quality that zero-spills at the 64K floor (reason 'best-fits'), never
a needless step down."""
entry = catalog_by_id()["qwen3.8-27b"]
# ~23.5 GiB usable: Q4 weights (16.7 GiB in-memory) + floor KV (2.2)
# + overhead (1.5 + 0.9 mmproj + ~1.0 MTP-posture logits) fits, but
# the 144K-target KV (+2.7 more) does not.
choice = select_variant(entry, budget(23.5))
assert choice is not None and choice.zero_spill
assert choice.reason_key == "best-fits"
assert choice.variant.quant == "UD-Q4_K_M"
def test_target_never_degrades_below_floor_choice():
"""The target preference may only IMPROVE the window, never the
floor guarantees: whenever the old floor rule found a zero-spill pick,
the new rule also finds one (possibly a smaller quant, never spill)."""
for entry in CATALOG:
for vram in (8, 12, 16, 24, 32, 48, 96):
choice = select_variant(entry, budget(vram, ram_gib=256))
if choice is None:
continue
# Rule 2: whatever was chosen zero-spill must genuinely clear
# the floor (the selector's own invariant, re-checked).
if choice.zero_spill:
assert choice.reason_key in ("best-large-window", "best-fits")
def test_find_entry_for_model_resolves_split_ids():
hit = find_entry_for_model("DeepSeek-V4-Flash-0731-UD-Q4_K_XL")
assert hit is not None
entry, variant = hit
assert entry.id == "deepseek-v4-flash"
assert variant.quant == "UD-Q4_K_XL"
def test_hybrid_long_context_stays_cheap():
"""The reason Nemotron/Qwen3.6 headline the catalog: their priced
64K-floor KV must be a small fraction of a dense model's."""
from hermes_cli.local_runtime.catalog import FLOOR
from hermes_cli.local_runtime.estimator import ctx_bytes
from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
hybrid = catalog_by_id()["qwen3.6-35b-a3b"]
hybrid_profile = hybrid.profile(hybrid.variants[-1])
# A fully-dense profile of the same layer count and per-layer cost:
# the contract is about LAYER ECONOMICS (recurrent layers pay no
# per-token KV), not about any particular catalog entry.
n_layers = len(hybrid_profile.layers)
dense_profile = ModelProfile(
name="synthetic-dense", weights_bytes=hybrid_profile.weights_bytes,
embd_table_bytes=0, n_ctx_train=hybrid.n_ctx_train,
layers=[(LayerKind.FULL, hybrid.per_layer_f16)] * n_layers)
dense_kv = ctx_bytes(dense_profile, FLOOR)
hybrid_kv = ctx_bytes(hybrid_profile, FLOOR)
# The contract is structural: recurrent layers pay no per-token KV,
# so the hybrid's KV must track its full-attention share (x kv_scale
# for MTP's draft context), not its total layer count.
full = sum(1 for kind, _ in hybrid_profile.layers if kind == LayerKind.FULL)
expected = dense_kv * full / n_layers * hybrid_profile.kv_scale
assert hybrid_kv < dense_kv, "hybrid must be cheaper than dense"
assert abs(hybrid_kv - expected) / expected < 0.25, (
f"hybrid KV ({hybrid_kv:,}) should track its full-attention share "
f"(expected ~{expected:,.0f})")
+410
View File
@@ -0,0 +1,410 @@
"""Context-policy decision-table tests (Rollout 3).
Per the design's verification plan: synthetic per-layer profiles ->
relationships, never exact numbers. Real-model spot checks pin the
estimator to constants measured on real GGUFs (those ARE relationships —
constants with tolerance bands, not change-detecting catalog snapshots).
"""
from __future__ import annotations
import pytest
from hermes_cli.local_runtime.context_policy import (
FLOOR,
SPEED_FLOOR_TOK_S,
GrowthDecision,
WindowDecision,
growth_decision,
initial_window,
ladder,
launch_args,
spill_overrides,
ub_logits_bytes,
)
from hermes_cli.local_runtime.estimator import (
HardwareBudget,
LayerKind,
ModelProfile,
PhysicsRefusal,
ctx_bytes,
kv_dtype_factor,
physics_check,
)
GIB = 1 << 30
KIB = 1024
# ── synthetic profiles (per-layer tuples, per the verification plan) ──
def dense(name="dense-32b", layers=64, per_token_f16=4096, weights_gib=20,
native=128 * 1024) -> ModelProfile:
return ModelProfile(
name=name, weights_bytes=weights_gib * GIB, embd_table_bytes=0,
n_ctx_train=native,
layers=[(LayerKind.FULL, per_token_f16)] * layers)
def hybrid(name="hybrid-30b", full_layers=16, recurrent_layers=48,
per_token_f16=4096, weights_gib=22, native=1024 * 1024) -> ModelProfile:
layers = ([(LayerKind.FULL, per_token_f16)] * full_layers
+ [(LayerKind.RECURRENT, 0)] * recurrent_layers)
return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
embd_table_bytes=0, n_ctx_train=native, layers=layers)
def moe(name="moe-30b", layers=48, per_token_f16=3072, weights_gib=17,
native=256 * 1024) -> ModelProfile:
return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
embd_table_bytes=0, n_ctx_train=native,
layers=[(LayerKind.FULL, per_token_f16)] * layers,
moe=True)
def card(vram_gib, ram_gib=64, uma=False) -> HardwareBudget:
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
total_device_bytes=int(vram_gib * GIB),
ram_available_bytes=int(ram_gib * GIB), uma=uma)
# ── estimator invariants ─────────────────────────────────────
def test_full_attention_linear_in_window():
p = dense()
b32, b64, b128 = (ctx_bytes(p, w * KIB) for w in (32, 64, 128))
assert abs(b64 / b32 - 2) < 0.01
assert abs(b128 / b64 - 2) < 0.01
def test_recurrent_state_constant_in_window():
p = hybrid()
full_share_32 = ctx_bytes(p, 32 * KIB)
full_share_1m = ctx_bytes(p, 1024 * KIB)
# Grows only through the 16 full-attn layers — the recurrent share is
# identical, so the ratio tracks the full-attn ratio exactly.
full_only = ModelProfile(name="x", weights_bytes=0, embd_table_bytes=0,
n_ctx_train=p.n_ctx_train,
layers=[(LayerKind.FULL, 4096)] * 16)
expected_delta = ctx_bytes(full_only, 1024 * KIB) - ctx_bytes(full_only, 32 * KIB)
assert abs((full_share_1m - full_share_32) - expected_delta) <= 1
def test_swa_layers_capped_at_window():
p = ModelProfile(
name="swa", weights_bytes=0, embd_table_bytes=0, n_ctx_train=128 * KIB,
layers=[(LayerKind.SWA, 4096)] * 5 + [(LayerKind.FULL, 4096)] * 1,
swa_window=1024)
small, big = ctx_bytes(p, 4 * KIB), ctx_bytes(p, 32 * KIB)
# Full layer grew 8x; the 5 SWA layers stayed capped at 1024 — total
# growth must land well under the all-full 8x (here ~4.1x).
assert big / small < 0.6 * 8
def test_q8_factor_is_exactly_34_over_64():
assert kv_dtype_factor(True) == pytest.approx(34 / 64)
assert kv_dtype_factor(False) == 1.0
def test_non_fa_fallback_doubles_ctx_cost():
p = dense()
assert ctx_bytes(p, FLOOR, flash_attention=False) == pytest.approx(
ctx_bytes(p, FLOOR, flash_attention=True) * 64 / 34, rel=0.001)
def test_hybrid_vs_dense_100x_class_spread():
"""The whole reason for the per-layer walk: equal-size models, ~100x
per-token spread between classic dense and a mostly-recurrent hybrid."""
d = dense(layers=64, per_token_f16=8192) # 256 KiB/tok class
h = hybrid(full_layers=4, recurrent_layers=60, per_token_f16=8192)
window = 256 * KIB
dense_cost = ctx_bytes(d, window)
hybrid_cost = ctx_bytes(h, window)
assert dense_cost / hybrid_cost > 10
# ── measured-constant spot checks (real models, tolerance bands) ──
def test_measured_dense_4b_per_token():
"""Qwen3-4B: 36 layers x 8 kv-heads x (128+128) x 2B = 144 KiB/tok f16."""
p = ModelProfile(name="qwen3-4b", weights_bytes=0, embd_table_bytes=0,
n_ctx_train=262144,
layers=[(LayerKind.FULL, 8 * 256 * 2)] * 36)
per_token_bytes = ctx_bytes(p, 32 * KIB, flash_attention=False) / (32 * KIB)
assert per_token_bytes == pytest.approx(144 * KIB, rel=0.02)
def test_measured_gdn_27b_per_token_q8():
"""Qwen3.6-27B: 16 full-attn of 64; measured 34.0 KiB/tok @ q8 (B4).
Per-layer f16 = 34 KiB * 64/34 / 16 = 4 KiB."""
per_layer_f16 = 4 * KIB
kv_only = ModelProfile(name="kv", weights_bytes=0, embd_table_bytes=0,
n_ctx_train=262144,
layers=[(LayerKind.FULL, per_layer_f16)] * 16)
per_token_bytes = ctx_bytes(kv_only, 128 * KIB) / (128 * KIB)
assert per_token_bytes == pytest.approx(34 * KIB, rel=0.02)
def test_measured_nemotron_1m_within_band():
"""1M @ q8 measured 3264 MiB KV (B3): ~3.19 KiB/token TOTAL across the
16 full-attn layers -> per-layer f16 ~384 B. Estimator must land in the
measured band, not the dense-formula 100x miss."""
p = hybrid(full_layers=16, recurrent_layers=46, per_token_f16=384,
native=1024 * KIB)
total = ctx_bytes(p, 1024 * KIB)
assert 2.5 * GIB < total < 4.0 * GIB
# ── physics check ────────────────────────────────────────────
def test_physics_refusal_only_past_vram_plus_ram():
p = dense(weights_gib=60)
ok = physics_check(p, card(24, ram_gib=64), FLOOR)
assert ok is None # 60 GiB weights fit in 24+64
refused = physics_check(p, card(24, ram_gib=16), FLOOR)
assert isinstance(refused, PhysicsRefusal)
assert "smaller quant" in refused.message
def test_physics_check_prices_at_floor_not_native():
"""A 1M-native hybrid must not be refused for its native window —
the check prices the floor only."""
p = hybrid(weights_gib=22)
assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
# ── ladder + initial window ──────────────────────────────────
def test_ladder_shape():
rungs = ladder(262144)
assert rungs[0] == FLOOR
assert rungs[-1] == 262144
assert all(a < b for a, b in zip(rungs, rungs[1:]))
# geometric-ish: each step grows, none more than 2x
assert all(b / a <= 2.0 for a, b in zip(rungs, rungs[1:]))
def test_initial_window_never_below_floor_and_never_above_native():
for profile in (dense(), hybrid(), moe(), dense(native=32 * KIB)):
for vram in (8, 16, 24, 32):
d = initial_window(profile, card(vram))
if isinstance(d, WindowDecision):
assert d.window >= min(FLOOR, profile.n_ctx_train)
assert d.window <= profile.n_ctx_train
def test_initial_window_monotone_in_vram():
p = dense()
windows = []
for vram in (8, 12, 16, 24, 32, 48):
d = initial_window(p, card(vram))
assert isinstance(d, WindowDecision)
windows.append(d.window)
assert all(a <= b for a, b in zip(windows, windows[1:]))
def test_flat_curve_reaches_native_where_dense_does_not():
"""Design invariant: equal-size hybrid rides to native spill-free where
the dense model cannot. Hybrid KV priced at the B3 class (~3.2 KiB/tok
total: per-layer f16 384 B x 16 layers)."""
h = hybrid(weights_gib=18, native=1024 * KIB, per_token_f16=384)
d = dense(weights_gib=18, per_token_f16=8192, native=1024 * KIB)
vram = card(24)
dh = initial_window(h, vram)
dd = initial_window(d, vram)
assert isinstance(dh, WindowDecision) and isinstance(dd, WindowDecision)
assert dh.window == 1024 * KIB and not dh.spilled
assert dd.window < 1024 * KIB
def test_dense_on_small_card_holds_floor_and_spills():
"""The deliberate price of the guarantee (design table: dense 32B on
24 GB starts at the floor with a few GiB spilled)."""
d = initial_window(dense(weights_gib=20), card(12))
assert isinstance(d, WindowDecision)
assert d.window == FLOOR
assert d.spilled
def test_uma_budget_caps_the_window_through_physics():
"""Unified memory needs no special context rule: the budget already
encodes the constraint (usable = RAM minus headroom, ram_available=0),
so the ladder stops where weights + KV genuinely stop fitting."""
p = hybrid(weights_gib=8, native=1024 * KIB)
unified = card(38.4, ram_gib=0, uma=True) # 48 GiB machine, 20% headroom
d = initial_window(p, unified)
assert isinstance(d, WindowDecision)
assert not d.spilled, "UMA budget must produce a resident decision"
need = 8 * GIB + ctx_bytes(p, d.window)
assert need <= unified.usable_vram_bytes
# ── growth ───────────────────────────────────────────────────
def _grow(profile, budget, **kw):
defaults = dict(current_window=FLOOR, session_tokens=int(FLOOR * 0.9),
measured_decode_tok_s=40.0, server_idle=True)
defaults.update(kw)
return growth_decision(profile, budget, **defaults)
def test_growth_holds_below_occupancy():
d = _grow(dense(), card(24), session_tokens=int(FLOOR * 0.5))
assert d.action == "hold"
def test_growth_requires_idle_server():
d = _grow(dense(), card(24), server_idle=False)
assert d.action == "hold"
assert "idle" in d.reason
def test_growth_steps_one_rung_and_is_monotone():
p = dense(native=262144)
d = _grow(p, card(32))
assert d.action == "grow"
assert d.next_window > FLOOR
rungs = ladder(262144)
assert d.next_window == rungs[rungs.index(FLOOR) + 1]
def test_growth_stops_at_native():
p = dense(native=128 * KIB)
d = _grow(p, card(48), current_window=128 * KIB,
session_tokens=int(128 * KIB * 0.9))
assert d.action == "compress-default"
def test_speed_floor_flips_default_to_compression():
d = _grow(dense(), card(24), measured_decode_tok_s=SPEED_FLOOR_TOK_S - 2)
assert d.action == "compress-default"
assert "explicit per-session choice" in d.reason
def test_growth_refits_against_live_budget():
"""V3C: a rung that no longer fits (external pressure ate the memory)
is not granted."""
p = dense(weights_gib=20)
starved = card(2, ram_gib=1)
d = _grow(p, starved)
assert d.action == "compress-default"
assert "physics" in d.reason
# ── spill placement + launch args ────────────────────────────
def test_spill_overrides_prefer_expert_and_recurrent_ffn():
assert "exps" in " ".join(spill_overrides(moe()))
assert "ffn" in " ".join(spill_overrides(hybrid()))
assert spill_overrides(dense()) == []
def test_launch_args_contract():
p = moe()
spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
resident = WindowDecision(window=131072, spill_bytes=0, kv_on_gpu=True)
a = launch_args(p, spilled, mtp_capable=True)
assert a[:2] == ["-c", str(FLOOR)] # explicit window, always
assert "q8_0" in a # q8 KV under flash attn
assert "-ot" in a # spill placement
assert "--spec-type" in a # MTP on spilled
# MTP is not gated on spill: resident decode measured +16% at depth 2.
b = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2)
assert "-ot" not in b, "placement is spill-only"
assert "--spec-type" in b, "MTP must run on resident configs too"
assert b[b.index("--spec-draft-n-max") + 1] == "2"
assert "--backend-sampling" in b
assert "--spec-draft-backend-sampling" in b
# Stacking MTP with the large microbatch is a FIT question, decided
# by the caller (presets' posture ladder) and passed as mtp_prefill.
# Default (no headroom proven): decode posture, small ubatch — the
# stacked logits buffers once packed a 32 GiB card 3.9 GiB past a
# fit that ignored them.
assert "-ub" not in b, "default MTP posture stays at the small ubatch"
# Headroom proven: the stacked posture carries the large microbatch
# (measured best on both axes where it fits: 93.3 vs 89.5 tok/s
# decode on Qwen3.8 Q4). ub_logits_bytes must price the same choice.
s = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2,
mtp_prefill=True)
assert "-ub" in s and s[s.index("-ub") + 1] == "2048"
assert "--spec-type" in s
v = 248320
assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
assert ub_logits_bytes(v, mtp_capable=True, mtp_prefill=True) == int(2048 * v * 4 * 1.5)
c = launch_args(p, resident, mtp_capable=False)
assert "-ub" in c and c[c.index("-ub") + 1] == "2048" # prefill hint
assert "--spec-type" not in c
d = launch_args(p, spilled, flash_attention=False, mtp_capable=False)
assert "q8_0" not in d # f16 on non-FA fallback
def test_launch_args_uma_never_pins_tensors():
"""On unified memory, -ot pinning is off even for spilled decisions:
"CPU" and "GPU" are the same silicon, and forcing FFN weights down
the host compute path measures far slower than letting the
allocator place everything. The discrete ~1.75x -ot win does not
transfer. Everything else about the launch shape is identical to
discrete."""
p = moe()
spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
u = launch_args(p, spilled, mtp_capable=False, uma=True)
assert "-ot" not in u, "UMA must never pin tensors to the host path"
assert u[:2] == ["-c", str(FLOOR)] # window contract unchanged
assert "q8_0" in u # KV policy unchanged
# Same call on discrete keeps the pinning — the flag is the ONLY delta.
disc = launch_args(p, spilled, mtp_capable=False, uma=False)
assert "-ot" in disc
assert [x for x in disc if x != "-ot" and not x.startswith("blk")] == \
[x for x in u if x != "-ot" and not x.startswith("blk")]
def test_ub_logits_bytes_prices_the_flag_choice():
"""The logits-buffer price must match the microbatch launch_args
chooses: 2048 x vocab x 4 for non-MTP, 512 x vocab x 4 x 2 for MTP
(draft context doubles it). 248320-vocab receipts: ~1.9 GiB at
ub2048, ~0.95 GiB under MTP."""
v = 248320
assert ub_logits_bytes(v, mtp_capable=False) == 2048 * v * 4
assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
assert ub_logits_bytes(0, mtp_capable=True) == 0 # unknown vocab: no charge
def test_no_refusal_branch_past_physics():
"""Design invariant: anything past the physics check is servable —
initial_window never refuses on its own."""
for vram in (4, 6, 8, 12):
d = initial_window(dense(weights_gib=20), card(vram, ram_gib=64))
assert isinstance(d, WindowDecision)
def test_kv_scale_prices_mtp_draft_context():
"""MTP profiles carry kv_scale > 1 (the draft context's KV share,
calibrated from measured server RSS); ctx_bytes must scale with it so
every consumer — launch fit, catalog rows, growth — prices what the
server actually allocates. Four-point calibration held within
+1.4 GiB conservative, never optimistic."""
import dataclasses
p = moe()
base = ctx_bytes(p, 131072)
scaled = ctx_bytes(dataclasses.replace(p, kv_scale=1.2), 131072)
assert scaled == int(base * 1.2)
# The safety direction: the estimate must never be BELOW measured.
# (Calibration receipts: predicted-measured was +233..+1400 MiB.)
assert scaled > base
@@ -0,0 +1,39 @@
"""The desktop subcommand's --local launch flag.
Local models ship on main behind this flag: `hermes desktop --local` (or
`Hermes.exe --local` directly) shows the local-models GUI surfaces; without
it the desktop hides them all, even when local models are configured. These
tests pin the argparse contract; the pass-through to the Electron argv lives
in cmd_gui's launch paths.
"""
import argparse
from hermes_cli.subcommands.gui import build_gui_parser
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="hermes")
subparsers = parser.add_subparsers(dest="command")
build_gui_parser(subparsers, cmd_gui=lambda args: None)
return parser
def test_local_flag_parses():
args = _parser().parse_args(["desktop", "--local"])
assert args.local is True
def test_local_flag_defaults_off():
args = _parser().parse_args(["desktop"])
assert args.local is False
def test_local_flag_composes_with_build_flags():
args = _parser().parse_args(["desktop", "--local", "--force-build"])
assert args.local is True
assert args.force_build is True
+186
View File
@@ -0,0 +1,186 @@
"""The HF browser: search the firehose, price it roughly, and let any
GGUF become a normal staged model.
Parsing contracts run against canned HF API shapes (no network); route
contracts run against the real FastAPI app with the HF client stubbed."""
from __future__ import annotations
import pytest
from fastapi.testclient import TestClient
from hermes_cli.local_runtime.estimator import HardwareBudget
from hermes_cli.local_runtime.hf_browse import (
HFFileGroup,
HFModelHit,
repo_files,
rough_fit,
search_models,
)
GIB = 1 << 30
@pytest.fixture
def client(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
(tmp_path / ".hermes").mkdir()
from hermes_cli import web_server
test_client = TestClient(web_server.app)
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
return test_client
def _budget(vram_gib, ram_gib=64):
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
total_device_bytes=int(vram_gib * GIB),
ram_available_bytes=int(ram_gib * GIB))
def test_search_parses_hf_hits(monkeypatch):
canned = [
{"id": "unsloth/Qwen3.8-27B-GGUF", "downloads": 872724, "likes": 47,
"lastModified": "2026-08-18", "gated": False},
{"id": "bartowski/whatever-GGUF", "downloads": 5, "likes": 0,
"lastModified": "2026-01-01", "gated": "auto"},
]
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
lambda url: canned)
hits = search_models("qwen")
assert hits[0].repo == "unsloth/Qwen3.8-27B-GGUF"
assert hits[0].downloads == 872724
assert hits[1].gated is True # HF 'auto'-gated counts as gated
def test_repo_files_groups_splits_and_excludes_companions(monkeypatch):
canned = [
{"path": "Qwen3.8-27B-Q4_K_M.gguf", "size": 17 * GIB},
{"path": "mmproj-BF16.gguf", "size": 1 * GIB},
{"path": "UD-Q8/model-00001-of-00002.gguf", "size": 30 * GIB},
{"path": "UD-Q8/model-00002-of-00002.gguf", "size": 12 * GIB},
{"path": "README.md", "size": 1000},
{"path": "dspark-draft-Q8_0.gguf", "size": 9 * GIB},
]
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
lambda url: canned)
groups = repo_files("any/repo")
labels = {g.label: g for g in groups}
assert "Q4_K_M" in labels and labels["Q4_K_M"].total_bytes == 17 * GIB
# Split parts collapse into one group, ordered, summed.
split = next(g for g in groups if len(g.paths) == 2)
assert split.total_bytes == 42 * GIB
assert split.paths[0].endswith("00001-of-00002.gguf")
# Companions (mmproj, draft) are not standalone models.
assert not any("mmproj" in p or "dspark" in p
for g in groups for p in g.paths)
# Largest first.
assert groups[0].total_bytes >= groups[-1].total_bytes
def test_rough_fit_bands():
b = _budget(29.6, ram_gib=64)
assert rough_fit(20 * GIB, b) == "fits-gpu" # + fill-ins under 29.6
assert rough_fit(28 * GIB, b) == "needs-ram" # weights spill
assert rough_fit(120 * GIB, b) == "too-big"
def test_search_route_requires_query_and_maps_errors(client, monkeypatch):
r = client.get("/api/local-models/search", params={"q": " "})
assert r.status_code == 200 and r.json() == {"hits": []}
def boom(q, limit):
raise RuntimeError("HF down")
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse.search_models", boom)
r = client.get("/api/local-models/search", params={"q": "qwen"})
assert r.status_code == 502
def test_browsed_download_stages_and_bounces(client, tmp_path, monkeypatch):
"""A browsed download must land in the machine-scoped models dir and
bounce the router — the seam that makes it a NORMAL model."""
body = b"GGUF" + b"\x00" * 60
class FakeResponse:
headers = {"Content-Length": str(len(body))}
def __init__(self):
self._data = body
def read(self, n=-1):
out, self._data = self._data, b""
return out
def __enter__(self):
return self
def __exit__(self, *a):
return False
monkeypatch.setattr("urllib.request.urlopen",
lambda *a, **k: FakeResponse())
bounced = {}
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
lambda: bounced.setdefault("yes", True))
r = client.post("/api/local-models/download-browsed",
json={"repo": "someone/Some-GGUF",
"paths": ["Some-Model-Q4_K_M.gguf"]})
assert r.status_code == 200
job_id = r.json()["job_id"]
import time as _time
deadline = _time.time() + 10
status = None
while _time.time() < deadline:
status = client.get(f"/api/local-models/jobs/{job_id}").json()
if status["status"] in ("done", "error"):
break
_time.sleep(0.05)
assert status["status"] == "done", status.get("error")
from hermes_cli.local_runtime.bootstrap import models_dir
assert (models_dir() / "Some-Model-Q4_K_M.gguf").exists()
assert bounced.get("yes") is True
def test_browsed_download_rejects_non_gguf(client):
r = client.post("/api/local-models/download-browsed",
json={"repo": "a/b", "paths": ["model.safetensors"]})
assert r.status_code == 422
def test_sideload_links_and_bounces(client, tmp_path, monkeypatch):
src = tmp_path / "My-Local-Model-Q5_K_M.gguf"
src.write_bytes(b"GGUF" + b"\x00" * 32)
bounced = {}
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
lambda: bounced.setdefault("yes", True))
r = client.post("/api/local-models/sideload", json={"path": str(src)})
assert r.status_code == 200
assert r.json()["model_id"] == "My-Local-Model-Q5_K_M"
from hermes_cli.local_runtime.bootstrap import models_dir
dest = models_dir() / src.name
assert dest.exists()
assert bounced.get("yes") is True
# The original must be untouched.
assert src.exists()
# Idempotent: sideloading again short-circuits.
r = client.post("/api/local-models/sideload", json={"path": str(src)})
assert r.json().get("already_present") is True
def test_sideload_rejects_non_gguf(client, tmp_path):
src = tmp_path / "model.bin"
src.write_bytes(b"nope")
r = client.post("/api/local-models/sideload", json={"path": str(src)})
assert r.status_code == 422
+260
View File
@@ -0,0 +1,260 @@
"""Model-load progress: SSE events -> composite percent -> wait notices.
The 40-second problem: a cold local model streams 16-21 GB of weights
before the first token, and the chat rendered that as the generic
"provider may be slow or overloaded" stall warning. llama-server's child
emits real per-tensor progress which the router relays over /models/sse
ONLY — these tests pin the consumer that turns that stream into the
status route's `loading` field and the chat's load notice."""
from __future__ import annotations
import json
import time
import hermes_cli.local_runtime.load_progress as lp
def setup_function(_fn):
with lp._lock:
lp._snapshot.clear()
# ── composite percent ────────────────────────────────────────
def test_composite_percent_text_stage_dominates():
stages = ["text_model", "spec_model", "mmproj_model"]
# Text model owns [0, 85): halfway through it reads ~42%.
assert lp._composite_percent(stages, "text_model", 0.5) == 42 # 0.5*85
# Extras start where text ends and never regress below it.
assert lp._composite_percent(stages, "spec_model", 0.0) == 85
assert lp._composite_percent(stages, "mmproj_model", 1.0) == 100
def test_composite_percent_monotone_across_stage_walk():
"""Walking the stages in llama-server's real order never moves the
bar backwards — the property that makes the bar trustworthy."""
stages = ["text_model", "spec_model", "mmproj_model"]
walk = [("text_model", v / 10) for v in range(11)] + \
[("spec_model", v / 10) for v in range(11)] + \
[("mmproj_model", v / 10) for v in range(11)]
seen = [lp._composite_percent(stages, s, v) for s, v in walk]
assert seen == sorted(seen)
assert seen[0] == 0 and seen[-1] == 100
def test_composite_percent_single_stage_is_plain():
assert lp._composite_percent(["text_model"], "text_model", 0.4) == 40
# ── event application ────────────────────────────────────────
def _loading_event(value: float, current: str = "text_model") -> dict:
return {"status": "loading",
"progress": {"stages": ["text_model", "mmproj_model"],
"current": current, "value": value}}
def test_loading_events_build_snapshot_and_terminal_clears():
lp._apply_event("m1", "status_change", _loading_event(0.5))
snap = lp.get_loading_progress()
assert "m1" in snap
assert snap["m1"]["percent"] == 42 # 0.5 * 85 within text stage
assert snap["m1"]["stage"] == "text_model"
lp._apply_event("m1", "status_change", {"status": "loaded", "info": {}})
assert lp.get_loading_progress() == {}
def test_unload_and_failure_clear_too():
lp._apply_event("m1", "status_change", _loading_event(0.2))
lp._apply_event("m1", "status_change", {"status": "unloaded", "exit_code": 1})
assert lp.get_loading_progress() == {}
lp._apply_event("m2", "status_change", _loading_event(0.9))
lp._apply_event("m2", "model_remove", {})
assert lp.get_loading_progress() == {}
def test_progressless_loading_event_keeps_entry_alive():
"""The router's first model_status event says just {status: loading} —
it must register the load (indeterminate) without inventing a percent."""
lp._apply_event("m1", "model_status", {"status": "loading"})
snap = lp.get_loading_progress()
assert snap["m1"]["percent"] == 0
def test_stale_entries_expire():
lp._apply_event("m1", "status_change", _loading_event(0.5))
with lp._lock:
lp._snapshot["m1"]["ts"] -= lp._STALE_ENTRY_TTL_S + 1
assert lp.get_loading_progress() == {}
# ── chat wait-notice ─────────────────────────────────────────
def test_load_notice_for_managed_model(tmp_path, monkeypatch):
from agent.chat_completion_helpers import _managed_local_load_notice
state = tmp_path / "server.json"
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
"api_key": "k"}), encoding="utf-8")
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
lambda: state)
lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
class _Agent:
base_url = "http://127.0.0.1:18434/v1"
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
assert notice is not None
assert notice.startswith("⏳ loading Qwen-Test into memory — 42%")
# Different endpoint (user's own server): never claim its loads.
class _Other:
base_url = "http://127.0.0.1:9999/v1"
assert _managed_local_load_notice(_Other(), {"model": "Qwen-Test"}) is None
# Managed endpoint but a model that isn't loading: no notice.
assert _managed_local_load_notice(_Agent(), {"model": "Elsewhere"}) is None
def test_load_notice_matches_desktop_wait_filter():
"""The notices must pass the desktop's providerWaitText regex and parse
under parseModelLoadWait's shapes — pinned here as plain string
contracts so the two sides can't drift silently."""
import re
accept = r"^(?:⏳|⚠|↻|⚙)\s*(?:waiting on|loading|processing prompt|no (?:output|response)|model returned)"
load = "⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 43% (responses start once the model is loaded)"
assert re.match(accept, load)
m = re.match(r"^⏳\s*loading\s+(.+?)\s+into memory\s+—\s+(\d{1,3})%", load)
assert m and m.group(1) == "Qwen3.6-35B-A3B-UD-Q4_K_M" and m.group(2) == "43"
prefill = "⚙ processing prompt — 31%"
assert re.match(accept, prefill)
p = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", prefill)
assert p and p.group(1) == "31"
bare = "⚙ processing prompt"
assert re.match(accept, bare)
b = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", bare)
assert b and b.group(1) is None
# ── prefill progress ─────────────────────────────────────────
def test_prefill_notice_for_managed_model(tmp_path, monkeypatch):
from agent.chat_completion_helpers import _managed_local_load_notice
state = tmp_path / "server.json"
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
"api_key": "k"}), encoding="utf-8")
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
lambda: state)
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
# No load in flight; a prefill counter is live.
monkeypatch.setattr(lp, "get_prefill_progress",
lambda model: {"processed": 12288})
import agent.chat_completion_helpers as cch
monkeypatch.setattr(cch, "estimate_request_context_tokens",
lambda kw: 39551)
class _Agent:
base_url = "http://127.0.0.1:18434/v1"
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
assert notice == "⚙ processing prompt — 31%"
# Counter past the estimate (estimator undercounted): no honest
# denominator, so no percent — never >100%.
monkeypatch.setattr(cch, "estimate_request_context_tokens", lambda kw: 100)
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
assert notice == "⚙ processing prompt"
def test_load_notice_outranks_prefill(tmp_path, monkeypatch):
"""While a load entry exists the load notice wins — prefill can't start
before the model is resident, so a simultaneous claim means the load
snapshot is authoritative."""
from agent.chat_completion_helpers import _managed_local_load_notice
state = tmp_path / "server.json"
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
"api_key": "k"}), encoding="utf-8")
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
lambda: state)
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
monkeypatch.setattr(lp, "get_prefill_progress",
lambda model: {"processed": 999})
class _Agent:
base_url = "http://127.0.0.1:18434/v1"
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
assert notice is not None and notice.startswith("⏳ loading")
def test_prefill_progress_reads_busiest_processing_slot(monkeypatch):
monkeypatch.setattr(lp, "_endpoint", lambda: ("http://127.0.0.1:1", "k"))
class _Resp:
def __init__(self, payload):
self._payload = payload
def read(self):
return json.dumps(self._payload).encode()
def __enter__(self):
return self
def __exit__(self, *a):
return False
slots = [
{"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500},
{"id": 1, "is_processing": True, "n_prompt_tokens_processed": 42},
{"id": 2, "is_processing": True, "n_prompt_tokens_processed": 32768},
]
monkeypatch.setattr(lp.urllib.request, "urlopen",
lambda req, timeout=0: _Resp(slots))
assert lp.get_prefill_progress("m") == {"processed": 32768}
# Nothing processing -> None (idle slots' counters are leftovers).
idle = [{"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500}]
monkeypatch.setattr(lp.urllib.request, "urlopen",
lambda req, timeout=0: _Resp(idle))
assert lp.get_prefill_progress("m") is None
# Unreachable server -> None, never an exception.
def _boom(req, timeout=0):
raise OSError("refused")
monkeypatch.setattr(lp.urllib.request, "urlopen", _boom)
assert lp.get_prefill_progress("m") is None
def test_endpoint_respects_ownership_guard(monkeypatch):
"""The watcher's endpoint MUST come from the ownership-guarded reader.
Regression: a raw state-file read attached the SSE watcher to a
foreign install's server on the shared stable port (health answers
for anyone; only the dead-pid check proves ownership)."""
import hermes_cli.local_runtime.load_progress as lp
# Guard says "not ours": no endpoint, regardless of state on disk.
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
lambda: None)
assert lp._endpoint() is None
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint._state_endpoint",
lambda: {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"})
assert lp._endpoint() == ("http://127.0.0.1:18434", "k")
@@ -0,0 +1,213 @@
"""Abandoned-request lifecycle: work sent to the managed local server must
die when its caller goes away, and teardown must never orphan VRAM.
The incident this guards: auxiliary calls (title generation + retries)
queued at the router behind a cold model load, their clients timed out
and hung up, and the router then dispatched them anyway. Non-streamed
responses write the socket only after the FULL generation, so nothing
noticed the dead clients — two uncapped decodes ran at full GPU for the
better part of an hour with nobody listening.
Three contracts, one per failure link:
1. Auxiliary requests to the managed local endpoint are always streamed
(a dead client then cancels decode at the first chunk write).
2. Explicit caller max_tokens caps reach the managed local endpoint
(title generation's 64-token cap must not be silently dropped).
3. Supervisor teardown terminates the whole process tree, and a router
respawn reaps orphaned model children first (each holds GiB of VRAM).
"""
from __future__ import annotations
import json
import subprocess
import types
import pytest
import agent.auxiliary_client as aux
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
MANAGED_URL = "http://127.0.0.1:18434/v1"
@pytest.fixture
def managed_state(tmp_path, monkeypatch):
"""A supervisor state file declaring the managed endpoint, cache reset."""
state = tmp_path / "server.json"
state.write_text(json.dumps({"base_url": MANAGED_URL, "api_key": "k",
"pid": 4242}), encoding="utf-8")
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
lambda: state)
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
return state
# ── 1. managed endpoint is always streamed ───────────────────
def test_managed_endpoint_requires_stream(managed_state):
assert aux._provider_requires_stream("custom", MANAGED_URL) is True
def test_managed_detection_matches_netloc_not_substring(managed_state):
# Same host, different port: a user's own external server — untouched.
assert aux._provider_requires_stream("custom", "http://127.0.0.1:8080/v1") is False
def test_no_state_file_means_no_managed_endpoint(tmp_path, monkeypatch):
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
lambda: tmp_path / "absent.json")
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
assert aux._is_managed_local_endpoint(MANAGED_URL) is False
def test_remote_providers_unaffected(managed_state):
assert aux._provider_requires_stream("nous",
"https://inference-api.nousresearch.com/v1/") is False
# ── 2. explicit caps reach the managed endpoint ──────────────
def test_explicit_max_tokens_forwarded_to_managed_local(managed_state, monkeypatch):
monkeypatch.setattr(aux, "_current_custom_base_url", lambda: MANAGED_URL)
kwargs = aux._build_call_kwargs(
"custom", "Qwen-Local", [{"role": "user", "content": "hi"}],
max_tokens=64, timeout=30.0, task="title_generation")
assert kwargs.get("max_tokens") == 64 or kwargs.get("max_completion_tokens") == 64, (
"explicit caller cap dropped on the managed local endpoint — an "
"EOS-less generation then runs to the full context window")
def test_no_default_cap_policy_unchanged_for_remote(monkeypatch):
# A generic remote provider still drops the cap (the forwarding gate
# is an allow-list). openrouter no longer qualifies as the example
# here: main forwards its caps deliberately (#41035, 402 affordability).
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
kwargs = aux._build_call_kwargs(
"openai", "some/model", [{"role": "user", "content": "hi"}],
max_tokens=64, timeout=30.0)
assert "max_tokens" not in kwargs and "max_completion_tokens" not in kwargs
# ── 3. teardown kills the tree; respawn reaps orphans ────────
class _FakeChild:
def __init__(self, pid):
self.pid = pid
self.terminated = False
self.killed = False
def terminate(self):
self.terminated = True
def is_running(self):
return not self.terminated and not self.killed
def kill(self):
self.killed = True
def test_terminate_tree_terminates_children_too(monkeypatch):
children = [_FakeChild(101), _FakeChild(102)]
class _FakeParentProc:
def __init__(self, pid):
self.pid = pid
def children(self, recursive=False):
assert recursive is True
return children
fake_psutil = types.SimpleNamespace(Process=_FakeParentProc)
monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
class _FakeRouter:
pid = 4242
terminated = False
def terminate(self):
_FakeRouter.terminated = True
def wait(self, timeout=None):
return 0
def poll(self):
return None
LlamaServerSupervisor._terminate_tree(_FakeRouter())
assert _FakeRouter.terminated
assert all(c.terminated for c in children), (
"router children orphaned on stop — each holds GiB of VRAM")
def test_terminate_tree_survives_missing_psutil(monkeypatch):
import builtins
real_import = builtins.__import__
def _no_psutil(name, *a, **k):
if name == "psutil":
raise ImportError("nope")
return real_import(name, *a, **k)
monkeypatch.setattr(builtins, "__import__", _no_psutil)
class _FakeRouter:
pid = 4242
terminated = False
def terminate(self):
_FakeRouter.terminated = True
def wait(self, timeout=None):
return 0
LlamaServerSupervisor._terminate_tree(_FakeRouter())
assert _FakeRouter.terminated # router still stopped without psutil
def test_reap_orphans_kills_only_our_parentless_binaries(tmp_path, monkeypatch):
exe = tmp_path / "llama-server.exe"
exe.write_text("")
orphan = _FakeChild(300)
adopted = _FakeChild(301) # parent alive -> not an orphan
foreign = _FakeChild(302) # different binary -> never touched
def _info(pid, exe_path, ppid):
p = _FakeChild(pid)
p.info = {"exe": exe_path, "ppid": ppid}
return p
procs = [
_info(300, str(exe), 9999), # dead parent -> reap
_info(301, str(exe), 1), # live parent -> keep
_info(302, str(tmp_path / "other.exe"), 9999), # foreign -> keep
]
reaped = []
for p in procs:
p.kill = lambda p=p: reaped.append(p.info and p.pid)
class _NoSuch(Exception):
pass
fake_psutil = types.SimpleNamespace(
process_iter=lambda attrs: procs,
pid_exists=lambda pid: pid == 1,
NoSuchProcess=_NoSuch,
AccessDenied=_NoSuch,
)
monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.server_binary",
lambda install_dir: exe)
sup = LlamaServerSupervisor.__new__(LlamaServerSupervisor)
sup.install_dir = tmp_path
sup.proc = None
sup._reap_orphaned_children()
assert reaped == [300], f"reaped {reaped}; wanted only the orphan (300)"
@@ -0,0 +1,120 @@
"""Context-length resolution for the managed llama.cpp router.
The incident: the statusbar showed 131K for a local model the server had
granted 262144 tokens. The router reports ``meta: null`` on /v1/models
for a model that is not currently LOADED (models autoload on first chat,
so at session start the model is routinely unloaded), and /v1/models/{id}
404s — every metadata probe missed, resolution fell through to the
name-pattern defaults, and the "qwen" family catch-all (131072) shipped
as the compressor's budget and the statusbar's denominator.
Contract: for a llama.cpp server, /props default_generation_settings.n_ctx
(the preset-backed RUNTIME window, served even for unloaded models) is
the authority, probed before the /v1/models fallbacks.
"""
from __future__ import annotations
import http.server
import json
import threading
import pytest
import agent.model_metadata as mm
GRANTED = 262144
@pytest.fixture
def router():
"""Stub of the llama-server router with the model UNLOADED:
/v1/models carries meta=null; /props answers from the preset."""
class _Router(http.server.BaseHTTPRequestHandler):
def do_GET(self):
if self.path.startswith("/props"):
body = {"default_generation_settings": {"n_ctx": GRANTED}}
elif self.path == "/v1/models":
body = {"data": [{
"id": "Qwen-Test-UD-Q4_K_M",
"owned_by": "llamacpp",
"meta": None,
"status": {"value": "unloaded"},
}]}
else: # /v1/models/{id} -> 404, as the real router answers
self.send_response(404)
self.send_header("Content-Length", "0")
self.end_headers()
return
raw = json.dumps(body).encode()
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
self.wfile.write(raw)
def log_message(self, *a):
pass
server = http.server.HTTPServer(("127.0.0.1", 0), _Router)
threading.Thread(target=server.serve_forever, daemon=True).start()
yield f"http://127.0.0.1:{server.server_address[1]}/v1"
server.shutdown()
def test_unloaded_llamacpp_model_resolves_granted_window(router, monkeypatch):
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
assert ctx == GRANTED, (
f"resolved {ctx}; an unloaded model must resolve the preset window "
"from /props, not fall through to name-pattern catch-alls")
def test_props_beats_meta_when_model_loaded(router, monkeypatch):
"""/props is probed first even when /v1/models would answer: n_ctx from
/props is the same runtime value, and probing it first keeps loaded and
unloaded models on one code path."""
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
assert ctx == GRANTED
def test_non_llamacpp_servers_skip_props(monkeypatch):
"""Ollama/LM Studio/vLLM keep their existing probe order — /props is
llama.cpp-shaped and must not be consulted for other server types."""
calls = []
class _FakeResp:
status_code = 404
def json(self):
return {}
class _FakeClient:
def __init__(self, *a, **k):
pass
def __enter__(self):
return self
def __exit__(self, *a):
return False
def get(self, url):
calls.append(url)
return _FakeResp()
def post(self, url, **k):
calls.append(url)
return _FakeResp()
import httpx
monkeypatch.setattr(httpx, "Client", _FakeClient)
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "vllm")
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
mm._query_local_context_length_uncached("m", "http://127.0.0.1:9999/v1")
assert not any("/props" in u for u in calls)
+280
View File
@@ -0,0 +1,280 @@
"""In-session growth contracts (growth.py + the presets override seam).
The live half of the window ladder: grow before compress, overrides
persist across boots, physics re-checked every boot, growth state dies
with the model."""
from __future__ import annotations
import pytest
@pytest.fixture
def hermes_home(tmp_path, monkeypatch):
home = tmp_path / ".hermes"
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
return home
def test_overrides_roundtrip_and_clear(hermes_home):
from hermes_cli.local_runtime.growth import (
clear_window_override,
load_window_overrides,
save_window_override,
)
assert load_window_overrides() == {}
save_window_override("model-a", 98304)
save_window_override("model-b", 262144)
assert load_window_overrides() == {"model-a": 98304, "model-b": 262144}
clear_window_override("model-a")
assert load_window_overrides() == {"model-b": 262144}
# Clearing a missing key is a no-op, not an error.
clear_window_override("never-existed")
def test_corrupt_overrides_read_as_empty(hermes_home):
from hermes_cli.local_runtime.growth import (
load_window_overrides,
window_overrides_path,
)
path = window_overrides_path()
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text("{not json", encoding="utf-8")
assert load_window_overrides() == {}
def test_growth_declines_foreign_endpoints(hermes_home):
"""Only the server THIS process supervises grows — a detected external
server or another process's endpoint returns None untouched."""
from hermes_cli.local_runtime.growth import maybe_grow_window
grown = maybe_grow_window(
"some-model", base_url="http://127.0.0.1:9999/v1",
session_tokens=100_000, current_window=65536)
assert grown is None
def test_occupancy_confirmed_skips_gate_one():
"""The agent's compression gate IS the occupancy signal: when it fired,
growth must not re-derive its own edge and hold. Decision-table check
with a synthetic profile."""
from hermes_cli.local_runtime.context_policy import growth_decision
from hermes_cli.local_runtime.estimator import (
HardwareBudget,
LayerKind,
ModelProfile,
)
gib = 1 << 30
profile = ModelProfile(
name="m", weights_bytes=2 * gib, embd_table_bytes=0,
n_ctx_train=262144,
layers=[(LayerKind.FULL, 4096)] * 16 + [(LayerKind.RECURRENT, 0)] * 48)
budget = HardwareBudget(usable_vram_bytes=26 * gib,
total_device_bytes=32 * gib,
ram_available_bytes=64 * gib)
# Hermes' threshold (e.g. 80% of window) can sit BELOW the ladder's 85%
# occupancy gate: 78K of a 96K window is 81%.
kwargs = dict(current_window=98304, session_tokens=78_000,
measured_decode_tok_s=None, server_idle=True)
ungated = growth_decision(profile, budget, **kwargs)
assert ungated.action == "hold", "sanity: below the ladder's own gate"
confirmed = growth_decision(profile, budget, occupancy_confirmed=True, **kwargs)
assert confirmed.action == "grow"
assert confirmed.next_window and confirmed.next_window > 98304
def _stage_fake_gguf(mdir, name):
mdir.mkdir(parents=True, exist_ok=True)
(mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 64)
def _header_stub(sampling: dict | None = None):
"""A read_gguf_header stand-in for tests that monkeypatch the reader:
just enough surface for preset generation (sampling ladder included)."""
class _Stub:
sampling_defaults = dict(sampling or {})
return _Stub()
def _tiny_profile(model_id: str):
from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
gib = 1 << 30
return ModelProfile(
name=model_id, weights_bytes=2 * gib, embd_table_bytes=0,
n_ctx_train=131072,
layers=[(LayerKind.FULL, 512)] * 4)
def test_preset_generation_for_catalog_model_with_mmproj(hermes_home, tmp_path, monkeypatch):
"""generate_presets must survive a model that IS in the catalog and
carries a vision projector — this executes the find_entry_for_model +
mmproj overhead branch that synthetic test models skip. Regression:
the branch once treated the (entry, variant) tuple as the entry and
crashed every real boot into the stock-fit fallback."""
import hermes_cli.local_runtime.presets as presets_mod
from hermes_cli.local_runtime.catalog import CATALOG
from hermes_cli.local_runtime.estimator import HardwareBudget
# A real catalog id with an mmproj (the recommended row has one).
entry = next(e for e in CATALOG if e.mmproj is not None)
variant = entry.variants[-1]
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, variant.model_id)
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(presets_mod, "profile_from_gguf",
lambda h: _tiny_profile(variant.model_id))
gib = 1 << 30
budget = HardwareBudget(usable_vram_bytes=24 * gib,
total_device_bytes=24 * gib,
ram_available_bytes=64 * gib)
entries = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")
assert len(entries) == 1
assert entries[0].refusal is None
assert entries[0].window > 0
def test_preset_restores_grown_window_capped_at_native(hermes_home, tmp_path, monkeypatch):
"""A persisted override lifts the preset window; an absurd override is
capped at native. GGUF parsing is stubbed — the contract under test is
the override plumbing, not the reader."""
import hermes_cli.local_runtime.presets as presets_mod
from hermes_cli.local_runtime.estimator import HardwareBudget
from hermes_cli.local_runtime.growth import save_window_override
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, "tiny-dense")
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(presets_mod, "profile_from_gguf",
lambda h: _tiny_profile("tiny-dense"))
gib = 1 << 30
budget = HardwareBudget(usable_vram_bytes=24 * gib,
total_device_bytes=24 * gib,
ram_available_bytes=64 * gib)
preset = tmp_path / "presets.ini"
baseline = presets_mod.generate_presets(mdir, budget, preset)[0]
assert baseline.window == 131072 # tiny model: native from the start
# Override above native must cap at native, not exceed it.
save_window_override("tiny-dense", 10_000_000)
capped = presets_mod.generate_presets(mdir, budget, preset)[0]
assert capped.window == 131072
def test_preset_ignores_override_below_launch_window(hermes_home, tmp_path, monkeypatch):
"""Overrides only ever RAISE the window (growth is monotone); a stale
smaller override never shrinks a launch decision."""
import hermes_cli.local_runtime.presets as presets_mod
from hermes_cli.local_runtime.estimator import HardwareBudget
from hermes_cli.local_runtime.growth import save_window_override
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, "tiny-dense")
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(presets_mod, "profile_from_gguf",
lambda h: _tiny_profile("tiny-dense"))
save_window_override("tiny-dense", 65536)
gib = 1 << 30
budget = HardwareBudget(usable_vram_bytes=24 * gib,
total_device_bytes=24 * gib,
ram_available_bytes=64 * gib)
entry = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")[0]
assert entry.window == 131072
def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypatch):
"""The real growth shape: launch at a lower rung, override to a middle
rung -> the preset window follows the override."""
import hermes_cli.local_runtime.presets as presets_mod
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile
from hermes_cli.local_runtime.growth import save_window_override
gib = 1 << 30
# Expensive dense KV so the launch decision lands BELOW native on this
# budget: 60 layers x 4 KiB/tok f16 -> q8 ~= 120 KiB/tok.
profile = ModelProfile(
name="big-dense", weights_bytes=20 * gib, embd_table_bytes=0,
n_ctx_train=262144,
layers=[(LayerKind.FULL, 4096)] * 60)
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, "big-dense")
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(presets_mod, "profile_from_gguf", lambda h: profile)
budget = HardwareBudget(usable_vram_bytes=28 * gib,
total_device_bytes=32 * gib,
ram_available_bytes=128 * gib)
baseline = presets_mod.generate_presets(mdir, budget, tmp_path / "a.ini")[0]
assert baseline.window < 262144, "sanity: launch below native"
grown = baseline.window * 2
save_window_override("big-dense", grown)
restored = presets_mod.generate_presets(mdir, budget, tmp_path / "b.ini")[0]
assert restored.window >= grown, "override must lift the launch window"
def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
"""The sampling deference ladder: the GGUF's own general.sampling.*
wins per key, catalog fills only what the file left silent, and a
model with neither gets no sampling keys at all (llama.cpp defaults).
Policy keys (ctx-size, cache types) must never be displaced."""
import configparser
import hermes_cli.local_runtime.presets as presets_mod
from hermes_cli.local_runtime.catalog import CATALOG
from hermes_cli.local_runtime.estimator import HardwareBudget
# A real catalog entry WITH catalog sampling, staged on disk.
entry = next(e for e in CATALOG if e.sampling)
variant = entry.variants[-1]
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, variant.model_id)
_stage_fake_gguf(mdir, "off-catalog-model")
gib = 1 << 30
budget = HardwareBudget(usable_vram_bytes=64 * gib,
total_device_bytes=64 * gib,
ram_available_bytes=64 * gib)
# The catalog model's file carries temp; catalog must fill the rest
# but NOT displace the file's value. The off-catalog file carries none.
def fake_header(path):
if variant.model_id in str(path):
return _header_stub({"temp": "0.42"})
return _header_stub()
monkeypatch.setattr(presets_mod, "read_gguf_header", fake_header)
monkeypatch.setattr(presets_mod, "profile_from_gguf",
lambda h: _tiny_profile("x"))
out = tmp_path / "presets.ini"
presets_mod.generate_presets(mdir, budget, out)
ini = configparser.ConfigParser()
ini.read(out)
sec = ini[variant.model_id]
assert sec["temp"] == "0.42", "file's own sampling must win per key"
for k, v in entry.sampling.items():
if k != "temp":
assert sec[k] == v, f"catalog must fill the silent key {k}"
assert "ctx-size" in sec, "policy keys survive the ladder"
off = ini["off-catalog-model"]
assert "temp" not in off and "top-p" not in off, (
"no file keys + no catalog entry = llama.cpp defaults, not ours")
@@ -0,0 +1,293 @@
"""Contract tests for the local-models dashboard routes (Rollout 4).
Real FastAPI TestClient against the real router; the runtime pieces
underneath are exercised against temp HERMES_HOME (autouse fixture). Network
downloads are stubbed at the urllib boundary — never live."""
from __future__ import annotations
import io
import json
import time
from pathlib import Path
import pytest
from fastapi.testclient import TestClient
@pytest.fixture
def client(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli import web_server
test_client = TestClient(web_server.app)
# Same auth pattern as the git-route tests: present the session token.
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
return test_client
def test_local_models_routes_require_auth(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli import web_server
unauth = TestClient(web_server.app)
assert unauth.get("/api/local-models/status").status_code == 401
def _write_fake_gguf(path: Path, size: int = 1024) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(b"GGUF" + b"\x00" * size)
# ── status ───────────────────────────────────────────────────
def test_status_shape_and_defaults(client):
r = client.get("/api/local-models/status")
assert r.status_code == 200
data = r.json()
# Contract: every key the pane's first paint needs, present and typed.
assert isinstance(data["enabled"], bool)
assert isinstance(data["tag"], str) and data["tag"].startswith("b")
assert isinstance(data["runtime_installed"], bool)
assert isinstance(data["server_running"], bool)
assert isinstance(data["models"], list)
def test_status_lists_staged_models_with_labels(client, tmp_path):
from hermes_cli.local_runtime.bootstrap import models_dir
_write_fake_gguf(models_dir() / "Some-Model.gguf", size=2048)
data = client.get("/api/local-models/status").json()
ids = [m["id"] for m in data["models"]]
assert "Some-Model" in ids
row = data["models"][ids.index("Some-Model")]
assert row["size_bytes"] > 0
assert row["size_label"].endswith("GB")
# ── hardware ─────────────────────────────────────────────────
def test_hardware_plain_facts(client):
data = client.get("/api/local-models/hardware").json()
assert isinstance(data["uma"], bool)
assert data["ram_total_bytes"] > 0
assert data["vram_total_bytes"] >= 0
# GPU fields are None-able (non-NVIDIA machines) but must exist.
assert "gpu_name" in data and "gpu_util_percent" in data and "vram_used_bytes" in data
# ── catalog ──────────────────────────────────────────────────
def test_catalog_prices_every_entry_for_this_machine(client):
data = client.get("/api/local-models/catalog").json()
assert len(data["models"]) >= 3
for row in data["models"]:
# The three user questions, answered on every row:
assert row["size_label"].endswith("GB") # how big
assert isinstance(row["fits"], bool) # will it fit
assert row["fit_summary"] # what shape
if row["fits"]:
assert row["start_window"] >= 1
assert row["start_window_label"].endswith("K")
else:
assert "memory" in row["fit_summary"].lower()
assert isinstance(row["downloaded"], bool)
def test_catalog_never_hides_unaffordable_models(client, monkeypatch):
"""Unaffordable entries stay visible with a plain reason — hiding them
is how users conclude the feature is broken."""
from hermes_cli.local_runtime.estimator import HardwareBudget
tiny = HardwareBudget(usable_vram_bytes=1 << 30, total_device_bytes=1 << 30,
ram_available_bytes=1 << 30)
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
lambda **kw: tiny)
data = client.get("/api/local-models/catalog").json()
from hermes_cli.local_runtime.catalog import CATALOG
assert len(data["models"]) == len(CATALOG)
refused = [m for m in data["models"] if not m["fits"]]
assert refused, "a 1 GiB machine must refuse the 20 GB models"
for row in refused:
assert row["fit_detail"] or row["fit_summary"]
# ── downloads ────────────────────────────────────────────────
def test_download_unknown_model_404s(client):
r = client.post("/api/local-models/download", json={"model_id": "nope"})
assert r.status_code == 404
def test_download_short_of_server_length_errors_and_cleans_up(client, monkeypatch):
"""Catalog sizes are advisory (upstream re-uploads may make them
stale — a mismatch against the CATALOG must not fail a download).
The server's own declared length is the only completeness check:
fewer bytes than the server promised means a dropped connection, so
the job errors and nothing is staged."""
class FakeResponse(io.BytesIO):
# Body is 17 bytes; the server promises 32 — a truncated stream.
headers = {"Content-Length": "32"}
def __enter__(self):
return self
def __exit__(self, *a):
return False
monkeypatch.setattr("urllib.request.urlopen",
lambda *a, **k: FakeResponse(b"not the real body"))
# Pin a generous budget: variant selection prices against the machine
# running the test, and a GPU-less CI runner honestly refuses every
# build (409) — this test is about the download path, not selection.
from hermes_cli.local_runtime.estimator import HardwareBudget
budget = HardwareBudget(usable_vram_bytes=64 << 30,
total_device_bytes=64 << 30,
ram_available_bytes=64 << 30)
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
lambda **kw: budget)
from hermes_cli.local_runtime.catalog import CATALOG
entry_id = CATALOG[0].id
r = client.post("/api/local-models/download", json={"model_id": entry_id})
assert r.status_code == 200
job_id = r.json()["job_id"]
assert job_id
deadline = time.time() + 10
status = None
while time.time() < deadline:
status = client.get(f"/api/local-models/jobs/{job_id}").json()
if status["status"] in ("done", "error"):
break
time.sleep(0.05)
assert status is not None and status["status"] == "error"
assert "bytes" in status["error"].lower()
from hermes_cli.local_runtime.bootstrap import models_dir
assert not (models_dir() / f"{entry_id}.gguf").exists()
assert not (models_dir() / f"{entry_id}.part").exists()
def test_download_already_downloaded_short_circuits(client, monkeypatch):
from hermes_cli.local_runtime.bootstrap import models_dir
from hermes_cli.local_runtime.catalog import CATALOG, select_variant
from hermes_cli.local_runtime.estimator import HardwareBudget
# Pin the budget so the selected variant is deterministic in the test.
budget = HardwareBudget(usable_vram_bytes=64 << 30, total_device_bytes=64 << 30,
ram_available_bytes=64 << 30)
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
lambda **kw: budget)
choice = select_variant(CATALOG[0], budget)
assert choice is not None
_write_fake_gguf(models_dir() / choice.variant.files[0].local_name)
r = client.post("/api/local-models/download", json={"model_id": CATALOG[0].id})
assert r.status_code == 200
assert r.json()["already_downloaded"] is True
def test_delete_model(client):
from hermes_cli.local_runtime.bootstrap import models_dir
_write_fake_gguf(models_dir() / "Doomed.gguf")
assert client.delete("/api/local-models/models/Doomed").status_code == 200
assert not (models_dir() / "Doomed.gguf").exists()
assert client.delete("/api/local-models/models/Doomed").status_code == 404
# ── runtime install ──────────────────────────────────────────
def test_runtime_install_rejects_impossible_combo(client, monkeypatch):
"""Impossible platform/backend combos fail the POST itself with the
resolver's honest message — not a background job that dies silently.
(win-arm64-vulkan; the old cuda case became real upstream at ~b1036x.)"""
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries._host_os_arch", lambda: ("win", "arm64"))
r = client.post("/api/local-models/runtime/install", json={"backend": "vulkan"})
assert r.status_code == 400
assert "arm64" in r.json()["detail"]
def test_job_poll_unknown_404s(client):
assert client.get("/api/local-models/jobs/deadbeef").status_code == 404
def test_eject_without_supervisor_is_not_a_500(client, monkeypatch):
"""Eject on an ADOPTED server (no in-process supervisor — the shape
every backend restart produces, since boot adopts the running server
via the state file) must route through the persisted endpoint, not
crash. Regression: _state_endpoint was only imported inside the
status route, so eject raised NameError -> 500 for every adopted-
server session."""
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.get_supervisor", lambda: None)
# No running server either: the route must answer 409 (no server),
# never a NameError 500.
monkeypatch.setattr(
"hermes_cli.web_routers.local_models._state_endpoint", lambda: None)
r = client.post("/api/local-models/eject", json={"model_id": "anything"})
assert r.status_code == 409, (r.status_code, r.text)
def test_download_tolerates_stale_catalog_size(client, monkeypatch):
"""Upstream re-uploads make catalog sizes stale; a download whose
delivered bytes are self-consistent with the SERVER's declared length
must succeed even when the catalog said something else. (This is the
tolerance the sha removal was for — being out of date must not break
downloads.)"""
body = b"x" * 48 # server-consistent: Content-Length == body length
class FakeResponse(io.BytesIO):
headers = {"Content-Length": str(len(body))}
def __enter__(self):
return self
def __exit__(self, *a):
return False
monkeypatch.setattr("urllib.request.urlopen",
lambda *a, **k: FakeResponse(body))
from hermes_cli.local_runtime.estimator import HardwareBudget
budget = HardwareBudget(usable_vram_bytes=64 << 30,
total_device_bytes=64 << 30,
ram_available_bytes=64 << 30)
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
lambda **kw: budget)
# Keep the post-download server bounce out of this unit.
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
lambda: False)
from hermes_cli.local_runtime.catalog import CATALOG
# Catalog size for this entry is in the tens of GB — wildly stale
# versus our 48-byte body. The download must still land.
entry_id = CATALOG[0].id
r = client.post("/api/local-models/download", json={"model_id": entry_id})
assert r.status_code == 200
job_id = r.json()["job_id"]
deadline = time.time() + 10
status = None
while time.time() < deadline:
status = client.get(f"/api/local-models/jobs/{job_id}").json()
if status["status"] in ("done", "error"):
break
time.sleep(0.05)
assert status is not None and status["status"] == "done", status.get("error")
@@ -0,0 +1,74 @@
"""The managed local server owns its picker identity.
A live session on the managed llama-server reports provider "custom"
(the resolution seam's generic label for a raw base_url). The picker
payload used to materialize that as a duplicate "Custom endpoint" group
above the Local row — same staged models listed twice, checkmark on the
wrong group. Contract: when the current session points at the managed
endpoint, the Local row is current and no custom-endpoint duplicate
exists; a user's own external endpoint keeps its row untouched."""
from __future__ import annotations
import dataclasses
import pytest
MANAGED = {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"}
STAGED = {"Qwen-A-UD-Q4_K_M", "Qwen-B-UD-Q4_K_M"}
@pytest.fixture
def ctx(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
import hermes_cli.inventory as inv
monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.staged_model_ids",
lambda: set(STAGED))
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
lambda: dict(MANAGED))
context = inv.load_picker_context()
return inv, context
def _rows(inv, context, **overrides):
context = dataclasses.replace(context, **overrides)
return inv.build_models_payload(context, explicit_only=True)["providers"]
def test_managed_custom_session_shows_only_the_local_row(ctx):
inv, context = ctx
rows = _rows(inv, context,
current_provider="custom",
current_model="Qwen-A-UD-Q4_K_M",
current_base_url=MANAGED["base_url"])
slugs = [r["slug"] for r in rows]
assert "llamacpp" in slugs
assert "custom" not in slugs, (
"managed endpoint leaked a duplicate 'Custom endpoint' group")
local = next(r for r in rows if r["slug"] == "llamacpp")
assert local["is_current"] is True
assert local["name"] == "Local"
def test_external_custom_endpoint_keeps_its_row(ctx):
inv, context = ctx
rows = _rows(inv, context,
current_provider="custom",
current_model="some-model",
current_base_url="http://my-vllm-box:8000/v1")
slugs = [r["slug"] for r in rows]
assert "custom" in slugs, "a real external endpoint must keep its row"
custom = next(r for r in rows if r["slug"] == "custom")
assert custom["is_current"] is True
local = next(r for r in rows if r["slug"] == "llamacpp")
assert local["is_current"] is False
def test_remote_provider_session_unaffected(ctx):
inv, context = ctx
rows = _rows(inv, context)
local = next(r for r in rows if r["slug"] == "llamacpp")
assert local["is_current"] is False
assert "custom" not in [r["slug"] for r in rows if r.get("is_current")]
+187
View File
@@ -0,0 +1,187 @@
"""Quickstart route: one POST from nothing to a working local default.
Contract, not implementation: the route must (a) preflight-fail
synchronously when nothing fits, (b) report which legs the job will run
(runtime install / model download), skipping legs already satisfied,
and (c) run install -> download -> activate through the same code paths
the individual routes use. The slow legs are stubbed at their module
boundaries; the sequencing and job bookkeeping are real.
"""
from __future__ import annotations
import time
from pathlib import Path
import pytest
from fastapi.testclient import TestClient
@pytest.fixture
def client(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli import web_server
test_client = TestClient(web_server.app)
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
return test_client
def _wait_job(client, job_id: str, timeout: float = 10.0) -> dict:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
job = client.get(f"/api/local-models/jobs/{job_id}").json()
if job["status"] != "running":
return job
time.sleep(0.05)
raise AssertionError(f"job {job_id} still running after {timeout}s")
def test_quickstart_unknown_model_404s(client):
r = client.post("/api/local-models/quickstart", json={"model_id": "no-such"})
assert r.status_code == 404
def test_quickstart_refuses_when_nothing_fits(client, monkeypatch):
"""Preflight is synchronous: a machine no catalog entry fits gets a 409
with guidance, not a doomed background job."""
monkeypatch.setattr(
"hermes_cli.local_runtime.catalog.select_variant", lambda *a, **k: None)
r = client.post("/api/local-models/quickstart", json={})
assert r.status_code == 409
assert "Local Models" in r.json()["detail"]
def test_quickstart_runs_all_three_legs(client, monkeypatch, tmp_path):
"""Fresh machine: install runtime -> download recommended -> activate.
Each leg is asserted by its observable call, in order."""
calls: list[str] = []
# Leg 1: no runtime installed yet; install is the stubbed binaries call.
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.installed_tags", lambda: [])
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.ensure_runtime_installed",
lambda tag, backend, progress=None: calls.append("install"))
# Leg 2: nothing staged; the download writes the files the plan names.
def _fake_download(url, dest, job, *, base_done=0, keep_totals=False):
Path(dest).parent.mkdir(parents=True, exist_ok=True)
Path(dest).write_bytes(b"GGUF\x00")
calls.append("download")
monkeypatch.setattr(
"hermes_cli.web_routers.local_models.download_file", _fake_download)
# Leg 3: activation — stub the server start and the model assignment.
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
lambda config, force=False: calls.append("server") or None)
monkeypatch.setattr(
"hermes_cli.web_routers.local_models._state_endpoint",
lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
from hermes_cli import web_deps
monkeypatch.setattr(
web_deps, "late",
lambda name: (lambda *a, **k: calls.append("assign")))
r = client.post("/api/local-models/quickstart", json={})
assert r.status_code == 200
body = r.json()
assert body["needs_runtime"] is True
assert body["needs_download"] is True
assert body["download_bytes"] > 0
job = _wait_job(client, body["job_id"])
assert job["status"] == "done", job["error"]
assert job["kind"] == "quickstart"
# Order is the contract: engine, weights, server, default.
assert calls[0] == "install"
assert "download" in calls
assert calls.index("install") < calls.index("download") < calls.index("assign")
# Durable effect: the runtime is enabled in config.
from hermes_cli.config import load_config
assert load_config()["local_runtime"]["enabled"] is True
def test_quickstart_skips_satisfied_legs(client, monkeypatch):
"""Runtime present and model already staged: the response says so and
the job goes straight to activation."""
calls: list[str] = []
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.ensure_runtime_installed",
lambda tag, backend, progress=None: calls.append("install"))
# Every catalog variant reads as staged.
from hermes_cli.local_runtime.catalog import CATALOG
all_ids = {v.model_id for e in CATALOG for v in e.variants}
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.staged_model_ids", lambda: all_ids)
monkeypatch.setattr(
"hermes_cli.web_routers.local_models.download_file",
lambda *a, **k: calls.append("download"))
monkeypatch.setattr(
"hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
lambda config, force=False: None)
monkeypatch.setattr(
"hermes_cli.web_routers.local_models._state_endpoint",
lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
from hermes_cli import web_deps
monkeypatch.setattr(
web_deps, "late",
lambda name: (lambda *a, **k: calls.append("assign")))
r = client.post("/api/local-models/quickstart", json={})
assert r.status_code == 200
body = r.json()
assert body["needs_runtime"] is False
assert body["needs_download"] is False
assert body["download_bytes"] == 0
job = _wait_job(client, body["job_id"])
assert job["status"] == "done", job["error"]
assert "install" not in calls and "download" not in calls
assert calls == ["assign"] or calls[-1] == "assign"
@pytest.fixture
def quickstart_ready(monkeypatch):
"""Preflight passes without hardware or network: the runtime reads as
installed and every entry's first variant is servable, so the POST
reaches the single-flight lock instead of 409ing at fit/engine
preflight on machines where nothing fits."""
from hermes_cli.local_runtime.catalog import VariantChoice
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
monkeypatch.setattr(
"hermes_cli.local_runtime.catalog.select_variant",
lambda entry, budget: VariantChoice(variant=entry.variants[0],
zero_spill=True,
reason_key="best-fits"))
monkeypatch.setattr(
"hermes_cli.web_routers.local_models._engine_too_old",
lambda min_engine: False)
def test_quickstart_is_single_flight(client, quickstart_ready, monkeypatch):
"""A second quickstart while one runs must 409, not start a twin job
(the job sequences installs, downloads, a server bounce, and a config
write — two interleaved runs corrupt all four)."""
import hermes_cli.web_routers.local_models as lm
lm._QUICKSTART_LOCK.acquire()
try:
r = client.post("/api/local-models/quickstart", json={})
assert r.status_code == 409
assert "already running" in r.json()["detail"].lower()
finally:
lm._QUICKSTART_LOCK.release()
@@ -0,0 +1,170 @@
"""The recommendation decision table — the reviewable matrix.
The recommendation itself is DERIVED (catalog.recommended_entry: best
quality among resident entries clearing the pleasant speed floor, else
fastest resident, else least-painful spilled), so nobody hand-maintains
per-hardware-class picks. This table is the editorial control on that
derivation: it enumerates the real memory size classes x {discrete,
unified} and pins every cell. A catalog change (new model, quality
re-rank, quant swap) flips cells HERE, and the diff of this file in
review IS the sign-off on what each machine class gets.
These are decision pins, not change-detectors: each cell is a choice a
human approved, exactly like a golden file. When a cell flips on
purpose, update it in the same commit and say why. When one flips by
surprise, that is the test doing its job.
Budgets mirror hardware.probe_budget's planning-mode shapes (margins,
UMA headroom) so the cells match what a real machine of that class
resolves.
"""
from __future__ import annotations
import pytest
from hermes_cli.local_runtime.catalog import (
CATALOG,
PLEASANT_FLOOR_TOK_S,
predicted_decode_tok_s,
recommended_entry,
recommended_id,
select_variant,
)
from hermes_cli.local_runtime.estimator import HardwareBudget
_GIB = 1 << 30
def _discrete(size_gb: int) -> HardwareBudget:
total = size_gb * _GIB
margin = max(2 * _GIB, int(total * 0.09))
return HardwareBudget(usable_vram_bytes=max(0, total - margin),
total_device_bytes=total,
ram_available_bytes=64 * _GIB, uma=False)
def _unified(size_gb: int) -> HardwareBudget:
total = size_gb * _GIB
return HardwareBudget(usable_vram_bytes=int(total * 0.80),
total_device_bytes=total,
ram_available_bytes=0, uma=True)
# The decision table. Cells were generated by the resolver and then
# reviewed as editorial decisions:
#
# VRAM | discrete | unified
# -----+-------------------------+------------------------
# 8 | qwen3.6-35b-a3b spilled | (none fits)
# 16 | qwen3.6-35b-a3b spilled | (none fits)
# 24 | qwen3.8-27b | (none fits)
# 32 | qwen3.8-27b | qwen3.6-35b-a3b
# 48 | qwen3.8-27b | qwen3.6-35b-a3b
# 96 | qwen3.8-27b | qwen3.6-35b-a3b
# 128 | qwen3.8-flash-next | qwen3.6-35b-a3b
# 256 | qwen3.8-flash-next | qwen3.8-flash-next
# 512 | qwen3.8-flash-next | qwen3.8-flash-next
#
# Reading guide for reviewers:
# - Discrete <=16 GB: nothing runs resident; the 35B MoE is the least
# painful spill (active slice streams from host; a dense spill reads
# every weight over the bus).
# - Discrete 24-96 GB: the 27B is the flagship experience — dense reads
# at ~1 TB/s clear the floor easily, so quality decides.
# - Discrete/unified where Flash Next fits resident (128 GB discrete,
# 256+ GB unified): the frontier model is the pick — highest quality,
# and its sparse decode clears the floor even at UMA bandwidth
# (~24 tok/s predicted at 210 GB/s).
# - Unified 32-128 GB — the Spark class, the reason this resolver
# exists: the dense 27B predicts ~13 tok/s at UMA bandwidth (below
# the pleasant floor), so the 35B-A3B (~60 tok/s) wins.
# - Unified <=24 GB: no entry passes the physics check inside the UMA
# budget (spilling is impossible on UMA by construction — the pool IS
# the RAM). The pane's browse flow is the path for those machines
# until a small catalog entry lands (revisit when one does).
DECISION_TABLE = [
(8, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
(8, "unified", None, None),
(16, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
(16, "unified", None, None),
(24, "discrete", "qwen3.8-27b", "best-quality-resident"),
(24, "unified", None, None),
(32, "discrete", "qwen3.8-27b", "best-quality-resident"),
(32, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
(48, "discrete", "qwen3.8-27b", "best-quality-resident"),
(48, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
(96, "discrete", "qwen3.8-27b", "best-quality-resident"),
(96, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
(128, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
(128, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
(256, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
(256, "unified", "qwen3.8-flash-next", "best-quality-resident"),
(512, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
(512, "unified", "qwen3.8-flash-next", "best-quality-resident"),
]
@pytest.mark.parametrize(
("size_gb", "kind", "expected", "expected_reason"),
DECISION_TABLE,
ids=[f"{s}GB-{k}" for s, k, _, _ in DECISION_TABLE])
def test_recommendation_decision_table(size_gb, kind, expected, expected_reason):
"""Pins the pick AND its reason per cell: the reason is user-facing
(the Recommended badge's tooltip), so a cell whose rationale flips
without the pick flipping is still a review-worthy change."""
budget = _discrete(size_gb) if kind == "discrete" else _unified(size_gb)
picked = recommended_entry(budget)
if expected is None:
assert picked is None
else:
assert picked is not None
assert (picked[0].id, picked[1]) == (expected, expected_reason)
# ── invariants behind the table (survive catalog changes) ──
def test_every_entry_carries_the_recommendation_axes():
"""quality and decode_fraction are authoring requirements: an entry
without them silently loses every quality comparison (quality=0) or
prices as dense (decode_fraction=1.0)."""
for entry in CATALOG:
assert entry.quality > 0, f"{entry.id} has no quality ordering"
assert 0.0 < entry.decode_fraction <= 1.0, entry.id
if not entry.moe:
assert entry.decode_fraction == 1.0, (
f"{entry.id} is dense — it reads every weight per token")
def test_unified_never_recommends_a_below_floor_dense_model():
"""The Spark rule, as an invariant: whatever the catalog holds, a
unified-memory machine must not be told to run a model whose
predicted decode is below the pleasant floor while a resident
alternative clears it."""
budget = _unified(128)
pick = recommended_id(budget)
assert pick is not None
entry = next(e for e in CATALOG if e.id == pick)
choice = select_variant(entry, budget)
assert choice is not None and choice.zero_spill
clears = [
e for e in CATALOG
if (c := select_variant(e, budget)) is not None and c.zero_spill
and predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
]
if clears:
assert predicted_decode_tok_s(entry, choice.variant, budget) >= PLEASANT_FLOOR_TOK_S
def test_quality_decides_where_speed_permits():
"""On big discrete hardware every resident entry clears the floor, so
the pick must be the highest-quality fitting entry — the axis that
justifies carrying an editorial field at all."""
budget = _discrete(512)
pick = recommended_id(budget)
resident = [
e for e in CATALOG
if (c := select_variant(e, budget)) is not None and c.zero_spill
]
assert pick == max(resident, key=lambda e: e.quality).id

Some files were not shown because too many files have changed in this diff Show More