feat: local models — managed llama.cpp runtime with one-click desktop setup
Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
This commit is contained in:
@@ -9187,6 +9187,14 @@ def _build_call_kwargs(
|
||||
_provider_norm == "openrouter"
|
||||
or base_url_host_matches(_effective_base, "openrouter.ai")
|
||||
)
|
||||
# The managed local llama-server honors explicit caps too: a local
|
||||
# decode burns the user's own GPU at full tilt, so a caller that
|
||||
# says "this is a 64-token task" must be believed — an uncapped
|
||||
# local generation whose EOS never comes runs to the full context
|
||||
# window. No wire-format quirks apply (llama.cpp accepts
|
||||
# max_tokens), and the no-default-cap policy is unchanged: this
|
||||
# only forwards caps callers explicitly set.
|
||||
_is_managed_local = _is_managed_local_endpoint(_effective_base)
|
||||
if (
|
||||
_is_anthropic_compat_endpoint(provider, _effective_base)
|
||||
or _nous_on_messages
|
||||
@@ -9194,6 +9202,7 @@ def _build_call_kwargs(
|
||||
or _is_moa
|
||||
or _is_gemini_native
|
||||
or _is_openrouter
|
||||
or _is_managed_local
|
||||
):
|
||||
# Use auxiliary_max_tokens_param() so models that require
|
||||
# max_completion_tokens (GPT-5 family, Copilot) get the right
|
||||
@@ -9539,6 +9548,49 @@ def _is_streaming_rejected_error(exc: Exception) -> bool:
|
||||
)
|
||||
|
||||
|
||||
_MANAGED_LOCAL_STATE_TTL_S = 15.0
|
||||
_managed_local_cache: "tuple[float, str]" = (0.0, "")
|
||||
|
||||
|
||||
def _managed_local_netloc() -> str:
|
||||
"""host:port of the managed local llama-server, or "" when none.
|
||||
|
||||
Read from the supervisor's state file (written at spawn, removed on
|
||||
stop) with a short TTL so per-request checks don't hit the disk. The
|
||||
state file is the same source provider resolution uses, so the match
|
||||
is exact — no false positives on other localhost endpoints.
|
||||
"""
|
||||
global _managed_local_cache
|
||||
now = time.monotonic()
|
||||
ts, cached = _managed_local_cache
|
||||
if now - ts < _MANAGED_LOCAL_STATE_TTL_S:
|
||||
return cached
|
||||
netloc = ""
|
||||
try:
|
||||
from hermes_cli.local_runtime.supervisor import state_path
|
||||
|
||||
raw = state_path().read_text(encoding="utf-8")
|
||||
base = str((json.loads(raw) or {}).get("base_url", ""))
|
||||
netloc = urlparse(base).netloc.lower()
|
||||
except Exception:
|
||||
netloc = ""
|
||||
_managed_local_cache = (now, netloc)
|
||||
return netloc
|
||||
|
||||
|
||||
def _is_managed_local_endpoint(base_url: Optional[str]) -> bool:
|
||||
"""True when *base_url* targets the llama-server this Hermes manages."""
|
||||
if not base_url:
|
||||
return False
|
||||
managed = _managed_local_netloc()
|
||||
if not managed:
|
||||
return False
|
||||
try:
|
||||
return urlparse(str(base_url)).netloc.lower() == managed
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
|
||||
"""Detect providers that only accept streaming (non-stream = HTTP 400).
|
||||
|
||||
@@ -9554,6 +9606,18 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
|
||||
Beyond the known-host list, users can mark ANY custom endpoint as
|
||||
stream-only via ``auxiliary.stream_only_base_urls`` in config.yaml
|
||||
(list of substrings matched against the endpoint URL).
|
||||
|
||||
The managed local llama-server is always streamed for a different
|
||||
reason: cancellation. llama-server only notices a dead client when it
|
||||
writes to the socket. A non-streamed request writes once — after the
|
||||
FULL generation — so an abandoned call (client timeout, retry, app
|
||||
exit) keeps the GPU decoding to the end of the context window with
|
||||
nobody listening; requests that queue behind a model load are the
|
||||
worst case, since the client is long gone before decode even starts.
|
||||
Streaming writes every few tokens, so an abandoned decode dies at the
|
||||
first post-disconnect chunk (verified against llama-server b10362:
|
||||
streamed disconnect cancels in <1s through the router; non-streamed
|
||||
survives until the server's next incidental socket poll, if ever).
|
||||
"""
|
||||
_url = str(base_url or "").lower()
|
||||
if not _url:
|
||||
@@ -9561,6 +9625,9 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
|
||||
# Tencent Copilot — "Non-stream chat request is currently not supported"
|
||||
if base_url_host_matches(_url, "copilot.tencent.com"):
|
||||
return True
|
||||
# Managed local llama-server — streamed so abandonment cancels decode.
|
||||
if _is_managed_local_endpoint(_url):
|
||||
return True
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
aux_cfg = (load_config() or {}).get("auxiliary", {})
|
||||
|
||||
@@ -1055,6 +1055,59 @@ def should_use_direct_api_call(agent) -> bool:
|
||||
_DIRECT_API_ACTIVITY_HEARTBEAT_SECONDS = 15.0
|
||||
|
||||
|
||||
def _managed_local_load_notice(agent, api_kwargs: dict) -> "Optional[str]":
|
||||
"""A live phase notice while the managed local server works before the
|
||||
first token, or None when neither phase (nor the managed server) applies:
|
||||
|
||||
- "⏳ loading <model> into memory — N%" (weights streaming off disk;
|
||||
real per-tensor percent from the router's SSE stream)
|
||||
- "⚙ processing prompt — N of ~M tokens (P%)" (prefill; live counter
|
||||
from /slots, denominator estimated from the request body)
|
||||
|
||||
A cold local model spends ~tens of seconds loading and a long-context
|
||||
turn spends tens more in prefill; without this, both windows render as
|
||||
the generic "no output yet (provider may be slow or overloaded)" stall
|
||||
warning — alarming copy for healthy, expected phases.
|
||||
"""
|
||||
try:
|
||||
base = str(getattr(agent, "base_url", "") or "")
|
||||
if not base:
|
||||
return None
|
||||
import json as _json
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from hermes_cli.local_runtime.load_progress import (
|
||||
get_loading_progress,
|
||||
get_prefill_progress,
|
||||
)
|
||||
from hermes_cli.local_runtime.supervisor import state_path
|
||||
|
||||
state = _json.loads(state_path().read_text(encoding="utf-8"))
|
||||
managed = urlparse(str(state.get("base_url", ""))).netloc.lower()
|
||||
if not managed or urlparse(base).netloc.lower() != managed:
|
||||
return None
|
||||
model = str(api_kwargs.get("model", ""))
|
||||
progress = get_loading_progress().get(model)
|
||||
if progress is not None:
|
||||
return (
|
||||
f"⏳ loading {model} into memory — {progress['percent']}% "
|
||||
"(responses start once the model is loaded)"
|
||||
)
|
||||
prefill = get_prefill_progress(model)
|
||||
if prefill is not None:
|
||||
processed = int(prefill["processed"])
|
||||
total = estimate_request_context_tokens(api_kwargs)
|
||||
if total and total >= processed:
|
||||
pct = max(0, min(100, round(processed / total * 100)))
|
||||
return f"⚙ processing prompt — {pct}%"
|
||||
# Counter past the estimate (estimator undercounted): no honest
|
||||
# denominator, so no percent — the UI shows label-only.
|
||||
return "⚙ processing prompt"
|
||||
return None
|
||||
except Exception: # noqa: BLE001 — a status nicety must never break a call
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_direct_stale_timeout(agent, api_kwargs: dict) -> float:
|
||||
"""Stale budget for the inline non-streaming call.
|
||||
|
||||
@@ -5322,9 +5375,54 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
|
||||
t.start()
|
||||
_last_heartbeat = time.time()
|
||||
_HEARTBEAT_INTERVAL = 30.0 # seconds between gateway activity touches
|
||||
# Managed local server: a cold model streams weights off disk for tens
|
||||
# of seconds before the first token can exist. Surface THAT immediately
|
||||
# (real per-tensor percent from the router's SSE stream) instead of
|
||||
# letting the wait fall through to the 30s "provider may be slow or
|
||||
# overloaded" copy. Checked on a ~1s cadence only while no chunks have
|
||||
# arrived; the probe is an in-memory snapshot read, not a network call.
|
||||
_last_load_poll = 0.0
|
||||
_load_notice_shown = False
|
||||
_load_notice_misses = 0
|
||||
_is_local_base = bool(agent.base_url) and is_local_endpoint(agent.base_url)
|
||||
while t.is_alive():
|
||||
t.join(timeout=0.3)
|
||||
|
||||
_hb_now = time.time()
|
||||
# Cold-load window: last_chunk_time is touched at request-client
|
||||
# creation and then only by REAL chunks, so "no chunk for 2s+" is
|
||||
# true through a model load (nothing can stream while the child is
|
||||
# still mapping weights) and false during healthy token flow —
|
||||
# which is what keeps this poll off the streaming hot path. The
|
||||
# probe itself is an in-memory snapshot read.
|
||||
if (
|
||||
_is_local_base
|
||||
and _hb_now - last_chunk_time["t"] >= 2.0
|
||||
and _hb_now - _last_load_poll >= 1.0
|
||||
):
|
||||
_last_load_poll = _hb_now
|
||||
_load_notice = _managed_local_load_notice(agent, api_kwargs)
|
||||
if _load_notice is not None:
|
||||
agent._emit_wait_notice(_load_notice)
|
||||
agent._touch_activity("local model loading")
|
||||
_load_notice_shown = True
|
||||
_load_notice_misses = 0
|
||||
# Loading IS liveness for the heartbeat; the stale detector
|
||||
# needs no help — the local floor (900s) dwarfs any load.
|
||||
_last_heartbeat = _hb_now
|
||||
continue
|
||||
if _load_notice_shown:
|
||||
# One missed sample is routine (a /slots read straddling a
|
||||
# batch boundary, a 2s probe timeout under load) — clearing
|
||||
# on it made the status line strobe blank once every few
|
||||
# seconds mid-prefill. Only a SUSTAINED absence means the
|
||||
# phase really ended.
|
||||
_load_notice_misses += 1
|
||||
if _load_notice_misses >= 3:
|
||||
_load_notice_shown = False
|
||||
_load_notice_misses = 0
|
||||
agent._emit_wait_notice("")
|
||||
|
||||
# Periodic heartbeat: touch the agent's activity tracker so the
|
||||
# gateway's inactivity monitor knows we're alive while waiting
|
||||
# for stream chunks. Without this, long thinking pauses (e.g.
|
||||
@@ -5333,7 +5431,6 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
|
||||
# activity on each chunk, but the gap between API call start
|
||||
# and first chunk can exceed the gateway timeout — especially
|
||||
# when the stale-stream timeout is disabled (local providers).
|
||||
_hb_now = time.time()
|
||||
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
|
||||
_last_heartbeat = _hb_now
|
||||
_waiting_secs = int(_hb_now - last_chunk_time["t"])
|
||||
|
||||
@@ -652,6 +652,40 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str
|
||||
)
|
||||
|
||||
|
||||
def _maybe_grow_local_window(agent: Any, compressor: Any,
|
||||
request_tokens: int) -> Optional[int]:
|
||||
"""Try growing the managed local model's context window before
|
||||
compressing. Returns the new window when the ladder granted one, else
|
||||
None (hold / at native / not a managed local session).
|
||||
|
||||
The window ladder's design order: models launch at their zero-spill
|
||||
window and grow toward native max as the session needs room;
|
||||
compression is the move of last resort. Cheap for every non-local
|
||||
provider: one lowercase compare, no imports.
|
||||
"""
|
||||
provider = (getattr(agent, "provider", "") or "").strip().lower()
|
||||
if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"):
|
||||
return None
|
||||
base_url = getattr(agent, "base_url", "") or ""
|
||||
if "127.0.0.1" not in base_url and "localhost" not in base_url:
|
||||
return None
|
||||
try:
|
||||
from hermes_cli.local_runtime.growth import maybe_grow_window
|
||||
|
||||
current_window = int(getattr(compressor, "context_length", 0) or 0)
|
||||
if current_window <= 0:
|
||||
return None
|
||||
return maybe_grow_window(
|
||||
getattr(agent, "model", "") or "",
|
||||
base_url=base_url,
|
||||
session_tokens=int(request_tokens),
|
||||
current_window=current_window,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — growth must never break a turn
|
||||
logger.debug("local window growth check failed: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def _ra():
|
||||
"""Lazy reference to ``run_agent`` so callers can patch
|
||||
``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` /
|
||||
@@ -2876,6 +2910,39 @@ def run_conversation(
|
||||
and not _compression_cooldown
|
||||
and _compressor.should_compress(request_pressure_tokens)
|
||||
):
|
||||
# Managed local runtime: try GROWING the context window before
|
||||
# compressing (the window ladder's design order — compression is
|
||||
# the move of last resort, once the window is at the model's
|
||||
# native max or physics/speed say stop). Only fires for a
|
||||
# llamacpp-flavored provider whose base_url is the server this
|
||||
# process supervises; every other provider falls straight
|
||||
# through to compression, exactly as before.
|
||||
_grown_window = _maybe_grow_local_window(
|
||||
agent, _compressor, request_pressure_tokens
|
||||
)
|
||||
if _grown_window:
|
||||
# The server now grants a bigger window: recalibrate the
|
||||
# compressor to it and skip compression this pass — the
|
||||
# request that was over the OLD threshold fits the new one.
|
||||
_compressor.update_model(
|
||||
agent.model,
|
||||
_grown_window,
|
||||
base_url=getattr(agent, "base_url", "") or "",
|
||||
api_key=getattr(agent, "api_key", "") or "",
|
||||
provider=getattr(agent, "provider", "") or "",
|
||||
api_mode=getattr(agent, "api_mode", "") or "",
|
||||
)
|
||||
agent._buffer_status(
|
||||
f"📈 Context window grown to {_grown_window // 1024}K "
|
||||
f"(local model; conversation continues uncompressed)"
|
||||
)
|
||||
# This preflight iteration never reached the provider —
|
||||
# refund the consumed call/budget exactly as the compression
|
||||
# path below does before ITS continue.
|
||||
api_call_count -= 1
|
||||
agent._api_call_count = api_call_count
|
||||
agent.iteration_budget.refund()
|
||||
continue
|
||||
if _moa_prepared_request is not None:
|
||||
pending_moa_prepared_request = _moa_prepared_request
|
||||
compression_attempts += 1
|
||||
|
||||
+44
-3
@@ -519,6 +519,28 @@ def _lookup_supports_vision(
|
||||
return override
|
||||
if not provider or not model:
|
||||
return None
|
||||
|
||||
# Managed local runtime: the server that would receive the image is
|
||||
# the authority on whether it can see (its /props reports modalities
|
||||
# when a vision projector is loaded; the catalog covers staged-but-
|
||||
# unloaded models). Cloud catalogs have never heard of a local GGUF,
|
||||
# so without this answer every local model reads as text-only and
|
||||
# images detour to a cloud auxiliary — wrong twice for a local-first
|
||||
# user (broken feature, and a screenshot leaving the machine).
|
||||
try:
|
||||
from hermes_cli.local_runtime.capabilities import (
|
||||
is_managed_provider,
|
||||
managed_model_supports_vision,
|
||||
)
|
||||
|
||||
if is_managed_provider(provider, _resolve_inference_base_url(cfg, provider) or ""):
|
||||
managed = managed_model_supports_vision(model)
|
||||
if managed is not None:
|
||||
return managed
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug("image_routing: managed-runtime caps lookup failed for %s:%s — %s",
|
||||
provider, model, exc)
|
||||
|
||||
caps = None
|
||||
try:
|
||||
from agent.models_dev import get_model_capabilities
|
||||
@@ -813,12 +835,31 @@ def _file_to_data_url(path: Path) -> Optional[str]:
|
||||
logger.warning("image_routing: failed to read %s — %s", path, exc)
|
||||
return None
|
||||
mime = _guess_mime(path, raw=raw)
|
||||
if mime not in _UNIVERSALLY_SUPPORTED_MIMES:
|
||||
accepted = _UNIVERSALLY_SUPPORTED_MIMES
|
||||
# The managed local server decodes fewer formats than cloud providers
|
||||
# (no WebP — and a WebP part fails SILENTLY: the model never sees an
|
||||
# image and confabulates a description). When the active main model is
|
||||
# served by the managed runtime, narrow the accepted set so those
|
||||
# formats transcode to PNG here instead of vanishing server-side.
|
||||
try:
|
||||
from agent.auxiliary_client import _runtime_main_value
|
||||
from hermes_cli.local_runtime.capabilities import (
|
||||
ACCEPTED_IMAGE_MIMES,
|
||||
is_managed_provider,
|
||||
)
|
||||
|
||||
if is_managed_provider(
|
||||
str(_runtime_main_value("provider") or ""),
|
||||
str(_runtime_main_value("base_url") or "")):
|
||||
accepted = ACCEPTED_IMAGE_MIMES
|
||||
except Exception: # noqa: BLE001 — best-effort narrowing only
|
||||
pass
|
||||
if mime not in accepted:
|
||||
transcoded = _transcode_to_png(raw)
|
||||
if transcoded is None:
|
||||
logger.warning(
|
||||
"image_routing: %s is %s which is not accepted by all major "
|
||||
"vision providers and could not be transcoded to PNG; "
|
||||
"image_routing: %s is %s which is not accepted by the "
|
||||
"active provider and could not be transcoded to PNG; "
|
||||
"skipping this attachment.",
|
||||
path, mime,
|
||||
)
|
||||
|
||||
@@ -1502,6 +1502,35 @@ def fetch_endpoint_model_metadata(
|
||||
model_alias = props.get("model_alias", "")
|
||||
if n_ctx and model_alias and model_alias in cache:
|
||||
cache[model_alias]["context_length"] = n_ctx
|
||||
else:
|
||||
# Router mode: bare /props 400s and telemetry is
|
||||
# per-child (?model=). Enumerate children via the
|
||||
# native /models (carries status) and read each
|
||||
# LOADED child's granted window — the value the
|
||||
# context policy actually granted, which the meter
|
||||
# and compressor must follow. Unloaded children are
|
||||
# skipped: probing them could trigger an autoload.
|
||||
native = requests.get(base + "/models", headers=headers, timeout=5, verify=_verify)
|
||||
if native.ok:
|
||||
children = (native.json() or {}).get("data", [])
|
||||
for child in children[:16]:
|
||||
if not isinstance(child, dict):
|
||||
continue
|
||||
child_id = child.get("id")
|
||||
status = (child.get("status") or {}).get("value")
|
||||
if not child_id or child_id not in cache or status not in ("loaded", "ready"):
|
||||
continue
|
||||
pr = requests.get(
|
||||
base + "/v1/props", params={"model": child_id},
|
||||
headers=headers, timeout=5, verify=_verify)
|
||||
if not pr.ok:
|
||||
pr = requests.get(
|
||||
base + "/props", params={"model": child_id},
|
||||
headers=headers, timeout=5, verify=_verify)
|
||||
if pr.ok:
|
||||
child_ctx = (pr.json().get("default_generation_settings") or {}).get("n_ctx")
|
||||
if child_ctx:
|
||||
cache[child_id]["context_length"] = child_ctx
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@@ -2375,6 +2404,27 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
return int(ctx)
|
||||
break
|
||||
|
||||
# llama.cpp: /props reports default_generation_settings.n_ctx —
|
||||
# the RUNTIME window the server grants. Critically, the router
|
||||
# answers this (from its preset) even for a model that is not
|
||||
# currently loaded, while /v1/models reports meta=null until
|
||||
# load. Without this probe, resolving a lazily-loaded model at
|
||||
# session start finds no metadata and falls through to the
|
||||
# name-pattern defaults, where a family catch-all (e.g. "qwen"
|
||||
# = 131072) misreports a server launched at 262144.
|
||||
if server_type == "llamacpp":
|
||||
for props_path in (f"/props?model={model}", "/props"):
|
||||
try:
|
||||
resp = client.get(f"{server_url}{props_path}")
|
||||
except httpx.HTTPError:
|
||||
break
|
||||
if resp.status_code != 200:
|
||||
continue
|
||||
n_ctx = (resp.json().get("default_generation_settings")
|
||||
or {}).get("n_ctx")
|
||||
if isinstance(n_ctx, (int, float)) and n_ctx:
|
||||
return int(n_ctx)
|
||||
|
||||
# LM Studio / vLLM / llama.cpp / Anthropic-compat proxies:
|
||||
# try /v1/models/{model}
|
||||
resp = client.get(f"{server_url}/v1/models/{model}")
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
"""Idle deferral for background reviews on the managed local runtime.
|
||||
|
||||
The post-turn review fork replays the whole conversation on the review
|
||||
runtime. On a cloud provider that costs seconds and runs concurrently
|
||||
with whatever the user does next. When the review runtime IS the managed
|
||||
llama-server, the same fork monopolizes the GPU the user's next prompt
|
||||
needs, for minutes — and the next live turn cancels it, so an active
|
||||
session tends to pay the decode cost AND lose the learning.
|
||||
|
||||
This module keeps the decision to learn exactly where it was (turn end,
|
||||
nudge intervals, full-strength model, full transcript) and moves only
|
||||
the execution moment: reviews bound for the managed local endpoint are
|
||||
queued and dispatched when the machine is quiet. Everything else runs
|
||||
immediately, as before.
|
||||
|
||||
Policy (auxiliary.background_review.defer):
|
||||
auto (default) — defer exactly when the resolved review runtime
|
||||
targets the managed local server.
|
||||
never — old behavior everywhere.
|
||||
Explicit /refine (focus set) never defers: an explicit ask runs now,
|
||||
matching its bypass of the enabled gate.
|
||||
|
||||
Queue semantics:
|
||||
- One slot per session, newest snapshot wins. A review replays the whole
|
||||
conversation, so a newer snapshot strictly supersedes an older one —
|
||||
coalescing is deduplication, not loss.
|
||||
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn
|
||||
wrapper observing the run token's cancel flag, not killed-and-forgotten.
|
||||
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless
|
||||
of idleness — deferral may delay learning, never lose it.
|
||||
- In-memory, best-effort: dropped on process exit, the same durability
|
||||
contract the immediate daemon-thread fork always had.
|
||||
|
||||
Idle truth comes from the supervisor's /slots (machine-level: it sees
|
||||
every client of the managed server, including other Hermes profiles) and
|
||||
must hold for a settle window so a review is not launched into the gap
|
||||
between two quick prompts. Local in-process turn liveness is tracked via
|
||||
note_turn_started/note_turn_finished from run_conversation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
import urllib.request
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Sustained-quiet window before dispatch. Long enough that "typed two
|
||||
# prompts back to back" does not look idle; short enough that walking
|
||||
# away for coffee runs the queue.
|
||||
_IDLE_SETTLE_S = 15.0
|
||||
# Poll cadence while the queue is non-empty. The thread parks when empty.
|
||||
_POLL_INTERVAL_S = 5.0
|
||||
# Age at which a queued review dispatches regardless of idleness.
|
||||
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
|
||||
|
||||
|
||||
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
|
||||
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
|
||||
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
|
||||
return raw if raw in ("auto", "never") else "auto"
|
||||
|
||||
|
||||
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
|
||||
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
|
||||
try:
|
||||
value = float(raw)
|
||||
except (TypeError, ValueError):
|
||||
return _MAX_AGE_DEFAULT_S
|
||||
return value if value > 0 else _MAX_AGE_DEFAULT_S
|
||||
|
||||
|
||||
def review_targets_managed_local(agent: Any,
|
||||
task_cfg: Optional[Dict[str, Any]]) -> bool:
|
||||
"""Would this review fork decode on the llama-server WE manage?
|
||||
|
||||
Resolves the review runtime the same way the fork itself will and
|
||||
exact-matches its netloc against the supervisor state file — the
|
||||
matcher that cannot false-positive on external local servers. Any
|
||||
failure reads False: immediate spawn is always the safe default.
|
||||
|
||||
Order matters: the netloc probe (one TTL-cached state-file read)
|
||||
runs FIRST, so machines with no managed server — every cloud-only
|
||||
install — return False without resolving the review runtime at all.
|
||||
This wrapper runs on the turn's tail; runtime resolution belongs on
|
||||
that path only when a managed server actually exists.
|
||||
"""
|
||||
try:
|
||||
from agent.auxiliary_client import (
|
||||
_is_managed_local_endpoint,
|
||||
_managed_local_netloc,
|
||||
)
|
||||
|
||||
if not _managed_local_netloc():
|
||||
return False
|
||||
from agent.background_review import _resolve_review_runtime
|
||||
|
||||
runtime = _resolve_review_runtime(agent, task_cfg)
|
||||
return _is_managed_local_endpoint(runtime.get("base_url"))
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
|
||||
class _PendingReview:
|
||||
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
|
||||
|
||||
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
|
||||
self.agent = agent
|
||||
self.session_key = session_key
|
||||
self.kwargs = kwargs
|
||||
self.enqueued_at = time.monotonic()
|
||||
|
||||
|
||||
class ReviewIdleQueue:
|
||||
"""Session-coalescing queue + idle-gated dispatcher thread."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._lock = threading.Lock()
|
||||
self._pending: Dict[str, _PendingReview] = {}
|
||||
self._wake = threading.Event()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
self._live_turns = 0
|
||||
self._quiet_since: Optional[float] = None
|
||||
# Test seams — replaced by unit tests, never in production.
|
||||
self._now: Callable[[], float] = time.monotonic
|
||||
self._server_idle: Callable[[], bool] = _managed_server_idle
|
||||
|
||||
# ── turn liveness (this process) ────────────────────────────
|
||||
|
||||
def note_turn_started(self) -> None:
|
||||
with self._lock:
|
||||
self._live_turns += 1
|
||||
self._quiet_since = None
|
||||
|
||||
def note_turn_finished(self) -> None:
|
||||
with self._lock:
|
||||
self._live_turns = max(0, self._live_turns - 1)
|
||||
if self._live_turns == 0:
|
||||
self._quiet_since = self._now()
|
||||
self._wake.set()
|
||||
|
||||
# ── queue ────────────────────────────────────────────────────
|
||||
|
||||
def enqueue(self, agent: Any, session_key: str,
|
||||
kwargs: Dict[str, Any]) -> None:
|
||||
"""Add (or replace — newest snapshot wins) a session's pending review."""
|
||||
with self._lock:
|
||||
existing = self._pending.get(session_key)
|
||||
item = _PendingReview(agent, session_key, kwargs)
|
||||
# Stamp through the queue's clock (test seam); keep the ORIGINAL
|
||||
# enqueue time on coalesce so a busy session cannot push its
|
||||
# review's age-out forever.
|
||||
item.enqueued_at = (existing.enqueued_at if existing is not None
|
||||
else self._now())
|
||||
self._pending[session_key] = item
|
||||
self._ensure_thread()
|
||||
self._wake.set()
|
||||
logger.info("Background review deferred (session=%s, queued=%d)",
|
||||
session_key[-12:], len(self._pending))
|
||||
|
||||
def pending_count(self) -> int:
|
||||
with self._lock:
|
||||
return len(self._pending)
|
||||
|
||||
# ── dispatcher ───────────────────────────────────────────────
|
||||
|
||||
def _ensure_thread(self) -> None:
|
||||
with self._lock:
|
||||
if self._thread is None or not self._thread.is_alive():
|
||||
self._thread = threading.Thread(
|
||||
target=self._run, daemon=True, name="bg-review-idle-queue")
|
||||
self._thread.start()
|
||||
|
||||
def _quiet_for(self) -> float:
|
||||
"""Seconds this process has been turn-free (0 while a turn runs)."""
|
||||
with self._lock:
|
||||
if self._live_turns > 0 or self._quiet_since is None:
|
||||
return 0.0
|
||||
return self._now() - self._quiet_since
|
||||
|
||||
def _pop_dispatchable(self) -> Optional[_PendingReview]:
|
||||
"""Oldest aged-out item, else any item once quiet+idle hold."""
|
||||
with self._lock:
|
||||
if not self._pending:
|
||||
return None
|
||||
items = sorted(self._pending.values(),
|
||||
key=lambda p: p.enqueued_at)
|
||||
aged = [p for p in items
|
||||
if self._now() - p.enqueued_at
|
||||
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
|
||||
candidate = aged[0] if aged else None
|
||||
if candidate is None:
|
||||
if self._quiet_for() < _IDLE_SETTLE_S:
|
||||
return None
|
||||
if not self._server_idle():
|
||||
return None
|
||||
with self._lock:
|
||||
if not self._pending:
|
||||
return None
|
||||
candidate = min(self._pending.values(),
|
||||
key=lambda p: p.enqueued_at)
|
||||
with self._lock:
|
||||
return self._pending.pop(candidate.session_key, None)
|
||||
|
||||
def _run(self) -> None:
|
||||
while True:
|
||||
self._wake.wait()
|
||||
with self._lock:
|
||||
if not self._pending:
|
||||
self._wake.clear()
|
||||
continue
|
||||
item = None
|
||||
try:
|
||||
item = self._pop_dispatchable()
|
||||
if item is not None:
|
||||
if not self._still_enabled(item):
|
||||
logger.info(
|
||||
"Deferred background review dropped: reviews "
|
||||
"were disabled while it was queued (session=%s)",
|
||||
item.session_key[-12:])
|
||||
continue
|
||||
logger.info(
|
||||
"Dispatching deferred background review "
|
||||
"(session=%s, waited=%.0fs, queued=%d)",
|
||||
item.session_key[-12:],
|
||||
self._now() - item.enqueued_at,
|
||||
self.pending_count())
|
||||
item.agent._spawn_background_review_now(**item.kwargs)
|
||||
except Exception: # noqa: BLE001 — dispatcher must survive anything
|
||||
logger.warning("Deferred review dispatch failed",
|
||||
exc_info=True)
|
||||
if item is None:
|
||||
time.sleep(_POLL_INTERVAL_S)
|
||||
|
||||
@staticmethod
|
||||
def _still_enabled(item: _PendingReview) -> bool:
|
||||
"""Re-check the enabled gate at DISPATCH time.
|
||||
|
||||
The entry wrapper gates at enqueue time, but minutes may pass in
|
||||
the queue — a user who sets background_review.enabled: false while
|
||||
a review waits means it, and the dispatch must not resurrect it.
|
||||
Fail-open like the gate itself (a broken config never silently
|
||||
disables reviews)."""
|
||||
try:
|
||||
from agent.background_review import load_background_review_settings
|
||||
|
||||
enabled, _ = load_background_review_settings()
|
||||
return enabled
|
||||
except Exception: # noqa: BLE001
|
||||
return True
|
||||
|
||||
|
||||
def _managed_server_idle() -> bool:
|
||||
"""Machine-level idle: no processing slot on any loaded model of the
|
||||
managed router. Unreachable/no state file reads idle (nothing to
|
||||
contend with). One /models + one /slots call per loaded model."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.supervisor import state_path
|
||||
|
||||
state = json.loads(state_path().read_text(encoding="utf-8"))
|
||||
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
|
||||
key = str(state.get("api_key", ""))
|
||||
if not base:
|
||||
return True
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
req = urllib.request.Request(f"{base}/models", headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=3) as r:
|
||||
models = json.loads(r.read())
|
||||
loaded = [m["id"] for m in models.get("data", [])
|
||||
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
|
||||
from urllib.parse import quote
|
||||
|
||||
for mid in loaded:
|
||||
req = urllib.request.Request(f"{base}/slots?model={quote(mid)}",
|
||||
headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=3) as r:
|
||||
slots = json.loads(r.read())
|
||||
if any(s.get("is_processing") for s in slots
|
||||
if isinstance(s, dict)):
|
||||
return False
|
||||
return True
|
||||
except Exception: # noqa: BLE001
|
||||
return True
|
||||
|
||||
|
||||
# Module singleton — one queue per process, like the load-progress watcher.
|
||||
QUEUE = ReviewIdleQueue()
|
||||
@@ -1128,6 +1128,34 @@ def build_turn_context(
|
||||
_compress_block_reason = _info(_preflight_tokens)[1]
|
||||
except Exception:
|
||||
_compress_block_reason = None
|
||||
if _should_compress_now:
|
||||
# Managed local runtime: growing the window beats compressing —
|
||||
# the ladder's design order (same seam as the conversation
|
||||
# loop's pre-API gate; see _maybe_grow_local_window there).
|
||||
try:
|
||||
from agent.conversation_loop import _maybe_grow_local_window
|
||||
|
||||
_grown = _maybe_grow_local_window(
|
||||
agent, _compressor, _preflight_tokens
|
||||
)
|
||||
except Exception:
|
||||
_grown = None
|
||||
if _grown:
|
||||
_compressor.update_model(
|
||||
agent.model,
|
||||
_grown,
|
||||
base_url=getattr(agent, "base_url", "") or "",
|
||||
api_key=getattr(agent, "api_key", "") or "",
|
||||
provider=getattr(agent, "provider", "") or "",
|
||||
api_mode=getattr(agent, "api_mode", "") or "",
|
||||
)
|
||||
agent._buffer_status(
|
||||
f"📈 Context window grown to {_grown // 1024}K "
|
||||
f"(local model; conversation continues uncompressed)"
|
||||
)
|
||||
_should_compress_now = _compressor.should_compress(
|
||||
_preflight_tokens
|
||||
)
|
||||
if _should_compress_now:
|
||||
_preflight_compressed = True
|
||||
# Compression is actually running (block cleared / was never
|
||||
|
||||
@@ -16741,6 +16741,15 @@ ipcMain.on('hermes:translucency:support', event => {
|
||||
event.returnValue = { glass: GLASS_SUPPORTED, translucency: TRANSLUCENCY_SUPPORTED }
|
||||
})
|
||||
|
||||
// Launch-flag facts the renderer needs before first paint (same sendSync
|
||||
// pattern as translucency). `--local` gates every local-models GUI surface;
|
||||
// it arrives from `hermes desktop --local` or directly on Hermes.exe (a
|
||||
// shortcut edit), and survives self-relaunches because collectRelaunchArgs
|
||||
// only strips internal flags.
|
||||
ipcMain.on('hermes:launch-flags', event => {
|
||||
event.returnValue = { localModels: process.argv.includes('--local') }
|
||||
})
|
||||
|
||||
ipcMain.on('hermes:translucency', (_event, payload) => {
|
||||
const next = normalizeTranslucency(payload, GLASS_SUPPORTED)
|
||||
const previous = translucencyState
|
||||
|
||||
@@ -10,10 +10,14 @@ import { contextBridge, ipcRenderer, webFrame, webUtils } from 'electron'
|
||||
const translucencySupport = ipcRenderer.sendSync('hermes:translucency:support')
|
||||
const hudWindowing = ipcRenderer.sendSync('hermes:hud:windowing')
|
||||
const hudNativeDrag = hudWindowing?.nativeDrag === true
|
||||
const launchFlags = ipcRenderer.sendSync('hermes:launch-flags')
|
||||
|
||||
contextBridge.exposeInMainWorld('hermesDesktop', {
|
||||
glassSupported: translucencySupport?.glass === true,
|
||||
translucencySupported: translucencySupport?.translucency === true,
|
||||
// Launch-flag fact: the app was started with --local, so the renderer may
|
||||
// show the local-models surfaces. Static for the window's lifetime.
|
||||
localModelsEnabled: launchFlags?.localModels === true,
|
||||
getConnection: profile => ipcRenderer.invoke('hermes:connection', profile),
|
||||
// Registry-scoped backend resolution: { connectionId, profile } → descriptor.
|
||||
getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload),
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
import type {
|
||||
LocalCatalogModel,
|
||||
LocalHardware,
|
||||
LocalModelsStatus,
|
||||
LocalRuntimeJob
|
||||
} from '@/types/hermes'
|
||||
|
||||
import { hermesApi, profileScoped } from './client'
|
||||
|
||||
// The desktop surface of the managed llama.cpp runtime: status/catalog
|
||||
// reads, download/install/activate jobs, and server control.
|
||||
|
||||
export function getLocalModelsStatus(): Promise<LocalModelsStatus> {
|
||||
return hermesApi<LocalModelsStatus>({
|
||||
...profileScoped(),
|
||||
path: '/api/local-models/status'
|
||||
})
|
||||
}
|
||||
|
||||
export function getLocalHardware(): Promise<LocalHardware> {
|
||||
return hermesApi<LocalHardware>({
|
||||
...profileScoped(),
|
||||
path: '/api/local-models/hardware'
|
||||
})
|
||||
}
|
||||
|
||||
export function getLocalCatalog(): Promise<{ models: LocalCatalogModel[] }> {
|
||||
return hermesApi<{ models: LocalCatalogModel[] }>({
|
||||
...profileScoped(),
|
||||
path: '/api/local-models/catalog'
|
||||
})
|
||||
}
|
||||
|
||||
export function installLocalRuntime(backend?: string): Promise<{ backend: string; job_id: string; tag: string }> {
|
||||
return hermesApi<{ backend: string; job_id: string; tag: string }>({
|
||||
...profileScoped(),
|
||||
body: { backend: backend ?? null },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/runtime/install'
|
||||
})
|
||||
}
|
||||
|
||||
export interface QuickstartResponse {
|
||||
display_name: string
|
||||
download_bytes: number
|
||||
job_id: string
|
||||
model_id: string
|
||||
needs_download: boolean
|
||||
needs_runtime: boolean
|
||||
}
|
||||
|
||||
export function quickstartLocalModels(modelId?: string): Promise<QuickstartResponse> {
|
||||
return hermesApi<QuickstartResponse>({
|
||||
...profileScoped(),
|
||||
body: { model_id: modelId ?? null },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/quickstart'
|
||||
})
|
||||
}
|
||||
|
||||
export function downloadLocalModel(modelId: string): Promise<{ already_downloaded?: boolean; job_id: null | string }> {
|
||||
return hermesApi<{ already_downloaded?: boolean; job_id: null | string }>({
|
||||
...profileScoped(),
|
||||
body: { model_id: modelId },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/download'
|
||||
})
|
||||
}
|
||||
|
||||
export function deleteLocalModel(modelId: string): Promise<{ ok: boolean }> {
|
||||
return hermesApi<{ ok: boolean }>({
|
||||
...profileScoped(),
|
||||
method: 'DELETE',
|
||||
path: `/api/local-models/models/${encodeURIComponent(modelId)}`
|
||||
})
|
||||
}
|
||||
|
||||
export function getLocalRuntimeJob(jobId: string): Promise<LocalRuntimeJob> {
|
||||
return hermesApi<LocalRuntimeJob>({
|
||||
...profileScoped(),
|
||||
path: `/api/local-models/jobs/${encodeURIComponent(jobId)}`
|
||||
})
|
||||
}
|
||||
|
||||
export function getLocalModelsJobs(): Promise<{ jobs: LocalRuntimeJob[] }> {
|
||||
return hermesApi<{ jobs: LocalRuntimeJob[] }>({
|
||||
...profileScoped(),
|
||||
path: '/api/local-models/jobs'
|
||||
})
|
||||
}
|
||||
|
||||
export function activateLocalModel(modelId: string): Promise<{ job_id: string }> {
|
||||
return hermesApi<{ job_id: string }>({
|
||||
...profileScoped(),
|
||||
body: { model_id: modelId },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/activate'
|
||||
})
|
||||
}
|
||||
|
||||
export function ejectLocalModel(modelId: string): Promise<{ ok: boolean }> {
|
||||
return hermesApi<{ ok: boolean }>({
|
||||
...profileScoped(),
|
||||
body: { model_id: modelId },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/eject'
|
||||
})
|
||||
}
|
||||
|
||||
export function setLocalServer(action: 'start' | 'stop'): Promise<{ ok: boolean }> {
|
||||
return hermesApi<{ ok: boolean }>({
|
||||
...profileScoped(),
|
||||
body: { action },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/server'
|
||||
})
|
||||
}
|
||||
|
||||
// ── Hugging Face browser + sideload ─────────────────────────────
|
||||
|
||||
export interface HFSearchHit {
|
||||
repo: string
|
||||
downloads: number
|
||||
likes: number
|
||||
updated: string
|
||||
gated: boolean
|
||||
}
|
||||
|
||||
export interface HFFileGroup {
|
||||
label: string
|
||||
paths: string[]
|
||||
total_bytes: number
|
||||
fit: 'fits-gpu' | 'needs-ram' | 'too-big' | 'unknown'
|
||||
}
|
||||
|
||||
export function searchHFModels(q: string, limit = 20): Promise<{ hits: HFSearchHit[] }> {
|
||||
return hermesApi<{ hits: HFSearchHit[] }>({
|
||||
...profileScoped(),
|
||||
path: `/api/local-models/search?q=${encodeURIComponent(q)}&limit=${limit}`
|
||||
})
|
||||
}
|
||||
|
||||
export function listHFRepoFiles(repo: string): Promise<{ files: HFFileGroup[] }> {
|
||||
return hermesApi<{ files: HFFileGroup[] }>({
|
||||
...profileScoped(),
|
||||
path: `/api/local-models/search/files?repo=${encodeURIComponent(repo)}`
|
||||
})
|
||||
}
|
||||
|
||||
export function downloadBrowsedModel(repo: string, paths: string[]): Promise<{ already_downloaded?: boolean; job_id: null | string; model_id: string }> {
|
||||
return hermesApi<{ already_downloaded?: boolean; job_id: null | string; model_id: string }>({
|
||||
...profileScoped(),
|
||||
body: { paths, repo },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/download-browsed'
|
||||
})
|
||||
}
|
||||
|
||||
export function sideloadLocalModel(path: string): Promise<{ already_present?: boolean; model_id: string; ok: boolean }> {
|
||||
return hermesApi<{ already_present?: boolean; model_id: string; ok: boolean }>({
|
||||
...profileScoped(),
|
||||
body: { path },
|
||||
method: 'POST',
|
||||
path: '/api/local-models/sideload'
|
||||
})
|
||||
}
|
||||
@@ -44,6 +44,7 @@ import {
|
||||
isCurrentGatewaySwitch,
|
||||
registerGatewaySwitchLifecycle
|
||||
} from '@/store/gateway-switch'
|
||||
import { checkLocalRuntimeUpdate, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import { notify, notifyError } from '@/store/notifications'
|
||||
import {
|
||||
$activeGatewayProfile,
|
||||
@@ -663,6 +664,12 @@ export function useGatewayBoot({
|
||||
|
||||
completeDesktopBoot()
|
||||
bootCompleted = true
|
||||
// Rediscover local-runtime jobs (model downloads, runtime installs)
|
||||
// that were running before a reload — the backend registry is the
|
||||
// authority; this just resumes following it.
|
||||
watchLocalRuntimeJobs()
|
||||
// One-per-session engine-update pointer (enabled runtimes only).
|
||||
void checkLocalRuntimeUpdate()
|
||||
} catch (err) {
|
||||
const mayPublishFailure =
|
||||
!cancelled && (switchToken === null ? !$gatewaySwitching.get() : isCurrentGatewaySwitch(switchToken))
|
||||
|
||||
@@ -12,6 +12,7 @@ import {
|
||||
Archive,
|
||||
BarChart3,
|
||||
Bell,
|
||||
Cpu,
|
||||
Download,
|
||||
Globe,
|
||||
Info,
|
||||
@@ -31,6 +32,7 @@ import { cn } from '@/lib/utils'
|
||||
import { $commandPaletteOpen, openCommandPalettePage } from '@/store/command-palette'
|
||||
import { confirm } from '@/store/confirm'
|
||||
import { bindingsFor } from '@/store/keybinds'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { notifyError } from '@/store/notifications'
|
||||
|
||||
import { useRouteEnumParam } from '../hooks/use-route-enum-param'
|
||||
@@ -217,7 +219,22 @@ export function SettingsView({ onClose, onConfigSaved, onMainModelChanged }: Set
|
||||
id: 'pview:custom-endpoints',
|
||||
label: t.settings.nav.providerCustomEndpoints,
|
||||
onSelect: () => openProviderView('custom-endpoints')
|
||||
}
|
||||
},
|
||||
// Local models ships behind the --local launch flag: no flag, no
|
||||
// nav entry (the pane itself also refuses to render, so a stale
|
||||
// ?pview=local deep link falls back to accounts-shaped emptiness
|
||||
// rather than a hidden feature).
|
||||
...($localModelsEnabled.get()
|
||||
? [
|
||||
{
|
||||
active: activeView === 'providers' && providerView === 'local',
|
||||
icon: Cpu,
|
||||
id: 'pview:local',
|
||||
label: t.settings.nav.providerLocalModels,
|
||||
onSelect: () => openProviderView('local')
|
||||
}
|
||||
]
|
||||
: [])
|
||||
],
|
||||
gapBefore: true,
|
||||
icon: Zap,
|
||||
|
||||
@@ -0,0 +1,556 @@
|
||||
import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
|
||||
import { MemoryRouter, useLocation } from 'react-router'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { I18nProvider } from '@/i18n'
|
||||
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import type { LocalCatalogModel, LocalHardware, LocalModelsStatus, LocalRuntimeJob } from '@/types/hermes'
|
||||
|
||||
import { LocalModelsSettings } from './local-models-settings'
|
||||
|
||||
// Mock the API layer — the pane's contract is what it RENDERS from these
|
||||
// payloads, not transport.
|
||||
vi.mock('@/hermes', () => ({
|
||||
activateLocalModel: vi.fn(),
|
||||
deleteLocalModel: vi.fn(),
|
||||
downloadBrowsedModel: vi.fn(),
|
||||
downloadLocalModel: vi.fn(),
|
||||
ejectLocalModel: vi.fn(),
|
||||
getLocalCatalog: vi.fn(),
|
||||
getLocalHardware: vi.fn(),
|
||||
getLocalModelsJobs: vi.fn(),
|
||||
getLocalModelsStatus: vi.fn(),
|
||||
getLocalRuntimeJob: vi.fn(),
|
||||
installLocalRuntime: vi.fn(),
|
||||
listHFRepoFiles: vi.fn(),
|
||||
quickstartLocalModels: vi.fn(),
|
||||
searchHFModels: vi.fn(),
|
||||
sideloadLocalModel: vi.fn()
|
||||
}))
|
||||
|
||||
import * as hermes from '@/hermes'
|
||||
|
||||
const mocked = vi.mocked(hermes)
|
||||
|
||||
const BASE_STATUS: LocalModelsStatus = {
|
||||
enabled: true,
|
||||
tag: 'b10290',
|
||||
configured_tag: 'b10290',
|
||||
update_available: false,
|
||||
runtime_installed: false,
|
||||
runtime_backend: null,
|
||||
server_running: false,
|
||||
server_base_url: null,
|
||||
active_model_id: null,
|
||||
loaded_models: {},
|
||||
models: [],
|
||||
models_dir: 'C:/somewhere/models'
|
||||
}
|
||||
|
||||
const BASE_HARDWARE: LocalHardware = {
|
||||
uma: false,
|
||||
vram_total_bytes: 32 * 2 ** 30,
|
||||
vram_usable_bytes: 26 * 2 ** 30,
|
||||
ram_total_bytes: 256 * 2 ** 30,
|
||||
ram_available_bytes: 200 * 2 ** 30,
|
||||
vram_label: '32.0 GB',
|
||||
gpu_name: 'NVIDIA GeForce RTX 5090',
|
||||
gpu_util_percent: 12,
|
||||
vram_used_bytes: 6 * 2 ** 30
|
||||
}
|
||||
|
||||
const FITTING_MODEL: LocalCatalogModel = {
|
||||
id: 'Qwen3.6-27B-UD-Q4_K_XL',
|
||||
display_name: 'Qwen3.6 27B',
|
||||
description: 'Best all-round agent model; long context stays fast',
|
||||
size_bytes: 17.6 * 2 ** 30,
|
||||
size_label: '17.6 GB',
|
||||
native_context: 262144,
|
||||
native_context_label: '256K',
|
||||
recommended: true,
|
||||
downloaded: false,
|
||||
mtp: false,
|
||||
fits: true,
|
||||
fit_summary: 'runs at its full 256K context',
|
||||
start_window: 262144,
|
||||
start_window_label: '256K',
|
||||
spilled: false
|
||||
}
|
||||
|
||||
const SPILLED_MODEL: LocalCatalogModel = {
|
||||
...FITTING_MODEL,
|
||||
id: 'Spilled-Model',
|
||||
display_name: 'Spilled Model',
|
||||
recommended: false,
|
||||
fits: true,
|
||||
spilled: true,
|
||||
start_window: 65536,
|
||||
start_window_label: '64K',
|
||||
fit_summary: 'starts at 64K and grows toward 256K as you use it (larger than your GPU memory — runs slower)'
|
||||
}
|
||||
|
||||
const REFUSED_MODEL: LocalCatalogModel = {
|
||||
...FITTING_MODEL,
|
||||
id: 'Huge-Model',
|
||||
display_name: 'Huge Model',
|
||||
recommended: false,
|
||||
fits: false,
|
||||
fit_summary: 'Needs more memory than this machine has',
|
||||
fit_detail: 'needs ~60 GiB at the 64K floor',
|
||||
start_window: undefined,
|
||||
start_window_label: undefined
|
||||
}
|
||||
|
||||
function renderPane() {
|
||||
return render(
|
||||
<MemoryRouter>
|
||||
<I18nProvider>
|
||||
<LocalModelsSettings />
|
||||
</I18nProvider>
|
||||
</MemoryRouter>
|
||||
)
|
||||
}
|
||||
|
||||
// The fresh-machine states these tests exercise now lead with the
|
||||
// quickstart card; the full pane (runtime rows, model list, browser)
|
||||
// is one 'Configure…' click away. Render and click through.
|
||||
async function renderFullPane() {
|
||||
const result = renderPane()
|
||||
const configure = await screen.findByRole('button', { name: /configure/i })
|
||||
|
||||
fireEvent.click(configure)
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue(BASE_STATUS)
|
||||
mocked.getLocalHardware.mockResolvedValue(BASE_HARDWARE)
|
||||
mocked.getLocalCatalog.mockResolvedValue({ models: [FITTING_MODEL, SPILLED_MODEL, REFUSED_MODEL] })
|
||||
mocked.getLocalModelsJobs.mockResolvedValue({ jobs: [] })
|
||||
$localRuntimeJobs.set([])
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
vi.clearAllMocks()
|
||||
})
|
||||
|
||||
describe('LocalModelsSettings', () => {
|
||||
it('offers the runtime install with a plain-language explanation', async () => {
|
||||
await renderFullPane()
|
||||
|
||||
expect(await screen.findByText('Install the local runtime')).toBeTruthy()
|
||||
expect(screen.getByText(/runs? entirely on this machine/i)).toBeTruthy()
|
||||
expect(screen.getByRole('button', { name: /install runtime/i })).toBeTruthy()
|
||||
})
|
||||
|
||||
it('shows every catalog model with fit pills; unaffordable ones stay visible with the reason', async () => {
|
||||
await renderFullPane()
|
||||
|
||||
expect(await screen.findByText('Qwen3.6 27B')).toBeTruthy()
|
||||
// The fitting model reads as pills, not prose: green memory pill +
|
||||
// green full-context pill (start_window == native, resident on GPU).
|
||||
expect(screen.getByText('Fits your GPU')).toBeTruthy()
|
||||
expect(screen.getByText('Full 256K context').className).toContain('emerald')
|
||||
|
||||
// The refused model is NOT hidden (discoverability rule): red memory
|
||||
// pill, plus the ceiling it would have had.
|
||||
expect(screen.getByText('Huge Model')).toBeTruthy()
|
||||
expect(screen.getByText('Too big for this machine')).toBeTruthy()
|
||||
|
||||
// The spilled model reads amber + ONE quiet ceiling pill — the same
|
||||
// 'Up to' shape the refused row wears; no start/grow pair.
|
||||
expect(screen.getByText('Spilled Model')).toBeTruthy()
|
||||
expect(screen.getByText('Uses system RAM')).toBeTruthy()
|
||||
expect(screen.getAllByText('Up to 256K context').length).toBe(2)
|
||||
expect(screen.queryByText(/Starts at/)).toBeNull()
|
||||
|
||||
// Its download button is disabled; the fitting model's is enabled once
|
||||
// the runtime exists (here runtime_installed=false, so both disabled —
|
||||
// asserted separately below).
|
||||
const buttons = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
|
||||
expect(buttons.every(b => (b as HTMLButtonElement).disabled)).toBe(true)
|
||||
})
|
||||
|
||||
it('orders the catalog by fit: resident first, then spilled, then too-big', async () => {
|
||||
// Scrambled input — the pane, not the backend, owns display order.
|
||||
mocked.getLocalCatalog.mockResolvedValue({ models: [REFUSED_MODEL, SPILLED_MODEL, FITTING_MODEL] })
|
||||
await renderFullPane()
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
|
||||
// The matched element is the row-title span; the recommended row's
|
||||
// includes its nested pill copy — strip it before comparing order.
|
||||
const names = screen
|
||||
.getAllByText(/^(Qwen3\.6 27B|Spilled Model|Huge Model)$/)
|
||||
.map(el => el.textContent?.replace('Recommended', ''))
|
||||
|
||||
expect(names).toEqual(['Qwen3.6 27B', 'Spilled Model', 'Huge Model'])
|
||||
})
|
||||
|
||||
it('never greens the full-context pill on a system-RAM model', async () => {
|
||||
// Full native window, but earned by spilling into system RAM: the
|
||||
// pill must not wear the green that would recommend exactly the
|
||||
// wrong model.
|
||||
const spilledFull: LocalCatalogModel = {
|
||||
...FITTING_MODEL,
|
||||
id: 'Spilled-Full',
|
||||
display_name: 'Spilled Full',
|
||||
recommended: false,
|
||||
spilled: true,
|
||||
fit_summary: 'runs its full 256K context, partly from system RAM'
|
||||
}
|
||||
|
||||
mocked.getLocalCatalog.mockResolvedValue({ models: [spilledFull] })
|
||||
await renderFullPane()
|
||||
await screen.findByText('Spilled Full')
|
||||
|
||||
expect(screen.getByText('Full 256K context').className).not.toContain('emerald')
|
||||
})
|
||||
|
||||
it('explains the Recommended pick on hover', async () => {
|
||||
// The tooltip is the resolver's own reason, and it must actually OPEN:
|
||||
// Tip works by asChild-cloning hover handlers onto the pill, so a Pill
|
||||
// that swallows its rest props kills the tooltip silently (the pill
|
||||
// still renders, nothing appears on hover).
|
||||
mocked.getLocalCatalog.mockResolvedValue({
|
||||
models: [{ ...FITTING_MODEL, recommended_reason: 'speed-gated-quality' }]
|
||||
})
|
||||
await renderFullPane()
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
|
||||
fireEvent.pointerMove(screen.getByText('Recommended'))
|
||||
fireEvent.pointerEnter(screen.getByText('Recommended'))
|
||||
|
||||
await waitFor(() =>
|
||||
expect(screen.getAllByText(/would respond too slowly on its memory bandwidth/).length).toBeGreaterThan(0)
|
||||
)
|
||||
})
|
||||
|
||||
it('enables downloads only once the runtime is installed', async () => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
runtime_installed: true,
|
||||
runtime_backend: 'cuda'
|
||||
})
|
||||
await renderFullPane()
|
||||
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
const [fittingButton] = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
|
||||
expect((fittingButton as HTMLButtonElement).disabled).toBe(false)
|
||||
})
|
||||
|
||||
it('shows hardware facts after backfill', async () => {
|
||||
await renderFullPane()
|
||||
|
||||
expect(await screen.findByText('NVIDIA GeForce RTX 5090')).toBeTruthy()
|
||||
expect(screen.getByText(/32\.0 GB GPU memory/)).toBeTruthy()
|
||||
expect(screen.getByText(/256\.0 GB RAM/)).toBeTruthy()
|
||||
})
|
||||
|
||||
it('tracks a download job to completion and refreshes', async () => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
runtime_installed: true,
|
||||
runtime_backend: 'cuda'
|
||||
})
|
||||
mocked.downloadLocalModel.mockResolvedValue({ job_id: 'j1' })
|
||||
|
||||
const running: LocalRuntimeJob = {
|
||||
job_id: 'j1',
|
||||
kind: 'model-download',
|
||||
target: 'Qwen3.6 27B',
|
||||
model_id: FITTING_MODEL.id,
|
||||
status: 'running',
|
||||
phase: 'downloading',
|
||||
detail: 'Qwen3.6 27B — 17.6 GB',
|
||||
total_bytes: 100,
|
||||
done_bytes: 40,
|
||||
percent: 40,
|
||||
error: null
|
||||
}
|
||||
|
||||
mocked.getLocalModelsJobs
|
||||
.mockResolvedValueOnce({ jobs: [running] })
|
||||
.mockResolvedValue({ jobs: [{ ...running, status: 'done', phase: 'done', done_bytes: 100, percent: 100 }] })
|
||||
|
||||
await renderFullPane()
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
|
||||
const [download] = screen.getAllByRole('button', { name: /download · 17\.6 GB/i })
|
||||
download.click()
|
||||
|
||||
// The app-level watcher follows the job; when it settles the pane
|
||||
// refreshes (status + catalog re-fetched).
|
||||
await waitFor(() => {
|
||||
expect(mocked.getLocalModelsJobs).toHaveBeenCalled()
|
||||
expect(mocked.getLocalModelsStatus.mock.calls.length).toBeGreaterThanOrEqual(2)
|
||||
})
|
||||
})
|
||||
|
||||
it('renders progress for a download discovered from the store (survives pane remount)', async () => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
runtime_installed: true,
|
||||
runtime_backend: 'cuda'
|
||||
})
|
||||
// A running job already in the app-level store — as after closing and
|
||||
// reopening the pane mid-download.
|
||||
$localRuntimeJobs.set([
|
||||
{
|
||||
job_id: 'j9',
|
||||
kind: 'model-download',
|
||||
target: 'Qwen3.6 27B',
|
||||
model_id: FITTING_MODEL.id,
|
||||
status: 'running',
|
||||
phase: 'downloading',
|
||||
detail: '',
|
||||
total_bytes: 100,
|
||||
done_bytes: 62,
|
||||
percent: 62,
|
||||
error: null
|
||||
}
|
||||
])
|
||||
|
||||
await renderFullPane()
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
|
||||
// The fitting row shows byte progress; the remaining download
|
||||
// buttons belong to the other rows (spilled + refused).
|
||||
expect(screen.getAllByText(/0\.0 GB of 0\.0 GB|of/).length).toBeGreaterThan(0)
|
||||
const remaining = screen.queryAllByRole('button', { name: /download · 17\.6 GB/i })
|
||||
expect(remaining.length).toBe(2)
|
||||
expect(remaining.some(b => (b as HTMLButtonElement).disabled)).toBe(true)
|
||||
})
|
||||
|
||||
it('surfaces a failed download with the backend message', async () => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
runtime_installed: true,
|
||||
runtime_backend: 'cuda'
|
||||
})
|
||||
$localRuntimeJobs.set([
|
||||
{
|
||||
job_id: 'j2',
|
||||
kind: 'model-download',
|
||||
target: 'Qwen3.6 27B',
|
||||
model_id: FITTING_MODEL.id,
|
||||
status: 'error',
|
||||
phase: 'verifying',
|
||||
detail: '',
|
||||
total_bytes: 100,
|
||||
done_bytes: 100,
|
||||
error: 'Downloaded file failed its integrity check and was removed — try again'
|
||||
}
|
||||
])
|
||||
|
||||
await renderFullPane()
|
||||
await screen.findByText('Qwen3.6 27B')
|
||||
|
||||
expect(await screen.findByText(/integrity check/)).toBeTruthy()
|
||||
})
|
||||
})
|
||||
|
||||
describe('quickstart', () => {
|
||||
it('leads with one button on a fresh machine and fires the quickstart job', async () => {
|
||||
mocked.quickstartLocalModels.mockResolvedValue({
|
||||
display_name: 'Qwen3.6 27B',
|
||||
download_bytes: FITTING_MODEL.size_bytes,
|
||||
job_id: 'q1',
|
||||
model_id: 'qwen3.6-27b',
|
||||
needs_download: true,
|
||||
needs_runtime: true
|
||||
})
|
||||
renderPane()
|
||||
|
||||
// The card names the recommended model and the one-click action; the
|
||||
// runtime/model machinery is NOT on screen.
|
||||
expect(await screen.findByRole('button', { name: /set up for me/i })).toBeTruthy()
|
||||
expect(screen.queryByText('Install the local runtime')).toBeNull()
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: /set up for me/i }))
|
||||
await waitFor(() => {
|
||||
expect(mocked.quickstartLocalModels).toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
it('pins the quickstart progress view while the job runs', async () => {
|
||||
$localRuntimeJobs.set([
|
||||
{
|
||||
job_id: 'q1',
|
||||
kind: 'quickstart',
|
||||
target: 'Qwen3.6 27B',
|
||||
model_id: 'qwen3.6-27b',
|
||||
status: 'running',
|
||||
phase: 'downloading',
|
||||
detail: 'Qwen3.6 27B — 17.6 GB',
|
||||
total_bytes: 100,
|
||||
done_bytes: 30,
|
||||
percent: 30,
|
||||
error: null
|
||||
}
|
||||
])
|
||||
renderPane()
|
||||
|
||||
expect(await screen.findByText('Qwen3.6 27B — 17.6 GB')).toBeTruthy()
|
||||
// One job, one view: no Set up / Configure buttons while it runs.
|
||||
expect(screen.queryByRole('button', { name: /set up for me/i })).toBeNull()
|
||||
})
|
||||
|
||||
it('skips the card entirely once a model is staged', async () => {
|
||||
mocked.getLocalModelsStatus.mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
runtime_installed: true,
|
||||
runtime_backend: 'cuda',
|
||||
models: [{ id: 'Qwen3.6-27B-UD-Q4_K_XL', size_bytes: 17 * 2 ** 30, size_label: '17.6 GB' }]
|
||||
})
|
||||
renderPane()
|
||||
|
||||
// Straight to the full pane — no quickstart hero for a working setup.
|
||||
expect(await screen.findByText('Qwen3.6 27B')).toBeTruthy()
|
||||
expect(screen.queryByRole('button', { name: /set up for me/i })).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('BrowseSection', () => {
|
||||
it('searches HF after a pause and shows fit-priced files on demand', async () => {
|
||||
vi.useFakeTimers()
|
||||
|
||||
try {
|
||||
vi.mocked(hermes.searchHFModels).mockResolvedValue({
|
||||
hits: [{ downloads: 872724, gated: false, likes: 47, repo: 'unsloth/Qwen3.8-27B-GGUF', updated: '2026-08-18' }]
|
||||
})
|
||||
vi.mocked(hermes.listHFRepoFiles).mockResolvedValue({
|
||||
files: [
|
||||
{ fit: 'fits-gpu', label: 'Q4_K_M', paths: ['Qwen3.8-27B-Q4_K_M.gguf'], total_bytes: 17 * 2 ** 30 },
|
||||
{ fit: 'too-big', label: 'F16', paths: ['Qwen3.8-27B-F16.gguf'], total_bytes: 56 * 2 ** 30 }
|
||||
]
|
||||
})
|
||||
|
||||
render(
|
||||
<MemoryRouter>
|
||||
<I18nProvider>
|
||||
<LocalModelsSettings />
|
||||
</I18nProvider>
|
||||
</MemoryRouter>
|
||||
)
|
||||
await act(async () => {
|
||||
await vi.runOnlyPendingTimersAsync()
|
||||
})
|
||||
// Fresh machine leads with the quickstart card — enter the full pane.
|
||||
fireEvent.click(screen.getByRole('button', { name: /configure/i }))
|
||||
|
||||
const box = screen.getByPlaceholderText(/search models/i)
|
||||
fireEvent.change(box, { target: { value: 'qwen' } })
|
||||
// Debounce: no call until the pause elapses.
|
||||
expect(hermes.searchHFModels).not.toHaveBeenCalled()
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(400)
|
||||
})
|
||||
expect(hermes.searchHFModels).toHaveBeenCalledWith('qwen')
|
||||
expect(screen.getByText('unsloth/Qwen3.8-27B-GGUF')).toBeTruthy()
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: /show files/i }))
|
||||
await act(async () => {
|
||||
await vi.runOnlyPendingTimersAsync()
|
||||
})
|
||||
expect(screen.getByText('Q4_K_M')).toBeTruthy()
|
||||
// Each tile has an explicit download button; the too-big quant's is
|
||||
// disabled, the fitting one is live and starts the download.
|
||||
const q4Btn = screen.getByRole('button', { name: 'Download Q4_K_M' })
|
||||
const f16Btn = screen.getByRole('button', { name: 'Download F16' })
|
||||
expect((f16Btn as HTMLButtonElement).disabled).toBe(true)
|
||||
expect((q4Btn as HTMLButtonElement).disabled).toBe(false)
|
||||
|
||||
vi.mocked(hermes.downloadBrowsedModel).mockResolvedValue({ job_id: 'j1', model_id: 'Qwen3.8-27B-Q4_K_M' })
|
||||
fireEvent.click(q4Btn)
|
||||
await act(async () => {
|
||||
await vi.runOnlyPendingTimersAsync()
|
||||
})
|
||||
expect(hermes.downloadBrowsedModel).toHaveBeenCalledWith('unsloth/Qwen3.8-27B-GGUF', ['Qwen3.8-27B-Q4_K_M.gguf'])
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe('added-by-you rows', () => {
|
||||
it('staged models outside the catalog get the full action set', async () => {
|
||||
vi.mocked(hermes.getLocalModelsStatus).mockResolvedValue({
|
||||
...BASE_STATUS,
|
||||
loaded_models: { 'Hermes-4.3-36B-Q5_K_M': 'loaded' },
|
||||
models: [{ id: 'Hermes-4.3-36B-Q5_K_M', size_bytes: 25 * 2 ** 30, size_label: '25.0 GB' }],
|
||||
placement: {
|
||||
'Hermes-4.3-36B-Q5_K_M': {
|
||||
granted_window_label: '96K',
|
||||
spilled: false,
|
||||
window: 98304,
|
||||
window_label: '96K'
|
||||
}
|
||||
},
|
||||
server_running: true
|
||||
})
|
||||
vi.mocked(hermes.getLocalCatalog).mockResolvedValue({ models: [] })
|
||||
|
||||
renderPane()
|
||||
await screen.findByText('Hermes-4.3-36B-Q5_K_M')
|
||||
|
||||
// Full management surface: Use, eject, delete, live placement pill.
|
||||
expect(screen.getByText(/added by you/i)).toBeTruthy()
|
||||
expect(screen.getByRole('button', { name: /use/i })).toBeTruthy()
|
||||
expect(screen.getByText(/96K/)).toBeTruthy()
|
||||
const buttons = screen.getAllByRole('button')
|
||||
expect(buttons.length).toBeGreaterThanOrEqual(3)
|
||||
})
|
||||
})
|
||||
|
||||
describe('quickstart completion navigation', () => {
|
||||
it('lands on a new chat when a quickstart it watched finishes; stale done jobs on mount never navigate', async () => {
|
||||
const routeProbe = vi.fn()
|
||||
|
||||
function Probe() {
|
||||
const loc = useLocation()
|
||||
routeProbe(loc.pathname)
|
||||
|
||||
return null
|
||||
}
|
||||
|
||||
const doneJob: LocalRuntimeJob = {
|
||||
done_bytes: 0,
|
||||
detail: '',
|
||||
error: null,
|
||||
job_id: 'stale-done',
|
||||
kind: 'quickstart',
|
||||
model_id: 'qwen3.8-27b',
|
||||
phase: 'done',
|
||||
status: 'done',
|
||||
target: 'Qwen3.8 27B',
|
||||
total_bytes: null
|
||||
}
|
||||
|
||||
// A finished quickstart already in history when the pane mounts —
|
||||
// must NOT trigger navigation.
|
||||
$localRuntimeJobs.set([doneJob])
|
||||
|
||||
render(
|
||||
<MemoryRouter initialEntries={['/settings']}>
|
||||
<I18nProvider>
|
||||
<LocalModelsSettings />
|
||||
</I18nProvider>
|
||||
<Probe />
|
||||
</MemoryRouter>
|
||||
)
|
||||
await act(async () => {})
|
||||
expect(routeProbe).not.toHaveBeenCalledWith('/')
|
||||
|
||||
// A quickstart the pane SAW running that then completes -> navigate.
|
||||
const running: LocalRuntimeJob = { ...doneJob, job_id: 'live-run', phase: 'downloading', status: 'running' }
|
||||
await act(async () => {
|
||||
$localRuntimeJobs.set([doneJob, running])
|
||||
})
|
||||
await act(async () => {
|
||||
$localRuntimeJobs.set([doneJob, { ...running, phase: 'done', status: 'done' }])
|
||||
})
|
||||
expect(routeProbe).toHaveBeenCalledWith('/')
|
||||
})
|
||||
})
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,4 +1,4 @@
|
||||
import type { ReactNode } from 'react'
|
||||
import type { ComponentProps, ReactNode } from 'react'
|
||||
|
||||
import { Badge } from '@/components/ui/badge'
|
||||
import { Button } from '@/components/ui/button'
|
||||
@@ -22,10 +22,28 @@ export function SettingsContent({ children, bare = false }: { children: ReactNod
|
||||
)
|
||||
}
|
||||
|
||||
const PILL_VARIANT = { muted: 'muted', primary: 'default', warn: 'warn' } as const
|
||||
const PILL_VARIANT = {
|
||||
muted: 'muted',
|
||||
primary: 'default',
|
||||
success: 'success',
|
||||
warn: 'warn',
|
||||
destructive: 'destructive'
|
||||
} as const
|
||||
|
||||
export function Pill({ tone = 'muted', children }: { tone?: keyof typeof PILL_VARIANT; children: ReactNode }) {
|
||||
return <Badge variant={PILL_VARIANT[tone]}>{children}</Badge>
|
||||
// Rest props spread through to the Badge's DOM node — REQUIRED for Radix
|
||||
// `asChild` composition (wrapping a Pill in `Tip` clones it with the hover
|
||||
// handlers and ref as props; swallowing them left every tooltip on a Pill
|
||||
// silently dead).
|
||||
export function Pill({
|
||||
tone = 'muted',
|
||||
children,
|
||||
...props
|
||||
}: { tone?: keyof typeof PILL_VARIANT; children: ReactNode } & Omit<ComponentProps<typeof Badge>, 'variant'>) {
|
||||
return (
|
||||
<Badge variant={PILL_VARIANT[tone]} {...props}>
|
||||
{children}
|
||||
</Badge>
|
||||
)
|
||||
}
|
||||
|
||||
export function SectionHeading({
|
||||
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
FEATURED_ID,
|
||||
FeaturedProviderRow,
|
||||
FireworksProviderRow,
|
||||
LocalModelsProviderRow,
|
||||
OpenRouterProviderRow,
|
||||
ProviderRow,
|
||||
providerTitle,
|
||||
@@ -21,6 +22,7 @@ import { Check, ChevronDown, ChevronRight, KeyRound, Loader2, Terminal, Trash2 }
|
||||
import { normalize } from '@/lib/text'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { confirm } from '@/store/confirm'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { notify, notifyError } from '@/store/notifications'
|
||||
import { $desktopOnboarding, startManualLocalEndpoint, startManualProviderOAuth } from '@/store/onboarding'
|
||||
import type { EnvVarInfo, OAuthProvider } from '@/types/hermes'
|
||||
@@ -29,6 +31,7 @@ import { isKeyVar, ProviderKeyRows } from './credential-key-ui'
|
||||
import { CustomEndpointsSettings } from './custom-endpoints-settings'
|
||||
import { SettingsCategoryHeading, useEnvCredentials } from './env-credentials'
|
||||
import { providerGroup, providerMeta, providerPriority } from './helpers'
|
||||
import { LocalModelsSettings } from './local-models-settings'
|
||||
import { SettingsContent, SettingsSkeleton } from './primitives'
|
||||
|
||||
// The embedded terminal (and thus the "run disconnect command" path) only
|
||||
@@ -46,7 +49,7 @@ function GroupLabel({ children }: { children: ReactNode }) {
|
||||
}
|
||||
|
||||
// Sub-views surfaced as a sidebar subnav: account sign-in vs raw API keys.
|
||||
export const PROVIDER_VIEWS = ['accounts', 'keys', 'custom-endpoints'] as const
|
||||
export const PROVIDER_VIEWS = ['accounts', 'keys', 'custom-endpoints', 'local'] as const
|
||||
|
||||
export type ProviderView = (typeof PROVIDER_VIEWS)[number]
|
||||
|
||||
@@ -117,24 +120,26 @@ function buildProviderKeyGroups(vars: Record<string, EnvVarInfo>): ProviderKeyGr
|
||||
|
||||
// Deliberately a near-1:1 replica of the first-run onboarding picker
|
||||
// (`Picker` in desktop-onboarding-overlay): same recommended card, same
|
||||
// Fireworks #2 quick-key row, same provider rows, same "Other providers"
|
||||
// disclosure, same OpenRouter quick-key row, and the same bottom-right
|
||||
// "I have an API key" affordance. The leaf cards are the exact shared
|
||||
// components, so the two surfaces stay visually identical. Selecting a
|
||||
// provider hands off to the shared onboarding overlay, which runs that
|
||||
// provider's real sign-in flow; the key affordances open the API-key
|
||||
// catalog below.
|
||||
// always-visible Local models row, same provider rows, same "Other
|
||||
// providers" disclosure (Fireworks and OpenRouter quick-key rows live
|
||||
// inside it on both surfaces), and the same bottom-right "I have an API
|
||||
// key" affordance. The leaf cards are the exact shared components, so
|
||||
// the two surfaces stay visually identical. Selecting a provider hands
|
||||
// off to the shared onboarding overlay, which runs that provider's real
|
||||
// sign-in flow; the key affordances open the API-key catalog below.
|
||||
function OAuthPicker({
|
||||
disconnecting,
|
||||
onDisconnect,
|
||||
onTerminalDisconnect,
|
||||
onWantApiKey,
|
||||
onWantLocalModels,
|
||||
providers
|
||||
}: {
|
||||
disconnecting: null | string
|
||||
onDisconnect: (provider: OAuthProvider) => void
|
||||
onTerminalDisconnect: (provider: OAuthProvider) => void
|
||||
onWantApiKey: () => void
|
||||
onWantLocalModels: () => void
|
||||
providers: OAuthProvider[]
|
||||
}) {
|
||||
const { t } = useI18n()
|
||||
@@ -176,8 +181,9 @@ function OAuthPicker({
|
||||
{p.intro}
|
||||
</p>
|
||||
{featured && <FeaturedProviderRow onSelect={select} provider={featured} />}
|
||||
{/* Slot #2 — always visible, matching onboarding / CANONICAL_PROVIDERS. */}
|
||||
<FireworksProviderRow onClick={onWantApiKey} />
|
||||
{/* Slot #2 — the no-account path, matching onboarding. Behind the
|
||||
--local launch flag like every local-models surface. */}
|
||||
{$localModelsEnabled.get() && <LocalModelsProviderRow onClick={onWantLocalModels} />}
|
||||
{connected.length > 0 && (
|
||||
<>
|
||||
<GroupLabel>{p.connected}</GroupLabel>
|
||||
@@ -199,6 +205,7 @@ function OAuthPicker({
|
||||
{others.map(p => (
|
||||
<ProviderRow key={p.id} onSelect={select} provider={p} />
|
||||
))}
|
||||
<FireworksProviderRow onClick={onWantApiKey} />
|
||||
<OpenRouterProviderRow onClick={onWantApiKey} />
|
||||
</>
|
||||
)}
|
||||
@@ -507,6 +514,13 @@ export function ProvidersSettings({
|
||||
return <CustomEndpointsSettings onConfigSaved={onConfigSaved} onMainModelChanged={onMainModelChanged} />
|
||||
}
|
||||
|
||||
if (view === 'local') {
|
||||
// Strict --local gate: without the launch flag the pane doesn't render
|
||||
// even when local models are configured — a stale ?pview=local deep link
|
||||
// (or an old shortcut) lands on the accounts view instead.
|
||||
return $localModelsEnabled.get() ? <LocalModelsSettings /> : null
|
||||
}
|
||||
|
||||
return (
|
||||
<SettingsContent>
|
||||
<OAuthPicker
|
||||
@@ -514,6 +528,7 @@ export function ProvidersSettings({
|
||||
onDisconnect={provider => void handleDisconnect(provider)}
|
||||
onTerminalDisconnect={provider => void handleTerminalDisconnect(provider)}
|
||||
onWantApiKey={() => onViewChange('keys')}
|
||||
onWantLocalModels={() => onViewChange('local')}
|
||||
providers={oauthProviders}
|
||||
/>
|
||||
</SettingsContent>
|
||||
|
||||
@@ -8,6 +8,7 @@ import { useApprovalModeStatusbarItem } from '@/app/shell/approval-mode-menu'
|
||||
import { ContextUsagePanel } from '@/app/shell/context-usage-panel'
|
||||
import { GatewayMenuPanel } from '@/app/shell/gateway-menu-panel'
|
||||
import { useContextBreakdown } from '@/app/shell/hooks/use-context-breakdown'
|
||||
import { useSystemResourcesStatusbarItem } from '@/app/shell/system-resources-statusbar'
|
||||
import { $paneVisible, togglePaneVisible } from '@/components/pane-shell/tree/store'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
@@ -268,6 +269,7 @@ export function useStatusbarItems({
|
||||
const contextBar = useMemo(() => contextBarLabel(gaugeUsage), [gaugeUsage])
|
||||
|
||||
const approvalModeItem = useApprovalModeStatusbarItem(activeGatewayProfile, requestGateway)
|
||||
const systemResourcesItem = useSystemResourcesStatusbarItem()
|
||||
|
||||
const gatewayMenuContent = useMemo(
|
||||
() => (close: () => void) => (
|
||||
@@ -546,9 +548,12 @@ export function useStatusbarItems({
|
||||
},
|
||||
{
|
||||
detail: contextBar || undefined,
|
||||
hidden: !contextUsage,
|
||||
// Never self-hide: the user opted this item in (it's hidden-by-
|
||||
// default), so an empty label must render as a waiting placeholder,
|
||||
// not a vanished item — an enabled-but-invisible toggle reads as
|
||||
// "another item took its spot".
|
||||
id: 'context-usage',
|
||||
label: contextUsage,
|
||||
label: contextUsage || '—',
|
||||
menuAlign: 'end',
|
||||
menuClassName: 'w-auto border-(--ui-stroke-secondary) p-0',
|
||||
menuContent: (
|
||||
@@ -565,6 +570,7 @@ export function useStatusbarItems({
|
||||
toggleLabel: copy.toggleSessionTimer,
|
||||
variant: 'text'
|
||||
},
|
||||
systemResourcesItem,
|
||||
{
|
||||
...approvalModeItem,
|
||||
hidden: gatewayState !== 'open',
|
||||
@@ -598,6 +604,7 @@ export function useStatusbarItems({
|
||||
gaugeUsage,
|
||||
sessionStartedAt,
|
||||
gatewayState,
|
||||
systemResourcesItem,
|
||||
terminalShowing,
|
||||
turnStartedAt
|
||||
]
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
|
||||
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
|
||||
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
|
||||
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { DropdownMenu, DropdownMenuContent } from '@/components/ui/dropdown-menu'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import {
|
||||
$modelVisibilityOpen,
|
||||
$visibleModels,
|
||||
@@ -10,6 +12,7 @@ import {
|
||||
setModelVisibilityOpen,
|
||||
setVisibleModels
|
||||
} from '@/store/model-visibility'
|
||||
import type { LocalRuntimeJob } from '@/types/hermes'
|
||||
|
||||
import { ModelCatalogMenu, type ModelMenuController } from './model-catalog-menu'
|
||||
|
||||
@@ -24,11 +27,23 @@ const getGlobalModelOptions = vi.fn()
|
||||
|
||||
vi.mock('@/hermes', () => ({
|
||||
getGlobalModelOptions: (...args: unknown[]) => getGlobalModelOptions(...args),
|
||||
// The menu kicks the app-level job poller on mount; echo the store so a
|
||||
// poll can't wipe the jobs a test staged (the real backend is authority,
|
||||
// and here the store plays that part).
|
||||
getLocalModelsJobs: vi.fn(async () => {
|
||||
const { $localRuntimeJobs } = await import('@/store/local-runtime-jobs')
|
||||
|
||||
return { jobs: [...$localRuntimeJobs.get()] }
|
||||
}),
|
||||
getLocalModelsStatus: vi.fn().mockResolvedValue({ loading: {} }),
|
||||
setApiRequestProfile: vi.fn()
|
||||
}))
|
||||
|
||||
beforeEach(() => {
|
||||
$visibleModels.set(null)
|
||||
$localRuntimeJobs.set([])
|
||||
// These suites exercise the local-models rows, which ship behind --local.
|
||||
$localModelsEnabled.set(true)
|
||||
setModelVisibilityOpen(false)
|
||||
getGlobalModelOptions.mockResolvedValue({
|
||||
providers: [{ models: ['gemini-3.1-pro', 'gemini-2.5-flash'], name: 'Google', slug: 'google' }]
|
||||
@@ -106,3 +121,77 @@ describe('the catalog owns model curation', () => {
|
||||
expect($modelVisibilityOpen.get()).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('in-flight local downloads', () => {
|
||||
const DOWNLOAD_JOB: LocalRuntimeJob = {
|
||||
job_id: 'dl1',
|
||||
kind: 'model-download',
|
||||
target: 'Qwen3.8 Flash Next (UD-Q4_K_XL)',
|
||||
model_id: 'qwen3.8-flash-next',
|
||||
status: 'running',
|
||||
phase: 'downloading',
|
||||
detail: '',
|
||||
total_bytes: 100,
|
||||
done_bytes: 41,
|
||||
percent: 41,
|
||||
error: null
|
||||
}
|
||||
|
||||
it('shows a downloading model as a disabled progress row in its own Local group', async () => {
|
||||
// No llamacpp provider in the catalog (first-ever download).
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderMenu()
|
||||
await screen.findByText(/Gemini 3\.1 Pro/i)
|
||||
|
||||
const row = screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
|
||||
|
||||
expect(row).toBeTruthy()
|
||||
expect(screen.getByText('41%')).toBeTruthy()
|
||||
expect(row.closest('[role="menuitem"]')?.getAttribute('aria-disabled')).toBe('true')
|
||||
})
|
||||
|
||||
it('shows the download inside the Local provider group when it exists', async () => {
|
||||
getGlobalModelOptions.mockResolvedValue({
|
||||
providers: [
|
||||
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
|
||||
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
|
||||
]
|
||||
})
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderMenu()
|
||||
|
||||
await screen.findByText(/Qwen3\.6 27B/i)
|
||||
expect(screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
|
||||
// One Local heading — the trailing fallback group must not double up.
|
||||
expect(screen.getAllByText('Local').length).toBe(1)
|
||||
})
|
||||
|
||||
it('drops the placeholder row once the download settles', async () => {
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderMenu()
|
||||
await screen.findByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
|
||||
|
||||
$localRuntimeJobs.set([{ ...DOWNLOAD_JOB, status: 'done', phase: 'done' }])
|
||||
await waitFor(() => {
|
||||
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
it('hides the local provider group and download rows without the --local flag (strict)', async () => {
|
||||
$localModelsEnabled.set(false)
|
||||
getGlobalModelOptions.mockResolvedValue({
|
||||
providers: [
|
||||
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
|
||||
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
|
||||
]
|
||||
})
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderMenu()
|
||||
|
||||
// Staged models exist and a download is running — none of it shows.
|
||||
await screen.findByText(/Gemini 3\.1 Pro/i)
|
||||
expect(screen.queryByText(/Qwen3\.6 27B/i)).toBeNull()
|
||||
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
|
||||
expect(screen.queryByText('Local')).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -19,12 +19,16 @@ import { HighlightMatches } from '@/components/ui/highlight-matches'
|
||||
import { usePointerQuiet } from '@/components/ui/keyboard-first'
|
||||
import { Skeleton } from '@/components/ui/skeleton'
|
||||
import type { HermesGateway } from '@/hermes'
|
||||
import { getLocalModelsStatus } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options'
|
||||
import { displayModelName, modelDisplayParts } from '@/lib/model-status-label'
|
||||
import { DEFAULT_REASONING_EFFORT, reasoningEffortLabel } from '@/lib/reasoning-effort'
|
||||
import { normalize } from '@/lib/text'
|
||||
import { useStoreSelector } from '@/lib/use-session-slice'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $localRuntimeJobs, runningModelDownloads, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import {
|
||||
$visibleModels,
|
||||
collapseModelFamilies,
|
||||
@@ -36,7 +40,7 @@ import {
|
||||
} from '@/store/model-visibility'
|
||||
import { $collapsedProviders, toggleCollapsedProvider } from '@/store/provider-collapse'
|
||||
import { $defaultReasoningEffort } from '@/store/session'
|
||||
import type { ModelOptionProvider, ModelOptionsResponse } from '@/types/hermes'
|
||||
import type { LocalModelLoadProgress, ModelOptionProvider, ModelOptionsResponse } from '@/types/hermes'
|
||||
|
||||
import { type FastControl, ModelEditSubmenu, resolveFastControl } from './model-edit-submenu'
|
||||
|
||||
@@ -134,6 +138,7 @@ export function ModelCatalogMenu({
|
||||
}: ModelCatalogMenuProps) {
|
||||
const { t } = useI18n()
|
||||
const copy = t.shell.modelMenu
|
||||
const copyPicker = t.modelPicker
|
||||
const closeMenu = useContext(ModelMenuCloseContext)
|
||||
const [search, setSearch] = useState('')
|
||||
const collapsedProviders = useStoreCollapsed()
|
||||
@@ -154,6 +159,78 @@ export function ModelCatalogMenu({
|
||||
|
||||
const loading = modelOptions.isPending && !modelOptions.data
|
||||
|
||||
// Every local-models read in this menu sits behind the --local launch
|
||||
// flag: no status polling, no download rows, and the llamacpp provider
|
||||
// group hides even when models are staged (the flag is strict).
|
||||
const localModelsEnabled = $localModelsEnabled.get()
|
||||
|
||||
// Live load state for the managed local server: which model is loading
|
||||
// into memory right now, with a REAL percent (per-tensor callback relayed
|
||||
// over the router's SSE stream). Polled only while this menu is mounted
|
||||
// (it unmounts on close); errors read as "nothing loading" — remote-only
|
||||
// installs have no local-models routes.
|
||||
const localStatus = useQuery({
|
||||
queryKey: ['local-models-loading', profile],
|
||||
queryFn: () => getLocalModelsStatus(),
|
||||
enabled: localModelsEnabled,
|
||||
refetchInterval: 2_000,
|
||||
retry: false
|
||||
})
|
||||
|
||||
const loadingModels: Record<string, LocalModelLoadProgress> = localStatus.data?.loading ?? {}
|
||||
|
||||
// Models on their way into the local library (downloads + quickstart runs
|
||||
// still fetching bytes) — rendered as disabled progress rows so the user
|
||||
// sees the model coming instead of wondering where it went. The jobs store
|
||||
// republishes every ~700ms with fresh byte counts while anything runs; a
|
||||
// whole-store subscription here would re-render the entire menu per tick
|
||||
// (breaking open submenus and focus — the #72163 class). Subscribe to a
|
||||
// STABLE identity projection instead: it changes only when a download
|
||||
// starts or ends. Each row selects its own percent scalar.
|
||||
const downloadsKey = useStoreSelector($localRuntimeJobs, jobs =>
|
||||
localModelsEnabled
|
||||
? runningModelDownloads(jobs)
|
||||
.map(job => `${job.job_id}\u0000${job.target}`)
|
||||
.join('\u0001')
|
||||
: ''
|
||||
)
|
||||
|
||||
const downloads = useMemo(
|
||||
() =>
|
||||
downloadsKey === ''
|
||||
? []
|
||||
: downloadsKey.split('\u0001').map(pair => {
|
||||
const [jobId, target] = pair.split('\u0000')
|
||||
|
||||
return { jobId, target }
|
||||
}),
|
||||
[downloadsKey]
|
||||
)
|
||||
|
||||
useEffect(() => {
|
||||
if (localModelsEnabled) {
|
||||
watchLocalRuntimeJobs()
|
||||
}
|
||||
}, [localModelsEnabled])
|
||||
|
||||
// A finished download turns into a real selectable model: refetch the
|
||||
// catalog so the placeholder row is replaced while the menu is open.
|
||||
const refetchOptions = modelOptions.refetch
|
||||
|
||||
useEffect(() => {
|
||||
let prevActive = runningModelDownloads($localRuntimeJobs.get()).length > 0
|
||||
|
||||
return $localRuntimeJobs.listen(next => {
|
||||
const active = runningModelDownloads(next).length > 0
|
||||
|
||||
if (prevActive && !active) {
|
||||
void refetchOptions()
|
||||
}
|
||||
|
||||
prevActive = active
|
||||
})
|
||||
}, [refetchOptions])
|
||||
|
||||
const error = modelOptions.error
|
||||
? modelOptions.error instanceof Error
|
||||
? modelOptions.error.message
|
||||
@@ -170,12 +247,27 @@ export function ModelCatalogMenu({
|
||||
)
|
||||
|
||||
const pickerProviders = useMemo(
|
||||
() => providers?.filter(provider => provider.slug.toLowerCase() !== 'moa') ?? [],
|
||||
[providers]
|
||||
() =>
|
||||
providers?.filter(
|
||||
provider =>
|
||||
provider.slug.toLowerCase() !== 'moa' &&
|
||||
// Strict --local gate: staged local models exist on disk, but
|
||||
// without the flag the GUI doesn't offer them.
|
||||
(localModelsEnabled || provider.slug !== LOCAL_PROVIDER_SLUG)
|
||||
) ?? [],
|
||||
[providers, localModelsEnabled]
|
||||
)
|
||||
|
||||
const current = controller.current
|
||||
|
||||
const q = normalize(search)
|
||||
|
||||
// In-flight downloads render inside the Local provider group when it
|
||||
// exists, else as their own trailing 'Local' group (first download —
|
||||
// nothing staged yet, so the catalog has no local provider row).
|
||||
const shownDownloads = q ? downloads.filter(job => (job.target || '').toLowerCase().includes(q)) : downloads
|
||||
const hasLocalGroup = pickerProviders.some(provider => provider.slug === LOCAL_PROVIDER_SLUG)
|
||||
|
||||
// Resolve visibility HERE, against the catalog we actually fetched: an empty
|
||||
// provider list would otherwise resolve to an empty key set that reads as
|
||||
// "user hid everything" and blanks the menu on first open.
|
||||
@@ -189,8 +281,6 @@ export function ModelCatalogMenu({
|
||||
[pickerProviders, search, current.model, current.provider, shownKeys]
|
||||
)
|
||||
|
||||
const q = normalize(search)
|
||||
|
||||
// Presets are searchable rows like everything else — an unfiltered preset
|
||||
// sitting under zero model matches would otherwise become the "first match"
|
||||
// Enter commits.
|
||||
@@ -367,7 +457,7 @@ export function ModelCatalogMenu({
|
||||
<DropdownMenuItem className={dropdownMenuRow} disabled>
|
||||
{error}
|
||||
</DropdownMenuItem>
|
||||
) : groups.length === 0 && moaPresets.length === 0 ? (
|
||||
) : groups.length === 0 && moaPresets.length === 0 && shownDownloads.length === 0 ? (
|
||||
<DropdownMenuItem className={dropdownMenuRow} disabled>
|
||||
{copy.noModels}
|
||||
</DropdownMenuItem>
|
||||
@@ -412,6 +502,10 @@ export function ModelCatalogMenu({
|
||||
const isCurrent = activeId !== null
|
||||
const name = modelDisplayParts(family.id).name
|
||||
const caps = group.provider.capabilities?.[family.id]
|
||||
// Managed local model loading into memory right now:
|
||||
// real load percent, keyed by exact model id (remote
|
||||
// providers never collide with GGUF stems).
|
||||
const loadProgress = loadingModels[family.id] ?? (family.fastId ? loadingModels[family.fastId] : undefined)
|
||||
|
||||
// Effective settings for this row: the live choice when it's
|
||||
// the active model, otherwise its remembered preset. Row
|
||||
@@ -461,8 +555,28 @@ export function ModelCatalogMenu({
|
||||
<HighlightMatches query={search} text={name} />
|
||||
{meta ? <span className="text-(--ui-text-tertiary)"> {meta}</span> : null}
|
||||
</span>
|
||||
{loadProgress ? (
|
||||
<span
|
||||
className="ml-auto flex shrink-0 items-center gap-1.5"
|
||||
title={copyPicker.loadingIntoMemory}
|
||||
>
|
||||
<span className="h-1 w-14 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<span
|
||||
className="block h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(2, loadProgress.percent)}%` }}
|
||||
/>
|
||||
</span>
|
||||
<span className="text-[0.62rem] tabular-nums text-(--ui-text-tertiary)">
|
||||
{loadProgress.percent}%
|
||||
</span>
|
||||
</span>
|
||||
) : null}
|
||||
{isCurrent ? (
|
||||
<Codicon className="ml-auto text-foreground" name="check" size="0.75rem" />
|
||||
<Codicon
|
||||
className={cn('text-foreground', loadProgress ? 'ml-1' : 'ml-auto')}
|
||||
name="check"
|
||||
size="0.75rem"
|
||||
/>
|
||||
) : null}
|
||||
</DropdownMenuSubTrigger>
|
||||
<ModelEditSubmenu
|
||||
@@ -486,9 +600,22 @@ export function ModelCatalogMenu({
|
||||
</DropdownMenuSub>
|
||||
)
|
||||
})}
|
||||
{!collapsed &&
|
||||
slug === LOCAL_PROVIDER_SLUG &&
|
||||
shownDownloads.map(job => <DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />)}
|
||||
</DropdownMenuGroup>
|
||||
)
|
||||
})}
|
||||
{!hasLocalGroup && shownDownloads.length > 0 && (
|
||||
<DropdownMenuGroup className="py-0.5" key="local-downloads">
|
||||
<DropdownMenuLabel className="px-2 pb-0.5 pt-0.5 text-[0.625rem] font-semibold uppercase tracking-wider text-(--ui-text-tertiary)">
|
||||
{copyPicker.localDownloadsHeading}
|
||||
</DropdownMenuLabel>
|
||||
{shownDownloads.map(job => (
|
||||
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
|
||||
))}
|
||||
</DropdownMenuGroup>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
@@ -540,6 +667,47 @@ export function ModelCatalogMenu({
|
||||
/** Re-exported so callers building a footer row match the catalog's rows. */
|
||||
export { dropdownMenuRow }
|
||||
|
||||
// The backend's provider row for staged local models (inventory.py's
|
||||
// _local_runtime_row). Downloads-in-flight attach to this group.
|
||||
const LOCAL_PROVIDER_SLUG = 'llamacpp'
|
||||
|
||||
// A model still downloading: visible so the user knows it's coming (and
|
||||
// where it will land), disabled so it can't be selected early, with the
|
||||
// same byte progress the Local Models pane shows. Percent is selected HERE,
|
||||
// per row, so the 700ms byte ticks repaint this leaf only — the menu tree
|
||||
// above subscribes to download identity, not progress.
|
||||
function DownloadingModelRow({ jobId, target }: { jobId: string; target: string }) {
|
||||
const { t } = useI18n()
|
||||
const copy = t.modelPicker
|
||||
|
||||
const percent = useStoreSelector(
|
||||
$localRuntimeJobs,
|
||||
jobs => jobs.find(job => job.job_id === jobId)?.percent ?? null
|
||||
)
|
||||
|
||||
return (
|
||||
<DropdownMenuItem
|
||||
className={cn(dropdownMenuRow, 'opacity-60')}
|
||||
disabled
|
||||
onSelect={event => event.preventDefault()}
|
||||
textValue=""
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate">{target}</span>
|
||||
<span className="ml-auto flex shrink-0 items-center gap-1.5" title={copy.downloading}>
|
||||
<span className="h-1 w-14 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<span
|
||||
className="block h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(2, percent ?? 0)}%` }}
|
||||
/>
|
||||
</span>
|
||||
<span className="text-[0.62rem] tabular-nums text-(--ui-text-tertiary)">
|
||||
{typeof percent === 'number' ? `${percent}%` : copy.downloading}
|
||||
</span>
|
||||
</span>
|
||||
</DropdownMenuItem>
|
||||
)
|
||||
}
|
||||
|
||||
// Collapsed we show the user's chosen models (or the curated default); typing
|
||||
// spans every available model so anything is reachable past the cut. A search
|
||||
// is itself a narrowing action, so we do NOT cap per-provider matches.
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
import { useStore } from '@nanostores/react'
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
import type { StatusbarItem } from '@/app/shell/statusbar-controls'
|
||||
import { getLocalHardware } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { Activity } from '@/lib/icons'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $statusbarHiddenIds } from '@/store/statusbar-prefs'
|
||||
import type { LocalHardware } from '@/types/hermes'
|
||||
|
||||
// Live host-resource readout for the bottom bar: GPU utilization + VRAM +
|
||||
// RAM, fed by /api/local-models/hardware. Hidden by default (an item most
|
||||
// users don't watch); the poll runs ONLY while the item is shown, so the
|
||||
// hidden default costs nothing. 5s cadence — resource numbers, not a
|
||||
// heartbeat.
|
||||
const POLL_MS = 5_000
|
||||
|
||||
function gb(bytes: number | null | undefined): string {
|
||||
return bytes ? `${(bytes / (1 << 30)).toFixed(0)}G` : '—'
|
||||
}
|
||||
|
||||
function gbLong(bytes: number | null | undefined): string {
|
||||
return bytes ? `${(bytes / (1 << 30)).toFixed(1)} GB` : '—'
|
||||
}
|
||||
|
||||
function MeterRow({ label, percent, value }: { label: string; percent: number | null; value: string }) {
|
||||
return (
|
||||
<div className="grid gap-1">
|
||||
<div className="flex items-baseline justify-between gap-2">
|
||||
{/* Label yields, value never does: if anything ever narrows the row
|
||||
again, a truncated label beats a clipped number — "15.2 GB" losing
|
||||
its tail reads as a wrong number, not a cut one. */}
|
||||
<span className="truncate text-muted-foreground">{label}</span>
|
||||
|
||||
<span className="shrink-0 whitespace-nowrap tabular-nums text-foreground">{value}</span>
|
||||
</div>
|
||||
|
||||
{percent !== null && (
|
||||
<div className="h-1.5 w-full overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<div
|
||||
className="h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(1, Math.min(100, percent))}%` }}
|
||||
/>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
export function useSystemResourcesStatusbarItem(): StatusbarItem {
|
||||
const { t } = useI18n()
|
||||
const copy = t.shell.statusbar.systemResources
|
||||
const hiddenIds = useStore($statusbarHiddenIds)
|
||||
// Behind the --local launch flag: without it the item is absent from the
|
||||
// bar AND from the customize menu (no toggleLabel), and never polls.
|
||||
const enabled = $localModelsEnabled.get()
|
||||
const shown = enabled && !hiddenIds.includes('system-resources')
|
||||
const [hardware, setHardware] = useState<LocalHardware | null>(null)
|
||||
|
||||
useEffect(() => {
|
||||
if (!shown) {
|
||||
return
|
||||
}
|
||||
|
||||
let cancelled = false
|
||||
let timer: number | null = null
|
||||
|
||||
const poll = async () => {
|
||||
try {
|
||||
const next = await getLocalHardware()
|
||||
|
||||
if (!cancelled) {
|
||||
setHardware(next)
|
||||
}
|
||||
} catch {
|
||||
if (!cancelled) {
|
||||
setHardware(null)
|
||||
}
|
||||
}
|
||||
|
||||
if (!cancelled) {
|
||||
timer = window.setTimeout(() => void poll(), POLL_MS)
|
||||
}
|
||||
}
|
||||
|
||||
void poll()
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
|
||||
if (timer !== null) {
|
||||
window.clearTimeout(timer)
|
||||
}
|
||||
}
|
||||
}, [shown])
|
||||
|
||||
const hasGpu = Boolean(hardware?.gpu_name)
|
||||
|
||||
const vramPercent =
|
||||
hardware?.vram_used_bytes != null && hardware.vram_total_bytes
|
||||
? Math.round((hardware.vram_used_bytes / hardware.vram_total_bytes) * 100)
|
||||
: null
|
||||
|
||||
const ramUsed = hardware ? hardware.ram_total_bytes - hardware.ram_available_bytes : null
|
||||
const ramPercent = hardware?.ram_total_bytes && ramUsed != null ? Math.round((ramUsed / hardware.ram_total_bytes) * 100) : null
|
||||
|
||||
// Compact bar label: the numbers a local-inference user glances at.
|
||||
// "GPU 34% · 18G/32G" with a GPU; "RAM 41G/256G" without.
|
||||
const label = hardware
|
||||
? hasGpu
|
||||
? `GPU ${hardware.gpu_util_percent ?? 0}%${
|
||||
hardware.vram_used_bytes != null ? ` · ${gb(hardware.vram_used_bytes)}/${gb(hardware.vram_total_bytes)}` : ''
|
||||
}`
|
||||
: `RAM ${gb(ramUsed)}/${gb(hardware.ram_total_bytes)}`
|
||||
: copy.loading
|
||||
|
||||
return {
|
||||
detail: undefined,
|
||||
hidden: !enabled,
|
||||
icon: <Activity className="size-3" />,
|
||||
id: 'system-resources',
|
||||
label,
|
||||
menuAlign: 'end',
|
||||
menuClassName: 'w-64 p-0',
|
||||
menuContent: (
|
||||
<div
|
||||
className="grid grid-cols-[minmax(0,1fr)] gap-3 p-3 text-[0.75rem]"
|
||||
data-slot="system-resources-panel"
|
||||
>
|
||||
{/* min-w-0 everywhere a flex/grid child must shrink: grid items
|
||||
default min-width:auto, so a long GPU name's nowrap min-content
|
||||
props the track open past the w-64 box and overflow-x:hidden
|
||||
shears off every right-aligned value. With the track clamped,
|
||||
`truncate` can finally act. */}
|
||||
<div className="flex min-w-0 items-baseline justify-between gap-2">
|
||||
<p className="shrink-0 font-medium text-foreground">{copy.title}</p>
|
||||
|
||||
{hardware?.gpu_name && (
|
||||
<span className="min-w-0 truncate text-[0.6875rem] text-muted-foreground">{hardware.gpu_name}</span>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{hasGpu && (
|
||||
<MeterRow
|
||||
label={copy.gpuUtilization}
|
||||
percent={hardware?.gpu_util_percent ?? null}
|
||||
value={`${hardware?.gpu_util_percent ?? 0}%`}
|
||||
/>
|
||||
)}
|
||||
|
||||
{hasGpu && (
|
||||
<MeterRow
|
||||
label={copy.gpuMemory}
|
||||
percent={vramPercent}
|
||||
value={`${gbLong(hardware?.vram_used_bytes)} / ${gbLong(hardware?.vram_total_bytes)}`}
|
||||
/>
|
||||
)}
|
||||
|
||||
<MeterRow
|
||||
label={copy.ram}
|
||||
percent={ramPercent}
|
||||
value={`${gbLong(ramUsed)} / ${gbLong(hardware?.ram_total_bytes)}`}
|
||||
/>
|
||||
|
||||
{hardware?.uma && <p className="text-[0.6875rem] text-muted-foreground">{copy.unifiedNote}</p>}
|
||||
</div>
|
||||
),
|
||||
toggleLabel: enabled ? copy.toggle : undefined,
|
||||
variant: 'menu'
|
||||
}
|
||||
}
|
||||
@@ -11,13 +11,17 @@ import { SCAFFOLD_LABEL_CLASS } from '@/components/chat/scaffold-row'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { Loader } from '@/components/ui/loader'
|
||||
import { StatusPulse } from '@/components/ui/status-pulse'
|
||||
import { getLocalModelsStatus } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { $backgroundResume } from '@/store/background-delegation'
|
||||
import { sessionCompacting } from '@/store/compaction'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { sessionAwaitingInput } from '@/store/prompts'
|
||||
import { sessionProviderWait } from '@/store/provider-wait'
|
||||
import { parseModelLoadWait, sessionProviderWait } from '@/store/provider-wait'
|
||||
import { $currentModel } from '@/store/session'
|
||||
import { type DraftingTool, sessionDraftingTool } from '@/store/tool-drafting'
|
||||
import type { LocalModelLoadProgress } from '@/types/hermes'
|
||||
|
||||
// A status line is scaffolding like any other — "Editing" while the model
|
||||
// drafts a call is the same kind of line as "Explored 3 files" once it has run,
|
||||
@@ -51,6 +55,100 @@ const HintText: FC<{ children: ReactNode }> = ({ children }) => (
|
||||
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shimmer min-w-0 flex-1 truncate')}>{children}</span>
|
||||
)
|
||||
|
||||
/** Renderer-side load synthesis: poll the local-models status while a turn
|
||||
* is busy with NO progress frame from the backend. The backend's wait loop
|
||||
* only narrates the MAIN chat request — a model load triggered while the
|
||||
* gateway is still initializing, or one consumed by a parallel auxiliary
|
||||
* call (title generation autoloads the same model), never gets a frame,
|
||||
* and the load looked like nothing was happening. The status route reads
|
||||
* the same SSE snapshot, so this bar carries the identical percent. */
|
||||
function useLocalModelLoad(active: boolean): LocalModelLoadProgress & { model: string } | null {
|
||||
const model = useStore($currentModel)
|
||||
const [progress, setProgress] = useState<(LocalModelLoadProgress & { model: string }) | null>(null)
|
||||
|
||||
// Behind the --local launch flag: without it, no status polling and no
|
||||
// load bar (the local server can't be the current provider anyway).
|
||||
const enabled = $localModelsEnabled.get()
|
||||
|
||||
useEffect(() => {
|
||||
if (!enabled || !active || !model) {
|
||||
setProgress(null)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
let cancelled = false
|
||||
let timer: number | undefined
|
||||
|
||||
const tick = async () => {
|
||||
try {
|
||||
const status = await getLocalModelsStatus()
|
||||
const entry = status.loading?.[model]
|
||||
|
||||
if (!cancelled) {
|
||||
setProgress(entry ? { ...entry, model } : null)
|
||||
}
|
||||
} catch {
|
||||
if (!cancelled) {
|
||||
setProgress(null)
|
||||
}
|
||||
}
|
||||
|
||||
if (!cancelled) {
|
||||
timer = window.setTimeout(() => void tick(), 1_500)
|
||||
}
|
||||
}
|
||||
|
||||
void tick()
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
|
||||
if (timer !== undefined) {
|
||||
window.clearTimeout(timer)
|
||||
}
|
||||
}
|
||||
}, [enabled, active, model])
|
||||
|
||||
return progress
|
||||
}
|
||||
|
||||
/** Wait hint with a real progress bar for managed-local model loads and
|
||||
* prompt processing. The percents come from llama-server itself (per-tensor
|
||||
* load callback / live prefill counter, via the gateway's wait frames), so a
|
||||
* determinate bar is honest — a 40s cold load or a long prefill reads as
|
||||
* visible progress instead of an alarming stall. */
|
||||
const WaitHint: FC<{ hint: string }> = ({ hint }) => {
|
||||
const { t } = useI18n()
|
||||
const load = parseModelLoadWait(hint)
|
||||
|
||||
if (!load) {
|
||||
return <HintText>{hint}</HintText>
|
||||
}
|
||||
|
||||
const label =
|
||||
load.kind === 'load' ? t.assistant.thread.loadingLocalModel(load.model) : t.assistant.thread.processingPrompt
|
||||
|
||||
return <ProgressHint label={label} percent={load.percent} />
|
||||
}
|
||||
|
||||
const ProgressHint: FC<{ label: string; percent: null | number }> = ({ label, percent }) => (
|
||||
<span className="flex min-w-0 flex-1 items-center gap-2">
|
||||
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shimmer min-w-0 shrink truncate')}>{label}</span>
|
||||
{percent !== null && (
|
||||
<>
|
||||
<span className="h-1 w-24 shrink-0 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<span
|
||||
className="block h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(2, percent)}%` }}
|
||||
/>
|
||||
</span>
|
||||
<span className={cn(SCAFFOLD_LABEL_CLASS, 'shrink-0 tabular-nums')}>{percent}%</span>
|
||||
</>
|
||||
)}
|
||||
</span>
|
||||
)
|
||||
|
||||
/** These indicators render inside whichever transcript mounted them, so every
|
||||
* session-scoped signal comes from that surface's view — a tile must never
|
||||
* show the primary chat's compaction, prompt-wait, or turn timer. */
|
||||
@@ -147,6 +245,10 @@ export const ResponseLoadingIndicator: FC = () => {
|
||||
const { compacting, drafting, providerWait, turnStartedAt } = useThreadSessionStatus()
|
||||
const elapsed = useElapsedSeconds(true, undefined, turnStartedAt)
|
||||
const hint = useStatusHint(compacting, drafting, providerWait)
|
||||
// Renderer-synthesized load bar: covers loads the backend's wait loop
|
||||
// can't narrate (gateway still initializing, or an auxiliary call — not
|
||||
// the main request — triggered the autoload). A real wait frame wins.
|
||||
const localLoad = useLocalModelLoad(!hint)
|
||||
|
||||
return (
|
||||
<StatusRow data-slot="aui_response-loading" label={hint || t.assistant.thread.loadingResponse}>
|
||||
@@ -155,7 +257,11 @@ export const ResponseLoadingIndicator: FC = () => {
|
||||
className="dither inline-block size-3 rounded-[2px] text-midground/80"
|
||||
kind="opacity"
|
||||
/>
|
||||
{hint && <HintText>{hint}</HintText>}
|
||||
{hint ? (
|
||||
<WaitHint hint={hint} />
|
||||
) : localLoad ? (
|
||||
<ProgressHint label={t.assistant.thread.loadingLocalModel(localLoad.model)} percent={localLoad.percent} />
|
||||
) : null}
|
||||
<ActivityTimerText seconds={elapsed} />
|
||||
</StatusRow>
|
||||
)
|
||||
@@ -207,6 +313,7 @@ export const BackgroundResumeNotice: FC = () => {
|
||||
// so that per-token updates re-render only this leaf, not the whole
|
||||
// AssistantMessage subtree.
|
||||
export const TurnActivityIndicator: FC = () => {
|
||||
const { t } = useI18n()
|
||||
const activity = useAuiState(s => activitySignature(s.message.content))
|
||||
|
||||
// Timestamp of the last visible progress, held from the moment the quiet
|
||||
@@ -227,6 +334,10 @@ export const TurnActivityIndicator: FC = () => {
|
||||
// turn of a fresh chat — so the row can't wait for the store to catch up.
|
||||
const messageRunning = useAuiState(s => s.message.status?.type === 'running')
|
||||
|
||||
// Renderer-synthesized load bar (see ResponseLoadingIndicator).
|
||||
const working = busy || messageRunning
|
||||
const localLoad = useLocalModelLoad(working && !hint && !toolNarrating)
|
||||
|
||||
useEffect(() => {
|
||||
setQuietSince(undefined)
|
||||
const seenAt = Date.now()
|
||||
@@ -240,8 +351,10 @@ export const TurnActivityIndicator: FC = () => {
|
||||
// TURN_QUIET_S first, or a run of quick calls would strobe a row between
|
||||
// each one. The two exemptions are waits already accounted for elsewhere: a
|
||||
// question the user is answering, and a tool call carrying its own timer.
|
||||
const working = busy || messageRunning
|
||||
const active = working && !awaitingInput && !toolNarrating && (Boolean(hint) || quietSince !== undefined)
|
||||
// A live local-model load is a named wait too — it must not wait out the
|
||||
// quiet window (the load IS the story from second one).
|
||||
const active =
|
||||
working && !awaitingInput && !toolNarrating && (Boolean(hint) || localLoad !== null || quietSince !== undefined)
|
||||
|
||||
// Compaction owns the whole turn, so it keeps counting from the turn's start;
|
||||
// anything else counts from the moment the turn last produced something — the
|
||||
@@ -263,7 +376,11 @@ export const TurnActivityIndicator: FC = () => {
|
||||
className="dither inline-block size-3 rounded-[2px] text-midground/80"
|
||||
kind="opacity"
|
||||
/>
|
||||
{hint && <HintText>{hint}</HintText>}
|
||||
{hint ? (
|
||||
<WaitHint hint={hint} />
|
||||
) : localLoad ? (
|
||||
<ProgressHint label={t.assistant.thread.loadingLocalModel(localLoad.model)} percent={localLoad.percent} />
|
||||
) : null}
|
||||
<ActivityTimerText seconds={elapsed} />
|
||||
</StatusRow>
|
||||
)
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
|
||||
import { cleanup, render, screen, waitFor } from '@testing-library/react'
|
||||
import type { ReactElement } from 'react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { I18nProvider } from '@/i18n'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import { stubMenuDomApis, stubResizeObserver } from '@/test/jsdom'
|
||||
import type { LocalRuntimeJob, ModelOptionsResponse } from '@/types/hermes'
|
||||
|
||||
import { ModelPickerDialog } from './model-picker'
|
||||
|
||||
vi.mock('@/hermes', () => ({
|
||||
getLocalModelsStatus: vi.fn().mockResolvedValue({ loading: {} })
|
||||
}))
|
||||
vi.mock('@/lib/model-options', async importOriginal => ({
|
||||
...(await importOriginal<Record<string, unknown>>()),
|
||||
requestModelOptions: vi.fn()
|
||||
}))
|
||||
|
||||
import { requestModelOptions } from '@/lib/model-options'
|
||||
|
||||
stubResizeObserver()
|
||||
stubMenuDomApis()
|
||||
|
||||
const OPTIONS: ModelOptionsResponse = {
|
||||
model: 'Qwen3.6-27B-UD-Q4_K_XL',
|
||||
provider: 'llamacpp',
|
||||
providers: [
|
||||
{
|
||||
slug: 'llamacpp',
|
||||
name: 'Local',
|
||||
models: ['Qwen3.6-27B-UD-Q4_K_XL'],
|
||||
is_current: true,
|
||||
authenticated: true
|
||||
},
|
||||
{
|
||||
slug: 'nous',
|
||||
name: 'Nous',
|
||||
models: ['Hermes-4.5'],
|
||||
authenticated: true
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
const DOWNLOAD_JOB: LocalRuntimeJob = {
|
||||
job_id: 'dl1',
|
||||
kind: 'model-download',
|
||||
target: 'Qwen3.8 Flash Next (UD-Q4_K_XL)',
|
||||
model_id: 'qwen3.8-flash-next',
|
||||
status: 'running',
|
||||
phase: 'downloading',
|
||||
detail: '',
|
||||
total_bytes: 100,
|
||||
done_bytes: 41,
|
||||
percent: 41,
|
||||
error: null
|
||||
}
|
||||
|
||||
function renderPicker(ui?: Partial<Parameters<typeof ModelPickerDialog>[0]>) {
|
||||
const client = new QueryClient({ defaultOptions: { queries: { retry: false } } })
|
||||
|
||||
const element: ReactElement = (
|
||||
<QueryClientProvider client={client}>
|
||||
<I18nProvider>
|
||||
<ModelPickerDialog
|
||||
currentModel="Qwen3.6-27B-UD-Q4_K_XL"
|
||||
currentProvider="llamacpp"
|
||||
onOpenChange={() => undefined}
|
||||
onSelect={() => undefined}
|
||||
open
|
||||
{...ui}
|
||||
/>
|
||||
</I18nProvider>
|
||||
</QueryClientProvider>
|
||||
)
|
||||
|
||||
return render(element)
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.mocked(requestModelOptions).mockResolvedValue(OPTIONS)
|
||||
$localRuntimeJobs.set([])
|
||||
// These suites exercise the local-models rows, which ship behind --local.
|
||||
$localModelsEnabled.set(true)
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
vi.clearAllMocks()
|
||||
})
|
||||
|
||||
describe('ModelPickerDialog download rows', () => {
|
||||
it('shows an in-flight download as a disabled progress row in the Local group', async () => {
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderPicker()
|
||||
|
||||
expect(await screen.findByText('Qwen3.6-27B-UD-Q4_K_XL')).toBeTruthy()
|
||||
|
||||
const row = screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
|
||||
|
||||
expect(row).toBeTruthy()
|
||||
expect(screen.getByText('41%')).toBeTruthy()
|
||||
|
||||
// Disabled: cmdk marks the item unselectable.
|
||||
const item = row.closest('[cmdk-item]')
|
||||
|
||||
expect(item?.getAttribute('aria-disabled')).toBe('true')
|
||||
})
|
||||
|
||||
it('shows a first-ever download under its own Local group when no local provider exists yet', async () => {
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
vi.mocked(requestModelOptions).mockResolvedValue({
|
||||
providers: [OPTIONS.providers![1]]
|
||||
})
|
||||
renderPicker()
|
||||
|
||||
expect(await screen.findByText('Hermes-4.5')).toBeTruthy()
|
||||
expect(screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
|
||||
expect(screen.getByText('41%')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('quickstart shows while downloading but not during later phases', async () => {
|
||||
const quickstart: LocalRuntimeJob = { ...DOWNLOAD_JOB, job_id: 'q1', kind: 'quickstart', phase: 'downloading' }
|
||||
|
||||
$localRuntimeJobs.set([quickstart])
|
||||
renderPicker()
|
||||
expect(await screen.findByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
|
||||
|
||||
// The model is staged once quickstart moves on to activating it — the
|
||||
// placeholder row must leave rather than sit beside the real model.
|
||||
$localRuntimeJobs.set([{ ...quickstart, phase: 'starting-server' }])
|
||||
await waitFor(() => {
|
||||
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
it('refetches the model options when a download it saw running completes', async () => {
|
||||
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
||||
renderPicker()
|
||||
await screen.findByText('Qwen3.6-27B-UD-Q4_K_XL')
|
||||
|
||||
expect(vi.mocked(requestModelOptions).mock.calls.length).toBe(1)
|
||||
|
||||
$localRuntimeJobs.set([{ ...DOWNLOAD_JOB, status: 'done', phase: 'done' }])
|
||||
await waitFor(() => {
|
||||
expect(vi.mocked(requestModelOptions).mock.calls.length).toBe(2)
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -1,12 +1,16 @@
|
||||
import { useQuery } from '@tanstack/react-query'
|
||||
import { useState } from 'react'
|
||||
import { useEffect, useMemo, useState } from 'react'
|
||||
|
||||
import { getLocalModelsStatus } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options'
|
||||
import { modelSearchText } from '@/lib/model-search-text'
|
||||
import { currentPickerSelection } from '@/lib/model-status-label'
|
||||
import { normalize } from '@/lib/text'
|
||||
import type { ModelOptionProvider, ModelPricing } from '@/types/hermes'
|
||||
import { useStoreSelector } from '@/lib/use-session-slice'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $localRuntimeJobs, runningModelDownloads, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
|
||||
import type { LocalModelLoadProgress, ModelOptionProvider, ModelPricing } from '@/types/hermes'
|
||||
|
||||
import type { HermesGateway } from '../hermes'
|
||||
import { cn } from '../lib/utils'
|
||||
@@ -67,6 +71,81 @@ export function ModelPickerDialog({
|
||||
enabled: open
|
||||
})
|
||||
|
||||
// Live load state for the managed local server: which model is loading
|
||||
// into memory right now, with a REAL percent (per-tensor callback relayed
|
||||
// over the router's SSE stream). Polled only while the picker is open —
|
||||
// 2s idle cadence is enough for a bar under a ~40s load. Errors read as
|
||||
// "nothing loading" (remote-only installs have no local-models routes).
|
||||
// Every local-models read here sits behind the --local launch flag (strict:
|
||||
// the llamacpp provider group hides even with staged models on disk).
|
||||
const localModelsEnabled = $localModelsEnabled.get()
|
||||
|
||||
const localStatus = useQuery({
|
||||
queryKey: ['local-models-loading', profile],
|
||||
queryFn: () => getLocalModelsStatus(),
|
||||
enabled: open && localModelsEnabled,
|
||||
refetchInterval: 2_000,
|
||||
retry: false
|
||||
})
|
||||
|
||||
const loadingModels: Record<string, LocalModelLoadProgress> = localStatus.data?.loading ?? {}
|
||||
|
||||
// Models on their way into the local library right now (downloads +
|
||||
// quickstart runs), rendered as grayed progress rows. The jobs store
|
||||
// republishes every ~700ms with fresh byte counts while anything runs —
|
||||
// and this dialog stays MOUNTED app-wide when closed — so subscribe only
|
||||
// to download identity (changes when a download starts/ends, and never
|
||||
// while closed); each row selects its own percent scalar (#72163 class).
|
||||
const downloadsKey = useStoreSelector($localRuntimeJobs, jobs =>
|
||||
open && localModelsEnabled
|
||||
? runningModelDownloads(jobs)
|
||||
.map(job => `${job.job_id}\u0000${job.target}`)
|
||||
.join('\u0001')
|
||||
: ''
|
||||
)
|
||||
|
||||
const downloads = useMemo(
|
||||
() =>
|
||||
downloadsKey === ''
|
||||
? []
|
||||
: downloadsKey.split('\u0001').map(pair => {
|
||||
const [jobId, target] = pair.split('\u0000')
|
||||
|
||||
return { jobId, target }
|
||||
}),
|
||||
[downloadsKey]
|
||||
)
|
||||
|
||||
// Rediscover in-flight work on open: the poller idles when nothing was
|
||||
// running, and a download can start from any surface.
|
||||
useEffect(() => {
|
||||
if (open && localModelsEnabled) {
|
||||
watchLocalRuntimeJobs()
|
||||
}
|
||||
}, [open, localModelsEnabled])
|
||||
|
||||
// A finished download turns into a real selectable model — refetch the
|
||||
// options so the placeholder row is replaced while the picker is open.
|
||||
const refetchOptions = modelOptions.refetch
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) {
|
||||
return
|
||||
}
|
||||
|
||||
let prevActive = runningModelDownloads($localRuntimeJobs.get()).length > 0
|
||||
|
||||
return $localRuntimeJobs.listen(next => {
|
||||
const active = runningModelDownloads(next).length > 0
|
||||
|
||||
if (prevActive && !active) {
|
||||
void refetchOptions()
|
||||
}
|
||||
|
||||
prevActive = active
|
||||
})
|
||||
}, [open, refetchOptions])
|
||||
|
||||
const providers = modelOptions.data?.providers ?? []
|
||||
|
||||
const { model: optionsModel, provider: optionsProvider } = currentPickerSelection(
|
||||
@@ -117,8 +196,10 @@ export function ModelPickerDialog({
|
||||
<ModelResults
|
||||
currentModel={optionsModel || currentModel}
|
||||
currentProvider={optionsProvider || currentProvider}
|
||||
downloads={downloads}
|
||||
error={error}
|
||||
loading={loading}
|
||||
loadingModels={loadingModels}
|
||||
onSelectModel={selectModel}
|
||||
providers={providers}
|
||||
search={search}
|
||||
@@ -145,6 +226,8 @@ function ModelResults({
|
||||
providers,
|
||||
currentModel,
|
||||
currentProvider,
|
||||
downloads,
|
||||
loadingModels,
|
||||
onSelectModel,
|
||||
search
|
||||
}: {
|
||||
@@ -153,6 +236,8 @@ function ModelResults({
|
||||
providers: ModelOptionProvider[]
|
||||
currentModel: string
|
||||
currentProvider: string
|
||||
downloads: { jobId: string; target: string }[]
|
||||
loadingModels: Record<string, LocalModelLoadProgress>
|
||||
onSelectModel: (provider: ModelOptionProvider, model: string) => void
|
||||
search: string
|
||||
}) {
|
||||
@@ -188,15 +273,29 @@ function ModelResults({
|
||||
// Only configured providers (those with curated models) are selectable
|
||||
// here. Switching to a NOT-yet-configured provider goes through the
|
||||
// "Add provider" footer button, which opens the full onboarding selector.
|
||||
const configured = providers.filter(p => (p.models ?? []).length > 0)
|
||||
// The local provider sits behind the --local launch flag (strict: staged
|
||||
// models on disk don't show without it). Module-level read — a launch flag
|
||||
// can't change mid-session.
|
||||
const localModelsShown = $localModelsEnabled.get()
|
||||
|
||||
const configured = providers.filter(
|
||||
p => (p.models ?? []).length > 0 && (localModelsShown || p.slug !== LOCAL_PROVIDER_SLUG)
|
||||
)
|
||||
|
||||
// In-flight local downloads render as disabled progress rows: inside the
|
||||
// Local group when it exists, else as their own group (first download —
|
||||
// nothing staged yet, so the backend reports no Local provider at all).
|
||||
const visibleDownloads = downloads.filter(job => !q || (job.target || '').toLowerCase().includes(q))
|
||||
const hasLocalGroup = configured.some(p => p.slug === LOCAL_PROVIDER_SLUG)
|
||||
|
||||
return (
|
||||
<>
|
||||
{configured.map(provider => {
|
||||
// Preserve the backend's curated order — filter in place, no re-sort.
|
||||
const models = (provider.models ?? []).filter(m => matches(provider, m))
|
||||
const groupDownloads = provider.slug === LOCAL_PROVIDER_SLUG ? visibleDownloads : []
|
||||
|
||||
if (models.length === 0) {
|
||||
if (models.length === 0 && groupDownloads.length === 0) {
|
||||
return null
|
||||
}
|
||||
|
||||
@@ -215,6 +314,10 @@ function ModelResults({
|
||||
const isCurrent = model === currentModel && provider.slug === currentProvider
|
||||
const price = provider.pricing?.[model]
|
||||
const locked = unavailable.has(model)
|
||||
// Managed local model loading into memory right now: show the
|
||||
// real load percent inline (keyed by exact model id — remote
|
||||
// providers never match).
|
||||
const loadProgress = loadingModels[model]
|
||||
|
||||
return (
|
||||
<CommandItem
|
||||
@@ -236,6 +339,19 @@ function ModelResults({
|
||||
<span className="min-w-0 flex-1 truncate">
|
||||
<HighlightMatches query={search} text={model} />
|
||||
</span>
|
||||
{loadProgress && (
|
||||
<span className="flex shrink-0 items-center gap-1.5" title={copy.loadingIntoMemory}>
|
||||
<span className="h-1 w-16 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<span
|
||||
className="block h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(2, loadProgress.percent)}%` }}
|
||||
/>
|
||||
</span>
|
||||
<span className="text-[0.62rem] tabular-nums text-muted-foreground">
|
||||
{loadProgress.percent}%
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
{locked && (
|
||||
<span className="shrink-0 text-[0.62rem] uppercase tracking-wide opacity-80">{copy.pro}</span>
|
||||
)}
|
||||
@@ -243,6 +359,9 @@ function ModelResults({
|
||||
</CommandItem>
|
||||
)
|
||||
})}
|
||||
{groupDownloads.map(job => (
|
||||
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
|
||||
))}
|
||||
{unavailable.size > 0 && (
|
||||
<div className="px-6 pb-2 pt-1 text-[0.62rem] leading-relaxed text-muted-foreground">
|
||||
{copy.proNeedsSubscription}
|
||||
@@ -251,10 +370,56 @@ function ModelResults({
|
||||
</CommandGroup>
|
||||
)
|
||||
})}
|
||||
{!hasLocalGroup && visibleDownloads.length > 0 && (
|
||||
<CommandGroup heading={copy.localDownloadsHeading} key="local-downloads">
|
||||
{visibleDownloads.map(job => (
|
||||
<DownloadingModelRow jobId={job.jobId} key={job.jobId} target={job.target} />
|
||||
))}
|
||||
</CommandGroup>
|
||||
)}
|
||||
</>
|
||||
)
|
||||
}
|
||||
|
||||
// The backend's provider row for staged local models (inventory.py's
|
||||
// _local_runtime_row). Downloads-in-flight attach to this group.
|
||||
const LOCAL_PROVIDER_SLUG = 'llamacpp'
|
||||
|
||||
// A model still downloading: visible so the user knows it's coming (and
|
||||
// where it will land), disabled so it can't be selected early, with the
|
||||
// same byte progress the settings pane shows. Percent is selected here, per
|
||||
// row, so the poller's 700ms byte ticks repaint this leaf only.
|
||||
function DownloadingModelRow({ jobId, target }: { jobId: string; target: string }) {
|
||||
const { t } = useI18n()
|
||||
const copy = t.modelPicker
|
||||
|
||||
const percent = useStoreSelector(
|
||||
$localRuntimeJobs,
|
||||
jobs => jobs.find(job => job.job_id === jobId)?.percent ?? null
|
||||
)
|
||||
|
||||
return (
|
||||
<CommandItem
|
||||
className="flex items-center gap-2 pl-6 font-mono opacity-60"
|
||||
disabled
|
||||
value={`downloading:${jobId}`}
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate">{target}</span>
|
||||
<span className="flex shrink-0 items-center gap-1.5" title={copy.downloading}>
|
||||
<span className="h-1 w-16 overflow-hidden rounded-full bg-(--ui-bg-tertiary)">
|
||||
<span
|
||||
className="block h-full rounded-full bg-primary transition-[width] duration-500"
|
||||
style={{ width: `${Math.max(2, percent ?? 0)}%` }}
|
||||
/>
|
||||
</span>
|
||||
<span className="text-[0.62rem] tabular-nums text-muted-foreground">
|
||||
{typeof percent === 'number' ? `${percent}%` : copy.downloading}
|
||||
</span>
|
||||
</span>
|
||||
</CommandItem>
|
||||
)
|
||||
}
|
||||
|
||||
// Compact In/Out $/Mtok price tag, mirroring the CLI picker's price columns.
|
||||
// Renders nothing when pricing is unavailable for the model.
|
||||
function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boolean }) {
|
||||
|
||||
@@ -11,6 +11,7 @@ import { Check, ChevronDown, ChevronLeft, KeyRound, Loader2 } from '@/lib/icons'
|
||||
import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { $desktopBoot, type DesktopBootState } from '@/store/boot'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import {
|
||||
$desktopOnboarding,
|
||||
clearPendingProviderOAuth,
|
||||
@@ -32,6 +33,7 @@ import { DocsLink, FlowPanel, Status } from './flow'
|
||||
import {
|
||||
FeaturedProviderRow,
|
||||
FireworksProviderRow,
|
||||
LocalModelsProviderRow,
|
||||
OpenRouterProviderRow,
|
||||
ProviderRow,
|
||||
sortProviders
|
||||
@@ -41,6 +43,7 @@ export {
|
||||
FeaturedProviderRow,
|
||||
FireworksProviderRow,
|
||||
KeyProviderRow,
|
||||
LocalModelsProviderRow,
|
||||
OpenRouterProviderRow,
|
||||
ProviderRow,
|
||||
providerTitle,
|
||||
@@ -478,10 +481,29 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
|
||||
const collapsible = Boolean(featured)
|
||||
const showRest = !collapsible || showAll
|
||||
|
||||
// "Run models locally" leaves the picker for Settings -> Providers ->
|
||||
// Local Models, where install/download live. First-run: persist the skip
|
||||
// (same contract as ChooseLaterLink) so the blocking overlay never
|
||||
// re-nags; manual mode just closes. window.location keeps this picker
|
||||
// router-independent (it renders outside the route tree on first run).
|
||||
const openLocalModels = () => {
|
||||
if (manual) {
|
||||
closeManualOnboarding()
|
||||
} else {
|
||||
dismissFirstRunOnboarding()
|
||||
}
|
||||
|
||||
window.location.hash = '#/settings?tab=providers&pview=local'
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="grid gap-2">
|
||||
<div className="grid max-h-[60dvh] gap-2 overflow-y-auto p-1">
|
||||
{featured ? <FeaturedProviderRow onSelect={select} provider={featured} /> : null}
|
||||
{/* The no-account path: everything runs on this machine. Shipped
|
||||
behind the --local launch flag. (Fireworks moved into the
|
||||
expanded list on main.) */}
|
||||
{$localModelsEnabled.get() ? <LocalModelsProviderRow onClick={openLocalModels} /> : null}
|
||||
{showRest ? (
|
||||
<>
|
||||
{/* Fireworks leads the expanded list, matching CANONICAL_PROVIDERS
|
||||
|
||||
@@ -95,6 +95,14 @@ export function FireworksProviderRow({ onClick }: { onClick: () => void }) {
|
||||
return <KeyProviderRow onClick={onClick} pitch={t.onboarding.fireworksPitch} title="Fireworks AI" />
|
||||
}
|
||||
|
||||
/** Onboarding row for the managed local runtime: no account, no key — the
|
||||
* destination is the Local Models pane where install/download live. */
|
||||
export function LocalModelsProviderRow({ onClick }: { onClick: () => void }) {
|
||||
const { t } = useI18n()
|
||||
|
||||
return <KeyProviderRow onClick={onClick} pitch={t.onboarding.localModelsPitch} title={t.onboarding.localModelsTitle} />
|
||||
}
|
||||
|
||||
export function OpenRouterProviderRow({ onClick }: { onClick: () => void }) {
|
||||
const { t } = useI18n()
|
||||
|
||||
|
||||
@@ -98,6 +98,7 @@ export function TipHost() {
|
||||
|
||||
return (
|
||||
<TipBubble
|
||||
action={tip.action}
|
||||
anchor={anchor}
|
||||
keybind={tip.keybind}
|
||||
onClose={retireActiveTip}
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
/**
|
||||
* The campaign offer against the live stores: eligibility fetch, the showTip
|
||||
* wiring (button included), retirement, the reshow clock, and the cursor
|
||||
* guard that keeps a campaign showing from restarting the rotation's walk.
|
||||
*/
|
||||
|
||||
import { cleanup } from '@testing-library/react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
const getLocalModelsStatus = vi.fn()
|
||||
const getLocalCatalog = vi.fn()
|
||||
|
||||
vi.mock('@/hermes', () => ({
|
||||
getLocalCatalog: (...args: unknown[]) => getLocalCatalog(...args),
|
||||
getLocalModelsStatus: (...args: unknown[]) => getLocalModelsStatus(...args)
|
||||
}))
|
||||
|
||||
import { en } from '@/i18n/en'
|
||||
import { LOCAL_SETUP_TIP_ID } from '@/lib/tips/local-cta'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $connection } from '@/store/session'
|
||||
import { $activeTip, $lastTipId, $retiredTips, $tipShownAt } from '@/store/tips'
|
||||
|
||||
import { offerLocalSetupTip, resetLocalSetupOfferCache } from './local-setup-offer'
|
||||
|
||||
function primeEligibleBackend() {
|
||||
getLocalModelsStatus.mockResolvedValue({ models: [], runtime_installed: false })
|
||||
getLocalCatalog.mockResolvedValue({ models: [{ fits: true, id: 'qwen3.8-27b' }] })
|
||||
}
|
||||
|
||||
async function flushFetch() {
|
||||
await Promise.resolve()
|
||||
await Promise.resolve()
|
||||
await Promise.resolve()
|
||||
}
|
||||
|
||||
describe('offerLocalSetupTip', () => {
|
||||
beforeEach(() => {
|
||||
resetLocalSetupOfferCache()
|
||||
// The campaign ships behind --local like every local-models surface.
|
||||
$localModelsEnabled.set(true)
|
||||
$activeTip.set(null)
|
||||
$retiredTips.set([])
|
||||
$tipShownAt.set({})
|
||||
$lastTipId.set(null)
|
||||
$connection.set({ mode: 'local' } as never)
|
||||
getLocalModelsStatus.mockReset()
|
||||
getLocalCatalog.mockReset()
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
})
|
||||
|
||||
it('holds the first quiet moment while the read flies, then shows on the next', async () => {
|
||||
primeEligibleBackend()
|
||||
|
||||
const openLocalModels = vi.fn()
|
||||
|
||||
// First offer: fetch in flight — the moment is HELD (true, so the
|
||||
// rotation's walk cannot take it and arm the cooldown ahead of the
|
||||
// campaign), but nothing is on screen yet.
|
||||
expect(offerLocalSetupTip(en.tips, openLocalModels)).toBe(true)
|
||||
expect($activeTip.get()).toBeNull()
|
||||
await flushFetch()
|
||||
|
||||
// Second offer: cached yes — bubble goes up with the CTA wired.
|
||||
expect(offerLocalSetupTip(en.tips, openLocalModels)).toBe(true)
|
||||
|
||||
const tip = $activeTip.get()
|
||||
|
||||
expect(tip?.tipId).toBe(LOCAL_SETUP_TIP_ID)
|
||||
expect(tip?.action?.label).toBe(en.tips.items['local-setup'].action)
|
||||
|
||||
tip?.action?.onSelect()
|
||||
expect(openLocalModels).toHaveBeenCalledTimes(1)
|
||||
// The CTA closes the bubble on its way to the pane.
|
||||
expect($activeTip.get()).toBeNull()
|
||||
})
|
||||
|
||||
it('never restarts the rotation walk: the campaign id stays out of the cursor', async () => {
|
||||
primeEligibleBackend()
|
||||
$lastTipId.set('cron')
|
||||
|
||||
offerLocalSetupTip(en.tips, vi.fn())
|
||||
await flushFetch()
|
||||
offerLocalSetupTip(en.tips, vi.fn())
|
||||
|
||||
expect($activeTip.get()?.tipId).toBe(LOCAL_SETUP_TIP_ID)
|
||||
expect($lastTipId.get()).toBe('cron')
|
||||
})
|
||||
|
||||
it('stays quiet on an ineligible machine without refetching', async () => {
|
||||
getLocalModelsStatus.mockResolvedValue({ models: [{ id: 'staged' }], runtime_installed: true })
|
||||
getLocalCatalog.mockResolvedValue({ models: [{ fits: true, id: 'qwen3.8-27b' }] })
|
||||
|
||||
offerLocalSetupTip(en.tips, vi.fn())
|
||||
await flushFetch()
|
||||
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
expect($activeTip.get()).toBeNull()
|
||||
expect(getLocalModelsStatus).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it('honors the ✕ forever and the ignored-bubble clock for a week', async () => {
|
||||
primeEligibleBackend()
|
||||
|
||||
$retiredTips.set([LOCAL_SETUP_TIP_ID])
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
expect(getLocalModelsStatus).not.toHaveBeenCalled()
|
||||
|
||||
$retiredTips.set([])
|
||||
$tipShownAt.set({ [LOCAL_SETUP_TIP_ID]: Date.now() - 60_000 })
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
expect(getLocalModelsStatus).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('asks nothing of a remote backend', () => {
|
||||
$connection.set({ mode: 'remote' } as never)
|
||||
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
expect(getLocalModelsStatus).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('never runs without the --local launch flag (strict), even on an eligible machine', () => {
|
||||
$localModelsEnabled.set(false)
|
||||
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
// Declined before any read: no fetch, no held moment, no cooldown spent.
|
||||
expect(getLocalModelsStatus).not.toHaveBeenCalled()
|
||||
expect($activeTip.get()).toBeNull()
|
||||
})
|
||||
|
||||
it('a failed read stands down for the session instead of retrying', async () => {
|
||||
getLocalModelsStatus.mockRejectedValue(new Error('backend gone'))
|
||||
getLocalCatalog.mockRejectedValue(new Error('backend gone'))
|
||||
|
||||
offerLocalSetupTip(en.tips, vi.fn())
|
||||
await flushFetch()
|
||||
|
||||
expect(offerLocalSetupTip(en.tips, vi.fn())).toBe(false)
|
||||
expect(getLocalModelsStatus).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,126 @@
|
||||
/**
|
||||
* The local-setup campaign: one bubble on the model pill for machines that
|
||||
* could run local models and haven't set them up.
|
||||
*
|
||||
* Not a rotation tip — a campaign the rotation CONSULTS first at each quiet
|
||||
* due moment (use-tip-rotation.ts): conditional (most machines qualify or
|
||||
* don't, permanently), actionable (it carries the one button a tip may
|
||||
* have), and perishable (setting up local models — or the ✕ — ends it).
|
||||
* A live "your GPU can run this, free and private" outranks the walk's
|
||||
* "the model name is a button" whenever both are true, and an ignored
|
||||
* bubble may return in a week rather than walking on forever.
|
||||
*
|
||||
* Eligibility is fetched, not assumed: the backend's own fit check (the
|
||||
* same catalog `fits` the Local Models pane prices its hero with) decides
|
||||
* whether this machine qualifies. Reads are lazy — nothing polls for a
|
||||
* bubble. The first quiet due moment kicks one status+catalog read and
|
||||
* holds the turn (no walk tip may spend the cooldown ahead of a pending
|
||||
* campaign); the cached answer serves every later one. Completing
|
||||
* setup flips the next read to ineligible, so the campaign retires itself
|
||||
* without bookkeeping — and the cache dies with a connection change,
|
||||
* because eligibility is a fact about the backend's machine.
|
||||
*/
|
||||
|
||||
import { getLocalCatalog, getLocalModelsStatus } from '@/hermes'
|
||||
import type { Translations } from '@/i18n/types'
|
||||
import { LOCAL_SETUP_TIP_ID, localSetupDue, localSetupEligible } from '@/lib/tips/local-cta'
|
||||
import { $localModelsEnabled } from '@/store/local-models-flag'
|
||||
import { $connection } from '@/store/session'
|
||||
import { $retiredTips, $tipShownAt, dismissTip, showTip } from '@/store/tips'
|
||||
|
||||
/** The pill the bubble points at — the same handle the rotation's
|
||||
* model-switch tip uses, so the two can never drift to different anchors. */
|
||||
const MODEL_PILL_TARGETS = ['[data-tour="model-pill"]'] as const
|
||||
|
||||
let eligibilityCache: { eligible: boolean } | null = null
|
||||
let eligibilityInFlight = false
|
||||
let boundToConnection = false
|
||||
|
||||
/** Reset the session cache — tests only. */
|
||||
export function resetLocalSetupOfferCache(): void {
|
||||
eligibilityCache = null
|
||||
eligibilityInFlight = false
|
||||
}
|
||||
|
||||
/**
|
||||
* Offer the campaign the current quiet moment. True = it put its bubble up
|
||||
* and the moment is spent; false = the rotation's walk may have it.
|
||||
*/
|
||||
export function offerLocalSetupTip(copy: Translations['tips'], openLocalModels: () => void): boolean {
|
||||
// Local models ship behind the --local launch flag; without it there is
|
||||
// no Local Models pane for the button to open, so the campaign never runs
|
||||
// (and never spends a status/catalog read).
|
||||
if (!$localModelsEnabled.get()) {
|
||||
return false
|
||||
}
|
||||
|
||||
if ($retiredTips.get().includes(LOCAL_SETUP_TIP_ID)) {
|
||||
return false
|
||||
}
|
||||
|
||||
if (!localSetupDue(Date.now(), $tipShownAt.get()[LOCAL_SETUP_TIP_ID])) {
|
||||
return false
|
||||
}
|
||||
|
||||
// Local backends only: on a remote connection (cloud resolves to remote)
|
||||
// the models would run on the far machine, and "stays on your computer"
|
||||
// would be promising someone else's computer. Checked before the cache so
|
||||
// a re-home mid-session can't serve a stale yes.
|
||||
if (($connection.get()?.mode ?? null) !== 'local') {
|
||||
return false
|
||||
}
|
||||
|
||||
if (!boundToConnection) {
|
||||
boundToConnection = true
|
||||
$connection.listen(() => resetLocalSetupOfferCache())
|
||||
}
|
||||
|
||||
if (!eligibilityCache) {
|
||||
if (!eligibilityInFlight) {
|
||||
eligibilityInFlight = true
|
||||
|
||||
void Promise.all([getLocalModelsStatus(), getLocalCatalog()])
|
||||
.then(([status, catalog]) => {
|
||||
eligibilityCache = {
|
||||
eligible: localSetupEligible($connection.get()?.mode ?? null, status, catalog.models)
|
||||
}
|
||||
})
|
||||
.catch(() => {
|
||||
// No backend answer, no campaign this session. The next launch —
|
||||
// or the next connection — asks again.
|
||||
eligibilityCache = { eligible: false }
|
||||
})
|
||||
.finally(() => {
|
||||
eligibilityInFlight = false
|
||||
})
|
||||
}
|
||||
|
||||
// Hold the moment while the read flies: nothing shows and no cooldown
|
||||
// arms, so the next tick answers from the cache. Handing this moment to
|
||||
// the rotation instead would put a walk tip up first and park the
|
||||
// campaign behind the six-hour cooldown — the exact inversion of the
|
||||
// priority. Costs an ineligible machine one 30s tick, once per session.
|
||||
return true
|
||||
}
|
||||
|
||||
if (!eligibilityCache.eligible) {
|
||||
return false
|
||||
}
|
||||
|
||||
showTip({
|
||||
action: {
|
||||
label: copy.items['local-setup'].action,
|
||||
onSelect: () => {
|
||||
dismissTip()
|
||||
openLocalModels()
|
||||
}
|
||||
},
|
||||
side: 'top',
|
||||
targets: MODEL_PILL_TARGETS,
|
||||
text: copy.items['local-setup'].text,
|
||||
tipId: LOCAL_SETUP_TIP_ID,
|
||||
title: copy.items['local-setup'].title
|
||||
})
|
||||
|
||||
return true
|
||||
}
|
||||
@@ -21,8 +21,11 @@ import { useI18n } from '@/i18n'
|
||||
import { iconSize, X } from '@/lib/icons'
|
||||
import { useKeybindHint } from '@/lib/keybinds/use-keybind-hint'
|
||||
import type { TipSide } from '@/lib/tips/catalog'
|
||||
import type { ActiveTip } from '@/store/tips'
|
||||
|
||||
export interface TipBubbleProps {
|
||||
/** A call to action rendered as the bubble's one button. See ActiveTip. */
|
||||
action?: ActiveTip['action']
|
||||
/** The element the arrow points at. */
|
||||
anchor: HTMLElement
|
||||
/** Keybind action id; its live combo prints under the text. */
|
||||
@@ -34,7 +37,7 @@ export interface TipBubbleProps {
|
||||
title?: string
|
||||
}
|
||||
|
||||
export function TipBubble({ anchor, keybind, onClose, side, text, title }: TipBubbleProps) {
|
||||
export function TipBubble({ action, anchor, keybind, onClose, side, text, title }: TipBubbleProps) {
|
||||
const { t } = useI18n()
|
||||
const combo = useKeybindHint(keybind ?? '')
|
||||
const anchorRef = useRef<HTMLElement | null>(anchor)
|
||||
@@ -81,6 +84,19 @@ export function TipBubble({ anchor, keybind, onClose, side, text, title }: TipBu
|
||||
{text}
|
||||
</p>
|
||||
{combo && <KbdCombo className="mt-2" combo={combo} size="sm" variant="inverted" />}
|
||||
{action && (
|
||||
// The CTA: still not a focus trap — the button is tabbable when
|
||||
// reached but nothing steals the caret to get there. Inverted
|
||||
// fill against the accent surface, same currentColor discipline
|
||||
// as the rest of the bubble.
|
||||
<button
|
||||
className="mt-2.5 inline-flex cursor-pointer items-center rounded-md bg-current/15 px-2.5 py-1 text-[length:var(--conversation-caption-font-size)] font-semibold transition-colors hover:bg-current/25"
|
||||
onClick={action.onSelect}
|
||||
type="button"
|
||||
>
|
||||
{action.label}
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
<button
|
||||
aria-label={t.tips.close}
|
||||
|
||||
@@ -20,7 +20,9 @@
|
||||
*/
|
||||
|
||||
import { useEffect } from 'react'
|
||||
import { useNavigate } from 'react-router'
|
||||
|
||||
import { SETTINGS_ROUTE } from '@/app/routes'
|
||||
import type { Translations } from '@/i18n/types'
|
||||
import { resolveTipAnchor } from '@/lib/tips/anchor'
|
||||
import { TIP_CATALOG } from '@/lib/tips/catalog'
|
||||
@@ -28,6 +30,8 @@ import { nextTip } from '@/lib/tips/rotation'
|
||||
import { $awaitingResponse, $busy } from '@/store/session'
|
||||
import { $activeTip, $lastTipId, $nextTipAt, $retiredTips, $tipsEnabled, showTip } from '@/store/tips'
|
||||
|
||||
import { offerLocalSetupTip } from './local-setup-offer'
|
||||
|
||||
const TICK_MS = 30_000
|
||||
/** Nothing in the first stretch of a launch, however long the cooldown says
|
||||
* it's been: you opened the app to do a thing, and the tip can wait until
|
||||
@@ -60,6 +64,8 @@ function appIsQuiet(lastTypedAt: number): boolean {
|
||||
|
||||
/** Drive the ambient rotation for as long as the host is mounted. */
|
||||
export function useTipRotation(copy: Translations['tips']) {
|
||||
const navigate = useNavigate()
|
||||
|
||||
useEffect(() => {
|
||||
let lastTypedAt = 0
|
||||
let settledAt = Date.now() + SETTLE_MIN_MS + Math.random() * SETTLE_SPREAD_MS
|
||||
@@ -83,6 +89,18 @@ export function useTipRotation(copy: Translations['tips']) {
|
||||
return
|
||||
}
|
||||
|
||||
// Campaigns outrank the walk: a conditional, actionable tip that is
|
||||
// live right now (the local-setup CTA) says something about THIS
|
||||
// machine, which beats the catalog's standing introduction. It shares
|
||||
// the cooldown, so taking the moment still costs it the usual hours.
|
||||
if (
|
||||
offerLocalSetupTip(copy, () => {
|
||||
navigate(`${SETTINGS_ROUTE}?tab=providers&pview=local`)
|
||||
})
|
||||
) {
|
||||
return
|
||||
}
|
||||
|
||||
// Only tips with something on screen to point at are candidates, so the
|
||||
// rotation never burns a turn on a pane the user isn't showing.
|
||||
const onScreen = TIP_CATALOG.filter(tip => resolveTipAnchor(document, tip.targets))
|
||||
@@ -128,5 +146,5 @@ export function useTipRotation(copy: Translations['tips']) {
|
||||
window.clearInterval(timer)
|
||||
window.removeEventListener('keydown', noteTyping, true)
|
||||
}
|
||||
}, [copy])
|
||||
}, [copy, navigate])
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ const badgeVariants = cva(
|
||||
variant: {
|
||||
default: 'bg-primary/10 text-primary',
|
||||
muted: 'bg-muted text-muted-foreground',
|
||||
success: 'bg-emerald-500/10 text-emerald-600 dark:text-emerald-300',
|
||||
warn: 'bg-amber-500/10 text-amber-600 dark:text-amber-300',
|
||||
destructive: 'bg-destructive/10 text-destructive',
|
||||
outline: 'border border-(--ui-stroke-secondary) text-muted-foreground',
|
||||
|
||||
Vendored
+3
@@ -301,6 +301,9 @@ declare global {
|
||||
glassSupported?: boolean
|
||||
/** Main-process fact: this OS can do any translucency at all (not Linux). */
|
||||
translucencySupported?: boolean
|
||||
/** Launch flag: the app was started with --local, enabling the
|
||||
* local-models GUI surfaces. Absent/false = every local surface hides. */
|
||||
localModelsEnabled?: boolean
|
||||
setTranslucency?: (payload: TranslucencyState) => void
|
||||
setKeepAwake?: (on: boolean) => void
|
||||
setDisableF12?: (blocked: boolean) => void
|
||||
|
||||
@@ -18,6 +18,7 @@ export {
|
||||
export type { ProfileScope } from './api/client'
|
||||
export * from './api/config'
|
||||
export * from './api/cron'
|
||||
export * from './api/local-models'
|
||||
export * from './api/mcp'
|
||||
export * from './api/messaging'
|
||||
export * from './api/models'
|
||||
|
||||
@@ -2890,6 +2890,11 @@ export const ar = defineLocale({
|
||||
title: 'بدّل النموذج أثناء المحادثة',
|
||||
text: 'اسم النموذج زر. غيّره كلما تغيّرت طبيعة العمل.'
|
||||
},
|
||||
'local-setup': {
|
||||
title: 'هذا الجهاز يمكنه تشغيل النماذج محليًا',
|
||||
text: 'عتادك قادر على تشغيل نموذج محلي. تبقى محادثاتك على جهازك ولا تكلف شيئًا.',
|
||||
action: 'إعداد الآن'
|
||||
},
|
||||
'right-pane': {
|
||||
title: 'لوحة العمل',
|
||||
text: 'الملفات والطرفية والمراجعة والمتصفح المدمج تتشارك اللوحة الجانبية.'
|
||||
|
||||
@@ -395,6 +395,7 @@ export const en: Translations = {
|
||||
providerAccounts: 'Accounts',
|
||||
providerApiKeys: 'API keys',
|
||||
providerCustomEndpoints: 'Custom Endpoints',
|
||||
providerLocalModels: 'Local Models',
|
||||
gateway: 'Gateways',
|
||||
apiKeys: 'Tools & Keys',
|
||||
keybinds: 'Keyboard Shortcuts',
|
||||
@@ -1117,6 +1118,121 @@ export const en: Translations = {
|
||||
curator: { label: 'Curator', hint: 'Skill-usage review' }
|
||||
}
|
||||
},
|
||||
localModels: {
|
||||
title: 'Local Models',
|
||||
runtimeTitle: 'Local runtime',
|
||||
runtimeReady: backend => `Ready · ${backend}`,
|
||||
serverRunning: 'Running',
|
||||
runtimeInstalled: 'llama.cpp runtime installed',
|
||||
runtimeInstalledDetail: (tag, backend) =>
|
||||
`Build ${tag}, ${backend} backend. Hermes starts and manages the server for you.`,
|
||||
installTitle: 'Install the local runtime',
|
||||
installDetail:
|
||||
'Downloads the llama.cpp inference engine (a few hundred MB). Models you download run entirely on this machine — no account, nothing leaves your computer.',
|
||||
installAction: 'Install runtime',
|
||||
installing: 'Installing runtime…',
|
||||
installFailed: 'Runtime install failed',
|
||||
hardwareTitle: 'This machine',
|
||||
hardwareLoading: 'Checking your hardware…',
|
||||
vram: label => `${label} GPU memory`,
|
||||
ram: label => `${label} RAM`,
|
||||
unifiedMemory: 'Unified memory',
|
||||
modelsTitle: 'Models',
|
||||
recommended: 'Recommended',
|
||||
/* The Recommended badge's tooltip, keyed by the resolver branch that
|
||||
made the pick. Qualitative on purpose: predictions order candidates,
|
||||
they are not promises to print. */
|
||||
recommendedReason: {
|
||||
'best-quality-resident':
|
||||
'The highest-quality model that runs entirely on your GPU at full speed. Picks weigh quality against predicted speed on this hardware.',
|
||||
'speed-gated-quality':
|
||||
'A higher-quality model fits this machine but would respond too slowly on its memory bandwidth — this is the best model that stays fast.',
|
||||
'fastest-resident':
|
||||
'No model reaches full speed on this hardware; this one comes closest while running entirely in GPU memory.',
|
||||
'least-painful-spilled':
|
||||
'No model fits entirely in GPU memory here — this one runs best from system RAM.'
|
||||
} as Record<string, string>,
|
||||
downloaded: 'Downloaded',
|
||||
downloadAction: size => `Download · ${size}`,
|
||||
downloadProgress: (done, total) => `Downloading ${done} of ${total}`,
|
||||
downloadDoneToast: model => `${model} is ready.`,
|
||||
installDoneToast: 'Local runtime installed and ready.',
|
||||
quickstartTitle: 'Run a model on this machine',
|
||||
quickstartDetail: (model, size) =>
|
||||
`One click sets everything up: the local engine, ${model} (${size} download), and your default for new chats. Nothing leaves this computer.`,
|
||||
quickstartDetailReady: model => `One click makes ${model} your default for new chats. Everything runs on this machine.`,
|
||||
quickstartAction: 'Set up for me',
|
||||
quickstartConfigure: 'Configure…',
|
||||
quickstartDoneToast: model => `${model} is set up — new chats run on this machine.`,
|
||||
quickstartFailed: 'Local model setup failed',
|
||||
quickstartStageEngine: 'Engine',
|
||||
quickstartStageModel: 'Model',
|
||||
quickstartStageFinish: 'Finish',
|
||||
useAction: 'Use',
|
||||
activePill: 'Default',
|
||||
updateTitle: 'Engine update available',
|
||||
updateDetail: (next, current) => `A newer llama.cpp build (${next}) is ready to install — you're on ${current}. Models keep working during the download.`,
|
||||
updateAction: 'Update engine',
|
||||
updating: 'Updating engine…',
|
||||
upToDateTitle: 'Engine up to date',
|
||||
upToDateDetail: (tag, backend) => `Running llama.cpp ${tag} (${backend}) — the latest build Hermes ships.`,
|
||||
updateToast: next => `A newer local engine build (${next}) is available. Update from Settings → Local Models.`,
|
||||
activeDetail: 'New chats use this model — it loads when you send your first message',
|
||||
activeNotLoaded: 'Loads on your first message',
|
||||
loadedPill: 'In memory',
|
||||
placementResident: 'all on GPU',
|
||||
placementSpilled: 'partly in RAM',
|
||||
placementResidentTip: 'Running entirely in GPU memory at this context window — full speed.',
|
||||
placementSpilledTip:
|
||||
'Part of this model runs from system RAM — it works, but slower. A more compact build or a smaller context would fit fully.',
|
||||
loadingPill: 'Loading…',
|
||||
ejectTip: 'Free GPU memory (loads again on the next message)',
|
||||
ejected: 'Model unloaded — GPU memory freed.',
|
||||
ejectFailed: 'Could not unload the model',
|
||||
stopServer: 'Turn off',
|
||||
startServer: 'Turn on',
|
||||
runtimeRunningDetail: 'The local server is running. Turning it off frees all GPU memory and stops new chats from using local models until you turn it back on.',
|
||||
serverStopped: 'Local server stopped — GPU memory freed.',
|
||||
serverStarted: 'Local server running.',
|
||||
serverStopFailed: 'Could not stop the local server',
|
||||
serverStartFailed: 'Could not start the local server',
|
||||
activating: 'Starting…',
|
||||
activateFailed: model => `Could not switch to ${model}`,
|
||||
activateDoneToast: model => `New chats use ${model}.`,
|
||||
downloadFailed: model => `Download of ${model} failed`,
|
||||
pillFitsGpu: 'Fits your GPU',
|
||||
pillUsesRam: 'Uses system RAM',
|
||||
pillTooBig: 'Too big for this machine',
|
||||
browseTitle: 'Find more models',
|
||||
browseHint: 'Search all of Hugging Face. Models you download here are sized to your machine automatically, but not tested by us.',
|
||||
browsePlaceholder: 'Search models by name or author…',
|
||||
browseSearching: 'Searching Hugging Face',
|
||||
browseListing: 'Reading model files',
|
||||
browseShowFiles: 'Show files',
|
||||
browseRefresh: 'Refresh',
|
||||
browseDownloads: 'downloads',
|
||||
browseLikes: 'likes',
|
||||
browseGated: 'requires Hugging Face sign-in',
|
||||
browseNoGguf: 'No compatible model files found.',
|
||||
browseFitUnknown: 'Fit unknown',
|
||||
browseAlreadyDownloaded: 'Already downloaded.',
|
||||
addedByYou: 'Added by you',
|
||||
browseDownloadStarted: 'Downloading {name}',
|
||||
browseDownloadAria: 'Download {name}',
|
||||
sideloadButton: 'Add model file',
|
||||
sideloadTitle: 'Choose a GGUF model file',
|
||||
sideloadDone: 'Added {name}.',
|
||||
sideloadAlreadyPresent: 'Already in your library.',
|
||||
pillFullContext: max => `Full ${max} context`,
|
||||
pillFullContextTip: "Runs at the model's complete context window from the start",
|
||||
pillUpTo: max => `Up to ${max} context`,
|
||||
pillGrowsTip: 'Grows automatically as your conversation needs more room',
|
||||
pillVision: 'Sees images',
|
||||
deleteAction: 'Delete model',
|
||||
deleteConfirm: model => `Delete ${model} from disk?`,
|
||||
deleted: model => `${model} deleted.`,
|
||||
deleteFailed: 'Delete failed'
|
||||
},
|
||||
providers: {
|
||||
connectAccount: 'Connect an account',
|
||||
haveApiKey: 'Have an API key instead?',
|
||||
@@ -2762,6 +2878,8 @@ export const en: Translations = {
|
||||
connected: 'Connected',
|
||||
featuredPitch: 'One subscription, 300+ frontier models — the recommended way to run Hermes',
|
||||
fireworksPitch: 'Direct model API — Fireworks-hosted frontier models',
|
||||
localModelsTitle: 'Run models locally',
|
||||
localModelsPitch: 'No account needed — download a model and run it on this machine',
|
||||
openRouterPitch: 'One key, hundreds of models — a solid default',
|
||||
apiKeyOptions: {
|
||||
fireworks: {
|
||||
@@ -2835,6 +2953,9 @@ export const en: Translations = {
|
||||
noModels: 'No models found.',
|
||||
addProvider: 'Add provider',
|
||||
loadFailed: 'Could not load models',
|
||||
loadingIntoMemory: 'Loading into memory',
|
||||
downloading: 'Downloading',
|
||||
localDownloadsHeading: 'Local',
|
||||
noAuthenticatedProviders: 'No authenticated providers.',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro models need a paid Nous subscription.',
|
||||
@@ -2961,6 +3082,15 @@ export const en: Translations = {
|
||||
openStarmap: 'Open memory graph',
|
||||
turnRunning: 'Running',
|
||||
contextUsage: 'Context usage',
|
||||
systemResources: {
|
||||
title: 'System Resources',
|
||||
loading: 'Resources…',
|
||||
gpuUtilization: 'GPU utilization',
|
||||
gpuMemory: 'GPU memory',
|
||||
ram: 'RAM',
|
||||
unifiedNote: 'Unified memory — the GPU and system share this pool.',
|
||||
toggle: 'System resources'
|
||||
},
|
||||
contextUsagePanel: {
|
||||
categories: {
|
||||
conversation: 'Conversation',
|
||||
@@ -3217,6 +3347,8 @@ export const en: Translations = {
|
||||
loadingSession: 'Loading session',
|
||||
showEarlier: 'Show earlier messages',
|
||||
loadingResponse: 'Hermes is loading a response',
|
||||
loadingLocalModel: model => `Loading ${model} into memory`,
|
||||
processingPrompt: 'Processing prompt',
|
||||
resumeWhenBackgroundDone: count =>
|
||||
count === 1
|
||||
? 'Will resume when the background task finishes'
|
||||
@@ -3545,6 +3677,11 @@ export const en: Translations = {
|
||||
title: 'Switch models mid-thread',
|
||||
text: 'The model name is a button. Change it whenever the work changes shape.'
|
||||
},
|
||||
'local-setup': {
|
||||
title: 'This machine can run models locally',
|
||||
text: 'Your hardware can serve a local model. Chats stay on your computer and cost nothing.',
|
||||
action: 'Set it up'
|
||||
},
|
||||
'right-pane': {
|
||||
title: 'The working pane',
|
||||
text: 'Files, terminal, review and the in-app browser share the right side.'
|
||||
|
||||
@@ -278,6 +278,7 @@ export const ja = defineLocale({
|
||||
providerAccounts: 'アカウント',
|
||||
providerApiKeys: 'API キー',
|
||||
providerCustomEndpoints: 'カスタムエンドポイント',
|
||||
providerLocalModels: 'ローカルモデル',
|
||||
gateway: 'ゲートウェイ',
|
||||
apiKeys: 'ツールとキー',
|
||||
keybinds: 'キーボードショートカット',
|
||||
@@ -1023,6 +1024,106 @@ export const ja = defineLocale({
|
||||
curator: { label: 'キュレーター', hint: 'スキル使用レビュー' }
|
||||
}
|
||||
},
|
||||
localModels: {
|
||||
title: 'ローカルモデル',
|
||||
runtimeTitle: 'ローカルランタイム',
|
||||
runtimeReady: backend => `準備完了 · ${backend}`,
|
||||
serverRunning: '実行中',
|
||||
runtimeInstalled: 'llama.cpp ランタイムをインストール済み',
|
||||
runtimeInstalledDetail: (tag, backend) =>
|
||||
`ビルド ${tag}、${backend} バックエンド。サーバーは Hermes が起動・管理します。`,
|
||||
installTitle: 'ローカルランタイムをインストール',
|
||||
installDetail:
|
||||
'llama.cpp 推論エンジン(数百 MB)をダウンロードします。ダウンロードしたモデルはすべてこのマシン上で動作します——アカウント不要、データが外部に送られることはありません。',
|
||||
installAction: 'ランタイムをインストール',
|
||||
installing: 'ランタイムをインストール中…',
|
||||
installFailed: 'ランタイムのインストールに失敗しました',
|
||||
hardwareTitle: 'このマシン',
|
||||
hardwareLoading: 'ハードウェアを確認中…',
|
||||
vram: label => `GPU メモリ ${label}`,
|
||||
ram: label => `RAM ${label}`,
|
||||
unifiedMemory: 'ユニファイドメモリ',
|
||||
modelsTitle: 'モデル',
|
||||
recommended: 'おすすめ',
|
||||
recommendedReason: {
|
||||
'best-quality-resident':
|
||||
'GPU に完全に載り、フルスピードで動くモデルの中で最高品質です。おすすめは品質とこのハードウェアでの予測速度を両立させて選ばれます。',
|
||||
'speed-gated-quality':
|
||||
'より高品質なモデルもこのマシンに載りますが、メモリ帯域の制約で応答が遅くなります — これは速度を保てる最良のモデルです。',
|
||||
'fastest-resident':
|
||||
'このハードウェアでフルスピードに達するモデルはありません。GPU メモリ内で動くものの中で最速です。',
|
||||
'least-painful-spilled':
|
||||
'GPU メモリに完全に収まるモデルはありません — システム RAM からの実行で最も快適なモデルです。'
|
||||
} as Record<string, string>,
|
||||
downloaded: 'ダウンロード済み',
|
||||
downloadAction: size => `ダウンロード · ${size}`,
|
||||
downloadProgress: (done, total) => `ダウンロード中 ${done} / ${total}`,
|
||||
downloadDoneToast: model => `${model} の準備ができました。`,
|
||||
installDoneToast: 'ローカルランタイムのインストールが完了しました。',
|
||||
useAction: '使用する',
|
||||
activePill: 'デフォルト',
|
||||
updateTitle: 'エンジンの更新があります',
|
||||
updateDetail: (next, current) => `新しい llama.cpp ビルド(${next})をインストールできます——現在は ${current} です。ダウンロード中もモデルは引き続き使えます。`,
|
||||
updateAction: 'エンジンを更新',
|
||||
updating: 'エンジンを更新中…',
|
||||
upToDateTitle: 'エンジンは最新です',
|
||||
upToDateDetail: (tag, backend) => `llama.cpp ${tag}(${backend})で動作中——Hermes が提供する最新ビルドです。`,
|
||||
updateToast: next => `ローカルエンジンの新しいビルド(${next})があります。設定 → ローカルモデル から更新できます。`,
|
||||
activeDetail: '新しいチャットはこのモデルを使用——最初のメッセージ送信時に読み込みます',
|
||||
activeNotLoaded: '最初のメッセージで読み込みます',
|
||||
loadedPill: '読み込み済み',
|
||||
placementResident: 'すべて GPU 上',
|
||||
placementSpilled: '一部 RAM 上',
|
||||
placementResidentTip: 'このコンテキストウィンドウで GPU メモリ内で完全に動作しています — フルスピード。',
|
||||
placementSpilledTip: 'モデルの一部がシステム RAM から動作しています — 動作しますが遅くなります。よりコンパクトなビルドか小さいコンテキストなら完全に収まります。',
|
||||
loadingPill: '読み込み中…',
|
||||
ejectTip: 'GPU メモリを解放(必要時に再読み込み)',
|
||||
ejected: 'モデルをアンロードしました——GPU メモリを解放しました。',
|
||||
ejectFailed: 'モデルをアンロードできませんでした',
|
||||
stopServer: 'オフにする',
|
||||
startServer: 'オンにする',
|
||||
runtimeRunningDetail: 'ローカルサーバーが実行中です。オフにすると GPU メモリを全て解放し、再度オンにするまで新しいチャットはローカルモデルを使用しません。',
|
||||
serverStopped: 'ローカルサーバーを停止しました——GPU メモリを解放しました。',
|
||||
serverStarted: 'ローカルサーバー実行中。',
|
||||
serverStopFailed: 'ローカルサーバーを停止できませんでした',
|
||||
serverStartFailed: 'ローカルサーバーを起動できませんでした',
|
||||
activating: '起動中…',
|
||||
activateFailed: model => `${model} への切り替えに失敗しました`,
|
||||
activateDoneToast: model => `新しいチャットは ${model} を使用します。`,
|
||||
downloadFailed: model => `${model} のダウンロードに失敗しました`,
|
||||
pillFitsGpu: 'GPU に完全に収まります',
|
||||
pillUsesRam: 'システム RAM を使用',
|
||||
pillTooBig: 'このマシンには大きすぎます',
|
||||
browseTitle: 'さらにモデルを探す',
|
||||
browseHint: 'Hugging Face 全体を検索できます。ここでダウンロードしたモデルは自動でマシンに合わせて動作しますが、当方でのテストは行われていません。',
|
||||
browsePlaceholder: 'モデル名または作者で検索…',
|
||||
browseSearching: 'Hugging Face を検索中',
|
||||
browseListing: 'モデルファイルを読み込み中',
|
||||
browseShowFiles: 'ファイルを表示',
|
||||
browseRefresh: '更新',
|
||||
browseDownloads: 'ダウンロード',
|
||||
browseLikes: 'いいね',
|
||||
browseGated: 'Hugging Face へのサインインが必要',
|
||||
browseNoGguf: '互換性のあるモデルファイルが見つかりません。',
|
||||
browseFitUnknown: '適合状況は不明',
|
||||
browseAlreadyDownloaded: 'ダウンロード済みです。',
|
||||
addedByYou: 'あなたが追加',
|
||||
browseDownloadStarted: '{name} をダウンロード中',
|
||||
browseDownloadAria: '{name} をダウンロード',
|
||||
sideloadButton: 'モデルファイルを追加',
|
||||
sideloadTitle: 'GGUF モデルファイルを選択',
|
||||
sideloadDone: '{name} を追加しました。',
|
||||
sideloadAlreadyPresent: '既にライブラリにあります。',
|
||||
pillFullContext: max => `フル ${max} コンテキスト`,
|
||||
pillFullContextTip: '最初からモデルの完全なコンテキストウィンドウで動作します',
|
||||
pillUpTo: max => `最大 ${max} コンテキスト`,
|
||||
pillGrowsTip: '会話が必要とするにつれて自動的に拡張します',
|
||||
pillVision: '画像対応',
|
||||
deleteAction: 'モデルを削除',
|
||||
deleteConfirm: model => `${model} をディスクから削除しますか?`,
|
||||
deleted: model => `${model} を削除しました。`,
|
||||
deleteFailed: '削除に失敗しました'
|
||||
},
|
||||
providers: {
|
||||
connectAccount: 'アカウントを接続',
|
||||
haveApiKey: 'API キーをお持ちですか?',
|
||||
@@ -2406,6 +2507,8 @@ export const ja = defineLocale({
|
||||
connected: '接続済み',
|
||||
featuredPitch: '1 つのサブスクリプションで 300 以上の最先端モデル — Hermes を実行するための推奨方法',
|
||||
fireworksPitch: '直接モデル API — Fireworks がホストする最先端モデル',
|
||||
localModelsTitle: 'モデルをローカルで実行',
|
||||
localModelsPitch: 'アカウント不要——モデルをダウンロードしてこのマシンで実行',
|
||||
openRouterPitch: '1 つのキーで数百のモデル — 堅実なデフォルト',
|
||||
apiKeyOptions: {
|
||||
fireworks: {
|
||||
@@ -2479,6 +2582,8 @@ export const ja = defineLocale({
|
||||
noModels: 'モデルが見つかりません。',
|
||||
addProvider: 'プロバイダーを追加',
|
||||
loadFailed: 'モデルを読み込めませんでした',
|
||||
downloading: 'ダウンロード中',
|
||||
localDownloadsHeading: 'ローカル',
|
||||
noAuthenticatedProviders: '認証済みプロバイダーがありません。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro モデルには有料の Nous サブスクリプションが必要です。',
|
||||
@@ -2591,6 +2696,15 @@ export const ja = defineLocale({
|
||||
openStarmap: 'メモリグラフを開く',
|
||||
turnRunning: '実行中',
|
||||
contextUsage: 'コンテキスト使用状況',
|
||||
systemResources: {
|
||||
title: 'システムリソース',
|
||||
loading: 'リソース…',
|
||||
gpuUtilization: 'GPU 使用率',
|
||||
gpuMemory: 'GPU メモリ',
|
||||
ram: 'RAM',
|
||||
unifiedNote: 'ユニファイドメモリ——GPU とシステムがこのプールを共有します。',
|
||||
toggle: 'システムリソース'
|
||||
},
|
||||
contextUsagePanel: {
|
||||
categories: {
|
||||
conversation: '会話',
|
||||
@@ -3160,6 +3274,11 @@ export const ja = defineLocale({
|
||||
title: '会話の途中でモデルを変更',
|
||||
text: 'モデル名はボタンです。作業の性質が変わったら切り替えてください。'
|
||||
},
|
||||
'local-setup': {
|
||||
title: 'このマシンはローカルでモデルを実行できます',
|
||||
text: 'お使いのハードウェアでローカルモデルを動かせます。会話はこのコンピュータから出ず、料金もかかりません。',
|
||||
action: 'セットアップ'
|
||||
},
|
||||
'right-pane': {
|
||||
title: '作業用ペイン',
|
||||
text: 'ファイル、ターミナル、レビュー、アプリ内ブラウザはサイドペインにまとまっています。'
|
||||
|
||||
@@ -338,6 +338,7 @@ export interface Translations {
|
||||
providerAccounts: string
|
||||
providerApiKeys: string
|
||||
providerCustomEndpoints: string
|
||||
providerLocalModels: string
|
||||
gateway: string
|
||||
apiKeys: string
|
||||
keybinds: string
|
||||
@@ -966,6 +967,107 @@ export interface Translations {
|
||||
notInCatalog: string
|
||||
tasks: Record<string, AuxTaskCopy>
|
||||
}
|
||||
localModels: {
|
||||
title: string
|
||||
runtimeTitle: string
|
||||
runtimeReady: (backend: string) => string
|
||||
serverRunning: string
|
||||
runtimeInstalled: string
|
||||
runtimeInstalledDetail: (tag: string, backend: string) => string
|
||||
installTitle: string
|
||||
installDetail: string
|
||||
installAction: string
|
||||
installing: string
|
||||
installFailed: string
|
||||
hardwareTitle: string
|
||||
hardwareLoading: string
|
||||
vram: (label: string) => string
|
||||
ram: (label: string) => string
|
||||
unifiedMemory: string
|
||||
modelsTitle: string
|
||||
recommended: string
|
||||
/** Recommended-badge tooltip by resolver branch; unknown keys (newer
|
||||
* backend) simply show no tooltip. */
|
||||
recommendedReason: Record<string, string>
|
||||
downloaded: string
|
||||
downloadAction: (size: string) => string
|
||||
downloadProgress: (done: string, total: string) => string
|
||||
downloadDoneToast: (model: string) => string
|
||||
installDoneToast: string
|
||||
quickstartTitle: string
|
||||
quickstartDetail: (model: string, size: string) => string
|
||||
quickstartDetailReady: (model: string) => string
|
||||
quickstartAction: string
|
||||
quickstartConfigure: string
|
||||
quickstartDoneToast: (model: string) => string
|
||||
quickstartFailed: string
|
||||
quickstartStageEngine: string
|
||||
quickstartStageModel: string
|
||||
quickstartStageFinish: string
|
||||
useAction: string
|
||||
activePill: string
|
||||
updateTitle: string
|
||||
updateDetail: (next: string, current: string) => string
|
||||
updateAction: string
|
||||
updating: string
|
||||
upToDateTitle: string
|
||||
upToDateDetail: (tag: string, backend: string) => string
|
||||
updateToast: (next: string) => string
|
||||
activeDetail: string
|
||||
activeNotLoaded: string
|
||||
loadedPill: string
|
||||
placementResident: string
|
||||
placementSpilled: string
|
||||
placementResidentTip: string
|
||||
placementSpilledTip: string
|
||||
loadingPill: string
|
||||
ejectTip: string
|
||||
ejected: string
|
||||
ejectFailed: string
|
||||
stopServer: string
|
||||
startServer: string
|
||||
runtimeRunningDetail: string
|
||||
serverStopped: string
|
||||
serverStarted: string
|
||||
serverStopFailed: string
|
||||
serverStartFailed: string
|
||||
activating: string
|
||||
activateFailed: (model: string) => string
|
||||
activateDoneToast: (model: string) => string
|
||||
downloadFailed: (model: string) => string
|
||||
pillFitsGpu: string
|
||||
pillUsesRam: string
|
||||
pillTooBig: string
|
||||
browseTitle: string
|
||||
browseHint: string
|
||||
browsePlaceholder: string
|
||||
browseSearching: string
|
||||
browseListing: string
|
||||
browseShowFiles: string
|
||||
browseRefresh: string
|
||||
browseDownloads: string
|
||||
browseLikes: string
|
||||
browseGated: string
|
||||
browseNoGguf: string
|
||||
browseFitUnknown: string
|
||||
browseAlreadyDownloaded: string
|
||||
addedByYou: string
|
||||
browseDownloadStarted: string
|
||||
browseDownloadAria: string
|
||||
sideloadButton: string
|
||||
sideloadTitle: string
|
||||
sideloadDone: string
|
||||
sideloadAlreadyPresent: string
|
||||
pillFullContext: (max: string) => string
|
||||
pillFullContextTip: string
|
||||
pillUpTo: (max: string) => string
|
||||
pillGrowsTip: string
|
||||
pillVision: string
|
||||
deleteAction: string
|
||||
deleteConfirm: (model: string) => string
|
||||
deleted: (model: string) => string
|
||||
deleteFailed: string
|
||||
}
|
||||
providers: {
|
||||
connectAccount: string
|
||||
haveApiKey: string
|
||||
@@ -2350,6 +2452,8 @@ export interface Translations {
|
||||
connected: string
|
||||
featuredPitch: string
|
||||
fireworksPitch: string
|
||||
localModelsTitle: string
|
||||
localModelsPitch: string
|
||||
openRouterPitch: string
|
||||
apiKeyOptions: Record<string, { short: string; description: string }>
|
||||
backToSignIn: string
|
||||
@@ -2400,6 +2504,9 @@ export interface Translations {
|
||||
noModels: string
|
||||
addProvider: string
|
||||
loadFailed: string
|
||||
loadingIntoMemory: string
|
||||
downloading: string
|
||||
localDownloadsHeading: string
|
||||
noAuthenticatedProviders: string
|
||||
pro: string
|
||||
proNeedsSubscription: string
|
||||
@@ -2526,6 +2633,15 @@ export interface Translations {
|
||||
openStarmap: string
|
||||
turnRunning: string
|
||||
contextUsage: string
|
||||
systemResources: {
|
||||
title: string
|
||||
loading: string
|
||||
gpuUtilization: string
|
||||
gpuMemory: string
|
||||
ram: string
|
||||
unifiedNote: string
|
||||
toggle: string
|
||||
}
|
||||
contextUsagePanel: {
|
||||
categories: {
|
||||
conversation: string
|
||||
@@ -2777,6 +2893,8 @@ export interface Translations {
|
||||
loadingSession: string
|
||||
showEarlier: string
|
||||
loadingResponse: string
|
||||
loadingLocalModel: (model: string) => string
|
||||
processingPrompt: string
|
||||
resumeWhenBackgroundDone: (count: number) => string
|
||||
thinking: string
|
||||
thought: string
|
||||
@@ -3027,8 +3145,12 @@ export interface Translations {
|
||||
|
||||
tips: {
|
||||
close: string
|
||||
/** Keyed by `TipId`, so a new tip without copy is a type error. */
|
||||
items: Record<TipId, { title: string; text: string }>
|
||||
/** Keyed by `TipId`, so a new tip without copy is a type error. Plus the
|
||||
* campaign tips, which live outside the rotation's catalog: they carry
|
||||
* a button, and `action` is its label. */
|
||||
items: Record<TipId, { title: string; text: string }> & {
|
||||
'local-setup': { title: string; text: string; action: string }
|
||||
}
|
||||
}
|
||||
|
||||
errors: {
|
||||
|
||||
@@ -269,6 +269,7 @@ export const zhHant = defineLocale({
|
||||
providerAccounts: '帳號',
|
||||
providerApiKeys: 'API 金鑰',
|
||||
providerCustomEndpoints: '自訂端點',
|
||||
providerLocalModels: '本地模型',
|
||||
gateway: '閘道',
|
||||
apiKeys: '工具與金鑰',
|
||||
keybinds: '鍵盤快捷鍵',
|
||||
@@ -986,6 +987,101 @@ export const zhHant = defineLocale({
|
||||
curator: { label: '策展器', hint: '技能使用審查' }
|
||||
}
|
||||
},
|
||||
localModels: {
|
||||
title: '本地模型',
|
||||
runtimeTitle: '本地執行環境',
|
||||
runtimeReady: backend => `就緒 · ${backend}`,
|
||||
serverRunning: '執行中',
|
||||
runtimeInstalled: '已安裝 llama.cpp 執行環境',
|
||||
runtimeInstalledDetail: (tag, backend) => `組建 ${tag},${backend} 後端。Hermes 會為您啟動並管理伺服器。`,
|
||||
installTitle: '安裝本地執行環境',
|
||||
installDetail:
|
||||
'下載 llama.cpp 推理引擎(數百 MB)。下載的模型完全在本機執行——無需帳號,資料不會離開您的電腦。',
|
||||
installAction: '安裝執行環境',
|
||||
installing: '正在安裝執行環境…',
|
||||
installFailed: '執行環境安裝失敗',
|
||||
hardwareTitle: '本機配置',
|
||||
hardwareLoading: '正在檢測硬體…',
|
||||
vram: label => `${label} 顯示記憶體`,
|
||||
ram: label => `${label} 記憶體`,
|
||||
unifiedMemory: '統一記憶體',
|
||||
modelsTitle: '模型',
|
||||
recommended: '推薦',
|
||||
recommendedReason: {
|
||||
'best-quality-resident': '在完全駐留 GPU 且保持全速的模型中品質最高。推薦會在品質與該硬體的預計速度之間權衡。',
|
||||
'speed-gated-quality': '有更高品質的模型可以裝入這台機器,但受記憶體頻寬限制回應會太慢——這是保持流暢的最佳模型。',
|
||||
'fastest-resident': '沒有模型能在該硬體上達到全速;這是完全駐留 GPU 記憶體中最快的一個。',
|
||||
'least-painful-spilled': '沒有模型能完全裝入 GPU 記憶體——這是從系統記憶體執行表現最好的一個。'
|
||||
} as Record<string, string>,
|
||||
downloaded: '已下載',
|
||||
downloadAction: size => `下載 · ${size}`,
|
||||
downloadProgress: (done, total) => `正在下載 ${done} / ${total}`,
|
||||
downloadDoneToast: model => `${model} 已就緒。`,
|
||||
installDoneToast: '本地執行環境已安裝就緒。',
|
||||
useAction: '使用',
|
||||
activePill: '預設',
|
||||
updateTitle: '引擎有可用更新',
|
||||
updateDetail: (next, current) => `新的 llama.cpp 組建(${next})可以安裝——目前為 ${current}。下載期間模型仍可正常使用。`,
|
||||
updateAction: '更新引擎',
|
||||
updating: '正在更新引擎…',
|
||||
upToDateTitle: '引擎已是最新',
|
||||
upToDateDetail: (tag, backend) => `正在執行 llama.cpp ${tag}(${backend})——Hermes 提供的最新組建。`,
|
||||
updateToast: next => `本地引擎有新組建(${next})。可在 設定 → 本地模型 中更新。`,
|
||||
activeDetail: '新對話使用此模型——傳送首條訊息時載入',
|
||||
activeNotLoaded: '首條訊息時載入',
|
||||
loadedPill: '已載入',
|
||||
placementResident: '全部在 GPU',
|
||||
placementSpilled: '部分在記憶體',
|
||||
placementResidentTip: '完全在 GPU 記憶體中以此上下文視窗執行——全速。',
|
||||
placementSpilledTip: '模型的一部分從系統記憶體執行——可用但較慢。更緊湊的版本或更小的上下文可以完全放入顯示記憶體。',
|
||||
loadingPill: '載入中…',
|
||||
ejectTip: '釋放顯示記憶體(需要時重新載入)',
|
||||
ejected: '模型已卸載——顯示記憶體已釋放。',
|
||||
ejectFailed: '無法卸載模型',
|
||||
stopServer: '關閉',
|
||||
startServer: '開啟',
|
||||
runtimeRunningDetail: '本地伺服器執行中。關閉後將釋放全部顯示記憶體,新對話將不再使用本地模型,直到您重新開啟。',
|
||||
serverStopped: '本地伺服器已停止——顯示記憶體已釋放。',
|
||||
serverStarted: '本地伺服器執行中。',
|
||||
serverStopFailed: '無法停止本地伺服器',
|
||||
serverStartFailed: '無法啟動本地伺服器',
|
||||
activating: '啟動中…',
|
||||
activateFailed: model => `無法切換到 ${model}`,
|
||||
activateDoneToast: model => `新對話將使用 ${model}。`,
|
||||
downloadFailed: model => `${model} 下載失敗`,
|
||||
pillFitsGpu: '完全在 GPU 上執行',
|
||||
pillUsesRam: '使用系統記憶體',
|
||||
pillTooBig: '超出本機記憶體',
|
||||
browseTitle: '發現更多模型',
|
||||
browseHint: '搜尋整個 Hugging Face。在這裡下載的模型會自動適配你的機器,但未經我們測試。',
|
||||
browsePlaceholder: '按名稱或作者搜尋模型…',
|
||||
browseSearching: '正在搜尋 Hugging Face',
|
||||
browseListing: '正在讀取模型檔案',
|
||||
browseShowFiles: '查看檔案',
|
||||
browseRefresh: '重新整理',
|
||||
browseDownloads: '次下載',
|
||||
browseLikes: '個讚',
|
||||
browseGated: '需要登入 Hugging Face',
|
||||
browseNoGguf: '未找到相容的模型檔案。',
|
||||
browseFitUnknown: '適配情況未知',
|
||||
browseAlreadyDownloaded: '已下載。',
|
||||
addedByYou: '由你新增',
|
||||
browseDownloadStarted: '正在下載 {name}',
|
||||
browseDownloadAria: '下載 {name}',
|
||||
sideloadButton: '新增模型檔案',
|
||||
sideloadTitle: '選擇 GGUF 模型檔案',
|
||||
sideloadDone: '已新增 {name}。',
|
||||
sideloadAlreadyPresent: '已在你的庫中。',
|
||||
pillFullContext: max => `完整 ${max} 上下文`,
|
||||
pillFullContextTip: '從一開始就以模型的完整上下文視窗執行',
|
||||
pillUpTo: max => `最高 ${max} 上下文`,
|
||||
pillGrowsTip: '隨著對話需要更多空間自動增長',
|
||||
pillVision: '識圖',
|
||||
deleteAction: '刪除模型',
|
||||
deleteConfirm: model => `從磁碟刪除 ${model}?`,
|
||||
deleted: model => `已刪除 ${model}。`,
|
||||
deleteFailed: '刪除失敗'
|
||||
},
|
||||
providers: {
|
||||
connectAccount: '連結帳號',
|
||||
haveApiKey: '改用 API 金鑰?',
|
||||
@@ -2323,6 +2419,8 @@ export const zhHant = defineLocale({
|
||||
connected: '已連線',
|
||||
featuredPitch: '一個訂閱,300+ 前沿模型 — 執行 Hermes 的建議方式',
|
||||
fireworksPitch: '直接模型 API — Fireworks 託管的前沿模型',
|
||||
localModelsTitle: '本地執行模型',
|
||||
localModelsPitch: '無需帳號——下載模型,在本機執行',
|
||||
openRouterPitch: '一個金鑰,數百個模型 — 穩定的預設選擇',
|
||||
apiKeyOptions: {
|
||||
fireworks: { short: '直接模型 API', description: '直接存取 Fireworks AI 託管的模型。' },
|
||||
@@ -2387,6 +2485,8 @@ export const zhHant = defineLocale({
|
||||
noModels: '找不到模型。',
|
||||
addProvider: '新增提供方',
|
||||
loadFailed: '無法載入模型',
|
||||
downloading: '下載中',
|
||||
localDownloadsHeading: '本地',
|
||||
noAuthenticatedProviders: '沒有已驗證的提供方。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro 模型需要付費 Nous 訂閱。',
|
||||
@@ -2499,6 +2599,15 @@ export const zhHant = defineLocale({
|
||||
openStarmap: '開啟記憶圖譜',
|
||||
turnRunning: '執行中',
|
||||
contextUsage: '上下文使用量',
|
||||
systemResources: {
|
||||
title: '系統資源',
|
||||
loading: '資源…',
|
||||
gpuUtilization: 'GPU 使用率',
|
||||
gpuMemory: '顯示記憶體',
|
||||
ram: '記憶體',
|
||||
unifiedNote: '統一記憶體——GPU 與系統共享此記憶體池。',
|
||||
toggle: '系統資源'
|
||||
},
|
||||
contextUsagePanel: {
|
||||
categories: {
|
||||
conversation: '對話',
|
||||
@@ -3038,6 +3147,11 @@ export const zhHant = defineLocale({
|
||||
title: '對話中隨時換模型',
|
||||
text: '模型名稱就是按鈕。工作性質變了就換一個。'
|
||||
},
|
||||
'local-setup': {
|
||||
title: '這台電腦可以本地執行模型',
|
||||
text: '你的硬體可以執行本地模型。對話不離開你的電腦,而且完全免費。',
|
||||
action: '立即設定'
|
||||
},
|
||||
'right-pane': {
|
||||
title: '工作面板',
|
||||
text: '檔案、終端機、審閱與內建瀏覽器都在側邊面板裡。'
|
||||
|
||||
@@ -382,6 +382,7 @@ export const zh: Translations = {
|
||||
providerAccounts: '账号',
|
||||
providerApiKeys: 'API 密钥',
|
||||
providerCustomEndpoints: '自定义端点',
|
||||
providerLocalModels: '本地模型',
|
||||
gateway: '网关',
|
||||
apiKeys: '工具与密钥',
|
||||
keybinds: '键盘快捷键',
|
||||
@@ -1311,6 +1312,111 @@ export const zh: Translations = {
|
||||
curator: { label: '维护器', hint: '技能使用审查' }
|
||||
}
|
||||
},
|
||||
localModels: {
|
||||
title: '本地模型',
|
||||
runtimeTitle: '本地运行时',
|
||||
runtimeReady: backend => `就绪 · ${backend}`,
|
||||
serverRunning: '运行中',
|
||||
runtimeInstalled: '已安装 llama.cpp 运行时',
|
||||
runtimeInstalledDetail: (tag, backend) => `构建 ${tag},${backend} 后端。Hermes 会为您启动并管理服务器。`,
|
||||
installTitle: '安装本地运行时',
|
||||
installDetail: '下载 llama.cpp 推理引擎(几百 MB)。下载的模型完全在本机运行——无需账号,数据不会离开您的电脑。',
|
||||
installAction: '安装运行时',
|
||||
installing: '正在安装运行时…',
|
||||
installFailed: '运行时安装失败',
|
||||
quickstartTitle: '在本机运行模型',
|
||||
quickstartDetail: (model, size) =>
|
||||
`一键完成所有设置:本地引擎、${model}(需下载 ${size}),并设为新会话的默认模型。数据不会离开这台电脑。`,
|
||||
quickstartDetailReady: model => `一键将 ${model} 设为新会话的默认模型。所有内容都在本机运行。`,
|
||||
quickstartAction: '为我设置',
|
||||
quickstartConfigure: '自定义…',
|
||||
quickstartDoneToast: model => `${model} 已就绪——新会话将在本机运行。`,
|
||||
quickstartFailed: '本地模型设置失败',
|
||||
quickstartStageEngine: '引擎',
|
||||
quickstartStageModel: '模型',
|
||||
quickstartStageFinish: '完成',
|
||||
hardwareTitle: '本机配置',
|
||||
hardwareLoading: '正在检测硬件…',
|
||||
vram: label => `${label} 显存`,
|
||||
ram: label => `${label} 内存`,
|
||||
unifiedMemory: '统一内存',
|
||||
modelsTitle: '模型',
|
||||
recommended: '推荐',
|
||||
recommendedReason: {
|
||||
'best-quality-resident': '在完全驻留 GPU 且保持全速的模型中质量最高。推荐会在质量与该硬件的预计速度之间权衡。',
|
||||
'speed-gated-quality': '有更高质量的模型可以装入这台机器,但受内存带宽限制响应会太慢——这是保持流畅的最佳模型。',
|
||||
'fastest-resident': '没有模型能在该硬件上达到全速;这是完全驻留 GPU 内存中最快的一个。',
|
||||
'least-painful-spilled': '没有模型能完全装入 GPU 内存——这是从系统内存运行表现最好的一个。'
|
||||
} as Record<string, string>,
|
||||
downloaded: '已下载',
|
||||
downloadAction: size => `下载 · ${size}`,
|
||||
downloadProgress: (done, total) => `正在下载 ${done} / ${total}`,
|
||||
downloadDoneToast: model => `${model} 已就绪。`,
|
||||
installDoneToast: '本地运行时已安装就绪。',
|
||||
useAction: '使用',
|
||||
activePill: '默认',
|
||||
updateTitle: '引擎有可用更新',
|
||||
updateDetail: (next, current) => `新的 llama.cpp 构建(${next})可以安装——当前为 ${current}。下载期间模型仍可正常使用。`,
|
||||
updateAction: '更新引擎',
|
||||
updating: '正在更新引擎…',
|
||||
upToDateTitle: '引擎已是最新',
|
||||
upToDateDetail: (tag, backend) => `正在运行 llama.cpp ${tag}(${backend})——Hermes 提供的最新构建。`,
|
||||
updateToast: next => `本地引擎有新构建(${next})。可在 设置 → 本地模型 中更新。`,
|
||||
activeDetail: '新对话使用此模型——发送首条消息时加载',
|
||||
activeNotLoaded: '首条消息时加载',
|
||||
loadedPill: '已加载',
|
||||
placementResident: '全部在 GPU',
|
||||
placementSpilled: '部分在内存',
|
||||
placementResidentTip: '完全在 GPU 显存中以此上下文窗口运行——全速。',
|
||||
placementSpilledTip: '模型的一部分从系统内存运行——可用但较慢。更紧凑的版本或更小的上下文可以完全放入显存。',
|
||||
loadingPill: '加载中…',
|
||||
ejectTip: '释放显存(需要时重新加载)',
|
||||
ejected: '模型已卸载——显存已释放。',
|
||||
ejectFailed: '无法卸载模型',
|
||||
stopServer: '关闭',
|
||||
startServer: '开启',
|
||||
runtimeRunningDetail: '本地服务器正在运行。关闭后将释放全部显存,新对话将不再使用本地模型,直到您重新开启。',
|
||||
serverStopped: '本地服务器已停止——显存已释放。',
|
||||
serverStarted: '本地服务器运行中。',
|
||||
serverStopFailed: '无法停止本地服务器',
|
||||
serverStartFailed: '无法启动本地服务器',
|
||||
activating: '启动中…',
|
||||
activateFailed: model => `无法切换到 ${model}`,
|
||||
activateDoneToast: model => `新对话将使用 ${model}。`,
|
||||
downloadFailed: model => `${model} 下载失败`,
|
||||
pillFitsGpu: '完全在 GPU 上运行',
|
||||
pillUsesRam: '使用系统内存',
|
||||
pillTooBig: '超出本机内存',
|
||||
browseTitle: '发现更多模型',
|
||||
browseHint: '搜索整个 Hugging Face。在这里下载的模型会自动适配你的机器,但未经我们测试。',
|
||||
browsePlaceholder: '按名称或作者搜索模型…',
|
||||
browseSearching: '正在搜索 Hugging Face',
|
||||
browseListing: '正在读取模型文件',
|
||||
browseShowFiles: '查看文件',
|
||||
browseRefresh: '刷新',
|
||||
browseDownloads: '次下载',
|
||||
browseLikes: '个赞',
|
||||
browseGated: '需要登录 Hugging Face',
|
||||
browseNoGguf: '未找到兼容的模型文件。',
|
||||
browseFitUnknown: '适配情况未知',
|
||||
browseAlreadyDownloaded: '已下载。',
|
||||
addedByYou: '由你添加',
|
||||
browseDownloadStarted: '正在下载 {name}',
|
||||
browseDownloadAria: '下载 {name}',
|
||||
sideloadButton: '添加模型文件',
|
||||
sideloadTitle: '选择 GGUF 模型文件',
|
||||
sideloadDone: '已添加 {name}。',
|
||||
sideloadAlreadyPresent: '已在你的库中。',
|
||||
pillFullContext: max => `完整 ${max} 上下文`,
|
||||
pillFullContextTip: '从一开始就以模型的完整上下文窗口运行',
|
||||
pillUpTo: max => `最高 ${max} 上下文`,
|
||||
pillGrowsTip: '随着对话需要更多空间自动增长',
|
||||
pillVision: '识图',
|
||||
deleteAction: '删除模型',
|
||||
deleteConfirm: model => `从磁盘删除 ${model}?`,
|
||||
deleted: model => `已删除 ${model}。`,
|
||||
deleteFailed: '删除失败'
|
||||
},
|
||||
providers: {
|
||||
connectAccount: '连接账号',
|
||||
haveApiKey: '改用 API 密钥?',
|
||||
@@ -2933,6 +3039,8 @@ export const zh: Translations = {
|
||||
connected: '已连接',
|
||||
featuredPitch: '一个订阅,300+ 前沿模型 — 运行 Hermes 的推荐方式',
|
||||
fireworksPitch: '直接模型 API — Fireworks 托管的前沿模型',
|
||||
localModelsTitle: '本地运行模型',
|
||||
localModelsPitch: '无需账号——下载模型,在本机运行',
|
||||
openRouterPitch: '一个密钥,数百个模型 — 稳妥的默认选择',
|
||||
apiKeyOptions: {
|
||||
fireworks: { short: '直接模型 API', description: '直接访问 Fireworks AI 托管的模型。' },
|
||||
@@ -2998,6 +3106,9 @@ export const zh: Translations = {
|
||||
noModels: '未找到模型。',
|
||||
addProvider: '添加提供方',
|
||||
loadFailed: '无法加载模型',
|
||||
loadingIntoMemory: '正在载入内存',
|
||||
downloading: '下载中',
|
||||
localDownloadsHeading: '本地',
|
||||
noAuthenticatedProviders: '没有已认证的提供方。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro 模型需要付费 Nous 订阅。',
|
||||
@@ -3124,6 +3235,15 @@ export const zh: Translations = {
|
||||
openStarmap: '打开记忆图谱',
|
||||
turnRunning: '运行中',
|
||||
contextUsage: '上下文用量',
|
||||
systemResources: {
|
||||
title: '系统资源',
|
||||
loading: '资源…',
|
||||
gpuUtilization: 'GPU 利用率',
|
||||
gpuMemory: '显存',
|
||||
ram: '内存',
|
||||
unifiedNote: '统一内存——GPU 与系统共享此内存池。',
|
||||
toggle: '系统资源'
|
||||
},
|
||||
contextUsagePanel: {
|
||||
categories: {
|
||||
conversation: '对话',
|
||||
@@ -3377,6 +3497,8 @@ export const zh: Translations = {
|
||||
loadingSession: '正在加载会话',
|
||||
showEarlier: '显示更早的消息',
|
||||
loadingResponse: 'Hermes 正在加载回复',
|
||||
loadingLocalModel: model => `正在将 ${model} 载入内存`,
|
||||
processingPrompt: '正在处理提示词',
|
||||
resumeWhenBackgroundDone: count =>
|
||||
count === 1 ? '后台任务完成后将自动继续' : `${count} 个后台任务完成后将自动继续`,
|
||||
thinking: '思考中',
|
||||
@@ -3688,6 +3810,11 @@ export const zh: Translations = {
|
||||
title: '对话中随时换模型',
|
||||
text: '模型名称就是按钮。工作性质变了就换一个。'
|
||||
},
|
||||
'local-setup': {
|
||||
title: '这台电脑可以本地运行模型',
|
||||
text: '你的硬件可以运行本地模型。对话不离开你的电脑,而且完全免费。',
|
||||
action: '立即设置'
|
||||
},
|
||||
'right-pane': {
|
||||
title: '工作面板',
|
||||
text: '文件、终端、审阅和内置浏览器都在侧边面板里。'
|
||||
|
||||
@@ -40,6 +40,7 @@ import {
|
||||
IconEar as Ear,
|
||||
IconEarOff as EarOff,
|
||||
IconEgg as Egg,
|
||||
IconPlayerEjectFilled as Eject,
|
||||
IconExternalLink as ExternalLink,
|
||||
IconEye as Eye,
|
||||
IconEyeOff as EyeOff,
|
||||
@@ -170,6 +171,7 @@ export {
|
||||
Ear,
|
||||
EarOff,
|
||||
Egg,
|
||||
Eject,
|
||||
ExternalLink,
|
||||
Eye,
|
||||
EyeOff,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { currentPickerSelection, displayModelName, formatModelStatusLabel } from './model-status-label'
|
||||
import { currentPickerSelection, displayModelName, formatModelStatusLabel, modelDisplayParts } from './model-status-label'
|
||||
import { reasoningEffortLabel } from './reasoning-effort'
|
||||
|
||||
describe('model-status-label', () => {
|
||||
@@ -16,6 +16,18 @@ describe('model-status-label', () => {
|
||||
expect(displayModelName('anthropic/claude-haiku-4-5-20251001')).toBe('Haiku 4 5')
|
||||
})
|
||||
|
||||
it('renders local GGUF ids as a clean name with a quant tag', () => {
|
||||
expect(modelDisplayParts('Qwen3.6-27B-UD-Q4_K_XL')).toEqual({ name: 'Qwen3.6 27B', tag: 'Q4' })
|
||||
expect(modelDisplayParts('Nemotron-3-Nano-30B-A3B-UD-Q4_K_XL')).toEqual({
|
||||
name: 'Nemotron 3 Nano 30B A3B',
|
||||
tag: 'Q4'
|
||||
})
|
||||
expect(modelDisplayParts('Qwen3-4B-Instruct-2507-UD-Q8_K_XL')).toEqual({ name: 'Qwen3 4B', tag: 'Q8' })
|
||||
expect(modelDisplayParts('some-model-Q6_K')).toEqual({ name: 'Some Model', tag: 'Q6' })
|
||||
// Cloud ids keep their existing behavior.
|
||||
expect(modelDisplayParts('anthropic/claude-opus-4.8-fast').tag).toBe('Fast')
|
||||
})
|
||||
|
||||
it('maps reasoning effort to compact labels', () => {
|
||||
expect(reasoningEffortLabel('high')).toBe('High')
|
||||
expect(reasoningEffortLabel('xhigh')).toBe('XHigh')
|
||||
|
||||
@@ -75,12 +75,26 @@ export function modelDisplayParts(model: string): { name: string; tag: string }
|
||||
let base = modelBaseId(model)
|
||||
let tag = ''
|
||||
|
||||
for (const [pattern, label] of VARIANT_TAGS) {
|
||||
if (pattern.test(base)) {
|
||||
tag = label
|
||||
base = base.replace(pattern, '')
|
||||
// Local GGUF ids carry a quant suffix (`…-UD-Q4_K_XL`, `…-Q8_0`). Render it
|
||||
// as a quiet tag — "Qwen3.6 27B · Q4" — never as part of the name. Without
|
||||
// this the composer pill reads raw quant soup ("Qwen3.6 27B UD Q4 K XL").
|
||||
const quant = base.match(/-(?:UD-)?(Q\d(?:_[A-Z0-9]+)*|IQ\d(?:_[A-Z0-9]+)*|F16|BF16)$/i)
|
||||
|
||||
break
|
||||
if (quant) {
|
||||
tag = quant[1].split('_')[0].toUpperCase()
|
||||
base = base.slice(0, -quant[0].length)
|
||||
// Instruct/chat markers are noise once the quant confirmed a local build.
|
||||
base = base.replace(/-(?:Instruct|Chat)(?:-\d{4})?$/i, '')
|
||||
}
|
||||
|
||||
if (!tag) {
|
||||
for (const [pattern, label] of VARIANT_TAGS) {
|
||||
if (pattern.test(base)) {
|
||||
tag = label
|
||||
base = base.replace(pattern, '')
|
||||
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
/**
|
||||
* The local-setup campaign's policy, tested as the pure decisions they are:
|
||||
* who qualifies (eligibility), when the bubble may return (the clock), and
|
||||
* that the campaign cannot corrupt the rotation's cursor.
|
||||
*/
|
||||
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { TIP_CATALOG } from '@/lib/tips/catalog'
|
||||
import { LOCAL_SETUP_RESHOW_MS, LOCAL_SETUP_TIP_ID, localSetupDue, localSetupEligible } from '@/lib/tips/local-cta'
|
||||
import type { LocalCatalogModel, LocalModelsStatus } from '@/types/hermes'
|
||||
|
||||
function status(overrides: Partial<LocalModelsStatus> = {}): LocalModelsStatus {
|
||||
return {
|
||||
active_model_id: null,
|
||||
loading: {},
|
||||
models: [],
|
||||
models_dir: '',
|
||||
placement: null,
|
||||
runtime_installed: false,
|
||||
server_running: false,
|
||||
...overrides
|
||||
} as LocalModelsStatus
|
||||
}
|
||||
|
||||
function fittingModel(): LocalCatalogModel {
|
||||
return { fits: true, id: 'qwen3.8-27b' } as LocalCatalogModel
|
||||
}
|
||||
|
||||
describe('localSetupEligible', () => {
|
||||
it('offers setup to a local backend with a fitting model and nothing staged', () => {
|
||||
expect(localSetupEligible('local', status(), [fittingModel()])).toBe(true)
|
||||
})
|
||||
|
||||
it('never promises local privacy on a remote or cloud backend', () => {
|
||||
for (const mode of ['remote', 'cloud', 'ssh', null]) {
|
||||
expect(localSetupEligible(mode, status(), [fittingModel()])).toBe(false)
|
||||
}
|
||||
})
|
||||
|
||||
it('stays quiet when no catalog model fits the machine', () => {
|
||||
expect(localSetupEligible('local', status(), [{ fits: false } as LocalCatalogModel])).toBe(false)
|
||||
expect(localSetupEligible('local', status(), [])).toBe(false)
|
||||
})
|
||||
|
||||
it('retires itself once the machine is set up', () => {
|
||||
const setUp = status({
|
||||
models: [{ id: 'qwen3.8-27b' }] as LocalModelsStatus['models'],
|
||||
runtime_installed: true
|
||||
})
|
||||
|
||||
expect(localSetupEligible('local', setUp, [fittingModel()])).toBe(false)
|
||||
})
|
||||
|
||||
it('still offers when the runtime exists but no model is staged', () => {
|
||||
expect(localSetupEligible('local', status({ runtime_installed: true }), [fittingModel()])).toBe(true)
|
||||
})
|
||||
|
||||
it('answers false while state is still loading', () => {
|
||||
expect(localSetupEligible('local', null, [fittingModel()])).toBe(false)
|
||||
expect(localSetupEligible('local', status(), null)).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('localSetupDue', () => {
|
||||
it('is due when never shown', () => {
|
||||
expect(localSetupDue(Date.now(), undefined)).toBe(true)
|
||||
})
|
||||
|
||||
it('holds for a week after an ignored showing, then returns', () => {
|
||||
const now = Date.now()
|
||||
|
||||
expect(localSetupDue(now, now - LOCAL_SETUP_RESHOW_MS + 1000)).toBe(false)
|
||||
expect(localSetupDue(now, now - LOCAL_SETUP_RESHOW_MS)).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('campaign identity', () => {
|
||||
it('lives outside the rotation catalog — the walk must never land on it', () => {
|
||||
// TipId already excludes the campaign id at the type level; this guards
|
||||
// the runtime data against someone re-adding it as a catalog entry.
|
||||
const ids: readonly string[] = TIP_CATALOG.map(def => def.id)
|
||||
|
||||
expect(ids).not.toContain(LOCAL_SETUP_TIP_ID)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,55 @@
|
||||
/**
|
||||
* The local-setup call to action — eligibility as data in, data out.
|
||||
*
|
||||
* The one campaign tip the app currently runs: a machine that could serve a
|
||||
* local model, on a backend that has none staged, gets one bubble on the
|
||||
* model pill saying so, with the button that starts the already-built
|
||||
* one-click setup. Everything here is a pure decision over fetched state so
|
||||
* the policy is testable without a DOM or a backend.
|
||||
*
|
||||
* Why this is not a rotation tip: the rotation teaches the app the user
|
||||
* already has, on a weeks-long walk. This tip is conditional (most machines
|
||||
* either qualify or don't, permanently), actionable (it carries a button),
|
||||
* and perishable (setting up local models — or ✕ — ends it forever). It
|
||||
* outranks the walk when live because "your GPU can do this" beats "the
|
||||
* model name is a button" every time both are true.
|
||||
*/
|
||||
|
||||
import type { LocalCatalogModel, LocalModelsStatus } from '@/types/hermes'
|
||||
|
||||
/** Retirement/shown-at ledger id. Not a `TipId` — the rotation never walks it. */
|
||||
export const LOCAL_SETUP_TIP_ID = 'local-setup'
|
||||
|
||||
/** A CTA ignored (timed out) may return, but on a much longer clock than the
|
||||
* rotation's: it is the same message twice, not a tour moving on. */
|
||||
export const LOCAL_SETUP_RESHOW_MS = 7 * 24 * 60 * 60_000
|
||||
|
||||
/**
|
||||
* Machines qualify when the catalog has a model that FITS (the backend's
|
||||
* physics check, same answer the pane's hero uses) and nothing is servable
|
||||
* yet — no runtime or no staged models. A set-up machine never qualifies, so
|
||||
* completing setup retires this tip without any bookkeeping.
|
||||
*
|
||||
* `connectionMode` must be 'local': on a remote backend (cloud resolves to
|
||||
* remote) the models would run on the far machine, and a bubble promising
|
||||
* "stays on your computer" would be promising someone else's computer.
|
||||
*/
|
||||
export function localSetupEligible(
|
||||
connectionMode: null | string,
|
||||
status: LocalModelsStatus | null,
|
||||
catalog: readonly LocalCatalogModel[] | null
|
||||
): boolean {
|
||||
if (connectionMode !== 'local' || !status || !catalog) {
|
||||
return false
|
||||
}
|
||||
|
||||
const needsSetup = !status.runtime_installed || status.models.length === 0
|
||||
|
||||
return needsSetup && catalog.some(model => model.fits)
|
||||
}
|
||||
|
||||
/** Due = never shown, or shown long enough ago that repeating it reads as a
|
||||
* reminder rather than a nag. Retirement is the caller's ledger, not ours. */
|
||||
export function localSetupDue(now: number, shownAt: number | undefined): boolean {
|
||||
return shownAt === undefined || now - shownAt >= LOCAL_SETUP_RESHOW_MS
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
describe('$localModelsEnabled', () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules()
|
||||
})
|
||||
|
||||
it('reads true when the preload bridge reports the --local launch flag', async () => {
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
value: { localModelsEnabled: true }
|
||||
})
|
||||
|
||||
const { $localModelsEnabled } = await import('./local-models-flag')
|
||||
|
||||
expect($localModelsEnabled.get()).toBe(true)
|
||||
})
|
||||
|
||||
it('defaults to false when the bridge omits the flag (older preload, web)', async () => {
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
value: {}
|
||||
})
|
||||
|
||||
const { $localModelsEnabled } = await import('./local-models-flag')
|
||||
|
||||
expect($localModelsEnabled.get()).toBe(false)
|
||||
})
|
||||
|
||||
it('defaults to false with no bridge at all', async () => {
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
value: undefined
|
||||
})
|
||||
|
||||
const { $localModelsEnabled } = await import('./local-models-flag')
|
||||
|
||||
expect($localModelsEnabled.get()).toBe(false)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,16 @@
|
||||
import { atom } from 'nanostores'
|
||||
|
||||
/**
|
||||
* Launch-flag gate for every local-models surface in the GUI.
|
||||
*
|
||||
* Local models ship on main behind `--local` (either `hermes desktop --local`
|
||||
* or the flag on Hermes.exe itself). The flag is strict: without it the GUI
|
||||
* shows no local-models surface at all, even on a machine where local models
|
||||
* are configured and running — the backend routes stay live, only the
|
||||
* desktop's presentation is gated. Read once from the preload bridge at
|
||||
* module load; a launch flag can't change mid-session, so nothing rewrites
|
||||
* it outside tests.
|
||||
*/
|
||||
export const $localModelsEnabled = atom<boolean>(
|
||||
typeof window !== 'undefined' && window.hermesDesktop?.localModelsEnabled === true
|
||||
)
|
||||
@@ -0,0 +1,169 @@
|
||||
import { atom } from 'nanostores'
|
||||
|
||||
import { getLocalModelsJobs, getLocalModelsStatus } from '@/hermes'
|
||||
import { translateNow } from '@/i18n'
|
||||
import { notify, notifyError } from '@/store/notifications'
|
||||
import type { LocalRuntimeJob } from '@/types/hermes'
|
||||
|
||||
// App-level tracker for local-runtime jobs (runtime installs, model
|
||||
// downloads). The AUTHORITY is the backend job registry — this store is a
|
||||
// cache of it (desktop guide: server truth is cached, not owned). Living at
|
||||
// the store layer, not in the settings pane, is what makes a download
|
||||
// survive the pane unmounting: anything can start a job, the poller follows
|
||||
// it to completion, and completion/failure notify app-wide exactly once.
|
||||
|
||||
export const $localRuntimeJobs = atom<readonly LocalRuntimeJob[]>([])
|
||||
|
||||
const POLL_ACTIVE_MS = 700
|
||||
let timer: null | number = null
|
||||
let polling = false
|
||||
// Jobs we've already toasted for, so a poll race can't double-notify.
|
||||
const settledNotified = new Set<string>()
|
||||
|
||||
function jobsEqual(a: readonly LocalRuntimeJob[], b: readonly LocalRuntimeJob[]) {
|
||||
if (a.length !== b.length) {
|
||||
return false
|
||||
}
|
||||
|
||||
return a.every((job, i) => {
|
||||
const other = b[i]
|
||||
|
||||
return (
|
||||
job.job_id === other.job_id &&
|
||||
job.status === other.status &&
|
||||
job.phase === other.phase &&
|
||||
job.done_bytes === other.done_bytes
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
function notifySettled(previous: readonly LocalRuntimeJob[], next: readonly LocalRuntimeJob[]) {
|
||||
const wasRunning = new Set(previous.filter(j => j.status === 'running').map(j => j.job_id))
|
||||
|
||||
for (const job of next) {
|
||||
if (job.status === 'running' || !wasRunning.has(job.job_id) || settledNotified.has(job.job_id)) {
|
||||
continue
|
||||
}
|
||||
|
||||
settledNotified.add(job.job_id)
|
||||
|
||||
if (job.status === 'done') {
|
||||
notify({
|
||||
durationMs: 6_000,
|
||||
kind: 'success',
|
||||
title: translateNow('settings.localModels.title'),
|
||||
message:
|
||||
job.kind === 'model-download'
|
||||
? translateNow('settings.localModels.downloadDoneToast', job.target)
|
||||
: job.kind === 'model-activate'
|
||||
? translateNow('settings.localModels.activateDoneToast', job.target)
|
||||
: job.kind === 'quickstart'
|
||||
? translateNow('settings.localModels.quickstartDoneToast', job.target)
|
||||
: translateNow('settings.localModels.installDoneToast')
|
||||
})
|
||||
} else {
|
||||
notifyError(
|
||||
new Error(job.error ?? job.detail ?? 'failed'),
|
||||
job.kind === 'model-download'
|
||||
? translateNow('settings.localModels.downloadFailed', job.target)
|
||||
: job.kind === 'model-activate'
|
||||
? translateNow('settings.localModels.activateFailed', job.target)
|
||||
: job.kind === 'quickstart'
|
||||
? translateNow('settings.localModels.quickstartFailed')
|
||||
: translateNow('settings.localModels.installFailed')
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function poll() {
|
||||
try {
|
||||
const { jobs } = await getLocalModelsJobs()
|
||||
const previous = $localRuntimeJobs.get()
|
||||
|
||||
if (!jobsEqual(previous, jobs)) {
|
||||
notifySettled(previous, jobs)
|
||||
$localRuntimeJobs.set(jobs)
|
||||
}
|
||||
} catch {
|
||||
// Backend unreachable — keep the last snapshot; the next poll retries.
|
||||
}
|
||||
|
||||
const anyRunning = $localRuntimeJobs.get().some(j => j.status === 'running')
|
||||
|
||||
if (anyRunning) {
|
||||
timer = window.setTimeout(() => void poll(), POLL_ACTIVE_MS)
|
||||
} else {
|
||||
polling = false
|
||||
timer = null
|
||||
}
|
||||
}
|
||||
|
||||
// Idempotent kick: start (or keep) the poll loop while work is in flight.
|
||||
// Call after starting a job AND on app boot (to rediscover work started
|
||||
// before a reload).
|
||||
export function watchLocalRuntimeJobs() {
|
||||
if (polling) {
|
||||
return
|
||||
}
|
||||
|
||||
polling = true
|
||||
|
||||
if (timer !== null) {
|
||||
window.clearTimeout(timer)
|
||||
}
|
||||
|
||||
void poll()
|
||||
}
|
||||
|
||||
// Selector: the running download job for a catalog model id, if any.
|
||||
export function runningDownloadFor(jobs: readonly LocalRuntimeJob[], modelId: string): LocalRuntimeJob | null {
|
||||
return jobs.find(j => j.kind === 'model-download' && j.status === 'running' && j.model_id === modelId) ?? null
|
||||
}
|
||||
|
||||
// Selector: every model on its way to the library right now — plain
|
||||
// downloads plus quickstart runs while they are still fetching bytes
|
||||
// (later quickstart phases mean the model is staged and activating).
|
||||
// The model picker renders these as disabled progress rows.
|
||||
const DOWNLOAD_PHASES = new Set(['starting', 'installing-runtime', 'downloading'])
|
||||
|
||||
export function runningModelDownloads(jobs: readonly LocalRuntimeJob[]): LocalRuntimeJob[] {
|
||||
return jobs.filter(
|
||||
j =>
|
||||
j.status === 'running' &&
|
||||
(j.kind === 'model-download' || (j.kind === 'quickstart' && DOWNLOAD_PHASES.has(j.phase)))
|
||||
)
|
||||
}
|
||||
|
||||
export function runningRuntimeInstall(jobs: readonly LocalRuntimeJob[]): LocalRuntimeJob | null {
|
||||
return jobs.find(j => j.kind === 'runtime-install' && j.status === 'running') ?? null
|
||||
}
|
||||
|
||||
// One engine-update toast per app session: checked at boot (after the
|
||||
// gateway is ready), only when the user runs the local engine. The
|
||||
// download itself is always a button click in Local Models — this is a
|
||||
// pointer, not an installer.
|
||||
let updateNotified = false
|
||||
|
||||
export async function checkLocalRuntimeUpdate() {
|
||||
if (updateNotified) {
|
||||
return
|
||||
}
|
||||
|
||||
try {
|
||||
const status = await getLocalModelsStatus()
|
||||
|
||||
if (status.enabled && status.update_available) {
|
||||
updateNotified = true
|
||||
notify({
|
||||
durationMs: 10_000,
|
||||
kind: 'info',
|
||||
title: translateNow('settings.localModels.title'),
|
||||
message: translateNow('settings.localModels.updateToast', status.configured_tag)
|
||||
})
|
||||
}
|
||||
} catch {
|
||||
// Backend without the endpoint (older runtime) or transient failure —
|
||||
// silently skip; the pane still shows the update row when opened.
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { parseModelLoadWait, providerWaitText } from './provider-wait'
|
||||
|
||||
// The load-notice string is minted by the backend
|
||||
// (agent/chat_completion_helpers._managed_local_load_notice) and parsed
|
||||
// here — these tests pin the desktop side of that cross-language contract
|
||||
// (the backend pins its side in tests/hermes_cli/test_load_progress.py).
|
||||
describe('providerWaitText', () => {
|
||||
it('accepts the managed-local load frame', () => {
|
||||
const frame = '⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 42% (responses start once the model is loaded)'
|
||||
|
||||
expect(providerWaitText(frame)).toBe(frame)
|
||||
})
|
||||
|
||||
it('still accepts classic wait frames and rejects spinner noise', () => {
|
||||
expect(providerWaitText('⏳ waiting on local-model — 30s with no output yet')).not.toBe('')
|
||||
expect(providerWaitText('◉_◉ cogitating...')).toBe('')
|
||||
})
|
||||
})
|
||||
|
||||
describe('parseModelLoadWait', () => {
|
||||
it('extracts model and percent from a load frame', () => {
|
||||
expect(
|
||||
parseModelLoadWait('⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 42% (responses start once the model is loaded)')
|
||||
).toEqual({ kind: 'load', model: 'Qwen3.6-35B-A3B-UD-Q4_K_M', percent: 42 })
|
||||
})
|
||||
|
||||
it('extracts the percent from a prefill frame', () => {
|
||||
expect(parseModelLoadWait('⚙ processing prompt — 31%')).toEqual({
|
||||
kind: 'prefill',
|
||||
model: '',
|
||||
percent: 31
|
||||
})
|
||||
})
|
||||
|
||||
it('parses a percentless prefill frame with a null percent (no fake bar)', () => {
|
||||
expect(parseModelLoadWait('⚙ processing prompt')).toEqual({
|
||||
kind: 'prefill',
|
||||
model: '',
|
||||
percent: null
|
||||
})
|
||||
})
|
||||
|
||||
it('returns null for every other wait frame', () => {
|
||||
expect(parseModelLoadWait('⏳ waiting on qwen — 30s with no output yet')).toBeNull()
|
||||
expect(parseModelLoadWait('⚠ no output from provider for 900s — reconnecting...')).toBeNull()
|
||||
expect(parseModelLoadWait('')).toBeNull()
|
||||
})
|
||||
|
||||
it('clamps out-of-range percents', () => {
|
||||
expect(parseModelLoadWait('⏳ loading m into memory — 999%')?.percent).toBe(100)
|
||||
})
|
||||
})
|
||||
|
||||
describe('providerWaitText accepts prefill frames', () => {
|
||||
it('passes the ⚙ processing-prompt frame through', () => {
|
||||
const frame = '⚙ processing prompt — 31%'
|
||||
|
||||
expect(providerWaitText(frame)).toBe(frame)
|
||||
})
|
||||
})
|
||||
@@ -45,5 +45,42 @@ export function clearAllProviderWaits(): void {
|
||||
export function providerWaitText(text: string): string {
|
||||
const value = text.trim()
|
||||
|
||||
return /^(?:⏳|⚠|↻)\s*(?:waiting on|no (?:output|response)|model returned)/i.test(value) ? value : ''
|
||||
return /^(?:⏳|⚠|↻|⚙)\s*(?:waiting on|loading|processing prompt|no (?:output|response)|model returned)/i.test(value)
|
||||
? value
|
||||
: ''
|
||||
}
|
||||
|
||||
/** Parse a managed-local progress frame into bar-renderable parts, or null
|
||||
* for every other wait frame. Two shapes, both minted by the backend's
|
||||
* _managed_local_load_notice (the percents are real — per-tensor load
|
||||
* callback / live prefill counter — so a determinate bar is honest):
|
||||
* "⏳ loading <model> into memory — 43% …" -> kind: 'load'
|
||||
* "⚙ processing prompt — 31%" -> kind: 'prefill'
|
||||
* A percentless prefill frame ("⚙ processing prompt") parses with
|
||||
* percent: null and renders as label-only, no fake bar. */
|
||||
export function parseModelLoadWait(
|
||||
text: string
|
||||
): null | { kind: 'load' | 'prefill'; model: string; percent: null | number } {
|
||||
const value = text.trim()
|
||||
const load = /^⏳\s*loading\s+(.+?)\s+into memory\s+—\s+(\d{1,3})%/i.exec(value)
|
||||
|
||||
if (load) {
|
||||
return {
|
||||
kind: 'load',
|
||||
model: load[1],
|
||||
percent: Math.max(0, Math.min(100, Number(load[2])))
|
||||
}
|
||||
}
|
||||
|
||||
const prefill = /^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?/i.exec(value)
|
||||
|
||||
if (prefill) {
|
||||
return {
|
||||
kind: 'prefill',
|
||||
model: '',
|
||||
percent: prefill[1] === undefined ? null : Math.max(0, Math.min(100, Number(prefill[1])))
|
||||
}
|
||||
}
|
||||
|
||||
return null
|
||||
}
|
||||
|
||||
@@ -25,6 +25,7 @@ export const STATUSBAR_HIDDEN_BY_DEFAULT: readonly string[] = [
|
||||
'cron',
|
||||
'running-timer',
|
||||
'session-timer',
|
||||
'system-resources',
|
||||
'terminal',
|
||||
'webhooks'
|
||||
]
|
||||
|
||||
@@ -26,7 +26,7 @@
|
||||
import { atom } from 'nanostores'
|
||||
|
||||
import { Codecs, persistentAtom } from '@/lib/persisted'
|
||||
import type { TipSide } from '@/lib/tips/catalog'
|
||||
import { TIP_CATALOG, type TipSide } from '@/lib/tips/catalog'
|
||||
import { mirrorDisplayToggle } from '@/store/display-toggles'
|
||||
|
||||
/** Hours, not minutes. The catalog is ten tips and it should take weeks. */
|
||||
@@ -34,6 +34,11 @@ const COOLDOWN_MS = 6 * 60 * 60_000
|
||||
|
||||
/** A tip as the bubble needs it: resolved copy, resolved anchor. */
|
||||
export interface ActiveTip {
|
||||
/** A call to action: one button under the text. What separates a campaign
|
||||
* tip from the rotation's — the rotation teaches, this one offers to DO
|
||||
* the thing, and the button is the only path (an ambient bubble must
|
||||
* never make its whole face clickable). Clicking closes the tip. */
|
||||
action?: { label: string; onSelect: () => void }
|
||||
/** Keybind action id whose live combo the bubble prints. */
|
||||
keybind?: string
|
||||
side: TipSide
|
||||
@@ -65,6 +70,23 @@ export const $activeTip = atom<ActiveTip | null>(null)
|
||||
// model's schema entirely rather than staying on offer and being dropped.
|
||||
mirrorDisplayToggle('display.in_app_tips', ENABLED_KEY, $tipsEnabled)
|
||||
|
||||
/** When each campaign tip (id outside the rotation catalog) last showed.
|
||||
* Campaign tips re-offer on their own long clock instead of walking on;
|
||||
* `$retiredTips` still owns the hard ✕. */
|
||||
export const $tipShownAt = persistentAtom<Record<string, number>>(
|
||||
'hermes.desktop.tips.shownAt.v1',
|
||||
{},
|
||||
Codecs.json(value => {
|
||||
if (!value || typeof value !== 'object' || Array.isArray(value)) {
|
||||
return {}
|
||||
}
|
||||
|
||||
return Object.fromEntries(
|
||||
Object.entries(value).filter((entry): entry is [string, number] => typeof entry[1] === 'number')
|
||||
)
|
||||
})
|
||||
)
|
||||
|
||||
export function setTipsEnabled(enabled: boolean): void {
|
||||
if (!enabled) {
|
||||
// Including whichever one is up: the switch is answering a bubble on
|
||||
@@ -85,7 +107,14 @@ export function resetTips(): void {
|
||||
/** Put a tip on screen, replacing whatever was there. */
|
||||
export function showTip(tip: ActiveTip): void {
|
||||
if (tip.tipId) {
|
||||
$lastTipId.set(tip.tipId)
|
||||
// The cursor belongs to the rotation's walk. A campaign tip (an id the
|
||||
// catalog doesn't hold) records when it showed but must not move the
|
||||
// cursor — nextTip treats an unknown id as "start over at the top".
|
||||
if (TIP_CATALOG.some(def => def.id === tip.tipId)) {
|
||||
$lastTipId.set(tip.tipId)
|
||||
}
|
||||
|
||||
$tipShownAt.set({ ...$tipShownAt.get(), [tip.tipId]: Date.now() })
|
||||
}
|
||||
|
||||
// Any tip starts the cooldown, an agent's included: whoever just pointed at
|
||||
|
||||
@@ -1237,6 +1237,97 @@ export interface StatusResponse {
|
||||
version: string
|
||||
}
|
||||
|
||||
// ── Managed local runtime (llama.cpp) ──────────────────────────
|
||||
|
||||
export interface LocalModelPlacement {
|
||||
window?: number
|
||||
window_label?: string
|
||||
spilled?: boolean
|
||||
granted_window?: number
|
||||
granted_window_label?: string
|
||||
}
|
||||
|
||||
export interface LocalModelLoadProgress {
|
||||
stage: string
|
||||
value: number
|
||||
percent: number
|
||||
}
|
||||
|
||||
export interface LocalModelsStatus {
|
||||
enabled: boolean
|
||||
tag: string
|
||||
configured_tag: string
|
||||
update_available: boolean
|
||||
runtime_installed: boolean
|
||||
runtime_backend: string | null
|
||||
server_running: boolean
|
||||
server_base_url: string | null
|
||||
active_model_id: string | null
|
||||
loaded_models: Record<string, string>
|
||||
/** Models loading into memory right now: real per-tensor load percent. */
|
||||
loading?: Record<string, LocalModelLoadProgress>
|
||||
placement?: Record<string, LocalModelPlacement>
|
||||
models: { id: string; size_bytes: number; size_label: string }[]
|
||||
models_dir: string
|
||||
}
|
||||
|
||||
export interface LocalHardware {
|
||||
uma: boolean
|
||||
vram_total_bytes: number
|
||||
vram_usable_bytes: number
|
||||
ram_total_bytes: number
|
||||
ram_available_bytes: number
|
||||
vram_label: string
|
||||
gpu_name: string | null
|
||||
gpu_util_percent: number | null
|
||||
vram_used_bytes: number | null
|
||||
}
|
||||
|
||||
export interface LocalCatalogModel {
|
||||
id: string
|
||||
display_name: string
|
||||
description: string
|
||||
size_bytes: number
|
||||
size_label: string
|
||||
native_context: number
|
||||
native_context_label: string
|
||||
recommended: boolean
|
||||
/** Why the resolver picked this entry (recommended rows only):
|
||||
* best-quality-resident | speed-gated-quality | fastest-resident |
|
||||
* least-painful-spilled. Renders as the Recommended badge's tooltip. */
|
||||
recommended_reason?: string | null
|
||||
downloaded: boolean
|
||||
downloaded_model_id?: string | null
|
||||
downloaded_quant?: string | null
|
||||
mtp: boolean
|
||||
vision?: boolean
|
||||
fits: boolean
|
||||
fit_summary: string
|
||||
fit_detail?: string
|
||||
model_id?: string
|
||||
quant?: string
|
||||
quant_reason?: string
|
||||
quant_validated?: boolean
|
||||
variant_count?: number
|
||||
start_window?: number
|
||||
start_window_label?: string
|
||||
spilled?: boolean
|
||||
}
|
||||
|
||||
export interface LocalRuntimeJob {
|
||||
job_id: string
|
||||
kind: 'model-activate' | 'model-download' | 'quickstart' | 'runtime-install'
|
||||
target: string
|
||||
model_id: string | null
|
||||
status: 'running' | 'done' | 'error'
|
||||
phase: string
|
||||
detail: string
|
||||
total_bytes: number | null
|
||||
done_bytes: number
|
||||
percent?: number
|
||||
error: string | null
|
||||
}
|
||||
|
||||
export interface ActionResponse {
|
||||
name: string
|
||||
ok: boolean
|
||||
|
||||
+91
-44
@@ -107,6 +107,34 @@ _EXCLUDED_DIRS = {
|
||||
".ruff_cache",
|
||||
}
|
||||
|
||||
# Hermes-managed runtime downloads that only exist at the top of a profile
|
||||
# home: local GGUF models, llama.cpp runtime binaries, and the managed Node
|
||||
# installation. All of them are re-downloaded on demand (model catalog,
|
||||
# runtime bootstrap, node installer) and routinely reach tens to hundreds of
|
||||
# GB, so zipping them turns a backup into an hours-long compress of
|
||||
# incompressible weights (the "backup stuck at N files" symptom). Matched
|
||||
# ONLY at the root of HERMES_HOME and at ``profiles/<name>/`` — a deeper
|
||||
# directory that happens to share one of these names (a skill's ``models/``,
|
||||
# a user checkout) is user data and stays in the backup.
|
||||
_EXCLUDED_ROOT_DIRS = {
|
||||
"models",
|
||||
"runtimes",
|
||||
"node",
|
||||
}
|
||||
|
||||
|
||||
def _in_excluded_root_dir(rel_path: Path) -> bool:
|
||||
"""True when *rel_path* (relative to HERMES_HOME) is, or sits inside, a
|
||||
Hermes-managed runtime tree at the top of a profile home."""
|
||||
parts = rel_path.parts
|
||||
if not parts:
|
||||
return False
|
||||
if parts[0] in _EXCLUDED_ROOT_DIRS:
|
||||
return True
|
||||
# Named profiles are profile homes too: profiles/<name>/models etc.
|
||||
return len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS
|
||||
|
||||
|
||||
# File-name suffixes to skip
|
||||
_EXCLUDED_SUFFIXES = (
|
||||
".pyc",
|
||||
@@ -128,6 +156,16 @@ _EXCLUDED_NAMES = {
|
||||
"cron.pid",
|
||||
}
|
||||
|
||||
# File-name prefixes to skip. The desktop updater's pre-flight drops
|
||||
# ``state.db.pre-update-emergency-<timestamp>.bak`` at the HERMES_HOME root
|
||||
# (apps/desktop/electron/main.ts preflightStateDb) — a backup artifact in
|
||||
# the same class as ``backups/`` and ``state-snapshots/``, so a full backup
|
||||
# must not re-ship it. Matched by prefix because the name carries a
|
||||
# timestamp; a plain ``.bak`` suffix rule would drop user files.
|
||||
_EXCLUDED_PREFIXES = (
|
||||
"state.db.pre-update-emergency-",
|
||||
)
|
||||
|
||||
# File names that ``hermes import`` must never overwrite, matched by basename so
|
||||
# they're caught for the root profile (``gateway_state.json``) and for named
|
||||
# profiles alike (``profiles/<name>/gateway_state.json``).
|
||||
@@ -335,6 +373,9 @@ def _should_exclude(rel_path: Path) -> bool:
|
||||
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
|
||||
parts = rel_path.parts
|
||||
|
||||
if _in_excluded_root_dir(rel_path):
|
||||
return True
|
||||
|
||||
for part in parts:
|
||||
if part not in _EXCLUDED_DIRS:
|
||||
continue
|
||||
@@ -350,6 +391,9 @@ def _should_exclude(rel_path: Path) -> bool:
|
||||
if name in _EXCLUDED_NAMES:
|
||||
return True
|
||||
|
||||
if name.startswith(_EXCLUDED_PREFIXES):
|
||||
return True
|
||||
|
||||
if name.endswith(_EXCLUDED_SUFFIXES):
|
||||
return True
|
||||
|
||||
@@ -372,6 +416,48 @@ def _should_skip_backup_file(abs_path: Path, rel_path: Path, out_path: Path) ->
|
||||
return False
|
||||
|
||||
|
||||
def _iter_backup_files(
|
||||
hermes_root: Path,
|
||||
out_path: Path,
|
||||
skipped_dirs: Optional[set] = None,
|
||||
):
|
||||
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
|
||||
|
||||
The one owner of the backup walk policy: directory pruning (so os.walk
|
||||
never descends a multi-GB excluded tree), the root-only ``hermes-agent``
|
||||
carve-out, profile-home-root runtime trees, and the per-file exclusion
|
||||
rules — shared by the manual ``hermes backup`` path and the automatic
|
||||
pre-update/pre-migration path so the two can never drift.
|
||||
|
||||
``skipped_dirs``, when given, collects pruned directories (root-relative,
|
||||
as strings) for the end-of-run summary.
|
||||
"""
|
||||
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
||||
rel_dir = Path(dirpath).relative_to(hermes_root)
|
||||
|
||||
# ``hermes-agent`` is only pruned at the root level; nested dirs
|
||||
# with the same name (e.g. in skills/) must be preserved. Managed
|
||||
# runtime trees (models/, runtimes/, node/) are pruned only at a
|
||||
# profile-home root — see _EXCLUDED_ROOT_DIRS.
|
||||
is_root = rel_dir == Path(".")
|
||||
orig_dirnames = dirnames[:]
|
||||
dirnames[:] = [
|
||||
d for d in dirnames
|
||||
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
|
||||
and not _in_excluded_root_dir(rel_dir / d)
|
||||
]
|
||||
if skipped_dirs is not None:
|
||||
for removed in set(orig_dirnames) - set(dirnames):
|
||||
skipped_dirs.add(str(rel_dir / removed))
|
||||
|
||||
for fname in filenames:
|
||||
rel = rel_dir / fname
|
||||
fpath = hermes_root / rel
|
||||
if _should_skip_backup_file(fpath, rel, out_path):
|
||||
continue
|
||||
yield fpath, rel
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# SQLite safe copy
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -869,33 +955,10 @@ def _run_backup_locked(args, hermes_root: Path) -> None:
|
||||
scan_started = time.monotonic()
|
||||
logger.info("backup phase=scan status=started")
|
||||
print(f"Scanning {display_hermes_home()} ...")
|
||||
files_to_add: list[tuple[Path, Path]] = [] # (absolute, relative)
|
||||
skipped_dirs = set()
|
||||
|
||||
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
||||
dp = Path(dirpath)
|
||||
rel_dir = dp.relative_to(hermes_root)
|
||||
|
||||
# Prune excluded directories in-place so os.walk doesn't descend
|
||||
# ``hermes-agent`` is only pruned at the root level; nested dirs
|
||||
# with the same name (e.g. in skills/) must be preserved.
|
||||
is_root = rel_dir == Path(".")
|
||||
orig_dirnames = dirnames[:]
|
||||
dirnames[:] = [
|
||||
d for d in dirnames
|
||||
if d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root)
|
||||
]
|
||||
for removed in set(orig_dirnames) - set(dirnames):
|
||||
skipped_dirs.add(str(rel_dir / removed))
|
||||
|
||||
for fname in filenames:
|
||||
fpath = dp / fname
|
||||
rel = fpath.relative_to(hermes_root)
|
||||
|
||||
if _should_skip_backup_file(fpath, rel, out_path):
|
||||
continue
|
||||
|
||||
files_to_add.append((fpath, rel))
|
||||
skipped_dirs: set = set()
|
||||
files_to_add: list[tuple[Path, Path]] = list(
|
||||
_iter_backup_files(hermes_root, out_path, skipped_dirs)
|
||||
)
|
||||
|
||||
# External memory-provider state (e.g. ~/.honcho, ~/.hindsight) lives
|
||||
# outside HERMES_HOME, so the walk above never sees it. Ask the active
|
||||
@@ -2328,24 +2391,8 @@ def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional
|
||||
"""
|
||||
scan_started = time.monotonic()
|
||||
logger.info("automatic backup phase=scan status=started")
|
||||
files_to_add: list[tuple[Path, Path]] = []
|
||||
try:
|
||||
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
||||
dp = Path(dirpath)
|
||||
# Prune excluded directories in-place so os.walk doesn't descend
|
||||
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
|
||||
|
||||
for fname in filenames:
|
||||
fpath = dp / fname
|
||||
try:
|
||||
rel = fpath.relative_to(hermes_root)
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
if _should_skip_backup_file(fpath, rel, out_path):
|
||||
continue
|
||||
|
||||
files_to_add.append((fpath, rel))
|
||||
files_to_add = list(_iter_backup_files(hermes_root, out_path))
|
||||
except OSError as exc:
|
||||
logger.warning("Full-zip backup: walk failed: %s", exc)
|
||||
return None
|
||||
|
||||
@@ -3013,6 +3013,7 @@ class CLICommandsMixin:
|
||||
review_memory=True,
|
||||
review_skills=review_skills,
|
||||
focus=focus or None,
|
||||
explicit=True,
|
||||
)
|
||||
except Exception as exc:
|
||||
_cprint(f" /refine failed to start: {exc}")
|
||||
|
||||
@@ -4012,6 +4012,29 @@ DEFAULT_CONFIG = {
|
||||
"region": "global",
|
||||
},
|
||||
|
||||
# Managed llama.cpp local runtime (see docs: user-guide/local-models).
|
||||
# Hermes downloads official llama.cpp release binaries, then spawns and
|
||||
# supervises one llama-server in router mode. Context sizing is policy,
|
||||
# not preference: there are deliberately no context/VRAM knobs here.
|
||||
"local_runtime": {
|
||||
# Master switch for the managed runtime. Off = detection-only
|
||||
# (Hermes still finds an external llama-server you run yourself).
|
||||
"enabled": False,
|
||||
# Pinned llama.cpp release tag (rolling bNNNN). Bumped by Hermes
|
||||
# releases after the validation suite re-runs, not tracked live.
|
||||
"tag": "b10679",
|
||||
# Inference backend: auto = CUDA on NVIDIA, Metal on macOS, Vulkan on
|
||||
# other GPUs, else CPU. Explicit values: cuda|metal|vulkan|hip|cpu.
|
||||
"backend": "auto",
|
||||
# Router process: how many models may be resident at once.
|
||||
"models_max": 4,
|
||||
# Port for the managed server. 0 = pick a free port at spawn.
|
||||
"port": 0,
|
||||
# Extra ports detection probes for an external llama-server, in
|
||||
# addition to the default 8080.
|
||||
"detect_ports": [],
|
||||
},
|
||||
|
||||
# Config schema version - bump this when adding new required fields
|
||||
"_config_version": 39,
|
||||
}
|
||||
|
||||
+93
-3
@@ -206,6 +206,31 @@ def build_models_payload(
|
||||
excluded_providers=ctx.excluded_providers or [],
|
||||
)
|
||||
|
||||
# Managed local runtime: staged GGUFs are selectable like any provider's
|
||||
# models. list_authenticated_providers can't know about them (no
|
||||
# credential, no custom_providers entry — the credential is
|
||||
# reachability), so inject the row here where every picker surface
|
||||
# inherits it. Present whenever models are staged; picking one routes
|
||||
# through the llamacpp alias -> managed/detected server resolution.
|
||||
local_row = _local_runtime_row(ctx)
|
||||
if local_row is not None:
|
||||
rows = [r for r in rows if str(r.get("slug", "")).lower() != "llamacpp"]
|
||||
rows.append(local_row)
|
||||
# A live session on the managed server reports provider "custom"
|
||||
# (the resolution seam's generic label for a raw base_url), which
|
||||
# would otherwise materialize a duplicate "Custom endpoint" row
|
||||
# carrying the same staged models and stealing the checkmark. The
|
||||
# Local row owns the managed server's identity — drop custom rows
|
||||
# that point at the managed endpoint.
|
||||
if local_row.get("is_current"):
|
||||
def _is_managed_custom(row: dict) -> bool:
|
||||
if str(row.get("slug", "")).lower() != "custom":
|
||||
return False
|
||||
models = {str(m) for m in (row.get("models") or [])}
|
||||
return bool(models) and models <= set(local_row["models"])
|
||||
|
||||
rows = [r for r in rows if not _is_managed_custom(r)]
|
||||
|
||||
moa_row = _moa_provider_row(ctx.current_provider)
|
||||
if moa_row is not None:
|
||||
rows = [moa_row] + [r for r in rows if str(r.get("slug", "")).lower() != "moa"]
|
||||
@@ -217,9 +242,15 @@ def build_models_payload(
|
||||
# has lost its credential, list_authenticated_providers() omits it;
|
||||
# keep that one row visible so the UI can show the saved selection and
|
||||
# a re-auth affordance instead of appearing to jump to another provider.
|
||||
rows = list(rows) + _append_unconfigured_rows(
|
||||
rows, ctx, current_only=True
|
||||
)
|
||||
# Exception: a "custom" current whose endpoint is the managed local
|
||||
# server is already represented (with the checkmark) by the Local row
|
||||
# — the skeleton would resurrect the duplicate the dedup above removed.
|
||||
_local_owns_current = bool(local_row and local_row.get("is_current")
|
||||
and (ctx.current_provider or "").lower() == "custom")
|
||||
if not _local_owns_current:
|
||||
rows = list(rows) + _append_unconfigured_rows(
|
||||
rows, ctx, current_only=True
|
||||
)
|
||||
|
||||
# --- Deduplicate: remove models from aggregators that overlap with
|
||||
# user-defined providers. When a local proxy (e.g. litellm-proxy)
|
||||
@@ -735,6 +766,15 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
|
||||
if current_slug and slug == current_slug:
|
||||
kept.append(row)
|
||||
continue
|
||||
if row.get("source") == "local-runtime":
|
||||
# Managed local models are explicit configuration by existence:
|
||||
# the user downloaded gigabytes into the machine-scoped models
|
||||
# dir. There is deliberately no config credential to find
|
||||
# (credential is reachability), so without this clause the row
|
||||
# only survives on the profile where Use was last clicked —
|
||||
# every other profile loses local models from its picker.
|
||||
kept.append(row)
|
||||
continue
|
||||
if slug == "moa":
|
||||
# MoA is a virtual routing mode, not an independently configured
|
||||
# provider. Hide it from explicit-only pickers unless it is the
|
||||
@@ -986,6 +1026,56 @@ def _apply_pricing(
|
||||
row["unavailable_models"] = []
|
||||
|
||||
|
||||
def _local_runtime_row(ctx: "ConfigContext") -> dict | None:
|
||||
"""Build the ``llamacpp`` provider row from staged local models.
|
||||
|
||||
Present whenever GGUFs are staged in the managed models directory —
|
||||
downloaded models must be selectable even before the server is running
|
||||
(selection starts it via the runtime_provider seam / activate flow).
|
||||
Returns ``None`` when nothing is staged.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.local_runtime.bootstrap import staged_model_ids
|
||||
|
||||
staged = staged_model_ids()
|
||||
if not staged:
|
||||
return None
|
||||
current = (ctx.current_provider or "").strip().lower() in (
|
||||
"llamacpp", "llama.cpp", "llama-cpp")
|
||||
if not current:
|
||||
# A LIVE session on the managed server reports provider "custom"
|
||||
# (the resolution seam's label) with the managed base_url. Match
|
||||
# on the endpoint so the picker still marks this row current —
|
||||
# otherwise the session the user is chatting in shows no
|
||||
# selection.
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
managed = _state_endpoint()
|
||||
current = bool(
|
||||
managed
|
||||
and (ctx.current_base_url or "").strip().rstrip("/")
|
||||
== managed["base_url"].rstrip("/"))
|
||||
except Exception:
|
||||
current = False
|
||||
return {
|
||||
"slug": "llamacpp",
|
||||
# Bare "Local" everywhere user-facing: the engine name is an
|
||||
# implementation detail (the pane brands this "Local models").
|
||||
"name": "Local",
|
||||
"is_current": current,
|
||||
"is_user_defined": False,
|
||||
"models": staged,
|
||||
"total_models": len(staged),
|
||||
"source": "local-runtime",
|
||||
"authenticated": True, # the credential is reachability
|
||||
"auth_type": "local",
|
||||
"warning": None,
|
||||
}
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _moa_provider_row(current_provider: str = "") -> dict | None:
|
||||
"""Build the virtual ``moa`` provider row for model pickers.
|
||||
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Managed llama.cpp runtime.
|
||||
|
||||
Hermes downloads, verifies, supervises, and updates one llama-server, and
|
||||
decides per machine which model build and context window to run. Key
|
||||
modules:
|
||||
|
||||
- ``binaries`` — resolve/download/verify official llama.cpp release zips
|
||||
into ``$HERMES_HOME/runtimes/llamacpp/<tag>/``.
|
||||
- ``supervisor``— spawn and supervise one llama-server in router mode;
|
||||
readiness is a touch generation, never health-200 alone.
|
||||
- ``detect`` — find an already-running llama-server (external or ours).
|
||||
- ``estimator`` / ``context_policy`` / ``growth`` — price context memory
|
||||
per architecture and run the window ladder (zero-spill start, grow
|
||||
toward native max, compress only at the top).
|
||||
- ``catalog`` / ``presets`` — the curated model list and the per-model
|
||||
launch flags that carry policy decisions to the router.
|
||||
|
||||
Everything is driven by the ``local_runtime`` section of config.yaml.
|
||||
"""
|
||||
|
||||
from hermes_cli.local_runtime.binaries import ( # noqa: F401
|
||||
BinaryResolutionError,
|
||||
ensure_runtime_installed,
|
||||
resolve_assets,
|
||||
select_backend,
|
||||
)
|
||||
from hermes_cli.local_runtime.bootstrap import ( # noqa: F401
|
||||
ensure_local_runtime,
|
||||
shutdown_local_runtime,
|
||||
)
|
||||
from hermes_cli.local_runtime.context_policy import ( # noqa: F401
|
||||
FLOOR,
|
||||
growth_decision,
|
||||
initial_window,
|
||||
ladder,
|
||||
launch_args,
|
||||
)
|
||||
from hermes_cli.local_runtime.growth import ( # noqa: F401
|
||||
clear_window_override,
|
||||
load_window_overrides,
|
||||
maybe_grow_window,
|
||||
save_window_override,
|
||||
)
|
||||
from hermes_cli.local_runtime.detect import detect_server # noqa: F401
|
||||
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint # noqa: F401
|
||||
from hermes_cli.local_runtime.estimator import ( # noqa: F401
|
||||
HardwareBudget,
|
||||
ctx_bytes,
|
||||
physics_check,
|
||||
profile_from_gguf,
|
||||
)
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header # noqa: F401
|
||||
from hermes_cli.local_runtime.hardware import probe_budget # noqa: F401
|
||||
from hermes_cli.local_runtime.presets import generate_presets # noqa: F401
|
||||
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor # noqa: F401
|
||||
@@ -0,0 +1,351 @@
|
||||
"""Binary acquisition for the managed llama.cpp runtime.
|
||||
|
||||
llama.cpp publishes per-tag assets (rolling ``bNNNN`` tags, no semver).
|
||||
Backends are dlopen'd plugins, so a runtime = CPU/base zip + backend zip
|
||||
extracted into one directory, plus the cudart runtime zip on Windows CUDA
|
||||
(end users have no CUDA toolkit). We pin the tag in config, sha256-verify
|
||||
every download, and keep the previous tag for rollback (N-1).
|
||||
|
||||
Layout: ``$HERMES_HOME/runtimes/llamacpp/<tag>/<backend>/<binaries>``
|
||||
with a ``manifest.json`` recording zips, sha256s, and the verified
|
||||
llama-server version string.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import platform
|
||||
import shutil
|
||||
import subprocess
|
||||
import urllib.request
|
||||
import zipfile
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
RELEASE_URL = "https://github.com/ggml-org/llama.cpp/releases/download/{tag}/{asset}"
|
||||
|
||||
# Windows CUDA zips ship per CUDA major; the runtime zip must be paired with
|
||||
# its cudart zip so end users need no toolkit. 13.3 verified on 13.1 and
|
||||
# 13.2 drivers.
|
||||
_WIN_CUDA_VERSION = "13.3"
|
||||
# arm64 Windows CUDA prebuilts landed upstream (~b1036x) on CUDA 13.4 —
|
||||
# verified against live asset lists (b10362, b10630, b10679). Tags at or before
|
||||
# b10290 don't have them; resolution succeeds and the download 404s
|
||||
# honestly on such tags, which only arises if a user pins backward.
|
||||
_WIN_CUDA_VERSION_ARM64 = "13.4"
|
||||
|
||||
|
||||
# Fallback when the config section is missing entirely (deep-merge normally
|
||||
# guarantees the key). Single source: DEFAULT_CONFIG owns the shipped tag.
|
||||
def default_tag() -> str:
|
||||
from hermes_cli.config_defaults import DEFAULT_CONFIG
|
||||
|
||||
return DEFAULT_CONFIG["local_runtime"]["tag"]
|
||||
|
||||
|
||||
class BinaryResolutionError(RuntimeError):
|
||||
"""No usable asset combination for this platform/backend."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetPlan:
|
||||
"""The exact zips one runtime install needs, in extraction order."""
|
||||
|
||||
tag: str
|
||||
backend: str # cuda | metal | vulkan | hip | cpu
|
||||
assets: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def install_dir(self) -> Path:
|
||||
return runtimes_root() / self.tag / self.backend
|
||||
|
||||
|
||||
def runtimes_root() -> Path:
|
||||
"""Machine-scoped, deliberately NOT profile-scoped. Engine binaries,
|
||||
presets, and server state describe this machine's hardware and its one
|
||||
managed server (stable port) — a second profile re-downloading the
|
||||
engine or fighting over the port would be the bug. Profile-scoped
|
||||
things (which model is the default, enabled) live in each profile's
|
||||
config.yaml as ever."""
|
||||
from hermes_constants import get_default_hermes_root
|
||||
|
||||
return get_default_hermes_root() / "runtimes" / "llamacpp"
|
||||
|
||||
|
||||
def installed_tags() -> list[str]:
|
||||
"""Tags with a verified install (manifest carries verified_version),
|
||||
newest first by release number. The boot ladder and the update check
|
||||
both read installed-ness from here — one resolver, every caller."""
|
||||
root = runtimes_root()
|
||||
if not root.exists():
|
||||
return []
|
||||
found: list[str] = []
|
||||
for entry in root.iterdir():
|
||||
if not entry.is_dir() or entry.name == "downloads":
|
||||
continue
|
||||
for manifest in entry.glob("*/manifest.json"):
|
||||
try:
|
||||
if json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
|
||||
found.append(entry.name)
|
||||
break
|
||||
except (json.JSONDecodeError, OSError):
|
||||
continue
|
||||
|
||||
def _release_number(tag: str) -> int:
|
||||
digits = "".join(ch for ch in tag if ch.isdigit())
|
||||
return int(digits) if digits else 0
|
||||
|
||||
return sorted(set(found), key=_release_number, reverse=True)
|
||||
|
||||
|
||||
def _host_os_arch() -> tuple[str, str]:
|
||||
"""(os, arch) normalized to release-asset vocabulary.
|
||||
|
||||
PITFALL: PROCESSOR_ARCHITECTURE lies under x64 emulation on
|
||||
ARM64 Windows. platform.machine() reads the same env on some Pythons, so
|
||||
on Windows prefer PROCESSOR_IDENTIFIER's text when present.
|
||||
"""
|
||||
system = platform.system().lower()
|
||||
os_name = {"windows": "win", "darwin": "macos", "linux": "ubuntu"}.get(system, system)
|
||||
machine = platform.machine().lower()
|
||||
arch = "arm64" if machine in ("arm64", "aarch64") else "x64"
|
||||
if os_name == "win":
|
||||
import os as _os
|
||||
ident = _os.environ.get("PROCESSOR_IDENTIFIER", "")
|
||||
if "armv8" in ident.lower() or "arm " in ident.lower():
|
||||
arch = "arm64"
|
||||
return os_name, arch
|
||||
|
||||
|
||||
def select_backend(gpu_vendor: str | None, os_name: str | None = None) -> str:
|
||||
"""Backend choice per design: CUDA if NVIDIA, Metal on macOS, Vulkan if
|
||||
a non-NVIDIA GPU is present, else CPU. ``--list-devices`` validates the
|
||||
choice post-install; the supervisor's touch generation is ground truth."""
|
||||
if os_name is None:
|
||||
os_name, _ = _host_os_arch()
|
||||
if os_name == "macos":
|
||||
return "metal"
|
||||
vendor = (gpu_vendor or "").lower()
|
||||
if "nvidia" in vendor:
|
||||
return "cuda"
|
||||
if vendor in ("amd", "intel") or "radeon" in vendor or "arc" in vendor:
|
||||
return "vulkan"
|
||||
return "cpu"
|
||||
|
||||
|
||||
def resolve_assets(tag: str, backend: str, os_name: str | None = None,
|
||||
arch: str | None = None) -> AssetPlan:
|
||||
"""Compose the asset list for (tag, backend, platform).
|
||||
|
||||
Raises BinaryResolutionError for combinations the release does not ship
|
||||
(a platform/backend pair upstream publishes no artifact for). Callers
|
||||
fall back down the backend ladder: cuda -> vulkan -> cpu.
|
||||
"""
|
||||
host_os, host_arch = _host_os_arch()
|
||||
os_name = os_name or host_os
|
||||
arch = arch or host_arch
|
||||
plan = AssetPlan(tag=tag, backend=backend)
|
||||
|
||||
if os_name == "macos":
|
||||
# macOS tarballs are unified (Metal built in).
|
||||
plan.assets = [f"llama-{tag}-bin-macos-{arch}.tar.gz"]
|
||||
return plan
|
||||
|
||||
if os_name == "ubuntu":
|
||||
if backend == "cuda":
|
||||
# No prebuilt Linux CUDA zips at current tags — Linux CUDA users
|
||||
# build from source or use vulkan; resolver is honest about it.
|
||||
raise BinaryResolutionError(
|
||||
f"no prebuilt linux CUDA asset at {tag}; use vulkan/cpu or a source build")
|
||||
suffix = {"vulkan": f"vulkan-{arch}", "hip": f"rocm-7.2-{arch}",
|
||||
"cpu": arch}.get(backend)
|
||||
if suffix is None:
|
||||
raise BinaryResolutionError(f"unsupported linux backend {backend}")
|
||||
plan.assets = [f"llama-{tag}-bin-ubuntu-{suffix}.tar.gz"]
|
||||
return plan
|
||||
|
||||
if os_name == "win":
|
||||
if backend == "cuda":
|
||||
cuda_ver = _WIN_CUDA_VERSION_ARM64 if arch == "arm64" else _WIN_CUDA_VERSION
|
||||
plan.assets = [
|
||||
f"llama-{tag}-bin-win-cuda-{cuda_ver}-{arch}.zip",
|
||||
f"cudart-llama-bin-win-cuda-{cuda_ver}-{arch}.zip",
|
||||
]
|
||||
elif backend == "vulkan":
|
||||
if arch == "arm64":
|
||||
raise BinaryResolutionError(f"no win-vulkan-arm64 asset at {tag}")
|
||||
plan.assets = [f"llama-{tag}-bin-win-vulkan-x64.zip"]
|
||||
elif backend == "hip":
|
||||
plan.assets = [f"llama-{tag}-bin-win-hip-radeon-x64.zip"]
|
||||
elif backend == "cpu":
|
||||
plan.assets = [f"llama-{tag}-bin-win-cpu-{arch}.zip"]
|
||||
else:
|
||||
raise BinaryResolutionError(f"unsupported windows backend {backend}")
|
||||
return plan
|
||||
|
||||
raise BinaryResolutionError(f"unsupported platform {os_name}-{arch}")
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
h = hashlib.sha256()
|
||||
with open(path, "rb") as f:
|
||||
for chunk in iter(lambda: f.read(1 << 22), b""):
|
||||
h.update(chunk)
|
||||
return h.hexdigest()
|
||||
|
||||
|
||||
def _download(url: str, dest: Path,
|
||||
progress: "Callable[[int, int], None] | None" = None) -> None:
|
||||
"""Stream url -> dest. ``progress(done_bytes, total_bytes)`` ticks per
|
||||
chunk (total 0 when the server sends no Content-Length) — a several-
|
||||
hundred-MB archive on a slow line must never look hung."""
|
||||
logger.info("downloading %s", url)
|
||||
tmp = dest.with_suffix(dest.suffix + ".part")
|
||||
with urllib.request.urlopen(url, timeout=120) as r, open(tmp, "wb") as f:
|
||||
total = int(r.headers.get("Content-Length") or 0)
|
||||
done = 0
|
||||
while True:
|
||||
chunk = r.read(1 << 20)
|
||||
if not chunk:
|
||||
break
|
||||
f.write(chunk)
|
||||
done += len(chunk)
|
||||
if progress is not None:
|
||||
progress(done, total)
|
||||
tmp.replace(dest)
|
||||
|
||||
|
||||
def _extract(archive: Path, dest: Path,
|
||||
progress: "Callable[[int, int], None] | None" = None) -> None:
|
||||
"""Extract member by member so ``progress(done, total)`` can tick in
|
||||
uncompressed bytes — big archives take real time on laptop disks."""
|
||||
if archive.name.endswith(".zip"):
|
||||
with zipfile.ZipFile(archive) as z:
|
||||
members = z.infolist()
|
||||
total = sum(m.file_size for m in members)
|
||||
done = 0
|
||||
for m in members:
|
||||
z.extract(m, dest)
|
||||
done += m.file_size
|
||||
if progress is not None:
|
||||
progress(done, total)
|
||||
else:
|
||||
import tarfile
|
||||
with tarfile.open(archive) as t:
|
||||
members = t.getmembers()
|
||||
total = sum(m.size for m in members)
|
||||
done = 0
|
||||
for m in members:
|
||||
t.extract(m, dest, filter="data")
|
||||
done += m.size
|
||||
if progress is not None:
|
||||
progress(done, total)
|
||||
|
||||
|
||||
def server_binary(install_dir: Path) -> Path:
|
||||
"""Locate llama-server within an extracted runtime (zips differ in
|
||||
whether they nest a build/bin directory)."""
|
||||
names = ("llama-server.exe", "llama-server")
|
||||
for name in names:
|
||||
direct = install_dir / name
|
||||
if direct.exists():
|
||||
return direct
|
||||
for name in names:
|
||||
hits = sorted(install_dir.rglob(name))
|
||||
if hits:
|
||||
return hits[0]
|
||||
raise BinaryResolutionError(f"llama-server not found under {install_dir}")
|
||||
|
||||
|
||||
def verify_install(install_dir: Path, tag: str) -> str:
|
||||
"""Run --version; require the tag's build number in the output.
|
||||
(The binary prints the tag WITHOUT the 'b' prefix.)"""
|
||||
exe = server_binary(install_dir)
|
||||
out = subprocess.run([str(exe), "--version"], capture_output=True,
|
||||
text=True, encoding="utf-8", errors="replace",
|
||||
timeout=60, cwd=str(exe.parent))
|
||||
text = (out.stdout + out.stderr).strip()
|
||||
if tag.lstrip("b") not in text:
|
||||
raise BinaryResolutionError(
|
||||
f"version check failed for {exe}: expected {tag}, got: {text[:120]}")
|
||||
return text.splitlines()[0] if text else ""
|
||||
|
||||
|
||||
def prune_old_tags(keep: list[str]) -> None:
|
||||
"""Retain only the tags in ``keep`` (current + previous — N-1 rollback).
|
||||
The shared ``downloads/`` archive cache is not a tag and always survives."""
|
||||
root = runtimes_root()
|
||||
if not root.exists():
|
||||
return
|
||||
for entry in root.iterdir():
|
||||
if entry.is_dir() and entry.name != "downloads" and entry.name not in keep:
|
||||
shutil.rmtree(entry, ignore_errors=True)
|
||||
logger.info("pruned old runtime %s", entry.name)
|
||||
|
||||
|
||||
def ensure_runtime_installed(tag: str, backend: str,
|
||||
expected_sha256: dict[str, str] | None = None,
|
||||
progress: "Callable[[str, int, int, str], None] | None" = None) -> Path:
|
||||
"""Idempotent: resolve, download, verify, extract, version-check.
|
||||
|
||||
``expected_sha256`` maps asset name -> hash when the catalog pins them;
|
||||
without pins the computed hash is recorded in the manifest (trust on
|
||||
first download, verified on every reinstall).
|
||||
``progress(stage, done_bytes, total_bytes, label)`` ticks through the
|
||||
slow parts — stage is "download" | "extract" | "verify", label is the
|
||||
asset counter ("1/2") when the plan has several archives.
|
||||
Returns the install directory containing llama-server.
|
||||
"""
|
||||
plan = resolve_assets(tag, backend)
|
||||
install_dir = plan.install_dir
|
||||
manifest_path = install_dir / "manifest.json"
|
||||
if manifest_path.exists():
|
||||
try:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
if manifest.get("verified_version"):
|
||||
return install_dir
|
||||
except (json.JSONDecodeError, OSError):
|
||||
pass # damaged manifest -> reinstall
|
||||
|
||||
install_dir.mkdir(parents=True, exist_ok=True)
|
||||
downloads = runtimes_root() / "downloads"
|
||||
downloads.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
recorded: dict[str, str] = {}
|
||||
n_assets = len(plan.assets)
|
||||
for i, asset in enumerate(plan.assets, 1):
|
||||
label = f"{i}/{n_assets}" if n_assets > 1 else ""
|
||||
archive = downloads / asset
|
||||
if not archive.exists():
|
||||
_download(RELEASE_URL.format(tag=tag, asset=asset), archive,
|
||||
progress=(lambda d, t, _l=label: progress("download", d, t, _l))
|
||||
if progress is not None else None)
|
||||
if progress is not None:
|
||||
progress("verify", 0, 0, label)
|
||||
digest = _sha256(archive)
|
||||
expected = (expected_sha256 or {}).get(asset)
|
||||
if expected and digest != expected:
|
||||
archive.unlink(missing_ok=True)
|
||||
raise BinaryResolutionError(
|
||||
f"sha256 mismatch for {asset}: expected {expected}, got {digest}")
|
||||
recorded[asset] = digest
|
||||
_extract(archive, install_dir,
|
||||
progress=(lambda d, t, _l=label: progress("extract", d, t, _l))
|
||||
if progress is not None else None)
|
||||
|
||||
if progress is not None:
|
||||
progress("verify", 0, 0, "")
|
||||
version = verify_install(install_dir, tag)
|
||||
manifest_path.write_text(json.dumps({
|
||||
"tag": tag, "backend": plan.backend, "assets": recorded,
|
||||
"verified_version": version,
|
||||
}, indent=2), encoding="utf-8")
|
||||
logger.info("installed llama.cpp %s (%s): %s", tag, backend, version)
|
||||
return install_dir
|
||||
@@ -0,0 +1,337 @@
|
||||
"""Bootstrap for the managed runtime: config -> installed binaries ->
|
||||
running supervised server.
|
||||
|
||||
One public call, ``ensure_local_runtime(config)``, safe to call at any
|
||||
session start:
|
||||
- disabled or already-running (state file answers /health) -> no-op
|
||||
- enabled -> install binaries if missing (idempotent), spawn supervisor
|
||||
|
||||
Kept import-light: callers gate on config before importing this module so
|
||||
sessions with local_runtime disabled never pay the import.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_constants import get_hermes_home # noqa: F401 — config paths
|
||||
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_SUPERVISOR = None # process-wide singleton; one router per Hermes process
|
||||
|
||||
|
||||
def _detect_gpu_vendor() -> str | None:
|
||||
"""Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi
|
||||
(resolved by the hardware probe's PATH-independent ladder — a stripped
|
||||
service PATH must not demote an NVIDIA box to vulkan/cpu); anything
|
||||
else defers to select_backend's fallback ladder."""
|
||||
from hermes_cli.local_runtime.hardware import _nvidia_smi_path
|
||||
|
||||
smi = _nvidia_smi_path()
|
||||
if smi is None:
|
||||
return None
|
||||
try:
|
||||
out = subprocess.run(
|
||||
[smi, "--query-gpu=name", "--format=csv,noheader"],
|
||||
capture_output=True, text=True, timeout=10)
|
||||
if out.returncode == 0 and out.stdout.strip():
|
||||
return "nvidia " + out.stdout.strip().splitlines()[0]
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def models_dir() -> Path:
|
||||
"""Machine-scoped, deliberately NOT profile-scoped: a 20 GB GGUF is a
|
||||
machine asset, and every profile shares the one managed server that
|
||||
serves it. See runtimes_root() for the same rule on the engine."""
|
||||
from hermes_constants import get_default_hermes_root
|
||||
|
||||
return get_default_hermes_root() / "models"
|
||||
|
||||
|
||||
def assets_dir() -> Path:
|
||||
"""Non-model companion files (mmproj vision projectors, spec-decode
|
||||
draft models). A subdirectory so the router's model listing — and our
|
||||
staged_models() — never mistakes an asset for a servable model."""
|
||||
return models_dir() / "assets"
|
||||
|
||||
|
||||
def staged_models() -> "list[Path]":
|
||||
"""Servable staged models: single-file GGUFs count when present; a
|
||||
split GGUF counts once, by its first part, and only when EVERY part
|
||||
is on disk — a mid-download split is not servable and must not
|
||||
surface anywhere as a model. Continuation parts and assets/ never
|
||||
count."""
|
||||
import re
|
||||
|
||||
part = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$")
|
||||
files = sorted(models_dir().glob("*.gguf"))
|
||||
names = {p.name for p in files}
|
||||
out = []
|
||||
for p in files:
|
||||
m = part.search(p.name)
|
||||
if m is None:
|
||||
out.append(p)
|
||||
continue
|
||||
if m.group(1) != "00001":
|
||||
continue
|
||||
stem = p.name[: m.start()]
|
||||
total = int(m.group(2))
|
||||
if all(f"{stem}-{i:05d}-of-{m.group(2)}.gguf" in names
|
||||
for i in range(2, total + 1)):
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
def staged_model_ids() -> "list[str]":
|
||||
import re
|
||||
|
||||
return [re.sub(r"-\d{5}-of-\d{5}$", "", p.stem) for p in staged_models()]
|
||||
|
||||
|
||||
def _presets_stale() -> bool:
|
||||
"""True when a staged model has no section in the preset INI — it
|
||||
would autoload with stock fit instead of a policy decision."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.presets import read_preset_decisions
|
||||
|
||||
known = set(read_preset_decisions())
|
||||
return any(mid not in known for mid in staged_model_ids())
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
|
||||
def _stop_state_server(state: dict) -> None:
|
||||
"""Best-effort stop of the server the state file points at (an
|
||||
incumbent this process doesn't supervise). The state pid is ours by
|
||||
contract — the file only ever describes the managed server."""
|
||||
from hermes_cli.local_runtime.endpoint import _pid_alive
|
||||
|
||||
pid = state.get("pid")
|
||||
try:
|
||||
pid = int(pid)
|
||||
except (TypeError, ValueError):
|
||||
return
|
||||
if pid <= 0:
|
||||
return
|
||||
try:
|
||||
import signal
|
||||
|
||||
os.kill(pid, signal.SIGTERM)
|
||||
except (OSError, ValueError):
|
||||
return
|
||||
# Give it a moment to release the port and the GPU. Liveness via
|
||||
# psutil — on Windows os.kill(pid, 0) TERMINATES the process, it is
|
||||
# not a probe (the endpoint.py pitfall note; #local-models review).
|
||||
for _ in range(50):
|
||||
if not _pid_alive(pid):
|
||||
return
|
||||
time.sleep(0.1)
|
||||
|
||||
|
||||
def refresh_local_runtime() -> bool:
|
||||
"""Restart the managed server so it rescans the models directory.
|
||||
|
||||
The router's model list is SPAWN-ONLY: a GGUF added after start is
|
||||
invisible to GET /models and 400s on completion, so anything that
|
||||
changes the staged set while the server runs must bounce it. Covers
|
||||
both ownership shapes: a supervised server restarts in-process; an
|
||||
ADOPTED server (started by a previous backend session — the normal
|
||||
shape after any restart) is stopped via its state-file pid and
|
||||
replaced with a supervised boot. Without the adopted branch, every
|
||||
download/delete in a post-restart session silently no-ops the bounce
|
||||
and the router serves a stale catalog. Returns False when there is
|
||||
nothing to refresh (no server anywhere; next boot scans fresh).
|
||||
"""
|
||||
global _SUPERVISOR
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
if _SUPERVISOR is None:
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
state = _state_endpoint()
|
||||
if state is None:
|
||||
return False
|
||||
logger.info("bouncing adopted llama-server (pid=%s) to rescan models",
|
||||
state.get("pid"))
|
||||
_stop_state_server(state)
|
||||
else:
|
||||
shutdown_local_runtime()
|
||||
return ensure_local_runtime(load_config(), force=True) is not None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("local runtime refresh failed: %s", exc)
|
||||
return False
|
||||
|
||||
|
||||
def ensure_local_runtime(config: dict, force: bool = False) -> "object | None":
|
||||
"""Idempotent boot of the managed runtime. Returns the supervisor (or
|
||||
None when disabled/unavailable). Never raises into a session start —
|
||||
failures log and return None; chat falls back to configured providers.
|
||||
|
||||
``force=True`` skips the enabled gate — used by the explicit "Use this
|
||||
model" action, where the click IS the opt-in (the caller records it in
|
||||
config so future boots auto-start).
|
||||
"""
|
||||
global _SUPERVISOR
|
||||
section = (config or {}).get("local_runtime") or {}
|
||||
if not force and not section.get("enabled"):
|
||||
return None
|
||||
if _SUPERVISOR is not None:
|
||||
return _SUPERVISOR
|
||||
|
||||
# Residency: no staged models means nothing to serve — don't boot an
|
||||
# empty server. The walked-away story handled with zero configuration
|
||||
# (delete your last model and boots stop); Use force-boots as ever.
|
||||
if not force and not staged_models():
|
||||
logger.info("local runtime enabled but no models staged; not booting")
|
||||
return None
|
||||
|
||||
# Another Hermes process may already be supervising — reuse via state,
|
||||
# but ONLY while its launch policy still covers every staged model. A
|
||||
# server whose preset file predates a download serves the new model
|
||||
# with no policy at all (--models-autoload + stock fit: f16 KV at max
|
||||
# context, no placement — the silent-demotion busy-wait on WDDM). A
|
||||
# stale incumbent gets stopped and replaced by a fresh boot with
|
||||
# regenerated presets; sessions ride through exactly like any other
|
||||
# supervised restart (stable port + persisted key).
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
state = _state_endpoint()
|
||||
if state is not None:
|
||||
if not _presets_stale():
|
||||
logger.info("managed llama-server already running (another process)")
|
||||
return None
|
||||
logger.info("running server's presets predate the staged models; "
|
||||
"replacing it so every model launches with a policy")
|
||||
_stop_state_server(state)
|
||||
|
||||
try:
|
||||
from hermes_cli.local_runtime.binaries import (
|
||||
ensure_runtime_installed,
|
||||
select_backend,
|
||||
)
|
||||
from hermes_cli.local_runtime.hardware import probe_budget
|
||||
from hermes_cli.local_runtime.presets import generate_presets
|
||||
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
||||
|
||||
backend = section.get("backend", "auto")
|
||||
if backend == "auto":
|
||||
backend = select_backend(_detect_gpu_vendor())
|
||||
# Boot ladder: serve what is INSTALLED, never download here. The
|
||||
# configured tag (config root-of-trust; deep-merge supplies the
|
||||
# Hermes-release default when unpinned) is preferred; when it isn't
|
||||
# installed yet, the newest installed tag serves and the status
|
||||
# endpoint reports the pending update — the download is a deliberate
|
||||
# button click in the pane, not a boot-path surprise (a multi-minute
|
||||
# inline download here is exactly how the onboarding bounce returns).
|
||||
from hermes_cli.local_runtime.binaries import default_tag, installed_tags
|
||||
|
||||
tag = section.get("tag") or default_tag()
|
||||
have = installed_tags()
|
||||
if tag not in have:
|
||||
if not have:
|
||||
logger.info("local runtime enabled but no build installed; "
|
||||
"install happens in the Local Models pane")
|
||||
return None
|
||||
logger.info("configured tag %s not installed; serving %s "
|
||||
"(update is a click in Local Models)", tag, have[0])
|
||||
tag = have[0]
|
||||
install_dir = ensure_runtime_installed(tag, backend)
|
||||
|
||||
mdir = models_dir()
|
||||
mdir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Context policy: one launch decision per staged model, carried to
|
||||
# the router via the preset INI. Priced against CAPACITY, not live
|
||||
# free VRAM: this runs while the outgoing server instance may still
|
||||
# hold the card (restart, refresh after a download), and its memory
|
||||
# is freed before the new instance loads anything. Pricing against
|
||||
# live-free here once pinned a fitting model's weights to CPU
|
||||
# because the probe saw the predecessor's VRAM as gone.
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
try:
|
||||
entries = generate_presets(mdir, probe_budget(planning=True), preset_path)
|
||||
for entry in entries:
|
||||
if entry.refusal:
|
||||
logger.warning("model refused by physics check: %s", entry.refusal)
|
||||
except Exception as exc: # noqa: BLE001 — policy failure must not block serving
|
||||
# Degradation ladder: a STALE policy still beats no policy —
|
||||
# stock fit (f16 KV at max context, no placement) is the
|
||||
# silent-busy-wait failure on Windows. Keep serving with the
|
||||
# previous INI when one exists; only a first boot with no INI
|
||||
# at all falls to stock fit.
|
||||
if preset_path.exists():
|
||||
logger.error("preset generation failed (%s); serving with the "
|
||||
"PREVIOUS launch policies — models staged since "
|
||||
"the last successful generation run unpoliced "
|
||||
"until this is fixed", exc)
|
||||
else:
|
||||
logger.error("preset generation failed (%s) and no previous "
|
||||
"policy file exists; router runs stock fit", exc)
|
||||
preset_path = None
|
||||
|
||||
sup = LlamaServerSupervisor(
|
||||
install_dir, mdir,
|
||||
models_max=int(section.get("models_max", 4)),
|
||||
port=int(section.get("port", 0)) or None,
|
||||
preset_path=preset_path,
|
||||
)
|
||||
try:
|
||||
sup.start()
|
||||
except Exception:
|
||||
# start() can fail after the router process exists (health
|
||||
# timeout, spawn error): leaving it running unsupervised
|
||||
# strands its VRAM behind a port nothing will clean up.
|
||||
try:
|
||||
sup.stop()
|
||||
except Exception: # noqa: BLE001 — cleanup is best-effort
|
||||
pass
|
||||
raise
|
||||
_SUPERVISOR = sup
|
||||
logger.info("managed llama-server up at %s (backend=%s tag=%s)",
|
||||
sup.base_url, backend, tag)
|
||||
_start_idle_sweeper(sup)
|
||||
return sup
|
||||
except Exception as exc: # noqa: BLE001 — never break session start
|
||||
logger.warning("managed local runtime unavailable: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def shutdown_local_runtime() -> None:
|
||||
global _SUPERVISOR
|
||||
if _SUPERVISOR is not None:
|
||||
_SUPERVISOR.stop()
|
||||
_SUPERVISOR = None
|
||||
|
||||
|
||||
def get_supervisor():
|
||||
"""The process-local supervisor, or None (server may still be running
|
||||
under another process — check the state file)."""
|
||||
return _SUPERVISOR
|
||||
|
||||
|
||||
def _start_idle_sweeper(sup) -> None:
|
||||
"""Idle-residency loop: every couple of minutes, unload non-primary
|
||||
models idle past the supervisor's threshold. Daemon thread tied to the
|
||||
supervisor's lifetime — exits when the server stops."""
|
||||
import threading
|
||||
|
||||
def _loop():
|
||||
while sup.proc is not None and sup.proc.poll() is None:
|
||||
time.sleep(120)
|
||||
try:
|
||||
sup.sweep_idle()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("idle sweep skipped: %s", exc)
|
||||
|
||||
threading.Thread(target=_loop, daemon=True,
|
||||
name="local-runtime-idle-sweep").start()
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Capability answers for models served by the managed runtime.
|
||||
|
||||
Capability lookups (vision, and whatever comes next) consult cloud-shaped
|
||||
catalogs that have never heard of a local GGUF, so a vision-capable local
|
||||
model reads as text-only and images detour to an auxiliary cloud model —
|
||||
the wrong behavior twice over for a local-first user (broken feature, and
|
||||
a screenshot silently leaving the machine).
|
||||
|
||||
The managed runtime can answer from ground truth instead, best source
|
||||
first:
|
||||
|
||||
1. The RUNNING child's /props: llama-server reports a ``modalities`` block
|
||||
when a vision projector is loaded. The server that will receive the
|
||||
image says whether it can see — no inference, no catalog.
|
||||
2. The catalog entry's declared capability (the ``vision`` tag + mmproj
|
||||
asset) for staged-but-unloaded models: what the model WILL support once
|
||||
its projector loads beside it.
|
||||
3. None — not one of ours, or nothing known; the caller falls through to
|
||||
its other sources.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import urllib.request
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
|
||||
|
||||
# Image formats the managed server's decoder actually handles. llama.cpp
|
||||
# decodes with stb_image: PNG/JPEG/GIF/BMP yes, WebP NO — and a WebP part
|
||||
# fails SILENTLY (no HTTP error, no log line; the model just never sees an
|
||||
# image and confabulates a description). Anything outside this set must be
|
||||
# transcoded before the request. Measured against the live server: the
|
||||
# same red square answered 'Red' as PNG and 'Unseen' as WebP.
|
||||
ACCEPTED_IMAGE_MIMES = frozenset({"image/png", "image/jpeg"})
|
||||
|
||||
|
||||
def is_managed_provider(provider: str, base_url: str = "") -> bool:
|
||||
"""True when this provider/base_url pair points at the managed server.
|
||||
``custom`` only counts when the base_url IS the managed endpoint —
|
||||
background lookups must never claim someone else's custom server."""
|
||||
p = (provider or "").strip().lower()
|
||||
if p in _LLAMACPP_ALIASES:
|
||||
return True
|
||||
if p == "custom" and base_url:
|
||||
try:
|
||||
from hermes_cli.local_runtime.growth import is_managed_endpoint
|
||||
|
||||
return is_managed_endpoint(base_url)
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
return False
|
||||
|
||||
|
||||
def _props_modalities(model_id: str) -> "bool | None":
|
||||
"""Ask the running server whether this loaded child sees images.
|
||||
None when the server is down, the model isn't loaded, or the build
|
||||
doesn't report modalities."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
state = _state_endpoint()
|
||||
if state is None:
|
||||
return None
|
||||
base = state["base_url"].rsplit("/v1", 1)[0]
|
||||
req = urllib.request.Request(
|
||||
f"{base}/props?model={model_id}",
|
||||
headers={"Authorization": f"Bearer {state.get('api_key', '')}"})
|
||||
with urllib.request.urlopen(req, timeout=3) as r:
|
||||
props = json.load(r)
|
||||
modalities = props.get("modalities")
|
||||
if isinstance(modalities, dict) and "vision" in modalities:
|
||||
return bool(modalities["vision"])
|
||||
return None
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
def managed_model_supports_vision(model_id: str) -> "bool | None":
|
||||
"""Ground-truth vision capability for a staged model, or None when the
|
||||
model isn't ours / nothing is known (caller keeps falling through)."""
|
||||
if not model_id:
|
||||
return None
|
||||
|
||||
# Only answer for models actually staged with us.
|
||||
try:
|
||||
from hermes_cli.local_runtime.bootstrap import staged_model_ids
|
||||
|
||||
if model_id not in staged_model_ids():
|
||||
return None
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
live = _props_modalities(model_id)
|
||||
if live is not None:
|
||||
return live
|
||||
|
||||
# Staged but not loaded (or an older server build): the catalog knows
|
||||
# whether this model ships a vision projector.
|
||||
try:
|
||||
from hermes_cli.local_runtime.bootstrap import assets_dir
|
||||
from hermes_cli.local_runtime.catalog import find_entry_for_model
|
||||
|
||||
hit = find_entry_for_model(model_id)
|
||||
if hit is None:
|
||||
return None
|
||||
entry = hit[0]
|
||||
if entry.mmproj is None:
|
||||
return False
|
||||
# Capability requires the projector to actually be on disk — a
|
||||
# model downloaded before its mmproj (partial delete, old layout)
|
||||
# genuinely cannot see.
|
||||
return (assets_dir() / entry.mmproj.local_name).exists()
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
@@ -0,0 +1,174 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"models": [
|
||||
{
|
||||
"id": "qwen3.8-27b",
|
||||
"display_name": "Qwen3.8 27B",
|
||||
"description": "Best all-round agent model; sees images; long context stays fast",
|
||||
"repo": "unsloth/Qwen3.8-27B-GGUF",
|
||||
"variants": [
|
||||
{
|
||||
"quant": "UD-Q4_K_M",
|
||||
"files": [
|
||||
{
|
||||
"path": "Qwen3.8-27B-UD-Q4_K_M.gguf",
|
||||
"size_bytes": 16464440224
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"n_ctx_train": 262144,
|
||||
"full_layers": 16,
|
||||
"recurrent_layers": 48,
|
||||
"per_layer_f16": 4096,
|
||||
"n_vocab": 248320,
|
||||
"mmproj": {
|
||||
"path": "mmproj-BF16.gguf",
|
||||
"size_bytes": 931146432,
|
||||
"local": "mmproj-Qwen3.8-27B-BF16.gguf"
|
||||
},
|
||||
"mtp": true,
|
||||
"mtp_draft_depth": 2,
|
||||
"sampling": {
|
||||
"temp": "1.0",
|
||||
"top-p": "0.95",
|
||||
"top-k": "20",
|
||||
"min-p": "0.0"
|
||||
},
|
||||
"quality": 90,
|
||||
"decode_fraction": 1.0
|
||||
},
|
||||
{
|
||||
"id": "qwen3.8-flash-next",
|
||||
"display_name": "Qwen3.8 Flash Next",
|
||||
"description": "Frontier-scale model; needs a very large GPU to run well",
|
||||
"repo": "unsloth/Qwen3.8-Flash-Next-GGUF",
|
||||
"variants": [
|
||||
{
|
||||
"quant": "UD-Q4_K_XL",
|
||||
"files": [
|
||||
{
|
||||
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf",
|
||||
"size_bytes": 10946624
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf",
|
||||
"size_bytes": 49859583136
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf",
|
||||
"size_bytes": 49376141504
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf",
|
||||
"size_bytes": 12087983520
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"n_ctx_train": 262144,
|
||||
"full_layers": 12,
|
||||
"recurrent_layers": 36,
|
||||
"per_layer_f16": 2048,
|
||||
"moe": true,
|
||||
"n_vocab": 248320,
|
||||
"mmproj": {
|
||||
"path": "mmproj-BF16.gguf",
|
||||
"size_bytes": 907542944,
|
||||
"local": "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
|
||||
},
|
||||
"min_engine": "b10678",
|
||||
"quality": 95,
|
||||
"decode_fraction": 0.08
|
||||
},
|
||||
{
|
||||
"id": "qwen3.6-35b-a3b",
|
||||
"display_name": "Qwen3.6 35B-A3B",
|
||||
"description": "Bigger mixture-of-experts with multi-token prediction; sees images",
|
||||
"repo": "unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
|
||||
"variants": [
|
||||
{
|
||||
"quant": "UD-Q4_K_M",
|
||||
"files": [
|
||||
{
|
||||
"path": "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf",
|
||||
"size_bytes": 22663387424
|
||||
}
|
||||
],
|
||||
"validated": true
|
||||
}
|
||||
],
|
||||
"n_ctx_train": 262144,
|
||||
"full_layers": 10,
|
||||
"recurrent_layers": 30,
|
||||
"per_layer_f16": 2048,
|
||||
"moe": true,
|
||||
"mtp": true,
|
||||
"n_vocab": 248320,
|
||||
"mtp_draft_depth": 2,
|
||||
"mmproj": {
|
||||
"path": "mmproj-BF16.gguf",
|
||||
"size_bytes": 902822528,
|
||||
"local": "mmproj-Qwen3.6-35B-A3B-BF16.gguf"
|
||||
},
|
||||
"sampling": {
|
||||
"temp": "1.0",
|
||||
"top-p": "0.95",
|
||||
"top-k": "20",
|
||||
"min-p": "0.0"
|
||||
},
|
||||
"quality": 80,
|
||||
"decode_fraction": 0.15
|
||||
},
|
||||
{
|
||||
"id": "deepseek-v4-flash",
|
||||
"display_name": "DeepSeek V4 Flash",
|
||||
"description": "Frontier-class model for machines with 128GB+ memory",
|
||||
"repo": "unsloth/DeepSeek-V4-Flash-0731-GGUF",
|
||||
"variants": [
|
||||
{
|
||||
"quant": "UD-Q4_K_XL",
|
||||
"files": [
|
||||
{
|
||||
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00001-of-00005.gguf",
|
||||
"size_bytes": 5257408
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00002-of-00005.gguf",
|
||||
"size_bytes": 48935523072
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00003-of-00005.gguf",
|
||||
"size_bytes": 48980787136
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00004-of-00005.gguf",
|
||||
"size_bytes": 49999168416
|
||||
},
|
||||
{
|
||||
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00005-of-00005.gguf",
|
||||
"size_bytes": 7174505088
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"n_ctx_train": 1048576,
|
||||
"full_layers": 43,
|
||||
"recurrent_layers": 0,
|
||||
"per_layer_f16": 1152,
|
||||
"moe": true,
|
||||
"n_vocab": 163840,
|
||||
"draft": {
|
||||
"path": "dspark-DeepSeek-V4-Flash-0731-Q8_0.gguf",
|
||||
"size_bytes": 10896057440
|
||||
},
|
||||
"sampling": {
|
||||
"temp": "1.0",
|
||||
"top-p": "0.95",
|
||||
"min-p": "0.01"
|
||||
},
|
||||
"quality": 85,
|
||||
"decode_fraction": 0.1
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,464 @@
|
||||
"""Curated starter catalog for the managed local runtime.
|
||||
|
||||
Small and honest: every entry carries the estimator inputs (measured on
|
||||
real GGUFs) so the picker can price a model BEFORE the user downloads
|
||||
gigabytes. Once a file is on disk, profile_from_gguf() is the authority
|
||||
and the catalog numbers are only used for the download decision. Entries
|
||||
whose base config is gated upstream carry a same-family conservative
|
||||
prior (commented) — the GGUF header corrects it at load time.
|
||||
|
||||
Each model ships ONE build, Q4-class (UD-Q4_K_M where the repo has it,
|
||||
UD-Q4_K_XL elsewhere). Q4 is the quant class current engines optimize
|
||||
for and the sweet spot of the size/quality curve, so there is no quant
|
||||
ladder: headroom buys a bigger context window, never a bigger quant,
|
||||
and every machine runs the same well-tested build. Below Q4 the quality
|
||||
loss is too severe to ship as someone's first local-AI experience; the
|
||||
fit policy prices the build honestly (zero-spill, spilled, or refused by
|
||||
the physics check).
|
||||
|
||||
Validation lifecycle: builds proven end-to-end on real hardware are
|
||||
marked validated. Day-0 entries ship before that proof (they simply lack
|
||||
the validated flag) — ensure_model_ready's touch generation still gates
|
||||
every first load at runtime.
|
||||
|
||||
Multi-file models: variants may carry split-GGUF parts (llama-server loads
|
||||
from the first part; all parts download together). Entries may carry an
|
||||
mmproj (vision projector) and a speculative-decode draft model — both
|
||||
download alongside the weights. MTP-integrated models run spec decode
|
||||
wherever they load; a separate draft model attaches only when the launch
|
||||
decision spills, where its speedup is largest.
|
||||
|
||||
File sizes come from HF LFS metadata and feed the estimator, the fit
|
||||
pills, and download progress. There is no download-time integrity check
|
||||
by design: a corrupt or truncated file surfaces as a llama.cpp
|
||||
load error at first use, and the reachability test catches upstream
|
||||
re-uploads by size drift before users do.
|
||||
|
||||
This is deliberately not a live registry feed: entries are reviewed like a
|
||||
version bump (the same policy governs vendor recipe ingestion — parsed
|
||||
data, never executed commands).
|
||||
|
||||
Vendor recipes overlay: a per-SKU recipes repo may SUPPLEMENT these
|
||||
entries where applicable — vendor SKUs only, never the base layer for
|
||||
other platforms. A recipe may enrich identity (GGUF/quant/sha), perf
|
||||
hints (-b/-ub, spec-decode), and sampling defaults; it never carries
|
||||
context/slots/placement/serving flags (the fit policy owns those).
|
||||
Resolution: exact SKU -> GPU-class bucket -> fit-only. Snapshot-synced,
|
||||
reviewed like a tag bump.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
import urllib.request
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
FLOOR,
|
||||
RUNTIME_OVERHEAD_BYTES,
|
||||
TARGET_WINDOW,
|
||||
ub_logits_bytes,
|
||||
)
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget,
|
||||
LayerKind,
|
||||
ModelProfile,
|
||||
ctx_bytes,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_GIB = 1 << 30
|
||||
_PART_SUFFIX = re.compile(r"-\d{5}-of-\d{5}$")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AssetFile:
|
||||
"""One downloadable file: repo-relative path and exact bytes (the size
|
||||
feeds the estimator and the download progress bar; there is no
|
||||
download-time integrity check by design — a corrupt file surfaces as a
|
||||
llama.cpp load error). ``local`` overrides the on-disk name (repos
|
||||
reuse generic names like mmproj-BF16.gguf across models). Non-model
|
||||
extras live under the models dir's assets/ subdirectory so the router
|
||||
never lists them."""
|
||||
|
||||
path: str # repo-relative (may include a subdir)
|
||||
size_bytes: int
|
||||
local: str | None = None
|
||||
|
||||
@property
|
||||
def local_name(self) -> str:
|
||||
return self.local or PurePosixPath(self.path).name
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class QuantVariant:
|
||||
"""One downloadable build of a model. Split GGUFs list every part in
|
||||
files; the model loads from the first part."""
|
||||
|
||||
quant: str # e.g. "UD-Q4_K_M"
|
||||
files: tuple # AssetFile, first = the load target
|
||||
validated: bool = False # proven end-to-end on real hardware
|
||||
|
||||
@property
|
||||
def model_id(self) -> str:
|
||||
stem = PurePosixPath(self.files[0].path).name.removesuffix(".gguf")
|
||||
return _PART_SUFFIX.sub("", stem)
|
||||
|
||||
@property
|
||||
def size_bytes(self) -> int:
|
||||
return sum(f.size_bytes for f in self.files)
|
||||
|
||||
@property
|
||||
def weights_bytes(self) -> int:
|
||||
"""Pre-download weights estimate: GGUF bytes ≈ tensor bytes + a
|
||||
small header (<2%) — a safe, slightly conservative stand-in until
|
||||
profile_from_gguf reads the real table."""
|
||||
return self.size_bytes
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CatalogEntry:
|
||||
id: str # stable family id (variant-independent)
|
||||
display_name: str
|
||||
description: str # one line, plain language
|
||||
repo: str # HF repo
|
||||
variants: tuple # QuantVariant (exactly one, Q4-class)
|
||||
# Estimator inputs (measured or config-derived; quant changes weights,
|
||||
# never KV). Entries with gated upstream configs carry a conservative
|
||||
# same-family prior — the GGUF header is the authority after download.
|
||||
n_ctx_train: int
|
||||
full_layers: int
|
||||
recurrent_layers: int
|
||||
per_layer_f16: int # KV bytes/token per full-attention layer
|
||||
swa_layers: int = 0
|
||||
swa_window: int = 0
|
||||
moe: bool = False
|
||||
mtp: bool = False # ships MTP heads (spec decode when loaded)
|
||||
# Speculative draft depth for MTP models. Per-model and measured:
|
||||
# deeper drafting pays only while draft acceptance holds, and the
|
||||
# break-even depth differs by model.
|
||||
mtp_draft_depth: int = 3
|
||||
# Vocab size prices the GPU logits buffers (ubatch x vocab x fp32,
|
||||
# doubled under MTP backend sampling) — a multi-GiB term at large
|
||||
# vocab sizes that a weights-only fit would miss.
|
||||
n_vocab: int = 0
|
||||
mmproj: "AssetFile | None" = None # vision projector, downloads with model
|
||||
draft: "AssetFile | None" = None # spec-decode draft model (e.g. DSpark)
|
||||
sampling: dict = field(default_factory=dict) # INI long-form launch defaults
|
||||
# Oldest llama.cpp release tag that can load this model (day-0
|
||||
# architectures need the release where their support landed). Empty
|
||||
# means any installed engine. The pane gates download/activate on it.
|
||||
min_engine: str = ""
|
||||
# Editorial quality ordering (higher = smarter), authored once,
|
||||
# globally, at catalog-authoring time — Artificial Analysis-informed
|
||||
# where they cover the model (scripts/aa_quality_sync.py proposes,
|
||||
# the commit decides), editorial elsewhere. Ranks entries for the
|
||||
# per-machine recommendation; never displayed as a score (it grades
|
||||
# the full-precision model, not our Q4 build).
|
||||
quality: int = 0
|
||||
# Fraction of the build's bytes read per decoded token: 1.0 for dense
|
||||
# models (every weight streams every token), the active slice for MoE
|
||||
# (attention + shared + routed experts over total). With memory
|
||||
# bandwidth this predicts decode speed — the physics half of the
|
||||
# recommendation.
|
||||
decode_fraction: float = 1.0
|
||||
|
||||
def profile(self, variant: QuantVariant) -> ModelProfile:
|
||||
layers = ([(LayerKind.FULL, self.per_layer_f16)] * self.full_layers
|
||||
+ [(LayerKind.SWA, self.per_layer_f16)] * self.swa_layers
|
||||
+ [(LayerKind.RECURRENT, 0)] * self.recurrent_layers)
|
||||
return ModelProfile(
|
||||
name=variant.model_id, weights_bytes=variant.weights_bytes,
|
||||
embd_table_bytes=0, n_ctx_train=self.n_ctx_train,
|
||||
layers=layers, swa_window=self.swa_window, moe=self.moe,
|
||||
n_vocab=self.n_vocab,
|
||||
kv_scale=1.2 if self.mtp else 1.0)
|
||||
|
||||
def download_files(self, variant: QuantVariant) -> tuple:
|
||||
"""Everything a download job fetches for this variant, in order."""
|
||||
extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
|
||||
return tuple(variant.files) + extras
|
||||
|
||||
def download_bytes(self, variant: QuantVariant) -> int:
|
||||
return sum(f.size_bytes for f in self.download_files(variant))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class VariantChoice:
|
||||
"""Selection result: which build this machine should download and why.
|
||||
reason_key is a UI-copy discriminator, not display text."""
|
||||
|
||||
variant: QuantVariant
|
||||
zero_spill: bool
|
||||
reason_key: str # "best-large-window" | "best-fits" | "smallest-fits-spilled"
|
||||
|
||||
|
||||
def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice | None:
|
||||
"""Fit the entry's one build (Q4-class) to this machine.
|
||||
|
||||
Every entry ships exactly one variant (see the module docstring for
|
||||
why there is no quant ladder); headroom buys a bigger window, never
|
||||
a bigger quant. The fit shapes:
|
||||
|
||||
- "best-large-window": zero-spills at TARGET_WINDOW
|
||||
- "best-fits": zero-spills at the 64K floor
|
||||
- "smallest-fits-spilled": weights spill to host RAM, priced honestly
|
||||
- None: even spilled, physics refuses (the machine can't run it)
|
||||
"""
|
||||
overhead = (RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
native = entry.n_ctx_train or FLOOR
|
||||
variant = entry.variants[-1]
|
||||
profile = entry.profile(variant)
|
||||
need = variant.weights_bytes + overhead
|
||||
if (need + ctx_bytes(profile, min(TARGET_WINDOW, native))
|
||||
<= budget.usable_vram_bytes):
|
||||
return VariantChoice(variant=variant, zero_spill=True,
|
||||
reason_key="best-large-window")
|
||||
floor_kv = ctx_bytes(profile, min(FLOOR, native))
|
||||
if need + floor_kv <= budget.usable_vram_bytes:
|
||||
return VariantChoice(variant=variant, zero_spill=True,
|
||||
reason_key="best-fits")
|
||||
if need + floor_kv <= budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
return VariantChoice(variant=variant, zero_spill=False,
|
||||
reason_key="smallest-fits-spilled")
|
||||
return None
|
||||
|
||||
|
||||
# ── recommendation: best quality that fits and isn't miserably slow ──
|
||||
#
|
||||
# Two axes, each living where it belongs. QUALITY is a judgment made once,
|
||||
# globally, at authoring time (entry.quality — AA-informed, editorially
|
||||
# owned). SPEED is physics computed per machine: decode is memory-bound,
|
||||
# so predicted tok/s ≈ bandwidth / bytes-read-per-token, and the bytes per
|
||||
# token are the build's size scaled by its decode fraction (dense reads
|
||||
# everything; MoE reads the active slice). The pick: highest quality among
|
||||
# entries that run resident and clear a pleasant speed floor; else the
|
||||
# fastest resident entry; else the least-painful spilled one.
|
||||
#
|
||||
# The bandwidth axis is the `uma` flag for now: every discrete card that
|
||||
# matters is 900+ GB/s GDDR while the unified-memory class measures ~1/5th
|
||||
# of that, so the flag IS the high/low split. A measured per-machine
|
||||
# bandwidth (one cached memcpy probe) can replace these class constants
|
||||
# without touching the rule; predictions order candidates and gate the
|
||||
# floor — they are not display values.
|
||||
|
||||
_DISCRETE_BANDWIDTH_GB_S = 1000.0 # representative GDDR6X/GDDR7 class
|
||||
_UMA_BANDWIDTH_GB_S = 210.0 # measured on unified-memory NVIDIA
|
||||
_HOST_BANDWIDTH_GB_S = 80.0 # spilled weights stream over host DRAM
|
||||
|
||||
# The one editorial constant in the tree: below this predicted decode
|
||||
# speed a model stops feeling pleasant for agentic use (roughly reading
|
||||
# speed with headroom for tool-call bursts). Distinct from the growth
|
||||
# policy's 6 tok/s compress floor, which marks unusable, not unpleasant.
|
||||
PLEASANT_FLOOR_TOK_S = 20.0
|
||||
|
||||
|
||||
def predicted_decode_tok_s(entry: CatalogEntry, variant: QuantVariant,
|
||||
budget: HardwareBudget, *,
|
||||
spilled: bool = False) -> float:
|
||||
"""Memory-bound decode prediction for ordering and floor-gating."""
|
||||
bandwidth = (_HOST_BANDWIDTH_GB_S if spilled
|
||||
else _UMA_BANDWIDTH_GB_S if budget.uma
|
||||
else _DISCRETE_BANDWIDTH_GB_S)
|
||||
bytes_per_token = max(1.0, variant.size_bytes * entry.decode_fraction)
|
||||
return bandwidth * 1e9 / bytes_per_token
|
||||
|
||||
|
||||
def recommended_entry(budget: HardwareBudget,
|
||||
entries: "tuple[CatalogEntry, ...] | None" = None
|
||||
) -> "tuple[CatalogEntry, str] | None":
|
||||
"""The catalog's default pick for THIS machine, with its reason.
|
||||
|
||||
Callers pass pre-filtered entries when some are ineligible for
|
||||
reasons the catalog can't know (engine too old); default is the full
|
||||
catalog. Returns (entry, reason) — the reason is a key the UI turns
|
||||
into the Recommended badge's tooltip, so the rationale shown to the
|
||||
user is the branch that actually fired, never a parallel explanation
|
||||
that can drift:
|
||||
|
||||
best-quality-resident quality won among resident entries that
|
||||
clear the pleasant floor
|
||||
speed-gated-quality same, but the floor eliminated a HIGHER
|
||||
quality candidate — the exact 'why not the
|
||||
big model?' a unified-memory owner asks
|
||||
fastest-resident nothing resident clears the floor; the
|
||||
quickest resident entry wins
|
||||
least-painful-spilled nothing runs resident; fastest from host
|
||||
memory (MoE by construction)
|
||||
|
||||
Returns None only when nothing fits at all.
|
||||
"""
|
||||
pool = CATALOG if entries is None else entries
|
||||
fitting: list[tuple[CatalogEntry, VariantChoice]] = []
|
||||
for entry in pool:
|
||||
choice = select_variant(entry, budget)
|
||||
if choice is not None:
|
||||
fitting.append((entry, choice))
|
||||
if not fitting:
|
||||
return None
|
||||
|
||||
resident = [(e, c) for e, c in fitting if c.zero_spill]
|
||||
pleasant = [
|
||||
(e, c) for e, c in resident
|
||||
if predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
|
||||
]
|
||||
if pleasant:
|
||||
pick = max(pleasant, key=lambda t: (t[0].quality, -t[1].variant.size_bytes))[0]
|
||||
floor_gated = any(e.quality > pick.quality for e, _ in resident)
|
||||
return (pick, "speed-gated-quality" if floor_gated
|
||||
else "best-quality-resident")
|
||||
if resident:
|
||||
pick = max(resident,
|
||||
key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget))[0]
|
||||
return (pick, "fastest-resident")
|
||||
# Everything spills: take the least painful — fastest predicted decode
|
||||
# from host memory (MoE wins here by construction; a dense spill
|
||||
# streams every weight over the host bus).
|
||||
pick = max(fitting,
|
||||
key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget,
|
||||
spilled=True))[0]
|
||||
return (pick, "least-painful-spilled")
|
||||
|
||||
|
||||
def recommended_id(budget: HardwareBudget,
|
||||
entries: "tuple[CatalogEntry, ...] | None" = None) -> str | None:
|
||||
picked = recommended_entry(budget, entries)
|
||||
return picked[0].id if picked is not None else None
|
||||
|
||||
|
||||
# ── catalog data: packaged JSON, refreshed from GitHub in memory ─
|
||||
#
|
||||
# The catalog DATA lives in catalog.json (checked in beside this module
|
||||
# and shipped as package data); this module keeps all policy. At import
|
||||
# we load the packaged copy — no network on the import path. A TTL-gated
|
||||
# background refresh fetches the same file from the repo's main branch
|
||||
# and swaps it in memory only: nothing on disk changes, so a git
|
||||
# checkout never sees a dirty tracked file and the packaged copy remains
|
||||
# the offline truth. A reverted commit on main heals every install on
|
||||
# its next fetch, and day-0 entries reach users without an app release.
|
||||
|
||||
_CATALOG_URL = ("https://raw.githubusercontent.com/NousResearch/hermes-agent"
|
||||
"/main/hermes_cli/local_runtime/catalog.json")
|
||||
_SCHEMA_VERSION = 1
|
||||
_REFRESH_TTL_S = 6 * 3600
|
||||
_refresh_lock = threading.Lock()
|
||||
_last_refresh_attempt = 0.0
|
||||
|
||||
|
||||
def _asset_from(d: "dict | None") -> "AssetFile | None":
|
||||
if not d:
|
||||
return None
|
||||
return AssetFile(path=d["path"], size_bytes=int(d["size_bytes"]),
|
||||
local=d.get("local"))
|
||||
|
||||
|
||||
def _load_catalog(doc: dict) -> "tuple[CatalogEntry, ...]":
|
||||
"""Parse a catalog document into entries. Unknown fields are ignored
|
||||
(newer catalogs stay readable by older apps); a major schema bump is
|
||||
the signal that they wouldn't be, and the caller skips the document."""
|
||||
if int(doc.get("schema_version", 0)) != _SCHEMA_VERSION:
|
||||
raise ValueError(f"catalog schema {doc.get('schema_version')!r} "
|
||||
f"(this build reads {_SCHEMA_VERSION})")
|
||||
entries = []
|
||||
for m in doc["models"]:
|
||||
variants = tuple(
|
||||
QuantVariant(quant=v["quant"],
|
||||
files=tuple(_asset_from(f) for f in v["files"]),
|
||||
validated=bool(v.get("validated")))
|
||||
for v in m["variants"])
|
||||
entries.append(CatalogEntry(
|
||||
id=m["id"], display_name=m["display_name"],
|
||||
description=m["description"], repo=m["repo"], variants=variants,
|
||||
n_ctx_train=int(m["n_ctx_train"]),
|
||||
full_layers=int(m["full_layers"]),
|
||||
recurrent_layers=int(m["recurrent_layers"]),
|
||||
per_layer_f16=int(m["per_layer_f16"]),
|
||||
swa_layers=int(m.get("swa_layers", 0)),
|
||||
swa_window=int(m.get("swa_window", 0)),
|
||||
moe=bool(m.get("moe")), mtp=bool(m.get("mtp")),
|
||||
mtp_draft_depth=int(m.get("mtp_draft_depth", 3)),
|
||||
n_vocab=int(m.get("n_vocab", 0)),
|
||||
mmproj=_asset_from(m.get("mmproj")),
|
||||
draft=_asset_from(m.get("draft")),
|
||||
sampling=dict(m.get("sampling", {})),
|
||||
min_engine=str(m.get("min_engine", "")),
|
||||
quality=int(m.get("quality", 0)),
|
||||
decode_fraction=float(m.get("decode_fraction", 1.0)),
|
||||
))
|
||||
return tuple(entries)
|
||||
|
||||
|
||||
def _packaged_catalog() -> "tuple[CatalogEntry, ...]":
|
||||
from importlib.resources import files
|
||||
|
||||
raw = files("hermes_cli.local_runtime").joinpath("catalog.json").read_text(
|
||||
encoding="utf-8")
|
||||
return _load_catalog(json.loads(raw))
|
||||
|
||||
|
||||
CATALOG: "tuple[CatalogEntry, ...]" = _packaged_catalog()
|
||||
|
||||
|
||||
def refresh_catalog(force: bool = False) -> bool:
|
||||
"""Fetch the current catalog from the repo and swap it in memory.
|
||||
|
||||
Best-effort by design: any failure (offline, GitHub down, unreadable
|
||||
schema) leaves the running catalog untouched and retries after the
|
||||
TTL. Returns True when a fetched document replaced the catalog."""
|
||||
global CATALOG, _last_refresh_attempt
|
||||
|
||||
now = time.monotonic()
|
||||
with _refresh_lock:
|
||||
if not force and now - _last_refresh_attempt < _REFRESH_TTL_S:
|
||||
return False
|
||||
_last_refresh_attempt = now
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
_CATALOG_URL, headers={"User-Agent": "hermes-local-runtime"})
|
||||
with urllib.request.urlopen(req, timeout=10) as r:
|
||||
fetched = _load_catalog(json.load(r))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("catalog refresh skipped: %s", exc)
|
||||
return False
|
||||
if fetched != CATALOG:
|
||||
logger.info("catalog refreshed from repo (%d models)", len(fetched))
|
||||
CATALOG = fetched
|
||||
return True
|
||||
|
||||
|
||||
def refresh_catalog_soon() -> None:
|
||||
"""TTL-gated background refresh; returns immediately. The caller's
|
||||
current request serves the catalog it already has — the refresh
|
||||
lands for the next one."""
|
||||
if time.monotonic() - _last_refresh_attempt < _REFRESH_TTL_S:
|
||||
return
|
||||
threading.Thread(target=refresh_catalog, daemon=True,
|
||||
name="catalog-refresh").start()
|
||||
|
||||
|
||||
def catalog_by_id() -> dict[str, CatalogEntry]:
|
||||
return {entry.id: entry for entry in CATALOG}
|
||||
|
||||
|
||||
def find_variant(entry_id: str, model_id: str) -> QuantVariant | None:
|
||||
entry = catalog_by_id().get(entry_id)
|
||||
if entry is None:
|
||||
return None
|
||||
return next((v for v in entry.variants if v.model_id == model_id), None)
|
||||
|
||||
|
||||
def find_entry_for_model(model_id: str) -> "tuple[CatalogEntry, QuantVariant] | None":
|
||||
"""Locate the entry + variant that owns a staged model id."""
|
||||
for entry in CATALOG:
|
||||
for variant in entry.variants:
|
||||
if variant.model_id == model_id:
|
||||
return entry, variant
|
||||
return None
|
||||
@@ -0,0 +1,286 @@
|
||||
"""Context policy — the window ladder for managed local models.
|
||||
|
||||
One contract: any model runs at any window up to its native max; hardware
|
||||
and session depth only change tokens/s. Constants, not knobs — nothing in
|
||||
this module reads config.
|
||||
|
||||
The policy encodes behavior measured on real hardware (llama.cpp,
|
||||
discrete NVIDIA GPUs on Windows/WDDM, and unified-memory devices):
|
||||
|
||||
- Windows never over-allocates VRAM ahead of need. On WDDM, allocating
|
||||
past residency slows decode roughly 9x even at identical conversation
|
||||
depth — the driver silently demotes pages instead of failing. Every
|
||||
window grant therefore re-fits against live memory at grant time.
|
||||
- Models launch at the largest window that fits entirely in GPU memory
|
||||
(zero-spill) and grow toward their native max as the session needs
|
||||
room, at request boundaries only.
|
||||
- Growth re-prefills the conversation into the larger window. Measured
|
||||
cost is comparable to save/restore on discrete GPUs, and recurrent or
|
||||
hybrid-attention models cannot rewind mid-sequence anyway, so
|
||||
re-prefill is the only mechanism that works for every architecture.
|
||||
- Every recommended model gets at least a 64K window. When weights alone
|
||||
exceed VRAM, the fit deliberately spills weights to host RAM to
|
||||
protect that floor (measured: an explicit context size makes the fit
|
||||
spill weights and hold the window rather than shrink it).
|
||||
- Below ~6 tok/s decode, growth stops and compression becomes the
|
||||
default; deeper context is an explicit per-session choice. The deepest
|
||||
measured host-spilled configuration bottomed out near this rate.
|
||||
- Spilled mixture-of-experts configs pin expert/FFN weights to host so
|
||||
attention and KV stay GPU-resident — measured ~1.75x faster than
|
||||
spilling layers naively at the same host byte count.
|
||||
- Speculative decoding (MTP) defaults on only for spilled configs, where
|
||||
its speedup is largest (measured 1.43x spilled vs 1.35x resident).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget,
|
||||
ModelProfile,
|
||||
PhysicsRefusal,
|
||||
ctx_bytes,
|
||||
physics_check,
|
||||
)
|
||||
|
||||
FLOOR = 64 * 1024 # = target; one internal constant
|
||||
_LADDER_GROWTH = 1.5
|
||||
_GROW_AT_OCCUPANCY = 0.85 # of the current window, at turn boundary
|
||||
SPEED_FLOOR_TOK_S = 6.0 # deepest measured spill bottomed near this
|
||||
_EARLY_COST_CTX_FRACTION = 0.15 # bounded early cost when weights spill
|
||||
|
||||
# TARGET_WINDOW: the smallest ladder rung at which compression becomes the
|
||||
# exception rather than the routine. Measured over 161 real agentic
|
||||
# sessions: 66% complete uncompressed in 64K, 82% in 96K, 91% in 144K —
|
||||
# and the marginal gain past 144K (+6 points for 216K) falls below the
|
||||
# quality cost of stepping down another quant. Quant selection prefers
|
||||
# the best build that reaches this; the FLOOR remains the guarantee.
|
||||
TARGET_WINDOW = 144 * 1024
|
||||
|
||||
# What a load really costs beyond weights + KV: CUDA contexts and compute
|
||||
# buffers at the DEFAULT microbatch (-ub 512, no MTP). Measured on a
|
||||
# 32 GiB card: a model estimated at 29.3 GiB (weights+KV) loaded at
|
||||
# ~31.2 GiB resident and the server's own fit still shaved a layer to
|
||||
# CPU. Microbatch/MTP logits buffers are priced separately per model
|
||||
# (ub_logits_bytes — they scale with the model's vocab and doubled once
|
||||
# packed a card 3.9 GiB past this constant). Callers add mmproj bytes on
|
||||
# top.
|
||||
RUNTIME_OVERHEAD_BYTES = int(1.5 * (1 << 30))
|
||||
|
||||
|
||||
def ladder(native: int) -> list[int]:
|
||||
"""64K -> 96K -> 128K -> ... -> native (native always the last rung)."""
|
||||
rungs: list[int] = []
|
||||
step = float(FLOOR)
|
||||
while step < native:
|
||||
rungs.append(int(step))
|
||||
step *= _LADDER_GROWTH
|
||||
rungs.append(native)
|
||||
return rungs
|
||||
|
||||
|
||||
@dataclass
|
||||
class WindowDecision:
|
||||
window: int
|
||||
spill_bytes: int # weights displaced to host at this window
|
||||
kv_on_gpu: bool
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def spilled(self) -> bool:
|
||||
return self.spill_bytes > 0
|
||||
|
||||
|
||||
def initial_window(profile: ModelProfile, budget: HardwareBudget,
|
||||
*, flash_attention: bool = True,
|
||||
overhead_bytes: int = 0) -> WindowDecision | PhysicsRefusal:
|
||||
"""The launch decision: largest cheap rung, never below the floor.
|
||||
|
||||
Zero-spill rung: weights + ctx + overhead fit usable VRAM entirely.
|
||||
Bounded-early-cost rung: weights already exceed VRAM; take the largest
|
||||
rung whose ctx stays <= ~15% of usable VRAM.
|
||||
Floor everywhere, capped at native.
|
||||
|
||||
``overhead_bytes``: runtime cost beyond weights+KV (RUNTIME_OVERHEAD
|
||||
plus the vision projector when one loads). Zero keeps this function
|
||||
pure physics for decision-table tests; production callers pass it.
|
||||
"""
|
||||
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
|
||||
if refusal:
|
||||
return refusal
|
||||
|
||||
native = profile.n_ctx_train or FLOOR
|
||||
rungs = ladder(native)
|
||||
|
||||
reasons: list[str] = []
|
||||
best_zero_spill: int | None = None
|
||||
for rung in rungs:
|
||||
need = (profile.weights_bytes + overhead_bytes
|
||||
+ ctx_bytes(profile, rung, flash_attention=flash_attention))
|
||||
if need <= budget.usable_vram_bytes:
|
||||
best_zero_spill = rung
|
||||
else:
|
||||
break
|
||||
|
||||
if best_zero_spill is not None and best_zero_spill >= min(FLOOR, native):
|
||||
window = best_zero_spill
|
||||
reasons.append(f"largest zero-spill rung ({window // 1024}K)")
|
||||
else:
|
||||
# Weights spill from turn one (steep-curve model on a small card) —
|
||||
# hold the floor, bound the early ctx cost.
|
||||
cap = int(budget.usable_vram_bytes * _EARLY_COST_CTX_FRACTION)
|
||||
window = min(FLOOR, native)
|
||||
for rung in rungs:
|
||||
if rung < window:
|
||||
continue
|
||||
if ctx_bytes(profile, rung, flash_attention=flash_attention) <= cap:
|
||||
window = rung
|
||||
else:
|
||||
break
|
||||
reasons.append(f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)")
|
||||
|
||||
kv = ctx_bytes(profile, window, flash_attention=flash_attention)
|
||||
spill = max(0, profile.weights_bytes + kv - budget.usable_vram_bytes)
|
||||
return WindowDecision(window=window, spill_bytes=spill,
|
||||
kv_on_gpu=kv <= budget.usable_vram_bytes,
|
||||
reasons=reasons)
|
||||
|
||||
|
||||
@dataclass
|
||||
class GrowthDecision:
|
||||
action: str # "grow" | "hold" | "compress-default"
|
||||
next_window: int | None = None
|
||||
reason: str = ""
|
||||
|
||||
|
||||
def growth_decision(profile: ModelProfile, budget: HardwareBudget, *,
|
||||
current_window: int, session_tokens: int,
|
||||
measured_decode_tok_s: float | None,
|
||||
server_idle: bool,
|
||||
flash_attention: bool = True,
|
||||
occupancy_confirmed: bool = False) -> GrowthDecision:
|
||||
"""One growth evaluation, END-OF-TURN ONLY (caller guarantees the turn
|
||||
boundary; recurrent state cannot rewind mid-sequence).
|
||||
|
||||
Gate ordering:
|
||||
1. occupancy (~85%) — nothing to do before the edge;
|
||||
2. native cap — the contract tops out at trained context;
|
||||
3. idleness — growth re-grants only on an otherwise-idle
|
||||
server (concurrency design);
|
||||
4. speed floor — below it, compression becomes the default and deeper
|
||||
is an explicit user choice;
|
||||
5. re-fit against LIVE free memory (the rung must fit residency
|
||||
NOW, not at launch time — over-allocation is the slow path).
|
||||
|
||||
``occupancy_confirmed``: the caller has independently established that
|
||||
the session is at its window's edge (the agent's compression gate fired
|
||||
on its own threshold). Skips gate 1 so two separately-derived edge
|
||||
definitions can't deadlock into compress-before-grow.
|
||||
"""
|
||||
if not occupancy_confirmed and session_tokens < current_window * _GROW_AT_OCCUPANCY:
|
||||
return GrowthDecision("hold", reason="session below growth occupancy")
|
||||
|
||||
native = profile.n_ctx_train or current_window
|
||||
if current_window >= native:
|
||||
return GrowthDecision("compress-default",
|
||||
reason="at native window; compression is the only move")
|
||||
|
||||
if not server_idle:
|
||||
return GrowthDecision("hold", reason="server busy; re-grant deferred to idle")
|
||||
|
||||
if measured_decode_tok_s is not None and measured_decode_tok_s < SPEED_FLOOR_TOK_S:
|
||||
return GrowthDecision(
|
||||
"compress-default",
|
||||
reason=(f"decode {measured_decode_tok_s:.1f} tok/s below the "
|
||||
f"~{SPEED_FLOOR_TOK_S:.0f} tok/s floor; growth is now an "
|
||||
"explicit per-session choice"))
|
||||
|
||||
next_rung = next((r for r in ladder(native) if r > current_window), native)
|
||||
|
||||
# Re-fit against live free memory: allocation beyond residency is the
|
||||
# slow path, so a rung that no longer fits doesn't get granted.
|
||||
kv = ctx_bytes(profile, next_rung, flash_attention=flash_attention)
|
||||
total_need = profile.weights_bytes + kv
|
||||
if total_need > budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
return GrowthDecision("compress-default",
|
||||
reason="next rung exceeds physics; compression instead")
|
||||
|
||||
return GrowthDecision("grow", next_window=next_rung,
|
||||
reason=f"rung {current_window // 1024}K -> {next_rung // 1024}K")
|
||||
|
||||
|
||||
def spill_overrides(profile: ModelProfile) -> list[str]:
|
||||
"""-ot placement for spilled configs: expert/FFN weights to host so
|
||||
attention + KV stay GPU-resident. MoE gets the expert pattern;
|
||||
hybrids push recurrent-layer FFNs (their n_head_kv==0 layers carry no
|
||||
KV worth protecting)."""
|
||||
if profile.moe:
|
||||
return ["-ot", r"blk\.\d+\.ffn_.*_exps\.weight=CPU"]
|
||||
if profile.recurrent_layer_count:
|
||||
return ["-ot", r"blk\.\d+\.ffn_.*\.weight=CPU"]
|
||||
return [] # dense: fit's back-to-front layer cut is the only axis
|
||||
|
||||
|
||||
def launch_args(profile: ModelProfile, decision: WindowDecision, *,
|
||||
flash_attention: bool = True,
|
||||
mtp_capable: bool = False,
|
||||
mtp_draft_depth: int = 3,
|
||||
uma: bool = False,
|
||||
mtp_prefill: bool = False) -> list[str]:
|
||||
"""Per-model launch flags from a window decision. Explicit -c puts fit
|
||||
into spill-weights-and-hold-ctx; q8 KV cache wherever flash attention
|
||||
exists; -ot placement on spilled configs — DISCRETE cards only.
|
||||
|
||||
``uma``: on unified memory there is no bus to protect tensors from —
|
||||
"CPU" and "GPU" are the same silicon, and pinning FFN weights to the
|
||||
host path just forces CPU compute (measured well over 2x slower than
|
||||
letting the allocator place everything). The discrete
|
||||
~1.75x win the -ot pattern encodes does not transfer; a spilled UMA
|
||||
config runs unpinned.
|
||||
|
||||
MTP and the large prefill microbatch both win, and whether they may
|
||||
STACK is a fit question, not a rule: backend sampling keeps a
|
||||
ubatch x vocab x fp32 logits buffer on the GPU and MTP's draft
|
||||
context doubles it, so the stacked posture costs a few GiB extra at
|
||||
large vocab. Where it fits, it measures best on both axes (Qwen3.8
|
||||
Q4 on a 32 GiB card: 93.3 tok/s decode vs 89.5 at ub512, prefill
|
||||
slightly better too); where it doesn't, ub512 keeps the decode win
|
||||
without packing the card. ``mtp_prefill`` is that fit verdict —
|
||||
presets decide it against the priced margin, and ub_logits_bytes()
|
||||
prices the same choice so the flag and its cost travel together."""
|
||||
args = ["-c", str(decision.window)]
|
||||
if mtp_capable:
|
||||
args += ["--spec-type", "draft-mtp",
|
||||
"--spec-draft-n-max", str(mtp_draft_depth),
|
||||
"--backend-sampling", "--spec-draft-backend-sampling"]
|
||||
if mtp_prefill:
|
||||
args += ["-b", "4096", "-ub", "2048"]
|
||||
else:
|
||||
args += ["-b", "2048", "-ub", "2048"]
|
||||
if flash_attention:
|
||||
args += ["-ctk", "q8_0", "-ctv", "q8_0", "-fa", "on"]
|
||||
if decision.spilled and not uma:
|
||||
args += spill_overrides(profile)
|
||||
return args
|
||||
|
||||
|
||||
def ub_logits_bytes(n_vocab: int, *, mtp_capable: bool,
|
||||
mtp_prefill: bool = False) -> int:
|
||||
"""GPU logits/compute-buffer cost of the microbatch posture chosen by
|
||||
launch_args, priced from the model's own vocab and calibrated against
|
||||
measured server RSS (Qwen3.8 Q4, both postures, three windows):
|
||||
|
||||
stacked (MTP + ub2048): ubatch x vocab x fp32 x 1.5 (~2.9 GiB at
|
||||
248K vocab; fitted 2.5, rounded up)
|
||||
decode (MTP + ub512): ubatch x vocab x fp32 x 2 (~1.0 GiB)
|
||||
plain (ub2048): ubatch x vocab x fp32 (~1.9 GiB)
|
||||
|
||||
Callers add this to RUNTIME_OVERHEAD per model — the flag and its
|
||||
price travel together or the fit lies."""
|
||||
v = max(0, int(n_vocab))
|
||||
if mtp_capable and mtp_prefill:
|
||||
return int(2048 * v * 4 * 1.5)
|
||||
if mtp_capable:
|
||||
return 512 * v * 4 * 2
|
||||
return 2048 * v * 4
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Detection of running llama-server instances.
|
||||
|
||||
Probes well-known local roots and fingerprints genuine llama-server via
|
||||
/props (build_info + model fields — Ollama and LM Studio answer /v1/models
|
||||
but not /props). The credential is reachability; detection never needs a
|
||||
key, but honors one if the probed server requires it (401 -> detected,
|
||||
auth_required=True).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from dataclasses import dataclass
|
||||
|
||||
# Always 127.0.0.1 — resolving localhost costs ~2s/request on Windows.
|
||||
DEFAULT_PROBE_PORTS = (8080,) # llama-server default; managed port comes from config
|
||||
|
||||
|
||||
@dataclass
|
||||
class DetectedServer:
|
||||
base_url: str # OpenAI-compatible /v1 root
|
||||
build_info: str # e.g. "b10290-c8e03ce81"
|
||||
model_path: str # currently loaded model (may be empty in router mode)
|
||||
n_ctx: int | None
|
||||
router_mode: bool # GET /models answered -> router management available
|
||||
auth_required: bool
|
||||
|
||||
|
||||
def _get(url: str, timeout_s: int = 3) -> tuple[int, dict | None]:
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=timeout_s) as r:
|
||||
raw = r.read()
|
||||
return r.status, (json.loads(raw) if raw else None)
|
||||
except urllib.error.HTTPError as exc:
|
||||
return exc.code, None
|
||||
except (urllib.error.URLError, OSError, TimeoutError, json.JSONDecodeError):
|
||||
return 0, None
|
||||
|
||||
|
||||
def probe_port(port: int) -> DetectedServer | None:
|
||||
"""One port: /props fingerprint, then /models for router capability."""
|
||||
root = f"http://127.0.0.1:{port}"
|
||||
status, props = _get(f"{root}/props")
|
||||
if status == 401:
|
||||
return DetectedServer(base_url=f"{root}/v1", build_info="", model_path="",
|
||||
n_ctx=None, router_mode=False, auth_required=True)
|
||||
if status != 200 or not isinstance(props, dict):
|
||||
return None
|
||||
build = str(props.get("build_info", ""))
|
||||
if not build:
|
||||
return None # answers /props but isn't llama-server
|
||||
n_ctx = None
|
||||
dgs = props.get("default_generation_settings")
|
||||
if isinstance(dgs, dict):
|
||||
n_ctx = dgs.get("n_ctx")
|
||||
models_status, models = _get(f"{root}/models")
|
||||
return DetectedServer(
|
||||
base_url=f"{root}/v1",
|
||||
build_info=build,
|
||||
model_path=str(props.get("model_path", "")),
|
||||
n_ctx=n_ctx,
|
||||
router_mode=(models_status == 200 and isinstance(models, dict)
|
||||
and "data" in models),
|
||||
auth_required=False,
|
||||
)
|
||||
|
||||
|
||||
def detect_server(extra_ports: tuple[int, ...] = ()) -> DetectedServer | None:
|
||||
"""First hit across default + extra ports (managed port, config port)."""
|
||||
seen = set()
|
||||
for port in (*DEFAULT_PROBE_PORTS, *extra_ports):
|
||||
if port in seen:
|
||||
continue
|
||||
seen.add(port)
|
||||
hit = probe_port(port)
|
||||
if hit:
|
||||
return hit
|
||||
return None
|
||||
@@ -0,0 +1,194 @@
|
||||
"""Endpoint resolution for llamacpp-alias requests (provider integration).
|
||||
|
||||
The seam between the existing provider mechanism and the managed runtime:
|
||||
``provider: llamacpp`` with no explicit base_url resolves, in order, to
|
||||
|
||||
1. the managed server this Hermes is supervising (state file written by
|
||||
LlamaServerSupervisor.start, removed on stop, staleness-checked), or
|
||||
2. a detected external llama-server.
|
||||
|
||||
Returns None when neither exists — the caller falls through to the normal
|
||||
custom-provider path and its own error reporting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _pid_alive(pid: int) -> bool:
|
||||
"""Liveness for the state file's supervisor-child pid.
|
||||
|
||||
psutil when available; otherwise fall back to True (optimistic) — on
|
||||
Windows ``os.kill(pid, 0)`` TERMINATES the process, so it must never be
|
||||
used as a probe (windows-git-bash interop pitfall).
|
||||
"""
|
||||
if not pid or pid < 0:
|
||||
return False
|
||||
try:
|
||||
import psutil # type: ignore
|
||||
|
||||
return psutil.pid_exists(pid)
|
||||
except Exception: # noqa: BLE001
|
||||
return True
|
||||
|
||||
|
||||
def _state_endpoint() -> dict | None:
|
||||
from hermes_cli.local_runtime.supervisor import state_path
|
||||
|
||||
path = state_path()
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (json.JSONDecodeError, OSError):
|
||||
return None
|
||||
base_url = state.get("base_url", "")
|
||||
if not base_url:
|
||||
return None
|
||||
endpoint = {"base_url": base_url, "api_key": state.get("api_key", "")}
|
||||
# Ownership proof: the stable port means a SECOND install (different
|
||||
# HERMES_HOME — a scratch profile, say) can own 127.0.0.1:18434 with a
|
||||
# different api key while this install's state file still points there.
|
||||
# /health is a public route, so it answers 200 for ANYONE's server —
|
||||
# trusting it alone sent every chat request and the load-progress
|
||||
# watcher at a server that 401s our key, silently. The recorded
|
||||
# supervisor pid is the tiebreaker: health-200 from a server whose
|
||||
# recorded child is DEAD is someone else's server, never a starting one.
|
||||
pid_ok = _pid_alive(int(state.get("pid") or 0))
|
||||
# Healthy server: done (when it's ours).
|
||||
try:
|
||||
health = base_url.rsplit("/v1", 1)[0] + "/health"
|
||||
with urllib.request.urlopen(health, timeout=3) as r:
|
||||
if r.status == 200:
|
||||
return endpoint if pid_ok else None
|
||||
except (urllib.error.URLError, OSError, TimeoutError):
|
||||
pass
|
||||
# Not healthy YET: a live supervisor child is a STARTING server (state
|
||||
# is written at spawn; llama-server takes seconds to listen). Resolve
|
||||
# optimistically so readiness probes racing the boot see a configured
|
||||
# provider, not missing credentials. A dead pid is a crashed-without-
|
||||
# cleanup leftover — ignore it so requests don't blackhole.
|
||||
if pid_ok:
|
||||
return endpoint
|
||||
return None
|
||||
|
||||
|
||||
def resolve_llamacpp_endpoint(config: dict | None = None,
|
||||
wait_for_boot_s: float = 8.0) -> dict | None:
|
||||
"""Managed-first, detection-second endpoint for llamacpp aliases.
|
||||
|
||||
Returns {"base_url", "api_key"} or None. api_key is empty for keyless
|
||||
external servers (callers substitute the SDK placeholder).
|
||||
|
||||
Boot-race rung: on a fresh backend start there is NO state file yet —
|
||||
the lifespan boot thread is still spawning the server (config load +
|
||||
preset generation + spawn ≈ 1-3 s) while the desktop's readiness probe
|
||||
fires the moment the WebSocket connects. When the runtime is enabled
|
||||
and installed, a missing endpoint means BOOTING, not unconfigured:
|
||||
poll briefly for the state file instead of failing the probe (twice
|
||||
observed as 'no usable credentials' → onboarding on restart).
|
||||
"""
|
||||
managed = _state_endpoint()
|
||||
if managed:
|
||||
return managed
|
||||
|
||||
from hermes_cli.local_runtime.detect import detect_server
|
||||
|
||||
extra = ()
|
||||
if config:
|
||||
ports = (config.get("local_runtime") or {}).get("detect_ports") or []
|
||||
extra = tuple(int(p) for p in ports)
|
||||
hit = detect_server(extra_ports=extra)
|
||||
if hit and not hit.auth_required:
|
||||
return {"base_url": hit.base_url, "api_key": ""}
|
||||
|
||||
if wait_for_boot_s > 0 and _boot_in_flight(config):
|
||||
_kick_managed_boot(config)
|
||||
deadline = time.monotonic() + wait_for_boot_s
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(0.25)
|
||||
managed = _state_endpoint()
|
||||
if managed:
|
||||
return managed
|
||||
return None
|
||||
|
||||
|
||||
_KICK_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def _kick_managed_boot(config: dict | None) -> None:
|
||||
"""Actively start the managed server when resolution finds it missing.
|
||||
|
||||
The wait loop above assumes some OTHER thread is bringing the server
|
||||
up — true only at backend start (the lifespan boot thread). A router
|
||||
that dies LATER leaves no boot in flight: the backend process was
|
||||
killed with the router as part of its tree, or another install took
|
||||
the stable port and the ownership guard rightly refused it. In those
|
||||
states the wait just expired and agent init failed with 'no provider
|
||||
configured', even though the fix is the same idempotent ensure call
|
||||
the lifespan makes. Kick it here, off-thread (the resolver's wait
|
||||
stays bounded; ensure's own state checks make a concurrent lifespan
|
||||
boot harmless) and non-reentrant (racing resolutions kick once).
|
||||
"""
|
||||
if not _KICK_LOCK.acquire(blocking=False):
|
||||
return # a kick is already in flight
|
||||
|
||||
def _boot() -> None:
|
||||
try:
|
||||
cfg = config
|
||||
if cfg is None:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
cfg = load_config()
|
||||
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
|
||||
|
||||
ensure_local_runtime(cfg)
|
||||
except Exception: # noqa: BLE001 — best-effort; resolution falls back
|
||||
logger.warning("on-demand managed-server boot failed", exc_info=True)
|
||||
finally:
|
||||
_KICK_LOCK.release()
|
||||
|
||||
threading.Thread(target=_boot, daemon=True,
|
||||
name="lr-on-demand-boot").start()
|
||||
|
||||
|
||||
def _boot_in_flight(config: dict | None) -> bool:
|
||||
"""True when the managed runtime is enabled and installed — the state
|
||||
a lifespan boot thread is (or is about to be) bringing up.
|
||||
|
||||
Installed-ness is a verified-manifest scan under runtimes_root(), NOT a
|
||||
server_binary() call — that helper requires an install_dir argument, and
|
||||
calling it bare made this gate throw-and-return-False forever, silently
|
||||
disabling the boot wait (the regression
|
||||
test had monkeypatched this function instead of exercising it).
|
||||
"""
|
||||
try:
|
||||
if config is None:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
config = load_config()
|
||||
if not ((config or {}).get("local_runtime") or {}).get("enabled"):
|
||||
return False
|
||||
import json as _json
|
||||
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
|
||||
for manifest in runtimes_root().glob("*/*/manifest.json"):
|
||||
try:
|
||||
if _json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
|
||||
return True
|
||||
except (ValueError, OSError):
|
||||
continue
|
||||
return False
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
@@ -0,0 +1,179 @@
|
||||
"""Per-layer context-memory estimator + physics check.
|
||||
|
||||
The whole-model dense formula misprices 1M-context hybrids by ~100x; the
|
||||
per-layer walk fixes that, and every column is measured on real GGUFs:
|
||||
|
||||
- full-attention layer: linear in T (B1: 144.0 KiB/tok on Qwen3-4B
|
||||
f16 — formula-exact)
|
||||
- SWA layer: capped at the sliding window
|
||||
- recurrent layer (n_head_kv == 0): constant (state is ~context-free)
|
||||
- q8_0 KV = exactly 34/64 of f16 (holds on CUDA and CPU)
|
||||
- weights: exact from the tensor table (within 0.01% of the loader)
|
||||
|
||||
The estimator is ADVISORY: fit's allocation is authoritative at launch and
|
||||
the touch generation is ground truth after it. Unknown shapes round UP
|
||||
(never underestimate memory).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
|
||||
from hermes_cli.local_runtime.gguf import GGUFHeader
|
||||
|
||||
# q8_0: 34-byte blocks of 32 f16-equivalent elements (exact).
|
||||
_Q8_BYTES_PER_ELEM = 34 / 32
|
||||
_F16_BYTES_PER_ELEM = 2.0
|
||||
|
||||
# Architectures with a known SWA layer pattern: arch -> fraction of layers
|
||||
# that are sliding-window. Unknown SWA archs conservatively treat every
|
||||
# layer as full attention (overestimate; safe direction).
|
||||
_SWA_LAYER_FRACTION = {"gemma3": 5 / 6, "gemma2": 1 / 2}
|
||||
|
||||
# Per-recurrent-layer state allowance (bytes/seq). Deliberately generous —
|
||||
# Measured: an entire hybrid slot state is ~99 MB including 8K tokens of
|
||||
# full-attn KV, so tens of MiB total is the right order; unknown SSM shapes
|
||||
# must never underestimate.
|
||||
_RECURRENT_STATE_PER_LAYER = 4 << 20
|
||||
|
||||
|
||||
class LayerKind(Enum):
|
||||
FULL = "full"
|
||||
SWA = "swa"
|
||||
RECURRENT = "recurrent"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ModelProfile:
|
||||
"""Everything the policy needs, decoupled from GGUF parsing so the
|
||||
decision-table tests can construct profiles directly (design's
|
||||
verification plan)."""
|
||||
|
||||
name: str
|
||||
weights_bytes: int
|
||||
embd_table_bytes: int
|
||||
n_ctx_train: int
|
||||
layers: list[tuple[LayerKind, int]] # (kind, kv_bytes_per_token_f16);
|
||||
# SWA/recurrent reuse the same
|
||||
# per-token figure, capped/ignored
|
||||
swa_window: int = 0
|
||||
moe: bool = False
|
||||
architecture: str = ""
|
||||
n_vocab: int = 0 # prices logits buffers (ubatch x vocab)
|
||||
# Context-cost multiplier. MTP spec decode keeps a small draft
|
||||
# context beside the main one. Calibrated against four measured
|
||||
# server-RSS points on Qwen3.8 Q4 (128K/221K/256K, both postures):
|
||||
# the draft adds ~17% to per-token KV; 1.2 rounds up so the error
|
||||
# stays on the safe side (+250 MiB at 256K, never negative).
|
||||
kv_scale: float = 1.0
|
||||
|
||||
@property
|
||||
def per_token_kv_f16(self) -> int:
|
||||
"""Uncapped per-token KV cost (full + SWA share)."""
|
||||
return sum(b for kind, b in self.layers if kind != LayerKind.RECURRENT)
|
||||
|
||||
@property
|
||||
def recurrent_layer_count(self) -> int:
|
||||
return sum(1 for kind, _ in self.layers if kind == LayerKind.RECURRENT)
|
||||
|
||||
|
||||
@dataclass
|
||||
class HardwareBudget:
|
||||
"""Memory the physics check may budget against.
|
||||
|
||||
Budget-source rule: discrete cards may trust the device query
|
||||
(measured honest); unified-memory devices must budget from OS free
|
||||
physical memory minus headroom — their device queries have been
|
||||
observed off by 3x. Callers construct
|
||||
this accordingly; the estimator just consumes it.
|
||||
"""
|
||||
|
||||
usable_vram_bytes: int # live free (discrete) / derived (UMA)
|
||||
total_device_bytes: int
|
||||
ram_available_bytes: int
|
||||
uma: bool = False
|
||||
|
||||
|
||||
def profile_from_gguf(header: GGUFHeader) -> ModelProfile:
|
||||
kv_heads = header.head_counts_kv()
|
||||
dk, dv = header.head_dim_k, header.head_dim_v
|
||||
swa_fraction = _SWA_LAYER_FRACTION.get(header.architecture, 0.0)
|
||||
has_swa = header.sliding_window > 0 and swa_fraction > 0
|
||||
|
||||
layers: list[tuple[LayerKind, int]] = []
|
||||
n_attn_seen = 0
|
||||
n_attn_total = sum(1 for h in kv_heads if h > 0)
|
||||
n_swa = round(n_attn_total * swa_fraction) if has_swa else 0
|
||||
for heads in kv_heads:
|
||||
if heads == 0:
|
||||
layers.append((LayerKind.RECURRENT, 0))
|
||||
continue
|
||||
per_token = round(heads * (dk + dv) * _F16_BYTES_PER_ELEM)
|
||||
# Distribute the SWA share across the first n_swa attention layers;
|
||||
# only the full/SWA SPLIT matters to the totals, not which indexes.
|
||||
kind = LayerKind.SWA if n_attn_seen < n_swa else LayerKind.FULL
|
||||
layers.append((kind, per_token))
|
||||
n_attn_seen += 1
|
||||
|
||||
return ModelProfile(
|
||||
name=header.path,
|
||||
weights_bytes=header.tensor_bytes,
|
||||
embd_table_bytes=header.embd_table_bytes,
|
||||
n_ctx_train=header.n_ctx_train,
|
||||
layers=layers,
|
||||
swa_window=header.sliding_window,
|
||||
moe=header.expert_count > 0,
|
||||
architecture=header.architecture,
|
||||
n_vocab=header.n_vocab,
|
||||
)
|
||||
|
||||
|
||||
def kv_dtype_factor(flash_attention: bool) -> float:
|
||||
"""q8_0 with FA (every backend we ship); f16 on exotic non-FA fallbacks
|
||||
— the 64K guarantee stands either way, the physics check just prices
|
||||
the doubled KV (design: KV dtype is behavior, not config)."""
|
||||
return (_Q8_BYTES_PER_ELEM / _F16_BYTES_PER_ELEM) if flash_attention else 1.0
|
||||
|
||||
|
||||
def ctx_bytes(profile: ModelProfile, window: int, *,
|
||||
flash_attention: bool = True) -> int:
|
||||
"""Context memory for one window: full layers linear in T, SWA layers
|
||||
capped at the sliding window, recurrent layers constant. Scaled by
|
||||
profile.kv_scale (MTP draft context)."""
|
||||
factor = kv_dtype_factor(flash_attention)
|
||||
total = 0.0
|
||||
for kind, per_token_f16 in profile.layers:
|
||||
if kind == LayerKind.RECURRENT:
|
||||
total += _RECURRENT_STATE_PER_LAYER
|
||||
elif kind == LayerKind.SWA:
|
||||
total += per_token_f16 * factor * min(window, profile.swa_window)
|
||||
else:
|
||||
total += per_token_f16 * factor * window
|
||||
return int(total * profile.kv_scale)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PhysicsRefusal:
|
||||
"""The only true refusal: weights + floor-KV + state exceed VRAM + RAM.
|
||||
The remedy is a smaller quant, never a smaller window."""
|
||||
|
||||
needed_bytes: int
|
||||
available_bytes: int
|
||||
message: str
|
||||
|
||||
|
||||
def physics_check(profile: ModelProfile, budget: HardwareBudget,
|
||||
floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
|
||||
needed = (profile.weights_bytes
|
||||
+ ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
|
||||
flash_attention=flash_attention))
|
||||
available = budget.usable_vram_bytes + budget.ram_available_bytes
|
||||
if needed > available:
|
||||
gib = 1 << 30
|
||||
return PhysicsRefusal(
|
||||
needed_bytes=needed, available_bytes=available,
|
||||
message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
|
||||
f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
|
||||
"of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
|
||||
return None
|
||||
@@ -0,0 +1,220 @@
|
||||
"""GGUF metadata + tensor-table reader (stdlib only).
|
||||
|
||||
Feeds the per-layer context estimator: architecture, layer count, per-layer
|
||||
KV head counts (0 = recurrent layer — the hybrid discriminator), head dims,
|
||||
sliding-window config, trained context, and exact weight bytes summed from
|
||||
the tensor table (validated to within 0.01% of the loader's buffer).
|
||||
|
||||
Reads the header only (metadata + tensor infos); never touches tensor data,
|
||||
so it is fast enough to run at picker time on multi-GB files.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import struct
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
_GGUF_MAGIC = b"GGUF"
|
||||
|
||||
# ggml tensor type sizes: type_id -> (block_bytes, block_elems).
|
||||
# IQ-family sizes verified against ggml-common.h.
|
||||
_GGML_TYPE_SIZES = {
|
||||
0: (4, 1), 1: (2, 1), 2: (18, 32), 3: (20, 32), 6: (22, 32), 7: (24, 32),
|
||||
8: (34, 32), 9: (36, 32), 10: (84, 256), 11: (110, 256), 12: (144, 256),
|
||||
13: (176, 256), 14: (210, 256), 15: (292, 256), 16: (66, 256),
|
||||
17: (74, 256), 18: (98, 256), 19: (50, 256), 20: (18, 32),
|
||||
21: (110, 256), 22: (82, 256), 23: (136, 256), 24: (1, 1), 25: (2, 1),
|
||||
26: (4, 1), 27: (8, 1), 28: (8, 1), 29: (56, 256), 30: (2, 1),
|
||||
}
|
||||
|
||||
# GGUF metadata value types.
|
||||
_V_UINT8, _V_INT8, _V_UINT16, _V_INT16 = 0, 1, 2, 3
|
||||
_V_UINT32, _V_INT32, _V_FLOAT32, _V_BOOL = 4, 5, 6, 7
|
||||
_V_STRING, _V_ARRAY, _V_UINT64, _V_INT64, _V_FLOAT64 = 8, 9, 10, 11, 12
|
||||
|
||||
_SCALAR_FMT = {
|
||||
_V_UINT8: "<B", _V_INT8: "<b", _V_UINT16: "<H", _V_INT16: "<h",
|
||||
_V_UINT32: "<I", _V_INT32: "<i", _V_FLOAT32: "<f", _V_BOOL: "<?",
|
||||
_V_UINT64: "<Q", _V_INT64: "<q", _V_FLOAT64: "<d",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class GGUFHeader:
|
||||
path: str
|
||||
version: int
|
||||
metadata: dict = field(default_factory=dict)
|
||||
n_tensors: int = 0
|
||||
tensor_bytes: int = 0 # exact sum over the tensor table
|
||||
embd_table_bytes: int = 0 # token_embd.weight (duplicated host-side
|
||||
# when fully offloaded)
|
||||
|
||||
# ── typed accessors ──────────────────────────────────────
|
||||
|
||||
@property
|
||||
def architecture(self) -> str:
|
||||
return str(self.metadata.get("general.architecture", ""))
|
||||
|
||||
def _arch_key(self, suffix: str):
|
||||
return self.metadata.get(f"{self.architecture}.{suffix}")
|
||||
|
||||
@property
|
||||
def n_layer(self) -> int:
|
||||
return int(self._arch_key("block_count") or 0)
|
||||
|
||||
@property
|
||||
def n_vocab(self) -> int:
|
||||
"""Vocabulary size: prices the GPU logits buffers (they scale
|
||||
ubatch x vocab). vocab_size metadata when present, else the
|
||||
tokenizer list length."""
|
||||
v = self._arch_key("vocab_size")
|
||||
if v:
|
||||
return int(v)
|
||||
toks = self.metadata.get("tokenizer.ggml.tokens")
|
||||
return len(toks) if isinstance(toks, list) else 0
|
||||
|
||||
@property
|
||||
def n_ctx_train(self) -> int:
|
||||
return int(self._arch_key("context_length") or 0)
|
||||
|
||||
@property
|
||||
def sampling_defaults(self) -> dict:
|
||||
"""Upstream's recommended sampling, when the file carries it.
|
||||
|
||||
Model publishers bake general.sampling.* keys into the GGUF
|
||||
(llama-server reads them as that model's default generation
|
||||
settings), so the file itself is the source of truth for how its
|
||||
publisher wants it run — it arrives with the download and updates
|
||||
with every re-upload, no catalog required. Returned as preset INI
|
||||
keys; empty when the file carries none.
|
||||
"""
|
||||
ini_key = {"temp": "temp", "temperature": "temp", "top_p": "top-p",
|
||||
"top_k": "top-k", "min_p": "min-p",
|
||||
"repeat_penalty": "repeat-penalty",
|
||||
"presence_penalty": "presence-penalty"}
|
||||
out = {}
|
||||
for key, value in self.metadata.items():
|
||||
if not key.startswith("general.sampling."):
|
||||
continue
|
||||
name = ini_key.get(key.rsplit(".", 1)[-1])
|
||||
if name is not None and isinstance(value, (int, float)):
|
||||
num = round(float(value), 4)
|
||||
out[name] = str(int(num)) if num == int(num) else str(num)
|
||||
return out
|
||||
|
||||
@property
|
||||
def n_embd(self) -> int:
|
||||
return int(self._arch_key("embedding_length") or 0)
|
||||
|
||||
@property
|
||||
def n_head(self) -> int:
|
||||
v = self._arch_key("attention.head_count")
|
||||
if isinstance(v, list):
|
||||
return int(max(v))
|
||||
return int(v or 0)
|
||||
|
||||
@property
|
||||
def full_attention_interval(self) -> int:
|
||||
"""GDN-hybrid discriminator (qwen35 family): every Nth layer is full
|
||||
attention, the rest are linear/recurrent. 0 = not present."""
|
||||
return int(self._arch_key("full_attention_interval") or 0)
|
||||
|
||||
def head_counts_kv(self) -> list[int]:
|
||||
"""Per-layer KV head counts; 0 marks a recurrent/linear layer (the
|
||||
n_head_kv == 0 discriminator).
|
||||
|
||||
Three GGUF shapes, each verified against real files:
|
||||
- per-layer array (nemotron_h_moe): use as-is;
|
||||
- scalar + full_attention_interval (qwen35): the scalar applies to
|
||||
every INTERVAL-th layer (1-indexed: layers where (i+1) % N == 0),
|
||||
zero elsewhere — pricing all layers as attention was a 4x
|
||||
overestimate on Qwen3.6-27B;
|
||||
- plain scalar (dense): broadcast to every layer.
|
||||
"""
|
||||
v = self._arch_key("attention.head_count_kv")
|
||||
if isinstance(v, list):
|
||||
return [int(x) for x in v]
|
||||
scalar = int(v or 0)
|
||||
interval = self.full_attention_interval
|
||||
if interval > 1:
|
||||
return [scalar if (i + 1) % interval == 0 else 0
|
||||
for i in range(self.n_layer)]
|
||||
return [scalar] * self.n_layer
|
||||
|
||||
@property
|
||||
def head_dim_k(self) -> int:
|
||||
v = self._arch_key("attention.key_length")
|
||||
if v:
|
||||
return int(v)
|
||||
return self.n_embd // self.n_head if self.n_head else 0
|
||||
|
||||
@property
|
||||
def head_dim_v(self) -> int:
|
||||
v = self._arch_key("attention.value_length")
|
||||
if v:
|
||||
return int(v)
|
||||
return self.head_dim_k
|
||||
|
||||
@property
|
||||
def sliding_window(self) -> int:
|
||||
return int(self._arch_key("attention.sliding_window") or 0)
|
||||
|
||||
@property
|
||||
def expert_count(self) -> int:
|
||||
return int(self._arch_key("expert_count") or 0)
|
||||
|
||||
|
||||
def read_gguf_header(path: str | Path) -> GGUFHeader:
|
||||
path = Path(path)
|
||||
|
||||
def read_str(f) -> str:
|
||||
(n,) = struct.unpack("<Q", f.read(8))
|
||||
return f.read(n).decode("utf-8", errors="replace")
|
||||
|
||||
def read_value(f, vtype: int):
|
||||
if vtype == _V_STRING:
|
||||
return read_str(f)
|
||||
if vtype == _V_ARRAY:
|
||||
(etype,) = struct.unpack("<I", f.read(4))
|
||||
(n,) = struct.unpack("<Q", f.read(8))
|
||||
return [read_value(f, etype) for _ in range(n)]
|
||||
fmt = _SCALAR_FMT[vtype]
|
||||
(value,) = struct.unpack(fmt, f.read(struct.calcsize(fmt)))
|
||||
return value
|
||||
|
||||
with open(path, "rb") as f:
|
||||
if f.read(4) != _GGUF_MAGIC:
|
||||
raise ValueError(f"not a GGUF file: {path}")
|
||||
(version,) = struct.unpack("<I", f.read(4))
|
||||
n_tensors, n_kv = struct.unpack("<QQ", f.read(16))
|
||||
|
||||
metadata: dict = {}
|
||||
for _ in range(n_kv):
|
||||
key = read_str(f)
|
||||
(vtype,) = struct.unpack("<I", f.read(4))
|
||||
metadata[key] = read_value(f, vtype)
|
||||
|
||||
tensor_bytes = 0
|
||||
embd_bytes = 0
|
||||
for _ in range(n_tensors):
|
||||
name = read_str(f)
|
||||
(n_dims,) = struct.unpack("<I", f.read(4))
|
||||
dims = struct.unpack(f"<{n_dims}Q", f.read(8 * n_dims))
|
||||
(ttype,) = struct.unpack("<I", f.read(4))
|
||||
f.read(8) # offset
|
||||
size = _GGML_TYPE_SIZES.get(ttype)
|
||||
if size is None:
|
||||
raise ValueError(f"unknown ggml tensor type {ttype} in {path}")
|
||||
block_bytes, block_elems = size
|
||||
elems = 1
|
||||
for d in dims:
|
||||
elems *= d
|
||||
nbytes = (elems // block_elems) * block_bytes
|
||||
tensor_bytes += nbytes
|
||||
if name == "token_embd.weight":
|
||||
embd_bytes = nbytes
|
||||
|
||||
return GGUFHeader(path=str(path), version=version, metadata=metadata,
|
||||
n_tensors=n_tensors, tensor_bytes=tensor_bytes,
|
||||
embd_table_bytes=embd_bytes)
|
||||
@@ -0,0 +1,143 @@
|
||||
"""In-session context growth for the managed llama.cpp runtime.
|
||||
|
||||
The live half of the window ladder (context_policy.growth_decision): when a
|
||||
session reaches the edge of its granted window, Hermes grows the window
|
||||
toward the model's native max INSTEAD of compressing. Compression becomes
|
||||
what the design says it is — the move of last resort, once the window is at
|
||||
native (or the speed floor / physics say stop).
|
||||
|
||||
Mechanism: growth is re-prefill. A per-model window
|
||||
override is persisted, presets regenerate with the bigger window, the
|
||||
supervised server bounces, and the next request autoloads the model at the
|
||||
new window and re-prefills the conversation. Nothing about the Hermes
|
||||
conversation mutates — no prompt-cache or role-alternation risk; the whole
|
||||
operation is server-side.
|
||||
|
||||
Scope guard: only a server THIS process supervises grows. Detected external
|
||||
servers and other-process supervisors keep their own policies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def window_overrides_path():
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
|
||||
return runtimes_root() / "window_overrides.json"
|
||||
|
||||
|
||||
def load_window_overrides() -> dict:
|
||||
"""model_id -> granted window (int). Empty on any read problem."""
|
||||
try:
|
||||
with open(window_overrides_path(), encoding="utf-8") as fh:
|
||||
data = json.load(fh)
|
||||
return {str(k): int(v) for k, v in data.items()}
|
||||
except Exception: # noqa: BLE001
|
||||
return {}
|
||||
|
||||
|
||||
def save_window_override(model_id: str, window: int) -> None:
|
||||
overrides = load_window_overrides()
|
||||
overrides[model_id] = int(window)
|
||||
path = window_overrides_path()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(overrides, indent=1), encoding="utf-8")
|
||||
|
||||
|
||||
def clear_window_override(model_id: str) -> None:
|
||||
"""Drop a model's growth state (delete/re-download paths)."""
|
||||
overrides = load_window_overrides()
|
||||
if model_id in overrides:
|
||||
del overrides[model_id]
|
||||
window_overrides_path().write_text(
|
||||
json.dumps(overrides, indent=1), encoding="utf-8")
|
||||
|
||||
|
||||
def is_managed_endpoint(base_url: str) -> bool:
|
||||
"""True when base_url is the server this process's state file points at."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
state = _state_endpoint()
|
||||
if state is None:
|
||||
return False
|
||||
return (base_url or "").rstrip("/") == str(
|
||||
state.get("base_url", "")).rstrip("/")
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
|
||||
def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
current_window: int,
|
||||
measured_decode_tok_s: float | None = None) -> int | None:
|
||||
"""One growth evaluation + execution. Returns the NEW window when the
|
||||
ladder granted a bigger one, else None (hold / compress / not ours).
|
||||
|
||||
The caller sits at a request boundary by construction (the pre-API
|
||||
compression gate), so re-prefill growth is safe at any call: the next
|
||||
request rebuilds server state from scratch in the larger window —
|
||||
nothing rewinds.
|
||||
"""
|
||||
from hermes_cli.local_runtime.bootstrap import (
|
||||
get_supervisor,
|
||||
refresh_local_runtime,
|
||||
staged_models,
|
||||
)
|
||||
from hermes_cli.local_runtime.context_policy import growth_decision
|
||||
from hermes_cli.local_runtime.estimator import profile_from_gguf
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header
|
||||
from hermes_cli.local_runtime.hardware import probe_budget
|
||||
|
||||
sup = get_supervisor()
|
||||
if sup is None or not is_managed_endpoint(base_url):
|
||||
return None
|
||||
|
||||
gguf = next((p for p in staged_models()
|
||||
if p.stem.startswith(model_id) or model_id in p.stem), None)
|
||||
if gguf is None:
|
||||
return None
|
||||
|
||||
try:
|
||||
profile = profile_from_gguf(read_gguf_header(gguf))
|
||||
except (ValueError, OSError) as exc:
|
||||
logger.debug("growth skip %s: unreadable gguf (%s)", model_id, exc)
|
||||
return None
|
||||
|
||||
try:
|
||||
server_idle = sup.is_idle(model_id)
|
||||
except Exception: # noqa: BLE001
|
||||
server_idle = False
|
||||
|
||||
decision = growth_decision(
|
||||
# Capacity budget, not live-free: growth executes via a server
|
||||
# bounce, so the grown instance loads onto a freed card. Live-free
|
||||
# here is distorted by the very model being grown — it reads its
|
||||
# own residency as unavailable and vetoes rungs that fit.
|
||||
profile, probe_budget(planning=True),
|
||||
current_window=current_window,
|
||||
session_tokens=session_tokens,
|
||||
measured_decode_tok_s=measured_decode_tok_s,
|
||||
server_idle=server_idle,
|
||||
# The caller IS the occupancy signal: this runs from the agent's
|
||||
# compression gate, which fired on its own threshold. Two
|
||||
# separately-derived edges must not deadlock into
|
||||
# compress-before-grow.
|
||||
occupancy_confirmed=True,
|
||||
)
|
||||
if decision.action != "grow" or not decision.next_window:
|
||||
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
|
||||
return None
|
||||
|
||||
logger.info("context growth %s: %s", model_id, decision.reason)
|
||||
save_window_override(model_id, decision.next_window)
|
||||
if not refresh_local_runtime():
|
||||
# The override still lands at the next boot; report no growth NOW
|
||||
# so the caller compresses instead of overflowing a stale window.
|
||||
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
|
||||
return None
|
||||
return decision.next_window
|
||||
@@ -0,0 +1,379 @@
|
||||
"""Live hardware budget probe.
|
||||
|
||||
Budget-source rule: discrete cards may trust the device query (measured
|
||||
honest within rounding); unified-memory devices must budget from OS free
|
||||
physical memory minus headroom — their device queries have been observed
|
||||
off by 3x in both directions. The probe classifies the device and
|
||||
constructs the right HardwareBudget for the estimator.
|
||||
|
||||
Vendor probe quirk (WDDM carve-out): on unified-memory NVIDIA devices
|
||||
under Windows, nvidia-smi answers from the legacy dedicated-VRAM
|
||||
carve-out — a fraction of the pool the CUDA allocator actually
|
||||
addresses uniformly at full bandwidth. The CUDA driver API is
|
||||
the tiebreaker: cuDeviceGetAttribute(INTEGRATED) is the vendor's own
|
||||
declaration and always wins — 1 budgets unified, 0 stays discrete no
|
||||
matter what any other number says. Only when the driver API is
|
||||
unreachable does the engine's --list-devices view apply, and then only
|
||||
behind two independent conditions no discrete card can meet.
|
||||
|
||||
Every probe here must work under a stripped PATH — gateway and service
|
||||
sessions don't inherit the interactive environment. nvcuda/libcuda load
|
||||
through the system loader (PATH plays no part), so classification never
|
||||
depends on PATH; nvidia-smi resolves through an explicit candidate
|
||||
ladder (PATH first, then the driver's known install locations) and its
|
||||
absence only softens the live number, never the verdict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_GIB = 1 << 30
|
||||
# Reserve carved off the card before any grant: the desktop's own
|
||||
# co-residents (compositor, browser, Electron) measure ~2-2.5 GiB on a
|
||||
# working machine, and a window granted into that space demotes silently
|
||||
# under WDDM. 7% covers big cards; the 2 GiB floor is what the margin's
|
||||
# old 512 MiB floor failed to cover in practice (a 221K grant measured
|
||||
# 31.9/32.6 GiB with the desktop running — 'fits' by the math, demoted
|
||||
# in reality). Small cards give up window to this; spill mode is their
|
||||
# path to big models regardless.
|
||||
_MARGIN_FLOOR = 2 << 30
|
||||
_MARGIN_FRACTION = 0.09
|
||||
# UMA headroom: on unified-memory machines (Apple Silicon, unified-memory
|
||||
# NVIDIA) the model shares physical memory with the OS and every app, so
|
||||
# budget from RAM minus this fraction.
|
||||
_UMA_HEADROOM_FRACTION = 0.20
|
||||
|
||||
# Engine-fallback gates for the unified-pool quirk — BOTH must hold, and
|
||||
# no discrete card can meet either: (1) the allocator's pool exceeds the
|
||||
# smi report by well past rounding/ECC slack (discrete cards agree within
|
||||
# ~2%; carve-out disagreement runs to whole multiples), and (2) the pool is
|
||||
# system-RAM-sized — a workstation card in a RAM-matched box fails (1)
|
||||
# because its smi and allocator AGREE, and a big discrete card in a
|
||||
# bigger box fails (2). The driver's INTEGRATED attribute, when
|
||||
# readable, bypasses both gates in whichever direction it points.
|
||||
_POOL_DISAGREEMENT_FACTOR = 1.5
|
||||
_POOL_RAM_FRACTION = 0.75
|
||||
|
||||
# cuDeviceGetAttribute enum: device is integrated with host memory.
|
||||
_CU_DEVICE_ATTRIBUTE_INTEGRATED = 18
|
||||
|
||||
# One probe per process once a device answers (silicon doesn't change);
|
||||
# a miss retries after this long so a runtime installed mid-session gets
|
||||
# picked up by the engine fallback.
|
||||
_POOL_NEGATIVE_TTL_S = 60.0
|
||||
_pool_probe_cache: tuple[float, "tuple[int, bool | None] | None"] | None = None
|
||||
|
||||
# ' CUDA0: NVIDIA Example Device (1234-core Example GPU) (46464 MiB, 46284 MiB free)'
|
||||
# — greedy .* pins the LAST parenthesized group, so device names carrying
|
||||
# their own parentheses parse correctly.
|
||||
_DEVICE_LINE_RE = re.compile(r"CUDA\d+:.*\((\d+)\s*MiB,\s*\d+\s*MiB free\)\s*$")
|
||||
|
||||
|
||||
def _ram_bytes() -> tuple[int, int]:
|
||||
"""(total, available) physical memory, cross-platform stdlib."""
|
||||
try:
|
||||
import ctypes
|
||||
|
||||
class MEMORYSTATUSEX(ctypes.Structure):
|
||||
_fields_ = [("dwLength", ctypes.c_ulong),
|
||||
("dwMemoryLoad", ctypes.c_ulong),
|
||||
("ullTotalPhys", ctypes.c_ulonglong),
|
||||
("ullAvailPhys", ctypes.c_ulonglong),
|
||||
("ullTotalPageFile", ctypes.c_ulonglong),
|
||||
("ullAvailPageFile", ctypes.c_ulonglong),
|
||||
("ullTotalVirtual", ctypes.c_ulonglong),
|
||||
("ullAvailVirtual", ctypes.c_ulonglong),
|
||||
("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
|
||||
|
||||
stat = MEMORYSTATUSEX()
|
||||
stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
|
||||
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat))
|
||||
return stat.ullTotalPhys, stat.ullAvailPhys
|
||||
except (AttributeError, OSError):
|
||||
pass
|
||||
if sys.platform == "darwin":
|
||||
# macOS getconf has no _PHYS_PAGES/_AVPHYS_PAGES (exit 64, "no such
|
||||
# configuration parameter") — the POSIX branch below returns (0, 0)
|
||||
# and every model reads unavailable. sysctl is the platform truth.
|
||||
try:
|
||||
total = int(subprocess.run(
|
||||
["/usr/sbin/sysctl", "-n", "hw.memsize"],
|
||||
capture_output=True, text=True, timeout=5).stdout.strip() or 0)
|
||||
if total <= 0:
|
||||
return 0, 0
|
||||
avail = total // 2 # conservative fallback
|
||||
try:
|
||||
out = subprocess.run(["/usr/bin/vm_stat"], capture_output=True,
|
||||
text=True, timeout=5).stdout
|
||||
page_m = re.search(r"page size of (\d+)", out)
|
||||
page = int(page_m.group(1)) if page_m else 16384
|
||||
pages = 0
|
||||
# free + inactive + purgeable ≈ reclaimable-on-demand; the
|
||||
# speculative pool is dropped by the OS under pressure too.
|
||||
for key in ("Pages free", "Pages inactive", "Pages purgeable",
|
||||
"Pages speculative"):
|
||||
m = re.search(rf"{key}:\s+(\d+)\.", out)
|
||||
if m:
|
||||
pages += int(m.group(1))
|
||||
if pages > 0:
|
||||
avail = pages * page
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
return total, avail
|
||||
except (OSError, ValueError):
|
||||
return 0, 0
|
||||
# POSIX
|
||||
try:
|
||||
page = int(subprocess.run(["getconf", "PAGE_SIZE"], capture_output=True,
|
||||
text=True, timeout=5).stdout or 4096)
|
||||
total = int(subprocess.run(["getconf", "_PHYS_PAGES"], capture_output=True,
|
||||
text=True, timeout=5).stdout or 0) * page
|
||||
avail = total // 2 # conservative when _AVPHYS is unavailable
|
||||
try:
|
||||
avail = int(subprocess.run(["getconf", "_AVPHYS_PAGES"],
|
||||
capture_output=True, text=True,
|
||||
timeout=5).stdout or 0) * page or avail
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
return total, avail
|
||||
except (OSError, ValueError):
|
||||
return 0, 0
|
||||
|
||||
|
||||
# nvidia-smi lives at a fixed path under the driver install; PATH presence
|
||||
# varies by session type (services and gateways often run with a minimal
|
||||
# environment) and by driver generation (legacy NVSMI dir was never on
|
||||
# PATH). Resolution result is cached: the driver doesn't move mid-process.
|
||||
_smi_path_cache: "tuple[str | None] | None" = None
|
||||
|
||||
|
||||
def _nvidia_smi_path() -> str | None:
|
||||
"""Absolute path to nvidia-smi, or None. PATH first (respects user
|
||||
overrides), then the driver's known install locations on Windows;
|
||||
on Linux/WSL the PATH lookup is the whole ladder."""
|
||||
global _smi_path_cache
|
||||
if _smi_path_cache is not None:
|
||||
return _smi_path_cache[0]
|
||||
found = shutil.which("nvidia-smi")
|
||||
if found is None and os.name == "nt":
|
||||
windir = os.environ.get("SystemRoot", r"C:\Windows")
|
||||
for candidate in (
|
||||
# DCH drivers (every modern install) place it in System32.
|
||||
Path(windir) / "System32" / "nvidia-smi.exe",
|
||||
# Legacy standalone drivers used NVSMI, never on PATH.
|
||||
Path(os.environ.get("ProgramFiles", r"C:\Program Files"))
|
||||
/ "NVIDIA Corporation" / "NVSMI" / "nvidia-smi.exe",
|
||||
):
|
||||
if candidate.exists():
|
||||
found = str(candidate)
|
||||
break
|
||||
_smi_path_cache = (found,)
|
||||
return found
|
||||
|
||||
|
||||
def _nvidia_vram() -> tuple[int, int] | None:
|
||||
"""(total, free) MiB->bytes from nvidia-smi, or None."""
|
||||
exe = _nvidia_smi_path()
|
||||
if exe is None:
|
||||
return None
|
||||
try:
|
||||
out = subprocess.run(
|
||||
[exe, "--query-gpu=memory.total,memory.free",
|
||||
"--format=csv,noheader,nounits"],
|
||||
capture_output=True, text=True, timeout=10)
|
||||
if out.returncode != 0 or not out.stdout.strip():
|
||||
return None
|
||||
total_mib, free_mib = (int(x) for x in out.stdout.strip().splitlines()[0].split(","))
|
||||
return total_mib << 20, free_mib << 20
|
||||
except (OSError, ValueError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
|
||||
|
||||
def _cuda_driver_pool() -> "tuple[int, bool | None] | None":
|
||||
"""(allocator_total_bytes, integrated_or_None) from the CUDA driver
|
||||
API, or None when unreachable. ctypes against the driver's own DLL/SO
|
||||
— no toolkit, no subprocess, ~ms. INTEGRATED is the vendor's own
|
||||
unified-memory declaration; total is the pool the allocator will
|
||||
actually hand out (on carve-out devices, several times what
|
||||
nvidia-smi reports)."""
|
||||
import ctypes
|
||||
|
||||
for name in ("nvcuda.dll", "libcuda.so.1", "libcuda.so"):
|
||||
try:
|
||||
cuda = ctypes.CDLL(name)
|
||||
break
|
||||
except OSError:
|
||||
continue
|
||||
else:
|
||||
return None
|
||||
try:
|
||||
if cuda.cuInit(0) != 0:
|
||||
return None
|
||||
dev = ctypes.c_int()
|
||||
if cuda.cuDeviceGet(ctypes.byref(dev), 0) != 0:
|
||||
return None
|
||||
total = ctypes.c_size_t()
|
||||
getter = getattr(cuda, "cuDeviceTotalMem_v2", None) or cuda.cuDeviceTotalMem
|
||||
if getter(ctypes.byref(total), dev) != 0 or total.value <= 0:
|
||||
return None
|
||||
integrated: bool | None = None
|
||||
attr = ctypes.c_int()
|
||||
if cuda.cuDeviceGetAttribute(
|
||||
ctypes.byref(attr), _CU_DEVICE_ATTRIBUTE_INTEGRATED, dev) == 0:
|
||||
integrated = bool(attr.value)
|
||||
return total.value, integrated
|
||||
except (OSError, AttributeError):
|
||||
return None
|
||||
|
||||
|
||||
def _engine_device_pool() -> "tuple[int, bool | None] | None":
|
||||
"""(engine_total_bytes, None) from the installed runtime's own
|
||||
--list-devices, or None. The fallback truth source when the driver
|
||||
API is unreachable: asks the exact binary that will do the
|
||||
allocating. Carries no integrated verdict — callers must gate it."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.binaries import (
|
||||
installed_tags,
|
||||
runtimes_root,
|
||||
server_binary,
|
||||
)
|
||||
|
||||
tags = installed_tags()
|
||||
if not tags:
|
||||
return None
|
||||
tag_dir = runtimes_root() / tags[0]
|
||||
backend_dirs = [d for d in tag_dir.iterdir() if d.is_dir()]
|
||||
if not backend_dirs:
|
||||
return None
|
||||
exe = server_binary(backend_dirs[0])
|
||||
out = subprocess.run([str(exe), "--list-devices"], capture_output=True,
|
||||
text=True, timeout=30, cwd=str(exe.parent))
|
||||
if out.returncode != 0:
|
||||
return None
|
||||
for line in (out.stdout + out.stderr).splitlines():
|
||||
m = _DEVICE_LINE_RE.search(line)
|
||||
if m:
|
||||
return int(m.group(1)) << 20, None
|
||||
return None
|
||||
except Exception: # noqa: BLE001 — a probe miss must never block budgeting
|
||||
return None
|
||||
|
||||
|
||||
def _device_pool_view() -> "tuple[int, bool | None] | None":
|
||||
"""Best available allocator-side view, cached: a hit is permanent for
|
||||
the process, a miss retries after a short TTL (the engine binary can
|
||||
appear mid-session via a pane install)."""
|
||||
global _pool_probe_cache
|
||||
now = time.monotonic()
|
||||
if _pool_probe_cache is not None:
|
||||
stamp, view = _pool_probe_cache
|
||||
if view is not None or now - stamp < _POOL_NEGATIVE_TTL_S:
|
||||
return view
|
||||
view = _cuda_driver_pool() or _engine_device_pool()
|
||||
_pool_probe_cache = (now, view)
|
||||
return view
|
||||
|
||||
|
||||
def _unified_pool_bytes(smi_total: int, ram_total: int) -> int | None:
|
||||
"""The real pool size when this NVIDIA device is unified memory behind
|
||||
a WDDM carve-out, else None (trust nvidia-smi as ever).
|
||||
|
||||
The driver's INTEGRATED attribute decides when readable — in BOTH
|
||||
directions (0 pins discrete even if the numbers look weird; a driver
|
||||
that declares integrated is believed even at modest pool sizes). Only
|
||||
an attribute-less view (engine fallback) needs the two numeric gates;
|
||||
both must hold and no discrete card meets either.
|
||||
"""
|
||||
view = _device_pool_view()
|
||||
if view is None:
|
||||
return None
|
||||
pool, integrated = view
|
||||
if integrated is False:
|
||||
return None
|
||||
if integrated is True:
|
||||
return pool
|
||||
if (smi_total > 0 and pool >= int(smi_total * _POOL_DISAGREEMENT_FACTOR)
|
||||
and ram_total > 0 and pool >= int(ram_total * _POOL_RAM_FRACTION)):
|
||||
return pool
|
||||
return None
|
||||
|
||||
|
||||
def probe_budget(*, planning: bool = False) -> HardwareBudget:
|
||||
"""Construct the budget per the source rules above.
|
||||
|
||||
``planning=False`` (default): LIVE budget — free VRAM right now. The
|
||||
right input for launch-time fit decisions and growth re-grants.
|
||||
|
||||
``planning=True``: CAPACITY budget — what this machine can run once
|
||||
the runtime manages placement (total device memory minus the margin).
|
||||
The right input for catalog pricing and quant selection: pricing
|
||||
against live-free while a model is already loaded made every row read
|
||||
'larger than your GPU memory' and degraded quant picks to Q2 on a
|
||||
32 GiB card. The managed server
|
||||
unloads/relaunches models itself, so at load time the capacity is
|
||||
genuinely available.
|
||||
"""
|
||||
ram_total, ram_avail = _ram_bytes()
|
||||
vram = _nvidia_vram()
|
||||
|
||||
# Unified-memory NVIDIA: the CUDA allocator pool is the real
|
||||
# capacity. Classification comes from the driver API/engine — it
|
||||
# must not require nvidia-smi (stripped-PATH sessions lose smi but
|
||||
# nvcuda loads via the system loader regardless). Crossing the
|
||||
# carve-out costs nothing (effective bandwidth is flat through the
|
||||
# boundary; smi's used/total merely saturate at it) — the carve-out
|
||||
# is an OS accounting knob, not a GPU limit. Deliberately NOT
|
||||
# clamped to OS RAM: carved-out memory is invisible to
|
||||
# GlobalMemoryStatusEx (the OS reports correspondingly less total
|
||||
# RAM), so a RAM clamp would throw away exactly the carved capacity.
|
||||
unified = _unified_pool_bytes(vram[0] if vram else 0, ram_total)
|
||||
if unified is not None:
|
||||
logger.info(
|
||||
"unified-memory NVIDIA device: allocator pool %.1f GiB "
|
||||
"(nvidia-smi carve-out: %s); budgeting from the pool",
|
||||
unified / _GIB,
|
||||
f"{vram[0] / _GIB:.1f} GiB" if vram else "unavailable")
|
||||
if planning:
|
||||
base = unified
|
||||
else:
|
||||
# Live: dedicated-free plus what the OS can still give. smi's
|
||||
# free saturates at the carve-out so this under-counts a bit —
|
||||
# the safe direction (the pool edge is a measured soft cliff:
|
||||
# decode collapses ~3.5x when concurrent demand hits it).
|
||||
# Without smi, OS-available alone is the honest floor.
|
||||
live = (vram[1] + ram_avail) if vram else ram_avail
|
||||
base = min(unified, live)
|
||||
usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
|
||||
return HardwareBudget(usable_vram_bytes=usable,
|
||||
total_device_bytes=unified,
|
||||
ram_available_bytes=0, uma=True)
|
||||
|
||||
if vram is None:
|
||||
# No NVIDIA device visible: Metal/Vulkan/CPU paths budget from RAM
|
||||
# as UMA (Apple Silicon) — conservative for discrete AMD until a
|
||||
# vendor probe lands (E3 hardware).
|
||||
base = ram_total if planning else ram_avail
|
||||
usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
|
||||
return HardwareBudget(usable_vram_bytes=usable,
|
||||
total_device_bytes=ram_total,
|
||||
ram_available_bytes=0, uma=True)
|
||||
|
||||
total, free = vram
|
||||
margin = max(_MARGIN_FLOOR, int(total * _MARGIN_FRACTION))
|
||||
base = total if planning else free
|
||||
return HardwareBudget(usable_vram_bytes=max(0, base - margin),
|
||||
total_device_bytes=total,
|
||||
ram_available_bytes=ram_avail if not planning else ram_total,
|
||||
uma=False)
|
||||
@@ -0,0 +1,161 @@
|
||||
"""Browse Hugging Face for GGUF models the user can run.
|
||||
|
||||
The curated catalog is the front page; this module is the firehose behind
|
||||
it — day-0 models not yet in the catalog, community quants,
|
||||
anything. Three rules keep it safe and honest:
|
||||
|
||||
1. Acquisition only. Nothing here serves a model: a browsed download
|
||||
lands in the machine-scoped models dir and from that moment the
|
||||
normal machinery owns it — staleness bounce, preset generation from
|
||||
the real GGUF header, fit policy, placement pills.
|
||||
2. The fit verdict shown BEFORE download is a rough cut priced from file
|
||||
size alone (weights dominate; KV/overhead use conservative fill-ins).
|
||||
After download the GGUF header is the authority, as everywhere.
|
||||
3. HF is queried directly with short timeouts and a small in-process
|
||||
cache. No third-party proxy service; if HF rate limits ever bite at
|
||||
fleet scale, revisit with a caching proxy then.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_HF = "https://huggingface.co"
|
||||
_TIMEOUT_S = 15
|
||||
# Rough-fit fill-ins for pre-download pricing: a mid-size model's 64K-floor
|
||||
# KV plus runtime overhead. Deliberately round numbers — the verdict bands
|
||||
# are coarse (fits GPU / needs RAM / too big), not window grants.
|
||||
_ROUGH_KV_AND_OVERHEAD = 4 << 30
|
||||
|
||||
# Tiny TTL cache: the pane fires a search per keystroke pause and re-opens
|
||||
# repos the user flips between. Process-local, size-capped, no invalidation
|
||||
# subtleties — upstream truth changes slowly at this granularity.
|
||||
_CACHE: dict[str, tuple[float, object]] = {}
|
||||
_CACHE_TTL_S = 300
|
||||
_CACHE_MAX = 128
|
||||
|
||||
|
||||
def _get_json(url: str) -> object:
|
||||
now = time.monotonic()
|
||||
hit = _CACHE.get(url)
|
||||
if hit and now - hit[0] < _CACHE_TTL_S:
|
||||
return hit[1]
|
||||
req = urllib.request.Request(url, headers={"User-Agent": "hermes-local-models"})
|
||||
with urllib.request.urlopen(req, timeout=_TIMEOUT_S) as r:
|
||||
data = json.load(r)
|
||||
if len(_CACHE) >= _CACHE_MAX:
|
||||
_CACHE.pop(min(_CACHE, key=lambda k: _CACHE[k][0]))
|
||||
_CACHE[url] = (now, data)
|
||||
return data
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HFModelHit:
|
||||
repo: str # e.g. "unsloth/Qwen3.8-27B-GGUF"
|
||||
downloads: int
|
||||
likes: int
|
||||
updated: str # ISO date from HF
|
||||
gated: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HFFileGroup:
|
||||
"""One downloadable quant: a single GGUF or all parts of a split one."""
|
||||
|
||||
label: str # e.g. "Q4_K_M" or the file stem
|
||||
paths: tuple[str, ...] # repo-relative, split parts in order
|
||||
total_bytes: int
|
||||
fit: str = "unknown" # fits-gpu | needs-ram | too-big | unknown
|
||||
|
||||
|
||||
_QUANT_RE = re.compile(
|
||||
r"(?:IQ|Q)\d[_A-Z0-9]*|F16|BF16|F32", re.IGNORECASE)
|
||||
_SPLIT_RE = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$", re.IGNORECASE)
|
||||
|
||||
|
||||
def search_models(query: str, limit: int = 20) -> list[HFModelHit]:
|
||||
"""Full-text search over HF models that ship GGUF files, most
|
||||
downloaded first (the closest public signal to 'trending')."""
|
||||
q = urllib.parse.quote(query.strip())
|
||||
url = (f"{_HF}/api/models?search={q}&filter=gguf&sort=downloads"
|
||||
f"&direction=-1&limit={max(1, min(int(limit), 50))}")
|
||||
out: list[HFModelHit] = []
|
||||
for m in _get_json(url):
|
||||
out.append(HFModelHit(
|
||||
repo=str(m.get("id", "")),
|
||||
downloads=int(m.get("downloads") or 0),
|
||||
likes=int(m.get("likes") or 0),
|
||||
updated=str(m.get("lastModified") or ""),
|
||||
gated=bool(m.get("gated")),
|
||||
))
|
||||
return out
|
||||
|
||||
|
||||
def _quant_label(filename: str) -> str:
|
||||
m = _QUANT_RE.search(filename)
|
||||
return m.group(0).upper() if m else filename
|
||||
|
||||
|
||||
def repo_files(repo: str) -> list[HFFileGroup]:
|
||||
"""The servable GGUFs in a repo, grouped: split parts collapse into one
|
||||
entry (first part is what llama.cpp loads), mmproj/draft companions are
|
||||
excluded (they aren't standalone models). Largest quant first."""
|
||||
url = f"{_HF}/api/models/{urllib.parse.quote(repo)}/tree/main?recursive=true"
|
||||
files = _get_json(url)
|
||||
|
||||
singles: list[tuple[str, int]] = []
|
||||
splits: dict[str, list[tuple[int, str, int]]] = {}
|
||||
for f in files:
|
||||
path = str(f.get("path", ""))
|
||||
if not path.lower().endswith(".gguf"):
|
||||
continue
|
||||
name = path.rsplit("/", 1)[-1].lower()
|
||||
if name.startswith("mmproj") or name.startswith("dspark") or "draft" in name:
|
||||
continue
|
||||
size = int(f.get("size") or 0)
|
||||
m = _SPLIT_RE.search(path)
|
||||
if m:
|
||||
stem = path[: m.start()]
|
||||
splits.setdefault(stem, []).append((int(m.group(1)), path, size))
|
||||
else:
|
||||
singles.append((path, size))
|
||||
|
||||
groups: list[HFFileGroup] = []
|
||||
for path, size in singles:
|
||||
groups.append(HFFileGroup(label=_quant_label(path), paths=(path,),
|
||||
total_bytes=size))
|
||||
for stem, parts in splits.items():
|
||||
parts.sort()
|
||||
groups.append(HFFileGroup(
|
||||
label=_quant_label(stem),
|
||||
paths=tuple(p for _, p, _ in parts),
|
||||
total_bytes=sum(s for _, _, s in parts)))
|
||||
groups.sort(key=lambda g: g.total_bytes, reverse=True)
|
||||
return groups
|
||||
|
||||
|
||||
def rough_fit(total_bytes: int, budget) -> str:
|
||||
"""Coarse pre-download verdict from file size alone. The GGUF header
|
||||
refines this after download; bands match the catalog pills' language.
|
||||
File size ≈ in-memory weights for GGUF (mmap'd as-is)."""
|
||||
need = total_bytes + _ROUGH_KV_AND_OVERHEAD
|
||||
if need <= budget.usable_vram_bytes:
|
||||
return "fits-gpu"
|
||||
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
return "needs-ram"
|
||||
return "too-big"
|
||||
|
||||
|
||||
def priced_repo_files(repo: str, budget) -> list[HFFileGroup]:
|
||||
from dataclasses import replace
|
||||
|
||||
return [replace(g, fit=rough_fit(g.total_bytes, budget))
|
||||
for g in repo_files(repo)]
|
||||
@@ -0,0 +1,198 @@
|
||||
"""Live model-load progress from the managed llama-server router.
|
||||
|
||||
llama-server's child processes emit per-tensor load progress
|
||||
({stages, current, value}, throttled upstream to ~200ms) which the
|
||||
router relays ONLY over its /models/sse stream — GET /models carries
|
||||
just the coarse status string. This module owns one lazy background
|
||||
watcher on that stream and keeps an in-memory snapshot other code can
|
||||
poll cheaply:
|
||||
|
||||
get_loading_progress() -> {model_id: {"stage", "value", "percent"}}
|
||||
|
||||
"percent" is a composite across stages so a bar doesn't sprint 0->100
|
||||
once per stage: the text model dominates load time (its weights dwarf
|
||||
the mmproj/spec extras), so it gets the lion's share of the range and
|
||||
the extras split the remainder.
|
||||
|
||||
The watcher starts on first call, reconnects with backoff (the router
|
||||
bounces on model download/eject), and never raises into callers — no
|
||||
router, no state file, or no SSE support (older engines) all read as
|
||||
"nothing loading". Safe from any process on the machine: the endpoint
|
||||
comes from the supervisor's machine-scoped state file.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_TEXT_STAGE_SHARE = 0.85 # composite range share for the text model
|
||||
_RECONNECT_DELAY_S = 3.0
|
||||
_STALE_ENTRY_TTL_S = 120.0 # a loading entry with no events this long is dead
|
||||
|
||||
_lock = threading.Lock()
|
||||
_watcher: threading.Thread | None = None
|
||||
_snapshot: dict[str, dict] = {}
|
||||
|
||||
|
||||
def _composite_percent(stages: list[str], current: str, value: float) -> int:
|
||||
"""Map (stage, in-stage value) onto one 0-100 range, text-heavy."""
|
||||
if not stages or current not in stages or len(stages) == 1:
|
||||
return max(0, min(100, round(value * 100)))
|
||||
extras = [s for s in stages if s != "text_model"]
|
||||
extra_share = (1.0 - _TEXT_STAGE_SHARE) / len(extras) if extras else 0.0
|
||||
offset = 0.0
|
||||
for stage in stages:
|
||||
share = _TEXT_STAGE_SHARE if stage == "text_model" else extra_share
|
||||
if stage == current:
|
||||
return max(0, min(100, round((offset + share * value) * 100)))
|
||||
offset += share
|
||||
return max(0, min(100, round(value * 100)))
|
||||
|
||||
|
||||
def _endpoint() -> "tuple[str, str] | None":
|
||||
"""(base_root, api_key) of the managed router, or None.
|
||||
|
||||
Resolved through the endpoint module's ownership-guarded reader, not
|
||||
a raw state-file read: on the shared stable port, a foreign install's
|
||||
server answers /health for anyone, and a raw read would attach this
|
||||
watcher to someone else's SSE stream (or spin on 401s against it).
|
||||
The guard's dead-pid check is the ownership proof."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
||||
|
||||
state = _state_endpoint()
|
||||
if state is None:
|
||||
return None
|
||||
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
|
||||
return (base, str(state.get("api_key", ""))) if base else None
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
def _apply_event(model: str, event: str, data: dict) -> None:
|
||||
with _lock:
|
||||
status = str(data.get("status", ""))
|
||||
if event in ("status_change", "model_status") and status == "loading":
|
||||
progress = data.get("progress") or {}
|
||||
stages = [str(s) for s in (progress.get("stages") or [])]
|
||||
current = str(progress.get("current", ""))
|
||||
value = progress.get("value")
|
||||
entry = _snapshot.setdefault(model, {"stage": "", "value": 0.0,
|
||||
"percent": 0, "ts": 0.0})
|
||||
entry["ts"] = time.monotonic()
|
||||
if current and isinstance(value, (int, float)):
|
||||
entry["stage"] = current
|
||||
entry["value"] = float(value)
|
||||
entry["percent"] = _composite_percent(stages, current, float(value))
|
||||
elif event in ("status_change", "model_status", "model_remove"):
|
||||
# Any terminal status (loaded/unloaded/failed) ends the load.
|
||||
if status != "loading":
|
||||
_snapshot.pop(model, None)
|
||||
|
||||
|
||||
def _watch() -> None:
|
||||
while True:
|
||||
endpoint = _endpoint()
|
||||
if endpoint is None:
|
||||
with _lock:
|
||||
_snapshot.clear()
|
||||
time.sleep(_RECONNECT_DELAY_S)
|
||||
continue
|
||||
base, key = endpoint
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
f"{base}/models/sse",
|
||||
headers={"Authorization": f"Bearer {key}",
|
||||
"Accept": "text/event-stream"})
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
buf = b""
|
||||
while True:
|
||||
chunk = r.read1(4096) if hasattr(r, "read1") else r.read(4096)
|
||||
if not chunk:
|
||||
break
|
||||
buf += chunk
|
||||
while b"\n" in buf:
|
||||
line, buf = buf.split(b"\n", 1)
|
||||
text = line.decode("utf-8", "replace").strip()
|
||||
if not text.startswith("data:"):
|
||||
continue
|
||||
try:
|
||||
msg = json.loads(text[5:].strip())
|
||||
_apply_event(str(msg.get("model", "")),
|
||||
str(msg.get("event", "")),
|
||||
msg.get("data") or {})
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001 — watcher must never die loud
|
||||
logger.debug("load-progress SSE reconnecting: %s", exc)
|
||||
# Stream ended (router bounce, timeout, error): loading entries from
|
||||
# the dead connection are unverifiable — drop rather than freeze.
|
||||
with _lock:
|
||||
_snapshot.clear()
|
||||
time.sleep(_RECONNECT_DELAY_S)
|
||||
|
||||
|
||||
def _ensure_watcher() -> None:
|
||||
global _watcher
|
||||
with _lock:
|
||||
if _watcher is None or not _watcher.is_alive():
|
||||
_watcher = threading.Thread(target=_watch, daemon=True,
|
||||
name="llamacpp-load-progress")
|
||||
_watcher.start()
|
||||
|
||||
|
||||
def get_loading_progress() -> dict[str, dict]:
|
||||
"""{model_id: {"stage", "value", "percent"}} for models loading right
|
||||
now. Empty when nothing is loading (or nothing is knowable)."""
|
||||
_ensure_watcher()
|
||||
now = time.monotonic()
|
||||
with _lock:
|
||||
return {m: {"stage": e["stage"], "value": e["value"],
|
||||
"percent": e["percent"]}
|
||||
for m, e in _snapshot.items()
|
||||
if now - e["ts"] < _STALE_ENTRY_TTL_S}
|
||||
|
||||
|
||||
def get_prefill_progress(model: str) -> "dict | None":
|
||||
"""{"processed": tokens} while the managed server is prompt-processing
|
||||
for ``model``, or None (idle, decoding, unreachable, or foreign server).
|
||||
|
||||
llama-server's /slots reports ``n_prompt_tokens_processed`` climbing in
|
||||
real time during prefill, but exposes no total — callers supply their
|
||||
own denominator (the request's estimated token count). Busiest
|
||||
processing slot wins when several are active: a parallel small request
|
||||
(title generation) freezes its counter during decode while a live
|
||||
prefill keeps climbing past it. One authenticated HTTP call per poll;
|
||||
every failure reads as "no prefill" — this is garnish, never load-
|
||||
bearing.
|
||||
"""
|
||||
ep = _endpoint()
|
||||
if ep is None:
|
||||
return None
|
||||
base, key = ep
|
||||
try:
|
||||
from urllib.parse import quote
|
||||
|
||||
req = urllib.request.Request(
|
||||
f"{base}/slots?model={quote(model)}",
|
||||
headers={"Authorization": f"Bearer {key}"})
|
||||
with urllib.request.urlopen(req, timeout=2) as r:
|
||||
slots = json.loads(r.read())
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
best = 0
|
||||
for slot in slots if isinstance(slots, list) else []:
|
||||
if not slot.get("is_processing"):
|
||||
continue
|
||||
try:
|
||||
processed = int(slot.get("n_prompt_tokens_processed") or 0)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
best = max(best, processed)
|
||||
return {"processed": best} if best > 0 else None
|
||||
@@ -0,0 +1,265 @@
|
||||
"""Per-model preset generation (--models-preset INI) — the router-side
|
||||
carrier for context-policy launch decisions.
|
||||
|
||||
The INI shape is what the router itself generates per child: a
|
||||
[model-id] section whose keys are long-form
|
||||
llama-server flag names without the leading dashes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
RUNTIME_OVERHEAD_BYTES,
|
||||
WindowDecision,
|
||||
initial_window,
|
||||
launch_args,
|
||||
ub_logits_bytes,
|
||||
)
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget,
|
||||
PhysicsRefusal,
|
||||
profile_from_gguf,
|
||||
)
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# args list -> INI keys. Flags the policy owns; everything else stays out
|
||||
# of the preset (recipe sampling defaults merge in a later pass).
|
||||
_FLAG_TO_KEY = {
|
||||
"-c": "ctx-size",
|
||||
"-b": "batch-size",
|
||||
"-ub": "ubatch-size",
|
||||
"-ctk": "cache-type-k",
|
||||
"-ctv": "cache-type-v",
|
||||
"-fa": "flash-attn",
|
||||
"-ot": "override-tensor",
|
||||
"--spec-type": "spec-type",
|
||||
"--spec-draft-n-max": "spec-draft-n-max",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class PresetEntry:
|
||||
model_id: str
|
||||
window: int
|
||||
spilled: bool
|
||||
refusal: str | None = None
|
||||
keys: dict[str, str] | None = None
|
||||
|
||||
|
||||
def _args_to_keys(args: list[str]) -> dict[str, str]:
|
||||
keys: dict[str, str] = {}
|
||||
i = 0
|
||||
while i < len(args):
|
||||
flag = args[i]
|
||||
key = _FLAG_TO_KEY.get(flag)
|
||||
if key is None:
|
||||
i += 1
|
||||
continue
|
||||
keys[key] = args[i + 1]
|
||||
i += 2
|
||||
return keys
|
||||
|
||||
|
||||
def generate_presets(models_dir: Path, budget: HardwareBudget,
|
||||
preset_path: Path,
|
||||
mtp_capable: set[str] | None = None) -> list[PresetEntry]:
|
||||
"""Walk the staged models, run the launch decision per model, and
|
||||
write one INI. Refused models get no section (the router simply won't
|
||||
have policy for them; the picker surfaces the refusal + smaller-quant
|
||||
suggestion from the returned entries).
|
||||
|
||||
Catalog-declared companions merge in here: sampling defaults (policy
|
||||
keys always win), the vision projector when present, and a spec-decode
|
||||
draft model iff the decision spilled — the rule: speculative
|
||||
decode is a spill amplifier, so a resident draft accelerates a spilled
|
||||
main model; a zero-spill model doesn't pay the draft's memory."""
|
||||
from hermes_cli.local_runtime.bootstrap import assets_dir
|
||||
from hermes_cli.local_runtime.catalog import find_entry_for_model
|
||||
|
||||
entries: list[PresetEntry] = []
|
||||
sections: list[str] = []
|
||||
for gguf in _staged_in(models_dir):
|
||||
model_id = _strip_part(gguf.stem)
|
||||
try:
|
||||
header = read_gguf_header(gguf)
|
||||
profile = profile_from_gguf(header)
|
||||
except (ValueError, OSError) as exc:
|
||||
logger.warning("preset skip %s: %s", gguf.name, exc)
|
||||
continue
|
||||
# Overhead beyond weights+KV: runtime buffers, the vision projector
|
||||
# when this model ships one, and the logits buffers of whichever
|
||||
# microbatch/MTP posture launch_args will choose — flag and price
|
||||
# decided together, from the same facts.
|
||||
hit = find_entry_for_model(model_id)
|
||||
entry = hit[0] if hit is not None else None
|
||||
is_mtp = (entry.mtp if entry is not None
|
||||
else model_id in (mtp_capable or set()))
|
||||
if is_mtp and profile.kv_scale == 1.0:
|
||||
# Header-derived profiles don't know about MTP's draft
|
||||
# context; apply the calibrated KV multiplier here so the
|
||||
# launch fit prices what the server will actually allocate.
|
||||
import dataclasses
|
||||
|
||||
profile = dataclasses.replace(profile, kv_scale=1.2)
|
||||
mmproj_bytes = 0
|
||||
if entry is not None and entry.mmproj is not None:
|
||||
mmproj_path = assets_dir() / entry.mmproj.local_name
|
||||
if mmproj_path.exists():
|
||||
mmproj_bytes = entry.mmproj.size_bytes
|
||||
# MTP posture ladder — window first, prefill second: price the
|
||||
# launch under both postures and keep whichever grants the larger
|
||||
# window (the stacked posture's bigger compute buffer buys ~3x
|
||||
# short-prompt prefill but costs ~2 GiB that would otherwise be
|
||||
# window; measured at 256K the ub512 posture still prefills at
|
||||
# 2.7K tok/s, so window wins ties only one way: never trade
|
||||
# context away for prefill). Same window -> stacked.
|
||||
mtp_prefill = False
|
||||
logits_bytes = ub_logits_bytes(profile.n_vocab, mtp_capable=is_mtp)
|
||||
if is_mtp:
|
||||
stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True,
|
||||
mtp_prefill=True)
|
||||
stacked_probe = initial_window(
|
||||
profile, budget,
|
||||
overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
|
||||
+ stacked_logits))
|
||||
plain_probe = initial_window(
|
||||
profile, budget,
|
||||
overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
|
||||
+ logits_bytes))
|
||||
if (not isinstance(stacked_probe, PhysicsRefusal)
|
||||
and not stacked_probe.spilled
|
||||
and (isinstance(plain_probe, PhysicsRefusal)
|
||||
or stacked_probe.window >= plain_probe.window)):
|
||||
mtp_prefill = True
|
||||
logits_bytes = stacked_logits
|
||||
decision = initial_window(
|
||||
profile, budget,
|
||||
overhead_bytes=RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
|
||||
if isinstance(decision, PhysicsRefusal):
|
||||
entries.append(PresetEntry(model_id=model_id, window=0,
|
||||
spilled=False, refusal=decision.message))
|
||||
continue
|
||||
|
||||
# Session growth (growth.py): a persisted override lifts the launch
|
||||
# window to where the ladder last grew it — capped at native, and
|
||||
# only when physics still clears the bigger window on THIS boot's
|
||||
# budget (a smaller-VRAM day re-fits honestly back down).
|
||||
try:
|
||||
from hermes_cli.local_runtime.estimator import ctx_bytes
|
||||
from hermes_cli.local_runtime.growth import load_window_overrides
|
||||
|
||||
override = load_window_overrides().get(model_id)
|
||||
native = profile.n_ctx_train or decision.window
|
||||
if override and override > decision.window:
|
||||
target = min(int(override), native)
|
||||
kv = ctx_bytes(profile, target)
|
||||
need = (profile.weights_bytes + kv
|
||||
+ RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
|
||||
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
spill = max(0, need - budget.usable_vram_bytes)
|
||||
decision = WindowDecision(
|
||||
window=target, spill_bytes=spill,
|
||||
kv_on_gpu=kv <= budget.usable_vram_bytes,
|
||||
reasons=[f"grown window restored ({target // 1024}K)"])
|
||||
except Exception as exc: # noqa: BLE001 — overrides are advisory
|
||||
logger.debug("window override skipped for %s: %s", model_id, exc)
|
||||
|
||||
# (entry and is_mtp resolved above, where the overhead was priced —
|
||||
# the launch flags below MUST match that pricing.)
|
||||
args = launch_args(profile, decision, mtp_capable=is_mtp,
|
||||
mtp_draft_depth=(entry.mtp_draft_depth
|
||||
if entry is not None else 3),
|
||||
uma=budget.uma, mtp_prefill=mtp_prefill)
|
||||
keys = _args_to_keys(args)
|
||||
|
||||
if entry is not None and is_mtp:
|
||||
# Integrated-MTP targets sample on the backend, and so does
|
||||
# the draft (pairing validated against the vendor's published
|
||||
# llama.cpp recipes for these models).
|
||||
keys["backend-sampling"] = "on"
|
||||
keys["spec-draft-backend-sampling"] = "on"
|
||||
|
||||
# Sampling deference ladder, under the policy keys (policy wins
|
||||
# on clash). The GGUF's own general.sampling.* metadata is the
|
||||
# publisher's recommendation — it arrives with the file, updates
|
||||
# with every re-upload, and covers models the catalog has never
|
||||
# heard of. Catalog sampling applies only where the file is
|
||||
# silent; a model carrying neither runs llama.cpp defaults.
|
||||
for k, v in header.sampling_defaults.items():
|
||||
keys.setdefault(k, v)
|
||||
if entry is not None:
|
||||
for k, v in (entry.sampling or {}).items():
|
||||
keys.setdefault(k, v)
|
||||
if entry.mmproj is not None:
|
||||
mmproj_path = assets_dir() / entry.mmproj.local_name
|
||||
if mmproj_path.exists():
|
||||
keys["mmproj"] = str(mmproj_path)
|
||||
if entry.draft is not None and decision.spilled:
|
||||
draft_path = assets_dir() / entry.draft.local_name
|
||||
if draft_path.exists():
|
||||
keys["model-draft"] = str(draft_path)
|
||||
keys["spec-type"] = "draft-dspark"
|
||||
# Unsloth's measured cliff: acceptance 83% at 2-3
|
||||
# drafts, collapses at 4.
|
||||
keys["spec-draft-n-max"] = "3"
|
||||
|
||||
entries.append(PresetEntry(model_id=model_id, window=decision.window,
|
||||
spilled=decision.spilled, keys=keys))
|
||||
body = "\n".join(f"{k} = {v}" for k, v in keys.items())
|
||||
sections.append(f"[{model_id}]\n{body}\n")
|
||||
|
||||
preset_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
preset_path.write_text("\n".join(sections), encoding="utf-8")
|
||||
logger.info("wrote %d preset sections to %s", len(sections), preset_path)
|
||||
return entries
|
||||
|
||||
|
||||
def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEntry]:
|
||||
"""The launch decisions the running server was actually given, read
|
||||
back from the preset INI (the INI is the record — it's what spawned
|
||||
the children). Missing/unparseable file returns {}."""
|
||||
import configparser
|
||||
|
||||
if preset_path is None:
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
out: dict[str, PresetEntry] = {}
|
||||
try:
|
||||
parser = configparser.ConfigParser()
|
||||
parser.read(preset_path, encoding="utf-8")
|
||||
for section in parser.sections():
|
||||
window = parser.getint(section, "ctx-size", fallback=0)
|
||||
spilled = parser.has_option(section, "override-tensor")
|
||||
out[section] = PresetEntry(model_id=section, window=window,
|
||||
spilled=spilled)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("preset read-back failed: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
def _strip_part(stem: str) -> str:
|
||||
import re
|
||||
|
||||
return re.sub(r"-\d{5}-of-\d{5}$", "", stem)
|
||||
|
||||
|
||||
def _staged_in(models_dir: Path) -> "list[Path]":
|
||||
"""Servable models in an arbitrary directory (split first-parts only) —
|
||||
the validation harness points at non-default dirs."""
|
||||
import re
|
||||
|
||||
part = re.compile(r"-(\d{5})-of-\d{5}\.gguf$")
|
||||
out = []
|
||||
for p in sorted(models_dir.glob("*.gguf")):
|
||||
m = part.search(p.name)
|
||||
if m and m.group(1) != "00001":
|
||||
continue
|
||||
out.append(p)
|
||||
return out
|
||||
@@ -0,0 +1,500 @@
|
||||
"""Supervision of one llama-server in router mode.
|
||||
|
||||
The router process is ours (restart with backoff on crash); router children
|
||||
are its problem — child failures surface via GET /models exit_code, never
|
||||
auto-retried here.
|
||||
|
||||
Readiness rules (each learned the hard way on real hardware):
|
||||
- health-200 is NOT readiness; every readiness claim requires a touch
|
||||
generation (temp-0, expected token, generous budget, reasoning_content
|
||||
scanned).
|
||||
- Always dial 127.0.0.1 — resolving localhost adds ~2s per request on
|
||||
Windows via IPv6 fallback.
|
||||
- /metrics is opt-in (--metrics) and carries no KV-usage metric;
|
||||
idleness = requests_processing == 0 and no slot is_processing.
|
||||
- The router's LRU eviction has no pin for the primary model: until an
|
||||
upstream pin exists, keep_primary_loaded re-touches the primary after
|
||||
any other model load.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import secrets
|
||||
import socket
|
||||
import subprocess
|
||||
import threading
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_cli.local_runtime.binaries import server_binary, runtimes_root
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
TOUCH_PROMPT = "Reply with exactly one word: the capital of France."
|
||||
TOUCH_EXPECT = "paris"
|
||||
_RESTART_BACKOFF_S = (1, 5, 15, 60)
|
||||
|
||||
|
||||
def state_path() -> Path:
|
||||
"""Endpoint state for other Hermes processes (provider resolution reads
|
||||
this to route llamacpp-alias requests at the managed server)."""
|
||||
return runtimes_root() / "server.json"
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket() as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return s.getsockname()[1]
|
||||
|
||||
|
||||
# Default port for the managed server, chosen once and reused across
|
||||
# restarts. Sessions persist the resolved base_url; an ephemeral port
|
||||
# would strand every resumed session on a dead endpoint after each
|
||||
# restart. Deliberately NOT 8080 so we never collide with a user's own
|
||||
# llama-server/Ollama-adjacent stack.
|
||||
_DEFAULT_PORT = 18434
|
||||
|
||||
|
||||
def _stable_port() -> int:
|
||||
"""The stable default port, falling back to an ephemeral one only when
|
||||
something else already listens there (and it isn't a leftover managed
|
||||
server, which stop() would have cleaned up)."""
|
||||
try:
|
||||
with socket.socket() as s:
|
||||
s.bind(("127.0.0.1", _DEFAULT_PORT))
|
||||
return _DEFAULT_PORT
|
||||
except OSError:
|
||||
logger.warning(
|
||||
"port %d busy; managed llama-server falling back to an ephemeral "
|
||||
"port — existing sessions may need a model re-pick", _DEFAULT_PORT)
|
||||
return _free_port()
|
||||
|
||||
|
||||
def _stable_api_key() -> str:
|
||||
"""One key for the life of the install, persisted beside the runtimes.
|
||||
|
||||
Endpoint identity must survive restarts as a UNIT — sessions persist the
|
||||
resolved base_url + api_key, so a per-boot key strands every resumed
|
||||
session on HTTP 401 exactly the way a per-boot port would strand them
|
||||
on connection errors. Rotating it buys nothing: the key exists to stop
|
||||
other loopback processes free-riding, and it lives on the same disk as
|
||||
the state file that would leak it. Delete the file to rotate manually.
|
||||
"""
|
||||
key_path = runtimes_root() / ".api_key"
|
||||
try:
|
||||
existing = key_path.read_text(encoding="utf-8").strip()
|
||||
if len(existing) >= 16:
|
||||
return existing
|
||||
except OSError:
|
||||
pass
|
||||
key = secrets.token_urlsafe(24)
|
||||
try:
|
||||
key_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
key_path.write_text(key, encoding="utf-8")
|
||||
except OSError as exc:
|
||||
logger.warning("could not persist api key (%s); sessions will need "
|
||||
"a re-pick after restart", exc)
|
||||
return key
|
||||
|
||||
|
||||
class LlamaServerSupervisor:
|
||||
"""Own one llama-server router process for the life of a Hermes session.
|
||||
|
||||
Usage::
|
||||
|
||||
sup = LlamaServerSupervisor(install_dir, models_dir)
|
||||
sup.start() # spawn + wait healthy
|
||||
sup.ensure_model_ready(name) # load + touch-generate
|
||||
... sup.base_url is the /v1 endpoint, sup.api_key its key ...
|
||||
sup.stop()
|
||||
"""
|
||||
|
||||
def __init__(self, install_dir: Path, models_dir: Path, *,
|
||||
models_max: int = 4, port: int | None = None,
|
||||
extra_args: list[str] | None = None,
|
||||
log_path: Path | None = None,
|
||||
preset_path: Path | None = None):
|
||||
self.install_dir = Path(install_dir)
|
||||
self.models_dir = Path(models_dir)
|
||||
self.models_max = models_max
|
||||
self.port = port or _stable_port()
|
||||
self.api_key = _stable_api_key()
|
||||
self.extra_args = list(extra_args or [])
|
||||
self.log_path = log_path or (self.models_dir.parent / "logs" / "llama-server.log")
|
||||
self.preset_path = preset_path
|
||||
self.proc: subprocess.Popen | None = None
|
||||
self.primary_model: str | None = None
|
||||
self._restarts = 0
|
||||
self._stopping = False
|
||||
self._watchdog: threading.Thread | None = None
|
||||
self._log_handle = None
|
||||
self._idle_since: dict[str, float] = {}
|
||||
|
||||
# ── endpoints ────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return f"http://127.0.0.1:{self.port}/v1"
|
||||
|
||||
def _url(self, route: str) -> str:
|
||||
return f"http://127.0.0.1:{self.port}{route}"
|
||||
|
||||
def _request(self, route: str, body: dict | None = None, timeout_s: int = 30) -> dict:
|
||||
req = urllib.request.Request(
|
||||
self._url(route),
|
||||
data=json.dumps(body).encode() if body is not None else None,
|
||||
headers={"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.api_key}"},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=timeout_s) as r:
|
||||
raw = r.read()
|
||||
return json.loads(raw) if raw else {}
|
||||
|
||||
# ── lifecycle ────────────────────────────────────────────
|
||||
|
||||
def _spawn(self) -> None:
|
||||
exe = server_binary(self.install_dir)
|
||||
cmd = [
|
||||
str(exe),
|
||||
"--host", "127.0.0.1",
|
||||
"--port", str(self.port),
|
||||
"--api-key", self.api_key,
|
||||
"--models-dir", str(self.models_dir),
|
||||
"--models-max", str(self.models_max),
|
||||
# The residency contract at the layer that sees every message:
|
||||
# a chat request to a staged-but-unloaded model loads it (slow
|
||||
# first token) instead of failing with 'model not found' —
|
||||
# without this flag, chat after an eject is a bare 400/404.
|
||||
"--models-autoload",
|
||||
"--metrics", # opt-in flag; supervisor telemetry needs it
|
||||
"--slots", # /slots endpoint is also opt-in; is_idle reads it
|
||||
"--no-webui",
|
||||
"--jinja",
|
||||
# Direct I/O on model load: bypasses the page cache, so a
|
||||
# multi-GB load doesn't evict half the OS cache — measured
|
||||
# faster loads on NVMe, and our router bounces (download/
|
||||
# delete/activate) reload models often enough to care.
|
||||
"-dio",
|
||||
]
|
||||
if self.preset_path and self.preset_path.exists():
|
||||
cmd += ["--models-preset", str(self.preset_path)]
|
||||
cmd += [
|
||||
*self.extra_args,
|
||||
]
|
||||
self.log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if self._log_handle is not None:
|
||||
# The crash-restart loop calls _spawn repeatedly; without
|
||||
# closing the prior handle each restart leaks one fd.
|
||||
try:
|
||||
self._log_handle.close()
|
||||
except Exception: # noqa: BLE001 — best-effort
|
||||
pass
|
||||
self._log_handle = open(self.log_path, "a", encoding="utf-8", errors="replace")
|
||||
self._log_handle.write(f"\n# spawn: {cmd}\n")
|
||||
self._log_handle.flush()
|
||||
# list-args, never a shell: spaced paths (user homes) must survive.
|
||||
self.proc = subprocess.Popen(cmd, stdout=self._log_handle,
|
||||
stderr=subprocess.STDOUT, cwd=str(exe.parent))
|
||||
logger.info("llama-server router spawned pid=%s port=%s", self.proc.pid, self.port)
|
||||
# State goes down at SPAWN, not after health: endpoint resolution
|
||||
# treats a live-pid-but-not-yet-healthy server as "starting" rather
|
||||
# than "unconfigured", so a readiness probe racing the boot doesn't
|
||||
# throw the app back to onboarding (observed on first restart test).
|
||||
self._write_state()
|
||||
|
||||
def start(self, timeout_s: int = 120) -> None:
|
||||
self._stopping = False
|
||||
self._spawn()
|
||||
self._wait_health(timeout_s)
|
||||
self._write_state()
|
||||
self._watchdog = threading.Thread(target=self._watch, daemon=True,
|
||||
name="llamacpp-supervisor")
|
||||
self._watchdog.start()
|
||||
|
||||
def _write_state(self) -> None:
|
||||
path = state_path()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps({
|
||||
"base_url": self.base_url,
|
||||
"api_key": self.api_key,
|
||||
"pid": self.proc.pid if self.proc else None,
|
||||
}), encoding="utf-8")
|
||||
|
||||
def _wait_health(self, timeout_s: int) -> None:
|
||||
deadline = time.monotonic() + timeout_s
|
||||
while time.monotonic() < deadline:
|
||||
if self.proc and self.proc.poll() is not None:
|
||||
raise RuntimeError(
|
||||
f"llama-server exited rc={self.proc.returncode} during startup "
|
||||
f"(log: {self.log_path})")
|
||||
try:
|
||||
with urllib.request.urlopen(self._url("/health"), timeout=3) as r:
|
||||
if r.status == 200:
|
||||
return
|
||||
except (urllib.error.URLError, OSError, TimeoutError):
|
||||
pass
|
||||
time.sleep(1)
|
||||
raise TimeoutError(f"llama-server not healthy after {timeout_s}s (log: {self.log_path})")
|
||||
|
||||
def _watch(self) -> None:
|
||||
"""Restart the router (not its children) on crash, with backoff."""
|
||||
while not self._stopping:
|
||||
proc = self.proc
|
||||
if proc is None:
|
||||
return
|
||||
rc = proc.poll()
|
||||
if rc is None:
|
||||
time.sleep(2)
|
||||
continue
|
||||
if self._stopping:
|
||||
return
|
||||
backoff = _RESTART_BACKOFF_S[min(self._restarts, len(_RESTART_BACKOFF_S) - 1)]
|
||||
logger.warning("llama-server exited rc=%s; restart #%s in %ss",
|
||||
rc, self._restarts + 1, backoff)
|
||||
time.sleep(backoff)
|
||||
self._restarts += 1
|
||||
try:
|
||||
self._reap_orphaned_children()
|
||||
self._spawn()
|
||||
self._wait_health(120)
|
||||
if self.primary_model:
|
||||
self.ensure_model_ready(self.primary_model)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("llama-server restart failed: %s", exc)
|
||||
|
||||
def stop(self) -> None:
|
||||
self._stopping = True
|
||||
state_path().unlink(missing_ok=True)
|
||||
if self.proc and self.proc.poll() is None:
|
||||
self._terminate_tree(self.proc)
|
||||
if self._log_handle:
|
||||
self._log_handle.close()
|
||||
self._log_handle = None
|
||||
|
||||
@staticmethod
|
||||
def _terminate_tree(proc: subprocess.Popen) -> None:
|
||||
"""Terminate the router AND its model children.
|
||||
|
||||
The router spawns one child llama-server per loaded model, each
|
||||
holding gigabytes of VRAM. Terminating only the router (on
|
||||
Windows, TerminateProcess — no signal handlers, no cleanup pass)
|
||||
orphans those children: the port goes quiet but the weights stay
|
||||
resident, and the next spawn re-loads models alongside a ghost
|
||||
still holding the memory. Enumerate children FIRST (the parent
|
||||
must be alive to walk them), then terminate parent and children
|
||||
together, escalating to kill for stragglers.
|
||||
"""
|
||||
children: list = []
|
||||
try:
|
||||
import psutil
|
||||
|
||||
children = psutil.Process(proc.pid).children(recursive=True)
|
||||
except Exception: # noqa: BLE001 — no psutil view; still stop the router
|
||||
children = []
|
||||
proc.terminate()
|
||||
for child in children:
|
||||
try:
|
||||
child.terminate()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
try:
|
||||
proc.wait(timeout=15)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
for child in children:
|
||||
try:
|
||||
if child.is_running():
|
||||
child.kill()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
def _reap_orphaned_children(self) -> None:
|
||||
"""Kill model children orphaned by a router crash, before respawn.
|
||||
|
||||
A crashed router can't clean up its children, and a dead parent
|
||||
can't be walked — so match by identity instead: any process
|
||||
running OUR llama-server binary whose parent is gone is an
|
||||
orphan of a previous router. Their VRAM must come back before
|
||||
the new router loads models next to the ghosts. External
|
||||
llama-servers (different binary path) never match.
|
||||
"""
|
||||
try:
|
||||
import psutil
|
||||
|
||||
exe = str(server_binary(self.install_dir))
|
||||
except Exception: # noqa: BLE001
|
||||
return
|
||||
for p in psutil.process_iter(["exe", "ppid"]):
|
||||
try:
|
||||
if p.info.get("exe") != exe:
|
||||
continue
|
||||
if self.proc is not None and p.pid == self.proc.pid:
|
||||
continue
|
||||
ppid = p.info.get("ppid") or 0
|
||||
if ppid and psutil.pid_exists(ppid):
|
||||
continue
|
||||
logger.warning("reaping orphaned llama-server child pid=%s", p.pid)
|
||||
p.kill()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
continue
|
||||
|
||||
# ── model management (router endpoints) ──────────────────
|
||||
|
||||
def models(self) -> dict:
|
||||
"""{model_id: status_value} from GET /models."""
|
||||
data = self._request("/models")
|
||||
return {m["id"]: m.get("status", {}).get("value", "unknown")
|
||||
for m in data.get("data", [])}
|
||||
|
||||
def model_failures(self) -> dict:
|
||||
"""{model_id: exit_code} for children that died — surfaced to the
|
||||
UI, never auto-retried (design: router children are its problem)."""
|
||||
data = self._request("/models")
|
||||
out = {}
|
||||
for m in data.get("data", []):
|
||||
status = m.get("status", {})
|
||||
if status.get("value") == "failed" or status.get("exit_code"):
|
||||
out[m["id"]] = status.get("exit_code")
|
||||
return out
|
||||
|
||||
def load_model(self, model_id: str, timeout_s: int = 600) -> None:
|
||||
self._request("/models/load", {"model": model_id}, timeout_s=timeout_s)
|
||||
|
||||
def unload_model(self, model_id: str) -> None:
|
||||
"""Free the child's VRAM now. Route existence verified empirically
|
||||
on b10290 (POST /models/unload; bogus name -> 400 'model is not
|
||||
found'). Momentary action: never touches primary_model — the
|
||||
declaration is durable, an eject is not (residency design).
|
||||
|
||||
Settle before returning: for a few seconds after unload returns,
|
||||
the router still routes to the dying child and answers chat with
|
||||
500 'proxy error: Could not establish connection' (probed on
|
||||
b10362). Waiting for the model to report unloaded means the next
|
||||
message autoloads cleanly instead of racing the teardown.
|
||||
"""
|
||||
self._request("/models/unload", {"model": model_id}, timeout_s=120)
|
||||
deadline = time.monotonic() + 15
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
if self.models().get(model_id) not in ("loaded", "ready", "unloading"):
|
||||
return
|
||||
except Exception: # noqa: BLE001
|
||||
return
|
||||
time.sleep(0.3)
|
||||
|
||||
# ── idle residency (non-primary models) ──────────────────
|
||||
|
||||
# A model that has gone quiet gets its VRAM back after this long. A
|
||||
# constant, not a knob: long enough that an active conversation never
|
||||
# trips it, short enough that a wandered-off session frees ~20 GiB
|
||||
# within the hour. No exemptions (residency v2): demand reloads
|
||||
# anything the user comes back to.
|
||||
IDLE_UNLOAD_S = 15 * 60
|
||||
|
||||
def sweep_idle(self, now: float | None = None) -> list[str]:
|
||||
"""Unload models idle past IDLE_UNLOAD_S. Returns the model ids
|
||||
unloaded. Idle means no busy slots and no queued work, tracked
|
||||
per model across calls; a model seen busy resets its clock."""
|
||||
now = time.monotonic() if now is None else now
|
||||
unloaded: list[str] = []
|
||||
try:
|
||||
statuses = self.models()
|
||||
except Exception: # noqa: BLE001
|
||||
return unloaded
|
||||
for model_id, status in statuses.items():
|
||||
if status not in ("loaded", "ready"):
|
||||
self._idle_since.pop(model_id, None)
|
||||
continue
|
||||
if not self.is_idle(model_id):
|
||||
self._idle_since.pop(model_id, None)
|
||||
continue
|
||||
first_idle = self._idle_since.setdefault(model_id, now)
|
||||
if now - first_idle >= self.IDLE_UNLOAD_S:
|
||||
try:
|
||||
self.unload_model(model_id)
|
||||
self._idle_since.pop(model_id, None)
|
||||
unloaded.append(model_id)
|
||||
logger.info("idle-unloaded %s (idle %ds)", model_id,
|
||||
int(now - first_idle))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("idle unload of %s failed: %s", model_id, exc)
|
||||
return unloaded
|
||||
|
||||
def touch_generate(self, model_id: str, timeout_s: int = 300) -> bool:
|
||||
"""The readiness proof. Generous budget + reasoning_content scan —
|
||||
small token budgets false-fail reasoning models, which spend their
|
||||
first tokens thinking."""
|
||||
try:
|
||||
resp = self._request("/v1/chat/completions", {
|
||||
"model": model_id,
|
||||
"messages": [{"role": "user", "content": TOUCH_PROMPT}],
|
||||
"max_tokens": 512, "temperature": 0,
|
||||
}, timeout_s=timeout_s)
|
||||
msg = resp["choices"][0]["message"]
|
||||
blob = (msg.get("content") or "") + " " + (msg.get("reasoning_content") or "")
|
||||
return TOUCH_EXPECT in blob.lower()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("touch generation failed for %s: %s", model_id, exc)
|
||||
return False
|
||||
|
||||
def ensure_model_ready(self, model_id: str, timeout_s: int = 600) -> bool:
|
||||
"""Load if needed, then prove readiness with a touch generation."""
|
||||
status = self.models().get(model_id)
|
||||
if status is None:
|
||||
raise KeyError(f"model {model_id} not present in models dir")
|
||||
if status not in ("loaded", "ready"):
|
||||
self.load_model(model_id, timeout_s=timeout_s)
|
||||
return self.touch_generate(model_id)
|
||||
|
||||
def actual_n_ctx(self, model_id: str) -> int | None:
|
||||
"""/props reconciliation: the granted window as the child reports
|
||||
it — the compressor's budget and the picker's 'running at 87K of
|
||||
262K' both read THIS value, never the request (design step 4)."""
|
||||
try:
|
||||
props = self._request(f"/props?model={model_id}")
|
||||
return props.get("default_generation_settings", {}).get("n_ctx")
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
def keep_primary_loaded(self) -> None:
|
||||
"""The router's LRU eviction has no pin, so after any other load
|
||||
re-touch the primary to keep it most-recently-used. Best-effort
|
||||
under bursty multi-model load — replaced when an upstream pin
|
||||
exists."""
|
||||
if self.primary_model and self.models().get(self.primary_model) in (
|
||||
"loaded", "ready"):
|
||||
self.touch_generate(self.primary_model, timeout_s=60)
|
||||
|
||||
# ── telemetry ────────────────────────────────────────────
|
||||
|
||||
def is_idle(self, model_id: str | None = None) -> bool:
|
||||
"""No processing requests and no busy slots. Router quirk: /slots
|
||||
and /metrics are per-child and require ?model= (bare calls 400),
|
||||
and no KV-usage metric exists. With ``model_id`` checks that one
|
||||
child; without, every loaded child."""
|
||||
try:
|
||||
if model_id is not None:
|
||||
loaded = [model_id]
|
||||
else:
|
||||
loaded = [m for m, status in self.models().items()
|
||||
if status in ("loaded", "ready")]
|
||||
for mid in loaded:
|
||||
slots = self._request(f"/slots?model={mid}")
|
||||
if any(s.get("is_processing") for s in slots):
|
||||
return False
|
||||
req = urllib.request.Request(
|
||||
self._url(f"/metrics?model={mid}"),
|
||||
headers={"Authorization": f"Bearer {self.api_key}"})
|
||||
with urllib.request.urlopen(req, timeout=10) as r:
|
||||
text = r.read().decode()
|
||||
for line in text.splitlines():
|
||||
if line.startswith("llamacpp:requests_processing"):
|
||||
if float(line.split()[-1]) != 0.0:
|
||||
return False
|
||||
return True
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
+6
-1
@@ -8658,7 +8658,10 @@ def cmd_gui(args: argparse.Namespace):
|
||||
|
||||
if source_mode:
|
||||
print("→ Launching Hermes Desktop from source build...")
|
||||
launch_result = subprocess.run([npm, "exec", "--", "electron", "."], cwd=desktop_dir, env=env, check=False)
|
||||
electron_argv = [npm, "exec", "--", "electron", "."]
|
||||
if getattr(args, "local", False):
|
||||
electron_argv.append("--local")
|
||||
launch_result = subprocess.run(electron_argv, cwd=desktop_dir, env=env, check=False)
|
||||
sys.exit(launch_result.returncode)
|
||||
|
||||
if packaged_executable is None:
|
||||
@@ -8677,6 +8680,8 @@ def cmd_gui(args: argparse.Namespace):
|
||||
launch_command.append("--disable-setuid-sandbox")
|
||||
|
||||
launch_command.extend(config_electron_flags)
|
||||
if getattr(args, "local", False):
|
||||
launch_command.append("--local")
|
||||
print(f"→ Launching packaged Hermes Desktop: {' '.join(launch_command)}")
|
||||
launch_result = subprocess.run(launch_command, cwd=desktop_dir, env=env, check=False)
|
||||
sys.exit(launch_result.returncode)
|
||||
|
||||
@@ -1023,6 +1023,29 @@ def resolve_provider_full(
|
||||
if custom_pdef is not None:
|
||||
return custom_pdef
|
||||
|
||||
# 2c. Managed local runtime: the llamacpp aliases are a real provider
|
||||
# whenever the managed server (or a detected external one) resolves —
|
||||
# no credential and no providers: entry required, the credential is
|
||||
# reachability. Without this rung the model-switch path rejected the
|
||||
# very provider the Local Models 'Use' flow writes to config
|
||||
# ("Unknown provider 'llamacpp'" from the desktop dropdown).
|
||||
if raw in ("llamacpp", "llama.cpp", "llama-cpp"):
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
|
||||
|
||||
endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0)
|
||||
except Exception:
|
||||
endpoint = None
|
||||
if endpoint:
|
||||
return ProviderDef(
|
||||
id="llamacpp",
|
||||
name="Local",
|
||||
transport="openai_chat",
|
||||
api_key_env_vars=(),
|
||||
base_url=endpoint["base_url"],
|
||||
source="local-runtime",
|
||||
)
|
||||
|
||||
# 3. Try models.dev directly (for providers not in our ALIASES)
|
||||
try:
|
||||
from agent.models_dev import get_provider_info as _mdev_provider
|
||||
|
||||
@@ -1228,6 +1228,53 @@ def _resolve_named_custom_runtime(
|
||||
# `provider: ollama` with a LAN/WireGuard `base_url` doesn't silently
|
||||
# fall through to OpenRouter.
|
||||
requested_norm = (requested_provider or "").strip().lower()
|
||||
|
||||
# Managed llama.cpp runtime: a llamacpp-flavored alias with no explicit
|
||||
# base_url resolves to the supervised server (or a detected external
|
||||
# one) before the generic custom fallthrough. Explicit base_url always
|
||||
# wins — a user pointing at a specific server means that server.
|
||||
if requested_norm in ("llamacpp", "llama.cpp", "llama-cpp") and not explicit_base_url:
|
||||
try:
|
||||
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
|
||||
|
||||
endpoint = resolve_llamacpp_endpoint()
|
||||
except Exception: # noqa: BLE001 — resolution is best-effort
|
||||
endpoint = None
|
||||
if endpoint:
|
||||
return {
|
||||
"provider": "custom",
|
||||
"api_mode": "chat_completions",
|
||||
"base_url": endpoint["base_url"],
|
||||
"api_key": (explicit_api_key or "").strip()
|
||||
or endpoint["api_key"] or "no-key-required",
|
||||
"source": "local-runtime",
|
||||
"requested_provider": requested_provider,
|
||||
}
|
||||
# No server to serve this model. Say so and stop — falling through
|
||||
# to the generic custom path sends the request to whatever provider
|
||||
# picks it up (OpenRouter with a placeholder key), and the user's
|
||||
# "local server is off" surfaces as that provider's baffling
|
||||
# "401 Invalid API key". The switch's own state picks the message:
|
||||
# the user who turned the server off gets pointed at the switch,
|
||||
# anyone else at the setup pane.
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_cfg
|
||||
|
||||
_lr_enabled = bool((_load_cfg().get("local_runtime") or {}).get("enabled"))
|
||||
except Exception: # noqa: BLE001
|
||||
_lr_enabled = False
|
||||
if _lr_enabled:
|
||||
raise ValueError(
|
||||
"The local model server isn't running. It may still be "
|
||||
"starting — try again in a moment, or check Settings → "
|
||||
"Providers → Local models."
|
||||
)
|
||||
raise ValueError(
|
||||
"The local model server is turned off. Turn it back on in "
|
||||
"Settings → Providers → Local models, or switch to another "
|
||||
"model."
|
||||
)
|
||||
|
||||
if requested_norm and requested_norm != "custom":
|
||||
try:
|
||||
from hermes_cli.auth import resolve_provider as _resolve_provider
|
||||
|
||||
@@ -55,6 +55,11 @@ def build_gui_parser(subparsers, *, cmd_gui: Callable) -> None:
|
||||
action="store_true",
|
||||
help="Skip npm install/package and launch the existing unpacked app from apps/desktop/release",
|
||||
)
|
||||
gui_parser.add_argument(
|
||||
"--local",
|
||||
action="store_true",
|
||||
help="Show the local-models UI in the desktop app (models pane, quickstart, picker rows)",
|
||||
)
|
||||
gui_parser.add_argument(
|
||||
"--force-build",
|
||||
action="store_true",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -511,6 +511,28 @@ async def _lifespan(app: "FastAPI"):
|
||||
# sweeping stale sessions on schedule, independent of list requests.
|
||||
auto_archive_task = asyncio.create_task(_auto_archive_ticker_loop())
|
||||
|
||||
# Managed local runtime: when the user opted in (local_runtime.enabled,
|
||||
# set by the Local Models 'Use' action), bring the llama-server back up
|
||||
# so a restart doesn't strand a llamacpp main model without a backend.
|
||||
# Off-thread and best-effort: binary check + spawn + health poll must
|
||||
# not delay the server socket, and failure falls back to configured
|
||||
# cloud providers exactly like a cold start.
|
||||
def _boot_local_runtime():
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
|
||||
|
||||
# Server only — models load on first inference, always (residency
|
||||
# design: downloaded = available; demand loads; idleness
|
||||
# evicts). An empty router holds no VRAM; warming a model at
|
||||
# boot would reload gigabytes nobody asked for yet.
|
||||
ensure_local_runtime(load_config())
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logging.getLogger(__name__).warning("local runtime boot failed: %s", exc)
|
||||
|
||||
threading.Thread(target=_boot_local_runtime, daemon=True,
|
||||
name="local-runtime-boot").start()
|
||||
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
@@ -523,6 +545,14 @@ async def _lifespan(app: "FastAPI"):
|
||||
selftest_task.cancel()
|
||||
auto_archive_task.cancel()
|
||||
await PTY_REGISTRY.close_all()
|
||||
# Stop the managed llama-server with its parent — a supervisor-less
|
||||
# orphan would keep VRAM pinned after the app closes.
|
||||
try:
|
||||
from hermes_cli.local_runtime.bootstrap import shutdown_local_runtime
|
||||
|
||||
shutdown_local_runtime()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
if os.getenv("HERMES_DESKTOP") == "1":
|
||||
_terminate_desktop_managed_gateway()
|
||||
|
||||
@@ -3289,6 +3319,10 @@ def _git_path(path: str) -> str:
|
||||
from hermes_cli.web_routers import git as _git_routes # noqa: E402
|
||||
|
||||
app.include_router(_git_routes.router)
|
||||
|
||||
from hermes_cli.web_routers import local_models as _local_models_routes # noqa: E402
|
||||
|
||||
app.include_router(_local_models_routes.router)
|
||||
from hermes_cli.web_routers.git import ( # noqa: E402,F401 — legacy re-exports; tests call these via web_server.<name>
|
||||
git_status_route,
|
||||
git_worktrees_route,
|
||||
|
||||
+1
-1
@@ -587,7 +587,7 @@ py-modules = [
|
||||
include = ["agent", "agent.*", "tools", "tools.*", "hermes_cli", "hermes_cli.*", "gateway", "gateway.*", "tui_gateway", "tui_gateway.*", "cron", "cron.*", "acp_adapter", "plugins", "plugins.*", "providers", "providers.*"]
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
hermes_cli = ["observability/schemas/*.json", "data/*.json"]
|
||||
hermes_cli = ["observability/schemas/*.json", "data/*.json", "local_runtime/*.json"]
|
||||
# gateway/assets/ ships status_phrases.yaml and the Telegram BotFather
|
||||
# screenshot. Without this, sealed venvs (uv2nix) silently lose both —
|
||||
# status phrases fall back to the tiny hardcoded set and the Telegram
|
||||
|
||||
+135
-22
@@ -1976,6 +1976,64 @@ class AIAgent:
|
||||
review_memory: bool = False,
|
||||
review_skills: bool = False,
|
||||
focus: Optional[str] = None,
|
||||
explicit: bool = False,
|
||||
) -> None:
|
||||
"""Post-turn review entry point: decide WHEN, then spawn.
|
||||
|
||||
The decision to review (nudge intervals, enabled gate) already
|
||||
happened at the call site. This wrapper adds one policy: a review
|
||||
whose runtime resolves to the MANAGED LOCAL llama-server is queued
|
||||
for machine idle instead of spawned into the user's GPU mid-session
|
||||
(auxiliary.background_review.defer: auto|never). Everything else —
|
||||
cloud runtimes, external local servers, explicit /refine — spawns
|
||||
immediately, exactly as before.
|
||||
|
||||
``explicit`` marks a user-initiated review (/refine, with or
|
||||
without focus text): never deferred. It does NOT touch the
|
||||
delegate/enabled gates below — those stay keyed on ``focus`` so a
|
||||
bare /refine keeps its historical gating behavior.
|
||||
"""
|
||||
# Delegation-subagent and enabled gates run here at enqueue/spawn
|
||||
# time; the idle dispatcher re-checks the enabled gate again at
|
||||
# dispatch time so a review queued for minutes cannot be
|
||||
# resurrected after the user disables reviews.
|
||||
if focus is None and getattr(self, "_delegate_depth", 0) > 0:
|
||||
return
|
||||
task_cfg = None
|
||||
if focus is None:
|
||||
from agent.background_review import load_background_review_settings
|
||||
enabled, task_cfg = load_background_review_settings()
|
||||
if not enabled:
|
||||
return
|
||||
|
||||
kwargs = dict(
|
||||
messages_snapshot=messages_snapshot,
|
||||
review_memory=review_memory,
|
||||
review_skills=review_skills,
|
||||
focus=focus,
|
||||
task_cfg=task_cfg,
|
||||
)
|
||||
if focus is None and not explicit:
|
||||
from agent.review_idle_queue import (
|
||||
QUEUE,
|
||||
defer_mode,
|
||||
review_targets_managed_local,
|
||||
)
|
||||
if (defer_mode(task_cfg) == "auto"
|
||||
and review_targets_managed_local(self, task_cfg)):
|
||||
session_key = str(getattr(self, "session_id", None) or id(self))
|
||||
QUEUE.enqueue(self, session_key, kwargs)
|
||||
return
|
||||
self._spawn_background_review_now(**kwargs)
|
||||
|
||||
def _spawn_background_review_now(
|
||||
self,
|
||||
messages_snapshot: List[Dict],
|
||||
review_memory: bool = False,
|
||||
review_skills: bool = False,
|
||||
focus: Optional[str] = None,
|
||||
task_cfg: Optional[Dict[str, Any]] = None,
|
||||
_requeue_attempts: int = 0,
|
||||
) -> None:
|
||||
"""Spawn the background memory/skill review thread.
|
||||
|
||||
@@ -1988,28 +2046,17 @@ class AIAgent:
|
||||
``focus`` is optional user-supplied steering (from ``/refine``)
|
||||
appended to the review prompt — e.g. "save the deploy workflow as a
|
||||
skill". The automatic post-turn triggers never set it.
|
||||
|
||||
``task_cfg`` is the pre-loaded ``auxiliary.background_review``
|
||||
block from the entry wrapper (None on direct calls, e.g. /refine —
|
||||
the spawn path reads config itself then).
|
||||
|
||||
A deferred review preempted by a live turn is REQUEUED (bounded by
|
||||
``_requeue_attempts``) instead of lost: on the managed local
|
||||
runtime a review takes minutes, so cancel-and-forget — harmless on
|
||||
cloud, where reviews finish in seconds — would silently discard
|
||||
most learning on an active session.
|
||||
"""
|
||||
# A delegation subagent (``_delegate_depth > 0``) must not run the
|
||||
# automatic post-turn review. Subagents are ephemeral workers already
|
||||
# barred from writing shared MEMORY.md (``DELEGATE_BLOCKED_TOOLS``) and
|
||||
# are spawned with ``skip_memory=True``, so a review here has little to
|
||||
# persist — yet it inherits the subagent's (often premium) delegation
|
||||
# model and replays the whole conversation at premium rates, silently
|
||||
# inflating token cost (#85859). An explicit ``/refine`` (``focus`` set)
|
||||
# is a deliberate user request and still runs.
|
||||
if focus is None and getattr(self, "_delegate_depth", 0) > 0:
|
||||
return
|
||||
# Explicit off-switch for automatic post-turn forks
|
||||
# (``auxiliary.background_review.enabled: false``). Manual ``/refine``
|
||||
# still works — same contract as zeroing the nudge intervals (#87250).
|
||||
# Load the task block once here and pass it into the spawn path so
|
||||
# aux routing does not re-read config.
|
||||
task_cfg = None
|
||||
if focus is None:
|
||||
from agent.background_review import load_background_review_settings
|
||||
enabled, task_cfg = load_background_review_settings()
|
||||
if not enabled:
|
||||
return
|
||||
from agent.background_review import (
|
||||
finish_background_review_run,
|
||||
prepare_background_review_run,
|
||||
@@ -2030,10 +2077,25 @@ class AIAgent:
|
||||
task_cfg=task_cfg,
|
||||
review_run=review_run,
|
||||
)
|
||||
|
||||
def _target_with_requeue() -> None:
|
||||
target()
|
||||
self._maybe_requeue_preempted_review(
|
||||
review_run,
|
||||
dict(
|
||||
messages_snapshot=messages_snapshot,
|
||||
review_memory=review_memory,
|
||||
review_skills=review_skills,
|
||||
focus=focus,
|
||||
task_cfg=task_cfg,
|
||||
_requeue_attempts=_requeue_attempts + 1,
|
||||
),
|
||||
)
|
||||
|
||||
# Carry the active profile into the review thread so MEMORY.md /
|
||||
# skill review writes land in the right profile (#54937).
|
||||
t = threading.Thread(
|
||||
target=propagate_context_to_thread(target),
|
||||
target=propagate_context_to_thread(_target_with_requeue),
|
||||
daemon=True,
|
||||
name="bg-review",
|
||||
)
|
||||
@@ -2042,6 +2104,42 @@ class AIAgent:
|
||||
finish_background_review_run(self, review_run)
|
||||
raise
|
||||
|
||||
_REVIEW_REQUEUE_MAX_ATTEMPTS = 3
|
||||
|
||||
def _maybe_requeue_preempted_review(self, review_run, kwargs) -> None:
|
||||
"""Requeue a deferred-mode review that a live turn cancelled.
|
||||
|
||||
Only fires for automatic reviews whose runtime targets the managed
|
||||
local server (the deferred population); bounded attempts prevent a
|
||||
busy box from cycling one review forever — past the cap it is
|
||||
dropped exactly like the pre-deferral behavior dropped every
|
||||
cancelled review.
|
||||
"""
|
||||
try:
|
||||
if not review_run.cancel_requested.is_set():
|
||||
return # ran to completion (or never admitted for other reasons)
|
||||
if kwargs.get("focus") is not None:
|
||||
return
|
||||
if kwargs.get("_requeue_attempts", 0) > self._REVIEW_REQUEUE_MAX_ATTEMPTS:
|
||||
logger.info("Preempted background review dropped after %d requeues",
|
||||
self._REVIEW_REQUEUE_MAX_ATTEMPTS)
|
||||
return
|
||||
from agent.review_idle_queue import (
|
||||
QUEUE,
|
||||
defer_mode,
|
||||
review_targets_managed_local,
|
||||
)
|
||||
task_cfg = kwargs.get("task_cfg")
|
||||
if (defer_mode(task_cfg) != "auto"
|
||||
or not review_targets_managed_local(self, task_cfg)):
|
||||
return
|
||||
session_key = str(getattr(self, "session_id", None) or id(self))
|
||||
# kwargs carries the incremented _requeue_attempts through the
|
||||
# queue so the cap survives the round trip.
|
||||
QUEUE.enqueue(self, session_key, dict(kwargs))
|
||||
except Exception: # noqa: BLE001 — requeue is best-effort
|
||||
logger.debug("Preempted-review requeue failed", exc_info=True)
|
||||
|
||||
def _build_memory_write_metadata(
|
||||
self,
|
||||
*,
|
||||
@@ -9095,6 +9193,13 @@ class AIAgent:
|
||||
|
||||
cancel_background_review_for_live_turn(self)
|
||||
|
||||
# Turn liveness for the deferred-review idle queue: a queued review
|
||||
# must not dispatch into the settle gap between two quick prompts.
|
||||
# Marked inside the try below so the balancing note_turn_finished in
|
||||
# its finally covers every exit; the actual start-mark happens as the
|
||||
# first statement of the try.
|
||||
from agent.review_idle_queue import QUEUE as _review_queue
|
||||
|
||||
from agent.aux_accounting import (
|
||||
reset_accounting_context,
|
||||
set_accounting_context,
|
||||
@@ -9180,6 +9285,7 @@ class AIAgent:
|
||||
_clear_if_owned()
|
||||
|
||||
try:
|
||||
_review_queue.note_turn_started()
|
||||
# Serialize the full load -> run -> flush region across Hermes
|
||||
# processes. Gateway's asyncio lease closes alias routing inside one
|
||||
# process; this durable lease covers Desktop, CLI resume, gateway,
|
||||
@@ -9726,6 +9832,13 @@ class AIAgent:
|
||||
reset_conversation_context(token)
|
||||
if affinity_token is not None:
|
||||
reset_affinity_scope(affinity_token)
|
||||
# Balance the note_turn_started above — every exit path
|
||||
# lands here, so the idle queue's live-turn count cannot
|
||||
# leak upward and starve deferred reviews.
|
||||
try:
|
||||
_review_queue.note_turn_finished()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def chat(self, message: str, stream_callback: Optional[callable] = None) -> str:
|
||||
"""
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Propose catalog quality updates from Artificial Analysis.
|
||||
|
||||
Authoring-time helper — NEVER called at runtime (their terms forbid
|
||||
client-side keys, the fleet would burn the rate limit, and a
|
||||
recommendation must not change because a third-party endpoint
|
||||
hiccuped). Run it when adding a model or refreshing the ordering;
|
||||
review the printed diff and edit catalog.json yourself. The script
|
||||
proposes, the commit decides.
|
||||
|
||||
The catalog's `quality` stays OUR field: AA-informed where they cover a
|
||||
model, editorially set where they don't (day-0 releases lag their evals;
|
||||
some entries never appear). AA's Intelligence Index grades the
|
||||
full-precision cloud model, not our Q4 build — fine for ordering, never
|
||||
for display.
|
||||
|
||||
Usage:
|
||||
export AA_API_KEY=... # from https://artificialanalysis.ai (free tier)
|
||||
python scripts/aa_quality_sync.py
|
||||
|
||||
Attribution: scores by Artificial Analysis (https://artificialanalysis.ai).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
CATALOG_PATH = REPO_ROOT / "hermes_cli" / "local_runtime" / "catalog.json"
|
||||
AA_URL = "https://artificialanalysis.ai/api/v2/data/llms/models"
|
||||
|
||||
# Catalog entry id -> AA slug. Hand-maintained: AA's naming rarely matches
|
||||
# HF repo names, and a wrong match silently mis-ranks a model. An entry
|
||||
# absent here (or mapped to None) is editorial-only and never overwritten.
|
||||
AA_SLUG_BY_ENTRY = {
|
||||
"qwen3.8-27b": "qwen3-8-27b",
|
||||
"qwen3.8-flash-next": "qwen3-8-flash-next",
|
||||
"qwen3.6-35b-a3b": "qwen3-6-35b-a3b",
|
||||
"deepseek-v4-flash": "deepseek-v4-flash",
|
||||
}
|
||||
|
||||
|
||||
def fetch_aa_models(api_key: str) -> dict[str, dict]:
|
||||
req = urllib.request.Request(AA_URL, headers={"x-api-key": api_key})
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
doc = json.load(r)
|
||||
return {m["slug"]: m for m in doc.get("data", [])}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
api_key = os.environ.get("AA_API_KEY", "").strip()
|
||||
if not api_key:
|
||||
print("AA_API_KEY not set — create a free key at "
|
||||
"https://artificialanalysis.ai and export it.", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
catalog = json.loads(CATALOG_PATH.read_text(encoding="utf-8"))
|
||||
aa = fetch_aa_models(api_key)
|
||||
|
||||
print(f"{'entry':24s} {'catalog q':>9s} {'AA index':>9s} note")
|
||||
print("-" * 70)
|
||||
for model in catalog["models"]:
|
||||
entry_id = model["id"]
|
||||
current = model.get("quality", 0)
|
||||
slug = AA_SLUG_BY_ENTRY.get(entry_id)
|
||||
if not slug:
|
||||
print(f"{entry_id:24s} {current:>9d} {'—':>9s} editorial only (no AA mapping)")
|
||||
continue
|
||||
hit = aa.get(slug)
|
||||
if hit is None:
|
||||
print(f"{entry_id:24s} {current:>9d} {'—':>9s} not in AA data (slug {slug!r})")
|
||||
continue
|
||||
index = (hit.get("evaluations") or {}).get(
|
||||
"artificial_analysis_intelligence_index")
|
||||
if index is None:
|
||||
print(f"{entry_id:24s} {current:>9d} {'—':>9s} AA row lacks the index")
|
||||
continue
|
||||
proposed = round(float(index))
|
||||
marker = "" if proposed == current else " <-- proposes change"
|
||||
print(f"{entry_id:24s} {current:>9d} {proposed:>9d}{marker}")
|
||||
|
||||
print("\nReview against the decision table before editing: a quality "
|
||||
"change that flips cells in tests/hermes_cli/"
|
||||
"test_local_recommendation.py is the actual decision being made.")
|
||||
print("Attribution: scores by Artificial Analysis "
|
||||
"(https://artificialanalysis.ai).")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,347 @@
|
||||
"""Deferred background review on the managed local runtime.
|
||||
|
||||
Behavior contracts for agent/review_idle_queue.py and the decision
|
||||
wrapper in run_agent.AIAgent._spawn_background_review:
|
||||
|
||||
- defer: auto + review runtime == managed local -> queued, not spawned
|
||||
- defer: never, or non-managed runtime, or /refine -> immediate spawn
|
||||
- queue coalesces per session (newest snapshot wins, age preserved)
|
||||
- dispatch requires sustained process-quiet AND server idle
|
||||
- aged-out items dispatch regardless of idleness (delay, never lose)
|
||||
- preempted deferred reviews requeue with a bounded attempt cap
|
||||
"""
|
||||
|
||||
import threading
|
||||
import time
|
||||
import types
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.review_idle_queue import (
|
||||
ReviewIdleQueue,
|
||||
_IDLE_SETTLE_S,
|
||||
defer_max_age_s,
|
||||
defer_mode,
|
||||
)
|
||||
|
||||
|
||||
# ── config parsing ───────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_defer_mode_values():
|
||||
assert defer_mode(None) == "auto"
|
||||
assert defer_mode({}) == "auto"
|
||||
assert defer_mode({"defer": "auto"}) == "auto"
|
||||
assert defer_mode({"defer": "never"}) == "never"
|
||||
assert defer_mode({"defer": "NEVER"}) == "never"
|
||||
# Unknown values fall back to auto (the safe, documented default).
|
||||
assert defer_mode({"defer": "sometimes"}) == "auto"
|
||||
assert defer_mode({"defer": 3}) == "auto"
|
||||
|
||||
|
||||
def test_defer_max_age_parsing():
|
||||
assert defer_max_age_s(None) == 30 * 60
|
||||
assert defer_max_age_s({"defer_max_age_s": 120}) == 120.0
|
||||
assert defer_max_age_s({"defer_max_age_s": "600"}) == 600.0
|
||||
# Nonsense and non-positive fall back to the default.
|
||||
assert defer_max_age_s({"defer_max_age_s": "soon"}) == 30 * 60
|
||||
assert defer_max_age_s({"defer_max_age_s": 0}) == 30 * 60
|
||||
assert defer_max_age_s({"defer_max_age_s": -5}) == 30 * 60
|
||||
|
||||
|
||||
# ── queue harness ────────────────────────────────────────────────
|
||||
|
||||
|
||||
class _FakeAgent:
|
||||
def __init__(self):
|
||||
self.spawned = []
|
||||
self.session_id = "sess-x"
|
||||
|
||||
def _spawn_background_review_now(self, **kwargs):
|
||||
self.spawned.append(kwargs)
|
||||
|
||||
|
||||
def _make_queue(now=None, server_idle=True):
|
||||
q = ReviewIdleQueue()
|
||||
clock = {"t": 0.0}
|
||||
if now is None:
|
||||
q._now = lambda: clock["t"]
|
||||
else:
|
||||
q._now = now
|
||||
q._server_idle = lambda: server_idle
|
||||
# Never start the real dispatcher thread in unit tests.
|
||||
q._ensure_thread = lambda: None
|
||||
return q, clock
|
||||
|
||||
|
||||
def test_enqueue_coalesces_per_session_newest_wins_oldest_age():
|
||||
q, clock = _make_queue()
|
||||
agent = _FakeAgent()
|
||||
|
||||
clock["t"] = 100.0
|
||||
q.enqueue(agent, "s1", {"messages_snapshot": ["old"], "task_cfg": {}})
|
||||
clock["t"] = 200.0
|
||||
q.enqueue(agent, "s1", {"messages_snapshot": ["new"], "task_cfg": {}})
|
||||
q.enqueue(agent, "s2", {"messages_snapshot": ["other"], "task_cfg": {}})
|
||||
|
||||
assert q.pending_count() == 2
|
||||
with q._lock:
|
||||
item = q._pending["s1"]
|
||||
# Newest snapshot won, but the age clock kept the ORIGINAL enqueue
|
||||
# time so a busy session cannot push its own age-out forever.
|
||||
assert item.kwargs["messages_snapshot"] == ["new"]
|
||||
assert item.enqueued_at == 100.0
|
||||
|
||||
|
||||
def test_dispatch_waits_for_sustained_quiet():
|
||||
q, clock = _make_queue()
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {}})
|
||||
|
||||
# A live turn: nothing dispatches.
|
||||
q.note_turn_started()
|
||||
assert q._pop_dispatchable() is None
|
||||
|
||||
# Turn finished, but the settle window hasn't elapsed.
|
||||
q.note_turn_finished()
|
||||
assert q._pop_dispatchable() is None
|
||||
|
||||
# Quiet long enough -> dispatchable.
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
item = q._pop_dispatchable()
|
||||
assert item is not None and item.session_key == "s1"
|
||||
assert q.pending_count() == 0
|
||||
|
||||
|
||||
def test_dispatch_blocked_by_busy_server():
|
||||
q, clock = _make_queue(server_idle=False)
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {}})
|
||||
q.note_turn_started()
|
||||
q.note_turn_finished()
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
# Process is quiet but the managed server has a processing slot
|
||||
# (another profile's session, a live prefill): hold.
|
||||
assert q._pop_dispatchable() is None
|
||||
assert q.pending_count() == 1
|
||||
|
||||
|
||||
def test_aged_out_item_dispatches_despite_busy_server():
|
||||
q, clock = _make_queue(server_idle=False)
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {"defer_max_age_s": 60}})
|
||||
q.note_turn_started() # never goes quiet
|
||||
clock["t"] += 61
|
||||
item = q._pop_dispatchable()
|
||||
assert item is not None
|
||||
assert item.session_key == "s1"
|
||||
|
||||
|
||||
def test_new_turn_resets_the_quiet_clock():
|
||||
q, clock = _make_queue()
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {}})
|
||||
q.note_turn_started()
|
||||
q.note_turn_finished()
|
||||
clock["t"] += _IDLE_SETTLE_S - 2
|
||||
# A new prompt arrives just before the settle window closes.
|
||||
q.note_turn_started()
|
||||
clock["t"] += 30
|
||||
assert q._pop_dispatchable() is None # still live
|
||||
q.note_turn_finished()
|
||||
assert q._pop_dispatchable() is None # settle restarts
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
assert q._pop_dispatchable() is not None
|
||||
|
||||
|
||||
def test_nested_turns_require_all_to_finish():
|
||||
q, clock = _make_queue()
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {}})
|
||||
q.note_turn_started()
|
||||
q.note_turn_started()
|
||||
q.note_turn_finished()
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
assert q._pop_dispatchable() is None # one turn still live
|
||||
q.note_turn_finished()
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
assert q._pop_dispatchable() is not None
|
||||
|
||||
|
||||
# ── the decision wrapper ─────────────────────────────────────────
|
||||
|
||||
|
||||
def _wrapper_agent(monkeypatch, defer="auto", managed=True):
|
||||
"""A minimal object wearing the real _spawn_background_review."""
|
||||
import run_agent
|
||||
from agent import review_idle_queue as riq
|
||||
|
||||
agent = _FakeAgent()
|
||||
agent._delegate_depth = 0
|
||||
calls = {"enqueued": [], "spawned": []}
|
||||
|
||||
monkeypatch.setattr(
|
||||
"agent.background_review.load_background_review_settings",
|
||||
lambda: (True, {"defer": defer}),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
riq, "review_targets_managed_local", lambda a, cfg: managed
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
riq.QUEUE, "enqueue",
|
||||
lambda a, key, kw: calls["enqueued"].append((key, kw)),
|
||||
)
|
||||
agent._spawn_background_review_now = (
|
||||
lambda **kw: calls["spawned"].append(kw)
|
||||
)
|
||||
bound = types.MethodType(run_agent.AIAgent._spawn_background_review, agent)
|
||||
return bound, calls
|
||||
|
||||
|
||||
def test_wrapper_defers_managed_local_auto(monkeypatch):
|
||||
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
|
||||
spawn([{"role": "user", "content": "hi"}], review_memory=True)
|
||||
assert len(calls["enqueued"]) == 1
|
||||
assert calls["spawned"] == []
|
||||
key, kwargs = calls["enqueued"][0]
|
||||
assert key == "sess-x"
|
||||
assert kwargs["review_memory"] is True
|
||||
|
||||
|
||||
def test_wrapper_spawns_immediately_for_non_managed(monkeypatch):
|
||||
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=False)
|
||||
spawn([{"role": "user", "content": "hi"}], review_skills=True)
|
||||
assert calls["enqueued"] == []
|
||||
assert len(calls["spawned"]) == 1
|
||||
|
||||
|
||||
def test_wrapper_defer_never_is_old_behavior(monkeypatch):
|
||||
spawn, calls = _wrapper_agent(monkeypatch, defer="never", managed=True)
|
||||
spawn([{"role": "user", "content": "hi"}], review_memory=True)
|
||||
assert calls["enqueued"] == []
|
||||
assert len(calls["spawned"]) == 1
|
||||
|
||||
|
||||
def test_wrapper_refine_bypasses_queue(monkeypatch):
|
||||
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
|
||||
spawn([{"role": "user", "content": "hi"}], review_memory=True,
|
||||
focus="save the deploy workflow")
|
||||
assert calls["enqueued"] == []
|
||||
assert len(calls["spawned"]) == 1
|
||||
assert calls["spawned"][0]["focus"] == "save the deploy workflow"
|
||||
|
||||
|
||||
def test_wrapper_bare_refine_bypasses_queue(monkeypatch):
|
||||
"""/refine with no focus text is still explicit: never deferred."""
|
||||
spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
|
||||
spawn([{"role": "user", "content": "hi"}], review_memory=True,
|
||||
focus=None, explicit=True)
|
||||
assert calls["enqueued"] == []
|
||||
assert len(calls["spawned"]) == 1
|
||||
|
||||
|
||||
def test_wrapper_cloud_fast_path_skips_runtime_resolution(monkeypatch):
|
||||
"""No managed server on the machine -> the classifier answers from the
|
||||
TTL-cached netloc probe alone, without resolving the review runtime.
|
||||
Guards the cloud-only turn tail from growing new work."""
|
||||
from agent import review_idle_queue as riq
|
||||
|
||||
resolved = {"count": 0}
|
||||
|
||||
def _explode(agent, cfg):
|
||||
resolved["count"] += 1
|
||||
raise AssertionError("runtime resolution must not run")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"agent.auxiliary_client._managed_local_netloc", lambda: "")
|
||||
monkeypatch.setattr(
|
||||
"agent.background_review._resolve_review_runtime", _explode)
|
||||
assert riq.review_targets_managed_local(object(), {}) is False
|
||||
assert resolved["count"] == 0
|
||||
|
||||
|
||||
def test_dispatcher_rechecks_enabled_gate(monkeypatch):
|
||||
"""A review disabled while queued must not be resurrected at dispatch."""
|
||||
from agent import review_idle_queue as riq
|
||||
|
||||
q, clock = _make_queue()
|
||||
agent = _FakeAgent()
|
||||
q.enqueue(agent, "s1", {"task_cfg": {}})
|
||||
monkeypatch.setattr(
|
||||
"agent.background_review.load_background_review_settings",
|
||||
lambda: (False, {}),
|
||||
)
|
||||
item = None
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
q.note_turn_started()
|
||||
q.note_turn_finished()
|
||||
clock["t"] += _IDLE_SETTLE_S + 1
|
||||
item = q._pop_dispatchable()
|
||||
assert item is not None
|
||||
assert q._still_enabled(item) is False
|
||||
|
||||
|
||||
# ── requeue on preemption ────────────────────────────────────────
|
||||
|
||||
|
||||
class _Run:
|
||||
def __init__(self, cancelled):
|
||||
self.cancel_requested = threading.Event()
|
||||
if cancelled:
|
||||
self.cancel_requested.set()
|
||||
|
||||
|
||||
def _requeue_agent(monkeypatch, managed=True):
|
||||
import run_agent
|
||||
from agent import review_idle_queue as riq
|
||||
|
||||
agent = _FakeAgent()
|
||||
calls = {"enqueued": []}
|
||||
monkeypatch.setattr(
|
||||
riq, "review_targets_managed_local", lambda a, cfg: managed
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
riq.QUEUE, "enqueue",
|
||||
lambda a, key, kw: calls["enqueued"].append(kw),
|
||||
)
|
||||
agent._REVIEW_REQUEUE_MAX_ATTEMPTS = (
|
||||
run_agent.AIAgent._REVIEW_REQUEUE_MAX_ATTEMPTS
|
||||
)
|
||||
bound = types.MethodType(
|
||||
run_agent.AIAgent._maybe_requeue_preempted_review, agent
|
||||
)
|
||||
return bound, calls
|
||||
|
||||
|
||||
def test_preempted_review_requeues(monkeypatch):
|
||||
requeue, calls = _requeue_agent(monkeypatch)
|
||||
requeue(_Run(cancelled=True),
|
||||
{"task_cfg": {"defer": "auto"}, "focus": None,
|
||||
"_requeue_attempts": 1})
|
||||
assert len(calls["enqueued"]) == 1
|
||||
# The attempt counter rides along so the cap survives the round trip.
|
||||
assert calls["enqueued"][0]["_requeue_attempts"] == 1
|
||||
|
||||
|
||||
def test_completed_review_does_not_requeue(monkeypatch):
|
||||
requeue, calls = _requeue_agent(monkeypatch)
|
||||
requeue(_Run(cancelled=False),
|
||||
{"task_cfg": {"defer": "auto"}, "focus": None,
|
||||
"_requeue_attempts": 1})
|
||||
assert calls["enqueued"] == []
|
||||
|
||||
|
||||
def test_requeue_attempt_cap(monkeypatch):
|
||||
requeue, calls = _requeue_agent(monkeypatch)
|
||||
requeue(_Run(cancelled=True),
|
||||
{"task_cfg": {"defer": "auto"}, "focus": None,
|
||||
"_requeue_attempts": 4})
|
||||
assert calls["enqueued"] == []
|
||||
|
||||
|
||||
def test_requeue_skips_non_managed(monkeypatch):
|
||||
requeue, calls = _requeue_agent(monkeypatch, managed=False)
|
||||
requeue(_Run(cancelled=True),
|
||||
{"task_cfg": {"defer": "auto"}, "focus": None,
|
||||
"_requeue_attempts": 1})
|
||||
assert calls["enqueued"] == []
|
||||
@@ -158,6 +158,94 @@ class TestShouldExclude:
|
||||
# The .db itself is still included (and safe-copied separately)
|
||||
assert not _should_exclude(Path("state.db"))
|
||||
|
||||
def test_excludes_managed_runtime_trees_at_root(self):
|
||||
"""models/, runtimes/, and node/ at a profile-home root hold
|
||||
re-downloadable GGUF weights and runtime binaries that reach
|
||||
hundreds of GB — zipping them is the 20-minute-hang symptom."""
|
||||
from hermes_cli.backup import _should_exclude
|
||||
assert _should_exclude(Path("models/Qwen3.6-27B-Q4_K_M.gguf"))
|
||||
assert _should_exclude(Path("models/assets/mmproj.gguf"))
|
||||
assert _should_exclude(Path("runtimes/llamacpp/b10362/cuda/ggml-cuda.dll"))
|
||||
assert _should_exclude(Path("node/node.exe"))
|
||||
# Named profiles download their own copies.
|
||||
assert _should_exclude(Path("profiles/clean/models/big.gguf"))
|
||||
assert _should_exclude(Path("profiles/clean/runtimes/llamacpp/x.dll"))
|
||||
|
||||
def test_keeps_nested_dirs_named_like_runtime_trees(self):
|
||||
"""A deeper directory that happens to be called models/ or node/ is
|
||||
user data (a skill's assets, project files) and must survive."""
|
||||
from hermes_cli.backup import _should_exclude
|
||||
assert not _should_exclude(Path("skills/mlops/models/notes.md"))
|
||||
assert not _should_exclude(Path("scratch/node/index.js"))
|
||||
assert not _should_exclude(Path("profiles/clean/skills/x/models/a.txt"))
|
||||
|
||||
def test_excludes_desktop_emergency_state_db_baks(self):
|
||||
"""The desktop updater's pre-flight drops timestamped
|
||||
state.db.pre-update-emergency-*.bak files at the HERMES_HOME root —
|
||||
backup artifacts in the same class as backups/, so a full backup
|
||||
must not re-ship them."""
|
||||
from hermes_cli.backup import _should_exclude
|
||||
assert _should_exclude(
|
||||
Path("state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
|
||||
)
|
||||
assert _should_exclude(
|
||||
Path("profiles/coder/state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
|
||||
)
|
||||
# Other .bak files are user data and stay.
|
||||
assert not _should_exclude(Path("config.yaml.bak"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _iter_backup_files tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestIterBackupFiles:
|
||||
def test_manual_and_automatic_paths_share_one_walk(self, tmp_path):
|
||||
"""Both backup entry points must select the identical file set.
|
||||
|
||||
Before the walks were unified, the automatic pre-update path pruned
|
||||
``hermes-agent`` at ANY depth, silently dropping nested skill dirs
|
||||
like ``skills/autonomous-ai-agents/hermes-agent/`` that the manual
|
||||
path preserved. One shared iterator makes that drift impossible;
|
||||
this test pins the contract."""
|
||||
from hermes_cli.backup import _iter_backup_files
|
||||
|
||||
root = tmp_path / ".hermes"
|
||||
root.mkdir()
|
||||
_make_hermes_tree(root)
|
||||
|
||||
# The case the old automatic walk got wrong: a nested dir named
|
||||
# hermes-agent holding real skill content.
|
||||
nested = root / "skills" / "autonomous-ai-agents" / "hermes-agent"
|
||||
nested.mkdir(parents=True)
|
||||
(nested / "SKILL.md").write_text("# nested skill\n")
|
||||
|
||||
# A root-level managed runtime tree that both paths must prune.
|
||||
(root / "models").mkdir()
|
||||
(root / "models" / "big.gguf").write_bytes(b"\x00" * 64)
|
||||
|
||||
out_path = tmp_path / "out.zip"
|
||||
selected = {str(rel) for _, rel in _iter_backup_files(root, out_path)}
|
||||
|
||||
rel_nested = str(Path("skills/autonomous-ai-agents/hermes-agent/SKILL.md"))
|
||||
assert rel_nested in selected
|
||||
assert str(Path("models/big.gguf")) not in selected
|
||||
assert not any(s.startswith("hermes-agent") for s in selected)
|
||||
|
||||
def test_skipped_dirs_collected_for_summary(self, tmp_path):
|
||||
from hermes_cli.backup import _iter_backup_files
|
||||
|
||||
root = tmp_path / ".hermes"
|
||||
root.mkdir()
|
||||
_make_hermes_tree(root)
|
||||
(root / "models").mkdir()
|
||||
(root / "models" / "big.gguf").write_bytes(b"\x00")
|
||||
|
||||
skipped: set = set()
|
||||
list(_iter_backup_files(root, tmp_path / "out.zip", skipped))
|
||||
assert "models" in skipped
|
||||
assert "hermes-agent" in skipped
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Backup tests
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Every staged model must launch with a policy decision, never stock fit.
|
||||
|
||||
The managed server autoloads any GGUF in its models dir; a model missing
|
||||
from the preset INI loads with llama-server defaults (f16 KV at max
|
||||
context, no placement) — on Windows/WDDM that silently demotes VRAM and
|
||||
decodes at a crawl. Boot must therefore refuse to adopt a running server
|
||||
whose presets predate the staged set."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def hermes_home(tmp_path, monkeypatch):
|
||||
home = tmp_path / ".hermes"
|
||||
home.mkdir()
|
||||
monkeypatch.setenv("HERMES_HOME", str(home))
|
||||
return home
|
||||
|
||||
|
||||
def _stage(home, name):
|
||||
mdir = home / "models"
|
||||
mdir.mkdir(parents=True, exist_ok=True)
|
||||
(mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 32)
|
||||
|
||||
|
||||
def _write_presets(home, *model_ids):
|
||||
pdir = home / "runtimes" / "llamacpp"
|
||||
pdir.mkdir(parents=True, exist_ok=True)
|
||||
body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
|
||||
(pdir / "presets.ini").write_text(body, encoding="utf-8")
|
||||
|
||||
|
||||
def test_presets_stale_when_a_staged_model_has_no_section(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
_stage(hermes_home, "model-a")
|
||||
_stage(hermes_home, "model-b")
|
||||
_write_presets(hermes_home, "model-a")
|
||||
assert _presets_stale() is True
|
||||
|
||||
|
||||
def test_presets_current_when_every_staged_model_is_covered(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
_stage(hermes_home, "model-a")
|
||||
_write_presets(hermes_home, "model-a")
|
||||
assert _presets_stale() is False
|
||||
|
||||
|
||||
def test_no_models_is_never_stale(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
_write_presets(hermes_home, "model-a")
|
||||
assert _presets_stale() is False
|
||||
|
||||
|
||||
def test_boot_replaces_incumbent_with_stale_presets(hermes_home, monkeypatch):
|
||||
"""ensure_local_runtime must not adopt a running server whose presets
|
||||
miss a staged model — it stops it and boots fresh (boot itself is
|
||||
stubbed; the contract under test is the adopt/replace decision)."""
|
||||
import hermes_cli.local_runtime.bootstrap as boot
|
||||
|
||||
_stage(hermes_home, "model-a")
|
||||
_stage(hermes_home, "model-b")
|
||||
_write_presets(hermes_home, "model-a")
|
||||
|
||||
stopped = {}
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.endpoint._state_endpoint",
|
||||
lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 12345})
|
||||
monkeypatch.setattr(boot, "_stop_state_server",
|
||||
lambda state: stopped.setdefault("pid", state["pid"]))
|
||||
|
||||
sentinel = object()
|
||||
|
||||
def fake_boot(*a, **k):
|
||||
raise _BootReached()
|
||||
|
||||
class _BootReached(Exception):
|
||||
pass
|
||||
|
||||
# Fail fast once boot proper begins — reaching it IS the assertion.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.ensure_runtime_installed", fake_boot)
|
||||
|
||||
result = boot.ensure_local_runtime({"local_runtime": {"enabled": True}})
|
||||
assert stopped.get("pid") == 12345, "stale incumbent was not stopped"
|
||||
# Boot proceeded past adoption (our fake raised inside the try block,
|
||||
# which ensure_local_runtime swallows into a None return).
|
||||
assert result is None or result is sentinel
|
||||
|
||||
|
||||
def test_refresh_bounces_an_adopted_server(hermes_home, monkeypatch):
|
||||
"""refresh_local_runtime with no in-process supervisor but a running
|
||||
state-file server (the post-restart shape) must stop that server and
|
||||
boot fresh — NOT silently no-op. Regression: the no-op meant every
|
||||
download/delete after a backend restart left the router serving a
|
||||
stale model catalog, and picking the new model failed with
|
||||
'not found in this provider's model listing'."""
|
||||
import hermes_cli.local_runtime.bootstrap as boot
|
||||
|
||||
stopped = {}
|
||||
monkeypatch.setattr(boot, "_SUPERVISOR", None)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.endpoint._state_endpoint",
|
||||
lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 4242})
|
||||
monkeypatch.setattr(boot, "_stop_state_server",
|
||||
lambda state: stopped.setdefault("pid", state["pid"]))
|
||||
booted = {}
|
||||
monkeypatch.setattr(boot, "ensure_local_runtime",
|
||||
lambda cfg, force=False: booted.setdefault("force", force) or object())
|
||||
|
||||
assert boot.refresh_local_runtime() is True
|
||||
assert stopped.get("pid") == 4242, "adopted server was not stopped"
|
||||
assert booted.get("force") is True, "fresh boot did not follow the stop"
|
||||
|
||||
|
||||
def test_refresh_no_server_anywhere_is_a_noop(hermes_home, monkeypatch):
|
||||
import hermes_cli.local_runtime.bootstrap as boot
|
||||
|
||||
monkeypatch.setattr(boot, "_SUPERVISOR", None)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.endpoint._state_endpoint", lambda: None)
|
||||
assert boot.refresh_local_runtime() is False
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Launch and growth decisions must price against CAPACITY, not live-free
|
||||
VRAM. Both execute through a server bounce — the outgoing instance's memory
|
||||
is freed before the new one loads — so a probe that reads the predecessor's
|
||||
(or the grown model's own) residency as 'gone' vetoes configurations that
|
||||
genuinely fit. Symptom when this regresses: a model the pane promised
|
||||
'144K on GPU' launches with its weights pinned to CPU and single-digit
|
||||
tokens/s while the card sits 60% empty."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import inspect
|
||||
|
||||
|
||||
def _planning_probe_calls(source: str) -> list[bool]:
|
||||
"""Every probe_budget(...) call's planning= value in the source."""
|
||||
tree = ast.parse(source)
|
||||
out = []
|
||||
for node in ast.walk(tree):
|
||||
if (isinstance(node, ast.Call)
|
||||
and getattr(node.func, "id", getattr(node.func, "attr", ""))
|
||||
== "probe_budget"):
|
||||
planning = any(
|
||||
kw.arg == "planning"
|
||||
and isinstance(kw.value, ast.Constant)
|
||||
and kw.value.value is True
|
||||
for kw in node.keywords)
|
||||
out.append(planning)
|
||||
return out
|
||||
|
||||
|
||||
def test_bootstrap_presets_price_against_capacity():
|
||||
import hermes_cli.local_runtime.bootstrap as bootstrap
|
||||
|
||||
calls = _planning_probe_calls(inspect.getsource(bootstrap))
|
||||
assert calls, "bootstrap no longer probes a budget? update this test"
|
||||
assert all(calls), (
|
||||
"bootstrap prices launch decisions against live-free VRAM; a "
|
||||
"restart/refresh probes while the outgoing server still holds the "
|
||||
"card, pinning fitting models to CPU")
|
||||
|
||||
|
||||
def test_growth_refit_prices_against_capacity():
|
||||
import hermes_cli.local_runtime.growth as growth
|
||||
|
||||
calls = _planning_probe_calls(inspect.getsource(growth))
|
||||
assert calls, "growth no longer probes a budget? update this test"
|
||||
assert all(calls), (
|
||||
"growth re-fits against live-free VRAM; the grown model's own "
|
||||
"residency reads as unavailable and vetoes rungs that fit the "
|
||||
"post-bounce card")
|
||||
@@ -0,0 +1,118 @@
|
||||
"""The pulled catalog: packaged JSON is the offline truth, a GitHub fetch
|
||||
swaps entries in memory only, and min_engine gates day-0 models.
|
||||
|
||||
Nothing here touches disk beyond the packaged file — the design constraint
|
||||
is that a git checkout must never see a dirty tracked catalog.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import io
|
||||
import json
|
||||
import urllib.request
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.local_runtime.catalog as cat
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_refresh_state(monkeypatch):
|
||||
"""Each test starts outside the TTL window with the packaged catalog."""
|
||||
monkeypatch.setattr(cat, "_last_refresh_attempt", 0.0)
|
||||
packaged = cat._packaged_catalog()
|
||||
monkeypatch.setattr(cat, "CATALOG", packaged)
|
||||
yield
|
||||
|
||||
|
||||
def _doc_from(entries):
|
||||
"""A fetchable catalog document built by mutating the packaged JSON."""
|
||||
from importlib.resources import files
|
||||
|
||||
doc = json.loads(files("hermes_cli.local_runtime")
|
||||
.joinpath("catalog.json").read_text(encoding="utf-8"))
|
||||
doc["models"] = entries(doc["models"])
|
||||
return doc
|
||||
|
||||
|
||||
def _fetch_returns(monkeypatch, doc):
|
||||
body = json.dumps(doc).encode()
|
||||
|
||||
class R(io.BytesIO):
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr(urllib.request, "urlopen",
|
||||
lambda *a, **k: R(body))
|
||||
|
||||
|
||||
def test_packaged_json_round_trips_the_catalog():
|
||||
"""The packaged JSON must produce a complete, selection-ready catalog:
|
||||
every entry carries estimator inputs and at least one variant, and the
|
||||
known invariants (best-first ordering, Q4 floor) hold — the same
|
||||
contract the literals obeyed."""
|
||||
assert len(cat.CATALOG) >= 4
|
||||
for e in cat.CATALOG:
|
||||
assert e.variants and e.n_ctx_train > 0 and e.per_layer_f16 >= 0
|
||||
sizes = [v.size_bytes for v in e.variants]
|
||||
assert sizes == sorted(sizes, reverse=True), f"{e.id} not best-first"
|
||||
|
||||
|
||||
def test_refresh_swaps_in_memory_only(monkeypatch, tmp_path):
|
||||
"""A fetched catalog replaces CATALOG in memory; the packaged file on
|
||||
disk is untouched (checkout stays clean)."""
|
||||
from importlib.resources import files
|
||||
|
||||
packaged_path = files("hermes_cli.local_runtime").joinpath("catalog.json")
|
||||
before = packaged_path.read_text(encoding="utf-8")
|
||||
|
||||
def add_day0(models):
|
||||
day0 = dict(models[0])
|
||||
day0.update(id="day0-model", display_name="Day 0",
|
||||
description="new", min_engine="b99999")
|
||||
return models + [day0]
|
||||
|
||||
_fetch_returns(monkeypatch, _doc_from(add_day0))
|
||||
assert cat.refresh_catalog(force=True) is True
|
||||
assert "day0-model" in {e.id for e in cat.CATALOG}
|
||||
assert cat.catalog_by_id()["day0-model"].min_engine == "b99999"
|
||||
assert packaged_path.read_text(encoding="utf-8") == before
|
||||
|
||||
|
||||
def test_refresh_failure_keeps_current_catalog(monkeypatch):
|
||||
def boom(*a, **k):
|
||||
raise OSError("offline")
|
||||
|
||||
monkeypatch.setattr(urllib.request, "urlopen", boom)
|
||||
ids_before = [e.id for e in cat.CATALOG]
|
||||
assert cat.refresh_catalog(force=True) is False
|
||||
assert [e.id for e in cat.CATALOG] == ids_before
|
||||
|
||||
|
||||
def test_refresh_rejects_wrong_schema(monkeypatch):
|
||||
doc = _doc_from(lambda m: m)
|
||||
doc["schema_version"] = 2
|
||||
_fetch_returns(monkeypatch, doc)
|
||||
ids_before = [e.id for e in cat.CATALOG]
|
||||
assert cat.refresh_catalog(force=True) is False
|
||||
assert [e.id for e in cat.CATALOG] == ids_before
|
||||
|
||||
|
||||
def test_loader_ignores_unknown_fields():
|
||||
doc = _doc_from(lambda m: m)
|
||||
doc["models"][0]["future_field"] = {"anything": True}
|
||||
entries = cat._load_catalog(doc)
|
||||
assert entries[0].id == doc["models"][0]["id"]
|
||||
|
||||
|
||||
def test_min_engine_gate(monkeypatch):
|
||||
from hermes_cli.web_routers.local_models import _engine_too_old
|
||||
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_tags",
|
||||
lambda: ["b10362"])
|
||||
assert _engine_too_old("") is False, "no requirement, no gate"
|
||||
assert _engine_too_old("b10000") is False, "installed engine suffices"
|
||||
assert _engine_too_old("b10363") is True, "newer requirement gates"
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Catalog reachability: every entry's repo and files must exist upstream.
|
||||
|
||||
Existence is the contract; SIZES are advisory (they feed the estimator and
|
||||
progress bars, and downloads deliberately tolerate a stale size when
|
||||
upstream re-uploads — completeness is judged against the server's own
|
||||
declared length, never the catalog). Size drift prints as a warning so a
|
||||
catalog refresh can be batched deliberately; only a MISSING file or repo
|
||||
fails.
|
||||
|
||||
Network-marked (skipped in hermetic CI unless explicitly enabled) — this is
|
||||
the test that catches wrong repo names (the Nemotron 401) and moved files.
|
||||
Run before any catalog commit:
|
||||
|
||||
HERMES_TEST_NETWORK=1 scripts/run_tests.sh tests/hermes_cli/test_catalog_reachability.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import urllib.request
|
||||
|
||||
import pytest
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not os.environ.get("HERMES_TEST_NETWORK"),
|
||||
reason="network test; set HERMES_TEST_NETWORK=1 to run",
|
||||
)
|
||||
|
||||
|
||||
def test_every_catalog_file_resolves():
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
|
||||
problems = []
|
||||
drift = []
|
||||
for entry in CATALOG:
|
||||
url = f"https://huggingface.co/api/models/{entry.repo}/tree/main?recursive=true"
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=30) as r:
|
||||
files = {f["path"]: f.get("size") for f in json.load(r)}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
problems.append(f"{entry.id}: repo {entry.repo} unreachable ({exc})")
|
||||
continue
|
||||
for variant in entry.variants:
|
||||
for asset in entry.download_files(variant):
|
||||
if asset.path not in files:
|
||||
problems.append(
|
||||
f"{entry.id}/{variant.quant}: {asset.path} not in {entry.repo}")
|
||||
continue
|
||||
live_size = files[asset.path]
|
||||
if live_size and live_size != asset.size_bytes:
|
||||
drift.append(
|
||||
f"{entry.id}/{variant.quant}: size drift on {asset.path} — "
|
||||
f"catalog {asset.size_bytes} vs live {live_size}")
|
||||
if drift:
|
||||
print("\nADVISORY size drift (downloads tolerate this; refresh when convenient):")
|
||||
print("\n".join(drift))
|
||||
assert not problems, "\n".join(problems)
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
"""Variant-selection contracts: fit the catalog's single Q4-class build
|
||||
to a machine and price it honestly. Pure decision-table tests over
|
||||
synthetic budgets."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.local_runtime.catalog import (
|
||||
CATALOG,
|
||||
catalog_by_id,
|
||||
find_entry_for_model,
|
||||
select_variant,
|
||||
)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
GIB = 1 << 30
|
||||
|
||||
|
||||
def budget(vram_gib: float, ram_gib: float = 64) -> HardwareBudget:
|
||||
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
|
||||
total_device_bytes=int(vram_gib * GIB),
|
||||
ram_available_bytes=int(ram_gib * GIB))
|
||||
|
||||
|
||||
def test_every_entry_ships_exactly_one_q4_build():
|
||||
"""No quant ladder: one Q4-class build per entry (K_M where the repo
|
||||
ships it, XL elsewhere) — the quant class current engines optimize
|
||||
for. Nothing below Q4 ever ships. Validation status is explicit per
|
||||
variant in catalog.json; unvalidated builds are permitted (day-0
|
||||
entries) and surface as unbadged rows in the pane."""
|
||||
for entry in CATALOG:
|
||||
assert len(entry.variants) == 1, (
|
||||
f"{entry.id}: {len(entry.variants)} variants — expected exactly one")
|
||||
build = entry.variants[0]
|
||||
assert build.quant.startswith(("UD-Q4", "Q4")), (
|
||||
f"{entry.id}: ships {build.quant}, not a Q4-class build")
|
||||
for asset in entry.download_files(build):
|
||||
assert asset.size_bytes > 0, f"{entry.id}: no size on {asset.path}"
|
||||
|
||||
|
||||
def test_split_variants_have_coherent_parts():
|
||||
"""Multi-file variants: same model_id from every part, exact sizes,
|
||||
first file is the load target."""
|
||||
entry = catalog_by_id()["deepseek-v4-flash"]
|
||||
for v in entry.variants:
|
||||
assert len(v.files) >= 2, "deepseek ships split GGUFs"
|
||||
assert "00001-of" in v.files[0].path, "first part must be the load target"
|
||||
assert v.size_bytes == sum(f.size_bytes for f in v.files)
|
||||
assert entry.draft is not None, "DSpark draft rides along"
|
||||
|
||||
|
||||
def test_selection_is_the_q4_build_even_with_headroom():
|
||||
"""The selector picks the Q4 build even when bigger quants would fit
|
||||
with room to spare — headroom buys window, not quant. Larger builds
|
||||
stay one tile click away in the pane."""
|
||||
entry = catalog_by_id()["qwen3.8-27b"]
|
||||
choice = select_variant(entry, budget(60))
|
||||
assert choice is not None
|
||||
assert choice.zero_spill
|
||||
assert choice.variant.quant == entry.variants[-1].quant # the Q4 rung
|
||||
assert choice.reason_key == "best-large-window"
|
||||
|
||||
|
||||
def test_selected_build_constant_and_fit_shape_monotone_in_vram():
|
||||
"""More VRAM never changes the selected build (always the Q4 rung);
|
||||
what improves is the fit shape: spilled -> floor -> target window."""
|
||||
entry = catalog_by_id()["qwen3.8-27b"]
|
||||
quants = set()
|
||||
shapes = []
|
||||
rank = {"smallest-fits-spilled": 0, "best-fits": 1, "best-large-window": 2}
|
||||
for vram in (8, 12, 16, 24, 32, 48):
|
||||
choice = select_variant(entry, budget(vram))
|
||||
assert choice is not None
|
||||
quants.add(choice.variant.quant)
|
||||
shapes.append(rank[choice.reason_key])
|
||||
assert quants == {entry.variants[-1].quant}, f"selection not constant: {quants}"
|
||||
assert shapes == sorted(shapes), f"fit shape not monotone in VRAM: {shapes}"
|
||||
|
||||
|
||||
def test_small_card_gets_q4_spilled_never_below():
|
||||
"""8 GiB card + 27B: nothing zero-spills. The floor holds — the
|
||||
selector offers Q4 spilled (priced honestly), never a sub-Q4 build."""
|
||||
entry = catalog_by_id()["qwen3.8-27b"]
|
||||
choice = select_variant(entry, budget(8))
|
||||
assert choice is not None
|
||||
assert not choice.zero_spill
|
||||
assert choice.reason_key == "smallest-fits-spilled"
|
||||
assert choice.variant.quant == "UD-Q4_K_M"
|
||||
|
||||
|
||||
def test_frontier_model_refused_on_consumer_card_offered_on_big_ram():
|
||||
"""DeepSeek V4 Flash (161 GB at Q4): refused outright on a 32 GiB-RAM
|
||||
desktop; offered spilled on a 192 GiB-RAM workstation. The catalog
|
||||
carries frontier hardware honestly instead of hiding the model."""
|
||||
entry = catalog_by_id()["deepseek-v4-flash"]
|
||||
assert select_variant(entry, budget(32, ram_gib=32)) is None
|
||||
big = select_variant(entry, budget(32, ram_gib=192))
|
||||
assert big is not None and not big.zero_spill
|
||||
|
||||
|
||||
def test_selection_accounts_for_kv_not_just_weights():
|
||||
"""The zero-spill check prices weights + KV, not weights alone: give a
|
||||
machine exactly enough VRAM for the build's weights and the fit must
|
||||
come back spilled, not zero-spill."""
|
||||
entry = catalog_by_id()["qwen3.8-27b"]
|
||||
build = entry.variants[0]
|
||||
exactly_weights = HardwareBudget(
|
||||
usable_vram_bytes=build.size_bytes + (100 << 20),
|
||||
total_device_bytes=build.size_bytes + (100 << 20),
|
||||
ram_available_bytes=64 * GIB)
|
||||
choice = select_variant(entry, exactly_weights)
|
||||
assert choice is not None
|
||||
assert not choice.zero_spill, "KV cost ignored — weights alone can't zero-spill"
|
||||
|
||||
|
||||
def test_floor_fallback_when_target_window_does_not_fit():
|
||||
"""Cards where nothing clears the target keep the old rule: highest
|
||||
quality that zero-spills at the 64K floor (reason 'best-fits'), never
|
||||
a needless step down."""
|
||||
entry = catalog_by_id()["qwen3.8-27b"]
|
||||
# ~23.5 GiB usable: Q4 weights (16.7 GiB in-memory) + floor KV (2.2)
|
||||
# + overhead (1.5 + 0.9 mmproj + ~1.0 MTP-posture logits) fits, but
|
||||
# the 144K-target KV (+2.7 more) does not.
|
||||
choice = select_variant(entry, budget(23.5))
|
||||
assert choice is not None and choice.zero_spill
|
||||
assert choice.reason_key == "best-fits"
|
||||
assert choice.variant.quant == "UD-Q4_K_M"
|
||||
|
||||
|
||||
def test_target_never_degrades_below_floor_choice():
|
||||
"""The target preference may only IMPROVE the window, never the
|
||||
floor guarantees: whenever the old floor rule found a zero-spill pick,
|
||||
the new rule also finds one (possibly a smaller quant, never spill)."""
|
||||
for entry in CATALOG:
|
||||
for vram in (8, 12, 16, 24, 32, 48, 96):
|
||||
choice = select_variant(entry, budget(vram, ram_gib=256))
|
||||
if choice is None:
|
||||
continue
|
||||
# Rule 2: whatever was chosen zero-spill must genuinely clear
|
||||
# the floor (the selector's own invariant, re-checked).
|
||||
if choice.zero_spill:
|
||||
assert choice.reason_key in ("best-large-window", "best-fits")
|
||||
|
||||
|
||||
def test_find_entry_for_model_resolves_split_ids():
|
||||
hit = find_entry_for_model("DeepSeek-V4-Flash-0731-UD-Q4_K_XL")
|
||||
assert hit is not None
|
||||
entry, variant = hit
|
||||
assert entry.id == "deepseek-v4-flash"
|
||||
assert variant.quant == "UD-Q4_K_XL"
|
||||
|
||||
|
||||
def test_hybrid_long_context_stays_cheap():
|
||||
"""The reason Nemotron/Qwen3.6 headline the catalog: their priced
|
||||
64K-floor KV must be a small fraction of a dense model's."""
|
||||
from hermes_cli.local_runtime.catalog import FLOOR
|
||||
from hermes_cli.local_runtime.estimator import ctx_bytes
|
||||
|
||||
from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
|
||||
|
||||
hybrid = catalog_by_id()["qwen3.6-35b-a3b"]
|
||||
hybrid_profile = hybrid.profile(hybrid.variants[-1])
|
||||
# A fully-dense profile of the same layer count and per-layer cost:
|
||||
# the contract is about LAYER ECONOMICS (recurrent layers pay no
|
||||
# per-token KV), not about any particular catalog entry.
|
||||
n_layers = len(hybrid_profile.layers)
|
||||
dense_profile = ModelProfile(
|
||||
name="synthetic-dense", weights_bytes=hybrid_profile.weights_bytes,
|
||||
embd_table_bytes=0, n_ctx_train=hybrid.n_ctx_train,
|
||||
layers=[(LayerKind.FULL, hybrid.per_layer_f16)] * n_layers)
|
||||
dense_kv = ctx_bytes(dense_profile, FLOOR)
|
||||
hybrid_kv = ctx_bytes(hybrid_profile, FLOOR)
|
||||
# The contract is structural: recurrent layers pay no per-token KV,
|
||||
# so the hybrid's KV must track its full-attention share (x kv_scale
|
||||
# for MTP's draft context), not its total layer count.
|
||||
full = sum(1 for kind, _ in hybrid_profile.layers if kind == LayerKind.FULL)
|
||||
expected = dense_kv * full / n_layers * hybrid_profile.kv_scale
|
||||
assert hybrid_kv < dense_kv, "hybrid must be cheaper than dense"
|
||||
assert abs(hybrid_kv - expected) / expected < 0.25, (
|
||||
f"hybrid KV ({hybrid_kv:,}) should track its full-attention share "
|
||||
f"(expected ~{expected:,.0f})")
|
||||
@@ -0,0 +1,410 @@
|
||||
"""Context-policy decision-table tests (Rollout 3).
|
||||
|
||||
Per the design's verification plan: synthetic per-layer profiles ->
|
||||
relationships, never exact numbers. Real-model spot checks pin the
|
||||
estimator to constants measured on real GGUFs (those ARE relationships —
|
||||
constants with tolerance bands, not change-detecting catalog snapshots).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
FLOOR,
|
||||
SPEED_FLOOR_TOK_S,
|
||||
GrowthDecision,
|
||||
WindowDecision,
|
||||
growth_decision,
|
||||
initial_window,
|
||||
ladder,
|
||||
launch_args,
|
||||
spill_overrides,
|
||||
ub_logits_bytes,
|
||||
)
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget,
|
||||
LayerKind,
|
||||
ModelProfile,
|
||||
PhysicsRefusal,
|
||||
ctx_bytes,
|
||||
kv_dtype_factor,
|
||||
physics_check,
|
||||
)
|
||||
|
||||
GIB = 1 << 30
|
||||
KIB = 1024
|
||||
|
||||
|
||||
# ── synthetic profiles (per-layer tuples, per the verification plan) ──
|
||||
|
||||
|
||||
def dense(name="dense-32b", layers=64, per_token_f16=4096, weights_gib=20,
|
||||
native=128 * 1024) -> ModelProfile:
|
||||
return ModelProfile(
|
||||
name=name, weights_bytes=weights_gib * GIB, embd_table_bytes=0,
|
||||
n_ctx_train=native,
|
||||
layers=[(LayerKind.FULL, per_token_f16)] * layers)
|
||||
|
||||
|
||||
def hybrid(name="hybrid-30b", full_layers=16, recurrent_layers=48,
|
||||
per_token_f16=4096, weights_gib=22, native=1024 * 1024) -> ModelProfile:
|
||||
layers = ([(LayerKind.FULL, per_token_f16)] * full_layers
|
||||
+ [(LayerKind.RECURRENT, 0)] * recurrent_layers)
|
||||
return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
|
||||
embd_table_bytes=0, n_ctx_train=native, layers=layers)
|
||||
|
||||
|
||||
def moe(name="moe-30b", layers=48, per_token_f16=3072, weights_gib=17,
|
||||
native=256 * 1024) -> ModelProfile:
|
||||
return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
|
||||
embd_table_bytes=0, n_ctx_train=native,
|
||||
layers=[(LayerKind.FULL, per_token_f16)] * layers,
|
||||
moe=True)
|
||||
|
||||
|
||||
def card(vram_gib, ram_gib=64, uma=False) -> HardwareBudget:
|
||||
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
|
||||
total_device_bytes=int(vram_gib * GIB),
|
||||
ram_available_bytes=int(ram_gib * GIB), uma=uma)
|
||||
|
||||
|
||||
# ── estimator invariants ─────────────────────────────────────
|
||||
|
||||
|
||||
def test_full_attention_linear_in_window():
|
||||
p = dense()
|
||||
b32, b64, b128 = (ctx_bytes(p, w * KIB) for w in (32, 64, 128))
|
||||
assert abs(b64 / b32 - 2) < 0.01
|
||||
assert abs(b128 / b64 - 2) < 0.01
|
||||
|
||||
|
||||
def test_recurrent_state_constant_in_window():
|
||||
p = hybrid()
|
||||
full_share_32 = ctx_bytes(p, 32 * KIB)
|
||||
full_share_1m = ctx_bytes(p, 1024 * KIB)
|
||||
# Grows only through the 16 full-attn layers — the recurrent share is
|
||||
# identical, so the ratio tracks the full-attn ratio exactly.
|
||||
full_only = ModelProfile(name="x", weights_bytes=0, embd_table_bytes=0,
|
||||
n_ctx_train=p.n_ctx_train,
|
||||
layers=[(LayerKind.FULL, 4096)] * 16)
|
||||
expected_delta = ctx_bytes(full_only, 1024 * KIB) - ctx_bytes(full_only, 32 * KIB)
|
||||
assert abs((full_share_1m - full_share_32) - expected_delta) <= 1
|
||||
|
||||
|
||||
def test_swa_layers_capped_at_window():
|
||||
p = ModelProfile(
|
||||
name="swa", weights_bytes=0, embd_table_bytes=0, n_ctx_train=128 * KIB,
|
||||
layers=[(LayerKind.SWA, 4096)] * 5 + [(LayerKind.FULL, 4096)] * 1,
|
||||
swa_window=1024)
|
||||
small, big = ctx_bytes(p, 4 * KIB), ctx_bytes(p, 32 * KIB)
|
||||
# Full layer grew 8x; the 5 SWA layers stayed capped at 1024 — total
|
||||
# growth must land well under the all-full 8x (here ~4.1x).
|
||||
assert big / small < 0.6 * 8
|
||||
|
||||
|
||||
def test_q8_factor_is_exactly_34_over_64():
|
||||
assert kv_dtype_factor(True) == pytest.approx(34 / 64)
|
||||
assert kv_dtype_factor(False) == 1.0
|
||||
|
||||
|
||||
def test_non_fa_fallback_doubles_ctx_cost():
|
||||
p = dense()
|
||||
assert ctx_bytes(p, FLOOR, flash_attention=False) == pytest.approx(
|
||||
ctx_bytes(p, FLOOR, flash_attention=True) * 64 / 34, rel=0.001)
|
||||
|
||||
|
||||
def test_hybrid_vs_dense_100x_class_spread():
|
||||
"""The whole reason for the per-layer walk: equal-size models, ~100x
|
||||
per-token spread between classic dense and a mostly-recurrent hybrid."""
|
||||
d = dense(layers=64, per_token_f16=8192) # 256 KiB/tok class
|
||||
h = hybrid(full_layers=4, recurrent_layers=60, per_token_f16=8192)
|
||||
window = 256 * KIB
|
||||
dense_cost = ctx_bytes(d, window)
|
||||
hybrid_cost = ctx_bytes(h, window)
|
||||
assert dense_cost / hybrid_cost > 10
|
||||
|
||||
|
||||
# ── measured-constant spot checks (real models, tolerance bands) ──
|
||||
|
||||
|
||||
def test_measured_dense_4b_per_token():
|
||||
"""Qwen3-4B: 36 layers x 8 kv-heads x (128+128) x 2B = 144 KiB/tok f16."""
|
||||
p = ModelProfile(name="qwen3-4b", weights_bytes=0, embd_table_bytes=0,
|
||||
n_ctx_train=262144,
|
||||
layers=[(LayerKind.FULL, 8 * 256 * 2)] * 36)
|
||||
per_token_bytes = ctx_bytes(p, 32 * KIB, flash_attention=False) / (32 * KIB)
|
||||
assert per_token_bytes == pytest.approx(144 * KIB, rel=0.02)
|
||||
|
||||
|
||||
def test_measured_gdn_27b_per_token_q8():
|
||||
"""Qwen3.6-27B: 16 full-attn of 64; measured 34.0 KiB/tok @ q8 (B4).
|
||||
Per-layer f16 = 34 KiB * 64/34 / 16 = 4 KiB."""
|
||||
per_layer_f16 = 4 * KIB
|
||||
kv_only = ModelProfile(name="kv", weights_bytes=0, embd_table_bytes=0,
|
||||
n_ctx_train=262144,
|
||||
layers=[(LayerKind.FULL, per_layer_f16)] * 16)
|
||||
per_token_bytes = ctx_bytes(kv_only, 128 * KIB) / (128 * KIB)
|
||||
assert per_token_bytes == pytest.approx(34 * KIB, rel=0.02)
|
||||
|
||||
|
||||
def test_measured_nemotron_1m_within_band():
|
||||
"""1M @ q8 measured 3264 MiB KV (B3): ~3.19 KiB/token TOTAL across the
|
||||
16 full-attn layers -> per-layer f16 ~384 B. Estimator must land in the
|
||||
measured band, not the dense-formula 100x miss."""
|
||||
p = hybrid(full_layers=16, recurrent_layers=46, per_token_f16=384,
|
||||
native=1024 * KIB)
|
||||
total = ctx_bytes(p, 1024 * KIB)
|
||||
assert 2.5 * GIB < total < 4.0 * GIB
|
||||
|
||||
|
||||
# ── physics check ────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_physics_refusal_only_past_vram_plus_ram():
|
||||
p = dense(weights_gib=60)
|
||||
ok = physics_check(p, card(24, ram_gib=64), FLOOR)
|
||||
assert ok is None # 60 GiB weights fit in 24+64
|
||||
refused = physics_check(p, card(24, ram_gib=16), FLOOR)
|
||||
assert isinstance(refused, PhysicsRefusal)
|
||||
assert "smaller quant" in refused.message
|
||||
|
||||
|
||||
def test_physics_check_prices_at_floor_not_native():
|
||||
"""A 1M-native hybrid must not be refused for its native window —
|
||||
the check prices the floor only."""
|
||||
p = hybrid(weights_gib=22)
|
||||
assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
|
||||
|
||||
|
||||
# ── ladder + initial window ──────────────────────────────────
|
||||
|
||||
|
||||
def test_ladder_shape():
|
||||
rungs = ladder(262144)
|
||||
assert rungs[0] == FLOOR
|
||||
assert rungs[-1] == 262144
|
||||
assert all(a < b for a, b in zip(rungs, rungs[1:]))
|
||||
# geometric-ish: each step grows, none more than 2x
|
||||
assert all(b / a <= 2.0 for a, b in zip(rungs, rungs[1:]))
|
||||
|
||||
|
||||
def test_initial_window_never_below_floor_and_never_above_native():
|
||||
for profile in (dense(), hybrid(), moe(), dense(native=32 * KIB)):
|
||||
for vram in (8, 16, 24, 32):
|
||||
d = initial_window(profile, card(vram))
|
||||
if isinstance(d, WindowDecision):
|
||||
assert d.window >= min(FLOOR, profile.n_ctx_train)
|
||||
assert d.window <= profile.n_ctx_train
|
||||
|
||||
|
||||
def test_initial_window_monotone_in_vram():
|
||||
p = dense()
|
||||
windows = []
|
||||
for vram in (8, 12, 16, 24, 32, 48):
|
||||
d = initial_window(p, card(vram))
|
||||
assert isinstance(d, WindowDecision)
|
||||
windows.append(d.window)
|
||||
assert all(a <= b for a, b in zip(windows, windows[1:]))
|
||||
|
||||
|
||||
def test_flat_curve_reaches_native_where_dense_does_not():
|
||||
"""Design invariant: equal-size hybrid rides to native spill-free where
|
||||
the dense model cannot. Hybrid KV priced at the B3 class (~3.2 KiB/tok
|
||||
total: per-layer f16 384 B x 16 layers)."""
|
||||
h = hybrid(weights_gib=18, native=1024 * KIB, per_token_f16=384)
|
||||
d = dense(weights_gib=18, per_token_f16=8192, native=1024 * KIB)
|
||||
vram = card(24)
|
||||
dh = initial_window(h, vram)
|
||||
dd = initial_window(d, vram)
|
||||
assert isinstance(dh, WindowDecision) and isinstance(dd, WindowDecision)
|
||||
assert dh.window == 1024 * KIB and not dh.spilled
|
||||
assert dd.window < 1024 * KIB
|
||||
|
||||
|
||||
def test_dense_on_small_card_holds_floor_and_spills():
|
||||
"""The deliberate price of the guarantee (design table: dense 32B on
|
||||
24 GB starts at the floor with a few GiB spilled)."""
|
||||
d = initial_window(dense(weights_gib=20), card(12))
|
||||
assert isinstance(d, WindowDecision)
|
||||
assert d.window == FLOOR
|
||||
assert d.spilled
|
||||
|
||||
|
||||
def test_uma_budget_caps_the_window_through_physics():
|
||||
"""Unified memory needs no special context rule: the budget already
|
||||
encodes the constraint (usable = RAM minus headroom, ram_available=0),
|
||||
so the ladder stops where weights + KV genuinely stop fitting."""
|
||||
p = hybrid(weights_gib=8, native=1024 * KIB)
|
||||
unified = card(38.4, ram_gib=0, uma=True) # 48 GiB machine, 20% headroom
|
||||
d = initial_window(p, unified)
|
||||
assert isinstance(d, WindowDecision)
|
||||
assert not d.spilled, "UMA budget must produce a resident decision"
|
||||
need = 8 * GIB + ctx_bytes(p, d.window)
|
||||
assert need <= unified.usable_vram_bytes
|
||||
|
||||
|
||||
# ── growth ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _grow(profile, budget, **kw):
|
||||
defaults = dict(current_window=FLOOR, session_tokens=int(FLOOR * 0.9),
|
||||
measured_decode_tok_s=40.0, server_idle=True)
|
||||
defaults.update(kw)
|
||||
return growth_decision(profile, budget, **defaults)
|
||||
|
||||
|
||||
def test_growth_holds_below_occupancy():
|
||||
d = _grow(dense(), card(24), session_tokens=int(FLOOR * 0.5))
|
||||
assert d.action == "hold"
|
||||
|
||||
|
||||
def test_growth_requires_idle_server():
|
||||
d = _grow(dense(), card(24), server_idle=False)
|
||||
assert d.action == "hold"
|
||||
assert "idle" in d.reason
|
||||
|
||||
|
||||
def test_growth_steps_one_rung_and_is_monotone():
|
||||
p = dense(native=262144)
|
||||
d = _grow(p, card(32))
|
||||
assert d.action == "grow"
|
||||
assert d.next_window > FLOOR
|
||||
rungs = ladder(262144)
|
||||
assert d.next_window == rungs[rungs.index(FLOOR) + 1]
|
||||
|
||||
|
||||
def test_growth_stops_at_native():
|
||||
p = dense(native=128 * KIB)
|
||||
d = _grow(p, card(48), current_window=128 * KIB,
|
||||
session_tokens=int(128 * KIB * 0.9))
|
||||
assert d.action == "compress-default"
|
||||
|
||||
|
||||
def test_speed_floor_flips_default_to_compression():
|
||||
d = _grow(dense(), card(24), measured_decode_tok_s=SPEED_FLOOR_TOK_S - 2)
|
||||
assert d.action == "compress-default"
|
||||
assert "explicit per-session choice" in d.reason
|
||||
|
||||
|
||||
def test_growth_refits_against_live_budget():
|
||||
"""V3C: a rung that no longer fits (external pressure ate the memory)
|
||||
is not granted."""
|
||||
p = dense(weights_gib=20)
|
||||
starved = card(2, ram_gib=1)
|
||||
d = _grow(p, starved)
|
||||
assert d.action == "compress-default"
|
||||
assert "physics" in d.reason
|
||||
|
||||
|
||||
# ── spill placement + launch args ────────────────────────────
|
||||
|
||||
|
||||
def test_spill_overrides_prefer_expert_and_recurrent_ffn():
|
||||
assert "exps" in " ".join(spill_overrides(moe()))
|
||||
assert "ffn" in " ".join(spill_overrides(hybrid()))
|
||||
assert spill_overrides(dense()) == []
|
||||
|
||||
|
||||
def test_launch_args_contract():
|
||||
p = moe()
|
||||
spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
|
||||
resident = WindowDecision(window=131072, spill_bytes=0, kv_on_gpu=True)
|
||||
|
||||
a = launch_args(p, spilled, mtp_capable=True)
|
||||
assert a[:2] == ["-c", str(FLOOR)] # explicit window, always
|
||||
assert "q8_0" in a # q8 KV under flash attn
|
||||
assert "-ot" in a # spill placement
|
||||
assert "--spec-type" in a # MTP on spilled
|
||||
|
||||
# MTP is not gated on spill: resident decode measured +16% at depth 2.
|
||||
b = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2)
|
||||
assert "-ot" not in b, "placement is spill-only"
|
||||
assert "--spec-type" in b, "MTP must run on resident configs too"
|
||||
assert b[b.index("--spec-draft-n-max") + 1] == "2"
|
||||
assert "--backend-sampling" in b
|
||||
assert "--spec-draft-backend-sampling" in b
|
||||
|
||||
# Stacking MTP with the large microbatch is a FIT question, decided
|
||||
# by the caller (presets' posture ladder) and passed as mtp_prefill.
|
||||
# Default (no headroom proven): decode posture, small ubatch — the
|
||||
# stacked logits buffers once packed a 32 GiB card 3.9 GiB past a
|
||||
# fit that ignored them.
|
||||
assert "-ub" not in b, "default MTP posture stays at the small ubatch"
|
||||
|
||||
# Headroom proven: the stacked posture carries the large microbatch
|
||||
# (measured best on both axes where it fits: 93.3 vs 89.5 tok/s
|
||||
# decode on Qwen3.8 Q4). ub_logits_bytes must price the same choice.
|
||||
s = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2,
|
||||
mtp_prefill=True)
|
||||
assert "-ub" in s and s[s.index("-ub") + 1] == "2048"
|
||||
assert "--spec-type" in s
|
||||
v = 248320
|
||||
assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
|
||||
assert ub_logits_bytes(v, mtp_capable=True, mtp_prefill=True) == int(2048 * v * 4 * 1.5)
|
||||
|
||||
c = launch_args(p, resident, mtp_capable=False)
|
||||
assert "-ub" in c and c[c.index("-ub") + 1] == "2048" # prefill hint
|
||||
assert "--spec-type" not in c
|
||||
|
||||
d = launch_args(p, spilled, flash_attention=False, mtp_capable=False)
|
||||
assert "q8_0" not in d # f16 on non-FA fallback
|
||||
|
||||
|
||||
def test_launch_args_uma_never_pins_tensors():
|
||||
"""On unified memory, -ot pinning is off even for spilled decisions:
|
||||
"CPU" and "GPU" are the same silicon, and forcing FFN weights down
|
||||
the host compute path measures far slower than letting the
|
||||
allocator place everything. The discrete ~1.75x -ot win does not
|
||||
transfer. Everything else about the launch shape is identical to
|
||||
discrete."""
|
||||
p = moe()
|
||||
spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
|
||||
|
||||
u = launch_args(p, spilled, mtp_capable=False, uma=True)
|
||||
assert "-ot" not in u, "UMA must never pin tensors to the host path"
|
||||
assert u[:2] == ["-c", str(FLOOR)] # window contract unchanged
|
||||
assert "q8_0" in u # KV policy unchanged
|
||||
|
||||
# Same call on discrete keeps the pinning — the flag is the ONLY delta.
|
||||
disc = launch_args(p, spilled, mtp_capable=False, uma=False)
|
||||
assert "-ot" in disc
|
||||
assert [x for x in disc if x != "-ot" and not x.startswith("blk")] == \
|
||||
[x for x in u if x != "-ot" and not x.startswith("blk")]
|
||||
|
||||
|
||||
def test_ub_logits_bytes_prices_the_flag_choice():
|
||||
"""The logits-buffer price must match the microbatch launch_args
|
||||
chooses: 2048 x vocab x 4 for non-MTP, 512 x vocab x 4 x 2 for MTP
|
||||
(draft context doubles it). 248320-vocab receipts: ~1.9 GiB at
|
||||
ub2048, ~0.95 GiB under MTP."""
|
||||
v = 248320
|
||||
assert ub_logits_bytes(v, mtp_capable=False) == 2048 * v * 4
|
||||
assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
|
||||
assert ub_logits_bytes(0, mtp_capable=True) == 0 # unknown vocab: no charge
|
||||
|
||||
|
||||
def test_no_refusal_branch_past_physics():
|
||||
"""Design invariant: anything past the physics check is servable —
|
||||
initial_window never refuses on its own."""
|
||||
for vram in (4, 6, 8, 12):
|
||||
d = initial_window(dense(weights_gib=20), card(vram, ram_gib=64))
|
||||
assert isinstance(d, WindowDecision)
|
||||
|
||||
|
||||
def test_kv_scale_prices_mtp_draft_context():
|
||||
"""MTP profiles carry kv_scale > 1 (the draft context's KV share,
|
||||
calibrated from measured server RSS); ctx_bytes must scale with it so
|
||||
every consumer — launch fit, catalog rows, growth — prices what the
|
||||
server actually allocates. Four-point calibration held within
|
||||
+1.4 GiB conservative, never optimistic."""
|
||||
import dataclasses
|
||||
|
||||
p = moe()
|
||||
base = ctx_bytes(p, 131072)
|
||||
scaled = ctx_bytes(dataclasses.replace(p, kv_scale=1.2), 131072)
|
||||
assert scaled == int(base * 1.2)
|
||||
|
||||
# The safety direction: the estimate must never be BELOW measured.
|
||||
# (Calibration receipts: predicted-measured was +233..+1400 MiB.)
|
||||
assert scaled > base
|
||||
@@ -0,0 +1,39 @@
|
||||
"""The desktop subcommand's --local launch flag.
|
||||
|
||||
Local models ship on main behind this flag: `hermes desktop --local` (or
|
||||
`Hermes.exe --local` directly) shows the local-models GUI surfaces; without
|
||||
it the desktop hides them all, even when local models are configured. These
|
||||
tests pin the argparse contract; the pass-through to the Electron argv lives
|
||||
in cmd_gui's launch paths.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
|
||||
from hermes_cli.subcommands.gui import build_gui_parser
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(prog="hermes")
|
||||
subparsers = parser.add_subparsers(dest="command")
|
||||
build_gui_parser(subparsers, cmd_gui=lambda args: None)
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
def test_local_flag_parses():
|
||||
args = _parser().parse_args(["desktop", "--local"])
|
||||
|
||||
assert args.local is True
|
||||
|
||||
|
||||
def test_local_flag_defaults_off():
|
||||
args = _parser().parse_args(["desktop"])
|
||||
|
||||
assert args.local is False
|
||||
|
||||
|
||||
def test_local_flag_composes_with_build_flags():
|
||||
args = _parser().parse_args(["desktop", "--local", "--force-build"])
|
||||
|
||||
assert args.local is True
|
||||
assert args.force_build is True
|
||||
@@ -0,0 +1,186 @@
|
||||
"""The HF browser: search the firehose, price it roughly, and let any
|
||||
GGUF become a normal staged model.
|
||||
|
||||
Parsing contracts run against canned HF API shapes (no network); route
|
||||
contracts run against the real FastAPI app with the HF client stubbed."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
from hermes_cli.local_runtime.hf_browse import (
|
||||
HFFileGroup,
|
||||
HFModelHit,
|
||||
repo_files,
|
||||
rough_fit,
|
||||
search_models,
|
||||
)
|
||||
|
||||
GIB = 1 << 30
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
||||
(tmp_path / ".hermes").mkdir()
|
||||
from hermes_cli import web_server
|
||||
|
||||
test_client = TestClient(web_server.app)
|
||||
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
|
||||
return test_client
|
||||
|
||||
|
||||
def _budget(vram_gib, ram_gib=64):
|
||||
return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
|
||||
total_device_bytes=int(vram_gib * GIB),
|
||||
ram_available_bytes=int(ram_gib * GIB))
|
||||
|
||||
|
||||
def test_search_parses_hf_hits(monkeypatch):
|
||||
canned = [
|
||||
{"id": "unsloth/Qwen3.8-27B-GGUF", "downloads": 872724, "likes": 47,
|
||||
"lastModified": "2026-08-18", "gated": False},
|
||||
{"id": "bartowski/whatever-GGUF", "downloads": 5, "likes": 0,
|
||||
"lastModified": "2026-01-01", "gated": "auto"},
|
||||
]
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
|
||||
lambda url: canned)
|
||||
hits = search_models("qwen")
|
||||
assert hits[0].repo == "unsloth/Qwen3.8-27B-GGUF"
|
||||
assert hits[0].downloads == 872724
|
||||
assert hits[1].gated is True # HF 'auto'-gated counts as gated
|
||||
|
||||
|
||||
def test_repo_files_groups_splits_and_excludes_companions(monkeypatch):
|
||||
canned = [
|
||||
{"path": "Qwen3.8-27B-Q4_K_M.gguf", "size": 17 * GIB},
|
||||
{"path": "mmproj-BF16.gguf", "size": 1 * GIB},
|
||||
{"path": "UD-Q8/model-00001-of-00002.gguf", "size": 30 * GIB},
|
||||
{"path": "UD-Q8/model-00002-of-00002.gguf", "size": 12 * GIB},
|
||||
{"path": "README.md", "size": 1000},
|
||||
{"path": "dspark-draft-Q8_0.gguf", "size": 9 * GIB},
|
||||
]
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
|
||||
lambda url: canned)
|
||||
groups = repo_files("any/repo")
|
||||
labels = {g.label: g for g in groups}
|
||||
assert "Q4_K_M" in labels and labels["Q4_K_M"].total_bytes == 17 * GIB
|
||||
# Split parts collapse into one group, ordered, summed.
|
||||
split = next(g for g in groups if len(g.paths) == 2)
|
||||
assert split.total_bytes == 42 * GIB
|
||||
assert split.paths[0].endswith("00001-of-00002.gguf")
|
||||
# Companions (mmproj, draft) are not standalone models.
|
||||
assert not any("mmproj" in p or "dspark" in p
|
||||
for g in groups for p in g.paths)
|
||||
# Largest first.
|
||||
assert groups[0].total_bytes >= groups[-1].total_bytes
|
||||
|
||||
|
||||
def test_rough_fit_bands():
|
||||
b = _budget(29.6, ram_gib=64)
|
||||
assert rough_fit(20 * GIB, b) == "fits-gpu" # + fill-ins under 29.6
|
||||
assert rough_fit(28 * GIB, b) == "needs-ram" # weights spill
|
||||
assert rough_fit(120 * GIB, b) == "too-big"
|
||||
|
||||
|
||||
def test_search_route_requires_query_and_maps_errors(client, monkeypatch):
|
||||
r = client.get("/api/local-models/search", params={"q": " "})
|
||||
assert r.status_code == 200 and r.json() == {"hits": []}
|
||||
|
||||
def boom(q, limit):
|
||||
raise RuntimeError("HF down")
|
||||
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hf_browse.search_models", boom)
|
||||
r = client.get("/api/local-models/search", params={"q": "qwen"})
|
||||
assert r.status_code == 502
|
||||
|
||||
|
||||
def test_browsed_download_stages_and_bounces(client, tmp_path, monkeypatch):
|
||||
"""A browsed download must land in the machine-scoped models dir and
|
||||
bounce the router — the seam that makes it a NORMAL model."""
|
||||
body = b"GGUF" + b"\x00" * 60
|
||||
|
||||
class FakeResponse:
|
||||
headers = {"Content-Length": str(len(body))}
|
||||
|
||||
def __init__(self):
|
||||
self._data = body
|
||||
|
||||
def read(self, n=-1):
|
||||
out, self._data = self._data, b""
|
||||
return out
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr("urllib.request.urlopen",
|
||||
lambda *a, **k: FakeResponse())
|
||||
bounced = {}
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
|
||||
lambda: bounced.setdefault("yes", True))
|
||||
|
||||
r = client.post("/api/local-models/download-browsed",
|
||||
json={"repo": "someone/Some-GGUF",
|
||||
"paths": ["Some-Model-Q4_K_M.gguf"]})
|
||||
assert r.status_code == 200
|
||||
job_id = r.json()["job_id"]
|
||||
|
||||
import time as _time
|
||||
|
||||
deadline = _time.time() + 10
|
||||
status = None
|
||||
while _time.time() < deadline:
|
||||
status = client.get(f"/api/local-models/jobs/{job_id}").json()
|
||||
if status["status"] in ("done", "error"):
|
||||
break
|
||||
_time.sleep(0.05)
|
||||
assert status["status"] == "done", status.get("error")
|
||||
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
|
||||
assert (models_dir() / "Some-Model-Q4_K_M.gguf").exists()
|
||||
assert bounced.get("yes") is True
|
||||
|
||||
|
||||
def test_browsed_download_rejects_non_gguf(client):
|
||||
r = client.post("/api/local-models/download-browsed",
|
||||
json={"repo": "a/b", "paths": ["model.safetensors"]})
|
||||
assert r.status_code == 422
|
||||
|
||||
|
||||
def test_sideload_links_and_bounces(client, tmp_path, monkeypatch):
|
||||
src = tmp_path / "My-Local-Model-Q5_K_M.gguf"
|
||||
src.write_bytes(b"GGUF" + b"\x00" * 32)
|
||||
bounced = {}
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
|
||||
lambda: bounced.setdefault("yes", True))
|
||||
|
||||
r = client.post("/api/local-models/sideload", json={"path": str(src)})
|
||||
assert r.status_code == 200
|
||||
assert r.json()["model_id"] == "My-Local-Model-Q5_K_M"
|
||||
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
|
||||
dest = models_dir() / src.name
|
||||
assert dest.exists()
|
||||
assert bounced.get("yes") is True
|
||||
# The original must be untouched.
|
||||
assert src.exists()
|
||||
|
||||
# Idempotent: sideloading again short-circuits.
|
||||
r = client.post("/api/local-models/sideload", json={"path": str(src)})
|
||||
assert r.json().get("already_present") is True
|
||||
|
||||
|
||||
def test_sideload_rejects_non_gguf(client, tmp_path):
|
||||
src = tmp_path / "model.bin"
|
||||
src.write_bytes(b"nope")
|
||||
r = client.post("/api/local-models/sideload", json={"path": str(src)})
|
||||
assert r.status_code == 422
|
||||
@@ -0,0 +1,260 @@
|
||||
"""Model-load progress: SSE events -> composite percent -> wait notices.
|
||||
|
||||
The 40-second problem: a cold local model streams 16-21 GB of weights
|
||||
before the first token, and the chat rendered that as the generic
|
||||
"provider may be slow or overloaded" stall warning. llama-server's child
|
||||
emits real per-tensor progress which the router relays over /models/sse
|
||||
ONLY — these tests pin the consumer that turns that stream into the
|
||||
status route's `loading` field and the chat's load notice."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import time
|
||||
|
||||
import hermes_cli.local_runtime.load_progress as lp
|
||||
|
||||
|
||||
def setup_function(_fn):
|
||||
with lp._lock:
|
||||
lp._snapshot.clear()
|
||||
|
||||
|
||||
# ── composite percent ────────────────────────────────────────
|
||||
|
||||
|
||||
def test_composite_percent_text_stage_dominates():
|
||||
stages = ["text_model", "spec_model", "mmproj_model"]
|
||||
# Text model owns [0, 85): halfway through it reads ~42%.
|
||||
assert lp._composite_percent(stages, "text_model", 0.5) == 42 # 0.5*85
|
||||
# Extras start where text ends and never regress below it.
|
||||
assert lp._composite_percent(stages, "spec_model", 0.0) == 85
|
||||
assert lp._composite_percent(stages, "mmproj_model", 1.0) == 100
|
||||
|
||||
|
||||
def test_composite_percent_monotone_across_stage_walk():
|
||||
"""Walking the stages in llama-server's real order never moves the
|
||||
bar backwards — the property that makes the bar trustworthy."""
|
||||
stages = ["text_model", "spec_model", "mmproj_model"]
|
||||
walk = [("text_model", v / 10) for v in range(11)] + \
|
||||
[("spec_model", v / 10) for v in range(11)] + \
|
||||
[("mmproj_model", v / 10) for v in range(11)]
|
||||
seen = [lp._composite_percent(stages, s, v) for s, v in walk]
|
||||
assert seen == sorted(seen)
|
||||
assert seen[0] == 0 and seen[-1] == 100
|
||||
|
||||
|
||||
def test_composite_percent_single_stage_is_plain():
|
||||
assert lp._composite_percent(["text_model"], "text_model", 0.4) == 40
|
||||
|
||||
|
||||
# ── event application ────────────────────────────────────────
|
||||
|
||||
|
||||
def _loading_event(value: float, current: str = "text_model") -> dict:
|
||||
return {"status": "loading",
|
||||
"progress": {"stages": ["text_model", "mmproj_model"],
|
||||
"current": current, "value": value}}
|
||||
|
||||
|
||||
def test_loading_events_build_snapshot_and_terminal_clears():
|
||||
lp._apply_event("m1", "status_change", _loading_event(0.5))
|
||||
snap = lp.get_loading_progress()
|
||||
assert "m1" in snap
|
||||
assert snap["m1"]["percent"] == 42 # 0.5 * 85 within text stage
|
||||
assert snap["m1"]["stage"] == "text_model"
|
||||
|
||||
lp._apply_event("m1", "status_change", {"status": "loaded", "info": {}})
|
||||
assert lp.get_loading_progress() == {}
|
||||
|
||||
|
||||
def test_unload_and_failure_clear_too():
|
||||
lp._apply_event("m1", "status_change", _loading_event(0.2))
|
||||
lp._apply_event("m1", "status_change", {"status": "unloaded", "exit_code": 1})
|
||||
assert lp.get_loading_progress() == {}
|
||||
|
||||
lp._apply_event("m2", "status_change", _loading_event(0.9))
|
||||
lp._apply_event("m2", "model_remove", {})
|
||||
assert lp.get_loading_progress() == {}
|
||||
|
||||
|
||||
def test_progressless_loading_event_keeps_entry_alive():
|
||||
"""The router's first model_status event says just {status: loading} —
|
||||
it must register the load (indeterminate) without inventing a percent."""
|
||||
lp._apply_event("m1", "model_status", {"status": "loading"})
|
||||
snap = lp.get_loading_progress()
|
||||
assert snap["m1"]["percent"] == 0
|
||||
|
||||
|
||||
def test_stale_entries_expire():
|
||||
lp._apply_event("m1", "status_change", _loading_event(0.5))
|
||||
with lp._lock:
|
||||
lp._snapshot["m1"]["ts"] -= lp._STALE_ENTRY_TTL_S + 1
|
||||
assert lp.get_loading_progress() == {}
|
||||
|
||||
|
||||
# ── chat wait-notice ─────────────────────────────────────────
|
||||
|
||||
|
||||
def test_load_notice_for_managed_model(tmp_path, monkeypatch):
|
||||
from agent.chat_completion_helpers import _managed_local_load_notice
|
||||
|
||||
state = tmp_path / "server.json"
|
||||
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
|
||||
"api_key": "k"}), encoding="utf-8")
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
|
||||
lambda: state)
|
||||
lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
|
||||
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
|
||||
|
||||
class _Agent:
|
||||
base_url = "http://127.0.0.1:18434/v1"
|
||||
|
||||
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
|
||||
assert notice is not None
|
||||
assert notice.startswith("⏳ loading Qwen-Test into memory — 42%")
|
||||
|
||||
# Different endpoint (user's own server): never claim its loads.
|
||||
class _Other:
|
||||
base_url = "http://127.0.0.1:9999/v1"
|
||||
|
||||
assert _managed_local_load_notice(_Other(), {"model": "Qwen-Test"}) is None
|
||||
# Managed endpoint but a model that isn't loading: no notice.
|
||||
assert _managed_local_load_notice(_Agent(), {"model": "Elsewhere"}) is None
|
||||
|
||||
|
||||
def test_load_notice_matches_desktop_wait_filter():
|
||||
"""The notices must pass the desktop's providerWaitText regex and parse
|
||||
under parseModelLoadWait's shapes — pinned here as plain string
|
||||
contracts so the two sides can't drift silently."""
|
||||
import re
|
||||
|
||||
accept = r"^(?:⏳|⚠|↻|⚙)\s*(?:waiting on|loading|processing prompt|no (?:output|response)|model returned)"
|
||||
|
||||
load = "⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 43% (responses start once the model is loaded)"
|
||||
assert re.match(accept, load)
|
||||
m = re.match(r"^⏳\s*loading\s+(.+?)\s+into memory\s+—\s+(\d{1,3})%", load)
|
||||
assert m and m.group(1) == "Qwen3.6-35B-A3B-UD-Q4_K_M" and m.group(2) == "43"
|
||||
|
||||
prefill = "⚙ processing prompt — 31%"
|
||||
assert re.match(accept, prefill)
|
||||
p = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", prefill)
|
||||
assert p and p.group(1) == "31"
|
||||
|
||||
bare = "⚙ processing prompt"
|
||||
assert re.match(accept, bare)
|
||||
b = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", bare)
|
||||
assert b and b.group(1) is None
|
||||
|
||||
|
||||
# ── prefill progress ─────────────────────────────────────────
|
||||
|
||||
|
||||
def test_prefill_notice_for_managed_model(tmp_path, monkeypatch):
|
||||
from agent.chat_completion_helpers import _managed_local_load_notice
|
||||
|
||||
state = tmp_path / "server.json"
|
||||
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
|
||||
"api_key": "k"}), encoding="utf-8")
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
|
||||
lambda: state)
|
||||
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
|
||||
# No load in flight; a prefill counter is live.
|
||||
monkeypatch.setattr(lp, "get_prefill_progress",
|
||||
lambda model: {"processed": 12288})
|
||||
import agent.chat_completion_helpers as cch
|
||||
|
||||
monkeypatch.setattr(cch, "estimate_request_context_tokens",
|
||||
lambda kw: 39551)
|
||||
|
||||
class _Agent:
|
||||
base_url = "http://127.0.0.1:18434/v1"
|
||||
|
||||
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
|
||||
assert notice == "⚙ processing prompt — 31%"
|
||||
|
||||
# Counter past the estimate (estimator undercounted): no honest
|
||||
# denominator, so no percent — never >100%.
|
||||
monkeypatch.setattr(cch, "estimate_request_context_tokens", lambda kw: 100)
|
||||
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
|
||||
assert notice == "⚙ processing prompt"
|
||||
|
||||
|
||||
def test_load_notice_outranks_prefill(tmp_path, monkeypatch):
|
||||
"""While a load entry exists the load notice wins — prefill can't start
|
||||
before the model is resident, so a simultaneous claim means the load
|
||||
snapshot is authoritative."""
|
||||
from agent.chat_completion_helpers import _managed_local_load_notice
|
||||
|
||||
state = tmp_path / "server.json"
|
||||
state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
|
||||
"api_key": "k"}), encoding="utf-8")
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
|
||||
lambda: state)
|
||||
monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
|
||||
lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
|
||||
monkeypatch.setattr(lp, "get_prefill_progress",
|
||||
lambda model: {"processed": 999})
|
||||
|
||||
class _Agent:
|
||||
base_url = "http://127.0.0.1:18434/v1"
|
||||
|
||||
notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
|
||||
assert notice is not None and notice.startswith("⏳ loading")
|
||||
|
||||
|
||||
def test_prefill_progress_reads_busiest_processing_slot(monkeypatch):
|
||||
monkeypatch.setattr(lp, "_endpoint", lambda: ("http://127.0.0.1:1", "k"))
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, payload):
|
||||
self._payload = payload
|
||||
|
||||
def read(self):
|
||||
return json.dumps(self._payload).encode()
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
slots = [
|
||||
{"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500},
|
||||
{"id": 1, "is_processing": True, "n_prompt_tokens_processed": 42},
|
||||
{"id": 2, "is_processing": True, "n_prompt_tokens_processed": 32768},
|
||||
]
|
||||
monkeypatch.setattr(lp.urllib.request, "urlopen",
|
||||
lambda req, timeout=0: _Resp(slots))
|
||||
assert lp.get_prefill_progress("m") == {"processed": 32768}
|
||||
|
||||
# Nothing processing -> None (idle slots' counters are leftovers).
|
||||
idle = [{"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500}]
|
||||
monkeypatch.setattr(lp.urllib.request, "urlopen",
|
||||
lambda req, timeout=0: _Resp(idle))
|
||||
assert lp.get_prefill_progress("m") is None
|
||||
|
||||
# Unreachable server -> None, never an exception.
|
||||
def _boom(req, timeout=0):
|
||||
raise OSError("refused")
|
||||
|
||||
monkeypatch.setattr(lp.urllib.request, "urlopen", _boom)
|
||||
assert lp.get_prefill_progress("m") is None
|
||||
|
||||
|
||||
def test_endpoint_respects_ownership_guard(monkeypatch):
|
||||
"""The watcher's endpoint MUST come from the ownership-guarded reader.
|
||||
Regression: a raw state-file read attached the SSE watcher to a
|
||||
foreign install's server on the shared stable port (health answers
|
||||
for anyone; only the dead-pid check proves ownership)."""
|
||||
import hermes_cli.local_runtime.load_progress as lp
|
||||
|
||||
# Guard says "not ours": no endpoint, regardless of state on disk.
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
|
||||
lambda: None)
|
||||
assert lp._endpoint() is None
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.endpoint._state_endpoint",
|
||||
lambda: {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"})
|
||||
assert lp._endpoint() == ("http://127.0.0.1:18434", "k")
|
||||
@@ -0,0 +1,213 @@
|
||||
"""Abandoned-request lifecycle: work sent to the managed local server must
|
||||
die when its caller goes away, and teardown must never orphan VRAM.
|
||||
|
||||
The incident this guards: auxiliary calls (title generation + retries)
|
||||
queued at the router behind a cold model load, their clients timed out
|
||||
and hung up, and the router then dispatched them anyway. Non-streamed
|
||||
responses write the socket only after the FULL generation, so nothing
|
||||
noticed the dead clients — two uncapped decodes ran at full GPU for the
|
||||
better part of an hour with nobody listening.
|
||||
|
||||
Three contracts, one per failure link:
|
||||
1. Auxiliary requests to the managed local endpoint are always streamed
|
||||
(a dead client then cancels decode at the first chunk write).
|
||||
2. Explicit caller max_tokens caps reach the managed local endpoint
|
||||
(title generation's 64-token cap must not be silently dropped).
|
||||
3. Supervisor teardown terminates the whole process tree, and a router
|
||||
respawn reaps orphaned model children first (each holds GiB of VRAM).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import types
|
||||
|
||||
import pytest
|
||||
|
||||
import agent.auxiliary_client as aux
|
||||
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
||||
|
||||
|
||||
MANAGED_URL = "http://127.0.0.1:18434/v1"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def managed_state(tmp_path, monkeypatch):
|
||||
"""A supervisor state file declaring the managed endpoint, cache reset."""
|
||||
state = tmp_path / "server.json"
|
||||
state.write_text(json.dumps({"base_url": MANAGED_URL, "api_key": "k",
|
||||
"pid": 4242}), encoding="utf-8")
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
|
||||
lambda: state)
|
||||
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
|
||||
return state
|
||||
|
||||
|
||||
# ── 1. managed endpoint is always streamed ───────────────────
|
||||
|
||||
|
||||
def test_managed_endpoint_requires_stream(managed_state):
|
||||
assert aux._provider_requires_stream("custom", MANAGED_URL) is True
|
||||
|
||||
|
||||
def test_managed_detection_matches_netloc_not_substring(managed_state):
|
||||
# Same host, different port: a user's own external server — untouched.
|
||||
assert aux._provider_requires_stream("custom", "http://127.0.0.1:8080/v1") is False
|
||||
|
||||
|
||||
def test_no_state_file_means_no_managed_endpoint(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
|
||||
lambda: tmp_path / "absent.json")
|
||||
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
|
||||
assert aux._is_managed_local_endpoint(MANAGED_URL) is False
|
||||
|
||||
|
||||
def test_remote_providers_unaffected(managed_state):
|
||||
assert aux._provider_requires_stream("nous",
|
||||
"https://inference-api.nousresearch.com/v1/") is False
|
||||
|
||||
|
||||
# ── 2. explicit caps reach the managed endpoint ──────────────
|
||||
|
||||
|
||||
def test_explicit_max_tokens_forwarded_to_managed_local(managed_state, monkeypatch):
|
||||
monkeypatch.setattr(aux, "_current_custom_base_url", lambda: MANAGED_URL)
|
||||
kwargs = aux._build_call_kwargs(
|
||||
"custom", "Qwen-Local", [{"role": "user", "content": "hi"}],
|
||||
max_tokens=64, timeout=30.0, task="title_generation")
|
||||
assert kwargs.get("max_tokens") == 64 or kwargs.get("max_completion_tokens") == 64, (
|
||||
"explicit caller cap dropped on the managed local endpoint — an "
|
||||
"EOS-less generation then runs to the full context window")
|
||||
|
||||
|
||||
def test_no_default_cap_policy_unchanged_for_remote(monkeypatch):
|
||||
# A generic remote provider still drops the cap (the forwarding gate
|
||||
# is an allow-list). openrouter no longer qualifies as the example
|
||||
# here: main forwards its caps deliberately (#41035, 402 affordability).
|
||||
monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
|
||||
kwargs = aux._build_call_kwargs(
|
||||
"openai", "some/model", [{"role": "user", "content": "hi"}],
|
||||
max_tokens=64, timeout=30.0)
|
||||
assert "max_tokens" not in kwargs and "max_completion_tokens" not in kwargs
|
||||
|
||||
|
||||
# ── 3. teardown kills the tree; respawn reaps orphans ────────
|
||||
|
||||
|
||||
class _FakeChild:
|
||||
def __init__(self, pid):
|
||||
self.pid = pid
|
||||
self.terminated = False
|
||||
self.killed = False
|
||||
|
||||
def terminate(self):
|
||||
self.terminated = True
|
||||
|
||||
def is_running(self):
|
||||
return not self.terminated and not self.killed
|
||||
|
||||
def kill(self):
|
||||
self.killed = True
|
||||
|
||||
|
||||
def test_terminate_tree_terminates_children_too(monkeypatch):
|
||||
children = [_FakeChild(101), _FakeChild(102)]
|
||||
|
||||
class _FakeParentProc:
|
||||
def __init__(self, pid):
|
||||
self.pid = pid
|
||||
|
||||
def children(self, recursive=False):
|
||||
assert recursive is True
|
||||
return children
|
||||
|
||||
fake_psutil = types.SimpleNamespace(Process=_FakeParentProc)
|
||||
monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
|
||||
|
||||
class _FakeRouter:
|
||||
pid = 4242
|
||||
terminated = False
|
||||
|
||||
def terminate(self):
|
||||
_FakeRouter.terminated = True
|
||||
|
||||
def wait(self, timeout=None):
|
||||
return 0
|
||||
|
||||
def poll(self):
|
||||
return None
|
||||
|
||||
LlamaServerSupervisor._terminate_tree(_FakeRouter())
|
||||
assert _FakeRouter.terminated
|
||||
assert all(c.terminated for c in children), (
|
||||
"router children orphaned on stop — each holds GiB of VRAM")
|
||||
|
||||
|
||||
def test_terminate_tree_survives_missing_psutil(monkeypatch):
|
||||
import builtins
|
||||
|
||||
real_import = builtins.__import__
|
||||
|
||||
def _no_psutil(name, *a, **k):
|
||||
if name == "psutil":
|
||||
raise ImportError("nope")
|
||||
return real_import(name, *a, **k)
|
||||
|
||||
monkeypatch.setattr(builtins, "__import__", _no_psutil)
|
||||
|
||||
class _FakeRouter:
|
||||
pid = 4242
|
||||
terminated = False
|
||||
|
||||
def terminate(self):
|
||||
_FakeRouter.terminated = True
|
||||
|
||||
def wait(self, timeout=None):
|
||||
return 0
|
||||
|
||||
LlamaServerSupervisor._terminate_tree(_FakeRouter())
|
||||
assert _FakeRouter.terminated # router still stopped without psutil
|
||||
|
||||
|
||||
def test_reap_orphans_kills_only_our_parentless_binaries(tmp_path, monkeypatch):
|
||||
exe = tmp_path / "llama-server.exe"
|
||||
exe.write_text("")
|
||||
|
||||
orphan = _FakeChild(300)
|
||||
adopted = _FakeChild(301) # parent alive -> not an orphan
|
||||
foreign = _FakeChild(302) # different binary -> never touched
|
||||
|
||||
def _info(pid, exe_path, ppid):
|
||||
p = _FakeChild(pid)
|
||||
p.info = {"exe": exe_path, "ppid": ppid}
|
||||
return p
|
||||
|
||||
procs = [
|
||||
_info(300, str(exe), 9999), # dead parent -> reap
|
||||
_info(301, str(exe), 1), # live parent -> keep
|
||||
_info(302, str(tmp_path / "other.exe"), 9999), # foreign -> keep
|
||||
]
|
||||
reaped = []
|
||||
for p in procs:
|
||||
p.kill = lambda p=p: reaped.append(p.info and p.pid)
|
||||
|
||||
class _NoSuch(Exception):
|
||||
pass
|
||||
|
||||
fake_psutil = types.SimpleNamespace(
|
||||
process_iter=lambda attrs: procs,
|
||||
pid_exists=lambda pid: pid == 1,
|
||||
NoSuchProcess=_NoSuch,
|
||||
AccessDenied=_NoSuch,
|
||||
)
|
||||
monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.supervisor.server_binary",
|
||||
lambda install_dir: exe)
|
||||
|
||||
sup = LlamaServerSupervisor.__new__(LlamaServerSupervisor)
|
||||
sup.install_dir = tmp_path
|
||||
sup.proc = None
|
||||
sup._reap_orphaned_children()
|
||||
|
||||
assert reaped == [300], f"reaped {reaped}; wanted only the orphan (300)"
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Context-length resolution for the managed llama.cpp router.
|
||||
|
||||
The incident: the statusbar showed 131K for a local model the server had
|
||||
granted 262144 tokens. The router reports ``meta: null`` on /v1/models
|
||||
for a model that is not currently LOADED (models autoload on first chat,
|
||||
so at session start the model is routinely unloaded), and /v1/models/{id}
|
||||
404s — every metadata probe missed, resolution fell through to the
|
||||
name-pattern defaults, and the "qwen" family catch-all (131072) shipped
|
||||
as the compressor's budget and the statusbar's denominator.
|
||||
|
||||
Contract: for a llama.cpp server, /props default_generation_settings.n_ctx
|
||||
(the preset-backed RUNTIME window, served even for unloaded models) is
|
||||
the authority, probed before the /v1/models fallbacks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import http.server
|
||||
import json
|
||||
import threading
|
||||
|
||||
import pytest
|
||||
|
||||
import agent.model_metadata as mm
|
||||
|
||||
|
||||
GRANTED = 262144
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def router():
|
||||
"""Stub of the llama-server router with the model UNLOADED:
|
||||
/v1/models carries meta=null; /props answers from the preset."""
|
||||
|
||||
class _Router(http.server.BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
if self.path.startswith("/props"):
|
||||
body = {"default_generation_settings": {"n_ctx": GRANTED}}
|
||||
elif self.path == "/v1/models":
|
||||
body = {"data": [{
|
||||
"id": "Qwen-Test-UD-Q4_K_M",
|
||||
"owned_by": "llamacpp",
|
||||
"meta": None,
|
||||
"status": {"value": "unloaded"},
|
||||
}]}
|
||||
else: # /v1/models/{id} -> 404, as the real router answers
|
||||
self.send_response(404)
|
||||
self.send_header("Content-Length", "0")
|
||||
self.end_headers()
|
||||
return
|
||||
raw = json.dumps(body).encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(raw)))
|
||||
self.end_headers()
|
||||
self.wfile.write(raw)
|
||||
|
||||
def log_message(self, *a):
|
||||
pass
|
||||
|
||||
server = http.server.HTTPServer(("127.0.0.1", 0), _Router)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
yield f"http://127.0.0.1:{server.server_address[1]}/v1"
|
||||
server.shutdown()
|
||||
|
||||
|
||||
def test_unloaded_llamacpp_model_resolves_granted_window(router, monkeypatch):
|
||||
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
|
||||
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
|
||||
ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
|
||||
assert ctx == GRANTED, (
|
||||
f"resolved {ctx}; an unloaded model must resolve the preset window "
|
||||
"from /props, not fall through to name-pattern catch-alls")
|
||||
|
||||
|
||||
def test_props_beats_meta_when_model_loaded(router, monkeypatch):
|
||||
"""/props is probed first even when /v1/models would answer: n_ctx from
|
||||
/props is the same runtime value, and probing it first keeps loaded and
|
||||
unloaded models on one code path."""
|
||||
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
|
||||
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
|
||||
ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
|
||||
assert ctx == GRANTED
|
||||
|
||||
|
||||
def test_non_llamacpp_servers_skip_props(monkeypatch):
|
||||
"""Ollama/LM Studio/vLLM keep their existing probe order — /props is
|
||||
llama.cpp-shaped and must not be consulted for other server types."""
|
||||
calls = []
|
||||
|
||||
class _FakeResp:
|
||||
status_code = 404
|
||||
|
||||
def json(self):
|
||||
return {}
|
||||
|
||||
class _FakeClient:
|
||||
def __init__(self, *a, **k):
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
def get(self, url):
|
||||
calls.append(url)
|
||||
return _FakeResp()
|
||||
|
||||
def post(self, url, **k):
|
||||
calls.append(url)
|
||||
return _FakeResp()
|
||||
|
||||
import httpx
|
||||
monkeypatch.setattr(httpx, "Client", _FakeClient)
|
||||
monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "vllm")
|
||||
monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
|
||||
mm._query_local_context_length_uncached("m", "http://127.0.0.1:9999/v1")
|
||||
assert not any("/props" in u for u in calls)
|
||||
@@ -0,0 +1,280 @@
|
||||
"""In-session growth contracts (growth.py + the presets override seam).
|
||||
|
||||
The live half of the window ladder: grow before compress, overrides
|
||||
persist across boots, physics re-checked every boot, growth state dies
|
||||
with the model."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def hermes_home(tmp_path, monkeypatch):
|
||||
home = tmp_path / ".hermes"
|
||||
home.mkdir()
|
||||
monkeypatch.setenv("HERMES_HOME", str(home))
|
||||
return home
|
||||
|
||||
|
||||
def test_overrides_roundtrip_and_clear(hermes_home):
|
||||
from hermes_cli.local_runtime.growth import (
|
||||
clear_window_override,
|
||||
load_window_overrides,
|
||||
save_window_override,
|
||||
)
|
||||
|
||||
assert load_window_overrides() == {}
|
||||
save_window_override("model-a", 98304)
|
||||
save_window_override("model-b", 262144)
|
||||
assert load_window_overrides() == {"model-a": 98304, "model-b": 262144}
|
||||
clear_window_override("model-a")
|
||||
assert load_window_overrides() == {"model-b": 262144}
|
||||
# Clearing a missing key is a no-op, not an error.
|
||||
clear_window_override("never-existed")
|
||||
|
||||
|
||||
def test_corrupt_overrides_read_as_empty(hermes_home):
|
||||
from hermes_cli.local_runtime.growth import (
|
||||
load_window_overrides,
|
||||
window_overrides_path,
|
||||
)
|
||||
|
||||
path = window_overrides_path()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text("{not json", encoding="utf-8")
|
||||
assert load_window_overrides() == {}
|
||||
|
||||
|
||||
def test_growth_declines_foreign_endpoints(hermes_home):
|
||||
"""Only the server THIS process supervises grows — a detected external
|
||||
server or another process's endpoint returns None untouched."""
|
||||
from hermes_cli.local_runtime.growth import maybe_grow_window
|
||||
|
||||
grown = maybe_grow_window(
|
||||
"some-model", base_url="http://127.0.0.1:9999/v1",
|
||||
session_tokens=100_000, current_window=65536)
|
||||
assert grown is None
|
||||
|
||||
|
||||
def test_occupancy_confirmed_skips_gate_one():
|
||||
"""The agent's compression gate IS the occupancy signal: when it fired,
|
||||
growth must not re-derive its own edge and hold. Decision-table check
|
||||
with a synthetic profile."""
|
||||
from hermes_cli.local_runtime.context_policy import growth_decision
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget,
|
||||
LayerKind,
|
||||
ModelProfile,
|
||||
)
|
||||
|
||||
gib = 1 << 30
|
||||
profile = ModelProfile(
|
||||
name="m", weights_bytes=2 * gib, embd_table_bytes=0,
|
||||
n_ctx_train=262144,
|
||||
layers=[(LayerKind.FULL, 4096)] * 16 + [(LayerKind.RECURRENT, 0)] * 48)
|
||||
budget = HardwareBudget(usable_vram_bytes=26 * gib,
|
||||
total_device_bytes=32 * gib,
|
||||
ram_available_bytes=64 * gib)
|
||||
|
||||
# Hermes' threshold (e.g. 80% of window) can sit BELOW the ladder's 85%
|
||||
# occupancy gate: 78K of a 96K window is 81%.
|
||||
kwargs = dict(current_window=98304, session_tokens=78_000,
|
||||
measured_decode_tok_s=None, server_idle=True)
|
||||
ungated = growth_decision(profile, budget, **kwargs)
|
||||
assert ungated.action == "hold", "sanity: below the ladder's own gate"
|
||||
|
||||
confirmed = growth_decision(profile, budget, occupancy_confirmed=True, **kwargs)
|
||||
assert confirmed.action == "grow"
|
||||
assert confirmed.next_window and confirmed.next_window > 98304
|
||||
|
||||
|
||||
def _stage_fake_gguf(mdir, name):
|
||||
mdir.mkdir(parents=True, exist_ok=True)
|
||||
(mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 64)
|
||||
|
||||
|
||||
def _header_stub(sampling: dict | None = None):
|
||||
"""A read_gguf_header stand-in for tests that monkeypatch the reader:
|
||||
just enough surface for preset generation (sampling ladder included)."""
|
||||
|
||||
class _Stub:
|
||||
sampling_defaults = dict(sampling or {})
|
||||
|
||||
return _Stub()
|
||||
|
||||
|
||||
def _tiny_profile(model_id: str):
|
||||
from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
|
||||
|
||||
gib = 1 << 30
|
||||
return ModelProfile(
|
||||
name=model_id, weights_bytes=2 * gib, embd_table_bytes=0,
|
||||
n_ctx_train=131072,
|
||||
layers=[(LayerKind.FULL, 512)] * 4)
|
||||
|
||||
|
||||
def test_preset_generation_for_catalog_model_with_mmproj(hermes_home, tmp_path, monkeypatch):
|
||||
"""generate_presets must survive a model that IS in the catalog and
|
||||
carries a vision projector — this executes the find_entry_for_model +
|
||||
mmproj overhead branch that synthetic test models skip. Regression:
|
||||
the branch once treated the (entry, variant) tuple as the entry and
|
||||
crashed every real boot into the stock-fit fallback."""
|
||||
import hermes_cli.local_runtime.presets as presets_mod
|
||||
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
# A real catalog id with an mmproj (the recommended row has one).
|
||||
entry = next(e for e in CATALOG if e.mmproj is not None)
|
||||
variant = entry.variants[-1]
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, variant.model_id)
|
||||
|
||||
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets_mod, "profile_from_gguf",
|
||||
lambda h: _tiny_profile(variant.model_id))
|
||||
|
||||
gib = 1 << 30
|
||||
budget = HardwareBudget(usable_vram_bytes=24 * gib,
|
||||
total_device_bytes=24 * gib,
|
||||
ram_available_bytes=64 * gib)
|
||||
entries = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")
|
||||
assert len(entries) == 1
|
||||
assert entries[0].refusal is None
|
||||
assert entries[0].window > 0
|
||||
|
||||
|
||||
def test_preset_restores_grown_window_capped_at_native(hermes_home, tmp_path, monkeypatch):
|
||||
"""A persisted override lifts the preset window; an absurd override is
|
||||
capped at native. GGUF parsing is stubbed — the contract under test is
|
||||
the override plumbing, not the reader."""
|
||||
import hermes_cli.local_runtime.presets as presets_mod
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, "tiny-dense")
|
||||
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets_mod, "profile_from_gguf",
|
||||
lambda h: _tiny_profile("tiny-dense"))
|
||||
|
||||
gib = 1 << 30
|
||||
budget = HardwareBudget(usable_vram_bytes=24 * gib,
|
||||
total_device_bytes=24 * gib,
|
||||
ram_available_bytes=64 * gib)
|
||||
preset = tmp_path / "presets.ini"
|
||||
|
||||
baseline = presets_mod.generate_presets(mdir, budget, preset)[0]
|
||||
assert baseline.window == 131072 # tiny model: native from the start
|
||||
|
||||
# Override above native must cap at native, not exceed it.
|
||||
save_window_override("tiny-dense", 10_000_000)
|
||||
capped = presets_mod.generate_presets(mdir, budget, preset)[0]
|
||||
assert capped.window == 131072
|
||||
|
||||
|
||||
def test_preset_ignores_override_below_launch_window(hermes_home, tmp_path, monkeypatch):
|
||||
"""Overrides only ever RAISE the window (growth is monotone); a stale
|
||||
smaller override never shrinks a launch decision."""
|
||||
import hermes_cli.local_runtime.presets as presets_mod
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, "tiny-dense")
|
||||
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets_mod, "profile_from_gguf",
|
||||
lambda h: _tiny_profile("tiny-dense"))
|
||||
save_window_override("tiny-dense", 65536)
|
||||
|
||||
gib = 1 << 30
|
||||
budget = HardwareBudget(usable_vram_bytes=24 * gib,
|
||||
total_device_bytes=24 * gib,
|
||||
ram_available_bytes=64 * gib)
|
||||
entry = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")[0]
|
||||
assert entry.window == 131072
|
||||
|
||||
|
||||
def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypatch):
|
||||
"""The real growth shape: launch at a lower rung, override to a middle
|
||||
rung -> the preset window follows the override."""
|
||||
import hermes_cli.local_runtime.presets as presets_mod
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
|
||||
gib = 1 << 30
|
||||
# Expensive dense KV so the launch decision lands BELOW native on this
|
||||
# budget: 60 layers x 4 KiB/tok f16 -> q8 ~= 120 KiB/tok.
|
||||
profile = ModelProfile(
|
||||
name="big-dense", weights_bytes=20 * gib, embd_table_bytes=0,
|
||||
n_ctx_train=262144,
|
||||
layers=[(LayerKind.FULL, 4096)] * 60)
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, "big-dense")
|
||||
monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets_mod, "profile_from_gguf", lambda h: profile)
|
||||
|
||||
budget = HardwareBudget(usable_vram_bytes=28 * gib,
|
||||
total_device_bytes=32 * gib,
|
||||
ram_available_bytes=128 * gib)
|
||||
baseline = presets_mod.generate_presets(mdir, budget, tmp_path / "a.ini")[0]
|
||||
assert baseline.window < 262144, "sanity: launch below native"
|
||||
|
||||
grown = baseline.window * 2
|
||||
save_window_override("big-dense", grown)
|
||||
restored = presets_mod.generate_presets(mdir, budget, tmp_path / "b.ini")[0]
|
||||
assert restored.window >= grown, "override must lift the launch window"
|
||||
|
||||
|
||||
def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
|
||||
"""The sampling deference ladder: the GGUF's own general.sampling.*
|
||||
wins per key, catalog fills only what the file left silent, and a
|
||||
model with neither gets no sampling keys at all (llama.cpp defaults).
|
||||
Policy keys (ctx-size, cache types) must never be displaced."""
|
||||
import configparser
|
||||
|
||||
import hermes_cli.local_runtime.presets as presets_mod
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
# A real catalog entry WITH catalog sampling, staged on disk.
|
||||
entry = next(e for e in CATALOG if e.sampling)
|
||||
variant = entry.variants[-1]
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, variant.model_id)
|
||||
_stage_fake_gguf(mdir, "off-catalog-model")
|
||||
|
||||
gib = 1 << 30
|
||||
budget = HardwareBudget(usable_vram_bytes=64 * gib,
|
||||
total_device_bytes=64 * gib,
|
||||
ram_available_bytes=64 * gib)
|
||||
# The catalog model's file carries temp; catalog must fill the rest
|
||||
# but NOT displace the file's value. The off-catalog file carries none.
|
||||
def fake_header(path):
|
||||
if variant.model_id in str(path):
|
||||
return _header_stub({"temp": "0.42"})
|
||||
return _header_stub()
|
||||
|
||||
monkeypatch.setattr(presets_mod, "read_gguf_header", fake_header)
|
||||
monkeypatch.setattr(presets_mod, "profile_from_gguf",
|
||||
lambda h: _tiny_profile("x"))
|
||||
|
||||
out = tmp_path / "presets.ini"
|
||||
presets_mod.generate_presets(mdir, budget, out)
|
||||
ini = configparser.ConfigParser()
|
||||
ini.read(out)
|
||||
|
||||
sec = ini[variant.model_id]
|
||||
assert sec["temp"] == "0.42", "file's own sampling must win per key"
|
||||
for k, v in entry.sampling.items():
|
||||
if k != "temp":
|
||||
assert sec[k] == v, f"catalog must fill the silent key {k}"
|
||||
assert "ctx-size" in sec, "policy keys survive the ladder"
|
||||
|
||||
off = ini["off-catalog-model"]
|
||||
assert "temp" not in off and "top-p" not in off, (
|
||||
"no file keys + no catalog entry = llama.cpp defaults, not ours")
|
||||
@@ -0,0 +1,293 @@
|
||||
"""Contract tests for the local-models dashboard routes (Rollout 4).
|
||||
|
||||
Real FastAPI TestClient against the real router; the runtime pieces
|
||||
underneath are exercised against temp HERMES_HOME (autouse fixture). Network
|
||||
downloads are stubbed at the urllib boundary — never live."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
||||
from hermes_cli import web_server
|
||||
|
||||
test_client = TestClient(web_server.app)
|
||||
# Same auth pattern as the git-route tests: present the session token.
|
||||
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
|
||||
return test_client
|
||||
|
||||
|
||||
def test_local_models_routes_require_auth(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
||||
from hermes_cli import web_server
|
||||
|
||||
unauth = TestClient(web_server.app)
|
||||
assert unauth.get("/api/local-models/status").status_code == 401
|
||||
|
||||
|
||||
def _write_fake_gguf(path: Path, size: int = 1024) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(b"GGUF" + b"\x00" * size)
|
||||
|
||||
|
||||
# ── status ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_status_shape_and_defaults(client):
|
||||
r = client.get("/api/local-models/status")
|
||||
assert r.status_code == 200
|
||||
data = r.json()
|
||||
# Contract: every key the pane's first paint needs, present and typed.
|
||||
assert isinstance(data["enabled"], bool)
|
||||
assert isinstance(data["tag"], str) and data["tag"].startswith("b")
|
||||
assert isinstance(data["runtime_installed"], bool)
|
||||
assert isinstance(data["server_running"], bool)
|
||||
assert isinstance(data["models"], list)
|
||||
|
||||
|
||||
def test_status_lists_staged_models_with_labels(client, tmp_path):
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
|
||||
_write_fake_gguf(models_dir() / "Some-Model.gguf", size=2048)
|
||||
data = client.get("/api/local-models/status").json()
|
||||
ids = [m["id"] for m in data["models"]]
|
||||
assert "Some-Model" in ids
|
||||
row = data["models"][ids.index("Some-Model")]
|
||||
assert row["size_bytes"] > 0
|
||||
assert row["size_label"].endswith("GB")
|
||||
|
||||
|
||||
# ── hardware ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_hardware_plain_facts(client):
|
||||
data = client.get("/api/local-models/hardware").json()
|
||||
assert isinstance(data["uma"], bool)
|
||||
assert data["ram_total_bytes"] > 0
|
||||
assert data["vram_total_bytes"] >= 0
|
||||
# GPU fields are None-able (non-NVIDIA machines) but must exist.
|
||||
assert "gpu_name" in data and "gpu_util_percent" in data and "vram_used_bytes" in data
|
||||
|
||||
|
||||
# ── catalog ──────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_catalog_prices_every_entry_for_this_machine(client):
|
||||
data = client.get("/api/local-models/catalog").json()
|
||||
assert len(data["models"]) >= 3
|
||||
for row in data["models"]:
|
||||
# The three user questions, answered on every row:
|
||||
assert row["size_label"].endswith("GB") # how big
|
||||
assert isinstance(row["fits"], bool) # will it fit
|
||||
assert row["fit_summary"] # what shape
|
||||
if row["fits"]:
|
||||
assert row["start_window"] >= 1
|
||||
assert row["start_window_label"].endswith("K")
|
||||
else:
|
||||
assert "memory" in row["fit_summary"].lower()
|
||||
assert isinstance(row["downloaded"], bool)
|
||||
|
||||
|
||||
def test_catalog_never_hides_unaffordable_models(client, monkeypatch):
|
||||
"""Unaffordable entries stay visible with a plain reason — hiding them
|
||||
is how users conclude the feature is broken."""
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
tiny = HardwareBudget(usable_vram_bytes=1 << 30, total_device_bytes=1 << 30,
|
||||
ram_available_bytes=1 << 30)
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
|
||||
lambda **kw: tiny)
|
||||
data = client.get("/api/local-models/catalog").json()
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
|
||||
assert len(data["models"]) == len(CATALOG)
|
||||
refused = [m for m in data["models"] if not m["fits"]]
|
||||
assert refused, "a 1 GiB machine must refuse the 20 GB models"
|
||||
for row in refused:
|
||||
assert row["fit_detail"] or row["fit_summary"]
|
||||
|
||||
|
||||
# ── downloads ────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_download_unknown_model_404s(client):
|
||||
r = client.post("/api/local-models/download", json={"model_id": "nope"})
|
||||
assert r.status_code == 404
|
||||
|
||||
|
||||
def test_download_short_of_server_length_errors_and_cleans_up(client, monkeypatch):
|
||||
"""Catalog sizes are advisory (upstream re-uploads may make them
|
||||
stale — a mismatch against the CATALOG must not fail a download).
|
||||
The server's own declared length is the only completeness check:
|
||||
fewer bytes than the server promised means a dropped connection, so
|
||||
the job errors and nothing is staged."""
|
||||
|
||||
class FakeResponse(io.BytesIO):
|
||||
# Body is 17 bytes; the server promises 32 — a truncated stream.
|
||||
headers = {"Content-Length": "32"}
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr("urllib.request.urlopen",
|
||||
lambda *a, **k: FakeResponse(b"not the real body"))
|
||||
|
||||
# Pin a generous budget: variant selection prices against the machine
|
||||
# running the test, and a GPU-less CI runner honestly refuses every
|
||||
# build (409) — this test is about the download path, not selection.
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
budget = HardwareBudget(usable_vram_bytes=64 << 30,
|
||||
total_device_bytes=64 << 30,
|
||||
ram_available_bytes=64 << 30)
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
|
||||
lambda **kw: budget)
|
||||
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
|
||||
entry_id = CATALOG[0].id
|
||||
r = client.post("/api/local-models/download", json={"model_id": entry_id})
|
||||
assert r.status_code == 200
|
||||
job_id = r.json()["job_id"]
|
||||
assert job_id
|
||||
|
||||
deadline = time.time() + 10
|
||||
status = None
|
||||
while time.time() < deadline:
|
||||
status = client.get(f"/api/local-models/jobs/{job_id}").json()
|
||||
if status["status"] in ("done", "error"):
|
||||
break
|
||||
time.sleep(0.05)
|
||||
assert status is not None and status["status"] == "error"
|
||||
assert "bytes" in status["error"].lower()
|
||||
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
|
||||
assert not (models_dir() / f"{entry_id}.gguf").exists()
|
||||
assert not (models_dir() / f"{entry_id}.part").exists()
|
||||
|
||||
|
||||
def test_download_already_downloaded_short_circuits(client, monkeypatch):
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
from hermes_cli.local_runtime.catalog import CATALOG, select_variant
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
# Pin the budget so the selected variant is deterministic in the test.
|
||||
budget = HardwareBudget(usable_vram_bytes=64 << 30, total_device_bytes=64 << 30,
|
||||
ram_available_bytes=64 << 30)
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
|
||||
lambda **kw: budget)
|
||||
choice = select_variant(CATALOG[0], budget)
|
||||
assert choice is not None
|
||||
_write_fake_gguf(models_dir() / choice.variant.files[0].local_name)
|
||||
r = client.post("/api/local-models/download", json={"model_id": CATALOG[0].id})
|
||||
assert r.status_code == 200
|
||||
assert r.json()["already_downloaded"] is True
|
||||
|
||||
|
||||
def test_delete_model(client):
|
||||
from hermes_cli.local_runtime.bootstrap import models_dir
|
||||
|
||||
_write_fake_gguf(models_dir() / "Doomed.gguf")
|
||||
assert client.delete("/api/local-models/models/Doomed").status_code == 200
|
||||
assert not (models_dir() / "Doomed.gguf").exists()
|
||||
assert client.delete("/api/local-models/models/Doomed").status_code == 404
|
||||
|
||||
|
||||
# ── runtime install ──────────────────────────────────────────
|
||||
|
||||
|
||||
def test_runtime_install_rejects_impossible_combo(client, monkeypatch):
|
||||
"""Impossible platform/backend combos fail the POST itself with the
|
||||
resolver's honest message — not a background job that dies silently.
|
||||
(win-arm64-vulkan; the old cuda case became real upstream at ~b1036x.)"""
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries._host_os_arch", lambda: ("win", "arm64"))
|
||||
r = client.post("/api/local-models/runtime/install", json={"backend": "vulkan"})
|
||||
assert r.status_code == 400
|
||||
assert "arm64" in r.json()["detail"]
|
||||
|
||||
|
||||
def test_job_poll_unknown_404s(client):
|
||||
assert client.get("/api/local-models/jobs/deadbeef").status_code == 404
|
||||
|
||||
|
||||
def test_eject_without_supervisor_is_not_a_500(client, monkeypatch):
|
||||
"""Eject on an ADOPTED server (no in-process supervisor — the shape
|
||||
every backend restart produces, since boot adopts the running server
|
||||
via the state file) must route through the persisted endpoint, not
|
||||
crash. Regression: _state_endpoint was only imported inside the
|
||||
status route, so eject raised NameError -> 500 for every adopted-
|
||||
server session."""
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.get_supervisor", lambda: None)
|
||||
# No running server either: the route must answer 409 (no server),
|
||||
# never a NameError 500.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models._state_endpoint", lambda: None)
|
||||
r = client.post("/api/local-models/eject", json={"model_id": "anything"})
|
||||
assert r.status_code == 409, (r.status_code, r.text)
|
||||
|
||||
|
||||
def test_download_tolerates_stale_catalog_size(client, monkeypatch):
|
||||
"""Upstream re-uploads make catalog sizes stale; a download whose
|
||||
delivered bytes are self-consistent with the SERVER's declared length
|
||||
must succeed even when the catalog said something else. (This is the
|
||||
tolerance the sha removal was for — being out of date must not break
|
||||
downloads.)"""
|
||||
|
||||
body = b"x" * 48 # server-consistent: Content-Length == body length
|
||||
|
||||
class FakeResponse(io.BytesIO):
|
||||
headers = {"Content-Length": str(len(body))}
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr("urllib.request.urlopen",
|
||||
lambda *a, **k: FakeResponse(body))
|
||||
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
budget = HardwareBudget(usable_vram_bytes=64 << 30,
|
||||
total_device_bytes=64 << 30,
|
||||
ram_available_bytes=64 << 30)
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
|
||||
lambda **kw: budget)
|
||||
# Keep the post-download server bounce out of this unit.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
|
||||
lambda: False)
|
||||
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
|
||||
# Catalog size for this entry is in the tens of GB — wildly stale
|
||||
# versus our 48-byte body. The download must still land.
|
||||
entry_id = CATALOG[0].id
|
||||
r = client.post("/api/local-models/download", json={"model_id": entry_id})
|
||||
assert r.status_code == 200
|
||||
job_id = r.json()["job_id"]
|
||||
|
||||
deadline = time.time() + 10
|
||||
status = None
|
||||
while time.time() < deadline:
|
||||
status = client.get(f"/api/local-models/jobs/{job_id}").json()
|
||||
if status["status"] in ("done", "error"):
|
||||
break
|
||||
time.sleep(0.05)
|
||||
assert status is not None and status["status"] == "done", status.get("error")
|
||||
@@ -0,0 +1,74 @@
|
||||
"""The managed local server owns its picker identity.
|
||||
|
||||
A live session on the managed llama-server reports provider "custom"
|
||||
(the resolution seam's generic label for a raw base_url). The picker
|
||||
payload used to materialize that as a duplicate "Custom endpoint" group
|
||||
above the Local row — same staged models listed twice, checkmark on the
|
||||
wrong group. Contract: when the current session points at the managed
|
||||
endpoint, the Local row is current and no custom-endpoint duplicate
|
||||
exists; a user's own external endpoint keeps its row untouched."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
MANAGED = {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"}
|
||||
STAGED = {"Qwen-A-UD-Q4_K_M", "Qwen-B-UD-Q4_K_M"}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ctx(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
||||
import hermes_cli.inventory as inv
|
||||
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.staged_model_ids",
|
||||
lambda: set(STAGED))
|
||||
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
|
||||
lambda: dict(MANAGED))
|
||||
context = inv.load_picker_context()
|
||||
return inv, context
|
||||
|
||||
|
||||
def _rows(inv, context, **overrides):
|
||||
context = dataclasses.replace(context, **overrides)
|
||||
return inv.build_models_payload(context, explicit_only=True)["providers"]
|
||||
|
||||
|
||||
def test_managed_custom_session_shows_only_the_local_row(ctx):
|
||||
inv, context = ctx
|
||||
rows = _rows(inv, context,
|
||||
current_provider="custom",
|
||||
current_model="Qwen-A-UD-Q4_K_M",
|
||||
current_base_url=MANAGED["base_url"])
|
||||
slugs = [r["slug"] for r in rows]
|
||||
assert "llamacpp" in slugs
|
||||
assert "custom" not in slugs, (
|
||||
"managed endpoint leaked a duplicate 'Custom endpoint' group")
|
||||
local = next(r for r in rows if r["slug"] == "llamacpp")
|
||||
assert local["is_current"] is True
|
||||
assert local["name"] == "Local"
|
||||
|
||||
|
||||
def test_external_custom_endpoint_keeps_its_row(ctx):
|
||||
inv, context = ctx
|
||||
rows = _rows(inv, context,
|
||||
current_provider="custom",
|
||||
current_model="some-model",
|
||||
current_base_url="http://my-vllm-box:8000/v1")
|
||||
slugs = [r["slug"] for r in rows]
|
||||
assert "custom" in slugs, "a real external endpoint must keep its row"
|
||||
custom = next(r for r in rows if r["slug"] == "custom")
|
||||
assert custom["is_current"] is True
|
||||
local = next(r for r in rows if r["slug"] == "llamacpp")
|
||||
assert local["is_current"] is False
|
||||
|
||||
|
||||
def test_remote_provider_session_unaffected(ctx):
|
||||
inv, context = ctx
|
||||
rows = _rows(inv, context)
|
||||
local = next(r for r in rows if r["slug"] == "llamacpp")
|
||||
assert local["is_current"] is False
|
||||
assert "custom" not in [r["slug"] for r in rows if r.get("is_current")]
|
||||
@@ -0,0 +1,187 @@
|
||||
"""Quickstart route: one POST from nothing to a working local default.
|
||||
|
||||
Contract, not implementation: the route must (a) preflight-fail
|
||||
synchronously when nothing fits, (b) report which legs the job will run
|
||||
(runtime install / model download), skipping legs already satisfied,
|
||||
and (c) run install -> download -> activate through the same code paths
|
||||
the individual routes use. The slow legs are stubbed at their module
|
||||
boundaries; the sequencing and job bookkeeping are real.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
||||
from hermes_cli import web_server
|
||||
|
||||
test_client = TestClient(web_server.app)
|
||||
test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
|
||||
return test_client
|
||||
|
||||
|
||||
def _wait_job(client, job_id: str, timeout: float = 10.0) -> dict:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
job = client.get(f"/api/local-models/jobs/{job_id}").json()
|
||||
if job["status"] != "running":
|
||||
return job
|
||||
time.sleep(0.05)
|
||||
raise AssertionError(f"job {job_id} still running after {timeout}s")
|
||||
|
||||
|
||||
def test_quickstart_unknown_model_404s(client):
|
||||
r = client.post("/api/local-models/quickstart", json={"model_id": "no-such"})
|
||||
assert r.status_code == 404
|
||||
|
||||
|
||||
def test_quickstart_refuses_when_nothing_fits(client, monkeypatch):
|
||||
"""Preflight is synchronous: a machine no catalog entry fits gets a 409
|
||||
with guidance, not a doomed background job."""
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.catalog.select_variant", lambda *a, **k: None)
|
||||
r = client.post("/api/local-models/quickstart", json={})
|
||||
assert r.status_code == 409
|
||||
assert "Local Models" in r.json()["detail"]
|
||||
|
||||
|
||||
def test_quickstart_runs_all_three_legs(client, monkeypatch, tmp_path):
|
||||
"""Fresh machine: install runtime -> download recommended -> activate.
|
||||
Each leg is asserted by its observable call, in order."""
|
||||
calls: list[str] = []
|
||||
|
||||
# Leg 1: no runtime installed yet; install is the stubbed binaries call.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.installed_tags", lambda: [])
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.ensure_runtime_installed",
|
||||
lambda tag, backend, progress=None: calls.append("install"))
|
||||
|
||||
# Leg 2: nothing staged; the download writes the files the plan names.
|
||||
def _fake_download(url, dest, job, *, base_done=0, keep_totals=False):
|
||||
Path(dest).parent.mkdir(parents=True, exist_ok=True)
|
||||
Path(dest).write_bytes(b"GGUF\x00")
|
||||
calls.append("download")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models.download_file", _fake_download)
|
||||
|
||||
# Leg 3: activation — stub the server start and the model assignment.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
|
||||
lambda config, force=False: calls.append("server") or None)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models._state_endpoint",
|
||||
lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
|
||||
from hermes_cli import web_deps
|
||||
|
||||
monkeypatch.setattr(
|
||||
web_deps, "late",
|
||||
lambda name: (lambda *a, **k: calls.append("assign")))
|
||||
|
||||
r = client.post("/api/local-models/quickstart", json={})
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["needs_runtime"] is True
|
||||
assert body["needs_download"] is True
|
||||
assert body["download_bytes"] > 0
|
||||
|
||||
job = _wait_job(client, body["job_id"])
|
||||
assert job["status"] == "done", job["error"]
|
||||
assert job["kind"] == "quickstart"
|
||||
# Order is the contract: engine, weights, server, default.
|
||||
assert calls[0] == "install"
|
||||
assert "download" in calls
|
||||
assert calls.index("install") < calls.index("download") < calls.index("assign")
|
||||
|
||||
# Durable effect: the runtime is enabled in config.
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
assert load_config()["local_runtime"]["enabled"] is True
|
||||
|
||||
|
||||
def test_quickstart_skips_satisfied_legs(client, monkeypatch):
|
||||
"""Runtime present and model already staged: the response says so and
|
||||
the job goes straight to activation."""
|
||||
calls: list[str] = []
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.ensure_runtime_installed",
|
||||
lambda tag, backend, progress=None: calls.append("install"))
|
||||
|
||||
# Every catalog variant reads as staged.
|
||||
from hermes_cli.local_runtime.catalog import CATALOG
|
||||
|
||||
all_ids = {v.model_id for e in CATALOG for v in e.variants}
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.staged_model_ids", lambda: all_ids)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models.download_file",
|
||||
lambda *a, **k: calls.append("download"))
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
|
||||
lambda config, force=False: None)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models._state_endpoint",
|
||||
lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
|
||||
from hermes_cli import web_deps
|
||||
|
||||
monkeypatch.setattr(
|
||||
web_deps, "late",
|
||||
lambda name: (lambda *a, **k: calls.append("assign")))
|
||||
|
||||
r = client.post("/api/local-models/quickstart", json={})
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["needs_runtime"] is False
|
||||
assert body["needs_download"] is False
|
||||
assert body["download_bytes"] == 0
|
||||
|
||||
job = _wait_job(client, body["job_id"])
|
||||
assert job["status"] == "done", job["error"]
|
||||
assert "install" not in calls and "download" not in calls
|
||||
assert calls == ["assign"] or calls[-1] == "assign"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def quickstart_ready(monkeypatch):
|
||||
"""Preflight passes without hardware or network: the runtime reads as
|
||||
installed and every entry's first variant is servable, so the POST
|
||||
reaches the single-flight lock instead of 409ing at fit/engine
|
||||
preflight on machines where nothing fits."""
|
||||
from hermes_cli.local_runtime.catalog import VariantChoice
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.local_runtime.catalog.select_variant",
|
||||
lambda entry, budget: VariantChoice(variant=entry.variants[0],
|
||||
zero_spill=True,
|
||||
reason_key="best-fits"))
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.web_routers.local_models._engine_too_old",
|
||||
lambda min_engine: False)
|
||||
|
||||
|
||||
def test_quickstart_is_single_flight(client, quickstart_ready, monkeypatch):
|
||||
"""A second quickstart while one runs must 409, not start a twin job
|
||||
(the job sequences installs, downloads, a server bounce, and a config
|
||||
write — two interleaved runs corrupt all four)."""
|
||||
import hermes_cli.web_routers.local_models as lm
|
||||
|
||||
lm._QUICKSTART_LOCK.acquire()
|
||||
try:
|
||||
r = client.post("/api/local-models/quickstart", json={})
|
||||
assert r.status_code == 409
|
||||
assert "already running" in r.json()["detail"].lower()
|
||||
finally:
|
||||
lm._QUICKSTART_LOCK.release()
|
||||
@@ -0,0 +1,170 @@
|
||||
"""The recommendation decision table — the reviewable matrix.
|
||||
|
||||
The recommendation itself is DERIVED (catalog.recommended_entry: best
|
||||
quality among resident entries clearing the pleasant speed floor, else
|
||||
fastest resident, else least-painful spilled), so nobody hand-maintains
|
||||
per-hardware-class picks. This table is the editorial control on that
|
||||
derivation: it enumerates the real memory size classes x {discrete,
|
||||
unified} and pins every cell. A catalog change (new model, quality
|
||||
re-rank, quant swap) flips cells HERE, and the diff of this file in
|
||||
review IS the sign-off on what each machine class gets.
|
||||
|
||||
These are decision pins, not change-detectors: each cell is a choice a
|
||||
human approved, exactly like a golden file. When a cell flips on
|
||||
purpose, update it in the same commit and say why. When one flips by
|
||||
surprise, that is the test doing its job.
|
||||
|
||||
Budgets mirror hardware.probe_budget's planning-mode shapes (margins,
|
||||
UMA headroom) so the cells match what a real machine of that class
|
||||
resolves.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.local_runtime.catalog import (
|
||||
CATALOG,
|
||||
PLEASANT_FLOOR_TOK_S,
|
||||
predicted_decode_tok_s,
|
||||
recommended_entry,
|
||||
recommended_id,
|
||||
select_variant,
|
||||
)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget
|
||||
|
||||
_GIB = 1 << 30
|
||||
|
||||
|
||||
def _discrete(size_gb: int) -> HardwareBudget:
|
||||
total = size_gb * _GIB
|
||||
margin = max(2 * _GIB, int(total * 0.09))
|
||||
return HardwareBudget(usable_vram_bytes=max(0, total - margin),
|
||||
total_device_bytes=total,
|
||||
ram_available_bytes=64 * _GIB, uma=False)
|
||||
|
||||
|
||||
def _unified(size_gb: int) -> HardwareBudget:
|
||||
total = size_gb * _GIB
|
||||
return HardwareBudget(usable_vram_bytes=int(total * 0.80),
|
||||
total_device_bytes=total,
|
||||
ram_available_bytes=0, uma=True)
|
||||
|
||||
|
||||
# The decision table. Cells were generated by the resolver and then
|
||||
# reviewed as editorial decisions:
|
||||
#
|
||||
# VRAM | discrete | unified
|
||||
# -----+-------------------------+------------------------
|
||||
# 8 | qwen3.6-35b-a3b spilled | (none fits)
|
||||
# 16 | qwen3.6-35b-a3b spilled | (none fits)
|
||||
# 24 | qwen3.8-27b | (none fits)
|
||||
# 32 | qwen3.8-27b | qwen3.6-35b-a3b
|
||||
# 48 | qwen3.8-27b | qwen3.6-35b-a3b
|
||||
# 96 | qwen3.8-27b | qwen3.6-35b-a3b
|
||||
# 128 | qwen3.8-flash-next | qwen3.6-35b-a3b
|
||||
# 256 | qwen3.8-flash-next | qwen3.8-flash-next
|
||||
# 512 | qwen3.8-flash-next | qwen3.8-flash-next
|
||||
#
|
||||
# Reading guide for reviewers:
|
||||
# - Discrete <=16 GB: nothing runs resident; the 35B MoE is the least
|
||||
# painful spill (active slice streams from host; a dense spill reads
|
||||
# every weight over the bus).
|
||||
# - Discrete 24-96 GB: the 27B is the flagship experience — dense reads
|
||||
# at ~1 TB/s clear the floor easily, so quality decides.
|
||||
# - Discrete/unified where Flash Next fits resident (128 GB discrete,
|
||||
# 256+ GB unified): the frontier model is the pick — highest quality,
|
||||
# and its sparse decode clears the floor even at UMA bandwidth
|
||||
# (~24 tok/s predicted at 210 GB/s).
|
||||
# - Unified 32-128 GB — the Spark class, the reason this resolver
|
||||
# exists: the dense 27B predicts ~13 tok/s at UMA bandwidth (below
|
||||
# the pleasant floor), so the 35B-A3B (~60 tok/s) wins.
|
||||
# - Unified <=24 GB: no entry passes the physics check inside the UMA
|
||||
# budget (spilling is impossible on UMA by construction — the pool IS
|
||||
# the RAM). The pane's browse flow is the path for those machines
|
||||
# until a small catalog entry lands (revisit when one does).
|
||||
DECISION_TABLE = [
|
||||
(8, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
|
||||
(8, "unified", None, None),
|
||||
(16, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
|
||||
(16, "unified", None, None),
|
||||
(24, "discrete", "qwen3.8-27b", "best-quality-resident"),
|
||||
(24, "unified", None, None),
|
||||
(32, "discrete", "qwen3.8-27b", "best-quality-resident"),
|
||||
(32, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
|
||||
(48, "discrete", "qwen3.8-27b", "best-quality-resident"),
|
||||
(48, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
|
||||
(96, "discrete", "qwen3.8-27b", "best-quality-resident"),
|
||||
(96, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
|
||||
(128, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
|
||||
(128, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
|
||||
(256, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
|
||||
(256, "unified", "qwen3.8-flash-next", "best-quality-resident"),
|
||||
(512, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
|
||||
(512, "unified", "qwen3.8-flash-next", "best-quality-resident"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("size_gb", "kind", "expected", "expected_reason"),
|
||||
DECISION_TABLE,
|
||||
ids=[f"{s}GB-{k}" for s, k, _, _ in DECISION_TABLE])
|
||||
def test_recommendation_decision_table(size_gb, kind, expected, expected_reason):
|
||||
"""Pins the pick AND its reason per cell: the reason is user-facing
|
||||
(the Recommended badge's tooltip), so a cell whose rationale flips
|
||||
without the pick flipping is still a review-worthy change."""
|
||||
budget = _discrete(size_gb) if kind == "discrete" else _unified(size_gb)
|
||||
picked = recommended_entry(budget)
|
||||
if expected is None:
|
||||
assert picked is None
|
||||
else:
|
||||
assert picked is not None
|
||||
assert (picked[0].id, picked[1]) == (expected, expected_reason)
|
||||
|
||||
|
||||
# ── invariants behind the table (survive catalog changes) ──
|
||||
|
||||
|
||||
def test_every_entry_carries_the_recommendation_axes():
|
||||
"""quality and decode_fraction are authoring requirements: an entry
|
||||
without them silently loses every quality comparison (quality=0) or
|
||||
prices as dense (decode_fraction=1.0)."""
|
||||
for entry in CATALOG:
|
||||
assert entry.quality > 0, f"{entry.id} has no quality ordering"
|
||||
assert 0.0 < entry.decode_fraction <= 1.0, entry.id
|
||||
if not entry.moe:
|
||||
assert entry.decode_fraction == 1.0, (
|
||||
f"{entry.id} is dense — it reads every weight per token")
|
||||
|
||||
|
||||
def test_unified_never_recommends_a_below_floor_dense_model():
|
||||
"""The Spark rule, as an invariant: whatever the catalog holds, a
|
||||
unified-memory machine must not be told to run a model whose
|
||||
predicted decode is below the pleasant floor while a resident
|
||||
alternative clears it."""
|
||||
budget = _unified(128)
|
||||
pick = recommended_id(budget)
|
||||
assert pick is not None
|
||||
entry = next(e for e in CATALOG if e.id == pick)
|
||||
choice = select_variant(entry, budget)
|
||||
assert choice is not None and choice.zero_spill
|
||||
clears = [
|
||||
e for e in CATALOG
|
||||
if (c := select_variant(e, budget)) is not None and c.zero_spill
|
||||
and predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
|
||||
]
|
||||
if clears:
|
||||
assert predicted_decode_tok_s(entry, choice.variant, budget) >= PLEASANT_FLOOR_TOK_S
|
||||
|
||||
|
||||
def test_quality_decides_where_speed_permits():
|
||||
"""On big discrete hardware every resident entry clears the floor, so
|
||||
the pick must be the highest-quality fitting entry — the axis that
|
||||
justifies carrying an editorial field at all."""
|
||||
budget = _discrete(512)
|
||||
pick = recommended_id(budget)
|
||||
resident = [
|
||||
e for e in CATALOG
|
||||
if (c := select_variant(e, budget)) is not None and c.zero_spill
|
||||
]
|
||||
assert pick == max(resident, key=lambda e: e.quality).id
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user