refactor(hermes_cli/web_routers): reflow docstrings/comments, capability field table, compact literals (local_models 1043, models 334, dashboard_ui 314)
This commit is contained in:
@@ -14,11 +14,7 @@ from fastapi.responses import FileResponse
|
||||
|
||||
from hermes_cli.web_deps import LateState, late
|
||||
from hermes_cli.web_models import (
|
||||
FontSetBody,
|
||||
ThemeSetBody,
|
||||
_AgentPluginInstallBody,
|
||||
_PluginProvidersPutBody,
|
||||
_PluginVisibilityBody,
|
||||
FontSetBody, ThemeSetBody, _AgentPluginInstallBody, _PluginProvidersPutBody, _PluginVisibilityBody,
|
||||
)
|
||||
|
||||
_log = logging.getLogger("hermes_cli.web_server")
|
||||
@@ -206,32 +202,26 @@ def _named_plugin_action(request: Request, name: str, action: Callable[[str], di
|
||||
@router.post("/api/dashboard/agent-plugins/{name:path}/enable")
|
||||
async def post_agent_plugin_enable(request: Request, name: str):
|
||||
from hermes_cli.plugins_cmd import dashboard_set_agent_plugin_enabled
|
||||
|
||||
return _named_plugin_action(
|
||||
request, name, lambda n: dashboard_set_agent_plugin_enabled(n, enabled=True), "Enable failed.", rescan=False
|
||||
)
|
||||
return _named_plugin_action(request, name, lambda n: dashboard_set_agent_plugin_enabled(n, enabled=True),
|
||||
"Enable failed.", rescan=False)
|
||||
|
||||
|
||||
@router.post("/api/dashboard/agent-plugins/{name:path}/disable")
|
||||
async def post_agent_plugin_disable(request: Request, name: str):
|
||||
from hermes_cli.plugins_cmd import dashboard_set_agent_plugin_enabled
|
||||
|
||||
return _named_plugin_action(
|
||||
request, name, lambda n: dashboard_set_agent_plugin_enabled(n, enabled=False), "Disable failed.", rescan=False
|
||||
)
|
||||
return _named_plugin_action(request, name, lambda n: dashboard_set_agent_plugin_enabled(n, enabled=False),
|
||||
"Disable failed.", rescan=False)
|
||||
|
||||
|
||||
@router.post("/api/dashboard/agent-plugins/{name:path}/update")
|
||||
async def post_agent_plugin_update(request: Request, name: str):
|
||||
from hermes_cli.plugins_cmd import dashboard_update_user_plugin
|
||||
|
||||
return _named_plugin_action(request, name, dashboard_update_user_plugin, "Update failed.", rescan=True)
|
||||
|
||||
|
||||
@router.delete("/api/dashboard/agent-plugins/{name:path}")
|
||||
async def delete_agent_plugin(request: Request, name: str):
|
||||
from hermes_cli.plugins_cmd import dashboard_remove_user_plugin
|
||||
|
||||
return _named_plugin_action(request, name, dashboard_remove_user_plugin, "Remove failed.", rescan=True)
|
||||
|
||||
|
||||
@@ -269,12 +259,10 @@ async def post_plugin_visibility(request: Request, name: str, body: _PluginVisib
|
||||
hidden_list: list = config["dashboard"].get("hidden_plugins") or []
|
||||
if not isinstance(hidden_list, list):
|
||||
hidden_list = []
|
||||
|
||||
if body.hidden and name not in hidden_list:
|
||||
hidden_list.append(name)
|
||||
elif not body.hidden and name in hidden_list:
|
||||
hidden_list.remove(name)
|
||||
|
||||
config["dashboard"]["hidden_plugins"] = hidden_list
|
||||
save_config(config)
|
||||
_invalidate_plugins_hub_cache()
|
||||
@@ -287,23 +275,11 @@ async def post_plugin_visibility(request: Request, name: str, body: _PluginVisib
|
||||
# never leak ``.py`` backend sources, READMEs, ``.env.example`` templates, etc.
|
||||
# Add to it deliberately when a new asset type comes up; do NOT add a fallback.
|
||||
_PLUGIN_ASSET_CONTENT_TYPES = {
|
||||
".js": "application/javascript",
|
||||
".mjs": "application/javascript",
|
||||
".css": "text/css",
|
||||
".json": "application/json",
|
||||
".html": "text/html",
|
||||
".svg": "image/svg+xml",
|
||||
".png": "image/png",
|
||||
".jpg": "image/jpeg",
|
||||
".jpeg": "image/jpeg",
|
||||
".gif": "image/gif",
|
||||
".webp": "image/webp",
|
||||
".ico": "image/x-icon",
|
||||
".woff2": "font/woff2",
|
||||
".woff": "font/woff",
|
||||
".ttf": "font/ttf",
|
||||
".otf": "font/otf",
|
||||
".map": "application/json",
|
||||
".js": "application/javascript", ".mjs": "application/javascript", ".css": "text/css",
|
||||
".json": "application/json", ".html": "text/html", ".svg": "image/svg+xml",
|
||||
".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".gif": "image/gif",
|
||||
".webp": "image/webp", ".ico": "image/x-icon", ".woff2": "font/woff2", ".woff": "font/woff",
|
||||
".ttf": "font/ttf", ".otf": "font/otf", ".map": "application/json",
|
||||
}
|
||||
|
||||
|
||||
@@ -335,8 +311,4 @@ async def serve_plugin_asset(plugin_name: str, file_path: str):
|
||||
media_type = _PLUGIN_ASSET_CONTENT_TYPES.get(target.suffix.lower())
|
||||
if media_type is None:
|
||||
raise HTTPException(status_code=404, detail="File not found")
|
||||
return FileResponse(
|
||||
target,
|
||||
media_type=media_type,
|
||||
headers={"Cache-Control": "no-store, no-cache, must-revalidate"},
|
||||
)
|
||||
return FileResponse(target, media_type=media_type, headers={"Cache-Control": "no-store, no-cache, must-revalidate"})
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
"""Local-models dashboard routes — the desktop's window into the managed
|
||||
llama.cpp runtime.
|
||||
"""Local-models dashboard routes — the desktop's window into the managed llama.cpp runtime.
|
||||
|
||||
Every payload carries plain-language, pre-formatted facts the UI shows
|
||||
verbatim (what will this model do ON THIS MACHINE, how big is the download,
|
||||
what is the runtime doing), never raw internals. Long jobs follow the repo's
|
||||
job pattern: start-POST -> {job_id} -> GET poll with byte progress.
|
||||
Every payload carries plain-language, pre-formatted facts the UI shows verbatim
|
||||
(what will this model do ON THIS MACHINE, how big is the download, what is the
|
||||
runtime doing), never raw internals. Long jobs follow the repo's job pattern:
|
||||
start-POST -> {job_id} -> GET poll with byte progress.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -97,8 +96,8 @@ def _finish(job: Dict[str, Any], detail: str) -> None:
|
||||
|
||||
def _spawn_job(job: Dict[str, Any], name: str, body: Callable[[], None], *,
|
||||
fail_msg: str | None = None, on_exit: Callable[[], None] | None = None) -> None:
|
||||
"""Run ``body`` on a daemon thread; an exception marks the job errored
|
||||
(warning ``fail_msg`` when given). ``on_exit`` always runs last."""
|
||||
"""Run ``body`` on a daemon thread; an exception marks the job errored (warning
|
||||
``fail_msg`` when given). ``on_exit`` always runs last."""
|
||||
def _run():
|
||||
try:
|
||||
body()
|
||||
@@ -115,8 +114,8 @@ def _spawn_job(job: Dict[str, Any], name: str, body: Callable[[], None], *,
|
||||
|
||||
|
||||
def _refresh_runtime(skip_msg: str) -> None:
|
||||
"""Bounce a running router so it rescans the models dir (it only scans at
|
||||
spawn). Never raises — the file operation already succeeded."""
|
||||
"""Bounce a running router so it rescans the models dir (it only scans at spawn).
|
||||
Never raises — the file operation already succeeded."""
|
||||
try:
|
||||
bootstrap.refresh_local_runtime()
|
||||
except Exception: # noqa: BLE001
|
||||
@@ -125,8 +124,8 @@ def _refresh_runtime(skip_msg: str) -> None:
|
||||
|
||||
def _router_request(endpoint: Dict[str, Any], path: str, *, timeout: float,
|
||||
payload: dict | None = None) -> Any:
|
||||
"""Call the local router (base_url minus ``/v1``) with its bearer key.
|
||||
GET (no payload) returns the parsed JSON body; POST returns None."""
|
||||
"""Call the local router (base_url minus ``/v1``) with its bearer key. GET (no
|
||||
payload) returns the parsed JSON body; POST returns None."""
|
||||
headers = {"Authorization": f"Bearer {endpoint.get('api_key', '')}"}
|
||||
data = None
|
||||
if payload is not None:
|
||||
@@ -147,9 +146,8 @@ _CHUNK = 4 << 20
|
||||
|
||||
|
||||
def _probe_range_support(url: str) -> int:
|
||||
"""Total size when the server honors Range requests, else 0. A 401/403
|
||||
means the repo is gated or the catalog names a wrong repo — raise with a
|
||||
plain-language message, not a bare status."""
|
||||
"""Total size when the server honors Range requests, else 0. A 401/403 means a
|
||||
gated repo or a wrong catalog repo — raise a plain-language message, not a bare status."""
|
||||
req = urllib.request.Request(url, headers={"Range": "bytes=0-0"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
@@ -173,8 +171,8 @@ def _model_id_for(gguf: Path) -> str:
|
||||
|
||||
|
||||
def _variant_files_on_disk(model_id: str) -> "list[Path]":
|
||||
"""Every local file belonging to a staged model: all split parts plus
|
||||
its catalog-declared assets (mmproj/draft) when present."""
|
||||
"""Every local file belonging to a staged model: all split parts plus its
|
||||
catalog-declared assets (mmproj/draft) when present."""
|
||||
files = [p for p in _models_dir().glob("*.gguf") if _model_id_for(p) == model_id]
|
||||
hit = catalog.find_entry_for_model(model_id)
|
||||
if hit is not None:
|
||||
@@ -186,15 +184,14 @@ def _variant_files_on_disk(model_id: str) -> "list[Path]":
|
||||
|
||||
def download_file(url: str, dest: Path, job: Dict[str, Any], *,
|
||||
base_done: int = 0, keep_totals: bool = False) -> None:
|
||||
"""Download url -> dest with byte progress on ``job``; ranged-parallel when
|
||||
the server supports it, single-stream otherwise. Never leaves a .part.
|
||||
"""Download url -> dest with byte progress on ``job``; ranged-parallel when the
|
||||
server supports it, single-stream otherwise. Never leaves a .part.
|
||||
|
||||
No integrity check against the CATALOG by design (catalog sizes may lag a
|
||||
re-upload); completeness is checked only against what the SERVER declared
|
||||
(range-probe total / Content-Length), so a dropped connection still errors
|
||||
instead of staging a truncated file. Multi-file variants: ``base_done``
|
||||
offsets progress onto earlier files and ``keep_totals=True`` stops the
|
||||
per-file size from overwriting the variant's total.
|
||||
Completeness is checked only against what the SERVER declared (range-probe
|
||||
total / Content-Length), never the CATALOG (its sizes may lag a re-upload), so
|
||||
a dropped connection still errors instead of staging a truncated file.
|
||||
Multi-file variants: ``base_done`` offsets progress onto earlier files and
|
||||
``keep_totals=True`` stops the per-file size from overwriting the variant's total.
|
||||
"""
|
||||
tmp = dest.with_suffix(".part")
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
@@ -274,8 +271,7 @@ def _hf_url(repo: str, path: str) -> str:
|
||||
|
||||
|
||||
def _download_plan(entry, variant) -> list:
|
||||
"""Everything a variant needs: split parts + mmproj/draft assets, as
|
||||
(url, dest, bytes) tuples."""
|
||||
"""Everything a variant needs: split parts + mmproj/draft assets, as (url, dest, bytes) tuples."""
|
||||
plan = [(_hf_url(entry.repo, a.path), _models_dir() / a.local_name, a.size_bytes) for a in variant.files]
|
||||
plan += [(_hf_url(entry.repo, a.path), bootstrap.assets_dir() / a.local_name, a.size_bytes)
|
||||
for a in (entry.mmproj, entry.draft) if a is not None]
|
||||
@@ -283,8 +279,7 @@ def _download_plan(entry, variant) -> list:
|
||||
|
||||
|
||||
def _run_download_plan(job: Dict[str, Any], plan: list, label: str) -> None:
|
||||
"""Download every missing file in ``plan``; already-present files count
|
||||
toward progress without a transfer."""
|
||||
"""Download every missing file in ``plan``; already-present files count toward progress without a transfer."""
|
||||
total = sum(p[2] for p in plan)
|
||||
_step(job, "downloading", f"{label} — {_human_gb(total)}")
|
||||
done_before = 0
|
||||
@@ -297,9 +292,9 @@ def _run_download_plan(job: Dict[str, Any], plan: list, label: str) -> None:
|
||||
|
||||
|
||||
def _engine_too_old(min_engine: str) -> bool:
|
||||
"""True when the installed llama.cpp predates a model's requirement.
|
||||
Tags are release numbers (b10362); no engine installed compares as
|
||||
too old only when the model states a requirement."""
|
||||
"""True when the installed llama.cpp predates a model's requirement. Tags are
|
||||
release numbers (b10362); no engine installed compares as too old only when
|
||||
the model states a requirement."""
|
||||
if not min_engine:
|
||||
return False
|
||||
try:
|
||||
@@ -335,8 +330,7 @@ def _resolve_backend(section: dict, requested: str | None = None) -> str:
|
||||
|
||||
|
||||
def _eligible_entries():
|
||||
"""Catalog entries this engine can activate today (engine-gated ones
|
||||
can't be the recommendation either)."""
|
||||
"""Catalog entries this engine can activate today (engine-gated ones can't be the recommendation either)."""
|
||||
return tuple(e for e in catalog.CATALOG if not _engine_too_old(e.min_engine))
|
||||
|
||||
|
||||
@@ -347,8 +341,8 @@ def _resolve_assets_or_400(tag: str, backend: str):
|
||||
|
||||
|
||||
def _start_local_server(config: dict, fail_detail: str):
|
||||
"""Start the local server (force) and return the supervisor; raise
|
||||
``fail_detail`` when neither we nor another process ended up serving."""
|
||||
"""Start the local server (force) and return the supervisor; raise ``fail_detail``
|
||||
when neither we nor another process ended up serving."""
|
||||
sup = bootstrap.ensure_local_runtime(config, force=True)
|
||||
if sup is None and _state_endpoint() is None:
|
||||
raise RuntimeError(fail_detail)
|
||||
@@ -357,10 +351,9 @@ def _start_local_server(config: dict, fail_detail: str):
|
||||
|
||||
def _ensure_server(job: Dict[str, Any], config: dict, model_id: str, *,
|
||||
fail_detail: str, skip_msg: str) -> None:
|
||||
"""Start the local server if needed and self-heal a stale router: the
|
||||
model list is spawn-only, so a server started before ``model_id``
|
||||
finished downloading can't serve it — bounce it when it doesn't know
|
||||
the model."""
|
||||
"""Start the local server if needed and self-heal a stale router: the model
|
||||
list is spawn-only, so a server started before ``model_id`` finished
|
||||
downloading can't serve it — bounce it when it doesn't know the model."""
|
||||
_step(job, "starting-server", "Starting the local server")
|
||||
sup = _start_local_server(config, fail_detail)
|
||||
if sup is not None:
|
||||
@@ -373,23 +366,20 @@ def _ensure_server(job: Dict[str, Any], config: dict, model_id: str, *,
|
||||
|
||||
|
||||
def _assign_default(job: Dict[str, Any], model_id: str) -> None:
|
||||
"""Make ``model_id`` the main model through the same machinery as
|
||||
/api/model/set (late-bound so tests can stub web_deps.late)."""
|
||||
"""Make ``model_id`` the main model through the same machinery as /api/model/set
|
||||
(late-bound so tests can stub web_deps.late)."""
|
||||
_step(job, "setting-default", "Making it your default")
|
||||
web_deps.late("_apply_model_assignment_sync")("main", "llamacpp", model_id, "", "", "")
|
||||
|
||||
|
||||
# ── status: the one call the pane opens with ─────────────────
|
||||
|
||||
|
||||
def _loaded_models(running: Dict[str, Any]) -> "tuple[Dict[str, str], Dict[str, Any]]":
|
||||
"""Resident models right now, plus how each is placed (granted window
|
||||
from the child, spill facts from the preset decision) — the difference
|
||||
between 'fast' and 'why is my CPU busy', so it must be inspectable."""
|
||||
"""Resident models right now, plus how each is placed (granted window from the
|
||||
child, spill facts from the preset decision) — the difference between 'fast'
|
||||
and 'why is my CPU busy', so it must be inspectable."""
|
||||
data = _router_request(running, "/models", timeout=3)
|
||||
# Everything resident or becoming resident: 'loading' renders as its own
|
||||
# state in the pane (a 20-GB load in flight is the most important thing
|
||||
# the pane can show).
|
||||
# Everything resident or becoming resident: 'loading' renders as its own state
|
||||
# in the pane (a 20-GB load in flight is the most important thing it can show).
|
||||
loaded = {m["id"]: m.get("status", {}).get("value", "unknown") for m in data.get("data", [])
|
||||
if m.get("status", {}).get("value") in ("loaded", "ready", "loading")}
|
||||
placement: Dict[str, Any] = {}
|
||||
@@ -427,8 +417,8 @@ def _installed_backend(tag: str) -> str | None:
|
||||
|
||||
|
||||
def _active_llamacpp_model_id() -> str | None:
|
||||
"""The active main model when it is one of ours (config authority: the
|
||||
same model.provider + model.default that /api/model/set writes)."""
|
||||
"""The active main model when it is one of ours (config authority: the same
|
||||
model.provider + model.default that /api/model/set writes)."""
|
||||
try:
|
||||
model_section = (_load_config() or {}).get("model") or {}
|
||||
if str(model_section.get("provider", "")).strip().lower() in _LLAMACPP_PROVIDERS:
|
||||
@@ -447,13 +437,11 @@ def local_models_status():
|
||||
configured_tag = section.get("tag") or binaries.default_tag()
|
||||
have = binaries.installed_tags()
|
||||
|
||||
# The tag actually serving (boot ladder: configured if installed, else
|
||||
# newest installed).
|
||||
# The tag actually serving (boot ladder: configured if installed, else newest installed).
|
||||
tag = configured_tag if configured_tag in have else (have[0] if have else configured_tag)
|
||||
|
||||
# Update pending = engine in use (enabled + something installed) and the
|
||||
# configured tag (pinned or release default) isn't on disk. The download
|
||||
# is a button click, never automatic.
|
||||
# Update pending = engine in use (enabled + something installed) and the configured
|
||||
# tag (pinned or release default) isn't on disk. The download is a button click, never automatic.
|
||||
update_available = bool(section.get("enabled") and have and configured_tag not in have)
|
||||
runtime_backend = _installed_backend(tag)
|
||||
|
||||
@@ -501,12 +489,10 @@ def _loading_progress() -> Dict[str, Any]:
|
||||
|
||||
|
||||
# ── hardware: what this machine can do ───────────────────────
|
||||
|
||||
|
||||
@router.get("/api/local-models/hardware")
|
||||
def local_models_hardware():
|
||||
"""The budget as plain facts, polled by the pane and statusbar. Sync def
|
||||
on purpose: shells out to nvidia-smi — threadpool, not loop."""
|
||||
"""The budget as plain facts, polled by the pane and statusbar. Sync def on
|
||||
purpose: shells out to nvidia-smi — threadpool, not loop."""
|
||||
budget = hardware.probe_budget()
|
||||
ram_total, ram_avail = hardware._ram_bytes()
|
||||
out = {
|
||||
@@ -514,8 +500,7 @@ def local_models_hardware():
|
||||
"ram_total_bytes": ram_total, "ram_available_bytes": ram_avail, "vram_label": _human_gb(budget.total_device_bytes),
|
||||
"gpu_name": None, "gpu_util_percent": None, "vram_used_bytes": None,
|
||||
}
|
||||
# GPU identity + live utilization (NVIDIA; other vendors degrade to None
|
||||
# and the UI hides those readouts).
|
||||
# GPU identity + live utilization (NVIDIA; other vendors degrade to None and the UI hides those readouts).
|
||||
try:
|
||||
smi_exe = hardware._nvidia_smi_path()
|
||||
smi = subprocess.run(
|
||||
@@ -551,9 +536,9 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
|
||||
"recommended_reason": recommended_reason if entry.id == recommended else None,
|
||||
"downloaded": dl is not None, "downloaded_model_id": dl.model_id if dl else None,
|
||||
"downloaded_quant": dl.quant if dl else None, "mtp": entry.mtp, "vision": entry.mmproj is not None,
|
||||
# Day-0 architectures need the llama.cpp release where their support
|
||||
# landed: True gates download/activate until the engine updates, but
|
||||
# the row still renders (visible + explained beats hidden).
|
||||
# Day-0 architectures need the llama.cpp release where their support landed:
|
||||
# True gates download/activate until the engine updates, but the row still
|
||||
# renders (visible + explained beats hidden).
|
||||
"needs_engine": _engine_too_old(entry.min_engine),
|
||||
"min_engine": entry.min_engine or None,
|
||||
}
|
||||
@@ -569,9 +554,9 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
|
||||
return row
|
||||
|
||||
variant = choice.variant
|
||||
# Same overhead the launch decision prices (runtime buffers + vision
|
||||
# projector + microbatch/MTP logits): the row must advertise the window
|
||||
# the model will actually get, not a paper number.
|
||||
# Same overhead the launch decision prices (runtime buffers + vision projector +
|
||||
# microbatch/MTP logits): the row must advertise the window the model will
|
||||
# actually get, not a paper number.
|
||||
overhead = (context_policy.RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ context_policy.ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
@@ -599,19 +584,19 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
|
||||
|
||||
@router.get("/api/local-models/catalog")
|
||||
def local_models_catalog():
|
||||
"""Every entry answers up front: how big is the download, will it fit,
|
||||
and what context/speed shape will I get. The row advertises the BEST
|
||||
build for this machine (highest quality fully on GPU at the 64K floor;
|
||||
else the smallest that works, spilled and priced). No entry is hidden;
|
||||
unaffordable models show WHY. Sync def: blocking I/O -> threadpool."""
|
||||
# Serve the in-memory catalog; a TTL-gated background fetch lands new
|
||||
# entries for the next call (day-0 models without an app release).
|
||||
"""Every entry answers up front: how big is the download, will it fit, and what
|
||||
context/speed shape will I get. The row advertises the BEST build for this
|
||||
machine (highest quality fully on GPU at the 64K floor; else the smallest that
|
||||
works, spilled and priced). No entry is hidden; unaffordable models show WHY.
|
||||
Sync def: blocking I/O -> threadpool."""
|
||||
# Serve the in-memory catalog; a TTL-gated background fetch lands new entries
|
||||
# for the next call (day-0 models without an app release).
|
||||
catalog.refresh_catalog_soon()
|
||||
# Planning budget: machine capacity, not live-free VRAM — a loaded model
|
||||
# must not make every row unaffordable.
|
||||
# Planning budget: machine capacity, not live-free VRAM — a loaded model must
|
||||
# not make every row unaffordable.
|
||||
budget = hardware.probe_budget(planning=True)
|
||||
# The reason key ships with the row so the Recommended badge's tooltip is
|
||||
# the branch that actually fired, not a re-derivation that can drift.
|
||||
# The reason key ships with the row so the Recommended badge's tooltip is the
|
||||
# branch that actually fired, not a re-derivation that can drift.
|
||||
picked = catalog.recommended_entry(budget, _eligible_entries())
|
||||
recommended = picked[0].id if picked is not None else None
|
||||
recommended_reason = picked[1] if picked is not None else None
|
||||
@@ -622,18 +607,16 @@ def local_models_catalog():
|
||||
|
||||
|
||||
# ── runtime install (job) ────────────────────────────────────
|
||||
|
||||
|
||||
class RuntimeInstallBody(BaseModel):
|
||||
backend: Optional[str] = None # None/auto -> detect
|
||||
|
||||
|
||||
def _runtime_progress_hook(job: Dict[str, Any]):
|
||||
"""Adapter: ensure_runtime_installed's progress stream -> job fields,
|
||||
throttled to ~4 updates/s. Byte counters are CUMULATIVE across the plan
|
||||
(a multi-asset engine reads as one growing download, total growing as
|
||||
each asset's size becomes known); unpack/verify keep the counters — a bar
|
||||
bouncing back to zero after the bytes finished reads as failure."""
|
||||
"""Adapter: ensure_runtime_installed's progress stream -> job fields, throttled
|
||||
to ~4 updates/s. Byte counters are CUMULATIVE across the plan (a multi-asset
|
||||
engine reads as one growing download, total growing as each asset's size becomes
|
||||
known); unpack/verify keep the counters — a bar bouncing back to zero after the
|
||||
bytes finished reads as failure."""
|
||||
state = {"last": 0.0, "banked": 0, "asset": None, "asset_total": 0}
|
||||
|
||||
def hook(stage: str, done: int, total: int, label: str) -> None:
|
||||
@@ -677,9 +660,9 @@ async def local_models_runtime_install(body: RuntimeInstallBody):
|
||||
_step(job, "downloading", f"Fetching {len(plan.assets)} package(s) for {backend}")
|
||||
binaries.ensure_runtime_installed(tag, backend, progress=_runtime_progress_hook(job))
|
||||
|
||||
# Engine update path: a server already running on an older tag moves
|
||||
# to the new one now — the click was the consent. Fresh installs (no
|
||||
# server) skip this; Use/boot handles their start.
|
||||
# Engine update path: a server already running on an older tag moves to the
|
||||
# new one now — the click was the consent. Fresh installs (no server) skip
|
||||
# this; Use/boot handles their start.
|
||||
restarted = False
|
||||
try:
|
||||
if bootstrap.get_supervisor() is not None and previous and tag not in previous:
|
||||
@@ -691,8 +674,8 @@ async def local_models_runtime_install(body: RuntimeInstallBody):
|
||||
# The new build is installed either way; the next boot serves it.
|
||||
logger.warning("post-update restart skipped: %s", exc)
|
||||
|
||||
# N-1 retention, only after the new tag verified: keep it and the
|
||||
# newest previous build as the rollback pin target.
|
||||
# N-1 retention, only after the new tag verified: keep it and the newest
|
||||
# previous build as the rollback pin target.
|
||||
try:
|
||||
binaries.prune_old_tags([tag] + [t for t in previous if t != tag][:1])
|
||||
except Exception as exc: # noqa: BLE001
|
||||
@@ -705,15 +688,13 @@ async def local_models_runtime_install(body: RuntimeInstallBody):
|
||||
|
||||
|
||||
# ── model download (job with byte progress) ──────────────────
|
||||
|
||||
|
||||
class ModelDownloadBody(BaseModel):
|
||||
model_id: str
|
||||
|
||||
|
||||
def _download_job(job: Dict[str, Any], body: Callable[[], None], name: str, label: str, fail_msg: str | None = None) -> None:
|
||||
"""Spawn a download job: ``body`` fetches, then the job finishes as
|
||||
"<label> ready" and the router is bounced to pick the file up."""
|
||||
"""Spawn a download job: ``body`` fetches, then the job finishes as "<label> ready"
|
||||
and the router is bounced to pick the file up."""
|
||||
def _run():
|
||||
body()
|
||||
_finish(job, f"{label} ready")
|
||||
@@ -724,8 +705,7 @@ def _download_job(job: Dict[str, Any], body: Callable[[], None], name: str, labe
|
||||
|
||||
@router.post("/api/local-models/download")
|
||||
async def local_models_download(body: ModelDownloadBody):
|
||||
"""Accepts either a family id (downloads this machine's selected
|
||||
variant) or an exact variant model_id."""
|
||||
"""Accepts either a family id (downloads this machine's selected variant) or an exact variant model_id."""
|
||||
entry = catalog.catalog_by_id().get(body.model_id)
|
||||
variant = None
|
||||
if entry is not None:
|
||||
@@ -758,16 +738,16 @@ async def local_models_download(body: ModelDownloadBody):
|
||||
|
||||
@router.delete("/api/local-models/models/{model_id}")
|
||||
async def local_models_delete(model_id: str):
|
||||
"""Remove every split part plus private assets, then bounce the router off
|
||||
the request thread (deleting the active file mid-serve is exactly the
|
||||
stale state the refresh exists for)."""
|
||||
"""Remove every split part plus private assets, then bounce the router off the
|
||||
request thread (deleting the active file mid-serve is exactly the stale state
|
||||
the refresh exists for)."""
|
||||
files = _variant_files_on_disk(model_id)
|
||||
if not files:
|
||||
raise HTTPException(status_code=404, detail="model not found")
|
||||
for path in files:
|
||||
path.unlink(missing_ok=True)
|
||||
# Growth state dies with the model: a re-download starts back at its
|
||||
# zero-spill window instead of inheriting a stale grown one.
|
||||
# Growth state dies with the model: a re-download starts back at its zero-spill
|
||||
# window instead of inheriting a stale grown one.
|
||||
try:
|
||||
growth.clear_window_override(model_id)
|
||||
except Exception: # noqa: BLE001
|
||||
@@ -779,21 +759,19 @@ async def local_models_delete(model_id: str):
|
||||
|
||||
|
||||
# ── quickstart: one click from nothing to a working default ──
|
||||
|
||||
|
||||
class QuickstartBody(BaseModel):
|
||||
model_id: str | None = None # default: the catalog's recommended entry
|
||||
|
||||
|
||||
# One quickstart at a time: the job sequences installs, downloads, a
|
||||
# server bounce, and a config write — two racing runs would interleave
|
||||
# all four. Held for the job's lifetime, released in the worker.
|
||||
# One quickstart at a time: the job sequences installs, downloads, a server bounce,
|
||||
# and a config write — two racing runs would interleave all four. Held for the
|
||||
# job's lifetime, released in the worker.
|
||||
_QUICKSTART_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def _quickstart_target(body: QuickstartBody, budget):
|
||||
"""(entry, variant) to set up: explicit id, else this machine's
|
||||
recommendation, else the first catalog entry this machine can serve."""
|
||||
"""(entry, variant) to set up: explicit id, else this machine's recommendation,
|
||||
else the first catalog entry this machine can serve."""
|
||||
if body.model_id:
|
||||
entry = catalog.catalog_by_id().get(body.model_id)
|
||||
if entry is None:
|
||||
@@ -813,12 +791,12 @@ def _quickstart_target(body: QuickstartBody, budget):
|
||||
|
||||
@router.post("/api/local-models/quickstart")
|
||||
async def local_models_quickstart(body: QuickstartBody):
|
||||
"""One job: install the runtime (if missing), download this machine's
|
||||
build of the recommended model (if missing), make it the default. Each
|
||||
leg is the same code the individual routes run, so 'Configure' and
|
||||
quickstart can never disagree. Preflight rejects (no servable entry,
|
||||
engine too old) fail the POST synchronously so the button can explain
|
||||
itself; everything slow runs in the job with phase/byte progress."""
|
||||
"""One job: install the runtime (if missing), download this machine's build of
|
||||
the recommended model (if missing), make it the default. Each leg is the same
|
||||
code the individual routes run, so 'Configure' and quickstart can never
|
||||
disagree. Preflight rejects (no servable entry, engine too old) fail the POST
|
||||
synchronously so the button can explain itself; everything slow runs in the
|
||||
job with phase/byte progress."""
|
||||
entry, variant = _quickstart_target(body, hardware.probe_budget(planning=True))
|
||||
|
||||
section = _runtime_section()
|
||||
@@ -844,8 +822,8 @@ async def local_models_quickstart(body: QuickstartBody):
|
||||
binaries.ensure_runtime_installed(tag, backend, progress=_runtime_progress_hook(job))
|
||||
|
||||
if need_download:
|
||||
# The runtime leg repurposed the byte counters for its own
|
||||
# stages — reset them to the model plan before download.
|
||||
# The runtime leg repurposed the byte counters for its own stages —
|
||||
# reset them to the model plan before download.
|
||||
job["done_bytes"] = 0
|
||||
job["total_bytes"] = download_bytes
|
||||
_run_download_plan(job, download_plan, entry.display_name)
|
||||
@@ -864,8 +842,6 @@ async def local_models_quickstart(body: QuickstartBody):
|
||||
|
||||
|
||||
# ── server lifecycle: turn the engine on/off ─────────────────
|
||||
|
||||
|
||||
class ServerActionBody(BaseModel):
|
||||
action: str # "stop" | "start"
|
||||
|
||||
@@ -874,8 +850,8 @@ def _stop_server() -> None:
|
||||
if bootstrap.get_supervisor() is not None:
|
||||
bootstrap.shutdown_local_runtime()
|
||||
elif _state_endpoint() is not None:
|
||||
# Server owned by another process (or an orphan): best-effort
|
||||
# terminate via the state file's pid, then clear the state.
|
||||
# Server owned by another process (or an orphan): best-effort terminate
|
||||
# via the state file's pid, then clear the state.
|
||||
try:
|
||||
import psutil # type: ignore
|
||||
|
||||
@@ -898,9 +874,9 @@ _SERVER_ACTIONS = {"stop": _stop_server, "start": _start_server}
|
||||
|
||||
@router.post("/api/local-models/server")
|
||||
async def local_models_server(body: ServerActionBody):
|
||||
"""Turn the local engine off (stop the server, free ALL GPU memory,
|
||||
disable auto-start) or back on. Unlike per-model eject the off switch IS
|
||||
durable: the user said off, so boots stay off until they say on."""
|
||||
"""Turn the local engine off (stop the server, free ALL GPU memory, disable
|
||||
auto-start) or back on. Unlike per-model eject the off switch IS durable: the
|
||||
user said off, so boots stay off until they say on."""
|
||||
action = (body.action or "").strip().lower()
|
||||
if action not in _SERVER_ACTIONS:
|
||||
raise HTTPException(status_code=400, detail="action must be 'stop' or 'start'")
|
||||
@@ -910,25 +886,23 @@ async def local_models_server(body: ServerActionBody):
|
||||
|
||||
|
||||
# ── activate: make a downloaded model THE model ──────────────
|
||||
|
||||
|
||||
class ModelEjectBody(BaseModel):
|
||||
model_id: str
|
||||
|
||||
|
||||
@router.post("/api/local-models/eject")
|
||||
def local_models_eject(body: ModelEjectBody):
|
||||
"""Free a loaded model's GPU memory now; only demand (the next message)
|
||||
reloads it — residency v2 has no automatic loading anywhere. Sync def:
|
||||
the fallback path blocks on a 120s urlopen — threadpool, never the loop."""
|
||||
"""Free a loaded model's GPU memory now; only demand (the next message) reloads
|
||||
it — residency v2 has no automatic loading anywhere. Sync def: the fallback
|
||||
path blocks on a 120s urlopen — threadpool, never the loop."""
|
||||
sup = bootstrap.get_supervisor()
|
||||
if sup is not None:
|
||||
with _http_error(502):
|
||||
sup.unload_model(body.model_id)
|
||||
return {"ok": True}
|
||||
|
||||
# Server owned by another process (or state-file only): drive the
|
||||
# router directly with the persisted endpoint.
|
||||
# Server owned by another process (or state-file only): drive the router
|
||||
# directly with the persisted endpoint.
|
||||
endpoint = _state_endpoint()
|
||||
if endpoint is None:
|
||||
raise HTTPException(status_code=409, detail="local server is not running")
|
||||
@@ -943,10 +917,10 @@ class ModelActivateBody(BaseModel):
|
||||
|
||||
@router.post("/api/local-models/activate")
|
||||
async def local_models_activate(body: ModelActivateBody):
|
||||
"""Make a downloaded model the default for new chats: a config write via
|
||||
the same machinery as /api/model/set plus making sure the server is up.
|
||||
NO model loading (residency v2: models load on first inference; an empty
|
||||
router costs nothing). Kept as a job for UI continuity."""
|
||||
"""Make a downloaded model the default for new chats: a config write via the
|
||||
same machinery as /api/model/set plus making sure the server is up. NO model
|
||||
loading (residency v2: models load on first inference; an empty router costs
|
||||
nothing). Kept as a job for UI continuity."""
|
||||
# Split variants stage under their first part — resolve like the other routes.
|
||||
if body.model_id not in bootstrap.staged_model_ids():
|
||||
raise HTTPException(status_code=404, detail=f"{body.model_id} is not downloaded")
|
||||
@@ -968,12 +942,10 @@ async def local_models_activate(body: ModelActivateBody):
|
||||
|
||||
|
||||
# ── job polling ──────────────────────────────────────────────
|
||||
|
||||
|
||||
@router.get("/api/local-models/jobs")
|
||||
async def local_models_jobs():
|
||||
"""All recent jobs, running first — the pane and the app-level poller
|
||||
rediscover in-flight work here after a remount or app restart."""
|
||||
"""All recent jobs, running first — the pane and the app-level poller rediscover
|
||||
in-flight work here after a remount or app restart."""
|
||||
with _JOBS_LOCK:
|
||||
jobs = sorted(_JOBS.values(), key=lambda j: (j["status"] != "running", -j["started_at"]))
|
||||
return {"jobs": [_job_view(job) for job in jobs[:20]]}
|
||||
@@ -989,12 +961,10 @@ async def local_models_job(job_id: str):
|
||||
|
||||
|
||||
# ── Hugging Face browser: search, repo files, arbitrary download ─
|
||||
|
||||
|
||||
@router.get("/api/local-models/search")
|
||||
async def local_models_search(q: str, limit: int = 20):
|
||||
"""Full-text HF search over GGUF models — the firehose behind the curated
|
||||
catalog; per-quant fit pills come from the repo-files call."""
|
||||
"""Full-text HF search over GGUF models — the firehose behind the curated catalog;
|
||||
per-quant fit pills come from the repo-files call."""
|
||||
if not q.strip():
|
||||
return {"hits": []}
|
||||
with _http_error(502, "Hugging Face search unavailable: "):
|
||||
@@ -1003,8 +973,8 @@ async def local_models_search(q: str, limit: int = 20):
|
||||
|
||||
@router.get("/api/local-models/search/files")
|
||||
async def local_models_search_files(repo: str):
|
||||
"""Servable GGUFs in one HF repo with a rough pre-download fit verdict per
|
||||
quant (file size + conservative fill-ins; the GGUF header refines it)."""
|
||||
"""Servable GGUFs in one HF repo with a rough pre-download fit verdict per quant
|
||||
(file size + conservative fill-ins; the GGUF header refines it)."""
|
||||
with _http_error(502, f"Could not list {repo}: "):
|
||||
groups = await run_in_threadpool(hf_browse.priced_repo_files, repo, hardware.probe_budget(planning=True))
|
||||
return {"files": [dict(g.__dict__, paths=list(g.paths)) for g in groups]}
|
||||
@@ -1017,10 +987,10 @@ class BrowsedDownloadBody(BaseModel):
|
||||
|
||||
@router.post("/api/local-models/download-browsed")
|
||||
async def local_models_download_browsed(body: BrowsedDownloadBody):
|
||||
"""Download an arbitrary HF GGUF into the managed models dir. Once landed
|
||||
it is a normal staged model (the post-download bounce regenerates presets
|
||||
from its real header); with no catalog entry it serves 'unverified',
|
||||
capabilities answered from the live server only."""
|
||||
"""Download an arbitrary HF GGUF into the managed models dir. Once landed it is
|
||||
a normal staged model (the post-download bounce regenerates presets from its
|
||||
real header); with no catalog entry it serves 'unverified', capabilities
|
||||
answered from the live server only."""
|
||||
paths = [p for p in (body.paths or []) if p.lower().endswith(".gguf")]
|
||||
if not paths:
|
||||
raise HTTPException(status_code=422, detail="no .gguf files given")
|
||||
@@ -1052,9 +1022,9 @@ class SideloadBody(BaseModel):
|
||||
|
||||
@router.post("/api/local-models/sideload")
|
||||
async def local_models_sideload(body: SideloadBody):
|
||||
"""Register a GGUF already on this machine: link it into the managed
|
||||
models dir (copy only when linking is impossible) and bounce the router.
|
||||
The original stays put; delete-from-Hermes removes only our link."""
|
||||
"""Register a GGUF already on this machine: link it into the managed models dir
|
||||
(copy only when linking is impossible) and bounce the router. The original
|
||||
stays put; delete-from-Hermes removes only our link."""
|
||||
src = Path(body.path)
|
||||
if not src.is_file() or src.suffix.lower() != ".gguf":
|
||||
raise HTTPException(status_code=422, detail="Pick a .gguf model file")
|
||||
|
||||
@@ -29,13 +29,11 @@ _AUX_TASK_SLOTS = LateState("_AUX_TASK_SLOTS")
|
||||
|
||||
|
||||
_EMPTY_MODEL_INFO: dict = {
|
||||
"model": "",
|
||||
"provider": "",
|
||||
"auto_context_length": 0,
|
||||
"config_context_length": 0,
|
||||
"effective_context_length": 0,
|
||||
"capabilities": {},
|
||||
"model": "", "provider": "", "auto_context_length": 0, "config_context_length": 0,
|
||||
"effective_context_length": 0, "capabilities": {},
|
||||
}
|
||||
_CAPABILITY_FIELDS = ("supports_tools", "supports_vision", "supports_reasoning", "context_window",
|
||||
"max_output_tokens", "model_family")
|
||||
|
||||
|
||||
def _main_model_fields(model_cfg) -> tuple[str, str]:
|
||||
@@ -82,21 +80,12 @@ def get_model_info(profile: Optional[str] = None):
|
||||
from agent.models_dev import get_model_capabilities
|
||||
mc = get_model_capabilities(provider=provider, model=model_name)
|
||||
if mc is not None:
|
||||
caps = {
|
||||
"supports_tools": mc.supports_tools,
|
||||
"supports_vision": mc.supports_vision,
|
||||
"supports_reasoning": mc.supports_reasoning,
|
||||
"context_window": mc.context_window,
|
||||
"max_output_tokens": mc.max_output_tokens,
|
||||
"model_family": mc.model_family,
|
||||
}
|
||||
caps = {name: getattr(mc, name) for name in _CAPABILITY_FIELDS}
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return {
|
||||
"model": model_name,
|
||||
"provider": provider,
|
||||
"auto_context_length": auto_ctx,
|
||||
"model": model_name, "provider": provider, "auto_context_length": auto_ctx,
|
||||
"config_context_length": config_ctx_int,
|
||||
"effective_context_length": config_ctx_int or auto_ctx, # what the agent actually uses
|
||||
"capabilities": caps,
|
||||
@@ -151,22 +140,12 @@ async def get_model_options(
|
||||
|
||||
|
||||
def _nous_recommended_default() -> dict:
|
||||
from hermes_cli.models import (
|
||||
get_curated_nous_model_ids,
|
||||
get_pricing_for_provider,
|
||||
check_nous_free_tier,
|
||||
nous_policy_allowed_ids,
|
||||
partition_nous_models_by_tier,
|
||||
pick_silent_default_model,
|
||||
restrict_to_nous_policy,
|
||||
union_with_portal_free_recommendations,
|
||||
union_with_portal_paid_recommendations,
|
||||
)
|
||||
from hermes_cli import models as m
|
||||
from hermes_cli.auth import get_provider_auth_state
|
||||
|
||||
model_ids = get_curated_nous_model_ids()
|
||||
pricing = get_pricing_for_provider("nous") or {}
|
||||
free_tier = check_nous_free_tier(force_fresh=True)
|
||||
model_ids = m.get_curated_nous_model_ids()
|
||||
pricing = m.get_pricing_for_provider("nous") or {}
|
||||
free_tier = m.check_nous_free_tier(force_fresh=True)
|
||||
|
||||
try:
|
||||
portal_url = (get_provider_auth_state("nous") or {}).get("portal_base_url", "") or ""
|
||||
@@ -176,14 +155,14 @@ def _nous_recommended_default() -> dict:
|
||||
# This endpoint picks the model a user lands on without choosing it, so an
|
||||
# unreachable one here is worse than in a picker. Narrow to policy before
|
||||
# the tier split, so a rescued id still has to pass the free/paid predicate.
|
||||
policy_allowed = nous_policy_allowed_ids()
|
||||
union = union_with_portal_free_recommendations if free_tier else union_with_portal_paid_recommendations
|
||||
policy_allowed = m.nous_policy_allowed_ids()
|
||||
union = m.union_with_portal_free_recommendations if free_tier else m.union_with_portal_paid_recommendations
|
||||
model_ids, pricing = union(model_ids, pricing, portal_url)
|
||||
model_ids = restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
|
||||
model_ids = m.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
|
||||
if free_tier:
|
||||
model_ids, _unavailable = partition_nous_models_by_tier(model_ids, pricing, free_tier=True)
|
||||
model_ids, _unavailable = m.partition_nous_models_by_tier(model_ids, pricing, free_tier=True)
|
||||
|
||||
model = pick_silent_default_model(model_ids, provider="nous")
|
||||
model = m.pick_silent_default_model(model_ids, provider="nous")
|
||||
return {"provider": "nous", "model": model, "free_tier": bool(free_tier)}
|
||||
|
||||
|
||||
@@ -242,10 +221,8 @@ def get_auxiliary_models(profile: Optional[str] = None):
|
||||
for slot in _AUX_TASK_SLOTS:
|
||||
slot_cfg = aux_cfg.get(slot, {}) if isinstance(aux_cfg.get(slot), dict) else {}
|
||||
tasks.append({
|
||||
"task": slot,
|
||||
"provider": str(slot_cfg.get("provider", "auto") or "auto"),
|
||||
"model": str(slot_cfg.get("model", "") or ""),
|
||||
"base_url": str(slot_cfg.get("base_url", "") or ""),
|
||||
"task": slot, "provider": str(slot_cfg.get("provider", "auto") or "auto"),
|
||||
"model": str(slot_cfg.get("model", "") or ""), "base_url": str(slot_cfg.get("base_url", "") or ""),
|
||||
})
|
||||
|
||||
model, provider = _main_model_fields(cfg.get("model", {}))
|
||||
@@ -323,12 +300,9 @@ async def set_model_assignment(body: ModelAssignment, profile: Optional[str] = N
|
||||
Writes ``~/.hermes/config.yaml`` — applies to **new** sessions only; a
|
||||
running chat PTY hot-swaps via the ``/model`` slash command instead.
|
||||
"""
|
||||
scope = (body.scope or "").strip().lower()
|
||||
provider = (body.provider or "").strip()
|
||||
model = (body.model or "").strip()
|
||||
task = (body.task or "").strip().lower()
|
||||
base_url = (body.base_url or "").strip()
|
||||
api_key = (body.api_key or "").strip()
|
||||
scope, task = (body.scope or "").strip().lower(), (body.task or "").strip().lower()
|
||||
provider, model = (body.provider or "").strip(), (body.model or "").strip()
|
||||
base_url, api_key = (body.base_url or "").strip(), (body.api_key or "").strip()
|
||||
|
||||
if scope not in {"main", "auxiliary"}:
|
||||
raise HTTPException(status_code=400, detail="scope must be 'main' or 'auxiliary'")
|
||||
@@ -350,14 +324,8 @@ async def set_model_assignment(body: ModelAssignment, profile: Optional[str] = N
|
||||
except Exception:
|
||||
warning = None
|
||||
if warning is not None:
|
||||
return {
|
||||
"ok": False,
|
||||
"scope": scope,
|
||||
"provider": provider,
|
||||
"model": model,
|
||||
"confirm_required": True,
|
||||
"confirm_message": warning.message,
|
||||
}
|
||||
return {"ok": False, "scope": scope, "provider": provider, "model": model,
|
||||
"confirm_required": True, "confirm_message": warning.message}
|
||||
|
||||
def _apply_assignment():
|
||||
with _profile_scope(body.profile or profile):
|
||||
|
||||
Reference in New Issue
Block a user