Files
hermes-agent/tests/hermes_cli/test_local_runtime_picker_row.py
T
emozilla 43e67d872f feat: local models — managed llama.cpp runtime with one-click desktop setup
Run models locally as a first-class provider. The CLI grows a managed
llama.cpp runtime (engine install, model download, server supervision);
the desktop app grows the full setup and management story on top of it.
GUI surfaces ship behind the desktop --local launch flag (hermes desktop
--local, or the flag on the packaged app); backend routes and the CLI
are always live.

Runtime (hermes_cli/local_runtime/):
- curated GGUF catalog with per-machine variant selection: hardware
  probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice
  by context window
- derived recommendation: quality-ranked picks gated by a predicted
  decode-speed floor, bandwidth-aware on unified memory; the decision
  table is pinned as a test (pick AND reason per memory class), and the
  Recommended badge explains its pick in a tooltip fed by the resolver's
  actual branch
- engine install + model download with resumable split parts, cumulative
  plan-level progress, and staged-model integrity (a split GGUF counts
  only when every part is present)
- server supervision: spawn/adopt/stop, router mode with per-model load
  progress relayed over SSE, abandoned-request cleanup

Desktop:
- Settings -> Providers -> Local models: one-click quickstart (install
  engine, download the recommended model, boot) plus per-model download/
  activate/eject, fit-ranked catalog with context pills
- model pickers (composer dropdown + Cmd+K) show staged local models,
  in-flight downloads as live progress rows, and load-into-memory bars
- local-setup campaign tip for eligible hardware; System resources
  statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends
- friendly dead-server errors, and failed agent builds retry on the next
  send instead of wedging the session

Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
2026-09-01 16:01:53 -04:00

91 lines
3.3 KiB
Python

"""The llamacpp provider row in the model picker payload.
Contract: staged local GGUFs appear as a selectable provider row in
build_models_payload — the same payload /api/model/options and the desktop
picker consume — whenever models are staged, without any credential."""
from __future__ import annotations
import pytest
@pytest.fixture
def hermes_home(tmp_path, monkeypatch):
home = tmp_path / ".hermes"
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
return home
def _stage(home, *names):
mdir = home / "models"
mdir.mkdir(exist_ok=True)
for name in names:
(mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 64)
def test_no_staged_models_no_row(hermes_home):
from hermes_cli.inventory import _local_runtime_row, load_picker_context
assert _local_runtime_row(load_picker_context()) is None
def test_staged_models_make_a_selectable_row(hermes_home):
from hermes_cli.inventory import _local_runtime_row, load_picker_context
_stage(hermes_home, "Qwen3-4B-Instruct-2507-UD-Q8_K_XL", "Some-Other-Model")
row = _local_runtime_row(load_picker_context())
assert row is not None
assert row["slug"] == "llamacpp"
assert row["authenticated"] is True
assert "Qwen3-4B-Instruct-2507-UD-Q8_K_XL" in row["models"]
assert row["total_models"] == 2
def test_row_marks_current_when_config_points_at_llamacpp(hermes_home):
from hermes_cli.inventory import _local_runtime_row, load_picker_context
_stage(hermes_home, "M")
ctx = load_picker_context().with_overrides(current_provider="llamacpp")
row = _local_runtime_row(ctx)
assert row is not None and row["is_current"] is True
def test_full_payload_includes_local_row(hermes_home):
"""Through the REAL payload builder — the shape the desktop picker eats."""
from hermes_cli.inventory import build_models_payload, load_picker_context
_stage(hermes_home, "Local-Model-X")
payload = build_models_payload(
load_picker_context(),
probe_custom_providers=False,
probe_current_custom_provider=False,
)
slugs = [p["slug"] for p in payload["providers"]]
assert "llamacpp" in slugs
row = payload["providers"][slugs.index("llamacpp")]
assert row["models"] == ["Local-Model-X"]
def test_explicit_only_filter_keeps_local_row_on_any_profile(hermes_home):
"""The desktop dropdown requests explicit_only=True, and the local row
has no config credential by design (credential is reachability). The
filter must treat staged models as explicit configuration — otherwise
the row only survives on the profile whose config points at llamacpp,
and every other profile's dropdown silently loses local models."""
from hermes_cli.inventory import _filter_explicit_provider_rows, _local_runtime_row, load_picker_context
_stage(hermes_home, "Qwen3.8-27B-UD-Q5_K_XL")
ctx = load_picker_context()
row = _local_runtime_row(ctx)
assert row is not None
# Simulate a profile whose current provider is a cloud one (the normal
# profile's shape): explicit-only filtering must keep the local row.
import dataclasses
ctx = dataclasses.replace(ctx, current_provider="anthropic")
kept = _filter_explicit_provider_rows([row], ctx)
assert kept, "explicit-only filter dropped the local-runtime row"
assert kept[0]["slug"] == "llamacpp"