Files
hermes-agent/tests/hermes_cli/test_local_server_lifecycle.py
T
emozilla 43e67d872f feat: local models — managed llama.cpp runtime with one-click desktop setup
Run models locally as a first-class provider. The CLI grows a managed
llama.cpp runtime (engine install, model download, server supervision);
the desktop app grows the full setup and management story on top of it.
GUI surfaces ship behind the desktop --local launch flag (hermes desktop
--local, or the flag on the packaged app); backend routes and the CLI
are always live.

Runtime (hermes_cli/local_runtime/):
- curated GGUF catalog with per-machine variant selection: hardware
  probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice
  by context window
- derived recommendation: quality-ranked picks gated by a predicted
  decode-speed floor, bandwidth-aware on unified memory; the decision
  table is pinned as a test (pick AND reason per memory class), and the
  Recommended badge explains its pick in a tooltip fed by the resolver's
  actual branch
- engine install + model download with resumable split parts, cumulative
  plan-level progress, and staged-model integrity (a split GGUF counts
  only when every part is present)
- server supervision: spawn/adopt/stop, router mode with per-model load
  progress relayed over SSE, abandoned-request cleanup

Desktop:
- Settings -> Providers -> Local models: one-click quickstart (install
  engine, download the recommended model, boot) plus per-model download/
  activate/eject, fit-ranked catalog with context pills
- model pickers (composer dropdown + Cmd+K) show staged local models,
  in-flight downloads as live progress rows, and load-into-memory bars
- local-setup campaign tip for eligible hardware; System resources
  statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends
- friendly dead-server errors, and failed agent builds retry on the next
  send instead of wedging the session

Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
2026-09-01 16:01:53 -04:00

115 lines
4.0 KiB
Python

"""Server on/off lifecycle route (round-9 feedback: 'we should be able to
completely turn off the local engine'). Contract: stop tears the server
down AND persists enabled=false (durable, unlike eject); start persists
enabled=true and boots."""
from __future__ import annotations
import pytest
from fastapi.testclient import TestClient
@pytest.fixture
def client(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli import web_server
test_client = TestClient(web_server.app)
token = getattr(web_server, "_SESSION_TOKEN", "")
if token:
test_client.headers["Authorization"] = f"Bearer {token}"
return test_client
def test_stop_disables_and_tears_down(client, monkeypatch):
stopped = {"called": False}
def _shutdown():
stopped["called"] = True
class _FakeSup:
pass
monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.get_supervisor",
lambda: _FakeSup())
monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.shutdown_local_runtime",
_shutdown)
r = client.post("/api/local-models/server", json={"action": "stop"})
assert r.status_code == 200
assert stopped["called"] is True
from hermes_cli.config import load_config
assert load_config()["local_runtime"]["enabled"] is False
def test_start_enables_and_boots(client, monkeypatch):
booted = {"called": False}
class _FakeSup:
base_url = "http://127.0.0.1:18434/v1"
def _ensure(config, force=False):
booted["called"] = True
assert force is True
return _FakeSup()
monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
_ensure)
r = client.post("/api/local-models/server", json={"action": "start"})
assert r.status_code == 200
assert booted["called"] is True
from hermes_cli.config import load_config
assert load_config()["local_runtime"]["enabled"] is True
def test_bogus_action_rejected(client):
r = client.post("/api/local-models/server", json={"action": "reboot"})
assert r.status_code == 400
def test_status_reports_loaded_models_from_live_router(client, monkeypatch):
"""Round-11 regression: the loaded-models read inside the status route
raised NameError (missing json import), the blanket except swallowed it,
and {} shipped as truth — 'Not in memory' on a machine with 30 GB of
VRAM in use. This test exercises the REAL route against a stub router
and demands the loaded set comes through."""
import http.server
import json as _json
import threading
class _Router(http.server.BaseHTTPRequestHandler):
def do_GET(self):
body = _json.dumps({"data": [
{"id": "m-loaded", "status": {"value": "loaded"}},
{"id": "m-loading", "status": {"value": "loading"}},
{"id": "m-cold", "status": {"value": "unloaded"}},
]}).encode()
self.send_response(200)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def log_message(self, *a):
pass
server = http.server.HTTPServer(("127.0.0.1", 0), _Router)
threading.Thread(target=server.serve_forever, daemon=True).start()
try:
port = server.server_address[1]
# Patch the ROUTE's binding: local_models binds _state_endpoint via
# from-import at module load, so patching the endpoint module's
# attribute never reaches the name the route actually calls.
monkeypatch.setattr(
"hermes_cli.web_routers.local_models._state_endpoint",
lambda: {"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "k"})
payload = client.get("/api/local-models/status").json()
assert payload["server_running"] is True
assert payload["loaded_models"] == {"m-loaded": "loaded", "m-loading": "loading"}
finally:
server.shutdown()