43e67d872f
Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
101 lines
3.5 KiB
Python
101 lines
3.5 KiB
Python
"""A failed agent build must not wedge the session.
|
|
|
|
The first send with the local server off fails agent init and stores
|
|
``agent_error`` with ``agent_ready`` set. Before the fix, prompt.submit's
|
|
build kick was a no-op on that state (``agent_build_started`` stayed True),
|
|
so every later send — including the error card's Retry — replayed the stored
|
|
failure even after the server came back; only NEW sessions worked. The fix
|
|
routes prompt.submit through ``_restart_completed_failed_agent_build`` first,
|
|
which clears one completed failed generation and rebuilds with fresh
|
|
provider resolution.
|
|
|
|
These tests pin the restart helper's contract directly: it is the seam the
|
|
prompt path now calls, and its answer decides retry-vs-replay.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import threading
|
|
|
|
from tui_gateway import server
|
|
|
|
|
|
def failed_session(tmp_path, sid: str) -> dict:
|
|
"""A session record whose deferred agent build COMPLETED in failure."""
|
|
ready = threading.Event()
|
|
ready.set()
|
|
session = {
|
|
"agent": None,
|
|
"agent_ready": ready,
|
|
"agent_build_started": True,
|
|
"agent_error": "The local model server is turned off.",
|
|
"cwd": str(tmp_path),
|
|
"history": [],
|
|
"history_lock": threading.RLock(),
|
|
"profile_home": str(tmp_path),
|
|
"running": False,
|
|
"session_key": sid,
|
|
}
|
|
server._sessions[sid] = session
|
|
return session
|
|
|
|
|
|
def test_restart_clears_failed_generation_and_rebuilds(tmp_path, monkeypatch):
|
|
sid = "wedged-session"
|
|
session = failed_session(tmp_path, sid)
|
|
failed_ready = session["agent_ready"]
|
|
|
|
rebuilt = []
|
|
monkeypatch.setattr(server, "_start_agent_build",
|
|
lambda s, sess: rebuilt.append(s))
|
|
|
|
try:
|
|
assert server._restart_completed_failed_agent_build(
|
|
sid, session, failed_ready) is True
|
|
|
|
# The failed generation is gone: error cleared, fresh unset ready
|
|
# event, build flag dropped — the next build starts from zero.
|
|
assert session["agent_error"] is None
|
|
assert session["agent_ready"] is not failed_ready
|
|
assert not session["agent_ready"].is_set()
|
|
assert "agent_build_started" not in session
|
|
assert rebuilt == [sid]
|
|
finally:
|
|
server._sessions.pop(sid, None)
|
|
|
|
|
|
def test_restart_declines_every_non_failure_state(tmp_path, monkeypatch):
|
|
"""False (caller falls through to the normal build) when there is no
|
|
completed failure: healthy agent, build still in flight, or no error."""
|
|
sid = "healthy-session"
|
|
session = failed_session(tmp_path, sid)
|
|
monkeypatch.setattr(server, "_start_agent_build",
|
|
lambda s, sess: None)
|
|
|
|
try:
|
|
# Build still in flight: ready not set.
|
|
in_flight = threading.Event()
|
|
session["agent_ready"] = in_flight
|
|
assert server._restart_completed_failed_agent_build(
|
|
sid, session, in_flight) is False
|
|
|
|
# No error recorded.
|
|
done = threading.Event()
|
|
done.set()
|
|
session["agent_ready"] = done
|
|
session["agent_error"] = None
|
|
assert server._restart_completed_failed_agent_build(
|
|
sid, session, done) is False
|
|
|
|
# Agent actually built.
|
|
session["agent_error"] = "stale text"
|
|
session["agent"] = object()
|
|
assert server._restart_completed_failed_agent_build(
|
|
sid, session, done) is False
|
|
|
|
# No ready event at all.
|
|
assert server._restart_completed_failed_agent_build(
|
|
sid, session, None) is False
|
|
finally:
|
|
server._sessions.pop(sid, None)
|