"""Endpoint resolution for llamacpp-alias requests (provider integration). The seam between the existing provider mechanism and the managed runtime: ``provider: llamacpp`` with no explicit base_url resolves, in order, to """ from __future__ import annotations import json import logging import threading import time import urllib.error import urllib.request LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"}) logger = logging.getLogger(__name__) def _pid_alive(pid: int) -> bool: """Liveness for the state file's supervisor-child pid. psutil when available; otherwise fall back to True (optimistic) — on Windows ``os.kill(pid, 0)`` TERMINATES the process, so it must never be used as a probe (windows-git-bash interop pitfall). """ if not pid or pid < 0: return False try: import psutil # type: ignore return psutil.pid_exists(pid) except Exception: # noqa: BLE001 return True def _state_endpoint() -> dict | None: from hermes_cli.local_runtime.supervisor import state_path path = state_path() if not path.exists(): return None try: state = json.loads(path.read_text(encoding="utf-8")) except (json.JSONDecodeError, OSError): return None base_url = state.get("base_url", "") if not base_url: return None endpoint = {"base_url": base_url, "api_key": state.get("api_key", "")} # Ownership proof: the stable port means a SECOND install (different # HERMES_HOME — a scratch profile, say) can own 127.0.0.1:18434 with a # different api key while this install's state file still points there. # /health is a public route, so it answers 200 for ANYONE's server — # trusting it alone sent every chat request and the load-progress # watcher at a server that 401s our key, silently. The recorded # supervisor pid is the tiebreaker: health-200 from a server whose # recorded child is DEAD is someone else's server, never a starting one. pid_ok = _pid_alive(int(state.get("pid") or 0)) # Healthy server: done (when it's ours). try: health = base_url.rsplit("/v1", 1)[0] + "/health" with urllib.request.urlopen(health, timeout=3) as r: if r.status == 200: return endpoint if pid_ok else None except (urllib.error.URLError, OSError, TimeoutError): pass # Not healthy YET: a live supervisor child is a STARTING server (state # is written at spawn; llama-server takes seconds to listen). Resolve # optimistically so readiness probes racing the boot see a configured # provider, not missing credentials. A dead pid is a crashed-without- # cleanup leftover — ignore it so requests don't blackhole. return endpoint if pid_ok else None def resolve_llamacpp_endpoint(config: dict | None = None, wait_for_boot_s: float = 8.0) -> dict | None: """Managed-first, detection-second endpoint for llamacpp aliases. Boot-race rung: on a fresh backend start there is NO state file yet — the lifespan boot thread is still spawning the server (config load + preset generation + spawn ≈ 1-3 s) while the desktop's readiness probe fires the moment the WebSocket connects. """ managed = _state_endpoint() if managed: return managed from hermes_cli.local_runtime.detect import detect_server ports = ((config or {}).get("local_runtime") or {}).get("detect_ports") or [] hit = detect_server(extra_ports=tuple(int(p) for p in ports)) if hit and not hit.auth_required: return {"base_url": hit.base_url, "api_key": ""} if wait_for_boot_s > 0 and _boot_in_flight(config): _kick_managed_boot(config) deadline = time.monotonic() + wait_for_boot_s while time.monotonic() < deadline: time.sleep(0.25) managed = _state_endpoint() if managed: return managed return None _KICK_LOCK = threading.Lock() def _load_config_if_none(config: dict | None) -> dict | None: if config is not None: return config from hermes_cli.config import load_config return load_config() def _kick_managed_boot(config: dict | None) -> None: """Actively start the managed server when resolution finds it missing. The wait loop above assumes some OTHER thread is bringing the server up — true only at backend start (the lifespan boot thread). """ if not _KICK_LOCK.acquire(blocking=False): return # a kick is already in flight def _boot() -> None: try: from hermes_cli.local_runtime.bootstrap import ensure_local_runtime ensure_local_runtime(_load_config_if_none(config)) except Exception: # noqa: BLE001 — best-effort; resolution falls back logger.warning("on-demand managed-server boot failed", exc_info=True) finally: _KICK_LOCK.release() threading.Thread(target=_boot, daemon=True, name="lr-on-demand-boot").start() def _boot_in_flight(config: dict | None) -> bool: """True when the managed runtime is enabled and installed — the state Installed-ness is a verified-manifest scan under runtimes_root(), NOT a bare ``server_binary()`` call: that helper needs an install_dir, and calling it bare made this gate throw-and-return False forever, silently disabling the boot wait. """ try: config = _load_config_if_none(config) if not ((config or {}).get("local_runtime") or {}).get("enabled"): return False from hermes_cli.local_runtime.binaries import runtimes_root for manifest in runtimes_root().glob("*/*/manifest.json"): try: if json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"): return True except (ValueError, OSError): continue return False except Exception: # noqa: BLE001 return False