Files
hermes-agent/agent/lsp/manager.py
T
Teknium b6e1037730 refactor(agent/lsp): table-driven server registry, shared client/manager helpers, compact docs
- servers.py: per-language _root_*/_spawn_* functions collapsed into declarative
  ServerDef entries built by _make_spec/_simple_spawn/_markers_root (-707 LOC)
- client.py: _write/_send_reply/_cancel_task helpers replace repeated
  try/except blocks; handler dispatch tables for server->client requests
- manager.py: _broken_key unifies the (server_id, root) key derivation
- install.py/cli.py/eventlog.py/workspace.py/protocol.py/range_shift.py/
  reporter.py: dedupe helpers (_find_binary, _link_into_bin, _run_installer,
  _emit_once, _walk_up), drop dead log_no_server_configured, compact docs
2026-09-02 13:53:27 -07:00

604 lines
24 KiB
Python

"""Service-level orchestration for LSP clients.
:class:`LSPService` bridges the synchronous file_operations layer and the
async :class:`agent.lsp.client.LSPClient`:
- One asyncio loop in a background thread; :meth:`get_diagnostics_sync`
opens + waits + drains in one blocking call.
- One lazily spawned client per ``(server_id, workspace_root)``.
- A **broken-set** of pairs that failed to spawn/initialize — never retried
for the life of the service.
- A **delta baseline** per file: ``snapshot_baseline()`` runs BEFORE a write,
and the next ``get_diagnostics_sync()`` returns only diagnostics not in it.
The service is off unless config enables it; :meth:`is_active` says whether
it does anything, and file_operations falls back to the in-process syntax
check otherwise.
"""
from __future__ import annotations
import asyncio
import logging
import os
import threading
import time
from typing import Any, Callable, Dict, List, Optional, Tuple
from agent.lsp import eventlog
from agent.lsp.client import (
DIAGNOSTICS_DOCUMENT_WAIT,
LSPClient,
_diagnostic_key as _diag_key,
)
from agent.lsp.servers import (
ServerContext,
ServerDef,
find_server_for_file,
language_id_for,
)
from agent.lsp.workspace import (
clear_cache,
resolve_workspace_for_file,
)
logger = logging.getLogger("agent.lsp.manager")
DEFAULT_IDLE_TIMEOUT = 600 # seconds; servers idle for >10min get reaped
MIN_IDLE_TIMEOUT = 30 # floor for config values; must exceed any per-op wait budget
class _BackgroundLoop:
"""A daemon thread owning one asyncio loop; :meth:`run` blocks on a coroutine."""
def __init__(self) -> None:
self._loop: Optional[asyncio.AbstractEventLoop] = None
self._thread: Optional[threading.Thread] = None
self._ready = threading.Event()
def start(self) -> None:
if self._thread is not None:
return
self._thread = threading.Thread(
target=self._run_forever,
name="hermes-lsp-loop",
daemon=True,
)
self._thread.start()
self._ready.wait(timeout=5.0)
def _run_forever(self) -> None:
loop = asyncio.new_event_loop()
self._loop = loop
asyncio.set_event_loop(loop)
self._ready.set()
try:
loop.run_forever()
finally:
try:
loop.close()
except Exception: # noqa: BLE001
pass
def run(self, coro, *, timeout: Optional[float] = None) -> Any:
"""Submit a coroutine to the loop and block for its result (or raise)."""
from agent.async_utils import safe_schedule_threadsafe
if self._loop is None:
if asyncio.iscoroutine(coro):
coro.close()
raise RuntimeError("background loop not started")
fut = safe_schedule_threadsafe(coro, self._loop)
if fut is None:
raise RuntimeError("background loop not running")
try:
return fut.result(timeout=timeout)
except Exception:
fut.cancel()
raise
def stop(self) -> None:
loop = self._loop
if loop is None:
return
try:
loop.call_soon_threadsafe(loop.stop)
except RuntimeError:
pass
if self._thread is not None:
self._thread.join(timeout=2.0)
self._loop = None
self._thread = None
class LSPService:
"""The process-wide LSP service; use :func:`agent.lsp.get_service` rather than constructing directly."""
def __init__(
self,
*,
enabled: bool,
wait_mode: str,
wait_timeout: float,
install_strategy: str,
binary_overrides: Optional[Dict[str, List[str]]] = None,
env_overrides: Optional[Dict[str, Dict[str, str]]] = None,
init_overrides: Optional[Dict[str, Dict[str, Any]]] = None,
disabled_servers: Optional[List[str]] = None,
idle_timeout: float = DEFAULT_IDLE_TIMEOUT,
) -> None:
self._enabled = enabled
self._wait_mode = wait_mode if wait_mode in {"document", "full"} else "document"
self._wait_timeout = wait_timeout
self._install_strategy = install_strategy
self._binary_overrides = binary_overrides or {}
self._env_overrides = env_overrides or {}
self._init_overrides = init_overrides or {}
self._disabled_servers = set(disabled_servers or [])
self._idle_timeout = idle_timeout
self._loop = _BackgroundLoop()
if self._enabled:
self._loop.start()
# Per-(server_id, workspace_root) state
self._clients: Dict[Tuple[str, str], LSPClient] = {}
self._broken: set = set()
self._spawning: Dict[Tuple[str, str], asyncio.Future] = {}
self._last_used: Dict[Tuple[str, str], float] = {}
self._state_lock = threading.Lock()
self._idle_reaper_task: Optional[asyncio.Task] = None
# file path → diagnostics snapshot taken immediately before a write.
self._delta_baseline: Dict[str, List[Dict[str, Any]]] = {}
if self._enabled and self._idle_timeout > 0:
self._loop.run(self._start_idle_reaper(), timeout=2.0)
@classmethod
def create_from_config(cls) -> Optional["LSPService"]:
"""Build a service from ``hermes_cli.config``; ``None`` if config can't load."""
try:
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e: # noqa: BLE001
logger.debug("LSP config load failed: %s", e)
return None
lsp_cfg = (cfg.get("lsp") or {}) if isinstance(cfg, dict) else {}
if not isinstance(lsp_cfg, dict):
lsp_cfg = {}
enabled = bool(lsp_cfg.get("enabled", True))
wait_mode = lsp_cfg.get("wait_mode", "document")
wait_timeout = float(lsp_cfg.get("wait_timeout", DIAGNOSTICS_DOCUMENT_WAIT))
install_strategy = lsp_cfg.get("install_strategy", "auto")
try:
idle_timeout = float(lsp_cfg.get("idle_timeout", DEFAULT_IDLE_TIMEOUT))
except (TypeError, ValueError):
idle_timeout = DEFAULT_IDLE_TIMEOUT
if 0 < idle_timeout < MIN_IDLE_TIMEOUT:
# Below the per-op wait budget the reaper could kill a client
# mid-flight and the outer timeout would then mark the pair broken
# for the process lifetime. Clamp (0 still disables).
idle_timeout = MIN_IDLE_TIMEOUT
servers_cfg = lsp_cfg.get("servers") or {}
disabled = []
binary_overrides: Dict[str, List[str]] = {}
env_overrides: Dict[str, Dict[str, str]] = {}
init_overrides: Dict[str, Dict[str, Any]] = {}
if isinstance(servers_cfg, dict):
for name, sub in servers_cfg.items():
if not isinstance(sub, dict):
continue
if sub.get("disabled"):
disabled.append(name)
cmd = sub.get("command")
if isinstance(cmd, list) and cmd:
binary_overrides[name] = cmd
env = sub.get("env")
if isinstance(env, dict):
env_overrides[name] = {k: str(v) for k, v in env.items()}
init = sub.get("initialization_options")
if isinstance(init, dict):
init_overrides[name] = init
return cls(
enabled=enabled,
wait_mode=wait_mode,
wait_timeout=wait_timeout,
install_strategy=install_strategy,
binary_overrides=binary_overrides,
env_overrides=env_overrides,
init_overrides=init_overrides,
disabled_servers=disabled,
idle_timeout=idle_timeout,
)
# ------------------------------------------------------------------
# public API
# ------------------------------------------------------------------
def is_active(self) -> bool:
"""Return True iff this service should be consulted at all."""
return self._enabled
def _broken_key(self, srv: ServerDef, file_path: str) -> Optional[Tuple[str, str]]:
"""``(server_id, per-server root)`` broken-set key, or ``None`` when the file isn't gated in.
Falls back to the workspace root when the per-server resolver fails —
the same key ``_get_or_spawn`` would have used when it failed.
"""
ws_root, gated = resolve_workspace_for_file(file_path)
if not (ws_root and gated):
return None
try:
per_server_root = srv.resolve_root(file_path, ws_root) or ws_root
except Exception: # noqa: BLE001
per_server_root = ws_root
return (srv.server_id, per_server_root)
def enabled_for(self, file_path: str) -> bool:
"""Return True iff LSP should run for this file.
Gates on a registered, non-disabled server for the extension, on
git-workspace detection, and on the pair not being in the broken-set
(so a failed server costs no spawn attempts or timeouts until
``hermes lsp restart`` or process exit).
"""
if not self._enabled:
return False
srv = find_server_for_file(file_path)
if srv is None or srv.server_id in self._disabled_servers:
return False
key = self._broken_key(srv, file_path)
return key is not None and key not in self._broken
def snapshot_baseline(self, file_path: str) -> None:
"""Snapshot current diagnostics for ``file_path`` as the delta baseline (call BEFORE a write).
Best-effort: failures are swallowed so a flaky server can't break a
write, but outer timeouts mark the pair broken so later edits skip it.
"""
if not self.enabled_for(file_path):
return
try:
# Outer budget must exceed the inner wait or a slow-but-alive
# server gets falsely marked broken.
t = max(8.0, self._wait_timeout + 3.0)
diags = self._loop.run(self._snapshot_async(file_path), timeout=t)
self._delta_baseline[os.path.abspath(file_path)] = diags or []
except Exception as e: # noqa: BLE001
logger.debug("baseline snapshot failed for %s: %s", file_path, e)
self._mark_broken_for_file(file_path, e)
self._delta_baseline[os.path.abspath(file_path)] = []
def get_diagnostics_sync(
self,
file_path: str,
*,
delta: bool = True,
timeout: Optional[float] = None,
line_shift: Optional[Callable[[int], Optional[int]]] = None,
) -> List[Dict[str, Any]]:
"""Synchronously open ``file_path``, wait for diagnostics, return them.
With ``delta`` (default) the result excludes anything in the baseline
from :meth:`snapshot_baseline`. ``line_shift`` (built by
:func:`agent.lsp.range_shift.build_line_shift`) remaps the baseline
into post-edit coordinates first, so pre-existing diagnostics that
merely moved don't look introduced by this edit.
Returns ``[]`` when LSP is disabled, no workspace/server matches, or
the server can't be spawned. Never raises.
"""
if not self.enabled_for(file_path):
return []
# Resolve server_id eagerly for structured logs on the error paths.
srv = find_server_for_file(file_path)
server_id = srv.server_id if srv else "?"
try:
t = timeout if timeout is not None else self._wait_timeout + 2.0
diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t)
except asyncio.TimeoutError as e:
eventlog.log_timeout(server_id, file_path)
logger.debug("LSP diagnostics timeout for %s: %s", file_path, e)
self._mark_broken_for_file(file_path, e)
return []
except Exception as e: # noqa: BLE001
eventlog.log_server_error(server_id, file_path, e)
logger.debug("LSP diagnostics fetch failed for %s: %s", file_path, e)
self._mark_broken_for_file(file_path, e)
return []
if diags is None:
# Server alive but no verdict on the post-edit content in budget
# (common for tsserver on big projects). Report "no data" rather
# than stale stores — that would be the ghost-diagnostics bug.
# Not marked broken: slow is not dead.
eventlog.log_timeout(server_id, file_path, kind="fresh diagnostics")
return []
abs_path = os.path.abspath(file_path)
if delta:
baseline = self._delta_baseline.get(abs_path) or []
if baseline:
if line_shift is not None:
# Entries that map into a deleted region drop out — they no longer apply.
from agent.lsp.range_shift import shift_baseline
baseline = shift_baseline(baseline, line_shift)
seen = {_diag_key(d) for d in baseline}
diags = [d for d in diags if _diag_key(d) not in seen]
# Roll the baseline forward so the next call is a delta against this state.
try:
fresh = self._loop.run(self._current_diags_async(file_path), timeout=2.0) or []
except Exception: # noqa: BLE001
fresh = []
if fresh:
self._delta_baseline[abs_path] = fresh
if diags:
eventlog.log_diagnostics(server_id, file_path, len(diags))
else:
eventlog.log_clean(server_id, file_path)
return diags
def _mark_broken_for_file(self, file_path: str, exc: BaseException) -> None:
"""Mark the file's ``(server_id, root)`` pair broken after an outer timeout/error.
The outer ``_loop.run`` timeout cancels the in-flight spawn before
``_get_or_spawn`` could record the failure, so without this every
later write would re-pay the full timeout. Also kills any
half-initialized client left in ``_clients`` and logs the failure once.
``exc`` is used only for logging.
"""
srv = find_server_for_file(file_path)
if srv is None:
return
key = self._broken_key(srv, file_path)
if key is None:
return
already_broken = key in self._broken
self._broken.add(key)
with self._state_lock:
client = self._clients.pop(key, None)
self._last_used.pop(key, None)
if client is not None:
try:
# Fire-and-forget shutdown — we're already on a slow path.
self._loop.run(client.shutdown(), timeout=1.0)
except Exception: # noqa: BLE001
pass
if not already_broken:
eventlog.log_spawn_failed(srv.server_id, key[1], exc)
def shutdown(self) -> None:
"""Tear down all clients and stop the background loop."""
if not self._enabled:
return
try:
self._loop.run(self._shutdown_async(), timeout=10.0)
except Exception as e: # noqa: BLE001
logger.debug("LSP shutdown error: %s", e)
self._loop.stop()
clear_cache()
# ------------------------------------------------------------------
# async internals
# ------------------------------------------------------------------
async def _snapshot_async(self, file_path: str) -> List[Dict[str, Any]]:
# No fresh data for the pre-edit content → empty baseline. Safe: the
# delta filter then removes less, never more. Never seed from stale stores.
return await self._open_and_wait_async(file_path, snapshot=True) or []
async def _open_and_wait_async(self, file_path: str, *, snapshot: bool = False) -> Optional[List[Dict[str, Any]]]:
"""Open + wait for FRESH diagnostics.
Returns the fresh list, or ``None`` when the server produced no
post-change data in budget. ``[]`` means "checked, clean"; ``None``
means "no verdict" — callers must not substitute stale data for either.
``snapshot`` mode (pre-write baseline) skips didSave and uses the
default wait budget.
"""
client = await self._get_or_spawn(file_path)
if client is None:
return None
try:
version = await client.open_file(file_path, language_id=language_id_for(file_path))
if not snapshot:
await client.save_file(file_path)
fresh = await client.wait_for_diagnostics(
file_path, version, mode=self._wait_mode,
timeout=None if snapshot else self._wait_timeout,
)
except Exception as e: # noqa: BLE001
if snapshot:
logger.debug("snapshot open/wait failed: %s", e)
else:
logger.debug("open/wait failed for %s: %s", file_path, e)
return None
self._touch(client)
if not fresh:
return None
return list(client.diagnostics_for(file_path, fresh_only=True))
async def _current_diags_async(self, file_path: str) -> List[Dict[str, Any]]:
ws, gated = resolve_workspace_for_file(file_path)
srv = find_server_for_file(file_path)
if not (ws and gated and srv):
return []
with self._state_lock:
client = self._clients.get((srv.server_id, ws))
if client is None:
return []
return list(client.diagnostics_for(file_path, fresh_only=True))
async def _get_or_spawn(self, file_path: str) -> Optional[LSPClient]:
srv = find_server_for_file(file_path)
if srv is None:
return None
if srv.server_id in self._disabled_servers:
eventlog.log_disabled(srv.server_id, file_path, "disabled in config")
return None
ws_root, gated = resolve_workspace_for_file(file_path)
if not (ws_root and gated):
eventlog.log_no_project_root(srv.server_id, file_path)
return None
per_server_root = srv.resolve_root(file_path, ws_root)
if per_server_root is None:
eventlog.log_disabled(
srv.server_id, file_path, "exclude marker hit (server gated off)"
)
return None
key = (srv.server_id, per_server_root)
if key in self._broken:
return None
with self._state_lock:
client = self._clients.get(key)
if client is not None and client.is_running:
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
return client
spawning = self._spawning.get(key)
if spawning is not None:
try:
return await spawning
except Exception: # noqa: BLE001
return None
loop = asyncio.get_running_loop()
spawn_future: asyncio.Future = loop.create_future()
with self._state_lock:
self._spawning[key] = spawn_future
try:
ctx = ServerContext(
workspace_root=per_server_root,
install_strategy=self._install_strategy,
binary_overrides=self._binary_overrides,
env_overrides=self._env_overrides,
init_overrides=self._init_overrides,
)
spec = srv.build_spawn(per_server_root, ctx)
if spec is None:
# Binary not locatable (auto-install off, manual-only, or
# install failed) — surface once via the structured logger.
eventlog.log_server_unavailable(srv.server_id, srv.server_id)
self._broken.add(key)
spawn_future.set_result(None)
return None
client = LSPClient(
server_id=srv.server_id,
workspace_root=spec.workspace_root,
command=spec.command,
env=spec.env,
cwd=spec.cwd,
initialization_options=spec.initialization_options,
seed_diagnostics_on_first_push=spec.seed_diagnostics_on_first_push or srv.seed_first_push,
)
try:
await client.start()
except Exception as e: # noqa: BLE001
eventlog.log_spawn_failed(srv.server_id, per_server_root, e)
self._broken.add(key)
spawn_future.set_result(None)
return None
with self._state_lock:
self._clients[key] = client
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
spawn_future.set_result(client)
return client
finally:
with self._state_lock:
self._spawning.pop(key, None)
async def _start_idle_reaper(self) -> None:
self._idle_reaper_task = asyncio.create_task(self._idle_reaper_loop())
def _touch(self, client: LSPClient) -> None:
"""Refresh last-used; guarded on membership so a client reaped mid-operation can't resurrect its entry."""
key = (client.server_id, client.workspace_root)
with self._state_lock:
if key in self._clients:
self._last_used[key] = time.time()
async def _idle_reaper_loop(self) -> None:
interval = min(60.0, self._idle_timeout)
while True:
await asyncio.sleep(interval)
try:
await self._reap_idle_once()
except asyncio.CancelledError:
raise
except Exception as e: # noqa: BLE001
# A transient sweep error must not kill the reaper, or the
# unbounded-accumulation leak it exists to fix comes back.
logger.debug("LSP idle reaper sweep error: %s", e)
async def _reap_idle_once(self) -> None:
cutoff = time.time() - self._idle_timeout
with self._state_lock:
idle_keys = [
key
for key in self._clients
if self._last_used.get(key, 0) < cutoff
]
clients = [self._clients.pop(key) for key in idle_keys]
for key in idle_keys:
self._last_used.pop(key, None)
if clients:
eventlog.log_reaped(
[(c.server_id, c.workspace_root) for c in clients],
self._idle_timeout,
)
await asyncio.gather(
*(client.shutdown() for client in clients),
return_exceptions=True,
)
async def _shutdown_async(self) -> None:
reaper = self._idle_reaper_task
self._idle_reaper_task = None
if reaper is not None:
reaper.cancel()
await asyncio.gather(reaper, return_exceptions=True)
with self._state_lock:
clients = list(self._clients.values())
self._clients.clear()
self._broken.clear()
self._last_used.clear()
await asyncio.gather(
*(c.shutdown() for c in clients),
return_exceptions=True,
)
def get_status(self) -> Dict[str, Any]:
"""Return a snapshot of the service for ``hermes lsp status``."""
with self._state_lock:
clients = [
{
"server_id": k[0],
"workspace_root": k[1],
"state": c.state,
"running": c.is_running,
}
for k, c in self._clients.items()
]
broken = list(self._broken)
return {
"enabled": self._enabled,
"wait_mode": self._wait_mode,
"wait_timeout": self._wait_timeout,
"install_strategy": self._install_strategy,
"clients": clients,
"broken": broken,
"disabled_servers": sorted(self._disabled_servers),
}
__all__ = ["LSPService"]