"""Service-level orchestration for LSP clients. :class:`LSPService` bridges the synchronous file_operations layer and the async :class:`agent.lsp.client.LSPClient`: - One asyncio loop in a background thread; :meth:`get_diagnostics_sync` opens + waits + drains in one blocking call. - One lazily spawned client per ``(server_id, workspace_root)``. - A **broken-set** of pairs that failed to spawn/initialize — never retried for the life of the service. - A **delta baseline** per file: ``snapshot_baseline()`` runs BEFORE a write, and the next ``get_diagnostics_sync()`` returns only diagnostics not in it. The service is off unless config enables it; :meth:`is_active` says whether it does anything, and file_operations falls back to the in-process syntax check otherwise. """ from __future__ import annotations import asyncio import logging import os import threading import time from typing import Any, Callable, Dict, List, Optional, Tuple from agent.lsp import eventlog from agent.lsp.client import ( DIAGNOSTICS_DOCUMENT_WAIT, LSPClient, _diagnostic_key as _diag_key, ) from agent.lsp.servers import ( ServerContext, ServerDef, find_server_for_file, language_id_for, ) from agent.lsp.workspace import ( clear_cache, resolve_workspace_for_file, ) logger = logging.getLogger("agent.lsp.manager") DEFAULT_IDLE_TIMEOUT = 600 # seconds; servers idle for >10min get reaped MIN_IDLE_TIMEOUT = 30 # floor for config values; must exceed any per-op wait budget class _BackgroundLoop: """A daemon thread owning one asyncio loop; :meth:`run` blocks on a coroutine.""" def __init__(self) -> None: self._loop: Optional[asyncio.AbstractEventLoop] = None self._thread: Optional[threading.Thread] = None self._ready = threading.Event() def start(self) -> None: if self._thread is not None: return self._thread = threading.Thread( target=self._run_forever, name="hermes-lsp-loop", daemon=True, ) self._thread.start() self._ready.wait(timeout=5.0) def _run_forever(self) -> None: loop = asyncio.new_event_loop() self._loop = loop asyncio.set_event_loop(loop) self._ready.set() try: loop.run_forever() finally: try: loop.close() except Exception: # noqa: BLE001 pass def run(self, coro, *, timeout: Optional[float] = None) -> Any: """Submit a coroutine to the loop and block for its result (or raise).""" from agent.async_utils import safe_schedule_threadsafe if self._loop is None: if asyncio.iscoroutine(coro): coro.close() raise RuntimeError("background loop not started") fut = safe_schedule_threadsafe(coro, self._loop) if fut is None: raise RuntimeError("background loop not running") try: return fut.result(timeout=timeout) except Exception: fut.cancel() raise def stop(self) -> None: loop = self._loop if loop is None: return try: loop.call_soon_threadsafe(loop.stop) except RuntimeError: pass if self._thread is not None: self._thread.join(timeout=2.0) self._loop = None self._thread = None class LSPService: """The process-wide LSP service; use :func:`agent.lsp.get_service` rather than constructing directly.""" def __init__( self, *, enabled: bool, wait_mode: str, wait_timeout: float, install_strategy: str, binary_overrides: Optional[Dict[str, List[str]]] = None, env_overrides: Optional[Dict[str, Dict[str, str]]] = None, init_overrides: Optional[Dict[str, Dict[str, Any]]] = None, disabled_servers: Optional[List[str]] = None, idle_timeout: float = DEFAULT_IDLE_TIMEOUT, ) -> None: self._enabled = enabled self._wait_mode = wait_mode if wait_mode in {"document", "full"} else "document" self._wait_timeout = wait_timeout self._install_strategy = install_strategy self._binary_overrides = binary_overrides or {} self._env_overrides = env_overrides or {} self._init_overrides = init_overrides or {} self._disabled_servers = set(disabled_servers or []) self._idle_timeout = idle_timeout self._loop = _BackgroundLoop() if self._enabled: self._loop.start() # Per-(server_id, workspace_root) state self._clients: Dict[Tuple[str, str], LSPClient] = {} self._broken: set = set() self._spawning: Dict[Tuple[str, str], asyncio.Future] = {} self._last_used: Dict[Tuple[str, str], float] = {} self._state_lock = threading.Lock() self._idle_reaper_task: Optional[asyncio.Task] = None # file path → diagnostics snapshot taken immediately before a write. self._delta_baseline: Dict[str, List[Dict[str, Any]]] = {} if self._enabled and self._idle_timeout > 0: self._loop.run(self._start_idle_reaper(), timeout=2.0) @classmethod def create_from_config(cls) -> Optional["LSPService"]: """Build a service from ``hermes_cli.config``; ``None`` if config can't load.""" try: from hermes_cli.config import load_config_readonly cfg = load_config_readonly() except Exception as e: # noqa: BLE001 logger.debug("LSP config load failed: %s", e) return None lsp_cfg = (cfg.get("lsp") or {}) if isinstance(cfg, dict) else {} if not isinstance(lsp_cfg, dict): lsp_cfg = {} enabled = bool(lsp_cfg.get("enabled", True)) wait_mode = lsp_cfg.get("wait_mode", "document") wait_timeout = float(lsp_cfg.get("wait_timeout", DIAGNOSTICS_DOCUMENT_WAIT)) install_strategy = lsp_cfg.get("install_strategy", "auto") try: idle_timeout = float(lsp_cfg.get("idle_timeout", DEFAULT_IDLE_TIMEOUT)) except (TypeError, ValueError): idle_timeout = DEFAULT_IDLE_TIMEOUT if 0 < idle_timeout < MIN_IDLE_TIMEOUT: # Below the per-op wait budget the reaper could kill a client # mid-flight and the outer timeout would then mark the pair broken # for the process lifetime. Clamp (0 still disables). idle_timeout = MIN_IDLE_TIMEOUT servers_cfg = lsp_cfg.get("servers") or {} disabled = [] binary_overrides: Dict[str, List[str]] = {} env_overrides: Dict[str, Dict[str, str]] = {} init_overrides: Dict[str, Dict[str, Any]] = {} if isinstance(servers_cfg, dict): for name, sub in servers_cfg.items(): if not isinstance(sub, dict): continue if sub.get("disabled"): disabled.append(name) cmd = sub.get("command") if isinstance(cmd, list) and cmd: binary_overrides[name] = cmd env = sub.get("env") if isinstance(env, dict): env_overrides[name] = {k: str(v) for k, v in env.items()} init = sub.get("initialization_options") if isinstance(init, dict): init_overrides[name] = init return cls( enabled=enabled, wait_mode=wait_mode, wait_timeout=wait_timeout, install_strategy=install_strategy, binary_overrides=binary_overrides, env_overrides=env_overrides, init_overrides=init_overrides, disabled_servers=disabled, idle_timeout=idle_timeout, ) # ------------------------------------------------------------------ # public API # ------------------------------------------------------------------ def is_active(self) -> bool: """Return True iff this service should be consulted at all.""" return self._enabled def _broken_key(self, srv: ServerDef, file_path: str) -> Optional[Tuple[str, str]]: """``(server_id, per-server root)`` broken-set key, or ``None`` when the file isn't gated in. Falls back to the workspace root when the per-server resolver fails — the same key ``_get_or_spawn`` would have used when it failed. """ ws_root, gated = resolve_workspace_for_file(file_path) if not (ws_root and gated): return None try: per_server_root = srv.resolve_root(file_path, ws_root) or ws_root except Exception: # noqa: BLE001 per_server_root = ws_root return (srv.server_id, per_server_root) def enabled_for(self, file_path: str) -> bool: """Return True iff LSP should run for this file. Gates on a registered, non-disabled server for the extension, on git-workspace detection, and on the pair not being in the broken-set (so a failed server costs no spawn attempts or timeouts until ``hermes lsp restart`` or process exit). """ if not self._enabled: return False srv = find_server_for_file(file_path) if srv is None or srv.server_id in self._disabled_servers: return False key = self._broken_key(srv, file_path) return key is not None and key not in self._broken def snapshot_baseline(self, file_path: str) -> None: """Snapshot current diagnostics for ``file_path`` as the delta baseline (call BEFORE a write). Best-effort: failures are swallowed so a flaky server can't break a write, but outer timeouts mark the pair broken so later edits skip it. """ if not self.enabled_for(file_path): return try: # Outer budget must exceed the inner wait or a slow-but-alive # server gets falsely marked broken. t = max(8.0, self._wait_timeout + 3.0) diags = self._loop.run(self._snapshot_async(file_path), timeout=t) self._delta_baseline[os.path.abspath(file_path)] = diags or [] except Exception as e: # noqa: BLE001 logger.debug("baseline snapshot failed for %s: %s", file_path, e) self._mark_broken_for_file(file_path, e) self._delta_baseline[os.path.abspath(file_path)] = [] def get_diagnostics_sync( self, file_path: str, *, delta: bool = True, timeout: Optional[float] = None, line_shift: Optional[Callable[[int], Optional[int]]] = None, ) -> List[Dict[str, Any]]: """Synchronously open ``file_path``, wait for diagnostics, return them. With ``delta`` (default) the result excludes anything in the baseline from :meth:`snapshot_baseline`. ``line_shift`` (built by :func:`agent.lsp.range_shift.build_line_shift`) remaps the baseline into post-edit coordinates first, so pre-existing diagnostics that merely moved don't look introduced by this edit. Returns ``[]`` when LSP is disabled, no workspace/server matches, or the server can't be spawned. Never raises. """ if not self.enabled_for(file_path): return [] # Resolve server_id eagerly for structured logs on the error paths. srv = find_server_for_file(file_path) server_id = srv.server_id if srv else "?" try: t = timeout if timeout is not None else self._wait_timeout + 2.0 diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t) except asyncio.TimeoutError as e: eventlog.log_timeout(server_id, file_path) logger.debug("LSP diagnostics timeout for %s: %s", file_path, e) self._mark_broken_for_file(file_path, e) return [] except Exception as e: # noqa: BLE001 eventlog.log_server_error(server_id, file_path, e) logger.debug("LSP diagnostics fetch failed for %s: %s", file_path, e) self._mark_broken_for_file(file_path, e) return [] if diags is None: # Server alive but no verdict on the post-edit content in budget # (common for tsserver on big projects). Report "no data" rather # than stale stores — that would be the ghost-diagnostics bug. # Not marked broken: slow is not dead. eventlog.log_timeout(server_id, file_path, kind="fresh diagnostics") return [] abs_path = os.path.abspath(file_path) if delta: baseline = self._delta_baseline.get(abs_path) or [] if baseline: if line_shift is not None: # Entries that map into a deleted region drop out — they no longer apply. from agent.lsp.range_shift import shift_baseline baseline = shift_baseline(baseline, line_shift) seen = {_diag_key(d) for d in baseline} diags = [d for d in diags if _diag_key(d) not in seen] # Roll the baseline forward so the next call is a delta against this state. try: fresh = self._loop.run(self._current_diags_async(file_path), timeout=2.0) or [] except Exception: # noqa: BLE001 fresh = [] if fresh: self._delta_baseline[abs_path] = fresh if diags: eventlog.log_diagnostics(server_id, file_path, len(diags)) else: eventlog.log_clean(server_id, file_path) return diags def _mark_broken_for_file(self, file_path: str, exc: BaseException) -> None: """Mark the file's ``(server_id, root)`` pair broken after an outer timeout/error. The outer ``_loop.run`` timeout cancels the in-flight spawn before ``_get_or_spawn`` could record the failure, so without this every later write would re-pay the full timeout. Also kills any half-initialized client left in ``_clients`` and logs the failure once. ``exc`` is used only for logging. """ srv = find_server_for_file(file_path) if srv is None: return key = self._broken_key(srv, file_path) if key is None: return already_broken = key in self._broken self._broken.add(key) with self._state_lock: client = self._clients.pop(key, None) self._last_used.pop(key, None) if client is not None: try: # Fire-and-forget shutdown — we're already on a slow path. self._loop.run(client.shutdown(), timeout=1.0) except Exception: # noqa: BLE001 pass if not already_broken: eventlog.log_spawn_failed(srv.server_id, key[1], exc) def shutdown(self) -> None: """Tear down all clients and stop the background loop.""" if not self._enabled: return try: self._loop.run(self._shutdown_async(), timeout=10.0) except Exception as e: # noqa: BLE001 logger.debug("LSP shutdown error: %s", e) self._loop.stop() clear_cache() # ------------------------------------------------------------------ # async internals # ------------------------------------------------------------------ async def _snapshot_async(self, file_path: str) -> List[Dict[str, Any]]: # No fresh data for the pre-edit content → empty baseline. Safe: the # delta filter then removes less, never more. Never seed from stale stores. return await self._open_and_wait_async(file_path, snapshot=True) or [] async def _open_and_wait_async(self, file_path: str, *, snapshot: bool = False) -> Optional[List[Dict[str, Any]]]: """Open + wait for FRESH diagnostics. Returns the fresh list, or ``None`` when the server produced no post-change data in budget. ``[]`` means "checked, clean"; ``None`` means "no verdict" — callers must not substitute stale data for either. ``snapshot`` mode (pre-write baseline) skips didSave and uses the default wait budget. """ client = await self._get_or_spawn(file_path) if client is None: return None try: version = await client.open_file(file_path, language_id=language_id_for(file_path)) if not snapshot: await client.save_file(file_path) fresh = await client.wait_for_diagnostics( file_path, version, mode=self._wait_mode, timeout=None if snapshot else self._wait_timeout, ) except Exception as e: # noqa: BLE001 if snapshot: logger.debug("snapshot open/wait failed: %s", e) else: logger.debug("open/wait failed for %s: %s", file_path, e) return None self._touch(client) if not fresh: return None return list(client.diagnostics_for(file_path, fresh_only=True)) async def _current_diags_async(self, file_path: str) -> List[Dict[str, Any]]: ws, gated = resolve_workspace_for_file(file_path) srv = find_server_for_file(file_path) if not (ws and gated and srv): return [] with self._state_lock: client = self._clients.get((srv.server_id, ws)) if client is None: return [] return list(client.diagnostics_for(file_path, fresh_only=True)) async def _get_or_spawn(self, file_path: str) -> Optional[LSPClient]: srv = find_server_for_file(file_path) if srv is None: return None if srv.server_id in self._disabled_servers: eventlog.log_disabled(srv.server_id, file_path, "disabled in config") return None ws_root, gated = resolve_workspace_for_file(file_path) if not (ws_root and gated): eventlog.log_no_project_root(srv.server_id, file_path) return None per_server_root = srv.resolve_root(file_path, ws_root) if per_server_root is None: eventlog.log_disabled( srv.server_id, file_path, "exclude marker hit (server gated off)" ) return None key = (srv.server_id, per_server_root) if key in self._broken: return None with self._state_lock: client = self._clients.get(key) if client is not None and client.is_running: self._last_used[key] = time.time() eventlog.log_active(srv.server_id, per_server_root) return client spawning = self._spawning.get(key) if spawning is not None: try: return await spawning except Exception: # noqa: BLE001 return None loop = asyncio.get_running_loop() spawn_future: asyncio.Future = loop.create_future() with self._state_lock: self._spawning[key] = spawn_future try: ctx = ServerContext( workspace_root=per_server_root, install_strategy=self._install_strategy, binary_overrides=self._binary_overrides, env_overrides=self._env_overrides, init_overrides=self._init_overrides, ) spec = srv.build_spawn(per_server_root, ctx) if spec is None: # Binary not locatable (auto-install off, manual-only, or # install failed) — surface once via the structured logger. eventlog.log_server_unavailable(srv.server_id, srv.server_id) self._broken.add(key) spawn_future.set_result(None) return None client = LSPClient( server_id=srv.server_id, workspace_root=spec.workspace_root, command=spec.command, env=spec.env, cwd=spec.cwd, initialization_options=spec.initialization_options, seed_diagnostics_on_first_push=spec.seed_diagnostics_on_first_push or srv.seed_first_push, ) try: await client.start() except Exception as e: # noqa: BLE001 eventlog.log_spawn_failed(srv.server_id, per_server_root, e) self._broken.add(key) spawn_future.set_result(None) return None with self._state_lock: self._clients[key] = client self._last_used[key] = time.time() eventlog.log_active(srv.server_id, per_server_root) spawn_future.set_result(client) return client finally: with self._state_lock: self._spawning.pop(key, None) async def _start_idle_reaper(self) -> None: self._idle_reaper_task = asyncio.create_task(self._idle_reaper_loop()) def _touch(self, client: LSPClient) -> None: """Refresh last-used; guarded on membership so a client reaped mid-operation can't resurrect its entry.""" key = (client.server_id, client.workspace_root) with self._state_lock: if key in self._clients: self._last_used[key] = time.time() async def _idle_reaper_loop(self) -> None: interval = min(60.0, self._idle_timeout) while True: await asyncio.sleep(interval) try: await self._reap_idle_once() except asyncio.CancelledError: raise except Exception as e: # noqa: BLE001 # A transient sweep error must not kill the reaper, or the # unbounded-accumulation leak it exists to fix comes back. logger.debug("LSP idle reaper sweep error: %s", e) async def _reap_idle_once(self) -> None: cutoff = time.time() - self._idle_timeout with self._state_lock: idle_keys = [ key for key in self._clients if self._last_used.get(key, 0) < cutoff ] clients = [self._clients.pop(key) for key in idle_keys] for key in idle_keys: self._last_used.pop(key, None) if clients: eventlog.log_reaped( [(c.server_id, c.workspace_root) for c in clients], self._idle_timeout, ) await asyncio.gather( *(client.shutdown() for client in clients), return_exceptions=True, ) async def _shutdown_async(self) -> None: reaper = self._idle_reaper_task self._idle_reaper_task = None if reaper is not None: reaper.cancel() await asyncio.gather(reaper, return_exceptions=True) with self._state_lock: clients = list(self._clients.values()) self._clients.clear() self._broken.clear() self._last_used.clear() await asyncio.gather( *(c.shutdown() for c in clients), return_exceptions=True, ) def get_status(self) -> Dict[str, Any]: """Return a snapshot of the service for ``hermes lsp status``.""" with self._state_lock: clients = [ { "server_id": k[0], "workspace_root": k[1], "state": c.state, "running": c.is_running, } for k, c in self._clients.items() ] broken = list(self._broken) return { "enabled": self._enabled, "wait_mode": self._wait_mode, "wait_timeout": self._wait_timeout, "install_strategy": self._install_strategy, "clients": clients, "broken": broken, "disabled_servers": sorted(self._disabled_servers), } __all__ = ["LSPService"]