refactor(tools): compact MCP oauth/config/health/lifecycle/agent/content modules

Dead-code removal (_make_hermes_provider_class factory -> direct class def,
_same_endpoint/_context_var_value inlined), image/audio cache unified into
_cache_mcp_media_block, match->isinstance chain, closer hugging, docstring
compaction keeping every invariant. Tool schemas byte-identical.
This commit is contained in:
Teknium
2026-09-02 22:45:12 -07:00
parent 113f04616b
commit ee81b1abdd
9 changed files with 403 additions and 625 deletions
+48 -71
View File
@@ -31,14 +31,13 @@ _stdio_pgids: Dict[int, int] = {}
def _snapshot_child_pids() -> set:
"""Current direct-child PIDs: /proc on Linux, else psutil, else empty set."""
my_pid = os.getpid()
# /proc/<pid>/task/<tid>/children is per-THREAD, and stdio_client() spawns
# from the MCP loop thread, so union every task's children — reading only
# the main thread's file returns an empty set on every Linux install.
# /proc/<pid>/task/<tid>/children is per-THREAD, and stdio_client() spawns from the MCP
# loop thread, so union every task's children — reading only the main thread's file
# returns an empty set on every Linux install.
try:
task_dir = f"/proc/{my_pid}/task"
tids = os.listdir(task_dir)
found: set = set()
for tid in tids:
for tid in os.listdir(task_dir):
try:
with open(f"{task_dir}/{tid}/children", encoding="utf-8") as f:
found.update(int(p) for p in f.read().split() if p.strip())
@@ -63,8 +62,7 @@ _NON_MCP_CHILD_CMDLINE_MARKERS: tuple[str, ...] = (
"tui_gateway.entry",
"-dorg.eclipse.equinox.launcher", # jdtls (legacy arg style)
"eclipse.jdt.ls",
"org.eclipse.equinox.launcher_",
)
"org.eclipse.equinox.launcher_")
def _filter_mcp_children(pids: set) -> set:
@@ -106,50 +104,40 @@ def shutdown_mcp_servers(*, scope: Optional[str] = None):
selected = [name for name in _core._servers if scope is None or _core._server_scope_keys.get(name) == scope]
servers_snapshot = [_core._servers[name] for name in selected]
# Fast path: nothing to shut down. Still clear the connect-cooldown maps —
# a server that failed to connect is never in ``_servers``, so this is the
# most likely state for stale backoff entries; a restart must retry at once.
if not servers_snapshot:
if servers_snapshot:
async def _shutdown():
results = await asyncio.gather(*(server.shutdown() for server in servers_snapshot), return_exceptions=True)
for server, result in zip(servers_snapshot, results):
if isinstance(result, Exception):
logger.debug("Error closing MCP server '%s': %s", server.name, result)
with _core._lock:
for name in selected:
_core._servers.pop(name, None)
_core._server_scope_keys.pop(name, None)
_clear_connect_cooldowns()
with _core._lock:
_clear_connect_cooldowns()
_core._stop_mcp_loop(only_if_idle=scope is not None)
return
loop = _core._mcp_loop
if loop is not None and loop.is_running():
from agent.async_utils import safe_schedule_threadsafe
future = safe_schedule_threadsafe(_shutdown(), loop, logger=logger, log_message="MCP shutdown: failed to schedule")
if future is not None:
try:
future.result(timeout=15)
except BaseException as exc:
logger.debug("Error during MCP shutdown: %s", exc)
async def _shutdown():
results = await asyncio.gather(*(server.shutdown() for server in servers_snapshot), return_exceptions=True)
for server, result in zip(servers_snapshot, results):
if isinstance(result, Exception):
logger.debug("Error closing MCP server '%s': %s", server.name, result)
with _core._lock:
for name in selected:
_core._servers.pop(name, None)
_core._server_scope_keys.pop(name, None)
_clear_connect_cooldowns()
with _core._lock:
loop = _core._mcp_loop
if loop is not None and loop.is_running():
from agent.async_utils import safe_schedule_threadsafe
future = safe_schedule_threadsafe(
_shutdown(), loop, logger=logger, log_message="MCP shutdown: failed to schedule",
)
if future is not None:
try:
future.result(timeout=15)
except BaseException as exc:
logger.debug("Error during MCP shutdown: %s", exc)
# Unconditional final sweep: whether ``_shutdown`` ran, timed out, or was
# never scheduled, no stale connect-cooldown state may survive shutdown.
# Unconditional final sweep: whether ``_shutdown`` ran, timed out, or was never scheduled
# (a server that failed to connect is never in ``_servers`` — the most likely state for
# stale backoff entries), no connect-cooldown state may survive shutdown.
with _core._lock:
_clear_connect_cooldowns()
_core._stop_mcp_loop(only_if_idle=scope is not None)
def _take_reapable_pids(include_active: bool, server_name: Optional[str]) -> tuple[Dict[int, str], Dict[int, int]]:
"""Pop the PIDs to reap (and their spawn-time pgids) out of the ledgers under
the lock, so a future spawn can't collide with stale state.
Returns ``(pid -> owner, pid -> pgid)``."""
"""Pop the PIDs to reap (and their spawn-time pgids) out of the ledgers under the lock, so
a future spawn can't collide with stale state. Returns ``(pid -> owner, pid -> pgid)``."""
def _owned(entries: Dict[int, str]) -> Dict[int, str]:
return {pid: owner for pid, owner in entries.items() if server_name is None or owner == server_name}
@@ -173,25 +161,21 @@ def _signal_mcp_process(pid: int, sig: int, server_name: str, pgid: Optional[int
killpg = getattr(os, "killpg", None)
if pgid is not None and killpg is not None:
if my_pgid is not None and pgid == my_pgid:
# Child shares the gateway's pgroup: killpg would kill the gateway
# too, so use per-pid kill. Warn because per-pid kill can't reach
# grandchildren in this group (inherent trade-off).
# Child shares the gateway's pgroup: killpg would kill the gateway too, so use
# per-pid kill. Warn because per-pid kill can't reach grandchildren in this group.
logger.warning(
"MCP server '%s' pgid %d matches gateway pgid; skipping "
"killpg to avoid self-kill and using per-pid kill — any "
"grandchildren in this group may not be reaped",
server_name, pgid,
)
server_name, pgid)
else:
try:
killpg(pgid, sig)
return
except (ProcessLookupError, PermissionError, OSError) as exc:
# Pgroup gone or refused — still try the direct child.
logger.debug(
"killpg(%d, %d) failed for MCP server '%s': %s; falling back to kill(pid)",
pgid, sig, server_name, exc,
)
logger.debug("killpg(%d, %d) failed for MCP server '%s': %s; falling back to kill(pid)",
pgid, sig, server_name, exc)
try:
os.kill(pid, sig)
except (ProcessLookupError, PermissionError, OSError):
@@ -208,13 +192,10 @@ def _kill_orphaned_mcp_children(include_active: bool = False, server_name: Optio
import signal as _signal
pids, pgids = _take_reapable_pids(include_active, server_name)
# Fast path: nothing to reap — skip the 2s sleep every MCP-free shutdown
# would otherwise pay.
if not pids:
if not pids: # skip the 2s sleep every MCP-free shutdown would otherwise pay
return
# Our own pgid, so we never killpg() the gateway itself.
try:
try: # our own pgid, so we never killpg() the gateway itself
my_pgid = os.getpgrp()
except (AttributeError, OSError):
my_pgid = None # Windows or restricted environment
@@ -236,18 +217,17 @@ def _kill_orphaned_mcp_children(include_active: bool = False, server_name: Optio
def _stop_mcp_loop_if_idle() -> bool:
"""Stop the MCP loop only when no registered server still owns it. Probe
paths create temporary MCPServerTasks not placed in ``_servers``; they may
clean up an idle loop but must not tear down the process-global loop under
live agent tools, or later calls fail with ``MCP event loop is not running``."""
"""Stop the MCP loop only when no registered server still owns it. Probe paths create
temporary MCPServerTasks not placed in ``_servers``; they may clean up an idle loop but
must not tear down the process-global loop under live agent tools."""
return _core._stop_mcp_loop(only_if_idle=True)
async def _drain_mcp_loop_tasks(*, timeout: Optional[float] = None) -> None:
"""Cancel every task still pending on the MCP loop and reap it.
``Task.cancel()`` only schedules the throw, so tasks need a cancellation
cycle before the loop goes away; wait for them here, on their owning loop,
bounded so a task that suppresses cancellation cannot hang process exit."""
"""Cancel every task still pending on the MCP loop and reap it. ``Task.cancel()`` only
schedules the throw, so tasks need a cancellation cycle before the loop goes away; wait
for them here, on their owning loop, bounded so a task that suppresses cancellation
cannot hang process exit."""
if timeout is None:
timeout = _core._MCP_LOOP_DRAIN_TIMEOUT
current = asyncio.current_task()
@@ -257,7 +237,6 @@ async def _drain_mcp_loop_tasks(*, timeout: Optional[float] = None) -> None:
logger.debug("Draining %d pending task(s) from the MCP loop", len(pending))
for task in pending:
task.cancel()
done, still_pending = await asyncio.wait(pending, timeout=timeout)
for task in done:
try:
@@ -267,16 +246,14 @@ async def _drain_mcp_loop_tasks(*, timeout: Optional[float] = None) -> None:
pass
except Exception as exc:
logger.debug("Pending MCP loop task ended during shutdown: %s", exc)
if still_pending:
logger.warning("%d MCP loop task(s) still pending after %.1fs drain", len(still_pending), timeout)
async def _drain_and_stop_mcp_loop() -> None:
"""Drain pending tasks, then stop the loop from its owning thread. Both must
run as one loop-owned sequence: a ``loop.stop`` queued separately by a
timed-out caller can overtake the scheduled drain, leaving the drain
coroutine itself pending when the loop is closed."""
"""Drain pending tasks, then stop the loop from its owning thread. Both must run as one
loop-owned sequence: a ``loop.stop`` queued separately by a timed-out caller can overtake
the scheduled drain, leaving the drain coroutine itself pending when the loop is closed."""
loop = asyncio.get_running_loop()
try:
await _drain_mcp_loop_tasks(timeout=_core._MCP_LOOP_DRAIN_TIMEOUT)