From fb329e4aaa922edb1235e6d08b293a72c483e8ef Mon Sep 17 00:00:00 2001 From: Xi Zhang <106144707+X-iZhang@users.noreply.github.com> Date: Tue, 11 Aug 2026 08:37:58 +0100 Subject: [PATCH] =?UTF-8?q?perf:=20cut=20startup=20latency=20=E2=80=94=20l?= =?UTF-8?q?azy=20import=20surface,=20langgraph=20dev=20keepalive,=20indexe?= =?UTF-8?q?d=20thread=20listing=20(#407)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat: add thread metadata index for improved performance in thread listing - Implemented a new SQLite index on the `checkpoints` table to optimize thread listing queries by indexing relevant metadata fields. - Updated the `list_threads` function to ensure the index is created if it does not exist. - Added a test to verify the creation of the metadata index during thread listing. feat: enhance workspace sidecar management with owner tracking - Modified the workspace sidecar to include `owner_pids` to track the current process owners. - Updated tests to validate the new owner tracking functionality and ensure proper behavior when managing workspace sidecars. chore: introduce model registry for streamlined model management - Created a new `registry.py` file to maintain a comprehensive model registry, including model names, IDs, providers, and routing tables. - Added functions to retrieve models by provider and list available models, enhancing the modularity and maintainability of model management. * feat: enhance workspace sidecar management and improve thread metadata indexing * fix(tests): ensure sidecar correctly registers owner with original workspace and pid * refactor: simplify workspace sidecar management by removing owner tracking * feat(server): add commands to manage background langgraph dev server - Introduced `server_app` for managing the langgraph dev server with commands to check status and stop the server. - Enhanced workspace sidecar management to include configuration fingerprint for drift detection. - Updated deployment functions to handle server configuration and state more effectively. * feat(server): enhance server status command to display PID with stale record warning * feat(langgraph_dev): exclusion-set config fingerprint, webui keepalive, unified stop guidance * fix(cli): platform-specific manual-stop hint; document keepalive endpoint-change limitation --- EvoScientist/cli/_app.py | 6 + EvoScientist/cli/commands.py | 23 +- EvoScientist/cli/server_cmd.py | 99 +++++ EvoScientist/config/settings.py | 10 + EvoScientist/deploy/server.py | 27 +- EvoScientist/deploy/webui.py | 38 +- EvoScientist/gateway/__init__.py | 57 ++- EvoScientist/gateway/types.py | 6 +- EvoScientist/langgraph_dev/manager.py | 253 ++++++++++++- EvoScientist/llm/__init__.py | 7 +- EvoScientist/llm/models.py | 350 +----------------- EvoScientist/llm/registry.py | 343 +++++++++++++++++ EvoScientist/sessions.py | 45 +++ tests/test_cli_deploy.py | 1 + tests/test_langgraph_dev_workspace_sidecar.py | 331 +++++++++++++++++ tests/test_langgraph_manager.py | 5 + tests/test_sessions.py | 14 + tests/test_webui_launcher.py | 32 +- 18 files changed, 1276 insertions(+), 371 deletions(-) create mode 100644 EvoScientist/cli/server_cmd.py create mode 100644 EvoScientist/llm/registry.py diff --git a/EvoScientist/cli/_app.py b/EvoScientist/cli/_app.py index 812a629..a26beb2 100644 --- a/EvoScientist/cli/_app.py +++ b/EvoScientist/cli/_app.py @@ -59,6 +59,12 @@ sessions_app = typer.Typer( ) app.add_typer(sessions_app, name="sessions") +# Background langgraph dev server management — the explicit counterpart to +# langgraph_dev_keepalive: a server that outlives its CLI needs a first-class +# way to inspect and stop it. +server_app = typer.Typer(help="Manage the background langgraph dev server") +app.add_typer(server_app, name="server") + # Configure subcommand group — re-run a single onboarding section. configure_app = typer.Typer( help=( diff --git a/EvoScientist/cli/commands.py b/EvoScientist/cli/commands.py index 6b87afa..8ae937c 100644 --- a/EvoScientist/cli/commands.py +++ b/EvoScientist/cli/commands.py @@ -26,14 +26,15 @@ from ..gateway import ( GraphGateway, GraphTarget, RunRequest, - RuntimeGateways, - create_runtime_gateways, ) from ..llm.context_window import DEFAULT_CONTEXT_WINDOW_FALLBACK, resolve_context_window from ..paths import ensure_dirs, set_active_workspace, set_workspace_root from ..runtime import AsyncRuntime from ..stream.console import console -from . import async_notifier +from . import ( + async_notifier, + server_cmd, # noqa: F401 — registers `EvoSci server` commands +) from ._app import app, channel_app, config_app, configure_app, mcp_app, sessions_app from ._constants import build_metadata from .agent import ( @@ -72,6 +73,7 @@ if TYPE_CHECKING: from langgraph.graph.state import CompiledStateGraph from ..config import EvoScientistConfig + from ..gateway import RuntimeGateways _ASYNC_RUNTIME_META_KEY = "evoscientist.async_runtime" @@ -527,6 +529,16 @@ def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None: console.print(f"[red]{exc}[/red]") raise typer.Exit(1) from exc + from ..langgraph_dev import manager as _lg_manager + + if _lg_manager.CONFIG_DRIFT_SINCE_LAUNCH: + console.print( + "[yellow]⚠ Config changed since the background agent server was " + "launched — async sub-agents still use the old settings. Apply " + "them with [bold]EvoSci server stop[/bold], then restart " + "EvoSci.[/yellow]" + ) + # The backend is shared by every UI mode, so the exposure warning lives # here, not just in deploy/WebUI. Gated on the server being up: warning # about a bind that never happened would be worse than saying nothing. @@ -940,7 +952,7 @@ class ServeRuntimeState: thread_id: str workspace_dir: str | None config: "EvoScientistConfig | None" - runtime_gateways: RuntimeGateways + runtime_gateways: "RuntimeGateways" async_runtime: AsyncRuntime resume_warning_thread_id: str | None = None @@ -1553,6 +1565,8 @@ def serve( console.print("[dim]Loading agent...[/dim]") agent = _load_agent(workspace_dir=ws, config=config, runtime=async_runtime) + from ..gateway import create_runtime_gateways + runtime_gateways = create_runtime_gateways() tid = async_runtime.run_sync( lambda: runtime_gateways.graph_gateway.create_thread( @@ -2412,6 +2426,7 @@ def _main_callback( # Single-shot mode: wrap in persistent checkpointer import asyncio + from ..gateway import create_runtime_gateways from ..sessions import get_checkpointer from ..stream.json_sink import stream_json from .interactive import _wait_for_memory_workers_before_exit, cmd_run diff --git a/EvoScientist/cli/server_cmd.py b/EvoScientist/cli/server_cmd.py new file mode 100644 index 0000000..04e3515 --- /dev/null +++ b/EvoScientist/cli/server_cmd.py @@ -0,0 +1,99 @@ +"""``EvoSci server`` — inspect and stop the background langgraph dev server. + +The explicit counterpart to ``langgraph_dev_keepalive``: an opt-in server +that outlives its CLI needs an equally explicit way to see and stop it. +""" + +from __future__ import annotations + +import sys + +from ..stream.console import console +from ._app import server_app + + +@server_app.command("status") +def server_status() -> None: + """Show the background langgraph dev server's state.""" + from ..config import get_effective_config + from ..langgraph_dev.manager import ( + _DEFAULT_HOST, + _DEFAULT_PORT, + _pid_serves_port, + _read_workspace_sidecar, + is_langgraph_dev_running, + ) + + config = get_effective_config() + port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT)) + host = ( + str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST) + ).strip() or _DEFAULT_HOST + running = is_langgraph_dev_running(port=port, host=host) + sidecar = _read_workspace_sidecar() + + if not running and sidecar is None: + console.print("[dim]No background langgraph dev server is running.[/dim]") + return + state = "[green]running[/green]" if running else "[red]not responding[/red]" + console.print(f"[bold]langgraph dev[/bold] on port {port}: {state}") + if sidecar is not None: + console.print(f" workspace: {sidecar.get('workspace')}") + pid = sidecar.get("pid") + if _pid_serves_port(pid, port): + console.print(f" pid: {pid}") + else: + console.print( + f" pid: {pid} [yellow](stale record — this pid does " + f"not serve port {port})[/yellow]" + ) + elif running: + console.print( + " [yellow]no sidecar — externally managed or pre-keepalive server[/yellow]" + ) + + +@server_app.command("stop") +def server_stop() -> None: + """Stop the background langgraph dev server started by EvoSci.""" + from ..config import get_effective_config + from ..langgraph_dev.manager import ( + _DEFAULT_HOST, + _DEFAULT_PORT, + is_langgraph_dev_running, + stop_recorded_server, + ) + + pid = stop_recorded_server() + if pid is not None: + console.print(f"[green]✓[/green] Stopped langgraph dev (pid {pid}).") + return + config = get_effective_config() + port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT)) + host = ( + str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST) + ).strip() or _DEFAULT_HOST + if is_langgraph_dev_running(port=port, host=host): + # A server without ownership records (crashed session, deleted state + # files) can't be verified as ours — refuse to guess, hand the user + # the manual path instead of a silent no-op. + console.print( + f"[yellow]⚠ A langgraph dev is still serving port {port}, but " + f"EvoSci has no ownership record for it, so it was not " + f"touched.[/yellow]" + ) + if sys.platform == "win32": + manual = ( + f'powershell "Get-NetTCPConnection -LocalPort {port} | ' + f'Select-Object -ExpandProperty OwningProcess | Stop-Process"' + ) + else: + manual = f"kill $(lsof -ti :{port})" + console.print( + f"[dim]If it is yours, stop it manually: [bold]{manual}[/bold][/dim]" + ) + else: + console.print( + "[dim]No EvoSci-owned langgraph dev server to stop " + "(stale state, if any, was cleaned up).[/dim]" + ) diff --git a/EvoScientist/config/settings.py b/EvoScientist/config/settings.py index 118a8b7..ed2b49f 100644 --- a/EvoScientist/config/settings.py +++ b/EvoScientist/config/settings.py @@ -257,6 +257,15 @@ class EvoScientistConfig: # slowdown. langgraph_dev_jobs_per_worker: int = 10 + # Keep the auto-started langgraph dev subprocess running after the CLI + # exits. The next `EvoSci` start in the same workspace reuses it instantly + # instead of paying the cold boot (~15s). Starting in a DIFFERENT workspace + # raises WorkspaceMismatchError with the leftover server's pid — stop it + # manually (the server is pinned to one workspace per process). Known + # limitation: changing langgraph_dev_port/host while a keepalive server + # runs orphans its records — run `EvoSci server stop` before switching. + langgraph_dev_keepalive: bool = False + # Max LangGraph super-steps (LLM call / tool call / sub-agent delegation # each count as 1) before raising GraphRecursionError. Resets on every # ``agent.invoke()`` — i.e., this is per-turn, NOT per-conversation. For @@ -836,6 +845,7 @@ _ENV_MAPPINGS = { "sandbox_execute_timeout": "EVOSCIENTIST_SANDBOX_EXECUTE_TIMEOUT", "langgraph_dev_file_persistence": "EVOSCIENTIST_LANGGRAPH_DEV_FILE_PERSISTENCE", "langgraph_dev_jobs_per_worker": "EVOSCIENTIST_LANGGRAPH_DEV_JOBS_PER_WORKER", + "langgraph_dev_keepalive": "EVOSCIENTIST_LANGGRAPH_DEV_KEEPALIVE", "recursion_limit": "EVOSCIENTIST_RECURSION_LIMIT", "memory_profile_enabled": "EVOSCIENTIST_MEMORY_PROFILE_ENABLED", "memory_observations_enabled": "EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED", diff --git a/EvoScientist/deploy/server.py b/EvoScientist/deploy/server.py index 6976db6..02914ed 100644 --- a/EvoScientist/deploy/server.py +++ b/EvoScientist/deploy/server.py @@ -79,6 +79,9 @@ def deploy( _base_url, _is_loopback_host, _is_port_occupied, + _pid_serves_port, + _read_workspace_sidecar, + _server_config_fingerprint, is_langgraph_dev_running, read_tunnel_url, start_langgraph_dev, @@ -145,10 +148,25 @@ def deploy( f"[red]Port {effective_port} is already serving a langgraph dev " f"instance.[/red]" ) - console.print( - "[dim]Stop the existing EvoSci/serve session first, or use " - "[bold]--port[/bold] to deploy on a different port.[/dim]" - ) + sidecar = _read_workspace_sidecar() + if sidecar is not None and _pid_serves_port( + sidecar.get("pid"), effective_port + ): + # Surface what we know about the occupant — with keepalive it + # may be an ownerless leftover rather than a live session. + # Only when the recorded pid verifiably serves THIS port, so a + # stale or other-port record is never blamed. + console.print( + f"[dim]It serves workspace {sidecar.get('workspace')} " + f"(pid {sidecar.get('pid')}). Stop it with " + f"[bold]EvoSci server stop[/bold], or use " + f"[bold]--port[/bold] to deploy on a different port.[/dim]" + ) + else: + console.print( + "[dim]Stop the existing EvoSci/serve session first, or use " + "[bold]--port[/bold] to deploy on a different port.[/dim]" + ) else: console.print( f"[red]Port {effective_port} is occupied by another process.[/red]" @@ -231,6 +249,7 @@ def deploy( jobs_per_worker=jobs_per_worker, deploy_mode=True, tunnel=tunnel, + config_fingerprint=_server_config_fingerprint(config), ) atexit.register(stop_langgraph_dev, proc) except Exception as exc: diff --git a/EvoScientist/deploy/webui.py b/EvoScientist/deploy/webui.py index eec85df..45138ae 100644 --- a/EvoScientist/deploy/webui.py +++ b/EvoScientist/deploy/webui.py @@ -67,6 +67,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: _is_loopback_host, _is_port_occupied, _read_workspace_sidecar, + _server_config_fingerprint, is_langgraph_dev_running, start_langgraph_dev, stop_langgraph_dev, @@ -163,6 +164,31 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: f"[/dim]" ) raise typer.Exit(1) + if sidecar is not None and sidecar.get("deploy_mode") is False: + # A stripped (CLI-started) server has no MCP and no async + # sub-agents — silently reusing it would degrade the WebUI + # with no visible cause. Refuse; never auto-kill. + console.print( + f"[red]Port {backend_port} is serving a stripped " + f"(CLI-mode) langgraph dev — the WebUI needs the full " + f"deploy-mode server (MCP + async sub-agents).[/red]" + ) + console.print( + "[dim]Stop it with [bold]EvoSci server stop[/bold], then " + "re-run [bold]EvoSci[/bold].[/dim]" + ) + raise typer.Exit(1) + if sidecar is not None: + recorded_fp = sidecar.get("config_fingerprint") + if isinstance( + recorded_fp, str + ) and recorded_fp != _server_config_fingerprint(config): + console.print( + "[yellow]⚠ Config changed since this server was " + "launched — it still serves the old settings. Apply " + "them with [bold]EvoSci server stop[/bold], then " + "re-run EvoSci.[/yellow]" + ) console.print( f"[green]✓[/green] Reusing langgraph dev already serving " f"port {backend_port}" @@ -191,8 +217,18 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: file_persistence=file_persistence, jobs_per_worker=jobs_per_worker, deploy_mode=True, + config_fingerprint=_server_config_fingerprint(config), ) - atexit.register(stop_langgraph_dev, started_proc) + if getattr(config, "langgraph_dev_keepalive", False): + # Keepalive: the deploy-mode backend outlives this session so + # the next same-workspace launch reuses it instantly. The npx + # front-end below still stops on exit as usual. + console.print( + "[dim]keepalive: backend server stays up after exit — " + "stop it with [bold]EvoSci server stop[/bold].[/dim]" + ) + else: + atexit.register(stop_langgraph_dev, started_proc) except Exception as exc: console.print(f"[red]langgraph dev startup failed:[/red] {exc}") raise typer.Exit(1) from exc diff --git a/EvoScientist/gateway/__init__.py b/EvoScientist/gateway/__init__.py index 2c14a66..3397a39 100644 --- a/EvoScientist/gateway/__init__.py +++ b/EvoScientist/gateway/__init__.py @@ -4,19 +4,16 @@ The gateway package is the migration seam between UI surfaces and graph execution. CLI, TUI, channels, and future frontends should depend on this package for thread/run operations instead of reaching directly into ``sessions.py``, ``stream.events``, or the LangGraph SDK. + +Backend implementations are attached lazily via :mod:`lazy_loader` (SPEC-1 / +PEP 562): importing the shared :mod:`.types` protocols must not cascade into +``sessions``/langgraph/langgraph_sdk, which every CLI invocation would pay. """ -from . import background_runs -from .local import LocalGraphGateway, LocalThreadStore -from .runtime import ( - RuntimeGatewayBackend, - RuntimeGateways, - create_runtime_gateways, -) -from .server import ( - LangGraphServerGateway, - LangGraphServerThreadStore, -) +from typing import TYPE_CHECKING + +import lazy_loader as _lazy + from .types import ( DEFAULT_GRAPH_ID, GraphEvent, @@ -29,6 +26,44 @@ from .types import ( ThreadStore, ) +if TYPE_CHECKING: + # Static counterparts of the lazy attach below — type checkers don't + # infer names served through __getattr__. + from . import background_runs + from .local import LocalGraphGateway, LocalThreadStore + from .runtime import ( + RuntimeGatewayBackend, + RuntimeGateways, + create_runtime_gateways, + ) + from .server import ( + LangGraphServerGateway, + LangGraphServerThreadStore, + ) + +__getattr__, _attach_dir, _ = _lazy.attach( + __name__, + submodules=["background_runs"], + submod_attrs={ + "local": ["LocalGraphGateway", "LocalThreadStore"], + "runtime": [ + "RuntimeGatewayBackend", + "RuntimeGateways", + "create_runtime_gateways", + ], + "server": [ + "LangGraphServerGateway", + "LangGraphServerThreadStore", + ], + }, +) + + +def __dir__() -> list[str]: + # attach() only knows the lazy names; include the eager type exports too. + return sorted(set(_attach_dir()) | set(__all__)) + + __all__ = [ "DEFAULT_GRAPH_ID", "GraphEvent", diff --git a/EvoScientist/gateway/types.py b/EvoScientist/gateway/types.py index 01745ea..c658d0a 100644 --- a/EvoScientist/gateway/types.py +++ b/EvoScientist/gateway/types.py @@ -6,15 +6,15 @@ from collections.abc import AsyncIterator from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Protocol, TypeAlias -from langgraph.types import Command - if TYPE_CHECKING: from langgraph.graph.state import CompiledStateGraph + from langgraph.types import Command from ..middleware.events import SessionEvents GraphEvent: TypeAlias = dict[str, Any] -GraphRunInput: TypeAlias = str | Command +# String alias keeps this module langgraph-free at import time (~950 modules). +GraphRunInput: TypeAlias = "str | Command" GraphStateValues: TypeAlias = dict[str, Any] DEFAULT_GRAPH_ID = "EvoScientist" diff --git a/EvoScientist/langgraph_dev/manager.py b/EvoScientist/langgraph_dev/manager.py index 6eec613..c876024 100644 --- a/EvoScientist/langgraph_dev/manager.py +++ b/EvoScientist/langgraph_dev/manager.py @@ -12,6 +12,7 @@ Mirrors the lifecycle pattern used by ``ccproxy_manager.py``. from __future__ import annotations import atexit +import hashlib import json import logging import os @@ -21,6 +22,7 @@ import subprocess import threading import time from dataclasses import dataclass +from dataclasses import fields as dataclass_fields from pathlib import Path import httpx @@ -110,6 +112,13 @@ def needs_langgraph_dev(config: EvoScientistConfig) -> bool: _LOCK = threading.RLock() +# Set by ``ensure_langgraph_dev`` when it reuses a keepalive server whose +# recorded launch-time config fingerprint differs from the current effective +# config. The CLI reads it after startup to surface a "restart to apply" +# hint — the server itself is never restarted automatically. +CONFIG_DRIFT_SINCE_LAUNCH = False + + # Default port (Kaprekar's constant — see config/settings.py for the rationale). # Overridable per-call via ``start_langgraph_dev(port=...)`` / # ``ensure_langgraph_dev`` (which reads ``config.langgraph_dev_port``) and the @@ -211,9 +220,17 @@ class WorkspaceMismatchError(RuntimeError): """ -def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None: +def _write_workspace_sidecar( + workspace_dir: Path, + pid: int, + config_fingerprint: str | None = None, + deploy_mode: bool | None = None, +) -> None: """Record the workspace + pid of the langgraph dev we just started. + ``config_fingerprint`` (optional) captures the launch-time config subset + the server consumed; keepalive reuse compares it to detect drift. + Atomic write via temp-file + ``os.replace``: without this, a concurrent reader could observe a partially-written file, fail JSON parse, and silently downgrade to the "no sidecar" fallback path — which skips the @@ -228,9 +245,12 @@ def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None: try: RUNTIME.pid_dir.mkdir(parents=True, exist_ok=True) tmp = RUNTIME.workspace_sidecar.with_suffix(".json.tmp") - tmp.write_text( - json.dumps({"workspace": str(workspace_dir), "pid": pid}), encoding="utf-8" - ) + payload: dict = {"workspace": str(workspace_dir), "pid": pid} + if config_fingerprint is not None: + payload["config_fingerprint"] = config_fingerprint + if deploy_mode is not None: + payload["deploy_mode"] = deploy_mode + tmp.write_text(json.dumps(payload), encoding="utf-8") os.replace(tmp, RUNTIME.workspace_sidecar) except OSError as exc: logger.warning( @@ -569,6 +589,186 @@ def _kill_owned_stale_process(port: int) -> bool: return True +# Config fields that provably never reach the langgraph dev subprocess: +# the channel stack + STT run in the CLI process, display/workspace/frontend +# knobs shape the CLI itself, and keepalive is a lifecycle flag. Everything +# NOT listed here counts toward the drift fingerprint, so a newly added +# config field defaults to "affects the server" — the failure mode is a +# spurious restart hint, never silent staleness. +# Packaged sub-agent specs — consumed at graph build; module constant so +# tests can redirect it. +_SUBAGENTS_DIR = Path(__file__).resolve().parent.parent / "subagents" + +_FINGERPRINT_EXCLUDED_PREFIXES = ( + "channel_", + "imessage_", + "telegram_", + "discord_", + "slack_", + "feishu_", + "wechat_", + "dingtalk_", + "email_", + "qq_", + "signal_", + "stt_", +) +_FINGERPRINT_EXCLUDED_FIELDS = frozenset( + { + "require_mention", + "text_chunk_limit", + "allowed_channels", + "dm_policy", + "shared_webhook_port", + "show_thinking", + "ui_backend", + "log_level", + "default_mode", + "default_workdir", + "webui_port", + "webui_host", + "langgraph_dev_keepalive", + "shell_allow_list", + } +) + + +def _server_config_fingerprint(config: EvoScientistConfig) -> str: + """Hash of everything the langgraph dev subprocess consumes at launch. + + Deployed graphs read config once at import (``subagents/_factory.py``, + ``EvoScientist.py``), so a keepalive server keeps serving those values + until restarted. Iterates the full ``EvoScientistConfig`` field list + minus the explicit exclusion set above — a new config field counts + toward drift by default — and folds in ``mcp.yaml`` plus the packaged + ``subagents/*.yaml``, which are consumed at graph build too. Secrets + only feed a truncated one-way digest; nothing recoverable is stored. + getattr with defaults: deploy/WebUI (and their tests) routinely hand + this module duck-typed config objects missing dataclass fields. + """ + parts = [] + for field in dataclass_fields(EvoScientistConfig): + name = field.name + if name in _FINGERPRINT_EXCLUDED_FIELDS or name.startswith( + _FINGERPRINT_EXCLUDED_PREFIXES + ): + continue + parts.append((name, str(getattr(config, name, None)))) + digest = hashlib.sha256(repr(parts).encode("utf-8")) + try: + from EvoScientist.config.settings import get_config_dir + + mcp_yaml = get_config_dir() / "mcp.yaml" + if mcp_yaml.exists(): + digest.update(mcp_yaml.read_bytes()) + except OSError: + pass + try: + for yaml_path in sorted(_SUBAGENTS_DIR.glob("*.yaml")): + digest.update(yaml_path.name.encode("utf-8")) + digest.update(yaml_path.read_bytes()) + except OSError: + pass + return digest.hexdigest()[:16] + + +def stop_recorded_server() -> int | None: + """Explicitly stop the langgraph dev recorded in our PID file. + + Backs the user-facing ``EvoSci server stop`` command — the deliberate + counterpart to ``langgraph_dev_keepalive``: an opt-in server that + outlives its CLI needs a first-class way to stop it. Ownership = our + PID file + a live process whose cmdline still contains ``langgraph`` + (same loose anti-PID-recycling match as ``_kill_owned_stale_process``, + with PID-file ownership as the primary guard). Holds the cross-process + file lock so a concurrent start can't have its fresh PID/sidecar records + wiped by this stop's cleanup. Kills the whole process tree, then removes + the PID file + sidecar. Returns the stopped pid, or ``None`` when nothing + was stopped (stale/corrupt files, if any, are still cleaned up). + """ + try: + with FileLock(str(RUNTIME.lock_file), timeout=_FILE_LOCK_TIMEOUT): + return _stop_recorded_server_locked() + except FileLockTimeout: + logger.warning( + "Timed out waiting for the langgraph dev lock — another EvoSci " + "process is mid lifecycle change; not stopping anything." + ) + return None + + +def _stop_recorded_server_locked() -> int | None: + with _LOCK: + if _PROCESS is not None and _PROCESS.poll() is None: + pid = _PROCESS.pid + stop_langgraph_dev() + return pid + if not RUNTIME.pid_file.exists(): + return None + try: + owned_pid = int(RUNTIME.pid_file.read_text(encoding="utf-8").strip()) + except ValueError: + stop_langgraph_dev() # corrupt PID file — clean it up as promised + return None + except OSError: + return None + try: + proc = psutil.Process(owned_pid) + cmdline = proc.cmdline() + except (psutil.NoSuchProcess, psutil.AccessDenied): + stop_langgraph_dev() # dead/inaccessible — clean the stale files + return None + if not any("langgraph" in arg for arg in cmdline): + stop_langgraph_dev() # pid recycled by a foreign process — files only + return None + try: + children = proc.children(recursive=True) + for child in children: + try: + child.terminate() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + proc.terminate() + try: + proc.wait(timeout=5) + except psutil.TimeoutExpired: + for child in children: + try: + child.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + proc.kill() + # The parent exiting promptly doesn't prove its workers did — sweep + # the pre-kill snapshot for survivors. + for child in children: + try: + if child.is_running(): + child.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + stop_langgraph_dev() + return owned_pid + + +def _pid_serves_port(pid: object, port: int) -> bool: + """Best-effort check that ``pid`` is a langgraph dev serving ``port``. + + Used to attribute an occupied port to the sidecar's recorded server + before printing its details — avoids blaming a stale record. Relies on + ``--port`` always being in ``start_langgraph_dev``'s argv, not on + port→PID mapping (root-only on macOS via psutil). + """ + if not isinstance(pid, int) or isinstance(pid, bool) or pid <= 0: + return False + try: + cmdline = psutil.Process(pid).cmdline() + except (psutil.NoSuchProcess, psutil.AccessDenied): + return False + return any("langgraph" in arg for arg in cmdline) and str(port) in cmdline + + def _packaged_langgraph_config() -> Path: """Return path to the package-shipped ``langgraph.json``. @@ -595,6 +795,7 @@ def start_langgraph_dev( jobs_per_worker: int = 10, deploy_mode: bool = False, tunnel: bool = False, + config_fingerprint: str | None = None, ) -> subprocess.Popen: """Start langgraph dev as a background subprocess. @@ -816,7 +1017,12 @@ def start_langgraph_dev( except Exception: pass RUNTIME.pid_file.write_text(str(proc.pid), encoding="utf-8") - _write_workspace_sidecar(workspace_dir=workspace_dir, pid=proc.pid) + _write_workspace_sidecar( + workspace_dir=workspace_dir, + pid=proc.pid, + config_fingerprint=config_fingerprint, + deploy_mode=deploy_mode, + ) global _PROCESS_WORKSPACE _PROCESS = proc _PROCESS_WORKSPACE = workspace_dir @@ -993,7 +1199,8 @@ def ensure_langgraph_dev( still chat with sync sub-agents; only async sub-agent calls and EvoMemory background workers will fail. """ - global _ASYNC_SUBAGENTS_AVAILABLE + global _ASYNC_SUBAGENTS_AVAILABLE, CONFIG_DRIFT_SINCE_LAUNCH + CONFIG_DRIFT_SINCE_LAUNCH = False if not needs_langgraph_dev(config): _ASYNC_SUBAGENTS_AVAILABLE = False @@ -1032,7 +1239,8 @@ def _ensure_langgraph_dev_locked( workspace_dir: Path | str | None, ) -> subprocess.Popen | None: """Locked critical section of ``ensure_langgraph_dev`` — must hold ``_LOCK``.""" - global _ASYNC_SUBAGENTS_AVAILABLE + global _ASYNC_SUBAGENTS_AVAILABLE, CONFIG_DRIFT_SINCE_LAUNCH + config_fp = _server_config_fingerprint(config) port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT)) host = str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST) file_persistence = bool(getattr(config, "langgraph_dev_file_persistence", True)) @@ -1086,12 +1294,33 @@ def _ensure_langgraph_dev_locked( if sidecar is not None: recorded = Path(sidecar["workspace"]).resolve() if recorded != ws_path.resolve(): + hint = "" + if getattr(config, "langgraph_dev_keepalive", False): + # Only under keepalive can the server be an ownerless + # leftover; without the flag the mismatch means a live + # session, where a stop suggestion would be misleading. + # Point at `EvoSci server stop` (not a raw kill): it + # verifies ownership and cleans the PID/sidecar files, + # so no stale records are left behind. + hint = ( + " If it is a leftover keepalive server, stop it" + " with: EvoSci server stop." + ) raise WorkspaceMismatchError( f"An EvoSci langgraph dev is already running on " f"{_base_url(port, host)} for workspace {recorded}, but the " f"current process requested workspace {ws_path}. " f"Stop the other EvoSci session (deploy / TUI / serve) " - f"or rerun with --workdir {recorded}." + f"or rerun with --workdir {recorded}." + hint + ) + recorded_fp = sidecar.get("config_fingerprint") + if isinstance(recorded_fp, str) and recorded_fp != config_fp: + CONFIG_DRIFT_SINCE_LAUNCH = True + logger.warning( + "Config changed since the running langgraph dev was " + "launched — async sub-agents still use the old " + "settings until the server is restarted " + "(EvoSci server stop)." ) logger.info( "Reusing externally-managed langgraph dev on %s; sidecar " @@ -1125,6 +1354,7 @@ def _ensure_langgraph_dev_locked( host=host, file_persistence=file_persistence, jobs_per_worker=jobs_per_worker, + config_fingerprint=config_fp, ) except (FileNotFoundError, RuntimeError) as exc: # Startup failed — keep async subagents disabled so the main agent @@ -1141,5 +1371,10 @@ def _ensure_langgraph_dev_locked( return None _ASYNC_SUBAGENTS_AVAILABLE = True - atexit.register(stop_langgraph_dev, proc) + if getattr(config, "langgraph_dev_keepalive", False): + # Keepalive: leave the server (plus PID file + sidecar) behind on CLI + # exit so the next start in this workspace reuses it instantly. + logger.info("langgraph_dev_keepalive enabled — server will outlive this CLI.") + else: + atexit.register(stop_langgraph_dev, proc) return proc diff --git a/EvoScientist/llm/__init__.py b/EvoScientist/llm/__init__.py index 1704632..d0af1f0 100644 --- a/EvoScientist/llm/__init__.py +++ b/EvoScientist/llm/__init__.py @@ -14,7 +14,7 @@ import lazy_loader as _lazy __getattr__, __dir__, __all__ = _lazy.attach( __name__, - submodules=["context_window", "models", "patches"], + submodules=["context_window", "models", "patches", "registry"], submod_attrs={ "context_window": [ "DEFAULT_CONTEXT_WINDOW_FALLBACK", @@ -22,9 +22,12 @@ __getattr__, __dir__, __all__ = _lazy.attach( "resolve_context_window", ], "models": [ + "get_chat_model", + ], + # Registry data resolves without the langchain/provider-SDK stack. + "registry": [ "DEFAULT_MODEL", "MODELS", - "get_chat_model", "get_model_info", "get_models_for_provider", "list_models", diff --git a/EvoScientist/llm/models.py b/EvoScientist/llm/models.py index ef56c27..7096d29 100644 --- a/EvoScientist/llm/models.py +++ b/EvoScientist/llm/models.py @@ -36,21 +36,21 @@ from .patches import ( _patch_openrouter_strip_responses_reasoning, _patch_openrouter_structured_output, ) - -_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic" -_SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1" - -_ZHIPU_BASE_URL = "https://open.bigmodel.cn/api/paas/v4" -_ZHIPU_CODE_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4" -_VOLCENGINE_BASE_URL = "https://ark.cn-beijing.volces.com/api/v3" -_VOLCENGINE_CODE_BASE_URL = "https://ark.cn-beijing.volces.com/api/coding/v3" -_DASHSCOPE_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1" -_DASHSCOPE_CODE_BASE_URL = "https://coding.dashscope.aliyuncs.com/v1" - -_ATLASCLOUD_BASE_URL = "https://api.atlascloud.ai/v1" -_MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1" -_KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/" -_REQUESTY_BASE_URL = "https://router.requesty.ai/v1" +from .registry import ( + _ANTHROPIC_ROUTED_PROVIDERS, + _MODEL_ENTRIES, + _OPENAI_ROUTED_PROVIDERS, + _OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS, # noqa: F401 — re-exported + _THINKING_CAPABLE_PROVIDERS, + DEFAULT_MODEL, + MODELS, + _is_mandatory_thinking_kimi, + get_model_info, # noqa: F401 — re-exported for existing import sites + get_models_for_provider, # noqa: F401 — re-exported for existing import sites + list_model_picker_entries, # noqa: F401 — re-exported for existing import sites + list_models, # noqa: F401 — re-exported for existing import sites + list_models_by_provider, # noqa: F401 — re-exported for existing import sites +) # Minimum Codex CLI version advertised when no explicit override is set. Newer # installed versions are advertised automatically. @@ -106,36 +106,6 @@ def _is_deepseek_endpoint(base_url: str | None) -> bool: return False -# Providers routed through the OpenAI provider with a custom base_url. -# Maps provider name → (base_url or None, env var for API key). -_OPENAI_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = { - "atlascloud": (_ATLASCLOUD_BASE_URL, "ATLASCLOUD_API_KEY"), - "moonshot": (_MOONSHOT_BASE_URL, "MOONSHOT_API_KEY"), - "siliconflow": (_SILICONFLOW_BASE_URL, "SILICONFLOW_API_KEY"), - "zhipu": (_ZHIPU_BASE_URL, "ZHIPU_API_KEY"), - "zhipu-code": (_ZHIPU_CODE_BASE_URL, "ZHIPU_API_KEY"), - "volcengine": (_VOLCENGINE_BASE_URL, "VOLCENGINE_API_KEY"), - "volcengine-code": (_VOLCENGINE_CODE_BASE_URL, "VOLCENGINE_API_KEY"), - "dashscope": (_DASHSCOPE_BASE_URL, "DASHSCOPE_API_KEY"), - "dashscope-code": (_DASHSCOPE_CODE_BASE_URL, "DASHSCOPE_API_KEY"), - "requesty": (_REQUESTY_BASE_URL, "REQUESTY_API_KEY"), - "custom-openai": ( - None, - "CUSTOM_OPENAI_API_KEY", - ), # base_url from CUSTOM_OPENAI_BASE_URL env -} - -# Providers routed through the Anthropic provider with a custom base_url. -# Maps provider name → (base_url or None, env var for API key). -_ANTHROPIC_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = { - "minimax": (_MINIMAX_ANTHROPIC_BASE_URL, "MINIMAX_API_KEY"), - "kimi-coding": (_KIMI_CODING_BASE_URL, "KIMI_API_KEY"), - "custom-anthropic": (None, "CUSTOM_ANTHROPIC_API_KEY"), -} - -# Anthropic-routed providers that support extended thinking. -_THINKING_CAPABLE_PROVIDERS: set[str] = {"minimax"} - _TRUTHY_ENV_VALUES = {"1", "true", "yes", "on"} _FALSEY_ENV_VALUES = {"0", "false", "no", "off"} @@ -147,233 +117,6 @@ _FALSEY_ENV_VALUES = {"0", "false", "no", "off"} # capped to this many below. https://openrouter.ai/docs/app-attribution _OPENROUTER_MAX_CATEGORIES_PER_REQUEST = 2 -# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3 -# cannot disable thinking — structured output must use json_schema there. -# Moonshot-specific: do NOT widen to other mandatory-reasoning models. -_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset( - {"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"} -) - - -def _is_mandatory_thinking_kimi(model_id: str) -> bool: - """True for Kimi models whose thinking cannot be disabled (K3 family).""" - short_id = model_id.split("/")[-1] - return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding" - - -# Model registry: list of (short_name, model_id, provider) -# Allows same short_name across different providers. -_MODEL_ENTRIES: list[tuple[str, str, str]] = [ - # Custom Anthropic (third-party Claude-compatible endpoints, current-gen defaults) - # Listed BEFORE native anthropic so MODELS dict defaults to native provider - ("claude-sonnet-4-6", "claude-sonnet-4-6", "custom-anthropic"), - ("claude-haiku-4-5", "claude-haiku-4-5", "custom-anthropic"), - # Custom OpenAI (third-party OpenAI-compatible endpoints, 3 defaults) - # Listed BEFORE native openai so MODELS dict defaults to native provider - ("gpt-5.5-pro", "gpt-5.5-pro", "custom-openai"), - ("gpt-5.5", "gpt-5.5", "custom-openai"), - ("gpt-5.4", "gpt-5.4", "custom-openai"), - ("gpt-5.3-codex", "gpt-5.3-codex", "custom-openai"), - ("gpt-5-mini", "gpt-5-mini", "custom-openai"), - # Atlas Cloud (OpenAI-compatible) - ("qwen3.5-27b", "qwen/qwen3.5-27b", "atlascloud"), - # Anthropic (current generation) - ("claude-fable-5", "claude-fable-5", "anthropic"), - ("claude-opus-5", "claude-opus-5", "anthropic"), - ("claude-opus-4-8", "claude-opus-4-8", "anthropic"), - ("claude-sonnet-5", "claude-sonnet-5", "anthropic"), - ("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"), - ("claude-haiku-4-5", "claude-haiku-4-5", "anthropic"), - # OpenAI - ("gpt-5.6-sol", "gpt-5.6-sol", "openai"), - ("gpt-5.6-terra", "gpt-5.6-terra", "openai"), - ("gpt-5.6-luna", "gpt-5.6-luna", "openai"), - ("gpt-5.5-pro", "gpt-5.5-pro", "openai"), - ("gpt-5.5", "gpt-5.5", "openai"), - ("gpt-5.4", "gpt-5.4", "openai"), - ("gpt-5.4-mini", "gpt-5.4-mini", "openai"), - ("gpt-5.4-nano", "gpt-5.4-nano", "openai"), - ("gpt-5.3-codex", "gpt-5.3-codex", "openai"), - ("gpt-5.2-codex", "gpt-5.2-codex", "openai"), - ("gpt-5.2", "gpt-5.2", "openai"), - ("gpt-5.1", "gpt-5.1", "openai"), - ("gpt-5", "gpt-5", "openai"), - ("gpt-5-mini", "gpt-5-mini", "openai"), - ("gpt-5-nano", "gpt-5-nano", "openai"), - # Google GenAI - ("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"), - ("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"), - ("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"), - ("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"), - ( - "gemini-3.1-pro-customtools", - "gemini-3.1-pro-preview-customtools", - "google-genai", - ), - ("gemini-3.1-flash-lite", "gemini-3.1-flash-lite-preview", "google-genai"), - ("gemini-3-flash", "gemini-3-flash-preview", "google-genai"), - ("gemini-2.5-flash", "gemini-2.5-flash", "google-genai"), - ("gemini-2.5-flash-lite", "gemini-2.5-flash-lite", "google-genai"), - ("gemini-2.5-pro", "gemini-2.5-pro", "google-genai"), - # MiniMax (direct API — Anthropic-compatible; default: api.minimaxi.com, global: api.minimax.io) - ("minimax-m3", "MiniMax-M3", "minimax"), - ("minimax-m2.7", "MiniMax-M2.7", "minimax"), - ("minimax-m2.7-highspeed", "MiniMax-M2.7-highspeed", "minimax"), - ("minimax-m2.5", "MiniMax-M2.5", "minimax"), - ("minimax-m2.5-highspeed", "MiniMax-M2.5-highspeed", "minimax"), - # NVIDIA - ("nemotron-super", "nvidia/nemotron-3-super-120b-a12b", "nvidia"), - ("nemotron-nano", "nvidia/nemotron-3-nano-30b-a3b", "nvidia"), - ("glm-5.2", "z-ai/glm-5.2", "nvidia"), - ("glm4.7", "z-ai/glm4.7", "nvidia"), - ("deepseek-v3.2", "deepseek-ai/deepseek-v3.2", "nvidia"), - ("deepseek-v3.1", "deepseek-ai/deepseek-v3.1-terminus", "nvidia"), - ("kimi-k2.5", "moonshotai/kimi-k2.5", "nvidia"), - ("kimi-k2-thinking", "moonshotai/kimi-k2-thinking", "nvidia"), - ("minimax-m2.5", "minimaxai/minimax-m2.5", "nvidia"), - ("minimax-m2.1", "minimaxai/minimax-m2.1", "nvidia"), - ("qwen3.5-397b", "qwen/qwen3.5-397b-a17b", "nvidia"), - ("step-3.5-flash", "stepfun-ai/step-3.5-flash", "nvidia"), - # SiliconFlow - ("minimax-m2.5", "Pro/MiniMaxAI/MiniMax-M2.5", "siliconflow"), - ("glm-5.2", "Pro/zai-org/GLM-5.2", "siliconflow"), - ("glm-5", "Pro/zai-org/GLM-5", "siliconflow"), - ("kimi-k2.5", "Pro/moonshotai/Kimi-K2.5", "siliconflow"), - ("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"), - # Requesty (aggregator — OpenAI-compatible router, provider/model IDs). - # Listed before OpenRouter so that for model names shared with OpenRouter - # or a native provider, Requesty does not override them (the dict below is - # last-entry-wins); Requesty is selected explicitly via get_models_for_provider. - ("claude-sonnet-4.6", "anthropic/claude-sonnet-4-6", "requesty"), - ("claude-opus-4.8", "anthropic/claude-opus-4-8", "requesty"), - ("gemini-3.5-flash", "google/gemini-3.5-flash", "requesty"), - ("grok-4.3", "xai/grok-4.3", "requesty"), - ("grok-build-0.1", "xai/grok-build-0.1", "requesty"), - # OpenRouter - ("claude-fable-5", "anthropic/claude-fable-5", "openrouter"), - ("claude-opus-5", "anthropic/claude-opus-5", "openrouter"), - ("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"), - ("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"), - ("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"), - ("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"), - ("claude-sonnet-4.6", "anthropic/claude-sonnet-4.6", "openrouter"), - ("gpt-5.6-sol", "openai/gpt-5.6-sol", "openrouter"), - ("gpt-5.6-terra", "openai/gpt-5.6-terra", "openrouter"), - ("gpt-5.6-luna", "openai/gpt-5.6-luna", "openrouter"), - ("gpt-5.5-pro", "openai/gpt-5.5-pro", "openrouter"), - ("gpt-5.5", "openai/gpt-5.5", "openrouter"), - ("gpt-5.4", "openai/gpt-5.4", "openrouter"), - ("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"), - ("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"), - ("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"), - ("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"), - ("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"), - ("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"), - ("kimi-k3", "moonshotai/kimi-k3", "openrouter"), - ("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"), - ("glm-5.2", "z-ai/glm-5.2", "openrouter"), - ("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"), - ("minimax-m3", "minimax/minimax-m3", "openrouter"), - ("mimo-v2.5-pro", "xiaomi/mimo-v2.5-pro", "openrouter"), - ("mimo-v2.5", "xiaomi/mimo-v2.5", "openrouter"), - ("grok-build-0.1", "x-ai/grok-build-0.1", "openrouter"), - ("grok-4.5", "x-ai/grok-4.5", "openrouter"), - ("hy3", "tencent/hy3", "openrouter"), - ("qwen3.8-max", "qwen/qwen3.8-max", "openrouter"), - ("qwen3.7-max", "qwen/qwen3.7-max", "openrouter"), - ("qwen3.7-plus", "qwen/qwen3.7-plus", "openrouter"), - ("qwen3.6-flash", "qwen/qwen3.6-flash", "openrouter"), - ("qwen3.5-122b", "qwen/qwen3.5-122b-a10b", "openrouter"), - ("deepseek-v4-pro", "deepseek/deepseek-v4-pro", "openrouter"), - ("deepseek-v4-flash", "deepseek/deepseek-v4-flash", "openrouter"), - # Volcengine Coding Plan (火山引擎代码计划 — coding-only endpoint) - # Listed before Zhipu so simple GLM lookups keep their existing default. - ("glm-5.2", "glm-5-2", "volcengine-code"), - ("kimi-k2.5", "kimi-k2-5", "volcengine-code"), - # Zhipu CodePlan (智谱代码计划 — coding-only endpoint) - ("glm-5.2", "glm-5.2", "zhipu-code"), - ("glm-5.1", "glm-5.1", "zhipu-code"), - ("glm-5", "glm-5", "zhipu-code"), - ("glm-5-turbo", "glm-5-turbo", "zhipu-code"), - ("glm-5v-turbo", "glm-5v-turbo", "zhipu-code"), - ("glm-4.7", "glm-4.7", "zhipu-code"), - # Zhipu (智谱 — general endpoint, default for simple lookups) - ("glm-5.2", "glm-5.2", "zhipu"), - ("glm-5.1", "glm-5.1", "zhipu"), - ("glm-5", "glm-5", "zhipu"), - ("glm-5-turbo", "glm-5-turbo", "zhipu"), - ("glm-5v-turbo", "glm-5v-turbo", "zhipu"), - ("glm-4.7", "glm-4.7", "zhipu"), - # Volcengine (火山引擎 — Doubao models) - ("doubao-seed-2.0-pro", "doubao-seed-2-0-pro-260215", "volcengine"), - ("doubao-seed-2.0-lite", "doubao-seed-2-0-lite-260215", "volcengine"), - ("doubao-seed-2.0-mini", "doubao-seed-2-0-mini-260215", "volcengine"), - ("doubao-seed-2.0-code", "doubao-seed-2-0-code-preview-260215", "volcengine"), - ("doubao-seed-1.6", "doubao-seed-1.6", "volcengine"), - ("doubao-1.5-pro", "doubao-1.5-pro-256k", "volcengine"), - ("doubao-1.5-thinking-pro", "doubao-1.5-thinking-pro", "volcengine"), - # DashScope Coding Plan (阿里云代码计划 — subscription sk-sp-* endpoint) - ("qwen3.8-max", "qwen3.8-max", "dashscope-code"), - ("qwen3.7-max", "qwen3.7-max", "dashscope-code"), - ("qwen3.7-plus", "qwen3.7-plus", "dashscope-code"), - ("qwen3.6-max", "qwen3.6-max-preview", "dashscope-code"), - ("qwen3.6-plus", "qwen3.6-plus", "dashscope-code"), - ("qwen3.6-flash", "qwen3.6-flash", "dashscope-code"), - ("qwen3-coder", "qwen3-coder-plus", "dashscope-code"), - ("qwen3-coder-next", "qwen3-coder-next", "dashscope-code"), - ("qwen3-max", "qwen3-max", "dashscope-code"), - ("qwen3.5-plus", "qwen3.5-plus", "dashscope-code"), - # DashScope (阿里云 — Qwen models, default for simple lookups) - ("qwen3.8-max", "qwen3.8-max", "dashscope"), - ("qwen3.7-max", "qwen3.7-max", "dashscope"), - ("qwen3.7-plus", "qwen3.7-plus", "dashscope"), - ("qwen3.6-max", "qwen3.6-max-preview", "dashscope"), - ("qwen3.6-plus", "qwen3.6-plus", "dashscope"), - ("qwen3.6-flash", "qwen3.6-flash", "dashscope"), - ("qwen3-coder", "qwen3-coder-plus", "dashscope"), - ("qwen3-235b", "qwen3-235b-a22b", "dashscope"), - ("qwen-max", "qwen-max", "dashscope"), - ("qwq-plus", "qwq-plus", "dashscope"), - # DeepSeek - ("deepseek-v4-pro", "deepseek-v4-pro", "deepseek"), - ("deepseek-v4-flash", "deepseek-v4-flash", "deepseek"), - # Legacy aliases (deprecated 2026-07-24; route to v4-flash thinking/non-thinking) - ("deepseek-r1", "deepseek-reasoner", "deepseek"), - ("deepseek-v3", "deepseek-chat", "deepseek"), - # Moonshot (OpenAI-compatible) - ("kimi-k3", "kimi-k3", "moonshot"), - ("kimi-k2.6", "kimi-k2.6", "moonshot"), - ("kimi-k2.5", "kimi-k2.5", "moonshot"), - ("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"), - ("kimi-k2-thinking-turbo", "kimi-k2-thinking-turbo", "moonshot"), - ("moonshot-v1-auto", "moonshot-v1-auto", "moonshot"), - ("moonshot-v1-128k", "moonshot-v1-128k", "moonshot"), - ("moonshot-v1-32k", "moonshot-v1-32k", "moonshot"), - ("moonshot-v1-8k", "moonshot-v1-8k", "moonshot"), - # Kimi Coding Plan (Anthropic-compatible) - ("kimi-for-coding", "kimi-for-coding", "kimi-coding"), -] - -# Public dict for simple lookups (last entry wins for duplicate names). -# Use get_models_for_provider() for provider-aware lookups. -MODELS: dict[str, tuple[str, str]] = { - name: (model_id, provider) for name, model_id, provider in _MODEL_ENTRIES -} - -DEFAULT_MODEL = "claude-sonnet-4-6" - - -def get_models_for_provider(provider: str) -> list[tuple[str, str]]: - """Get all models for a specific provider. - - Args: - provider: Provider name (e.g., 'anthropic', 'openrouter'). - - Returns: - List of (short_name, model_id) tuples for the provider. - """ - return [(name, model_id) for name, model_id, p in _MODEL_ENTRIES if p == provider] - def _env_flag_enabled(name: str) -> bool: return os.environ.get(name, "").strip().lower() in _TRUTHY_ENV_VALUES @@ -839,66 +582,3 @@ def get_chat_model( apply_known_context_window(chat_model) return chat_model - - -def list_models() -> list[str]: - """List all available model short names. - - Returns: - List of unique model short names that can be passed to get_chat_model(). - """ - seen = set() - result = [] - for name, _, _ in _MODEL_ENTRIES: - if name not in seen: - seen.add(name) - result.append(name) - return result - - -def list_models_by_provider() -> list[tuple[str, str, str]]: - """List all unique (short_name, model_id, provider) entries. - - Returns: - De-duplicated list of model entries preserving registry order. - """ - seen: set[tuple[str, str]] = set() - result: list[tuple[str, str, str]] = [] - for name, model_id, provider in _MODEL_ENTRIES: - key = (name, provider) - if key not in seen: - seen.add(key) - result.append((name, model_id, provider)) - return result - - -async def list_model_picker_entries( - ollama_base_url: str | None, - *, - include_custom_ollama: bool, -) -> list[tuple[str, str, str]]: - """Return model picker entries, optionally including local Ollama models.""" - entries = list_models_by_provider() - if ollama_base_url: - from .ollama_discovery import discover_ollama_models - - for detected_name in await discover_ollama_models( - ollama_base_url, - timeout=1.5, - ): - entries.append((detected_name, detected_name, "ollama")) - if include_custom_ollama: - entries.append(("Custom Ollama model...", "__custom_ollama__", "ollama")) - return entries - - -def get_model_info(model: str) -> tuple[str, str] | None: - """Get the (model_id, provider) tuple for a short name. - - Args: - model: Short model name. - - Returns: - Tuple of (model_id, provider) or None if not found. - """ - return MODELS.get(model) diff --git a/EvoScientist/llm/registry.py b/EvoScientist/llm/registry.py new file mode 100644 index 0000000..d558e60 --- /dev/null +++ b/EvoScientist/llm/registry.py @@ -0,0 +1,343 @@ +"""Model registry data — short names, model ids, providers, routing tables. + +Pure data with no langchain/provider-SDK imports: the onboard wizard, the +``/model`` pickers, and provider validation read this registry without paying +for the chat-model construction stack in :mod:`.models` (~2000 modules). +""" + +from __future__ import annotations + +_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic" +_SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1" + +_ZHIPU_BASE_URL = "https://open.bigmodel.cn/api/paas/v4" +_ZHIPU_CODE_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4" +_VOLCENGINE_BASE_URL = "https://ark.cn-beijing.volces.com/api/v3" +_VOLCENGINE_CODE_BASE_URL = "https://ark.cn-beijing.volces.com/api/coding/v3" +_DASHSCOPE_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1" +_DASHSCOPE_CODE_BASE_URL = "https://coding.dashscope.aliyuncs.com/v1" + +_ATLASCLOUD_BASE_URL = "https://api.atlascloud.ai/v1" +_MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1" +_KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/" +_REQUESTY_BASE_URL = "https://router.requesty.ai/v1" + +# Providers routed through the OpenAI provider with a custom base_url. +# Maps provider name → (base_url or None, env var for API key). +_OPENAI_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = { + "atlascloud": (_ATLASCLOUD_BASE_URL, "ATLASCLOUD_API_KEY"), + "moonshot": (_MOONSHOT_BASE_URL, "MOONSHOT_API_KEY"), + "siliconflow": (_SILICONFLOW_BASE_URL, "SILICONFLOW_API_KEY"), + "zhipu": (_ZHIPU_BASE_URL, "ZHIPU_API_KEY"), + "zhipu-code": (_ZHIPU_CODE_BASE_URL, "ZHIPU_API_KEY"), + "volcengine": (_VOLCENGINE_BASE_URL, "VOLCENGINE_API_KEY"), + "volcengine-code": (_VOLCENGINE_CODE_BASE_URL, "VOLCENGINE_API_KEY"), + "dashscope": (_DASHSCOPE_BASE_URL, "DASHSCOPE_API_KEY"), + "dashscope-code": (_DASHSCOPE_CODE_BASE_URL, "DASHSCOPE_API_KEY"), + "requesty": (_REQUESTY_BASE_URL, "REQUESTY_API_KEY"), + "custom-openai": ( + None, + "CUSTOM_OPENAI_API_KEY", + ), # base_url from CUSTOM_OPENAI_BASE_URL env +} + +# Providers routed through the Anthropic provider with a custom base_url. +# Maps provider name → (base_url or None, env var for API key). +_ANTHROPIC_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = { + "minimax": (_MINIMAX_ANTHROPIC_BASE_URL, "MINIMAX_API_KEY"), + "kimi-coding": (_KIMI_CODING_BASE_URL, "KIMI_API_KEY"), + "custom-anthropic": (None, "CUSTOM_ANTHROPIC_API_KEY"), +} + +# Anthropic-routed providers that support extended thinking. +_THINKING_CAPABLE_PROVIDERS: set[str] = {"minimax"} + +# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3 +# cannot disable thinking — structured output must use json_schema there. +# Moonshot-specific: do NOT widen to other mandatory-reasoning models. +_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset( + {"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"} +) + + +def _is_mandatory_thinking_kimi(model_id: str) -> bool: + """True for Kimi models whose thinking cannot be disabled (K3 family).""" + short_id = model_id.split("/")[-1] + return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding" + + +# Model registry: list of (short_name, model_id, provider) +# Allows same short_name across different providers. +_MODEL_ENTRIES: list[tuple[str, str, str]] = [ + # Custom Anthropic (third-party Claude-compatible endpoints, current-gen defaults) + # Listed BEFORE native anthropic so MODELS dict defaults to native provider + ("claude-sonnet-4-6", "claude-sonnet-4-6", "custom-anthropic"), + ("claude-haiku-4-5", "claude-haiku-4-5", "custom-anthropic"), + # Custom OpenAI (third-party OpenAI-compatible endpoints, 3 defaults) + # Listed BEFORE native openai so MODELS dict defaults to native provider + ("gpt-5.5-pro", "gpt-5.5-pro", "custom-openai"), + ("gpt-5.5", "gpt-5.5", "custom-openai"), + ("gpt-5.4", "gpt-5.4", "custom-openai"), + ("gpt-5.3-codex", "gpt-5.3-codex", "custom-openai"), + ("gpt-5-mini", "gpt-5-mini", "custom-openai"), + # Atlas Cloud (OpenAI-compatible) + ("qwen3.5-27b", "qwen/qwen3.5-27b", "atlascloud"), + # Anthropic (current generation) + ("claude-fable-5", "claude-fable-5", "anthropic"), + ("claude-opus-5", "claude-opus-5", "anthropic"), + ("claude-opus-4-8", "claude-opus-4-8", "anthropic"), + ("claude-sonnet-5", "claude-sonnet-5", "anthropic"), + ("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"), + ("claude-haiku-4-5", "claude-haiku-4-5", "anthropic"), + # OpenAI + ("gpt-5.6-sol", "gpt-5.6-sol", "openai"), + ("gpt-5.6-terra", "gpt-5.6-terra", "openai"), + ("gpt-5.6-luna", "gpt-5.6-luna", "openai"), + ("gpt-5.5-pro", "gpt-5.5-pro", "openai"), + ("gpt-5.5", "gpt-5.5", "openai"), + ("gpt-5.4", "gpt-5.4", "openai"), + ("gpt-5.4-mini", "gpt-5.4-mini", "openai"), + ("gpt-5.4-nano", "gpt-5.4-nano", "openai"), + ("gpt-5.3-codex", "gpt-5.3-codex", "openai"), + ("gpt-5.2-codex", "gpt-5.2-codex", "openai"), + ("gpt-5.2", "gpt-5.2", "openai"), + ("gpt-5.1", "gpt-5.1", "openai"), + ("gpt-5", "gpt-5", "openai"), + ("gpt-5-mini", "gpt-5-mini", "openai"), + ("gpt-5-nano", "gpt-5-nano", "openai"), + # Google GenAI + ("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"), + ("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"), + ("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"), + ("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"), + ( + "gemini-3.1-pro-customtools", + "gemini-3.1-pro-preview-customtools", + "google-genai", + ), + ("gemini-3.1-flash-lite", "gemini-3.1-flash-lite-preview", "google-genai"), + ("gemini-3-flash", "gemini-3-flash-preview", "google-genai"), + ("gemini-2.5-flash", "gemini-2.5-flash", "google-genai"), + ("gemini-2.5-flash-lite", "gemini-2.5-flash-lite", "google-genai"), + ("gemini-2.5-pro", "gemini-2.5-pro", "google-genai"), + # MiniMax (direct API — Anthropic-compatible; default: api.minimaxi.com, global: api.minimax.io) + ("minimax-m3", "MiniMax-M3", "minimax"), + ("minimax-m2.7", "MiniMax-M2.7", "minimax"), + ("minimax-m2.7-highspeed", "MiniMax-M2.7-highspeed", "minimax"), + ("minimax-m2.5", "MiniMax-M2.5", "minimax"), + ("minimax-m2.5-highspeed", "MiniMax-M2.5-highspeed", "minimax"), + # NVIDIA + ("nemotron-super", "nvidia/nemotron-3-super-120b-a12b", "nvidia"), + ("nemotron-nano", "nvidia/nemotron-3-nano-30b-a3b", "nvidia"), + ("glm-5.2", "z-ai/glm-5.2", "nvidia"), + ("glm4.7", "z-ai/glm4.7", "nvidia"), + ("deepseek-v3.2", "deepseek-ai/deepseek-v3.2", "nvidia"), + ("deepseek-v3.1", "deepseek-ai/deepseek-v3.1-terminus", "nvidia"), + ("kimi-k2.5", "moonshotai/kimi-k2.5", "nvidia"), + ("kimi-k2-thinking", "moonshotai/kimi-k2-thinking", "nvidia"), + ("minimax-m2.5", "minimaxai/minimax-m2.5", "nvidia"), + ("minimax-m2.1", "minimaxai/minimax-m2.1", "nvidia"), + ("qwen3.5-397b", "qwen/qwen3.5-397b-a17b", "nvidia"), + ("step-3.5-flash", "stepfun-ai/step-3.5-flash", "nvidia"), + # SiliconFlow + ("minimax-m2.5", "Pro/MiniMaxAI/MiniMax-M2.5", "siliconflow"), + ("glm-5.2", "Pro/zai-org/GLM-5.2", "siliconflow"), + ("glm-5", "Pro/zai-org/GLM-5", "siliconflow"), + ("kimi-k2.5", "Pro/moonshotai/Kimi-K2.5", "siliconflow"), + ("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"), + # Requesty (aggregator — OpenAI-compatible router, provider/model IDs). + # Listed before OpenRouter so that for model names shared with OpenRouter + # or a native provider, Requesty does not override them (the dict below is + # last-entry-wins); Requesty is selected explicitly via get_models_for_provider. + ("claude-sonnet-4.6", "anthropic/claude-sonnet-4-6", "requesty"), + ("claude-opus-4.8", "anthropic/claude-opus-4-8", "requesty"), + ("gemini-3.5-flash", "google/gemini-3.5-flash", "requesty"), + ("grok-4.3", "xai/grok-4.3", "requesty"), + ("grok-build-0.1", "xai/grok-build-0.1", "requesty"), + # OpenRouter + ("claude-fable-5", "anthropic/claude-fable-5", "openrouter"), + ("claude-opus-5", "anthropic/claude-opus-5", "openrouter"), + ("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"), + ("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"), + ("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"), + ("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"), + ("claude-sonnet-4.6", "anthropic/claude-sonnet-4.6", "openrouter"), + ("gpt-5.6-sol", "openai/gpt-5.6-sol", "openrouter"), + ("gpt-5.6-terra", "openai/gpt-5.6-terra", "openrouter"), + ("gpt-5.6-luna", "openai/gpt-5.6-luna", "openrouter"), + ("gpt-5.5-pro", "openai/gpt-5.5-pro", "openrouter"), + ("gpt-5.5", "openai/gpt-5.5", "openrouter"), + ("gpt-5.4", "openai/gpt-5.4", "openrouter"), + ("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"), + ("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"), + ("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"), + ("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"), + ("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"), + ("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"), + ("kimi-k3", "moonshotai/kimi-k3", "openrouter"), + ("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"), + ("glm-5.2", "z-ai/glm-5.2", "openrouter"), + ("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"), + ("minimax-m3", "minimax/minimax-m3", "openrouter"), + ("mimo-v2.5-pro", "xiaomi/mimo-v2.5-pro", "openrouter"), + ("mimo-v2.5", "xiaomi/mimo-v2.5", "openrouter"), + ("grok-build-0.1", "x-ai/grok-build-0.1", "openrouter"), + ("grok-4.5", "x-ai/grok-4.5", "openrouter"), + ("hy3", "tencent/hy3", "openrouter"), + ("qwen3.8-max", "qwen/qwen3.8-max", "openrouter"), + ("qwen3.7-max", "qwen/qwen3.7-max", "openrouter"), + ("qwen3.7-plus", "qwen/qwen3.7-plus", "openrouter"), + ("qwen3.6-flash", "qwen/qwen3.6-flash", "openrouter"), + ("qwen3.5-122b", "qwen/qwen3.5-122b-a10b", "openrouter"), + ("deepseek-v4-pro", "deepseek/deepseek-v4-pro", "openrouter"), + ("deepseek-v4-flash", "deepseek/deepseek-v4-flash", "openrouter"), + # Volcengine Coding Plan (火山引擎代码计划 — coding-only endpoint) + # Listed before Zhipu so simple GLM lookups keep their existing default. + ("glm-5.2", "glm-5-2", "volcengine-code"), + ("kimi-k2.5", "kimi-k2-5", "volcengine-code"), + # Zhipu CodePlan (智谱代码计划 — coding-only endpoint) + ("glm-5.2", "glm-5.2", "zhipu-code"), + ("glm-5.1", "glm-5.1", "zhipu-code"), + ("glm-5", "glm-5", "zhipu-code"), + ("glm-5-turbo", "glm-5-turbo", "zhipu-code"), + ("glm-5v-turbo", "glm-5v-turbo", "zhipu-code"), + ("glm-4.7", "glm-4.7", "zhipu-code"), + # Zhipu (智谱 — general endpoint, default for simple lookups) + ("glm-5.2", "glm-5.2", "zhipu"), + ("glm-5.1", "glm-5.1", "zhipu"), + ("glm-5", "glm-5", "zhipu"), + ("glm-5-turbo", "glm-5-turbo", "zhipu"), + ("glm-5v-turbo", "glm-5v-turbo", "zhipu"), + ("glm-4.7", "glm-4.7", "zhipu"), + # Volcengine (火山引擎 — Doubao models) + ("doubao-seed-2.0-pro", "doubao-seed-2-0-pro-260215", "volcengine"), + ("doubao-seed-2.0-lite", "doubao-seed-2-0-lite-260215", "volcengine"), + ("doubao-seed-2.0-mini", "doubao-seed-2-0-mini-260215", "volcengine"), + ("doubao-seed-2.0-code", "doubao-seed-2-0-code-preview-260215", "volcengine"), + ("doubao-seed-1.6", "doubao-seed-1.6", "volcengine"), + ("doubao-1.5-pro", "doubao-1.5-pro-256k", "volcengine"), + ("doubao-1.5-thinking-pro", "doubao-1.5-thinking-pro", "volcengine"), + # DashScope Coding Plan (阿里云代码计划 — subscription sk-sp-* endpoint) + ("qwen3.8-max", "qwen3.8-max", "dashscope-code"), + ("qwen3.7-max", "qwen3.7-max", "dashscope-code"), + ("qwen3.7-plus", "qwen3.7-plus", "dashscope-code"), + ("qwen3.6-max", "qwen3.6-max-preview", "dashscope-code"), + ("qwen3.6-plus", "qwen3.6-plus", "dashscope-code"), + ("qwen3.6-flash", "qwen3.6-flash", "dashscope-code"), + ("qwen3-coder", "qwen3-coder-plus", "dashscope-code"), + ("qwen3-coder-next", "qwen3-coder-next", "dashscope-code"), + ("qwen3-max", "qwen3-max", "dashscope-code"), + ("qwen3.5-plus", "qwen3.5-plus", "dashscope-code"), + # DashScope (阿里云 — Qwen models, default for simple lookups) + ("qwen3.8-max", "qwen3.8-max", "dashscope"), + ("qwen3.7-max", "qwen3.7-max", "dashscope"), + ("qwen3.7-plus", "qwen3.7-plus", "dashscope"), + ("qwen3.6-max", "qwen3.6-max-preview", "dashscope"), + ("qwen3.6-plus", "qwen3.6-plus", "dashscope"), + ("qwen3.6-flash", "qwen3.6-flash", "dashscope"), + ("qwen3-coder", "qwen3-coder-plus", "dashscope"), + ("qwen3-235b", "qwen3-235b-a22b", "dashscope"), + ("qwen-max", "qwen-max", "dashscope"), + ("qwq-plus", "qwq-plus", "dashscope"), + # DeepSeek + ("deepseek-v4-pro", "deepseek-v4-pro", "deepseek"), + ("deepseek-v4-flash", "deepseek-v4-flash", "deepseek"), + # Legacy aliases (deprecated 2026-07-24; route to v4-flash thinking/non-thinking) + ("deepseek-r1", "deepseek-reasoner", "deepseek"), + ("deepseek-v3", "deepseek-chat", "deepseek"), + # Moonshot (OpenAI-compatible) + ("kimi-k3", "kimi-k3", "moonshot"), + ("kimi-k2.6", "kimi-k2.6", "moonshot"), + ("kimi-k2.5", "kimi-k2.5", "moonshot"), + ("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"), + ("kimi-k2-thinking-turbo", "kimi-k2-thinking-turbo", "moonshot"), + ("moonshot-v1-auto", "moonshot-v1-auto", "moonshot"), + ("moonshot-v1-128k", "moonshot-v1-128k", "moonshot"), + ("moonshot-v1-32k", "moonshot-v1-32k", "moonshot"), + ("moonshot-v1-8k", "moonshot-v1-8k", "moonshot"), + # Kimi Coding Plan (Anthropic-compatible) + ("kimi-for-coding", "kimi-for-coding", "kimi-coding"), +] + +# Public dict for simple lookups (last entry wins for duplicate names). +# Use get_models_for_provider() for provider-aware lookups. +MODELS: dict[str, tuple[str, str]] = { + name: (model_id, provider) for name, model_id, provider in _MODEL_ENTRIES +} + +DEFAULT_MODEL = "claude-sonnet-4-6" + + +def get_models_for_provider(provider: str) -> list[tuple[str, str]]: + """Get all models for a specific provider. + + Args: + provider: Provider name (e.g., 'anthropic', 'openrouter'). + + Returns: + List of (short_name, model_id) tuples for the provider. + """ + return [(name, model_id) for name, model_id, p in _MODEL_ENTRIES if p == provider] + + +def list_models() -> list[str]: + """List all available model short names. + + Returns: + List of unique model short names that can be passed to get_chat_model(). + """ + seen = set() + result = [] + for name, _, _ in _MODEL_ENTRIES: + if name not in seen: + seen.add(name) + result.append(name) + return result + + +def list_models_by_provider() -> list[tuple[str, str, str]]: + """List all unique (short_name, model_id, provider) entries. + + Returns: + De-duplicated list of model entries preserving registry order. + """ + seen: set[tuple[str, str]] = set() + result: list[tuple[str, str, str]] = [] + for name, model_id, provider in _MODEL_ENTRIES: + key = (name, provider) + if key not in seen: + seen.add(key) + result.append((name, model_id, provider)) + return result + + +async def list_model_picker_entries( + ollama_base_url: str | None, + *, + include_custom_ollama: bool, +) -> list[tuple[str, str, str]]: + """Return model picker entries, optionally including local Ollama models.""" + entries = list_models_by_provider() + if ollama_base_url: + from .ollama_discovery import discover_ollama_models + + for detected_name in await discover_ollama_models( + ollama_base_url, + timeout=1.5, + ): + entries.append((detected_name, detected_name, "ollama")) + if include_custom_ollama: + entries.append(("Custom Ollama model...", "__custom_ollama__", "ollama")) + return entries + + +def get_model_info(model: str) -> tuple[str, str] | None: + """Get the (model_id, provider) tuple for a short name. + + Args: + model: Short model name. + + Returns: + Tuple of (model_id, provider) or None if not found. + """ + return MODELS.get(model) diff --git a/EvoScientist/sessions.py b/EvoScientist/sessions.py index 03995a1..300c2b2 100644 --- a/EvoScientist/sessions.py +++ b/EvoScientist/sessions.py @@ -544,6 +544,50 @@ async def _table_exists(conn: aiosqlite.Connection, table: str) -> bool: return await cur.fetchone() is not None +async def _ensure_thread_meta_index(conn: aiosqlite.Connection, db_path: str) -> None: + """Create the expression index behind thread listing, if missing. + + ``list_threads`` filters + groups on ``json_extract(metadata, ...)``; + without an index SQLite scans every checkpoint row, dragging each row's + multi-KB checkpoint blob through the page cache (~1.3s on a 700MB DB vs + ~0.1s indexed). The four columns are exactly the per-row expressions: + agent_name + graph_id feed the WHERE and updated_at feeds the MAX, so + none of them may fall back to a per-row metadata fetch; workspace_dir / + model are only materialized for the surviving groups (~dozens), so they + stay out of the index (measured: same speed as a 6-column variant at + 60% of its size, and two fewer json_extract per checkpoint write). + + The existence probe is a cheap catalog read on the caller's connection + (it can still wait on an exclusive schema lock, bounded by that + connection's timeout); the one-time build runs on a separate connection + whose 2s timeout bounds its lock wait, with build time after acquiring + (~0.2s measured on a 700MB DB) on top. Best-effort: on any failure the + listing simply runs unindexed and the next call retries. + """ + try: + async with conn.execute( + "SELECT 1 FROM sqlite_master WHERE type='index' AND name=?", + ("idx_evoscientist_thread_meta",), + ) as cur: + if await cur.fetchone() is not None: + return + async with aiosqlite.connect(db_path, timeout=2.0) as ddl_conn: + await ddl_conn.execute( + """ + CREATE INDEX IF NOT EXISTS idx_evoscientist_thread_meta + ON checkpoints ( + json_extract(metadata, '$.agent_name'), + thread_id, + json_extract(metadata, '$.updated_at'), + json_extract(metadata, '$.graph_id') + ) + """ + ) + await ddl_conn.commit() + except aiosqlite.Error: + _logger.debug("Could not ensure thread meta index", exc_info=True) + + def _reduce_messages_delta( state: list[AnyMessage] | None, writes: list[Any] ) -> list[AnyMessage]: @@ -889,6 +933,7 @@ async def list_threads( async with aiosqlite.connect(db_path, timeout=30.0) as conn: if not await _table_exists(conn, "checkpoints"): return [] + await _ensure_thread_meta_index(conn, db_path) query = f""" SELECT thread_id, diff --git a/tests/test_cli_deploy.py b/tests/test_cli_deploy.py index db1bf36..d38a4cf 100644 --- a/tests/test_cli_deploy.py +++ b/tests/test_cli_deploy.py @@ -134,6 +134,7 @@ def _run_deploy_once( jobs_per_worker=10, deploy_mode=False, tunnel=False, + config_fingerprint=None, ): captured["langgraph_dev_started"] = True captured["workspace_passed"] = str(workspace_dir) if workspace_dir else None diff --git a/tests/test_langgraph_dev_workspace_sidecar.py b/tests/test_langgraph_dev_workspace_sidecar.py index ed14109..271cf2e 100644 --- a/tests/test_langgraph_dev_workspace_sidecar.py +++ b/tests/test_langgraph_dev_workspace_sidecar.py @@ -142,6 +142,36 @@ def test_ensure_langgraph_dev_refuses_on_workspace_mismatch( manager.ensure_langgraph_dev(cfg, workspace_dir=ws_b) assert str(ws_a.resolve()) in str(exc.value) assert str(ws_b) in str(exc.value) + # Without keepalive a mismatch means a live session — no stop suggestion, + # keeping the message identical to pre-keepalive behavior. + assert "EvoSci server stop" not in str(exc.value) + + +def test_mismatch_error_suggests_server_stop_under_keepalive( + tmp_path, monkeypatch, runtime_paths +): + """With keepalive on, the leftover may be ownerless — the error points at + `EvoSci server stop`, which kills AND cleans the records (a raw kill + would leave stale files behind).""" + ws_a = tmp_path / "A" + ws_b = tmp_path / "B" + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, workspace_sidecar=tmp_path / "ws.json"), + ) + manager._write_workspace_sidecar(workspace_dir=ws_a, pid=99999) + + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + cfg.langgraph_dev_keepalive = True + with pytest.raises(manager.WorkspaceMismatchError) as exc: + manager.ensure_langgraph_dev(cfg, workspace_dir=ws_b) + assert "EvoSci server stop" in str(exc.value) def test_ensure_langgraph_dev_refuses_on_mismatch_with_stale_process( @@ -236,3 +266,304 @@ def test_stop_langgraph_dev_removes_sidecar(tmp_path, monkeypatch, runtime_paths monkeypatch.setattr(manager, "_PROCESS", None) manager.stop_langgraph_dev() assert not sidecar.exists() + + +def test_keepalive_skips_atexit_registration(tmp_path, monkeypatch, runtime_paths): + """keepalive=True leaves the server running on exit; False registers stop.""" + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, workspace_sidecar=tmp_path / "ws.json"), + ) + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: False) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + fake_proc = object() + monkeypatch.setattr(manager, "start_langgraph_dev", lambda **kw: fake_proc) + registered = [] + monkeypatch.setattr( + manager.atexit, "register", lambda *a, **kw: registered.append(a) + ) + + cfg = manager.EvoScientistConfig() + assert cfg.langgraph_dev_keepalive is False, "keepalive must default to off" + cfg.enable_async_subagents = True + cfg.langgraph_dev_keepalive = True + assert manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path / "A") is fake_proc + assert registered == [] + + cfg.langgraph_dev_keepalive = False + assert manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path / "A") is fake_proc + assert registered + assert registered[0][0] is manager.stop_langgraph_dev + + +# --------------------------------------------------------------------------- +# Config fingerprint + drift detection + explicit server stop +# --------------------------------------------------------------------------- + + +def _dead_pid() -> int: + """Spawn-and-reap a process so its pid is reliably dead.""" + import subprocess + import sys + + proc = subprocess.Popen([sys.executable, "-c", "pass"]) + proc.wait() + return proc.pid + + +def test_sidecar_records_config_fingerprint_when_passed( + tmp_path, monkeypatch, runtime_paths +): + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, workspace_sidecar=tmp_path / "ws.json"), + ) + manager._write_workspace_sidecar( + workspace_dir=tmp_path / "ws", pid=1, config_fingerprint="abc123" + ) + assert json.loads((tmp_path / "ws.json").read_text())["config_fingerprint"] == ( + "abc123" + ) + + +def test_server_config_fingerprint_tracks_relevant_fields(): + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + assert base == manager._server_config_fingerprint(cfg), "must be deterministic" + cfg.model = "some-other-model" + assert manager._server_config_fingerprint(cfg) != base + + +def _reuse_setup(tmp_path, monkeypatch, runtime_paths, fingerprint): + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, workspace_sidecar=tmp_path / "ws.json"), + ) + manager._write_workspace_sidecar( + workspace_dir=tmp_path / "A", pid=99999, config_fingerprint=fingerprint + ) + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + return cfg + + +def test_reuse_sets_drift_flag_on_fingerprint_mismatch( + tmp_path, monkeypatch, runtime_paths +): + cfg = _reuse_setup(tmp_path, monkeypatch, runtime_paths, "stale-fingerprint") + manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path / "A") + assert manager.CONFIG_DRIFT_SINCE_LAUNCH is True + + +def test_reuse_clears_drift_flag_on_matching_fingerprint( + tmp_path, monkeypatch, runtime_paths +): + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + fp = manager._server_config_fingerprint(cfg) + cfg2 = _reuse_setup(tmp_path, monkeypatch, runtime_paths, fp) + manager.ensure_langgraph_dev(cfg2, workspace_dir=tmp_path / "A") + assert manager.CONFIG_DRIFT_SINCE_LAUNCH is False + + +def test_stop_recorded_server_none_when_no_pid_file( + tmp_path, monkeypatch, runtime_paths +): + monkeypatch.setattr(manager, "RUNTIME", runtime_paths) + monkeypatch.setattr(manager, "_PROCESS", None) + assert manager.stop_recorded_server() is None + + +def test_stop_recorded_server_cleans_stale_files_for_dead_pid( + tmp_path, monkeypatch, runtime_paths +): + pid_file = tmp_path / "pid.txt" + sidecar = tmp_path / "ws.json" + pid_file.write_text(str(_dead_pid())) + sidecar.write_text(json.dumps({"workspace": "/a", "pid": 1})) + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace( + runtime_paths, pid_file=pid_file, workspace_sidecar=sidecar + ), + ) + monkeypatch.setattr(manager, "_PROCESS", None) + assert manager.stop_recorded_server() is None + assert not pid_file.exists() + assert not sidecar.exists() + + +def test_stop_recorded_server_refuses_foreign_process( + tmp_path, monkeypatch, runtime_paths +): + import subprocess + import sys + + victim = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(30)"]) + try: + pid_file = tmp_path / "pid.txt" + pid_file.write_text(str(victim.pid)) + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, pid_file=pid_file), + ) + monkeypatch.setattr(manager, "_PROCESS", None) + assert manager.stop_recorded_server() is None + assert victim.poll() is None, "foreign process must not be killed" + finally: + victim.kill() + victim.wait() + + +def test_stop_recorded_server_kills_owned_langgraph_process( + tmp_path, monkeypatch, runtime_paths +): + import subprocess + import sys + + victim = subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(30)", "langgraph"] + ) + try: + pid_file = tmp_path / "pid.txt" + sidecar = tmp_path / "ws.json" + pid_file.write_text(str(victim.pid)) + sidecar.write_text(json.dumps({"workspace": "/a", "pid": victim.pid})) + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace( + runtime_paths, pid_file=pid_file, workspace_sidecar=sidecar + ), + ) + monkeypatch.setattr(manager, "_PROCESS", None) + assert manager.stop_recorded_server() == victim.pid + assert victim.poll() is not None, "owned langgraph process must be stopped" + assert not pid_file.exists() + assert not sidecar.exists() + finally: + if victim.poll() is None: + victim.kill() + victim.wait() + + +def test_stop_recorded_server_cleans_corrupt_pid_file( + tmp_path, monkeypatch, runtime_paths +): + """A corrupt PID file is cleaned up, matching the docstring's promise.""" + pid_file = tmp_path / "pid.txt" + pid_file.write_text("not-a-pid") + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, pid_file=pid_file), + ) + monkeypatch.setattr(manager, "_PROCESS", None) + assert manager.stop_recorded_server() is None + assert not pid_file.exists() + + +def test_server_config_fingerprint_tracks_dangerous_mode(): + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + cfg.dangerous_mode = True + assert manager._server_config_fingerprint(cfg) != base + + +def test_pid_serves_port_matrix(): + import subprocess + import sys + + assert manager._pid_serves_port(None, 6174) is False + assert manager._pid_serves_port(True, 6174) is False + assert manager._pid_serves_port(-1, 6174) is False + + victim = subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(30)", "langgraph", "6174"] + ) + try: + assert manager._pid_serves_port(victim.pid, 6174) is True + assert manager._pid_serves_port(victim.pid, 9999) is False, ( + "must not attribute a server bound to a different port" + ) + finally: + victim.kill() + victim.wait() + assert manager._pid_serves_port(victim.pid, 6174) is False, "dead pid" + + +def test_server_config_fingerprint_covers_new_fields_by_default(): + """The fingerprint iterates the dataclass minus an exclusion set, so + server-affecting values din0s called out — reasoning effort, provider + keys/base URLs — count without per-field registration.""" + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + cfg.reasoning_effort = "definitely-different" + after_effort = manager._server_config_fingerprint(cfg) + assert after_effort != base + cfg.anthropic_base_url = "https://proxy.example" + assert manager._server_config_fingerprint(cfg) != after_effort + + +def test_server_config_fingerprint_ignores_cli_only_fields(): + """Channel/UI-side fields never reach the server — no spurious drift.""" + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + cfg.show_thinking = not cfg.show_thinking + cfg.telegram_bot_token = "tg-token" + cfg.langgraph_dev_keepalive = True + assert manager._server_config_fingerprint(cfg) == base + + +def test_sidecar_records_deploy_mode(tmp_path, monkeypatch, runtime_paths): + monkeypatch.setattr( + manager, + "RUNTIME", + dataclasses.replace(runtime_paths, workspace_sidecar=tmp_path / "ws.json"), + ) + manager._write_workspace_sidecar( + workspace_dir=tmp_path / "ws", pid=1, deploy_mode=False + ) + assert json.loads((tmp_path / "ws.json").read_text())["deploy_mode"] is False + + +def test_server_config_fingerprint_ignores_shell_allow_list(): + """shell_allow_list is read only by CLI/channel approval resolvers — + it never reaches the subprocess, so it must not cause drift.""" + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + cfg.shell_allow_list = "git status,ls" + assert manager._server_config_fingerprint(cfg) == base + + +def test_server_config_fingerprint_tracks_mcp_yaml_content(tmp_path, monkeypatch): + from EvoScientist.config import settings as settings_mod + + monkeypatch.setattr(settings_mod, "get_config_dir", lambda: tmp_path) + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + (tmp_path / "mcp.yaml").write_text("server-a:\n transport: stdio\n") + changed = manager._server_config_fingerprint(cfg) + assert changed != base + (tmp_path / "mcp.yaml").write_text("server-b:\n transport: http\n") + assert manager._server_config_fingerprint(cfg) != changed + + +def test_server_config_fingerprint_tracks_subagent_yaml_content(tmp_path, monkeypatch): + fake_dir = tmp_path / "subagents" + fake_dir.mkdir() + (fake_dir / "research.yaml").write_text("name: research\nprompt: v1\n") + monkeypatch.setattr(manager, "_SUBAGENTS_DIR", fake_dir) + cfg = manager.EvoScientistConfig() + base = manager._server_config_fingerprint(cfg) + (fake_dir / "research.yaml").write_text("name: research\nprompt: v2\n") + assert manager._server_config_fingerprint(cfg) != base diff --git a/tests/test_langgraph_manager.py b/tests/test_langgraph_manager.py index c3ce0cd..74a59aa 100644 --- a/tests/test_langgraph_manager.py +++ b/tests/test_langgraph_manager.py @@ -419,6 +419,10 @@ class TestEnsureLanggraphDev: with ( patch.object(manager, "is_langgraph_dev_running", return_value=False), patch.object(manager, "start_langgraph_dev", return_value=proc) as start, + # Intercept the real atexit hook: letting it register means the + # callback fires at interpreter exit AFTER patches are restored, + # deleting the REAL ~/.config/evoscientist pid/sidecar files. + patch.object(manager.atexit, "register") as atexit_register, patch.object( manager, "RUNTIME", @@ -432,6 +436,7 @@ class TestEnsureLanggraphDev: assert result is proc start.assert_called_once() + atexit_register.assert_called_once_with(manager.stop_langgraph_dev, proc) assert manager.is_async_subagents_available() is True def test_skips_when_async_and_memory_workers_disabled( diff --git a/tests/test_sessions.py b/tests/test_sessions.py index 599d59a..8486c24 100644 --- a/tests/test_sessions.py +++ b/tests/test_sessions.py @@ -9,6 +9,7 @@ import uuid from datetime import UTC from unittest.mock import patch +import aiosqlite import pytest from langchain_core.messages import AIMessage, HumanMessage, RemoveMessage from langgraph.checkpoint.serde.jsonplus import JsonPlusSerializer @@ -232,6 +233,19 @@ class TestThreadFunctions(unittest.IsolatedAsyncioTestCase): threads = await list_threads(limit=10, include_message_count=True) assert "message_count" in threads[0] + async def test_list_threads_creates_meta_index(self): + """Thread listing must run off the expression index, not a blob scan.""" + async with aiosqlite.connect(self._db_path) as conn: + await conn.execute("DROP INDEX IF EXISTS idx_evoscientist_thread_meta") + await conn.commit() + await list_threads(limit=10) + async with aiosqlite.connect(self._db_path) as conn: + async with conn.execute( + "SELECT 1 FROM sqlite_master WHERE type='index' AND name=?", + ("idx_evoscientist_thread_meta",), + ) as cur: + assert await cur.fetchone() is not None + async def test_thread_exists_true(self): assert await thread_exists("abc12345") diff --git a/tests/test_webui_launcher.py b/tests/test_webui_launcher.py index 602888f..c4b6f7d 100644 --- a/tests/test_webui_launcher.py +++ b/tests/test_webui_launcher.py @@ -26,6 +26,7 @@ def _make_config( langgraph_dev_host: str = "127.0.0.1", webui_port: int = 4716, webui_host: str = "127.0.0.1", + langgraph_dev_keepalive: bool = False, ): return SimpleNamespace( default_workdir=default_workdir, @@ -35,6 +36,7 @@ def _make_config( webui_host=webui_host, langgraph_dev_jobs_per_worker=10, langgraph_dev_file_persistence=True, + langgraph_dev_keepalive=langgraph_dev_keepalive, ) @@ -121,7 +123,12 @@ def _run_webui_once(monkeypatch, config, *, backend_port_occupied: bool = False) return SimpleNamespace(pid=99999) monkeypatch.setattr(lgm, "start_langgraph_dev", _fake_start_langgraph_dev) - monkeypatch.setattr(lgm, "stop_langgraph_dev", lambda *_a, **_kw: None) + + def _fake_stop(*_a, **_kw): + return None + + monkeypatch.setattr(lgm, "stop_langgraph_dev", _fake_stop) + captured["stop_fn"] = _fake_stop class _FakeProc: pid = 12345 @@ -146,7 +153,10 @@ def _run_webui_once(monkeypatch, config, *, backend_port_occupied: bool = False) monkeypatch.setattr(subprocess, "Popen", _fake_popen) # _stop_webui shells out to taskkill on Windows — neutralize it. monkeypatch.setattr(webui_mod, "_stop_webui", lambda _proc: None) - monkeypatch.setattr(atexit, "register", lambda fn, *a, **k: fn) + captured["atexit_fns"] = [] + monkeypatch.setattr( + atexit, "register", lambda fn, *a, **k: captured["atexit_fns"].append(fn) or fn + ) monkeypatch.setattr(signal, "signal", lambda _sig, _handler: lambda *a: None) monkeypatch.setattr(threading, "Event", _ImmediateEvent) @@ -273,3 +283,21 @@ def test_no_remote_hint_when_both_exposed(monkeypatch): banner = "\n".join(captured["printed"]) assert "Remote visitors cannot reach" not in banner + + +# ============================================================================= +# Backend keepalive +# ============================================================================= + + +def test_backend_default_registers_stop_on_exit(monkeypatch): + """Without keepalive the WebUI-started backend dies with the session.""" + captured = _run_webui_once(monkeypatch, _make_config()) + assert captured["stop_fn"] in captured["atexit_fns"] + + +def test_backend_keepalive_skips_stop_registration(monkeypatch): + """With keepalive the backend outlives the WebUI session, so the next + same-workspace launch reuses it instead of paying the cold boot.""" + captured = _run_webui_once(monkeypatch, _make_config(langgraph_dev_keepalive=True)) + assert captured["stop_fn"] not in captured["atexit_fns"]