perf: cut startup latency — lazy import surface, langgraph dev keepalive, indexed thread listing (#407)
* feat: add thread metadata index for improved performance in thread listing - Implemented a new SQLite index on the `checkpoints` table to optimize thread listing queries by indexing relevant metadata fields. - Updated the `list_threads` function to ensure the index is created if it does not exist. - Added a test to verify the creation of the metadata index during thread listing. feat: enhance workspace sidecar management with owner tracking - Modified the workspace sidecar to include `owner_pids` to track the current process owners. - Updated tests to validate the new owner tracking functionality and ensure proper behavior when managing workspace sidecars. chore: introduce model registry for streamlined model management - Created a new `registry.py` file to maintain a comprehensive model registry, including model names, IDs, providers, and routing tables. - Added functions to retrieve models by provider and list available models, enhancing the modularity and maintainability of model management. * feat: enhance workspace sidecar management and improve thread metadata indexing * fix(tests): ensure sidecar correctly registers owner with original workspace and pid * refactor: simplify workspace sidecar management by removing owner tracking * feat(server): add commands to manage background langgraph dev server - Introduced `server_app` for managing the langgraph dev server with commands to check status and stop the server. - Enhanced workspace sidecar management to include configuration fingerprint for drift detection. - Updated deployment functions to handle server configuration and state more effectively. * feat(server): enhance server status command to display PID with stale record warning * feat(langgraph_dev): exclusion-set config fingerprint, webui keepalive, unified stop guidance * fix(cli): platform-specific manual-stop hint; document keepalive endpoint-change limitation
This commit is contained in:
@@ -59,6 +59,12 @@ sessions_app = typer.Typer(
|
||||
)
|
||||
app.add_typer(sessions_app, name="sessions")
|
||||
|
||||
# Background langgraph dev server management — the explicit counterpart to
|
||||
# langgraph_dev_keepalive: a server that outlives its CLI needs a first-class
|
||||
# way to inspect and stop it.
|
||||
server_app = typer.Typer(help="Manage the background langgraph dev server")
|
||||
app.add_typer(server_app, name="server")
|
||||
|
||||
# Configure subcommand group — re-run a single onboarding section.
|
||||
configure_app = typer.Typer(
|
||||
help=(
|
||||
|
||||
@@ -26,14 +26,15 @@ from ..gateway import (
|
||||
GraphGateway,
|
||||
GraphTarget,
|
||||
RunRequest,
|
||||
RuntimeGateways,
|
||||
create_runtime_gateways,
|
||||
)
|
||||
from ..llm.context_window import DEFAULT_CONTEXT_WINDOW_FALLBACK, resolve_context_window
|
||||
from ..paths import ensure_dirs, set_active_workspace, set_workspace_root
|
||||
from ..runtime import AsyncRuntime
|
||||
from ..stream.console import console
|
||||
from . import async_notifier
|
||||
from . import (
|
||||
async_notifier,
|
||||
server_cmd, # noqa: F401 — registers `EvoSci server` commands
|
||||
)
|
||||
from ._app import app, channel_app, config_app, configure_app, mcp_app, sessions_app
|
||||
from ._constants import build_metadata
|
||||
from .agent import (
|
||||
@@ -72,6 +73,7 @@ if TYPE_CHECKING:
|
||||
from langgraph.graph.state import CompiledStateGraph
|
||||
|
||||
from ..config import EvoScientistConfig
|
||||
from ..gateway import RuntimeGateways
|
||||
|
||||
|
||||
_ASYNC_RUNTIME_META_KEY = "evoscientist.async_runtime"
|
||||
@@ -527,6 +529,16 @@ def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None:
|
||||
console.print(f"[red]{exc}[/red]")
|
||||
raise typer.Exit(1) from exc
|
||||
|
||||
from ..langgraph_dev import manager as _lg_manager
|
||||
|
||||
if _lg_manager.CONFIG_DRIFT_SINCE_LAUNCH:
|
||||
console.print(
|
||||
"[yellow]⚠ Config changed since the background agent server was "
|
||||
"launched — async sub-agents still use the old settings. Apply "
|
||||
"them with [bold]EvoSci server stop[/bold], then restart "
|
||||
"EvoSci.[/yellow]"
|
||||
)
|
||||
|
||||
# The backend is shared by every UI mode, so the exposure warning lives
|
||||
# here, not just in deploy/WebUI. Gated on the server being up: warning
|
||||
# about a bind that never happened would be worse than saying nothing.
|
||||
@@ -940,7 +952,7 @@ class ServeRuntimeState:
|
||||
thread_id: str
|
||||
workspace_dir: str | None
|
||||
config: "EvoScientistConfig | None"
|
||||
runtime_gateways: RuntimeGateways
|
||||
runtime_gateways: "RuntimeGateways"
|
||||
async_runtime: AsyncRuntime
|
||||
resume_warning_thread_id: str | None = None
|
||||
|
||||
@@ -1553,6 +1565,8 @@ def serve(
|
||||
console.print("[dim]Loading agent...[/dim]")
|
||||
agent = _load_agent(workspace_dir=ws, config=config, runtime=async_runtime)
|
||||
|
||||
from ..gateway import create_runtime_gateways
|
||||
|
||||
runtime_gateways = create_runtime_gateways()
|
||||
tid = async_runtime.run_sync(
|
||||
lambda: runtime_gateways.graph_gateway.create_thread(
|
||||
@@ -2412,6 +2426,7 @@ def _main_callback(
|
||||
# Single-shot mode: wrap in persistent checkpointer
|
||||
import asyncio
|
||||
|
||||
from ..gateway import create_runtime_gateways
|
||||
from ..sessions import get_checkpointer
|
||||
from ..stream.json_sink import stream_json
|
||||
from .interactive import _wait_for_memory_workers_before_exit, cmd_run
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
"""``EvoSci server`` — inspect and stop the background langgraph dev server.
|
||||
|
||||
The explicit counterpart to ``langgraph_dev_keepalive``: an opt-in server
|
||||
that outlives its CLI needs an equally explicit way to see and stop it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
from ..stream.console import console
|
||||
from ._app import server_app
|
||||
|
||||
|
||||
@server_app.command("status")
|
||||
def server_status() -> None:
|
||||
"""Show the background langgraph dev server's state."""
|
||||
from ..config import get_effective_config
|
||||
from ..langgraph_dev.manager import (
|
||||
_DEFAULT_HOST,
|
||||
_DEFAULT_PORT,
|
||||
_pid_serves_port,
|
||||
_read_workspace_sidecar,
|
||||
is_langgraph_dev_running,
|
||||
)
|
||||
|
||||
config = get_effective_config()
|
||||
port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT))
|
||||
host = (
|
||||
str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST)
|
||||
).strip() or _DEFAULT_HOST
|
||||
running = is_langgraph_dev_running(port=port, host=host)
|
||||
sidecar = _read_workspace_sidecar()
|
||||
|
||||
if not running and sidecar is None:
|
||||
console.print("[dim]No background langgraph dev server is running.[/dim]")
|
||||
return
|
||||
state = "[green]running[/green]" if running else "[red]not responding[/red]"
|
||||
console.print(f"[bold]langgraph dev[/bold] on port {port}: {state}")
|
||||
if sidecar is not None:
|
||||
console.print(f" workspace: {sidecar.get('workspace')}")
|
||||
pid = sidecar.get("pid")
|
||||
if _pid_serves_port(pid, port):
|
||||
console.print(f" pid: {pid}")
|
||||
else:
|
||||
console.print(
|
||||
f" pid: {pid} [yellow](stale record — this pid does "
|
||||
f"not serve port {port})[/yellow]"
|
||||
)
|
||||
elif running:
|
||||
console.print(
|
||||
" [yellow]no sidecar — externally managed or pre-keepalive server[/yellow]"
|
||||
)
|
||||
|
||||
|
||||
@server_app.command("stop")
|
||||
def server_stop() -> None:
|
||||
"""Stop the background langgraph dev server started by EvoSci."""
|
||||
from ..config import get_effective_config
|
||||
from ..langgraph_dev.manager import (
|
||||
_DEFAULT_HOST,
|
||||
_DEFAULT_PORT,
|
||||
is_langgraph_dev_running,
|
||||
stop_recorded_server,
|
||||
)
|
||||
|
||||
pid = stop_recorded_server()
|
||||
if pid is not None:
|
||||
console.print(f"[green]✓[/green] Stopped langgraph dev (pid {pid}).")
|
||||
return
|
||||
config = get_effective_config()
|
||||
port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT))
|
||||
host = (
|
||||
str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST)
|
||||
).strip() or _DEFAULT_HOST
|
||||
if is_langgraph_dev_running(port=port, host=host):
|
||||
# A server without ownership records (crashed session, deleted state
|
||||
# files) can't be verified as ours — refuse to guess, hand the user
|
||||
# the manual path instead of a silent no-op.
|
||||
console.print(
|
||||
f"[yellow]⚠ A langgraph dev is still serving port {port}, but "
|
||||
f"EvoSci has no ownership record for it, so it was not "
|
||||
f"touched.[/yellow]"
|
||||
)
|
||||
if sys.platform == "win32":
|
||||
manual = (
|
||||
f'powershell "Get-NetTCPConnection -LocalPort {port} | '
|
||||
f'Select-Object -ExpandProperty OwningProcess | Stop-Process"'
|
||||
)
|
||||
else:
|
||||
manual = f"kill $(lsof -ti :{port})"
|
||||
console.print(
|
||||
f"[dim]If it is yours, stop it manually: [bold]{manual}[/bold][/dim]"
|
||||
)
|
||||
else:
|
||||
console.print(
|
||||
"[dim]No EvoSci-owned langgraph dev server to stop "
|
||||
"(stale state, if any, was cleaned up).[/dim]"
|
||||
)
|
||||
@@ -257,6 +257,15 @@ class EvoScientistConfig:
|
||||
# slowdown.
|
||||
langgraph_dev_jobs_per_worker: int = 10
|
||||
|
||||
# Keep the auto-started langgraph dev subprocess running after the CLI
|
||||
# exits. The next `EvoSci` start in the same workspace reuses it instantly
|
||||
# instead of paying the cold boot (~15s). Starting in a DIFFERENT workspace
|
||||
# raises WorkspaceMismatchError with the leftover server's pid — stop it
|
||||
# manually (the server is pinned to one workspace per process). Known
|
||||
# limitation: changing langgraph_dev_port/host while a keepalive server
|
||||
# runs orphans its records — run `EvoSci server stop` before switching.
|
||||
langgraph_dev_keepalive: bool = False
|
||||
|
||||
# Max LangGraph super-steps (LLM call / tool call / sub-agent delegation
|
||||
# each count as 1) before raising GraphRecursionError. Resets on every
|
||||
# ``agent.invoke()`` — i.e., this is per-turn, NOT per-conversation. For
|
||||
@@ -836,6 +845,7 @@ _ENV_MAPPINGS = {
|
||||
"sandbox_execute_timeout": "EVOSCIENTIST_SANDBOX_EXECUTE_TIMEOUT",
|
||||
"langgraph_dev_file_persistence": "EVOSCIENTIST_LANGGRAPH_DEV_FILE_PERSISTENCE",
|
||||
"langgraph_dev_jobs_per_worker": "EVOSCIENTIST_LANGGRAPH_DEV_JOBS_PER_WORKER",
|
||||
"langgraph_dev_keepalive": "EVOSCIENTIST_LANGGRAPH_DEV_KEEPALIVE",
|
||||
"recursion_limit": "EVOSCIENTIST_RECURSION_LIMIT",
|
||||
"memory_profile_enabled": "EVOSCIENTIST_MEMORY_PROFILE_ENABLED",
|
||||
"memory_observations_enabled": "EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
|
||||
@@ -79,6 +79,9 @@ def deploy(
|
||||
_base_url,
|
||||
_is_loopback_host,
|
||||
_is_port_occupied,
|
||||
_pid_serves_port,
|
||||
_read_workspace_sidecar,
|
||||
_server_config_fingerprint,
|
||||
is_langgraph_dev_running,
|
||||
read_tunnel_url,
|
||||
start_langgraph_dev,
|
||||
@@ -145,10 +148,25 @@ def deploy(
|
||||
f"[red]Port {effective_port} is already serving a langgraph dev "
|
||||
f"instance.[/red]"
|
||||
)
|
||||
console.print(
|
||||
"[dim]Stop the existing EvoSci/serve session first, or use "
|
||||
"[bold]--port[/bold] to deploy on a different port.[/dim]"
|
||||
)
|
||||
sidecar = _read_workspace_sidecar()
|
||||
if sidecar is not None and _pid_serves_port(
|
||||
sidecar.get("pid"), effective_port
|
||||
):
|
||||
# Surface what we know about the occupant — with keepalive it
|
||||
# may be an ownerless leftover rather than a live session.
|
||||
# Only when the recorded pid verifiably serves THIS port, so a
|
||||
# stale or other-port record is never blamed.
|
||||
console.print(
|
||||
f"[dim]It serves workspace {sidecar.get('workspace')} "
|
||||
f"(pid {sidecar.get('pid')}). Stop it with "
|
||||
f"[bold]EvoSci server stop[/bold], or use "
|
||||
f"[bold]--port[/bold] to deploy on a different port.[/dim]"
|
||||
)
|
||||
else:
|
||||
console.print(
|
||||
"[dim]Stop the existing EvoSci/serve session first, or use "
|
||||
"[bold]--port[/bold] to deploy on a different port.[/dim]"
|
||||
)
|
||||
else:
|
||||
console.print(
|
||||
f"[red]Port {effective_port} is occupied by another process.[/red]"
|
||||
@@ -231,6 +249,7 @@ def deploy(
|
||||
jobs_per_worker=jobs_per_worker,
|
||||
deploy_mode=True,
|
||||
tunnel=tunnel,
|
||||
config_fingerprint=_server_config_fingerprint(config),
|
||||
)
|
||||
atexit.register(stop_langgraph_dev, proc)
|
||||
except Exception as exc:
|
||||
|
||||
@@ -67,6 +67,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
_is_loopback_host,
|
||||
_is_port_occupied,
|
||||
_read_workspace_sidecar,
|
||||
_server_config_fingerprint,
|
||||
is_langgraph_dev_running,
|
||||
start_langgraph_dev,
|
||||
stop_langgraph_dev,
|
||||
@@ -163,6 +164,31 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
f"[/dim]"
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
if sidecar is not None and sidecar.get("deploy_mode") is False:
|
||||
# A stripped (CLI-started) server has no MCP and no async
|
||||
# sub-agents — silently reusing it would degrade the WebUI
|
||||
# with no visible cause. Refuse; never auto-kill.
|
||||
console.print(
|
||||
f"[red]Port {backend_port} is serving a stripped "
|
||||
f"(CLI-mode) langgraph dev — the WebUI needs the full "
|
||||
f"deploy-mode server (MCP + async sub-agents).[/red]"
|
||||
)
|
||||
console.print(
|
||||
"[dim]Stop it with [bold]EvoSci server stop[/bold], then "
|
||||
"re-run [bold]EvoSci[/bold].[/dim]"
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
if sidecar is not None:
|
||||
recorded_fp = sidecar.get("config_fingerprint")
|
||||
if isinstance(
|
||||
recorded_fp, str
|
||||
) and recorded_fp != _server_config_fingerprint(config):
|
||||
console.print(
|
||||
"[yellow]⚠ Config changed since this server was "
|
||||
"launched — it still serves the old settings. Apply "
|
||||
"them with [bold]EvoSci server stop[/bold], then "
|
||||
"re-run EvoSci.[/yellow]"
|
||||
)
|
||||
console.print(
|
||||
f"[green]✓[/green] Reusing langgraph dev already serving "
|
||||
f"port {backend_port}"
|
||||
@@ -191,8 +217,18 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
file_persistence=file_persistence,
|
||||
jobs_per_worker=jobs_per_worker,
|
||||
deploy_mode=True,
|
||||
config_fingerprint=_server_config_fingerprint(config),
|
||||
)
|
||||
atexit.register(stop_langgraph_dev, started_proc)
|
||||
if getattr(config, "langgraph_dev_keepalive", False):
|
||||
# Keepalive: the deploy-mode backend outlives this session so
|
||||
# the next same-workspace launch reuses it instantly. The npx
|
||||
# front-end below still stops on exit as usual.
|
||||
console.print(
|
||||
"[dim]keepalive: backend server stays up after exit — "
|
||||
"stop it with [bold]EvoSci server stop[/bold].[/dim]"
|
||||
)
|
||||
else:
|
||||
atexit.register(stop_langgraph_dev, started_proc)
|
||||
except Exception as exc:
|
||||
console.print(f"[red]langgraph dev startup failed:[/red] {exc}")
|
||||
raise typer.Exit(1) from exc
|
||||
|
||||
@@ -4,19 +4,16 @@ The gateway package is the migration seam between UI surfaces and graph
|
||||
execution. CLI, TUI, channels, and future frontends should depend on this
|
||||
package for thread/run operations instead of reaching directly into
|
||||
``sessions.py``, ``stream.events``, or the LangGraph SDK.
|
||||
|
||||
Backend implementations are attached lazily via :mod:`lazy_loader` (SPEC-1 /
|
||||
PEP 562): importing the shared :mod:`.types` protocols must not cascade into
|
||||
``sessions``/langgraph/langgraph_sdk, which every CLI invocation would pay.
|
||||
"""
|
||||
|
||||
from . import background_runs
|
||||
from .local import LocalGraphGateway, LocalThreadStore
|
||||
from .runtime import (
|
||||
RuntimeGatewayBackend,
|
||||
RuntimeGateways,
|
||||
create_runtime_gateways,
|
||||
)
|
||||
from .server import (
|
||||
LangGraphServerGateway,
|
||||
LangGraphServerThreadStore,
|
||||
)
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import lazy_loader as _lazy
|
||||
|
||||
from .types import (
|
||||
DEFAULT_GRAPH_ID,
|
||||
GraphEvent,
|
||||
@@ -29,6 +26,44 @@ from .types import (
|
||||
ThreadStore,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# Static counterparts of the lazy attach below — type checkers don't
|
||||
# infer names served through __getattr__.
|
||||
from . import background_runs
|
||||
from .local import LocalGraphGateway, LocalThreadStore
|
||||
from .runtime import (
|
||||
RuntimeGatewayBackend,
|
||||
RuntimeGateways,
|
||||
create_runtime_gateways,
|
||||
)
|
||||
from .server import (
|
||||
LangGraphServerGateway,
|
||||
LangGraphServerThreadStore,
|
||||
)
|
||||
|
||||
__getattr__, _attach_dir, _ = _lazy.attach(
|
||||
__name__,
|
||||
submodules=["background_runs"],
|
||||
submod_attrs={
|
||||
"local": ["LocalGraphGateway", "LocalThreadStore"],
|
||||
"runtime": [
|
||||
"RuntimeGatewayBackend",
|
||||
"RuntimeGateways",
|
||||
"create_runtime_gateways",
|
||||
],
|
||||
"server": [
|
||||
"LangGraphServerGateway",
|
||||
"LangGraphServerThreadStore",
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def __dir__() -> list[str]:
|
||||
# attach() only knows the lazy names; include the eager type exports too.
|
||||
return sorted(set(_attach_dir()) | set(__all__))
|
||||
|
||||
|
||||
__all__ = [
|
||||
"DEFAULT_GRAPH_ID",
|
||||
"GraphEvent",
|
||||
|
||||
@@ -6,15 +6,15 @@ from collections.abc import AsyncIterator
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Any, Protocol, TypeAlias
|
||||
|
||||
from langgraph.types import Command
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from langgraph.graph.state import CompiledStateGraph
|
||||
from langgraph.types import Command
|
||||
|
||||
from ..middleware.events import SessionEvents
|
||||
|
||||
GraphEvent: TypeAlias = dict[str, Any]
|
||||
GraphRunInput: TypeAlias = str | Command
|
||||
# String alias keeps this module langgraph-free at import time (~950 modules).
|
||||
GraphRunInput: TypeAlias = "str | Command"
|
||||
GraphStateValues: TypeAlias = dict[str, Any]
|
||||
DEFAULT_GRAPH_ID = "EvoScientist"
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@ Mirrors the lifecycle pattern used by ``ccproxy_manager.py``.
|
||||
from __future__ import annotations
|
||||
|
||||
import atexit
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
@@ -21,6 +22,7 @@ import subprocess
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import fields as dataclass_fields
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
@@ -110,6 +112,13 @@ def needs_langgraph_dev(config: EvoScientistConfig) -> bool:
|
||||
_LOCK = threading.RLock()
|
||||
|
||||
|
||||
# Set by ``ensure_langgraph_dev`` when it reuses a keepalive server whose
|
||||
# recorded launch-time config fingerprint differs from the current effective
|
||||
# config. The CLI reads it after startup to surface a "restart to apply"
|
||||
# hint — the server itself is never restarted automatically.
|
||||
CONFIG_DRIFT_SINCE_LAUNCH = False
|
||||
|
||||
|
||||
# Default port (Kaprekar's constant — see config/settings.py for the rationale).
|
||||
# Overridable per-call via ``start_langgraph_dev(port=...)`` /
|
||||
# ``ensure_langgraph_dev`` (which reads ``config.langgraph_dev_port``) and the
|
||||
@@ -211,9 +220,17 @@ class WorkspaceMismatchError(RuntimeError):
|
||||
"""
|
||||
|
||||
|
||||
def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None:
|
||||
def _write_workspace_sidecar(
|
||||
workspace_dir: Path,
|
||||
pid: int,
|
||||
config_fingerprint: str | None = None,
|
||||
deploy_mode: bool | None = None,
|
||||
) -> None:
|
||||
"""Record the workspace + pid of the langgraph dev we just started.
|
||||
|
||||
``config_fingerprint`` (optional) captures the launch-time config subset
|
||||
the server consumed; keepalive reuse compares it to detect drift.
|
||||
|
||||
Atomic write via temp-file + ``os.replace``: without this, a concurrent
|
||||
reader could observe a partially-written file, fail JSON parse, and
|
||||
silently downgrade to the "no sidecar" fallback path — which skips the
|
||||
@@ -228,9 +245,12 @@ def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None:
|
||||
try:
|
||||
RUNTIME.pid_dir.mkdir(parents=True, exist_ok=True)
|
||||
tmp = RUNTIME.workspace_sidecar.with_suffix(".json.tmp")
|
||||
tmp.write_text(
|
||||
json.dumps({"workspace": str(workspace_dir), "pid": pid}), encoding="utf-8"
|
||||
)
|
||||
payload: dict = {"workspace": str(workspace_dir), "pid": pid}
|
||||
if config_fingerprint is not None:
|
||||
payload["config_fingerprint"] = config_fingerprint
|
||||
if deploy_mode is not None:
|
||||
payload["deploy_mode"] = deploy_mode
|
||||
tmp.write_text(json.dumps(payload), encoding="utf-8")
|
||||
os.replace(tmp, RUNTIME.workspace_sidecar)
|
||||
except OSError as exc:
|
||||
logger.warning(
|
||||
@@ -569,6 +589,186 @@ def _kill_owned_stale_process(port: int) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
# Config fields that provably never reach the langgraph dev subprocess:
|
||||
# the channel stack + STT run in the CLI process, display/workspace/frontend
|
||||
# knobs shape the CLI itself, and keepalive is a lifecycle flag. Everything
|
||||
# NOT listed here counts toward the drift fingerprint, so a newly added
|
||||
# config field defaults to "affects the server" — the failure mode is a
|
||||
# spurious restart hint, never silent staleness.
|
||||
# Packaged sub-agent specs — consumed at graph build; module constant so
|
||||
# tests can redirect it.
|
||||
_SUBAGENTS_DIR = Path(__file__).resolve().parent.parent / "subagents"
|
||||
|
||||
_FINGERPRINT_EXCLUDED_PREFIXES = (
|
||||
"channel_",
|
||||
"imessage_",
|
||||
"telegram_",
|
||||
"discord_",
|
||||
"slack_",
|
||||
"feishu_",
|
||||
"wechat_",
|
||||
"dingtalk_",
|
||||
"email_",
|
||||
"qq_",
|
||||
"signal_",
|
||||
"stt_",
|
||||
)
|
||||
_FINGERPRINT_EXCLUDED_FIELDS = frozenset(
|
||||
{
|
||||
"require_mention",
|
||||
"text_chunk_limit",
|
||||
"allowed_channels",
|
||||
"dm_policy",
|
||||
"shared_webhook_port",
|
||||
"show_thinking",
|
||||
"ui_backend",
|
||||
"log_level",
|
||||
"default_mode",
|
||||
"default_workdir",
|
||||
"webui_port",
|
||||
"webui_host",
|
||||
"langgraph_dev_keepalive",
|
||||
"shell_allow_list",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _server_config_fingerprint(config: EvoScientistConfig) -> str:
|
||||
"""Hash of everything the langgraph dev subprocess consumes at launch.
|
||||
|
||||
Deployed graphs read config once at import (``subagents/_factory.py``,
|
||||
``EvoScientist.py``), so a keepalive server keeps serving those values
|
||||
until restarted. Iterates the full ``EvoScientistConfig`` field list
|
||||
minus the explicit exclusion set above — a new config field counts
|
||||
toward drift by default — and folds in ``mcp.yaml`` plus the packaged
|
||||
``subagents/*.yaml``, which are consumed at graph build too. Secrets
|
||||
only feed a truncated one-way digest; nothing recoverable is stored.
|
||||
getattr with defaults: deploy/WebUI (and their tests) routinely hand
|
||||
this module duck-typed config objects missing dataclass fields.
|
||||
"""
|
||||
parts = []
|
||||
for field in dataclass_fields(EvoScientistConfig):
|
||||
name = field.name
|
||||
if name in _FINGERPRINT_EXCLUDED_FIELDS or name.startswith(
|
||||
_FINGERPRINT_EXCLUDED_PREFIXES
|
||||
):
|
||||
continue
|
||||
parts.append((name, str(getattr(config, name, None))))
|
||||
digest = hashlib.sha256(repr(parts).encode("utf-8"))
|
||||
try:
|
||||
from EvoScientist.config.settings import get_config_dir
|
||||
|
||||
mcp_yaml = get_config_dir() / "mcp.yaml"
|
||||
if mcp_yaml.exists():
|
||||
digest.update(mcp_yaml.read_bytes())
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
for yaml_path in sorted(_SUBAGENTS_DIR.glob("*.yaml")):
|
||||
digest.update(yaml_path.name.encode("utf-8"))
|
||||
digest.update(yaml_path.read_bytes())
|
||||
except OSError:
|
||||
pass
|
||||
return digest.hexdigest()[:16]
|
||||
|
||||
|
||||
def stop_recorded_server() -> int | None:
|
||||
"""Explicitly stop the langgraph dev recorded in our PID file.
|
||||
|
||||
Backs the user-facing ``EvoSci server stop`` command — the deliberate
|
||||
counterpart to ``langgraph_dev_keepalive``: an opt-in server that
|
||||
outlives its CLI needs a first-class way to stop it. Ownership = our
|
||||
PID file + a live process whose cmdline still contains ``langgraph``
|
||||
(same loose anti-PID-recycling match as ``_kill_owned_stale_process``,
|
||||
with PID-file ownership as the primary guard). Holds the cross-process
|
||||
file lock so a concurrent start can't have its fresh PID/sidecar records
|
||||
wiped by this stop's cleanup. Kills the whole process tree, then removes
|
||||
the PID file + sidecar. Returns the stopped pid, or ``None`` when nothing
|
||||
was stopped (stale/corrupt files, if any, are still cleaned up).
|
||||
"""
|
||||
try:
|
||||
with FileLock(str(RUNTIME.lock_file), timeout=_FILE_LOCK_TIMEOUT):
|
||||
return _stop_recorded_server_locked()
|
||||
except FileLockTimeout:
|
||||
logger.warning(
|
||||
"Timed out waiting for the langgraph dev lock — another EvoSci "
|
||||
"process is mid lifecycle change; not stopping anything."
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _stop_recorded_server_locked() -> int | None:
|
||||
with _LOCK:
|
||||
if _PROCESS is not None and _PROCESS.poll() is None:
|
||||
pid = _PROCESS.pid
|
||||
stop_langgraph_dev()
|
||||
return pid
|
||||
if not RUNTIME.pid_file.exists():
|
||||
return None
|
||||
try:
|
||||
owned_pid = int(RUNTIME.pid_file.read_text(encoding="utf-8").strip())
|
||||
except ValueError:
|
||||
stop_langgraph_dev() # corrupt PID file — clean it up as promised
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
proc = psutil.Process(owned_pid)
|
||||
cmdline = proc.cmdline()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
stop_langgraph_dev() # dead/inaccessible — clean the stale files
|
||||
return None
|
||||
if not any("langgraph" in arg for arg in cmdline):
|
||||
stop_langgraph_dev() # pid recycled by a foreign process — files only
|
||||
return None
|
||||
try:
|
||||
children = proc.children(recursive=True)
|
||||
for child in children:
|
||||
try:
|
||||
child.terminate()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
pass
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except psutil.TimeoutExpired:
|
||||
for child in children:
|
||||
try:
|
||||
child.kill()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
pass
|
||||
proc.kill()
|
||||
# The parent exiting promptly doesn't prove its workers did — sweep
|
||||
# the pre-kill snapshot for survivors.
|
||||
for child in children:
|
||||
try:
|
||||
if child.is_running():
|
||||
child.kill()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
pass
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
pass
|
||||
stop_langgraph_dev()
|
||||
return owned_pid
|
||||
|
||||
|
||||
def _pid_serves_port(pid: object, port: int) -> bool:
|
||||
"""Best-effort check that ``pid`` is a langgraph dev serving ``port``.
|
||||
|
||||
Used to attribute an occupied port to the sidecar's recorded server
|
||||
before printing its details — avoids blaming a stale record. Relies on
|
||||
``--port`` always being in ``start_langgraph_dev``'s argv, not on
|
||||
port→PID mapping (root-only on macOS via psutil).
|
||||
"""
|
||||
if not isinstance(pid, int) or isinstance(pid, bool) or pid <= 0:
|
||||
return False
|
||||
try:
|
||||
cmdline = psutil.Process(pid).cmdline()
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
return False
|
||||
return any("langgraph" in arg for arg in cmdline) and str(port) in cmdline
|
||||
|
||||
|
||||
def _packaged_langgraph_config() -> Path:
|
||||
"""Return path to the package-shipped ``langgraph.json``.
|
||||
|
||||
@@ -595,6 +795,7 @@ def start_langgraph_dev(
|
||||
jobs_per_worker: int = 10,
|
||||
deploy_mode: bool = False,
|
||||
tunnel: bool = False,
|
||||
config_fingerprint: str | None = None,
|
||||
) -> subprocess.Popen:
|
||||
"""Start langgraph dev as a background subprocess.
|
||||
|
||||
@@ -816,7 +1017,12 @@ def start_langgraph_dev(
|
||||
except Exception:
|
||||
pass
|
||||
RUNTIME.pid_file.write_text(str(proc.pid), encoding="utf-8")
|
||||
_write_workspace_sidecar(workspace_dir=workspace_dir, pid=proc.pid)
|
||||
_write_workspace_sidecar(
|
||||
workspace_dir=workspace_dir,
|
||||
pid=proc.pid,
|
||||
config_fingerprint=config_fingerprint,
|
||||
deploy_mode=deploy_mode,
|
||||
)
|
||||
global _PROCESS_WORKSPACE
|
||||
_PROCESS = proc
|
||||
_PROCESS_WORKSPACE = workspace_dir
|
||||
@@ -993,7 +1199,8 @@ def ensure_langgraph_dev(
|
||||
still chat with sync sub-agents; only async sub-agent calls and EvoMemory
|
||||
background workers will fail.
|
||||
"""
|
||||
global _ASYNC_SUBAGENTS_AVAILABLE
|
||||
global _ASYNC_SUBAGENTS_AVAILABLE, CONFIG_DRIFT_SINCE_LAUNCH
|
||||
CONFIG_DRIFT_SINCE_LAUNCH = False
|
||||
|
||||
if not needs_langgraph_dev(config):
|
||||
_ASYNC_SUBAGENTS_AVAILABLE = False
|
||||
@@ -1032,7 +1239,8 @@ def _ensure_langgraph_dev_locked(
|
||||
workspace_dir: Path | str | None,
|
||||
) -> subprocess.Popen | None:
|
||||
"""Locked critical section of ``ensure_langgraph_dev`` — must hold ``_LOCK``."""
|
||||
global _ASYNC_SUBAGENTS_AVAILABLE
|
||||
global _ASYNC_SUBAGENTS_AVAILABLE, CONFIG_DRIFT_SINCE_LAUNCH
|
||||
config_fp = _server_config_fingerprint(config)
|
||||
port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT))
|
||||
host = str(getattr(config, "langgraph_dev_host", _DEFAULT_HOST) or _DEFAULT_HOST)
|
||||
file_persistence = bool(getattr(config, "langgraph_dev_file_persistence", True))
|
||||
@@ -1086,12 +1294,33 @@ def _ensure_langgraph_dev_locked(
|
||||
if sidecar is not None:
|
||||
recorded = Path(sidecar["workspace"]).resolve()
|
||||
if recorded != ws_path.resolve():
|
||||
hint = ""
|
||||
if getattr(config, "langgraph_dev_keepalive", False):
|
||||
# Only under keepalive can the server be an ownerless
|
||||
# leftover; without the flag the mismatch means a live
|
||||
# session, where a stop suggestion would be misleading.
|
||||
# Point at `EvoSci server stop` (not a raw kill): it
|
||||
# verifies ownership and cleans the PID/sidecar files,
|
||||
# so no stale records are left behind.
|
||||
hint = (
|
||||
" If it is a leftover keepalive server, stop it"
|
||||
" with: EvoSci server stop."
|
||||
)
|
||||
raise WorkspaceMismatchError(
|
||||
f"An EvoSci langgraph dev is already running on "
|
||||
f"{_base_url(port, host)} for workspace {recorded}, but the "
|
||||
f"current process requested workspace {ws_path}. "
|
||||
f"Stop the other EvoSci session (deploy / TUI / serve) "
|
||||
f"or rerun with --workdir {recorded}."
|
||||
f"or rerun with --workdir {recorded}." + hint
|
||||
)
|
||||
recorded_fp = sidecar.get("config_fingerprint")
|
||||
if isinstance(recorded_fp, str) and recorded_fp != config_fp:
|
||||
CONFIG_DRIFT_SINCE_LAUNCH = True
|
||||
logger.warning(
|
||||
"Config changed since the running langgraph dev was "
|
||||
"launched — async sub-agents still use the old "
|
||||
"settings until the server is restarted "
|
||||
"(EvoSci server stop)."
|
||||
)
|
||||
logger.info(
|
||||
"Reusing externally-managed langgraph dev on %s; sidecar "
|
||||
@@ -1125,6 +1354,7 @@ def _ensure_langgraph_dev_locked(
|
||||
host=host,
|
||||
file_persistence=file_persistence,
|
||||
jobs_per_worker=jobs_per_worker,
|
||||
config_fingerprint=config_fp,
|
||||
)
|
||||
except (FileNotFoundError, RuntimeError) as exc:
|
||||
# Startup failed — keep async subagents disabled so the main agent
|
||||
@@ -1141,5 +1371,10 @@ def _ensure_langgraph_dev_locked(
|
||||
return None
|
||||
|
||||
_ASYNC_SUBAGENTS_AVAILABLE = True
|
||||
atexit.register(stop_langgraph_dev, proc)
|
||||
if getattr(config, "langgraph_dev_keepalive", False):
|
||||
# Keepalive: leave the server (plus PID file + sidecar) behind on CLI
|
||||
# exit so the next start in this workspace reuses it instantly.
|
||||
logger.info("langgraph_dev_keepalive enabled — server will outlive this CLI.")
|
||||
else:
|
||||
atexit.register(stop_langgraph_dev, proc)
|
||||
return proc
|
||||
|
||||
@@ -14,7 +14,7 @@ import lazy_loader as _lazy
|
||||
|
||||
__getattr__, __dir__, __all__ = _lazy.attach(
|
||||
__name__,
|
||||
submodules=["context_window", "models", "patches"],
|
||||
submodules=["context_window", "models", "patches", "registry"],
|
||||
submod_attrs={
|
||||
"context_window": [
|
||||
"DEFAULT_CONTEXT_WINDOW_FALLBACK",
|
||||
@@ -22,9 +22,12 @@ __getattr__, __dir__, __all__ = _lazy.attach(
|
||||
"resolve_context_window",
|
||||
],
|
||||
"models": [
|
||||
"get_chat_model",
|
||||
],
|
||||
# Registry data resolves without the langchain/provider-SDK stack.
|
||||
"registry": [
|
||||
"DEFAULT_MODEL",
|
||||
"MODELS",
|
||||
"get_chat_model",
|
||||
"get_model_info",
|
||||
"get_models_for_provider",
|
||||
"list_models",
|
||||
|
||||
+15
-335
@@ -36,21 +36,21 @@ from .patches import (
|
||||
_patch_openrouter_strip_responses_reasoning,
|
||||
_patch_openrouter_structured_output,
|
||||
)
|
||||
|
||||
_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic"
|
||||
_SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1"
|
||||
|
||||
_ZHIPU_BASE_URL = "https://open.bigmodel.cn/api/paas/v4"
|
||||
_ZHIPU_CODE_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4"
|
||||
_VOLCENGINE_BASE_URL = "https://ark.cn-beijing.volces.com/api/v3"
|
||||
_VOLCENGINE_CODE_BASE_URL = "https://ark.cn-beijing.volces.com/api/coding/v3"
|
||||
_DASHSCOPE_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1"
|
||||
_DASHSCOPE_CODE_BASE_URL = "https://coding.dashscope.aliyuncs.com/v1"
|
||||
|
||||
_ATLASCLOUD_BASE_URL = "https://api.atlascloud.ai/v1"
|
||||
_MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1"
|
||||
_KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/"
|
||||
_REQUESTY_BASE_URL = "https://router.requesty.ai/v1"
|
||||
from .registry import (
|
||||
_ANTHROPIC_ROUTED_PROVIDERS,
|
||||
_MODEL_ENTRIES,
|
||||
_OPENAI_ROUTED_PROVIDERS,
|
||||
_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS, # noqa: F401 — re-exported
|
||||
_THINKING_CAPABLE_PROVIDERS,
|
||||
DEFAULT_MODEL,
|
||||
MODELS,
|
||||
_is_mandatory_thinking_kimi,
|
||||
get_model_info, # noqa: F401 — re-exported for existing import sites
|
||||
get_models_for_provider, # noqa: F401 — re-exported for existing import sites
|
||||
list_model_picker_entries, # noqa: F401 — re-exported for existing import sites
|
||||
list_models, # noqa: F401 — re-exported for existing import sites
|
||||
list_models_by_provider, # noqa: F401 — re-exported for existing import sites
|
||||
)
|
||||
|
||||
# Minimum Codex CLI version advertised when no explicit override is set. Newer
|
||||
# installed versions are advertised automatically.
|
||||
@@ -106,36 +106,6 @@ def _is_deepseek_endpoint(base_url: str | None) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
# Providers routed through the OpenAI provider with a custom base_url.
|
||||
# Maps provider name → (base_url or None, env var for API key).
|
||||
_OPENAI_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = {
|
||||
"atlascloud": (_ATLASCLOUD_BASE_URL, "ATLASCLOUD_API_KEY"),
|
||||
"moonshot": (_MOONSHOT_BASE_URL, "MOONSHOT_API_KEY"),
|
||||
"siliconflow": (_SILICONFLOW_BASE_URL, "SILICONFLOW_API_KEY"),
|
||||
"zhipu": (_ZHIPU_BASE_URL, "ZHIPU_API_KEY"),
|
||||
"zhipu-code": (_ZHIPU_CODE_BASE_URL, "ZHIPU_API_KEY"),
|
||||
"volcengine": (_VOLCENGINE_BASE_URL, "VOLCENGINE_API_KEY"),
|
||||
"volcengine-code": (_VOLCENGINE_CODE_BASE_URL, "VOLCENGINE_API_KEY"),
|
||||
"dashscope": (_DASHSCOPE_BASE_URL, "DASHSCOPE_API_KEY"),
|
||||
"dashscope-code": (_DASHSCOPE_CODE_BASE_URL, "DASHSCOPE_API_KEY"),
|
||||
"requesty": (_REQUESTY_BASE_URL, "REQUESTY_API_KEY"),
|
||||
"custom-openai": (
|
||||
None,
|
||||
"CUSTOM_OPENAI_API_KEY",
|
||||
), # base_url from CUSTOM_OPENAI_BASE_URL env
|
||||
}
|
||||
|
||||
# Providers routed through the Anthropic provider with a custom base_url.
|
||||
# Maps provider name → (base_url or None, env var for API key).
|
||||
_ANTHROPIC_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = {
|
||||
"minimax": (_MINIMAX_ANTHROPIC_BASE_URL, "MINIMAX_API_KEY"),
|
||||
"kimi-coding": (_KIMI_CODING_BASE_URL, "KIMI_API_KEY"),
|
||||
"custom-anthropic": (None, "CUSTOM_ANTHROPIC_API_KEY"),
|
||||
}
|
||||
|
||||
# Anthropic-routed providers that support extended thinking.
|
||||
_THINKING_CAPABLE_PROVIDERS: set[str] = {"minimax"}
|
||||
|
||||
_TRUTHY_ENV_VALUES = {"1", "true", "yes", "on"}
|
||||
_FALSEY_ENV_VALUES = {"0", "false", "no", "off"}
|
||||
|
||||
@@ -147,233 +117,6 @@ _FALSEY_ENV_VALUES = {"0", "false", "no", "off"}
|
||||
# capped to this many below. https://openrouter.ai/docs/app-attribution
|
||||
_OPENROUTER_MAX_CATEGORIES_PER_REQUEST = 2
|
||||
|
||||
# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3
|
||||
# cannot disable thinking — structured output must use json_schema there.
|
||||
# Moonshot-specific: do NOT widen to other mandatory-reasoning models.
|
||||
_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset(
|
||||
{"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"}
|
||||
)
|
||||
|
||||
|
||||
def _is_mandatory_thinking_kimi(model_id: str) -> bool:
|
||||
"""True for Kimi models whose thinking cannot be disabled (K3 family)."""
|
||||
short_id = model_id.split("/")[-1]
|
||||
return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding"
|
||||
|
||||
|
||||
# Model registry: list of (short_name, model_id, provider)
|
||||
# Allows same short_name across different providers.
|
||||
_MODEL_ENTRIES: list[tuple[str, str, str]] = [
|
||||
# Custom Anthropic (third-party Claude-compatible endpoints, current-gen defaults)
|
||||
# Listed BEFORE native anthropic so MODELS dict defaults to native provider
|
||||
("claude-sonnet-4-6", "claude-sonnet-4-6", "custom-anthropic"),
|
||||
("claude-haiku-4-5", "claude-haiku-4-5", "custom-anthropic"),
|
||||
# Custom OpenAI (third-party OpenAI-compatible endpoints, 3 defaults)
|
||||
# Listed BEFORE native openai so MODELS dict defaults to native provider
|
||||
("gpt-5.5-pro", "gpt-5.5-pro", "custom-openai"),
|
||||
("gpt-5.5", "gpt-5.5", "custom-openai"),
|
||||
("gpt-5.4", "gpt-5.4", "custom-openai"),
|
||||
("gpt-5.3-codex", "gpt-5.3-codex", "custom-openai"),
|
||||
("gpt-5-mini", "gpt-5-mini", "custom-openai"),
|
||||
# Atlas Cloud (OpenAI-compatible)
|
||||
("qwen3.5-27b", "qwen/qwen3.5-27b", "atlascloud"),
|
||||
# Anthropic (current generation)
|
||||
("claude-fable-5", "claude-fable-5", "anthropic"),
|
||||
("claude-opus-5", "claude-opus-5", "anthropic"),
|
||||
("claude-opus-4-8", "claude-opus-4-8", "anthropic"),
|
||||
("claude-sonnet-5", "claude-sonnet-5", "anthropic"),
|
||||
("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"),
|
||||
("claude-haiku-4-5", "claude-haiku-4-5", "anthropic"),
|
||||
# OpenAI
|
||||
("gpt-5.6-sol", "gpt-5.6-sol", "openai"),
|
||||
("gpt-5.6-terra", "gpt-5.6-terra", "openai"),
|
||||
("gpt-5.6-luna", "gpt-5.6-luna", "openai"),
|
||||
("gpt-5.5-pro", "gpt-5.5-pro", "openai"),
|
||||
("gpt-5.5", "gpt-5.5", "openai"),
|
||||
("gpt-5.4", "gpt-5.4", "openai"),
|
||||
("gpt-5.4-mini", "gpt-5.4-mini", "openai"),
|
||||
("gpt-5.4-nano", "gpt-5.4-nano", "openai"),
|
||||
("gpt-5.3-codex", "gpt-5.3-codex", "openai"),
|
||||
("gpt-5.2-codex", "gpt-5.2-codex", "openai"),
|
||||
("gpt-5.2", "gpt-5.2", "openai"),
|
||||
("gpt-5.1", "gpt-5.1", "openai"),
|
||||
("gpt-5", "gpt-5", "openai"),
|
||||
("gpt-5-mini", "gpt-5-mini", "openai"),
|
||||
("gpt-5-nano", "gpt-5-nano", "openai"),
|
||||
# Google GenAI
|
||||
("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"),
|
||||
("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"),
|
||||
("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"),
|
||||
("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"),
|
||||
(
|
||||
"gemini-3.1-pro-customtools",
|
||||
"gemini-3.1-pro-preview-customtools",
|
||||
"google-genai",
|
||||
),
|
||||
("gemini-3.1-flash-lite", "gemini-3.1-flash-lite-preview", "google-genai"),
|
||||
("gemini-3-flash", "gemini-3-flash-preview", "google-genai"),
|
||||
("gemini-2.5-flash", "gemini-2.5-flash", "google-genai"),
|
||||
("gemini-2.5-flash-lite", "gemini-2.5-flash-lite", "google-genai"),
|
||||
("gemini-2.5-pro", "gemini-2.5-pro", "google-genai"),
|
||||
# MiniMax (direct API — Anthropic-compatible; default: api.minimaxi.com, global: api.minimax.io)
|
||||
("minimax-m3", "MiniMax-M3", "minimax"),
|
||||
("minimax-m2.7", "MiniMax-M2.7", "minimax"),
|
||||
("minimax-m2.7-highspeed", "MiniMax-M2.7-highspeed", "minimax"),
|
||||
("minimax-m2.5", "MiniMax-M2.5", "minimax"),
|
||||
("minimax-m2.5-highspeed", "MiniMax-M2.5-highspeed", "minimax"),
|
||||
# NVIDIA
|
||||
("nemotron-super", "nvidia/nemotron-3-super-120b-a12b", "nvidia"),
|
||||
("nemotron-nano", "nvidia/nemotron-3-nano-30b-a3b", "nvidia"),
|
||||
("glm-5.2", "z-ai/glm-5.2", "nvidia"),
|
||||
("glm4.7", "z-ai/glm4.7", "nvidia"),
|
||||
("deepseek-v3.2", "deepseek-ai/deepseek-v3.2", "nvidia"),
|
||||
("deepseek-v3.1", "deepseek-ai/deepseek-v3.1-terminus", "nvidia"),
|
||||
("kimi-k2.5", "moonshotai/kimi-k2.5", "nvidia"),
|
||||
("kimi-k2-thinking", "moonshotai/kimi-k2-thinking", "nvidia"),
|
||||
("minimax-m2.5", "minimaxai/minimax-m2.5", "nvidia"),
|
||||
("minimax-m2.1", "minimaxai/minimax-m2.1", "nvidia"),
|
||||
("qwen3.5-397b", "qwen/qwen3.5-397b-a17b", "nvidia"),
|
||||
("step-3.5-flash", "stepfun-ai/step-3.5-flash", "nvidia"),
|
||||
# SiliconFlow
|
||||
("minimax-m2.5", "Pro/MiniMaxAI/MiniMax-M2.5", "siliconflow"),
|
||||
("glm-5.2", "Pro/zai-org/GLM-5.2", "siliconflow"),
|
||||
("glm-5", "Pro/zai-org/GLM-5", "siliconflow"),
|
||||
("kimi-k2.5", "Pro/moonshotai/Kimi-K2.5", "siliconflow"),
|
||||
("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"),
|
||||
# Requesty (aggregator — OpenAI-compatible router, provider/model IDs).
|
||||
# Listed before OpenRouter so that for model names shared with OpenRouter
|
||||
# or a native provider, Requesty does not override them (the dict below is
|
||||
# last-entry-wins); Requesty is selected explicitly via get_models_for_provider.
|
||||
("claude-sonnet-4.6", "anthropic/claude-sonnet-4-6", "requesty"),
|
||||
("claude-opus-4.8", "anthropic/claude-opus-4-8", "requesty"),
|
||||
("gemini-3.5-flash", "google/gemini-3.5-flash", "requesty"),
|
||||
("grok-4.3", "xai/grok-4.3", "requesty"),
|
||||
("grok-build-0.1", "xai/grok-build-0.1", "requesty"),
|
||||
# OpenRouter
|
||||
("claude-fable-5", "anthropic/claude-fable-5", "openrouter"),
|
||||
("claude-opus-5", "anthropic/claude-opus-5", "openrouter"),
|
||||
("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"),
|
||||
("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"),
|
||||
("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"),
|
||||
("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"),
|
||||
("claude-sonnet-4.6", "anthropic/claude-sonnet-4.6", "openrouter"),
|
||||
("gpt-5.6-sol", "openai/gpt-5.6-sol", "openrouter"),
|
||||
("gpt-5.6-terra", "openai/gpt-5.6-terra", "openrouter"),
|
||||
("gpt-5.6-luna", "openai/gpt-5.6-luna", "openrouter"),
|
||||
("gpt-5.5-pro", "openai/gpt-5.5-pro", "openrouter"),
|
||||
("gpt-5.5", "openai/gpt-5.5", "openrouter"),
|
||||
("gpt-5.4", "openai/gpt-5.4", "openrouter"),
|
||||
("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"),
|
||||
("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"),
|
||||
("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"),
|
||||
("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"),
|
||||
("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"),
|
||||
("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"),
|
||||
("kimi-k3", "moonshotai/kimi-k3", "openrouter"),
|
||||
("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"),
|
||||
("glm-5.2", "z-ai/glm-5.2", "openrouter"),
|
||||
("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"),
|
||||
("minimax-m3", "minimax/minimax-m3", "openrouter"),
|
||||
("mimo-v2.5-pro", "xiaomi/mimo-v2.5-pro", "openrouter"),
|
||||
("mimo-v2.5", "xiaomi/mimo-v2.5", "openrouter"),
|
||||
("grok-build-0.1", "x-ai/grok-build-0.1", "openrouter"),
|
||||
("grok-4.5", "x-ai/grok-4.5", "openrouter"),
|
||||
("hy3", "tencent/hy3", "openrouter"),
|
||||
("qwen3.8-max", "qwen/qwen3.8-max", "openrouter"),
|
||||
("qwen3.7-max", "qwen/qwen3.7-max", "openrouter"),
|
||||
("qwen3.7-plus", "qwen/qwen3.7-plus", "openrouter"),
|
||||
("qwen3.6-flash", "qwen/qwen3.6-flash", "openrouter"),
|
||||
("qwen3.5-122b", "qwen/qwen3.5-122b-a10b", "openrouter"),
|
||||
("deepseek-v4-pro", "deepseek/deepseek-v4-pro", "openrouter"),
|
||||
("deepseek-v4-flash", "deepseek/deepseek-v4-flash", "openrouter"),
|
||||
# Volcengine Coding Plan (火山引擎代码计划 — coding-only endpoint)
|
||||
# Listed before Zhipu so simple GLM lookups keep their existing default.
|
||||
("glm-5.2", "glm-5-2", "volcengine-code"),
|
||||
("kimi-k2.5", "kimi-k2-5", "volcengine-code"),
|
||||
# Zhipu CodePlan (智谱代码计划 — coding-only endpoint)
|
||||
("glm-5.2", "glm-5.2", "zhipu-code"),
|
||||
("glm-5.1", "glm-5.1", "zhipu-code"),
|
||||
("glm-5", "glm-5", "zhipu-code"),
|
||||
("glm-5-turbo", "glm-5-turbo", "zhipu-code"),
|
||||
("glm-5v-turbo", "glm-5v-turbo", "zhipu-code"),
|
||||
("glm-4.7", "glm-4.7", "zhipu-code"),
|
||||
# Zhipu (智谱 — general endpoint, default for simple lookups)
|
||||
("glm-5.2", "glm-5.2", "zhipu"),
|
||||
("glm-5.1", "glm-5.1", "zhipu"),
|
||||
("glm-5", "glm-5", "zhipu"),
|
||||
("glm-5-turbo", "glm-5-turbo", "zhipu"),
|
||||
("glm-5v-turbo", "glm-5v-turbo", "zhipu"),
|
||||
("glm-4.7", "glm-4.7", "zhipu"),
|
||||
# Volcengine (火山引擎 — Doubao models)
|
||||
("doubao-seed-2.0-pro", "doubao-seed-2-0-pro-260215", "volcengine"),
|
||||
("doubao-seed-2.0-lite", "doubao-seed-2-0-lite-260215", "volcengine"),
|
||||
("doubao-seed-2.0-mini", "doubao-seed-2-0-mini-260215", "volcengine"),
|
||||
("doubao-seed-2.0-code", "doubao-seed-2-0-code-preview-260215", "volcengine"),
|
||||
("doubao-seed-1.6", "doubao-seed-1.6", "volcengine"),
|
||||
("doubao-1.5-pro", "doubao-1.5-pro-256k", "volcengine"),
|
||||
("doubao-1.5-thinking-pro", "doubao-1.5-thinking-pro", "volcengine"),
|
||||
# DashScope Coding Plan (阿里云代码计划 — subscription sk-sp-* endpoint)
|
||||
("qwen3.8-max", "qwen3.8-max", "dashscope-code"),
|
||||
("qwen3.7-max", "qwen3.7-max", "dashscope-code"),
|
||||
("qwen3.7-plus", "qwen3.7-plus", "dashscope-code"),
|
||||
("qwen3.6-max", "qwen3.6-max-preview", "dashscope-code"),
|
||||
("qwen3.6-plus", "qwen3.6-plus", "dashscope-code"),
|
||||
("qwen3.6-flash", "qwen3.6-flash", "dashscope-code"),
|
||||
("qwen3-coder", "qwen3-coder-plus", "dashscope-code"),
|
||||
("qwen3-coder-next", "qwen3-coder-next", "dashscope-code"),
|
||||
("qwen3-max", "qwen3-max", "dashscope-code"),
|
||||
("qwen3.5-plus", "qwen3.5-plus", "dashscope-code"),
|
||||
# DashScope (阿里云 — Qwen models, default for simple lookups)
|
||||
("qwen3.8-max", "qwen3.8-max", "dashscope"),
|
||||
("qwen3.7-max", "qwen3.7-max", "dashscope"),
|
||||
("qwen3.7-plus", "qwen3.7-plus", "dashscope"),
|
||||
("qwen3.6-max", "qwen3.6-max-preview", "dashscope"),
|
||||
("qwen3.6-plus", "qwen3.6-plus", "dashscope"),
|
||||
("qwen3.6-flash", "qwen3.6-flash", "dashscope"),
|
||||
("qwen3-coder", "qwen3-coder-plus", "dashscope"),
|
||||
("qwen3-235b", "qwen3-235b-a22b", "dashscope"),
|
||||
("qwen-max", "qwen-max", "dashscope"),
|
||||
("qwq-plus", "qwq-plus", "dashscope"),
|
||||
# DeepSeek
|
||||
("deepseek-v4-pro", "deepseek-v4-pro", "deepseek"),
|
||||
("deepseek-v4-flash", "deepseek-v4-flash", "deepseek"),
|
||||
# Legacy aliases (deprecated 2026-07-24; route to v4-flash thinking/non-thinking)
|
||||
("deepseek-r1", "deepseek-reasoner", "deepseek"),
|
||||
("deepseek-v3", "deepseek-chat", "deepseek"),
|
||||
# Moonshot (OpenAI-compatible)
|
||||
("kimi-k3", "kimi-k3", "moonshot"),
|
||||
("kimi-k2.6", "kimi-k2.6", "moonshot"),
|
||||
("kimi-k2.5", "kimi-k2.5", "moonshot"),
|
||||
("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"),
|
||||
("kimi-k2-thinking-turbo", "kimi-k2-thinking-turbo", "moonshot"),
|
||||
("moonshot-v1-auto", "moonshot-v1-auto", "moonshot"),
|
||||
("moonshot-v1-128k", "moonshot-v1-128k", "moonshot"),
|
||||
("moonshot-v1-32k", "moonshot-v1-32k", "moonshot"),
|
||||
("moonshot-v1-8k", "moonshot-v1-8k", "moonshot"),
|
||||
# Kimi Coding Plan (Anthropic-compatible)
|
||||
("kimi-for-coding", "kimi-for-coding", "kimi-coding"),
|
||||
]
|
||||
|
||||
# Public dict for simple lookups (last entry wins for duplicate names).
|
||||
# Use get_models_for_provider() for provider-aware lookups.
|
||||
MODELS: dict[str, tuple[str, str]] = {
|
||||
name: (model_id, provider) for name, model_id, provider in _MODEL_ENTRIES
|
||||
}
|
||||
|
||||
DEFAULT_MODEL = "claude-sonnet-4-6"
|
||||
|
||||
|
||||
def get_models_for_provider(provider: str) -> list[tuple[str, str]]:
|
||||
"""Get all models for a specific provider.
|
||||
|
||||
Args:
|
||||
provider: Provider name (e.g., 'anthropic', 'openrouter').
|
||||
|
||||
Returns:
|
||||
List of (short_name, model_id) tuples for the provider.
|
||||
"""
|
||||
return [(name, model_id) for name, model_id, p in _MODEL_ENTRIES if p == provider]
|
||||
|
||||
|
||||
def _env_flag_enabled(name: str) -> bool:
|
||||
return os.environ.get(name, "").strip().lower() in _TRUTHY_ENV_VALUES
|
||||
@@ -839,66 +582,3 @@ def get_chat_model(
|
||||
apply_known_context_window(chat_model)
|
||||
|
||||
return chat_model
|
||||
|
||||
|
||||
def list_models() -> list[str]:
|
||||
"""List all available model short names.
|
||||
|
||||
Returns:
|
||||
List of unique model short names that can be passed to get_chat_model().
|
||||
"""
|
||||
seen = set()
|
||||
result = []
|
||||
for name, _, _ in _MODEL_ENTRIES:
|
||||
if name not in seen:
|
||||
seen.add(name)
|
||||
result.append(name)
|
||||
return result
|
||||
|
||||
|
||||
def list_models_by_provider() -> list[tuple[str, str, str]]:
|
||||
"""List all unique (short_name, model_id, provider) entries.
|
||||
|
||||
Returns:
|
||||
De-duplicated list of model entries preserving registry order.
|
||||
"""
|
||||
seen: set[tuple[str, str]] = set()
|
||||
result: list[tuple[str, str, str]] = []
|
||||
for name, model_id, provider in _MODEL_ENTRIES:
|
||||
key = (name, provider)
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
result.append((name, model_id, provider))
|
||||
return result
|
||||
|
||||
|
||||
async def list_model_picker_entries(
|
||||
ollama_base_url: str | None,
|
||||
*,
|
||||
include_custom_ollama: bool,
|
||||
) -> list[tuple[str, str, str]]:
|
||||
"""Return model picker entries, optionally including local Ollama models."""
|
||||
entries = list_models_by_provider()
|
||||
if ollama_base_url:
|
||||
from .ollama_discovery import discover_ollama_models
|
||||
|
||||
for detected_name in await discover_ollama_models(
|
||||
ollama_base_url,
|
||||
timeout=1.5,
|
||||
):
|
||||
entries.append((detected_name, detected_name, "ollama"))
|
||||
if include_custom_ollama:
|
||||
entries.append(("Custom Ollama model...", "__custom_ollama__", "ollama"))
|
||||
return entries
|
||||
|
||||
|
||||
def get_model_info(model: str) -> tuple[str, str] | None:
|
||||
"""Get the (model_id, provider) tuple for a short name.
|
||||
|
||||
Args:
|
||||
model: Short model name.
|
||||
|
||||
Returns:
|
||||
Tuple of (model_id, provider) or None if not found.
|
||||
"""
|
||||
return MODELS.get(model)
|
||||
|
||||
@@ -0,0 +1,343 @@
|
||||
"""Model registry data — short names, model ids, providers, routing tables.
|
||||
|
||||
Pure data with no langchain/provider-SDK imports: the onboard wizard, the
|
||||
``/model`` pickers, and provider validation read this registry without paying
|
||||
for the chat-model construction stack in :mod:`.models` (~2000 modules).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
_MINIMAX_ANTHROPIC_BASE_URL = "https://api.minimaxi.com/anthropic"
|
||||
_SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1"
|
||||
|
||||
_ZHIPU_BASE_URL = "https://open.bigmodel.cn/api/paas/v4"
|
||||
_ZHIPU_CODE_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4"
|
||||
_VOLCENGINE_BASE_URL = "https://ark.cn-beijing.volces.com/api/v3"
|
||||
_VOLCENGINE_CODE_BASE_URL = "https://ark.cn-beijing.volces.com/api/coding/v3"
|
||||
_DASHSCOPE_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1"
|
||||
_DASHSCOPE_CODE_BASE_URL = "https://coding.dashscope.aliyuncs.com/v1"
|
||||
|
||||
_ATLASCLOUD_BASE_URL = "https://api.atlascloud.ai/v1"
|
||||
_MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1"
|
||||
_KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/"
|
||||
_REQUESTY_BASE_URL = "https://router.requesty.ai/v1"
|
||||
|
||||
# Providers routed through the OpenAI provider with a custom base_url.
|
||||
# Maps provider name → (base_url or None, env var for API key).
|
||||
_OPENAI_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = {
|
||||
"atlascloud": (_ATLASCLOUD_BASE_URL, "ATLASCLOUD_API_KEY"),
|
||||
"moonshot": (_MOONSHOT_BASE_URL, "MOONSHOT_API_KEY"),
|
||||
"siliconflow": (_SILICONFLOW_BASE_URL, "SILICONFLOW_API_KEY"),
|
||||
"zhipu": (_ZHIPU_BASE_URL, "ZHIPU_API_KEY"),
|
||||
"zhipu-code": (_ZHIPU_CODE_BASE_URL, "ZHIPU_API_KEY"),
|
||||
"volcengine": (_VOLCENGINE_BASE_URL, "VOLCENGINE_API_KEY"),
|
||||
"volcengine-code": (_VOLCENGINE_CODE_BASE_URL, "VOLCENGINE_API_KEY"),
|
||||
"dashscope": (_DASHSCOPE_BASE_URL, "DASHSCOPE_API_KEY"),
|
||||
"dashscope-code": (_DASHSCOPE_CODE_BASE_URL, "DASHSCOPE_API_KEY"),
|
||||
"requesty": (_REQUESTY_BASE_URL, "REQUESTY_API_KEY"),
|
||||
"custom-openai": (
|
||||
None,
|
||||
"CUSTOM_OPENAI_API_KEY",
|
||||
), # base_url from CUSTOM_OPENAI_BASE_URL env
|
||||
}
|
||||
|
||||
# Providers routed through the Anthropic provider with a custom base_url.
|
||||
# Maps provider name → (base_url or None, env var for API key).
|
||||
_ANTHROPIC_ROUTED_PROVIDERS: dict[str, tuple[str | None, str]] = {
|
||||
"minimax": (_MINIMAX_ANTHROPIC_BASE_URL, "MINIMAX_API_KEY"),
|
||||
"kimi-coding": (_KIMI_CODING_BASE_URL, "KIMI_API_KEY"),
|
||||
"custom-anthropic": (None, "CUSTOM_ANTHROPIC_API_KEY"),
|
||||
}
|
||||
|
||||
# Anthropic-routed providers that support extended thinking.
|
||||
_THINKING_CAPABLE_PROVIDERS: set[str] = {"minimax"}
|
||||
|
||||
# Moonshot rejects a forced tool choice while thinking is enabled, and kimi-k3
|
||||
# cannot disable thinking — structured output must use json_schema there.
|
||||
# Moonshot-specific: do NOT widen to other mandatory-reasoning models.
|
||||
_OPENROUTER_JSON_SCHEMA_STRUCTURED_OUTPUT_MODELS = frozenset(
|
||||
{"moonshotai/kimi-k3", "moonshotai/kimi-k3-20260715"}
|
||||
)
|
||||
|
||||
|
||||
def _is_mandatory_thinking_kimi(model_id: str) -> bool:
|
||||
"""True for Kimi models whose thinking cannot be disabled (K3 family)."""
|
||||
short_id = model_id.split("/")[-1]
|
||||
return short_id.startswith("kimi-k3") or short_id == "kimi-for-coding"
|
||||
|
||||
|
||||
# Model registry: list of (short_name, model_id, provider)
|
||||
# Allows same short_name across different providers.
|
||||
_MODEL_ENTRIES: list[tuple[str, str, str]] = [
|
||||
# Custom Anthropic (third-party Claude-compatible endpoints, current-gen defaults)
|
||||
# Listed BEFORE native anthropic so MODELS dict defaults to native provider
|
||||
("claude-sonnet-4-6", "claude-sonnet-4-6", "custom-anthropic"),
|
||||
("claude-haiku-4-5", "claude-haiku-4-5", "custom-anthropic"),
|
||||
# Custom OpenAI (third-party OpenAI-compatible endpoints, 3 defaults)
|
||||
# Listed BEFORE native openai so MODELS dict defaults to native provider
|
||||
("gpt-5.5-pro", "gpt-5.5-pro", "custom-openai"),
|
||||
("gpt-5.5", "gpt-5.5", "custom-openai"),
|
||||
("gpt-5.4", "gpt-5.4", "custom-openai"),
|
||||
("gpt-5.3-codex", "gpt-5.3-codex", "custom-openai"),
|
||||
("gpt-5-mini", "gpt-5-mini", "custom-openai"),
|
||||
# Atlas Cloud (OpenAI-compatible)
|
||||
("qwen3.5-27b", "qwen/qwen3.5-27b", "atlascloud"),
|
||||
# Anthropic (current generation)
|
||||
("claude-fable-5", "claude-fable-5", "anthropic"),
|
||||
("claude-opus-5", "claude-opus-5", "anthropic"),
|
||||
("claude-opus-4-8", "claude-opus-4-8", "anthropic"),
|
||||
("claude-sonnet-5", "claude-sonnet-5", "anthropic"),
|
||||
("claude-sonnet-4-6", "claude-sonnet-4-6", "anthropic"),
|
||||
("claude-haiku-4-5", "claude-haiku-4-5", "anthropic"),
|
||||
# OpenAI
|
||||
("gpt-5.6-sol", "gpt-5.6-sol", "openai"),
|
||||
("gpt-5.6-terra", "gpt-5.6-terra", "openai"),
|
||||
("gpt-5.6-luna", "gpt-5.6-luna", "openai"),
|
||||
("gpt-5.5-pro", "gpt-5.5-pro", "openai"),
|
||||
("gpt-5.5", "gpt-5.5", "openai"),
|
||||
("gpt-5.4", "gpt-5.4", "openai"),
|
||||
("gpt-5.4-mini", "gpt-5.4-mini", "openai"),
|
||||
("gpt-5.4-nano", "gpt-5.4-nano", "openai"),
|
||||
("gpt-5.3-codex", "gpt-5.3-codex", "openai"),
|
||||
("gpt-5.2-codex", "gpt-5.2-codex", "openai"),
|
||||
("gpt-5.2", "gpt-5.2", "openai"),
|
||||
("gpt-5.1", "gpt-5.1", "openai"),
|
||||
("gpt-5", "gpt-5", "openai"),
|
||||
("gpt-5-mini", "gpt-5-mini", "openai"),
|
||||
("gpt-5-nano", "gpt-5-nano", "openai"),
|
||||
# Google GenAI
|
||||
("gemini-3.6-flash", "gemini-3.6-flash", "google-genai"),
|
||||
("gemini-3.5-flash", "gemini-3.5-flash", "google-genai"),
|
||||
("gemini-3.5-flash-lite", "gemini-3.5-flash-lite", "google-genai"),
|
||||
("gemini-3.1-pro", "gemini-3.1-pro-preview", "google-genai"),
|
||||
(
|
||||
"gemini-3.1-pro-customtools",
|
||||
"gemini-3.1-pro-preview-customtools",
|
||||
"google-genai",
|
||||
),
|
||||
("gemini-3.1-flash-lite", "gemini-3.1-flash-lite-preview", "google-genai"),
|
||||
("gemini-3-flash", "gemini-3-flash-preview", "google-genai"),
|
||||
("gemini-2.5-flash", "gemini-2.5-flash", "google-genai"),
|
||||
("gemini-2.5-flash-lite", "gemini-2.5-flash-lite", "google-genai"),
|
||||
("gemini-2.5-pro", "gemini-2.5-pro", "google-genai"),
|
||||
# MiniMax (direct API — Anthropic-compatible; default: api.minimaxi.com, global: api.minimax.io)
|
||||
("minimax-m3", "MiniMax-M3", "minimax"),
|
||||
("minimax-m2.7", "MiniMax-M2.7", "minimax"),
|
||||
("minimax-m2.7-highspeed", "MiniMax-M2.7-highspeed", "minimax"),
|
||||
("minimax-m2.5", "MiniMax-M2.5", "minimax"),
|
||||
("minimax-m2.5-highspeed", "MiniMax-M2.5-highspeed", "minimax"),
|
||||
# NVIDIA
|
||||
("nemotron-super", "nvidia/nemotron-3-super-120b-a12b", "nvidia"),
|
||||
("nemotron-nano", "nvidia/nemotron-3-nano-30b-a3b", "nvidia"),
|
||||
("glm-5.2", "z-ai/glm-5.2", "nvidia"),
|
||||
("glm4.7", "z-ai/glm4.7", "nvidia"),
|
||||
("deepseek-v3.2", "deepseek-ai/deepseek-v3.2", "nvidia"),
|
||||
("deepseek-v3.1", "deepseek-ai/deepseek-v3.1-terminus", "nvidia"),
|
||||
("kimi-k2.5", "moonshotai/kimi-k2.5", "nvidia"),
|
||||
("kimi-k2-thinking", "moonshotai/kimi-k2-thinking", "nvidia"),
|
||||
("minimax-m2.5", "minimaxai/minimax-m2.5", "nvidia"),
|
||||
("minimax-m2.1", "minimaxai/minimax-m2.1", "nvidia"),
|
||||
("qwen3.5-397b", "qwen/qwen3.5-397b-a17b", "nvidia"),
|
||||
("step-3.5-flash", "stepfun-ai/step-3.5-flash", "nvidia"),
|
||||
# SiliconFlow
|
||||
("minimax-m2.5", "Pro/MiniMaxAI/MiniMax-M2.5", "siliconflow"),
|
||||
("glm-5.2", "Pro/zai-org/GLM-5.2", "siliconflow"),
|
||||
("glm-5", "Pro/zai-org/GLM-5", "siliconflow"),
|
||||
("kimi-k2.5", "Pro/moonshotai/Kimi-K2.5", "siliconflow"),
|
||||
("glm-4.7", "Pro/zai-org/GLM-4.7", "siliconflow"),
|
||||
# Requesty (aggregator — OpenAI-compatible router, provider/model IDs).
|
||||
# Listed before OpenRouter so that for model names shared with OpenRouter
|
||||
# or a native provider, Requesty does not override them (the dict below is
|
||||
# last-entry-wins); Requesty is selected explicitly via get_models_for_provider.
|
||||
("claude-sonnet-4.6", "anthropic/claude-sonnet-4-6", "requesty"),
|
||||
("claude-opus-4.8", "anthropic/claude-opus-4-8", "requesty"),
|
||||
("gemini-3.5-flash", "google/gemini-3.5-flash", "requesty"),
|
||||
("grok-4.3", "xai/grok-4.3", "requesty"),
|
||||
("grok-build-0.1", "xai/grok-build-0.1", "requesty"),
|
||||
# OpenRouter
|
||||
("claude-fable-5", "anthropic/claude-fable-5", "openrouter"),
|
||||
("claude-opus-5", "anthropic/claude-opus-5", "openrouter"),
|
||||
("claude-opus-5-fast", "anthropic/claude-opus-5-fast", "openrouter"),
|
||||
("claude-opus-4.8", "anthropic/claude-opus-4.8", "openrouter"),
|
||||
("claude-opus-4.8-fast", "anthropic/claude-opus-4.8-fast", "openrouter"),
|
||||
("claude-sonnet-5", "anthropic/claude-sonnet-5", "openrouter"),
|
||||
("claude-sonnet-4.6", "anthropic/claude-sonnet-4.6", "openrouter"),
|
||||
("gpt-5.6-sol", "openai/gpt-5.6-sol", "openrouter"),
|
||||
("gpt-5.6-terra", "openai/gpt-5.6-terra", "openrouter"),
|
||||
("gpt-5.6-luna", "openai/gpt-5.6-luna", "openrouter"),
|
||||
("gpt-5.5-pro", "openai/gpt-5.5-pro", "openrouter"),
|
||||
("gpt-5.5", "openai/gpt-5.5", "openrouter"),
|
||||
("gpt-5.4", "openai/gpt-5.4", "openrouter"),
|
||||
("gpt-5.3-codex", "openai/gpt-5.3-codex", "openrouter"),
|
||||
("gemini-3.6-flash", "google/gemini-3.6-flash", "openrouter"),
|
||||
("gemini-3.5-flash", "google/gemini-3.5-flash", "openrouter"),
|
||||
("gemini-3.5-flash-lite", "google/gemini-3.5-flash-lite", "openrouter"),
|
||||
("gemini-3.1-pro", "google/gemini-3.1-pro-preview", "openrouter"),
|
||||
("gemini-3-flash", "google/gemini-3-flash-preview", "openrouter"),
|
||||
("kimi-k3", "moonshotai/kimi-k3", "openrouter"),
|
||||
("kimi-k2.6", "moonshotai/kimi-k2.6", "openrouter"),
|
||||
("glm-5.2", "z-ai/glm-5.2", "openrouter"),
|
||||
("glm-5v-turbo", "z-ai/glm-5v-turbo", "openrouter"),
|
||||
("minimax-m3", "minimax/minimax-m3", "openrouter"),
|
||||
("mimo-v2.5-pro", "xiaomi/mimo-v2.5-pro", "openrouter"),
|
||||
("mimo-v2.5", "xiaomi/mimo-v2.5", "openrouter"),
|
||||
("grok-build-0.1", "x-ai/grok-build-0.1", "openrouter"),
|
||||
("grok-4.5", "x-ai/grok-4.5", "openrouter"),
|
||||
("hy3", "tencent/hy3", "openrouter"),
|
||||
("qwen3.8-max", "qwen/qwen3.8-max", "openrouter"),
|
||||
("qwen3.7-max", "qwen/qwen3.7-max", "openrouter"),
|
||||
("qwen3.7-plus", "qwen/qwen3.7-plus", "openrouter"),
|
||||
("qwen3.6-flash", "qwen/qwen3.6-flash", "openrouter"),
|
||||
("qwen3.5-122b", "qwen/qwen3.5-122b-a10b", "openrouter"),
|
||||
("deepseek-v4-pro", "deepseek/deepseek-v4-pro", "openrouter"),
|
||||
("deepseek-v4-flash", "deepseek/deepseek-v4-flash", "openrouter"),
|
||||
# Volcengine Coding Plan (火山引擎代码计划 — coding-only endpoint)
|
||||
# Listed before Zhipu so simple GLM lookups keep their existing default.
|
||||
("glm-5.2", "glm-5-2", "volcengine-code"),
|
||||
("kimi-k2.5", "kimi-k2-5", "volcengine-code"),
|
||||
# Zhipu CodePlan (智谱代码计划 — coding-only endpoint)
|
||||
("glm-5.2", "glm-5.2", "zhipu-code"),
|
||||
("glm-5.1", "glm-5.1", "zhipu-code"),
|
||||
("glm-5", "glm-5", "zhipu-code"),
|
||||
("glm-5-turbo", "glm-5-turbo", "zhipu-code"),
|
||||
("glm-5v-turbo", "glm-5v-turbo", "zhipu-code"),
|
||||
("glm-4.7", "glm-4.7", "zhipu-code"),
|
||||
# Zhipu (智谱 — general endpoint, default for simple lookups)
|
||||
("glm-5.2", "glm-5.2", "zhipu"),
|
||||
("glm-5.1", "glm-5.1", "zhipu"),
|
||||
("glm-5", "glm-5", "zhipu"),
|
||||
("glm-5-turbo", "glm-5-turbo", "zhipu"),
|
||||
("glm-5v-turbo", "glm-5v-turbo", "zhipu"),
|
||||
("glm-4.7", "glm-4.7", "zhipu"),
|
||||
# Volcengine (火山引擎 — Doubao models)
|
||||
("doubao-seed-2.0-pro", "doubao-seed-2-0-pro-260215", "volcengine"),
|
||||
("doubao-seed-2.0-lite", "doubao-seed-2-0-lite-260215", "volcengine"),
|
||||
("doubao-seed-2.0-mini", "doubao-seed-2-0-mini-260215", "volcengine"),
|
||||
("doubao-seed-2.0-code", "doubao-seed-2-0-code-preview-260215", "volcengine"),
|
||||
("doubao-seed-1.6", "doubao-seed-1.6", "volcengine"),
|
||||
("doubao-1.5-pro", "doubao-1.5-pro-256k", "volcengine"),
|
||||
("doubao-1.5-thinking-pro", "doubao-1.5-thinking-pro", "volcengine"),
|
||||
# DashScope Coding Plan (阿里云代码计划 — subscription sk-sp-* endpoint)
|
||||
("qwen3.8-max", "qwen3.8-max", "dashscope-code"),
|
||||
("qwen3.7-max", "qwen3.7-max", "dashscope-code"),
|
||||
("qwen3.7-plus", "qwen3.7-plus", "dashscope-code"),
|
||||
("qwen3.6-max", "qwen3.6-max-preview", "dashscope-code"),
|
||||
("qwen3.6-plus", "qwen3.6-plus", "dashscope-code"),
|
||||
("qwen3.6-flash", "qwen3.6-flash", "dashscope-code"),
|
||||
("qwen3-coder", "qwen3-coder-plus", "dashscope-code"),
|
||||
("qwen3-coder-next", "qwen3-coder-next", "dashscope-code"),
|
||||
("qwen3-max", "qwen3-max", "dashscope-code"),
|
||||
("qwen3.5-plus", "qwen3.5-plus", "dashscope-code"),
|
||||
# DashScope (阿里云 — Qwen models, default for simple lookups)
|
||||
("qwen3.8-max", "qwen3.8-max", "dashscope"),
|
||||
("qwen3.7-max", "qwen3.7-max", "dashscope"),
|
||||
("qwen3.7-plus", "qwen3.7-plus", "dashscope"),
|
||||
("qwen3.6-max", "qwen3.6-max-preview", "dashscope"),
|
||||
("qwen3.6-plus", "qwen3.6-plus", "dashscope"),
|
||||
("qwen3.6-flash", "qwen3.6-flash", "dashscope"),
|
||||
("qwen3-coder", "qwen3-coder-plus", "dashscope"),
|
||||
("qwen3-235b", "qwen3-235b-a22b", "dashscope"),
|
||||
("qwen-max", "qwen-max", "dashscope"),
|
||||
("qwq-plus", "qwq-plus", "dashscope"),
|
||||
# DeepSeek
|
||||
("deepseek-v4-pro", "deepseek-v4-pro", "deepseek"),
|
||||
("deepseek-v4-flash", "deepseek-v4-flash", "deepseek"),
|
||||
# Legacy aliases (deprecated 2026-07-24; route to v4-flash thinking/non-thinking)
|
||||
("deepseek-r1", "deepseek-reasoner", "deepseek"),
|
||||
("deepseek-v3", "deepseek-chat", "deepseek"),
|
||||
# Moonshot (OpenAI-compatible)
|
||||
("kimi-k3", "kimi-k3", "moonshot"),
|
||||
("kimi-k2.6", "kimi-k2.6", "moonshot"),
|
||||
("kimi-k2.5", "kimi-k2.5", "moonshot"),
|
||||
("kimi-k2-thinking", "kimi-k2-thinking", "moonshot"),
|
||||
("kimi-k2-thinking-turbo", "kimi-k2-thinking-turbo", "moonshot"),
|
||||
("moonshot-v1-auto", "moonshot-v1-auto", "moonshot"),
|
||||
("moonshot-v1-128k", "moonshot-v1-128k", "moonshot"),
|
||||
("moonshot-v1-32k", "moonshot-v1-32k", "moonshot"),
|
||||
("moonshot-v1-8k", "moonshot-v1-8k", "moonshot"),
|
||||
# Kimi Coding Plan (Anthropic-compatible)
|
||||
("kimi-for-coding", "kimi-for-coding", "kimi-coding"),
|
||||
]
|
||||
|
||||
# Public dict for simple lookups (last entry wins for duplicate names).
|
||||
# Use get_models_for_provider() for provider-aware lookups.
|
||||
MODELS: dict[str, tuple[str, str]] = {
|
||||
name: (model_id, provider) for name, model_id, provider in _MODEL_ENTRIES
|
||||
}
|
||||
|
||||
DEFAULT_MODEL = "claude-sonnet-4-6"
|
||||
|
||||
|
||||
def get_models_for_provider(provider: str) -> list[tuple[str, str]]:
|
||||
"""Get all models for a specific provider.
|
||||
|
||||
Args:
|
||||
provider: Provider name (e.g., 'anthropic', 'openrouter').
|
||||
|
||||
Returns:
|
||||
List of (short_name, model_id) tuples for the provider.
|
||||
"""
|
||||
return [(name, model_id) for name, model_id, p in _MODEL_ENTRIES if p == provider]
|
||||
|
||||
|
||||
def list_models() -> list[str]:
|
||||
"""List all available model short names.
|
||||
|
||||
Returns:
|
||||
List of unique model short names that can be passed to get_chat_model().
|
||||
"""
|
||||
seen = set()
|
||||
result = []
|
||||
for name, _, _ in _MODEL_ENTRIES:
|
||||
if name not in seen:
|
||||
seen.add(name)
|
||||
result.append(name)
|
||||
return result
|
||||
|
||||
|
||||
def list_models_by_provider() -> list[tuple[str, str, str]]:
|
||||
"""List all unique (short_name, model_id, provider) entries.
|
||||
|
||||
Returns:
|
||||
De-duplicated list of model entries preserving registry order.
|
||||
"""
|
||||
seen: set[tuple[str, str]] = set()
|
||||
result: list[tuple[str, str, str]] = []
|
||||
for name, model_id, provider in _MODEL_ENTRIES:
|
||||
key = (name, provider)
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
result.append((name, model_id, provider))
|
||||
return result
|
||||
|
||||
|
||||
async def list_model_picker_entries(
|
||||
ollama_base_url: str | None,
|
||||
*,
|
||||
include_custom_ollama: bool,
|
||||
) -> list[tuple[str, str, str]]:
|
||||
"""Return model picker entries, optionally including local Ollama models."""
|
||||
entries = list_models_by_provider()
|
||||
if ollama_base_url:
|
||||
from .ollama_discovery import discover_ollama_models
|
||||
|
||||
for detected_name in await discover_ollama_models(
|
||||
ollama_base_url,
|
||||
timeout=1.5,
|
||||
):
|
||||
entries.append((detected_name, detected_name, "ollama"))
|
||||
if include_custom_ollama:
|
||||
entries.append(("Custom Ollama model...", "__custom_ollama__", "ollama"))
|
||||
return entries
|
||||
|
||||
|
||||
def get_model_info(model: str) -> tuple[str, str] | None:
|
||||
"""Get the (model_id, provider) tuple for a short name.
|
||||
|
||||
Args:
|
||||
model: Short model name.
|
||||
|
||||
Returns:
|
||||
Tuple of (model_id, provider) or None if not found.
|
||||
"""
|
||||
return MODELS.get(model)
|
||||
@@ -544,6 +544,50 @@ async def _table_exists(conn: aiosqlite.Connection, table: str) -> bool:
|
||||
return await cur.fetchone() is not None
|
||||
|
||||
|
||||
async def _ensure_thread_meta_index(conn: aiosqlite.Connection, db_path: str) -> None:
|
||||
"""Create the expression index behind thread listing, if missing.
|
||||
|
||||
``list_threads`` filters + groups on ``json_extract(metadata, ...)``;
|
||||
without an index SQLite scans every checkpoint row, dragging each row's
|
||||
multi-KB checkpoint blob through the page cache (~1.3s on a 700MB DB vs
|
||||
~0.1s indexed). The four columns are exactly the per-row expressions:
|
||||
agent_name + graph_id feed the WHERE and updated_at feeds the MAX, so
|
||||
none of them may fall back to a per-row metadata fetch; workspace_dir /
|
||||
model are only materialized for the surviving groups (~dozens), so they
|
||||
stay out of the index (measured: same speed as a 6-column variant at
|
||||
60% of its size, and two fewer json_extract per checkpoint write).
|
||||
|
||||
The existence probe is a cheap catalog read on the caller's connection
|
||||
(it can still wait on an exclusive schema lock, bounded by that
|
||||
connection's timeout); the one-time build runs on a separate connection
|
||||
whose 2s timeout bounds its lock wait, with build time after acquiring
|
||||
(~0.2s measured on a 700MB DB) on top. Best-effort: on any failure the
|
||||
listing simply runs unindexed and the next call retries.
|
||||
"""
|
||||
try:
|
||||
async with conn.execute(
|
||||
"SELECT 1 FROM sqlite_master WHERE type='index' AND name=?",
|
||||
("idx_evoscientist_thread_meta",),
|
||||
) as cur:
|
||||
if await cur.fetchone() is not None:
|
||||
return
|
||||
async with aiosqlite.connect(db_path, timeout=2.0) as ddl_conn:
|
||||
await ddl_conn.execute(
|
||||
"""
|
||||
CREATE INDEX IF NOT EXISTS idx_evoscientist_thread_meta
|
||||
ON checkpoints (
|
||||
json_extract(metadata, '$.agent_name'),
|
||||
thread_id,
|
||||
json_extract(metadata, '$.updated_at'),
|
||||
json_extract(metadata, '$.graph_id')
|
||||
)
|
||||
"""
|
||||
)
|
||||
await ddl_conn.commit()
|
||||
except aiosqlite.Error:
|
||||
_logger.debug("Could not ensure thread meta index", exc_info=True)
|
||||
|
||||
|
||||
def _reduce_messages_delta(
|
||||
state: list[AnyMessage] | None, writes: list[Any]
|
||||
) -> list[AnyMessage]:
|
||||
@@ -889,6 +933,7 @@ async def list_threads(
|
||||
async with aiosqlite.connect(db_path, timeout=30.0) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return []
|
||||
await _ensure_thread_meta_index(conn, db_path)
|
||||
|
||||
query = f"""
|
||||
SELECT thread_id,
|
||||
|
||||
Reference in New Issue
Block a user