Files
EvoScientist-Multi/EvoScientist/langgraph_dev/http.py
T
m4 5a581c78a2
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
feat: add scoped model runtime configuration
Introduce provider, model, and invocation contracts with encrypted configuration persistence. Add web runtime fencing, route fallback, recovery middleware, workspace scoping, and comprehensive tests.
2026-08-14 22:03:04 +08:00

429 lines
16 KiB
Python

"""Custom HTTP routes mounted alongside the langgraph dev server.
The langgraph-api host supports a top-level ``http`` key in
``langgraph.json`` that names an ASGI app to mount on the same
process as the graph. We use it to surface the registry the WebUI's
``/model`` picker needs.
Why this lives here and not as a separate sidecar: the WebUI talks to
``EvoSci deploy``'s langgraph endpoint anyway, so one origin keeps the
WebUI's fetch logic simple — no CORS dance, no extra port to configure.
Why Starlette and not FastAPI: ``langgraph_api`` already depends on
Starlette; adding FastAPI would pull in pydantic v1-vs-v2 reconciliation
the deploy doesn't need. The one route here has no input model, just a
JSON body, so the lower-level surface is sufficient.
Lightweight by design — module-level imports stick to ``config``,
``llm.models`` (registry only; no chat-model construction), and
Starlette itself. Nothing on this surface should pull the agent into
memory.
"""
from __future__ import annotations
import asyncio
import hashlib
import json
import os
import secrets
from pathlib import Path, PurePosixPath
from typing import Any
from uuid import UUID
from starlette.applications import Starlette
from starlette.requests import Request
from starlette.responses import JSONResponse
from starlette.routing import Route
from EvoScientist.config import get_effective_config
from EvoScientist.llm.models import list_model_picker_entries
_recoverable_run_lock = asyncio.Lock()
def _load_scope_service_token() -> str:
configured = (
os.getenv("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", "").strip()
or os.getenv("AI4SCI_EVO_RUNTIME_GRANT_SECRET", "").strip()
)
if configured:
return configured
from EvoScientist.scope_registry import get_scope_service_token
return get_scope_service_token()
_SCOPE_SERVICE_TOKEN = _load_scope_service_token()
async def get_models(_request: Request) -> JSONResponse:
"""Return the model registry as ``{entries, default}``.
``entries`` preserves the registry order so the WebUI picker can
rank providers per short name the same way the backend would.
Mirrors the TUI ``/model`` picker by appending locally-pulled
Ollama models when ``ollama_base_url`` is configured — same
``discover_ollama_models()`` call, same 1.5-s timeout, same
fail-soft semantics (the probe returns ``[]`` on any error, never
raises). The TUI's "Custom Ollama model…" sentinel is intentionally
omitted — that's a widget-specific input affordance, not part of
the registry surface.
``default`` reflects the deployment's currently-configured fallback
(``config.yaml``'s ``model`` / ``provider`` — what ``/model reset``
would land on). Returned even when the configured pair isn't in
the registry, so the picker can still label it.
Uses ``get_effective_config()`` (not ``load_config()``) so env-var
overrides like ``OLLAMA_BASE_URL`` from ``_ENV_MAPPINGS`` are
honored — matching the deploy's actual model-building behavior.
Offloaded to a thread because ``get_effective_config()`` calls
``find_dotenv(usecwd=True)`` which invokes ``os.getcwd()`` — a
blocking syscall that langgraph-dev's ``blockbuster`` middleware
refuses to allow on the async event loop (would surface as a 500).
"""
cfg = await asyncio.to_thread(get_effective_config)
entries = [
{"name": name, "model_id": model_id, "provider": provider}
for name, model_id, provider in await list_model_picker_entries(
getattr(cfg, "ollama_base_url", None),
include_custom_ollama=False,
)
]
return JSONResponse(
{
"entries": entries,
"default": {"name": cfg.model, "provider": cfg.provider},
}
)
async def recoverable_run_capabilities(_request: Request) -> JSONResponse:
"""Capabilities required by Ai4Sci's durable dispatch outbox."""
return JSONResponse(
{
"version": 1,
"deterministic_run_id": True,
"stream_resumable": True,
"durability_sync": True,
"multitask_enqueue": True,
"interrupt_resume": True,
"pending_interrupt_state": True,
"workspace_scope_v1": os.getenv("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full",
}
)
def _scope_service_authorized(request: Request) -> JSONResponse | None:
if not _SCOPE_SERVICE_TOKEN:
return JSONResponse({"code": "WORKSPACE_SERVICE_UNAVAILABLE"}, status_code=503)
header = request.headers.get("authorization", "")
if not header.startswith("Bearer ") or not secrets.compare_digest(
header[7:], _SCOPE_SERVICE_TOKEN
):
return JSONResponse({"code": "UNAUTHORIZED"}, status_code=401)
return None
def _scope_payload(record: Any) -> dict[str, Any]:
return {
"deployment_id": record.deployment_id,
"scope_id": record.scope_id,
"primary_thread_id": record.primary_thread_id,
"primary_owner_id": record.primary_owner_id,
"state": record.state,
"revision": record.revision,
}
def _run_payload(run: Any) -> dict[str, Any]:
return {
"run_request_id": run.run_request_id,
"turn_id": run.turn_id,
"interrupt_key": run.interrupt_key,
"request_hash": run.request_hash,
"run_owner_id": run.run_owner_id,
"run_id": run.run_id,
"state": run.state,
}
def _registry_call(method: str, *args: Any, **kwargs: Any) -> Any:
from EvoScientist.scope_registry import get_scope_registry
from EvoScientist.workspace_scope import current_deployment_id
return getattr(get_scope_registry(), method)(current_deployment_id(), *args, **kwargs)
def _provision_scope(thread_id: str) -> Any:
from EvoScientist.workspace_scope import (
current_deployment_id,
provision_conversation_scope,
)
return provision_conversation_scope(thread_id, deployment_id=current_deployment_id())
async def provision_workspace_scope(request: Request) -> JSONResponse:
if denied := _scope_service_authorized(request):
return denied
try:
payload = await request.json()
except json.JSONDecodeError:
payload = None
if not isinstance(payload, dict) or not isinstance(payload.get("thread_id"), str):
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
try:
record = await asyncio.to_thread(_provision_scope, payload["thread_id"])
except Exception as exc:
return JSONResponse({"code": "WORKSPACE_SCOPE_CONFLICT", "message": str(exc)}, status_code=409)
return JSONResponse(_scope_payload(record), status_code=201)
async def get_workspace_scope(request: Request) -> JSONResponse:
if denied := _scope_service_authorized(request):
return denied
try:
record = await asyncio.to_thread(
_registry_call, "get_by_thread", str(request.path_params["thread_id"])
)
except Exception as exc:
return JSONResponse({"code": "WORKSPACE_SCOPE_NOT_FOUND", "message": str(exc)}, status_code=404)
return JSONResponse(_scope_payload(record))
async def reserve_workspace_run(request: Request) -> JSONResponse:
if denied := _scope_service_authorized(request):
return denied
try:
payload = await request.json()
except json.JSONDecodeError:
payload = None
if not isinstance(payload, dict):
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
try:
run = await asyncio.to_thread(
_registry_call,
"reserve_run",
str(request.path_params["scope_id"]),
str(payload["run_request_id"]),
str(payload["turn_id"]),
str(payload["request_hash"]),
interrupt_key=(
str(payload["interrupt_key"]) if payload.get("interrupt_key") else None
),
)
except Exception as exc:
code = (
"INTERRUPT_ALREADY_RESOLVED"
if type(exc).__name__ == "ScopeInterruptResolvedError"
else "WORKSPACE_RUN_CONFLICT"
)
return JSONResponse({"code": code, "message": str(exc)}, status_code=409)
return JSONResponse(_run_payload(run), status_code=201)
async def bind_workspace_run(request: Request) -> JSONResponse:
if denied := _scope_service_authorized(request):
return denied
try:
payload = await request.json()
except json.JSONDecodeError:
payload = None
if not isinstance(payload, dict) or not isinstance(payload.get("run_id"), str):
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
try:
run = await asyncio.to_thread(
_registry_call,
"bind_run",
str(request.path_params["scope_id"]),
str(request.path_params["run_request_id"]),
payload["run_id"],
)
except Exception as exc:
return JSONResponse({"code": "WORKSPACE_RUN_CONFLICT", "message": str(exc)}, status_code=409)
return JSONResponse(_run_payload(run))
def _materialize_target(scope_id: str, raw_path: str) -> Path:
from EvoScientist.workspace_scope import conversation_files_dir
path = PurePosixPath(raw_path.replace("\\", "/"))
if path.is_absolute() or not path.parts or path.parts[0] != "uploads":
raise ValueError("only uploads/ paths are accepted")
if any(part in {"", ".", ".."} for part in path.parts):
raise ValueError("invalid upload path")
root = conversation_files_dir(scope_id).resolve()
target = root.joinpath(*path.parts)
target.parent.mkdir(parents=True, exist_ok=True)
try:
target.parent.resolve().relative_to(root)
except ValueError as exc:
raise ValueError("upload path escapes workspace scope") from exc
current = root
for part in path.parts[:-1]:
current = current / part
if current.is_symlink():
raise ValueError("symlink parents are rejected")
if target.is_symlink():
raise ValueError("symlink targets are rejected")
return target
async def materialize_workspace_file(request: Request) -> JSONResponse:
if denied := _scope_service_authorized(request):
return denied
scope_id = str(request.path_params["scope_id"])
try:
await asyncio.to_thread(_registry_call, "get", scope_id)
target = await asyncio.to_thread(
_materialize_target, scope_id, str(request.path_params["path"])
)
except Exception as exc:
return JSONResponse({"code": "WORKSPACE_PATH_INVALID", "message": str(exc)}, status_code=400)
expected_hash = request.headers.get("x-content-sha256", "").lower()
expected_size = int(request.headers.get("content-length") or 0)
if expected_size > 100 * 1024 * 1024:
return JSONResponse({"code": "WORKSPACE_FILE_TOO_LARGE"}, status_code=413)
temporary = target.with_name(f".{target.name}.{secrets.token_hex(8)}.tmp")
digest = hashlib.sha256()
size = 0
try:
with temporary.open("xb") as handle:
async for chunk in request.stream():
size += len(chunk)
if size > 100 * 1024 * 1024:
raise ValueError("workspace file exceeds 100 MiB")
digest.update(chunk)
handle.write(chunk)
handle.flush()
os.fsync(handle.fileno())
actual_hash = digest.hexdigest()
if expected_hash and not secrets.compare_digest(actual_hash, expected_hash):
raise ValueError("workspace file hash mismatch")
os.replace(temporary, target)
except Exception as exc:
temporary.unlink(missing_ok=True)
return JSONResponse({"code": "WORKSPACE_FILE_INVALID", "message": str(exc)}, status_code=409)
return JSONResponse({"virtual_path": str(request.path_params["path"]), "size": size, "sha256": actual_hash})
async def create_recoverable_run(request: Request) -> JSONResponse:
"""Create a LangGraph Run with a caller-owned deterministic UUID.
LangGraph's public create endpoint always generates its own UUID. This
adapter performs lookup and insertion while holding the process-wide run
creation lock and passes the durable request UUID to ``create_valid_run``.
Retrying after a lost HTTP response therefore cannot create another Run.
"""
if request.headers.get("x-auth-scheme") != "langsmith":
return JSONResponse({"code": "UNAUTHORIZED"}, status_code=401)
value = await request.json()
if not isinstance(value, dict):
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
try:
thread_id = str(UUID(str(value["thread_id"])))
run_id = UUID(str(value["run_id"]))
run_request_id = str(UUID(str(value["run_request_id"])))
request_hash = str(value["request_hash"])
assistant_id = str(value["assistant_id"])
operation = str(value.get("operation") or "start")
except (KeyError, TypeError, ValueError):
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
if (
str(run_id) != run_request_id
or len(request_hash) != 64
or operation not in {"start", "resume"}
):
return JSONResponse({"code": "INVALID_IDEMPOTENCY_KEY"}, status_code=400)
command = value.get("command")
if operation == "resume":
if (
value.get("input") is not None
or not isinstance(command, dict)
or set(command) != {"resume"}
):
return JSONResponse({"code": "INVALID_RESUME_REQUEST"}, status_code=400)
elif command is not None:
return JSONResponse({"code": "INVALID_START_REQUEST"}, status_code=400)
from langgraph_api.models.run import Runs, create_valid_run
from langgraph_api.utils import fetchone
from langgraph_runtime.database import connect
payload = {
"assistant_id": assistant_id,
"input": value.get("input"),
"command": command,
"metadata": value.get("metadata") or {},
"config": value.get("config") or {},
"stream_mode": value.get("stream_mode") or ["messages", "updates", "tasks", "custom"],
"stream_resumable": True,
"durability": "sync",
"multitask_strategy": "enqueue",
"if_not_exists": "create",
}
payload["metadata"] = {
**payload["metadata"],
"run_request_id": run_request_id,
"request_hash": request_hash,
}
async with _recoverable_run_lock:
async with connect() as conn:
existing_iter = await Runs.get(conn, run_id, thread_id=UUID(thread_id))
try:
existing = await fetchone(existing_iter)
except Exception as exc:
if getattr(exc, "status_code", None) != 404:
raise
existing = None
if existing is not None:
metadata = existing.get("metadata") or {}
if metadata.get("request_hash") != request_hash:
return JSONResponse(
{
"code": "RUN_REQUEST_CONFLICT",
"message": "run_request_id is bound to another request hash",
},
status_code=409,
)
return JSONResponse(
{"run_id": str(existing["run_id"]), "status": existing["status"], "created": False}
)
created = await create_valid_run(
conn,
thread_id,
payload,
dict(request.headers),
run_id=run_id,
)
return JSONResponse(
{"run_id": str(created["run_id"]), "status": created["status"], "created": True},
status_code=201,
)
app = Starlette(
routes=[
Route("/api/models", get_models, methods=["GET"]),
Route(
"/api/ai4sci/recoverable-runs/capabilities",
recoverable_run_capabilities,
methods=["GET"],
),
Route(
"/api/ai4sci/recoverable-runs/create",
create_recoverable_run,
methods=["POST"],
),
Route("/internal/workspace-scopes/provision", provision_workspace_scope, methods=["POST"]),
Route("/internal/workspace-scopes/by-thread/{thread_id}", get_workspace_scope, methods=["GET"]),
Route("/internal/workspace-scopes/{scope_id}/runs/reserve", reserve_workspace_run, methods=["POST"]),
Route("/internal/workspace-scopes/{scope_id}/runs/{run_request_id}", bind_workspace_run, methods=["PATCH"]),
Route("/internal/workspace-scopes/{scope_id}/files/{path:path}", materialize_workspace_file, methods=["PUT"]),
]
)