5a581c78a2
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Introduce provider, model, and invocation contracts with encrypted configuration persistence. Add web runtime fencing, route fallback, recovery middleware, workspace scoping, and comprehensive tests.
429 lines
16 KiB
Python
429 lines
16 KiB
Python
"""Custom HTTP routes mounted alongside the langgraph dev server.
|
|
|
|
The langgraph-api host supports a top-level ``http`` key in
|
|
``langgraph.json`` that names an ASGI app to mount on the same
|
|
process as the graph. We use it to surface the registry the WebUI's
|
|
``/model`` picker needs.
|
|
|
|
Why this lives here and not as a separate sidecar: the WebUI talks to
|
|
``EvoSci deploy``'s langgraph endpoint anyway, so one origin keeps the
|
|
WebUI's fetch logic simple — no CORS dance, no extra port to configure.
|
|
|
|
Why Starlette and not FastAPI: ``langgraph_api`` already depends on
|
|
Starlette; adding FastAPI would pull in pydantic v1-vs-v2 reconciliation
|
|
the deploy doesn't need. The one route here has no input model, just a
|
|
JSON body, so the lower-level surface is sufficient.
|
|
|
|
Lightweight by design — module-level imports stick to ``config``,
|
|
``llm.models`` (registry only; no chat-model construction), and
|
|
Starlette itself. Nothing on this surface should pull the agent into
|
|
memory.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import secrets
|
|
from pathlib import Path, PurePosixPath
|
|
from typing import Any
|
|
from uuid import UUID
|
|
|
|
from starlette.applications import Starlette
|
|
from starlette.requests import Request
|
|
from starlette.responses import JSONResponse
|
|
from starlette.routing import Route
|
|
|
|
from EvoScientist.config import get_effective_config
|
|
from EvoScientist.llm.models import list_model_picker_entries
|
|
|
|
_recoverable_run_lock = asyncio.Lock()
|
|
|
|
|
|
def _load_scope_service_token() -> str:
|
|
configured = (
|
|
os.getenv("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", "").strip()
|
|
or os.getenv("AI4SCI_EVO_RUNTIME_GRANT_SECRET", "").strip()
|
|
)
|
|
if configured:
|
|
return configured
|
|
from EvoScientist.scope_registry import get_scope_service_token
|
|
|
|
return get_scope_service_token()
|
|
|
|
|
|
_SCOPE_SERVICE_TOKEN = _load_scope_service_token()
|
|
|
|
|
|
async def get_models(_request: Request) -> JSONResponse:
|
|
"""Return the model registry as ``{entries, default}``.
|
|
|
|
``entries`` preserves the registry order so the WebUI picker can
|
|
rank providers per short name the same way the backend would.
|
|
Mirrors the TUI ``/model`` picker by appending locally-pulled
|
|
Ollama models when ``ollama_base_url`` is configured — same
|
|
``discover_ollama_models()`` call, same 1.5-s timeout, same
|
|
fail-soft semantics (the probe returns ``[]`` on any error, never
|
|
raises). The TUI's "Custom Ollama model…" sentinel is intentionally
|
|
omitted — that's a widget-specific input affordance, not part of
|
|
the registry surface.
|
|
|
|
``default`` reflects the deployment's currently-configured fallback
|
|
(``config.yaml``'s ``model`` / ``provider`` — what ``/model reset``
|
|
would land on). Returned even when the configured pair isn't in
|
|
the registry, so the picker can still label it.
|
|
|
|
Uses ``get_effective_config()`` (not ``load_config()``) so env-var
|
|
overrides like ``OLLAMA_BASE_URL`` from ``_ENV_MAPPINGS`` are
|
|
honored — matching the deploy's actual model-building behavior.
|
|
Offloaded to a thread because ``get_effective_config()`` calls
|
|
``find_dotenv(usecwd=True)`` which invokes ``os.getcwd()`` — a
|
|
blocking syscall that langgraph-dev's ``blockbuster`` middleware
|
|
refuses to allow on the async event loop (would surface as a 500).
|
|
"""
|
|
cfg = await asyncio.to_thread(get_effective_config)
|
|
entries = [
|
|
{"name": name, "model_id": model_id, "provider": provider}
|
|
for name, model_id, provider in await list_model_picker_entries(
|
|
getattr(cfg, "ollama_base_url", None),
|
|
include_custom_ollama=False,
|
|
)
|
|
]
|
|
return JSONResponse(
|
|
{
|
|
"entries": entries,
|
|
"default": {"name": cfg.model, "provider": cfg.provider},
|
|
}
|
|
)
|
|
|
|
|
|
async def recoverable_run_capabilities(_request: Request) -> JSONResponse:
|
|
"""Capabilities required by Ai4Sci's durable dispatch outbox."""
|
|
|
|
return JSONResponse(
|
|
{
|
|
"version": 1,
|
|
"deterministic_run_id": True,
|
|
"stream_resumable": True,
|
|
"durability_sync": True,
|
|
"multitask_enqueue": True,
|
|
"interrupt_resume": True,
|
|
"pending_interrupt_state": True,
|
|
"workspace_scope_v1": os.getenv("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full",
|
|
}
|
|
)
|
|
|
|
|
|
def _scope_service_authorized(request: Request) -> JSONResponse | None:
|
|
if not _SCOPE_SERVICE_TOKEN:
|
|
return JSONResponse({"code": "WORKSPACE_SERVICE_UNAVAILABLE"}, status_code=503)
|
|
header = request.headers.get("authorization", "")
|
|
if not header.startswith("Bearer ") or not secrets.compare_digest(
|
|
header[7:], _SCOPE_SERVICE_TOKEN
|
|
):
|
|
return JSONResponse({"code": "UNAUTHORIZED"}, status_code=401)
|
|
return None
|
|
|
|
|
|
def _scope_payload(record: Any) -> dict[str, Any]:
|
|
return {
|
|
"deployment_id": record.deployment_id,
|
|
"scope_id": record.scope_id,
|
|
"primary_thread_id": record.primary_thread_id,
|
|
"primary_owner_id": record.primary_owner_id,
|
|
"state": record.state,
|
|
"revision": record.revision,
|
|
}
|
|
|
|
|
|
def _run_payload(run: Any) -> dict[str, Any]:
|
|
return {
|
|
"run_request_id": run.run_request_id,
|
|
"turn_id": run.turn_id,
|
|
"interrupt_key": run.interrupt_key,
|
|
"request_hash": run.request_hash,
|
|
"run_owner_id": run.run_owner_id,
|
|
"run_id": run.run_id,
|
|
"state": run.state,
|
|
}
|
|
|
|
|
|
def _registry_call(method: str, *args: Any, **kwargs: Any) -> Any:
|
|
from EvoScientist.scope_registry import get_scope_registry
|
|
from EvoScientist.workspace_scope import current_deployment_id
|
|
|
|
return getattr(get_scope_registry(), method)(current_deployment_id(), *args, **kwargs)
|
|
|
|
|
|
def _provision_scope(thread_id: str) -> Any:
|
|
from EvoScientist.workspace_scope import (
|
|
current_deployment_id,
|
|
provision_conversation_scope,
|
|
)
|
|
|
|
return provision_conversation_scope(thread_id, deployment_id=current_deployment_id())
|
|
|
|
|
|
async def provision_workspace_scope(request: Request) -> JSONResponse:
|
|
if denied := _scope_service_authorized(request):
|
|
return denied
|
|
try:
|
|
payload = await request.json()
|
|
except json.JSONDecodeError:
|
|
payload = None
|
|
if not isinstance(payload, dict) or not isinstance(payload.get("thread_id"), str):
|
|
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
|
|
try:
|
|
record = await asyncio.to_thread(_provision_scope, payload["thread_id"])
|
|
except Exception as exc:
|
|
return JSONResponse({"code": "WORKSPACE_SCOPE_CONFLICT", "message": str(exc)}, status_code=409)
|
|
return JSONResponse(_scope_payload(record), status_code=201)
|
|
|
|
|
|
async def get_workspace_scope(request: Request) -> JSONResponse:
|
|
if denied := _scope_service_authorized(request):
|
|
return denied
|
|
try:
|
|
record = await asyncio.to_thread(
|
|
_registry_call, "get_by_thread", str(request.path_params["thread_id"])
|
|
)
|
|
except Exception as exc:
|
|
return JSONResponse({"code": "WORKSPACE_SCOPE_NOT_FOUND", "message": str(exc)}, status_code=404)
|
|
return JSONResponse(_scope_payload(record))
|
|
|
|
|
|
async def reserve_workspace_run(request: Request) -> JSONResponse:
|
|
if denied := _scope_service_authorized(request):
|
|
return denied
|
|
try:
|
|
payload = await request.json()
|
|
except json.JSONDecodeError:
|
|
payload = None
|
|
if not isinstance(payload, dict):
|
|
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
|
|
try:
|
|
run = await asyncio.to_thread(
|
|
_registry_call,
|
|
"reserve_run",
|
|
str(request.path_params["scope_id"]),
|
|
str(payload["run_request_id"]),
|
|
str(payload["turn_id"]),
|
|
str(payload["request_hash"]),
|
|
interrupt_key=(
|
|
str(payload["interrupt_key"]) if payload.get("interrupt_key") else None
|
|
),
|
|
)
|
|
except Exception as exc:
|
|
code = (
|
|
"INTERRUPT_ALREADY_RESOLVED"
|
|
if type(exc).__name__ == "ScopeInterruptResolvedError"
|
|
else "WORKSPACE_RUN_CONFLICT"
|
|
)
|
|
return JSONResponse({"code": code, "message": str(exc)}, status_code=409)
|
|
return JSONResponse(_run_payload(run), status_code=201)
|
|
|
|
|
|
async def bind_workspace_run(request: Request) -> JSONResponse:
|
|
if denied := _scope_service_authorized(request):
|
|
return denied
|
|
try:
|
|
payload = await request.json()
|
|
except json.JSONDecodeError:
|
|
payload = None
|
|
if not isinstance(payload, dict) or not isinstance(payload.get("run_id"), str):
|
|
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
|
|
try:
|
|
run = await asyncio.to_thread(
|
|
_registry_call,
|
|
"bind_run",
|
|
str(request.path_params["scope_id"]),
|
|
str(request.path_params["run_request_id"]),
|
|
payload["run_id"],
|
|
)
|
|
except Exception as exc:
|
|
return JSONResponse({"code": "WORKSPACE_RUN_CONFLICT", "message": str(exc)}, status_code=409)
|
|
return JSONResponse(_run_payload(run))
|
|
|
|
|
|
def _materialize_target(scope_id: str, raw_path: str) -> Path:
|
|
from EvoScientist.workspace_scope import conversation_files_dir
|
|
|
|
path = PurePosixPath(raw_path.replace("\\", "/"))
|
|
if path.is_absolute() or not path.parts or path.parts[0] != "uploads":
|
|
raise ValueError("only uploads/ paths are accepted")
|
|
if any(part in {"", ".", ".."} for part in path.parts):
|
|
raise ValueError("invalid upload path")
|
|
root = conversation_files_dir(scope_id).resolve()
|
|
target = root.joinpath(*path.parts)
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
target.parent.resolve().relative_to(root)
|
|
except ValueError as exc:
|
|
raise ValueError("upload path escapes workspace scope") from exc
|
|
current = root
|
|
for part in path.parts[:-1]:
|
|
current = current / part
|
|
if current.is_symlink():
|
|
raise ValueError("symlink parents are rejected")
|
|
if target.is_symlink():
|
|
raise ValueError("symlink targets are rejected")
|
|
return target
|
|
|
|
|
|
async def materialize_workspace_file(request: Request) -> JSONResponse:
|
|
if denied := _scope_service_authorized(request):
|
|
return denied
|
|
scope_id = str(request.path_params["scope_id"])
|
|
try:
|
|
await asyncio.to_thread(_registry_call, "get", scope_id)
|
|
target = await asyncio.to_thread(
|
|
_materialize_target, scope_id, str(request.path_params["path"])
|
|
)
|
|
except Exception as exc:
|
|
return JSONResponse({"code": "WORKSPACE_PATH_INVALID", "message": str(exc)}, status_code=400)
|
|
expected_hash = request.headers.get("x-content-sha256", "").lower()
|
|
expected_size = int(request.headers.get("content-length") or 0)
|
|
if expected_size > 100 * 1024 * 1024:
|
|
return JSONResponse({"code": "WORKSPACE_FILE_TOO_LARGE"}, status_code=413)
|
|
temporary = target.with_name(f".{target.name}.{secrets.token_hex(8)}.tmp")
|
|
digest = hashlib.sha256()
|
|
size = 0
|
|
try:
|
|
with temporary.open("xb") as handle:
|
|
async for chunk in request.stream():
|
|
size += len(chunk)
|
|
if size > 100 * 1024 * 1024:
|
|
raise ValueError("workspace file exceeds 100 MiB")
|
|
digest.update(chunk)
|
|
handle.write(chunk)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
actual_hash = digest.hexdigest()
|
|
if expected_hash and not secrets.compare_digest(actual_hash, expected_hash):
|
|
raise ValueError("workspace file hash mismatch")
|
|
os.replace(temporary, target)
|
|
except Exception as exc:
|
|
temporary.unlink(missing_ok=True)
|
|
return JSONResponse({"code": "WORKSPACE_FILE_INVALID", "message": str(exc)}, status_code=409)
|
|
return JSONResponse({"virtual_path": str(request.path_params["path"]), "size": size, "sha256": actual_hash})
|
|
|
|
|
|
async def create_recoverable_run(request: Request) -> JSONResponse:
|
|
"""Create a LangGraph Run with a caller-owned deterministic UUID.
|
|
|
|
LangGraph's public create endpoint always generates its own UUID. This
|
|
adapter performs lookup and insertion while holding the process-wide run
|
|
creation lock and passes the durable request UUID to ``create_valid_run``.
|
|
Retrying after a lost HTTP response therefore cannot create another Run.
|
|
"""
|
|
|
|
if request.headers.get("x-auth-scheme") != "langsmith":
|
|
return JSONResponse({"code": "UNAUTHORIZED"}, status_code=401)
|
|
value = await request.json()
|
|
if not isinstance(value, dict):
|
|
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
|
|
try:
|
|
thread_id = str(UUID(str(value["thread_id"])))
|
|
run_id = UUID(str(value["run_id"]))
|
|
run_request_id = str(UUID(str(value["run_request_id"])))
|
|
request_hash = str(value["request_hash"])
|
|
assistant_id = str(value["assistant_id"])
|
|
operation = str(value.get("operation") or "start")
|
|
except (KeyError, TypeError, ValueError):
|
|
return JSONResponse({"code": "INVALID_REQUEST"}, status_code=400)
|
|
if (
|
|
str(run_id) != run_request_id
|
|
or len(request_hash) != 64
|
|
or operation not in {"start", "resume"}
|
|
):
|
|
return JSONResponse({"code": "INVALID_IDEMPOTENCY_KEY"}, status_code=400)
|
|
command = value.get("command")
|
|
if operation == "resume":
|
|
if (
|
|
value.get("input") is not None
|
|
or not isinstance(command, dict)
|
|
or set(command) != {"resume"}
|
|
):
|
|
return JSONResponse({"code": "INVALID_RESUME_REQUEST"}, status_code=400)
|
|
elif command is not None:
|
|
return JSONResponse({"code": "INVALID_START_REQUEST"}, status_code=400)
|
|
|
|
from langgraph_api.models.run import Runs, create_valid_run
|
|
from langgraph_api.utils import fetchone
|
|
from langgraph_runtime.database import connect
|
|
|
|
payload = {
|
|
"assistant_id": assistant_id,
|
|
"input": value.get("input"),
|
|
"command": command,
|
|
"metadata": value.get("metadata") or {},
|
|
"config": value.get("config") or {},
|
|
"stream_mode": value.get("stream_mode") or ["messages", "updates", "tasks", "custom"],
|
|
"stream_resumable": True,
|
|
"durability": "sync",
|
|
"multitask_strategy": "enqueue",
|
|
"if_not_exists": "create",
|
|
}
|
|
payload["metadata"] = {
|
|
**payload["metadata"],
|
|
"run_request_id": run_request_id,
|
|
"request_hash": request_hash,
|
|
}
|
|
async with _recoverable_run_lock:
|
|
async with connect() as conn:
|
|
existing_iter = await Runs.get(conn, run_id, thread_id=UUID(thread_id))
|
|
try:
|
|
existing = await fetchone(existing_iter)
|
|
except Exception as exc:
|
|
if getattr(exc, "status_code", None) != 404:
|
|
raise
|
|
existing = None
|
|
if existing is not None:
|
|
metadata = existing.get("metadata") or {}
|
|
if metadata.get("request_hash") != request_hash:
|
|
return JSONResponse(
|
|
{
|
|
"code": "RUN_REQUEST_CONFLICT",
|
|
"message": "run_request_id is bound to another request hash",
|
|
},
|
|
status_code=409,
|
|
)
|
|
return JSONResponse(
|
|
{"run_id": str(existing["run_id"]), "status": existing["status"], "created": False}
|
|
)
|
|
created = await create_valid_run(
|
|
conn,
|
|
thread_id,
|
|
payload,
|
|
dict(request.headers),
|
|
run_id=run_id,
|
|
)
|
|
return JSONResponse(
|
|
{"run_id": str(created["run_id"]), "status": created["status"], "created": True},
|
|
status_code=201,
|
|
)
|
|
|
|
|
|
app = Starlette(
|
|
routes=[
|
|
Route("/api/models", get_models, methods=["GET"]),
|
|
Route(
|
|
"/api/ai4sci/recoverable-runs/capabilities",
|
|
recoverable_run_capabilities,
|
|
methods=["GET"],
|
|
),
|
|
Route(
|
|
"/api/ai4sci/recoverable-runs/create",
|
|
create_recoverable_run,
|
|
methods=["POST"],
|
|
),
|
|
Route("/internal/workspace-scopes/provision", provision_workspace_scope, methods=["POST"]),
|
|
Route("/internal/workspace-scopes/by-thread/{thread_id}", get_workspace_scope, methods=["GET"]),
|
|
Route("/internal/workspace-scopes/{scope_id}/runs/reserve", reserve_workspace_run, methods=["POST"]),
|
|
Route("/internal/workspace-scopes/{scope_id}/runs/{run_request_id}", bind_workspace_run, methods=["PATCH"]),
|
|
Route("/internal/workspace-scopes/{scope_id}/files/{path:path}", materialize_workspace_file, methods=["PUT"]),
|
|
]
|
|
)
|