feat: expose model registry at GET /api/models (#308)
* feat: expose model registry at GET /api/models * fix: include Ollama models in /api/models endpoint * fix: honor env vars override in /api/models endpoint * fix: offload get_effective_config to thread to satisfy blockbuster --------- Co-authored-by: Xi Zhang <106144707+X-iZhang@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,85 @@
|
|||||||
|
"""Custom HTTP routes mounted alongside the langgraph dev server.
|
||||||
|
|
||||||
|
The langgraph-api host supports a top-level ``http`` key in
|
||||||
|
``langgraph.json`` that names an ASGI app to mount on the same
|
||||||
|
process as the graph. We use it to surface the registry the WebUI's
|
||||||
|
``/model`` picker needs.
|
||||||
|
|
||||||
|
Why this lives here and not as a separate sidecar: the WebUI talks to
|
||||||
|
``EvoSci deploy``'s langgraph endpoint anyway, so one origin keeps the
|
||||||
|
WebUI's fetch logic simple — no CORS dance, no extra port to configure.
|
||||||
|
|
||||||
|
Why Starlette and not FastAPI: ``langgraph_api`` already depends on
|
||||||
|
Starlette; adding FastAPI would pull in pydantic v1-vs-v2 reconciliation
|
||||||
|
the deploy doesn't need. The one route here has no input model, just a
|
||||||
|
JSON body, so the lower-level surface is sufficient.
|
||||||
|
|
||||||
|
Lightweight by design — module-level imports stick to ``config``,
|
||||||
|
``llm.models`` (registry only; no chat-model construction), and
|
||||||
|
Starlette itself. Nothing on this surface should pull the agent into
|
||||||
|
memory.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
from starlette.applications import Starlette
|
||||||
|
from starlette.requests import Request
|
||||||
|
from starlette.responses import JSONResponse
|
||||||
|
from starlette.routing import Route
|
||||||
|
|
||||||
|
from EvoScientist.config import get_effective_config
|
||||||
|
from EvoScientist.llm.models import list_models_by_provider
|
||||||
|
|
||||||
|
|
||||||
|
async def get_models(_request: Request) -> JSONResponse:
|
||||||
|
"""Return the model registry as ``{entries, default}``.
|
||||||
|
|
||||||
|
``entries`` preserves the registry order so the WebUI picker can
|
||||||
|
rank providers per short name the same way the backend would.
|
||||||
|
Mirrors the TUI ``/model`` picker by appending locally-pulled
|
||||||
|
Ollama models when ``ollama_base_url`` is configured — same
|
||||||
|
``discover_ollama_models()`` call, same 1.5-s timeout, same
|
||||||
|
fail-soft semantics (the probe returns ``[]`` on any error, never
|
||||||
|
raises). The TUI's "Custom Ollama model…" sentinel is intentionally
|
||||||
|
omitted: that's a widget-specific input affordance, not part of
|
||||||
|
the registry surface.
|
||||||
|
|
||||||
|
``default`` reflects the deployment's currently-configured fallback
|
||||||
|
(``config.yaml``'s ``model`` / ``provider`` — what ``/model reset``
|
||||||
|
would land on). Returned even when the configured pair isn't in
|
||||||
|
the registry, so the picker can still label it.
|
||||||
|
|
||||||
|
Uses ``get_effective_config()`` (not ``load_config()``) so env-var
|
||||||
|
overrides like ``OLLAMA_BASE_URL`` from ``_ENV_MAPPINGS`` are
|
||||||
|
honored — matching the deploy's actual model-building behavior.
|
||||||
|
Offloaded to a thread because ``get_effective_config()`` calls
|
||||||
|
``find_dotenv(usecwd=True)`` which invokes ``os.getcwd()`` — a
|
||||||
|
blocking syscall that langgraph-dev's ``blockbuster`` middleware
|
||||||
|
refuses to allow on the async event loop (would surface as a 500).
|
||||||
|
"""
|
||||||
|
cfg = await asyncio.to_thread(get_effective_config)
|
||||||
|
entries = [
|
||||||
|
{"name": name, "model_id": model_id, "provider": provider}
|
||||||
|
for name, model_id, provider in list_models_by_provider()
|
||||||
|
]
|
||||||
|
ollama_base_url = getattr(cfg, "ollama_base_url", None)
|
||||||
|
if ollama_base_url:
|
||||||
|
from EvoScientist.llm.ollama_discovery import discover_ollama_models
|
||||||
|
|
||||||
|
for name in await discover_ollama_models(ollama_base_url, timeout=1.5):
|
||||||
|
entries.append({"name": name, "model_id": name, "provider": "ollama"})
|
||||||
|
return JSONResponse(
|
||||||
|
{
|
||||||
|
"entries": entries,
|
||||||
|
"default": {"name": cfg.model, "provider": cfg.provider},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
app = Starlette(
|
||||||
|
routes=[
|
||||||
|
Route("/api/models", get_models, methods=["GET"]),
|
||||||
|
]
|
||||||
|
)
|
||||||
@@ -15,5 +15,8 @@
|
|||||||
},
|
},
|
||||||
"config": {
|
"config": {
|
||||||
"recursion_limit": 1000000
|
"recursion_limit": 1000000
|
||||||
|
},
|
||||||
|
"http": {
|
||||||
|
"app": "EvoScientist.langgraph_dev.http:app"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,163 @@
|
|||||||
|
"""Smoke test for the /api/models route mounted via langgraph.json's
|
||||||
|
``http`` field. We test the FastAPI app directly — no need to spin up
|
||||||
|
langgraph dev.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from starlette.testclient import TestClient
|
||||||
|
|
||||||
|
from EvoScientist.config import EvoScientistConfig
|
||||||
|
from EvoScientist.langgraph_dev.http import app
|
||||||
|
|
||||||
|
client = TestClient(app)
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_models_returns_entries_and_default():
|
||||||
|
mock_cfg = EvoScientistConfig(
|
||||||
|
model="claude-sonnet-4-6", provider="custom-anthropic"
|
||||||
|
)
|
||||||
|
with patch(
|
||||||
|
"EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg
|
||||||
|
):
|
||||||
|
resp = client.get("/api/models")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
body = resp.json()
|
||||||
|
assert "entries" in body
|
||||||
|
assert "default" in body
|
||||||
|
assert body["default"] == {
|
||||||
|
"name": "claude-sonnet-4-6",
|
||||||
|
"provider": "custom-anthropic",
|
||||||
|
}
|
||||||
|
assert isinstance(body["entries"], list)
|
||||||
|
assert len(body["entries"]) > 0
|
||||||
|
# Every entry has the three required keys
|
||||||
|
for entry in body["entries"]:
|
||||||
|
assert set(entry.keys()) == {"name", "model_id", "provider"}
|
||||||
|
assert isinstance(entry["name"], str)
|
||||||
|
assert entry["name"]
|
||||||
|
assert isinstance(entry["model_id"], str)
|
||||||
|
assert entry["model_id"]
|
||||||
|
assert isinstance(entry["provider"], str)
|
||||||
|
assert entry["provider"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_entries_preserve_registry_order():
|
||||||
|
"""The picker uses position-in-list to rank providers per short name —
|
||||||
|
the JSON must preserve the order returned by ``list_models_by_provider``.
|
||||||
|
|
||||||
|
Stubs ``get_effective_config`` to keep the assertion focused on
|
||||||
|
registry order rather than implicitly depending on the ambient
|
||||||
|
deploy config.
|
||||||
|
"""
|
||||||
|
from EvoScientist.llm.models import list_models_by_provider
|
||||||
|
|
||||||
|
expected = [
|
||||||
|
{"name": n, "model_id": m, "provider": p}
|
||||||
|
for n, m, p in list_models_by_provider()
|
||||||
|
]
|
||||||
|
mock_cfg = EvoScientistConfig()
|
||||||
|
with patch(
|
||||||
|
"EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg
|
||||||
|
):
|
||||||
|
resp = client.get("/api/models")
|
||||||
|
assert resp.json()["entries"] == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_default_passes_through_arbitrary_config_pair():
|
||||||
|
"""If config.yaml names a (name, provider) pair that isn't in the
|
||||||
|
registry (typo, retired model), still report it as default — the
|
||||||
|
picker labels it as the active selection regardless.
|
||||||
|
"""
|
||||||
|
mock_cfg = EvoScientistConfig(model="some-retired-name", provider="some-provider")
|
||||||
|
with patch(
|
||||||
|
"EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg
|
||||||
|
):
|
||||||
|
resp = client.get("/api/models")
|
||||||
|
assert resp.json()["default"] == {
|
||||||
|
"name": "some-retired-name",
|
||||||
|
"provider": "some-provider",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_ollama_models_appended_when_base_url_configured():
|
||||||
|
"""Mirrors the TUI ``/model`` picker: when ``ollama_base_url`` is set,
|
||||||
|
locally-pulled Ollama models are appended after the static registry
|
||||||
|
as ``provider: "ollama"`` entries.
|
||||||
|
"""
|
||||||
|
mock_cfg = EvoScientistConfig(
|
||||||
|
model="claude-sonnet-4-6",
|
||||||
|
provider="custom-anthropic",
|
||||||
|
ollama_base_url="http://localhost:11434",
|
||||||
|
)
|
||||||
|
|
||||||
|
async def fake_discover(_base_url, *, timeout):
|
||||||
|
return ["llama3:8b", "mistral:7b"]
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http.get_effective_config",
|
||||||
|
return_value=mock_cfg,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.llm.ollama_discovery.discover_ollama_models",
|
||||||
|
new=fake_discover,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
body = client.get("/api/models").json()
|
||||||
|
|
||||||
|
# Assert the response is the static registry followed by the discovered
|
||||||
|
# Ollama suffix — robust to future static Ollama entries in the registry.
|
||||||
|
from EvoScientist.llm.models import list_models_by_provider
|
||||||
|
|
||||||
|
static_entries = [
|
||||||
|
{"name": n, "model_id": m, "provider": p}
|
||||||
|
for n, m, p in list_models_by_provider()
|
||||||
|
]
|
||||||
|
discovered_entries = [
|
||||||
|
{"name": "llama3:8b", "model_id": "llama3:8b", "provider": "ollama"},
|
||||||
|
{"name": "mistral:7b", "model_id": "mistral:7b", "provider": "ollama"},
|
||||||
|
]
|
||||||
|
assert body["entries"][: len(static_entries)] == static_entries
|
||||||
|
assert body["entries"][len(static_entries) :] == discovered_entries
|
||||||
|
# TUI's "Custom Ollama model…" sentinel is a widget-specific affordance —
|
||||||
|
# it must not appear on the HTTP surface.
|
||||||
|
assert not any(e["model_id"] == "__custom_ollama__" for e in body["entries"])
|
||||||
|
|
||||||
|
|
||||||
|
def test_ollama_discovery_skipped_when_base_url_absent():
|
||||||
|
"""No Ollama discovery should happen when ``ollama_base_url`` is unset —
|
||||||
|
matches the ``/model`` picker's gating. The probe function should never
|
||||||
|
be called in that case.
|
||||||
|
"""
|
||||||
|
mock_cfg = EvoScientistConfig(
|
||||||
|
model="claude-sonnet-4-6", provider="custom-anthropic"
|
||||||
|
)
|
||||||
|
calls: list[str | None] = []
|
||||||
|
|
||||||
|
async def spy_discover(base_url, *, timeout):
|
||||||
|
calls.append(base_url)
|
||||||
|
return []
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http.get_effective_config",
|
||||||
|
return_value=mock_cfg,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.llm.ollama_discovery.discover_ollama_models",
|
||||||
|
new=spy_discover,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
body = client.get("/api/models").json()
|
||||||
|
|
||||||
|
assert calls == []
|
||||||
|
# Response is exactly the static registry — no Ollama additions whatsoever.
|
||||||
|
from EvoScientist.llm.models import list_models_by_provider
|
||||||
|
|
||||||
|
assert body["entries"] == [
|
||||||
|
{"name": n, "model_id": m, "provider": p}
|
||||||
|
for n, m, p in list_models_by_provider()
|
||||||
|
]
|
||||||
Reference in New Issue
Block a user