From 7214d4099deb8c27e41c784f455ff2045de168fb Mon Sep 17 00:00:00 2001 From: jfilipiuk Date: Fri, 26 Jun 2026 23:48:37 +0200 Subject: [PATCH] feat: expose model registry at GET /api/models (#308) * feat: expose model registry at GET /api/models * fix: include Ollama models in /api/models endpoint * fix: honor env vars override in /api/models endpoint * fix: offload get_effective_config to thread to satisfy blockbuster --------- Co-authored-by: Xi Zhang <106144707+X-iZhang@users.noreply.github.com> --- EvoScientist/langgraph_dev/http.py | 85 +++++++++++ EvoScientist/langgraph_dev/langgraph.json | 3 + tests/test_langgraph_dev_http.py | 163 ++++++++++++++++++++++ 3 files changed, 251 insertions(+) create mode 100644 EvoScientist/langgraph_dev/http.py create mode 100644 tests/test_langgraph_dev_http.py diff --git a/EvoScientist/langgraph_dev/http.py b/EvoScientist/langgraph_dev/http.py new file mode 100644 index 0000000..10a047e --- /dev/null +++ b/EvoScientist/langgraph_dev/http.py @@ -0,0 +1,85 @@ +"""Custom HTTP routes mounted alongside the langgraph dev server. + +The langgraph-api host supports a top-level ``http`` key in +``langgraph.json`` that names an ASGI app to mount on the same +process as the graph. We use it to surface the registry the WebUI's +``/model`` picker needs. + +Why this lives here and not as a separate sidecar: the WebUI talks to +``EvoSci deploy``'s langgraph endpoint anyway, so one origin keeps the +WebUI's fetch logic simple — no CORS dance, no extra port to configure. + +Why Starlette and not FastAPI: ``langgraph_api`` already depends on +Starlette; adding FastAPI would pull in pydantic v1-vs-v2 reconciliation +the deploy doesn't need. The one route here has no input model, just a +JSON body, so the lower-level surface is sufficient. + +Lightweight by design — module-level imports stick to ``config``, +``llm.models`` (registry only; no chat-model construction), and +Starlette itself. Nothing on this surface should pull the agent into +memory. +""" + +from __future__ import annotations + +import asyncio + +from starlette.applications import Starlette +from starlette.requests import Request +from starlette.responses import JSONResponse +from starlette.routing import Route + +from EvoScientist.config import get_effective_config +from EvoScientist.llm.models import list_models_by_provider + + +async def get_models(_request: Request) -> JSONResponse: + """Return the model registry as ``{entries, default}``. + + ``entries`` preserves the registry order so the WebUI picker can + rank providers per short name the same way the backend would. + Mirrors the TUI ``/model`` picker by appending locally-pulled + Ollama models when ``ollama_base_url`` is configured — same + ``discover_ollama_models()`` call, same 1.5-s timeout, same + fail-soft semantics (the probe returns ``[]`` on any error, never + raises). The TUI's "Custom Ollama model…" sentinel is intentionally + omitted: that's a widget-specific input affordance, not part of + the registry surface. + + ``default`` reflects the deployment's currently-configured fallback + (``config.yaml``'s ``model`` / ``provider`` — what ``/model reset`` + would land on). Returned even when the configured pair isn't in + the registry, so the picker can still label it. + + Uses ``get_effective_config()`` (not ``load_config()``) so env-var + overrides like ``OLLAMA_BASE_URL`` from ``_ENV_MAPPINGS`` are + honored — matching the deploy's actual model-building behavior. + Offloaded to a thread because ``get_effective_config()`` calls + ``find_dotenv(usecwd=True)`` which invokes ``os.getcwd()`` — a + blocking syscall that langgraph-dev's ``blockbuster`` middleware + refuses to allow on the async event loop (would surface as a 500). + """ + cfg = await asyncio.to_thread(get_effective_config) + entries = [ + {"name": name, "model_id": model_id, "provider": provider} + for name, model_id, provider in list_models_by_provider() + ] + ollama_base_url = getattr(cfg, "ollama_base_url", None) + if ollama_base_url: + from EvoScientist.llm.ollama_discovery import discover_ollama_models + + for name in await discover_ollama_models(ollama_base_url, timeout=1.5): + entries.append({"name": name, "model_id": name, "provider": "ollama"}) + return JSONResponse( + { + "entries": entries, + "default": {"name": cfg.model, "provider": cfg.provider}, + } + ) + + +app = Starlette( + routes=[ + Route("/api/models", get_models, methods=["GET"]), + ] +) diff --git a/EvoScientist/langgraph_dev/langgraph.json b/EvoScientist/langgraph_dev/langgraph.json index 5414957..3cb6555 100644 --- a/EvoScientist/langgraph_dev/langgraph.json +++ b/EvoScientist/langgraph_dev/langgraph.json @@ -15,5 +15,8 @@ }, "config": { "recursion_limit": 1000000 + }, + "http": { + "app": "EvoScientist.langgraph_dev.http:app" } } diff --git a/tests/test_langgraph_dev_http.py b/tests/test_langgraph_dev_http.py new file mode 100644 index 0000000..88cd3ca --- /dev/null +++ b/tests/test_langgraph_dev_http.py @@ -0,0 +1,163 @@ +"""Smoke test for the /api/models route mounted via langgraph.json's +``http`` field. We test the FastAPI app directly — no need to spin up +langgraph dev. +""" + +from __future__ import annotations + +from unittest.mock import patch + +from starlette.testclient import TestClient + +from EvoScientist.config import EvoScientistConfig +from EvoScientist.langgraph_dev.http import app + +client = TestClient(app) + + +def test_get_models_returns_entries_and_default(): + mock_cfg = EvoScientistConfig( + model="claude-sonnet-4-6", provider="custom-anthropic" + ) + with patch( + "EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg + ): + resp = client.get("/api/models") + assert resp.status_code == 200 + body = resp.json() + assert "entries" in body + assert "default" in body + assert body["default"] == { + "name": "claude-sonnet-4-6", + "provider": "custom-anthropic", + } + assert isinstance(body["entries"], list) + assert len(body["entries"]) > 0 + # Every entry has the three required keys + for entry in body["entries"]: + assert set(entry.keys()) == {"name", "model_id", "provider"} + assert isinstance(entry["name"], str) + assert entry["name"] + assert isinstance(entry["model_id"], str) + assert entry["model_id"] + assert isinstance(entry["provider"], str) + assert entry["provider"] + + +def test_entries_preserve_registry_order(): + """The picker uses position-in-list to rank providers per short name — + the JSON must preserve the order returned by ``list_models_by_provider``. + + Stubs ``get_effective_config`` to keep the assertion focused on + registry order rather than implicitly depending on the ambient + deploy config. + """ + from EvoScientist.llm.models import list_models_by_provider + + expected = [ + {"name": n, "model_id": m, "provider": p} + for n, m, p in list_models_by_provider() + ] + mock_cfg = EvoScientistConfig() + with patch( + "EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg + ): + resp = client.get("/api/models") + assert resp.json()["entries"] == expected + + +def test_default_passes_through_arbitrary_config_pair(): + """If config.yaml names a (name, provider) pair that isn't in the + registry (typo, retired model), still report it as default — the + picker labels it as the active selection regardless. + """ + mock_cfg = EvoScientistConfig(model="some-retired-name", provider="some-provider") + with patch( + "EvoScientist.langgraph_dev.http.get_effective_config", return_value=mock_cfg + ): + resp = client.get("/api/models") + assert resp.json()["default"] == { + "name": "some-retired-name", + "provider": "some-provider", + } + + +def test_ollama_models_appended_when_base_url_configured(): + """Mirrors the TUI ``/model`` picker: when ``ollama_base_url`` is set, + locally-pulled Ollama models are appended after the static registry + as ``provider: "ollama"`` entries. + """ + mock_cfg = EvoScientistConfig( + model="claude-sonnet-4-6", + provider="custom-anthropic", + ollama_base_url="http://localhost:11434", + ) + + async def fake_discover(_base_url, *, timeout): + return ["llama3:8b", "mistral:7b"] + + with ( + patch( + "EvoScientist.langgraph_dev.http.get_effective_config", + return_value=mock_cfg, + ), + patch( + "EvoScientist.llm.ollama_discovery.discover_ollama_models", + new=fake_discover, + ), + ): + body = client.get("/api/models").json() + + # Assert the response is the static registry followed by the discovered + # Ollama suffix — robust to future static Ollama entries in the registry. + from EvoScientist.llm.models import list_models_by_provider + + static_entries = [ + {"name": n, "model_id": m, "provider": p} + for n, m, p in list_models_by_provider() + ] + discovered_entries = [ + {"name": "llama3:8b", "model_id": "llama3:8b", "provider": "ollama"}, + {"name": "mistral:7b", "model_id": "mistral:7b", "provider": "ollama"}, + ] + assert body["entries"][: len(static_entries)] == static_entries + assert body["entries"][len(static_entries) :] == discovered_entries + # TUI's "Custom Ollama model…" sentinel is a widget-specific affordance — + # it must not appear on the HTTP surface. + assert not any(e["model_id"] == "__custom_ollama__" for e in body["entries"]) + + +def test_ollama_discovery_skipped_when_base_url_absent(): + """No Ollama discovery should happen when ``ollama_base_url`` is unset — + matches the ``/model`` picker's gating. The probe function should never + be called in that case. + """ + mock_cfg = EvoScientistConfig( + model="claude-sonnet-4-6", provider="custom-anthropic" + ) + calls: list[str | None] = [] + + async def spy_discover(base_url, *, timeout): + calls.append(base_url) + return [] + + with ( + patch( + "EvoScientist.langgraph_dev.http.get_effective_config", + return_value=mock_cfg, + ), + patch( + "EvoScientist.llm.ollama_discovery.discover_ollama_models", + new=spy_discover, + ), + ): + body = client.get("/api/models").json() + + assert calls == [] + # Response is exactly the static registry — no Ollama additions whatsoever. + from EvoScientist.llm.models import list_models_by_provider + + assert body["entries"] == [ + {"name": n, "model_id": m, "provider": p} + for n, m, p in list_models_by_provider() + ]