feat: implement caching for model registry fetch and improve error handling

This commit is contained in:
Xi Zhang
2026-06-27 00:22:29 +01:00
parent 0be37b54e7
commit 964430441a
2 changed files with 83 additions and 60 deletions
+77 -55
View File
@@ -28,17 +28,80 @@ interface RegistryResponse {
const EMPTY: ModelRegistry = { entries: [], defaultEntry: null };
// Module-level cache keyed by normalised deploymentUrl. The registry is static
// between deployment restarts — one network round-trip per URL per page load
// is enough regardless of how many times ChatInterface mounts/unmounts.
// Failed fetches are evicted so the next mount retries.
const cache = new Map<string, Promise<ModelRegistry>>();
function fetchRegistry(deploymentUrl: string, apiKey: string): Promise<ModelRegistry> {
const key = deploymentUrl.replace(/\/$/, "");
const hit = cache.get(key);
if (hit) return hit;
const headers: Record<string, string> = {};
if (apiKey) headers["X-Api-Key"] = apiKey;
const p = fetch(`${key}/api/models`, { headers })
.then(async (r) => {
if (!r.ok) throw new Error(`HTTP ${r.status}`);
const body = (await r.json()) as RegistryResponse;
const entries: ModelRegistryEntry[] = [];
if (Array.isArray(body.entries)) {
for (const raw of body.entries) {
if (!raw || typeof raw !== "object") continue;
const e = raw as {
name?: unknown;
model_id?: unknown;
provider?: unknown;
};
if (
typeof e.name === "string" &&
typeof e.provider === "string" &&
e.name &&
e.provider
) {
entries.push({
name: e.name,
model_id: typeof e.model_id === "string" ? e.model_id : e.name,
provider: e.provider,
});
}
}
}
let defaultEntry: ModelRegistry["defaultEntry"] = null;
if (body.default && typeof body.default === "object") {
const d = body.default as { name?: unknown; provider?: unknown };
if (typeof d.name === "string" && d.name) {
defaultEntry = {
name: d.name,
provider:
typeof d.provider === "string" && d.provider
? d.provider
: null,
};
}
}
return { entries, defaultEntry } as ModelRegistry;
})
.catch((err: unknown) => {
cache.delete(key);
throw err;
});
cache.set(key, p);
return p;
}
/**
* Fetch the backend's authoritative model registry from
* `GET ${deploymentUrl}/api/models`. The endpoint is mounted into the
* langgraph-dev process (see `.backend-ref/notes/expose-available-models-endpoint.md`)
* so the URL lives on the same origin/port as the SDK, not the WebUI's
* Next.js server. We fetch once per session and cache in component state —
* the registry is large (~120 entries) but static between deployment
* restarts.
* `GET ${deploymentUrl}/api/models`. Results are cached at module level —
* the registry is static between deployment restarts, so remounting
* ChatInterface never triggers a redundant network request.
*
* Failures are non-fatal: the picker falls back to its curated
* `COMMON_MODELS` list when `entries` is empty.
* `COMMON_MODELS` list when `entries` is empty. Failed fetches are evicted
* from the cache so the next mount retries.
*/
export function useAvailableModels(): {
registry: ModelRegistry;
@@ -56,58 +119,16 @@ export function useAvailableModels(): {
return;
}
let cancelled = false;
setLoading(true);
setError(null);
const apiKey =
cfg.langsmithApiKey || process.env.NEXT_PUBLIC_LANGSMITH_API_KEY || "";
const headers: Record<string, string> = {};
if (apiKey) headers["X-Api-Key"] = apiKey;
fetch(`${cfg.deploymentUrl.replace(/\/$/, "")}/api/models`, { headers })
.then(async (r) => {
if (!r.ok) throw new Error(`HTTP ${r.status}`);
return (await r.json()) as RegistryResponse;
})
.then((body) => {
fetchRegistry(cfg.deploymentUrl, apiKey)
.then((result) => {
if (cancelled) return;
const entries: ModelRegistryEntry[] = [];
if (Array.isArray(body.entries)) {
for (const raw of body.entries) {
if (!raw || typeof raw !== "object") continue;
const e = raw as {
name?: unknown;
model_id?: unknown;
provider?: unknown;
};
if (
typeof e.name === "string" &&
typeof e.provider === "string" &&
e.name &&
e.provider
) {
entries.push({
name: e.name,
model_id: typeof e.model_id === "string" ? e.model_id : e.name,
provider: e.provider,
});
}
}
}
let defaultEntry: ModelRegistry["defaultEntry"] = null;
if (body.default && typeof body.default === "object") {
const d = body.default as { name?: unknown; provider?: unknown };
if (typeof d.name === "string" && d.name) {
defaultEntry = {
name: d.name,
provider:
typeof d.provider === "string" && d.provider
? d.provider
: null,
};
}
}
setRegistry({ entries, defaultEntry });
setRegistry(result);
setError(null);
})
.catch((err) => {
.catch((err: unknown) => {
if (cancelled) return;
setError(err instanceof Error ? err.message : "Failed to load models.");
setRegistry(EMPTY);
@@ -115,6 +136,7 @@ export function useAvailableModels(): {
.finally(() => {
if (!cancelled) setLoading(false);
});
return () => {
cancelled = true;
};
+6 -5
View File
@@ -7,11 +7,12 @@
// and persist the choice per-thread (in thread metadata, not localStorage —
// the choice should follow the conversation, not the browser tab).
//
// Listing available models is a separate concern: the backend has the
// authoritative registry in `EvoScientist/llm/models.py`, but no HTTP endpoint
// exposes it today. Until that lands we ship a curated `COMMON_MODELS` list
// for the picker. Names outside the list are NOT rejected — `/model <name>`
// passes through verbatim and the middleware accepts anything `init_chat_model`
// Listing available models: the backend's authoritative registry lives in
// `EvoScientist/llm/models.py` and is exposed at `GET /api/models` (mounted
// via langgraph.json). `useAvailableModels` fetches that endpoint at runtime;
// `COMMON_MODELS` below is a fallback for older deployments or network
// failures. Names outside the list are NOT rejected — `/model <name>` passes
// through verbatim and the middleware accepts anything `init_chat_model`
// recognises, so power users aren't blocked by our curation.
/** Thread metadata key carrying the per-thread model override. Mirrors the