feat: implement caching for model registry fetch and improve error handling
This commit is contained in:
@@ -28,17 +28,80 @@ interface RegistryResponse {
|
||||
|
||||
const EMPTY: ModelRegistry = { entries: [], defaultEntry: null };
|
||||
|
||||
// Module-level cache keyed by normalised deploymentUrl. The registry is static
|
||||
// between deployment restarts — one network round-trip per URL per page load
|
||||
// is enough regardless of how many times ChatInterface mounts/unmounts.
|
||||
// Failed fetches are evicted so the next mount retries.
|
||||
const cache = new Map<string, Promise<ModelRegistry>>();
|
||||
|
||||
function fetchRegistry(deploymentUrl: string, apiKey: string): Promise<ModelRegistry> {
|
||||
const key = deploymentUrl.replace(/\/$/, "");
|
||||
const hit = cache.get(key);
|
||||
if (hit) return hit;
|
||||
|
||||
const headers: Record<string, string> = {};
|
||||
if (apiKey) headers["X-Api-Key"] = apiKey;
|
||||
|
||||
const p = fetch(`${key}/api/models`, { headers })
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error(`HTTP ${r.status}`);
|
||||
const body = (await r.json()) as RegistryResponse;
|
||||
const entries: ModelRegistryEntry[] = [];
|
||||
if (Array.isArray(body.entries)) {
|
||||
for (const raw of body.entries) {
|
||||
if (!raw || typeof raw !== "object") continue;
|
||||
const e = raw as {
|
||||
name?: unknown;
|
||||
model_id?: unknown;
|
||||
provider?: unknown;
|
||||
};
|
||||
if (
|
||||
typeof e.name === "string" &&
|
||||
typeof e.provider === "string" &&
|
||||
e.name &&
|
||||
e.provider
|
||||
) {
|
||||
entries.push({
|
||||
name: e.name,
|
||||
model_id: typeof e.model_id === "string" ? e.model_id : e.name,
|
||||
provider: e.provider,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
let defaultEntry: ModelRegistry["defaultEntry"] = null;
|
||||
if (body.default && typeof body.default === "object") {
|
||||
const d = body.default as { name?: unknown; provider?: unknown };
|
||||
if (typeof d.name === "string" && d.name) {
|
||||
defaultEntry = {
|
||||
name: d.name,
|
||||
provider:
|
||||
typeof d.provider === "string" && d.provider
|
||||
? d.provider
|
||||
: null,
|
||||
};
|
||||
}
|
||||
}
|
||||
return { entries, defaultEntry } as ModelRegistry;
|
||||
})
|
||||
.catch((err: unknown) => {
|
||||
cache.delete(key);
|
||||
throw err;
|
||||
});
|
||||
|
||||
cache.set(key, p);
|
||||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch the backend's authoritative model registry from
|
||||
* `GET ${deploymentUrl}/api/models`. The endpoint is mounted into the
|
||||
* langgraph-dev process (see `.backend-ref/notes/expose-available-models-endpoint.md`)
|
||||
* so the URL lives on the same origin/port as the SDK, not the WebUI's
|
||||
* Next.js server. We fetch once per session and cache in component state —
|
||||
* the registry is large (~120 entries) but static between deployment
|
||||
* restarts.
|
||||
* `GET ${deploymentUrl}/api/models`. Results are cached at module level —
|
||||
* the registry is static between deployment restarts, so remounting
|
||||
* ChatInterface never triggers a redundant network request.
|
||||
*
|
||||
* Failures are non-fatal: the picker falls back to its curated
|
||||
* `COMMON_MODELS` list when `entries` is empty.
|
||||
* `COMMON_MODELS` list when `entries` is empty. Failed fetches are evicted
|
||||
* from the cache so the next mount retries.
|
||||
*/
|
||||
export function useAvailableModels(): {
|
||||
registry: ModelRegistry;
|
||||
@@ -56,58 +119,16 @@ export function useAvailableModels(): {
|
||||
return;
|
||||
}
|
||||
let cancelled = false;
|
||||
setLoading(true);
|
||||
setError(null);
|
||||
const apiKey =
|
||||
cfg.langsmithApiKey || process.env.NEXT_PUBLIC_LANGSMITH_API_KEY || "";
|
||||
const headers: Record<string, string> = {};
|
||||
if (apiKey) headers["X-Api-Key"] = apiKey;
|
||||
fetch(`${cfg.deploymentUrl.replace(/\/$/, "")}/api/models`, { headers })
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error(`HTTP ${r.status}`);
|
||||
return (await r.json()) as RegistryResponse;
|
||||
})
|
||||
.then((body) => {
|
||||
|
||||
fetchRegistry(cfg.deploymentUrl, apiKey)
|
||||
.then((result) => {
|
||||
if (cancelled) return;
|
||||
const entries: ModelRegistryEntry[] = [];
|
||||
if (Array.isArray(body.entries)) {
|
||||
for (const raw of body.entries) {
|
||||
if (!raw || typeof raw !== "object") continue;
|
||||
const e = raw as {
|
||||
name?: unknown;
|
||||
model_id?: unknown;
|
||||
provider?: unknown;
|
||||
};
|
||||
if (
|
||||
typeof e.name === "string" &&
|
||||
typeof e.provider === "string" &&
|
||||
e.name &&
|
||||
e.provider
|
||||
) {
|
||||
entries.push({
|
||||
name: e.name,
|
||||
model_id: typeof e.model_id === "string" ? e.model_id : e.name,
|
||||
provider: e.provider,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
let defaultEntry: ModelRegistry["defaultEntry"] = null;
|
||||
if (body.default && typeof body.default === "object") {
|
||||
const d = body.default as { name?: unknown; provider?: unknown };
|
||||
if (typeof d.name === "string" && d.name) {
|
||||
defaultEntry = {
|
||||
name: d.name,
|
||||
provider:
|
||||
typeof d.provider === "string" && d.provider
|
||||
? d.provider
|
||||
: null,
|
||||
};
|
||||
}
|
||||
}
|
||||
setRegistry({ entries, defaultEntry });
|
||||
setRegistry(result);
|
||||
setError(null);
|
||||
})
|
||||
.catch((err) => {
|
||||
.catch((err: unknown) => {
|
||||
if (cancelled) return;
|
||||
setError(err instanceof Error ? err.message : "Failed to load models.");
|
||||
setRegistry(EMPTY);
|
||||
@@ -115,6 +136,7 @@ export function useAvailableModels(): {
|
||||
.finally(() => {
|
||||
if (!cancelled) setLoading(false);
|
||||
});
|
||||
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
|
||||
@@ -7,11 +7,12 @@
|
||||
// and persist the choice per-thread (in thread metadata, not localStorage —
|
||||
// the choice should follow the conversation, not the browser tab).
|
||||
//
|
||||
// Listing available models is a separate concern: the backend has the
|
||||
// authoritative registry in `EvoScientist/llm/models.py`, but no HTTP endpoint
|
||||
// exposes it today. Until that lands we ship a curated `COMMON_MODELS` list
|
||||
// for the picker. Names outside the list are NOT rejected — `/model <name>`
|
||||
// passes through verbatim and the middleware accepts anything `init_chat_model`
|
||||
// Listing available models: the backend's authoritative registry lives in
|
||||
// `EvoScientist/llm/models.py` and is exposed at `GET /api/models` (mounted
|
||||
// via langgraph.json). `useAvailableModels` fetches that endpoint at runtime;
|
||||
// `COMMON_MODELS` below is a fallback for older deployments or network
|
||||
// failures. Names outside the list are NOT rejected — `/model <name>` passes
|
||||
// through verbatim and the middleware accepts anything `init_chat_model`
|
||||
// recognises, so power users aren't blocked by our curation.
|
||||
|
||||
/** Thread metadata key carrying the per-thread model override. Mirrors the
|
||||
|
||||
Reference in New Issue
Block a user