diff --git a/src/app/hooks/useAvailableModels.ts b/src/app/hooks/useAvailableModels.ts index 88a38e4..dcee6d8 100644 --- a/src/app/hooks/useAvailableModels.ts +++ b/src/app/hooks/useAvailableModels.ts @@ -28,17 +28,80 @@ interface RegistryResponse { const EMPTY: ModelRegistry = { entries: [], defaultEntry: null }; +// Module-level cache keyed by normalised deploymentUrl. The registry is static +// between deployment restarts — one network round-trip per URL per page load +// is enough regardless of how many times ChatInterface mounts/unmounts. +// Failed fetches are evicted so the next mount retries. +const cache = new Map>(); + +function fetchRegistry(deploymentUrl: string, apiKey: string): Promise { + const key = deploymentUrl.replace(/\/$/, ""); + const hit = cache.get(key); + if (hit) return hit; + + const headers: Record = {}; + if (apiKey) headers["X-Api-Key"] = apiKey; + + const p = fetch(`${key}/api/models`, { headers }) + .then(async (r) => { + if (!r.ok) throw new Error(`HTTP ${r.status}`); + const body = (await r.json()) as RegistryResponse; + const entries: ModelRegistryEntry[] = []; + if (Array.isArray(body.entries)) { + for (const raw of body.entries) { + if (!raw || typeof raw !== "object") continue; + const e = raw as { + name?: unknown; + model_id?: unknown; + provider?: unknown; + }; + if ( + typeof e.name === "string" && + typeof e.provider === "string" && + e.name && + e.provider + ) { + entries.push({ + name: e.name, + model_id: typeof e.model_id === "string" ? e.model_id : e.name, + provider: e.provider, + }); + } + } + } + let defaultEntry: ModelRegistry["defaultEntry"] = null; + if (body.default && typeof body.default === "object") { + const d = body.default as { name?: unknown; provider?: unknown }; + if (typeof d.name === "string" && d.name) { + defaultEntry = { + name: d.name, + provider: + typeof d.provider === "string" && d.provider + ? d.provider + : null, + }; + } + } + return { entries, defaultEntry } as ModelRegistry; + }) + .catch((err: unknown) => { + cache.delete(key); + throw err; + }); + + cache.set(key, p); + return p; +} + /** * Fetch the backend's authoritative model registry from - * `GET ${deploymentUrl}/api/models`. The endpoint is mounted into the - * langgraph-dev process (see `.backend-ref/notes/expose-available-models-endpoint.md`) - * so the URL lives on the same origin/port as the SDK, not the WebUI's - * Next.js server. We fetch once per session and cache in component state — - * the registry is large (~120 entries) but static between deployment - * restarts. + * `GET ${deploymentUrl}/api/models`. Results are cached at module level — + * the registry is static between deployment restarts, so remounting + * ChatInterface never triggers a redundant network request. * * Failures are non-fatal: the picker falls back to its curated - * `COMMON_MODELS` list when `entries` is empty. + * `COMMON_MODELS` list when `entries` is empty. Failed fetches are evicted + * from the cache so the next mount retries. */ export function useAvailableModels(): { registry: ModelRegistry; @@ -56,58 +119,16 @@ export function useAvailableModels(): { return; } let cancelled = false; - setLoading(true); - setError(null); const apiKey = cfg.langsmithApiKey || process.env.NEXT_PUBLIC_LANGSMITH_API_KEY || ""; - const headers: Record = {}; - if (apiKey) headers["X-Api-Key"] = apiKey; - fetch(`${cfg.deploymentUrl.replace(/\/$/, "")}/api/models`, { headers }) - .then(async (r) => { - if (!r.ok) throw new Error(`HTTP ${r.status}`); - return (await r.json()) as RegistryResponse; - }) - .then((body) => { + + fetchRegistry(cfg.deploymentUrl, apiKey) + .then((result) => { if (cancelled) return; - const entries: ModelRegistryEntry[] = []; - if (Array.isArray(body.entries)) { - for (const raw of body.entries) { - if (!raw || typeof raw !== "object") continue; - const e = raw as { - name?: unknown; - model_id?: unknown; - provider?: unknown; - }; - if ( - typeof e.name === "string" && - typeof e.provider === "string" && - e.name && - e.provider - ) { - entries.push({ - name: e.name, - model_id: typeof e.model_id === "string" ? e.model_id : e.name, - provider: e.provider, - }); - } - } - } - let defaultEntry: ModelRegistry["defaultEntry"] = null; - if (body.default && typeof body.default === "object") { - const d = body.default as { name?: unknown; provider?: unknown }; - if (typeof d.name === "string" && d.name) { - defaultEntry = { - name: d.name, - provider: - typeof d.provider === "string" && d.provider - ? d.provider - : null, - }; - } - } - setRegistry({ entries, defaultEntry }); + setRegistry(result); + setError(null); }) - .catch((err) => { + .catch((err: unknown) => { if (cancelled) return; setError(err instanceof Error ? err.message : "Failed to load models."); setRegistry(EMPTY); @@ -115,6 +136,7 @@ export function useAvailableModels(): { .finally(() => { if (!cancelled) setLoading(false); }); + return () => { cancelled = true; }; diff --git a/src/lib/modelCommand.ts b/src/lib/modelCommand.ts index 0b903f0..e2e4fe1 100644 --- a/src/lib/modelCommand.ts +++ b/src/lib/modelCommand.ts @@ -7,11 +7,12 @@ // and persist the choice per-thread (in thread metadata, not localStorage — // the choice should follow the conversation, not the browser tab). // -// Listing available models is a separate concern: the backend has the -// authoritative registry in `EvoScientist/llm/models.py`, but no HTTP endpoint -// exposes it today. Until that lands we ship a curated `COMMON_MODELS` list -// for the picker. Names outside the list are NOT rejected — `/model ` -// passes through verbatim and the middleware accepts anything `init_chat_model` +// Listing available models: the backend's authoritative registry lives in +// `EvoScientist/llm/models.py` and is exposed at `GET /api/models` (mounted +// via langgraph.json). `useAvailableModels` fetches that endpoint at runtime; +// `COMMON_MODELS` below is a fallback for older deployments or network +// failures. Names outside the list are NOT rejected — `/model ` passes +// through verbatim and the middleware accepts anything `init_chat_model` // recognises, so power users aren't blocked by our curation. /** Thread metadata key carrying the per-thread model override. Mirrors the