From c401756a6a4a71092fa0a21a4ffae43ef67e3c59 Mon Sep 17 00:00:00 2001 From: ClintonEmok <54935030+ClintonEmok@users.noreply.github.com> Date: Thu, 3 Sep 2026 12:12:24 +0530 Subject: [PATCH] fix(desktop): pool sizing as a live device preference in Settings (#91545) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hover-intent prewarm sweeps across the Bots rail spawned past the pool cap, LRU-evicting the backend the user was about to click — an evict/respawn cascade that made profile switching progressively slower (#91545). - prewarmProfileBackend skips speculative spawns once every pool slot holds an open socket; the real click still spawns on demand. - Pool max/idle become a device preference (Settings -> Advanced), persisted atomically in userData (pool-limits.json) and applied live over IPC; the HERMES_DESKTOP_POOL_* env vars remain the initial fallback. Defaults are unchanged (3 backends / 10 min idle). Squash of the 3-commit PR #92581 branch (a00dc088c5..783899d12f) applied via diff onto the spawn-coordinator salvage; import + constant-block conflicts resolved so the coordinator is constructed from, and follows, the live preference (setLimit added in the next commit). --- apps/desktop/electron/main.ts | 110 +++++++++++++++-- apps/desktop/electron/pool-limits.test.ts | 47 ++++++++ apps/desktop/electron/pool-limits.ts | 81 +++++++++++++ apps/desktop/electron/preload.ts | 2 + .../src/app/gateway/hooks/use-gateway-boot.ts | 5 + .../src/app/settings/config-settings.tsx | 2 + .../src/app/settings/pool-limits-setting.tsx | 113 ++++++++++++++++++ apps/desktop/src/global.d.ts | 10 ++ apps/desktop/src/store/gateway.ts | 18 +++ apps/desktop/src/store/pool-limits.ts | 65 ++++++++++ apps/desktop/src/store/profile.test.ts | 56 ++++++++- apps/desktop/src/store/profile.ts | 15 ++- 12 files changed, 510 insertions(+), 14 deletions(-) create mode 100644 apps/desktop/electron/pool-limits.test.ts create mode 100644 apps/desktop/electron/pool-limits.ts create mode 100644 apps/desktop/src/app/settings/pool-limits-setting.tsx create mode 100644 apps/desktop/src/store/pool-limits.ts diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index d466e13725..a9bb25a291 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -281,6 +281,7 @@ import { undialedSshRouteSeeds } from './plugin-profile-routes' import { selectPoolEvictions } from './pool-eviction' +import { clampPoolLimits, parsePoolLimits, POOL_LIMITS_DEFAULTS } from './pool-limits' import { LocalBackendSpawnCoordinator, type LocalBackendSpawnRequest, @@ -1414,14 +1415,87 @@ const profileDeletionGate = new ProfileDeletionGate() // Keep the pool light: cap concurrent profile backends (LRU eviction) and reap // idle ones. A user idles at exactly the primary backend; pool backends only // exist while a non-primary profile is actively being chatted through. -const POOL_MAX_BACKENDS = Math.max(1, Number(process.env.HERMES_DESKTOP_POOL_MAX) || 3) -const POOL_IDLE_MS = Math.max(60_000, Number(process.env.HERMES_DESKTOP_POOL_IDLE_MS) || 10 * 60_000) -const localBackendSpawnCoordinator = new LocalBackendSpawnCoordinator(POOL_MAX_BACKENDS) +// Pool sizing is a device preference (Settings → Advanced → pool rows), not a +// launch constant: mutable at runtime, persisted in userData, applied live. +// The legacy HERMES_DESKTOP_POOL_* env vars remain the initial-value fallback +// for scripted/headless setups; after launch the stored preference wins. +const POOL_LIMITS_PATH = path.join(app.getPath('userData'), 'pool-limits.json') + +function readPersistedPoolLimits() { + try { + const limits = parsePoolLimits(fs.readFileSync(POOL_LIMITS_PATH, 'utf8')) + rememberLog( + `[pool-limits] loaded from ${POOL_LIMITS_PATH}: maxBackends=${limits.maxBackends}, idleMs=${limits.idleMs}` + ) + + return limits + } catch { + // No persisted file yet — fall back to the legacy env vars so scripted + // setups keep working. Log which source won: a silently-ignored env var + // here costs a scripted-setup user a debugging session. + const fromEnv = clampPoolLimits({ + maxBackends: Number(process.env.HERMES_DESKTOP_POOL_MAX) || undefined, + idleMs: Number(process.env.HERMES_DESKTOP_POOL_IDLE_MS) || undefined + }) + + if (fromEnv.maxBackends !== POOL_LIMITS_DEFAULTS.maxBackends || fromEnv.idleMs !== POOL_LIMITS_DEFAULTS.idleMs) { + rememberLog(`[pool-limits] no saved file; using env-var overrides: maxBackends=${fromEnv.maxBackends}, idleMs=${fromEnv.idleMs}`) + } else { + rememberLog('[pool-limits] no saved file and no env overrides; using defaults') + } + + return fromEnv + } +} + +function persistPoolLimits(limits) { + try { + fs.mkdirSync(path.dirname(POOL_LIMITS_PATH), { recursive: true }) + // Atomic write: write to a temp file in the same directory, then rename. + // A crash mid-write would otherwise leave truncated JSON and silently + // lose the user's saved sizing. + const tmpPath = `${POOL_LIMITS_PATH}.tmp` + fs.writeFileSync(tmpPath, JSON.stringify(limits, null, 2), 'utf8') + fs.renameSync(tmpPath, POOL_LIMITS_PATH) + } catch (error) { + rememberLog(`[pool-limits] write failed: ${error.message}`) + } +} + +let poolLimits = readPersistedPoolLimits() +// Hard cap on local backends that are starting OR running (the LRU eviction +// above is soft — it spares keepalive-fresh entries). Follows the live +// preference: setPoolLimits() pushes a new max into the coordinator. +const localBackendSpawnCoordinator = new LocalBackendSpawnCoordinator(poolLimits.maxBackends) // How long a spawn may wait for a free local slot. Must stay under the // renderer's BACKEND_BOOT_WAIT_TIMEOUT_MS (45s, src/lib/with-timeout.ts) so // the queued ticket fails before the renderer does and the user sees why. const POOL_SLOT_WAIT_MS = 30_000 +function poolMaxBackends() { + return poolLimits.maxBackends +} + +function poolIdleMs() { + return poolLimits.idleMs +} + +/** + * Apply new limits live: persist, then converge the running pool — evict + * LRU backends down to the new max, and let the (already running) idle + * reaper handle a shortened idle window on its next tick. Returns the + * limits actually in force (post-clamp). + */ +function setPoolLimits(raw) { + poolLimits = clampPoolLimits(raw) + persistPoolLimits(poolLimits) + localBackendSpawnCoordinator.setLimit(poolLimits.maxBackends) + evictLruPoolBackends(poolMaxBackends()) + startPoolIdleReaper() + + return { ...poolLimits } +} + // A backend touched within this window has a live renderer socket (the keepalive // pings every 60s for every open profile). LRU eviction must spare these — a // concurrent multi-profile session keeps several backends "fresh" at once, and @@ -1440,7 +1514,7 @@ const POOL_SLOT_WAIT_MS = 30_000 // re-allocating pooled gateway secondaries ~700×/day). // * 3× ping + 60s headroom = ~4 min, comfortable margin for two missed // pings + WSL2 IPC stall. The hard ceiling for the cap-eligible set is -// POOL_IDLE_MS above (default 10 min) — this constant only governs the +// pool idle window above (default 10 min) — this constant only governs the // "is this backend plausibly still alive" question for LRU eviction, // not when the idle reaper definitively tears a backend down. const POOL_KEEPALIVE_FRESH_MS = Math.max( @@ -11271,7 +11345,7 @@ async function ensureBackend(profile) { return connection } - evictLruPoolBackends(POOL_MAX_BACKENDS - 1) + evictLruPoolBackends(poolMaxBackends() - 1) const entry = { process: null, @@ -11437,7 +11511,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela return existingLocal.connectionPromise } - evictLruPoolBackends(POOL_MAX_BACKENDS - 1) + evictLruPoolBackends(poolMaxBackends() - 1) const localEntry = { process: null, @@ -11508,7 +11582,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela ) } - evictLruPoolBackends(POOL_MAX_BACKENDS - 1) + evictLruPoolBackends(poolMaxBackends() - 1) const entry = { process: null, @@ -12163,7 +12237,7 @@ function evictLruPoolBackends(keep) { const evictions = selectPoolEvictions(backendPool.entries(), Math.max(0, keep), Date.now(), POOL_KEEPALIVE_FRESH_MS) for (const profile of evictions) { - rememberLog(`Evicting idle profile backend "${profile}" (LRU cap ${POOL_MAX_BACKENDS})`) + rememberLog(`Evicting idle profile backend "${profile}" (LRU cap ${poolMaxBackends()})`) stopPoolBackend(profile) } } @@ -12177,8 +12251,8 @@ function startPoolIdleReaper() { const now = Date.now() for (const [profile, entry] of [...backendPool.entries()]) { - if (now - (entry.lastActiveAt || 0) > POOL_IDLE_MS) { - rememberLog(`Reaping idle profile backend "${profile}" (idle > ${Math.round(POOL_IDLE_MS / 1000)}s)`) + if (now - (entry.lastActiveAt || 0) > poolIdleMs()) { + rememberLog(`Reaping idle profile backend "${profile}" (idle > ${Math.round(poolIdleMs() / 1000)}s)`) stopPoolBackend(profile) } } @@ -12303,9 +12377,9 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po entry.localBackendSlotKey = poolKey entry.localBackendSpawnRequest = spawnRequest - if (localBackendSpawnCoordinator.activeCount >= POOL_MAX_BACKENDS) { + if (localBackendSpawnCoordinator.activeCount >= poolMaxBackends()) { rememberLog( - `Profile backend "${profile}" waiting for a free local slot (${localBackendSpawnCoordinator.activeCount}/${POOL_MAX_BACKENDS} busy, ${localBackendSpawnCoordinator.queuedCount} queued)` + `Profile backend "${profile}" waiting for a free local slot (${localBackendSpawnCoordinator.activeCount}/${poolMaxBackends()} busy, ${localBackendSpawnCoordinator.queuedCount} queued)` ) } @@ -14724,6 +14798,18 @@ ipcMain.handle('hermes:backend:touch', async (_event, profile) => { return { ok: true } }) +// Pool sizing (Settings → Advanced): device-local, live-applied. Main is +// authoritative (it owns the pool and the persisted copy); the returned +// limits are what actually took effect post-clamp. +ipcMain.handle('hermes:pool-limits:get', async () => ({ ...poolLimits })) +ipcMain.handle('hermes:pool-limits:set', async (_event, raw) => { + const next = setPoolLimits({ + maxBackends: typeof raw?.maxBackends === 'number' ? raw.maxBackends : poolLimits.maxBackends, + idleMs: typeof raw?.idleMs === 'number' ? raw.idleMs : poolLimits.idleMs + }) + + return { ok: true, limits: next } +}) ipcMain.handle('hermes:gateway:ws-url', async (_event, profile) => { return gatewayWsUrlIpcResult(() => freshGatewayWsUrl(profile)) }) diff --git a/apps/desktop/electron/pool-limits.test.ts b/apps/desktop/electron/pool-limits.test.ts new file mode 100644 index 0000000000..3dc86f0ed3 --- /dev/null +++ b/apps/desktop/electron/pool-limits.test.ts @@ -0,0 +1,47 @@ +import { describe, expect, it } from 'vitest' + +import { clampPoolLimits, parsePoolLimits, POOL_LIMITS_BOUNDS, POOL_LIMITS_DEFAULTS, POOL_LIMITS_MIN } from './pool-limits' + +describe('parsePoolLimits', () => { + it('falls back to defaults for null/empty/corrupt input', () => { + expect(parsePoolLimits(null)).toEqual(POOL_LIMITS_DEFAULTS) + expect(parsePoolLimits(undefined)).toEqual(POOL_LIMITS_DEFAULTS) + expect(parsePoolLimits('')).toEqual(POOL_LIMITS_DEFAULTS) + expect(parsePoolLimits('not json {')).toEqual(POOL_LIMITS_DEFAULTS) + }) + + it('parses a valid persisted blob', () => { + expect(parsePoolLimits(JSON.stringify({ maxBackends: 11, idleMs: 7_200_000 }))).toEqual({ + maxBackends: 11, + idleMs: 7_200_000 + }) + }) + + it('fills missing keys from defaults', () => { + expect(parsePoolLimits(JSON.stringify({ maxBackends: 5 }))).toEqual({ ...POOL_LIMITS_DEFAULTS, maxBackends: 5 }) + expect(parsePoolLimits('{}')).toEqual(POOL_LIMITS_DEFAULTS) + }) + + it('ignores non-numeric junk instead of NaN-poisoning the pool', () => { + expect(parsePoolLimits(JSON.stringify({ maxBackends: 'lots', idleMs: null }))).toEqual(POOL_LIMITS_DEFAULTS) + }) +}) + +describe('clampPoolLimits', () => { + it('clamps below the floors', () => { + expect(clampPoolLimits({ maxBackends: 0 }).maxBackends).toBe(POOL_LIMITS_MIN.maxBackends) + expect(clampPoolLimits({ idleMs: 100 }).idleMs).toBe(POOL_LIMITS_MIN.idleMs) + }) + + it('clamps absurdly high backend counts', () => { + expect(clampPoolLimits({ maxBackends: 10_000 }).maxBackends).toBeLessThanOrEqual(64) + }) + + it('clamps idleMs to the shared ceiling (7 days)', () => { + expect(clampPoolLimits({ idleMs: 999_000_000 }).idleMs).toBe(POOL_LIMITS_BOUNDS.idleMsMax) + }) + + it('floors fractional values', () => { + expect(clampPoolLimits({ maxBackends: 2.9 }).maxBackends).toBe(2) + }) +}) diff --git a/apps/desktop/electron/pool-limits.ts b/apps/desktop/electron/pool-limits.ts new file mode 100644 index 0000000000..ee7e5b8a8a --- /dev/null +++ b/apps/desktop/electron/pool-limits.ts @@ -0,0 +1,81 @@ +/** + * Pool limits — how many bot backends may stay spawned, and how long an + * unused one survives. + * + * A device-local preference (each machine trades RAM against switching + * speed for itself), stored in userData like keep-awake. The main process + * is authoritative: it owns the pool AND the persisted copy, and applies a + * new max IMMEDIATELY by evicting least-recently-used idle backends — no + * app restart. The renderer mirrors the values for its UI and prewarm + * guard over IPC. + * + * Defaults preserve the historical hard-coded behavior (3 backends, 10min + * idle) so machines that never open Settings behave exactly as before. + */ + +export interface PoolLimits { + /** Max concurrently spawned non-primary profile backends. */ + maxBackends: number + /** Idle lifetime of an unused pool backend, in milliseconds. */ + idleMs: number +} + +export const POOL_LIMITS_DEFAULTS: PoolLimits = { + maxBackends: 3, + idleMs: 10 * 60_000 +} + +/** Hard floors — match the clamps the env-var path always applied. */ +export const POOL_LIMITS_MIN: PoolLimits = { + maxBackends: 1, + idleMs: 60_000 +} + +/** Shared bounds for both pool knobs — imported by the Settings UI so the + * advertised input ranges can never drift from what main actually clamps + * to. idleMs has no ceiling: a user who wants backends kept warm all week + * may have exactly that. */ +export const POOL_LIMITS_BOUNDS = { + maxBackendsMax: 64, + /** 7 days, matching the UI's suggestion ceiling. */ + idleMsMax: 7 * 24 * 60 * 60_000 +} as const + +const MAX_BACKENDS_CEILING = POOL_LIMITS_BOUNDS.maxBackendsMax +const IDLE_MS_CEILING = POOL_LIMITS_BOUNDS.idleMsMax + +/** Clamp a raw partial to the floors/ceilings; missing keys fall to defaults. */ +export function clampPoolLimits(raw: Partial): PoolLimits { + const maxBackends = Number.isFinite(raw.maxBackends) + ? Math.min(MAX_BACKENDS_CEILING, Math.max(POOL_LIMITS_MIN.maxBackends, Math.floor(Number(raw.maxBackends)))) + : POOL_LIMITS_DEFAULTS.maxBackends + + const idleMs = Number.isFinite(raw.idleMs) + ? Math.min(IDLE_MS_CEILING, Math.max(POOL_LIMITS_MIN.idleMs, Math.floor(Number(raw.idleMs)))) + : POOL_LIMITS_DEFAULTS.idleMs + + return { maxBackends, idleMs } +} + +function clampLimits(raw: Partial): PoolLimits { + return clampPoolLimits(raw) +} + +/** Parse + clamp a persisted JSON blob; anything unreadable falls back to + * defaults so a corrupted file can never wedge the pool. */ +export function parsePoolLimits(json: string | null | undefined): PoolLimits { + if (!json) { + return { ...POOL_LIMITS_DEFAULTS } + } + + try { + const parsed = JSON.parse(json) + + return clampLimits({ + maxBackends: typeof parsed?.maxBackends === 'number' ? parsed.maxBackends : undefined, + idleMs: typeof parsed?.idleMs === 'number' ? parsed.idleMs : undefined + }) + } catch { + return { ...POOL_LIMITS_DEFAULTS } + } +} diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index 8176ebfc07..fd9668752b 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -24,6 +24,8 @@ contextBridge.exposeInMainWorld('hermesDesktop', { getProfileRoutes: profiles => ipcRenderer.invoke('hermes:plugin-profile-routes', profiles), revalidateConnection: () => ipcRenderer.invoke('hermes:connection:revalidate'), touchBackend: profile => ipcRenderer.invoke('hermes:backend:touch', profile), + getPoolLimits: () => ipcRenderer.invoke('hermes:pool-limits:get'), + setPoolLimits: limits => ipcRenderer.invoke('hermes:pool-limits:set', limits), getGatewayWsUrl: profile => ipcRenderer.invoke('hermes:gateway:ws-url', profile), // Registry-scoped fresh WS URL: { connectionId, profile } → result shape of // getGatewayWsUrl, minted against that connection's backend. diff --git a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts index 29dd2c7a5a..f5ee2ca66c 100644 --- a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts +++ b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts @@ -46,6 +46,7 @@ import { } from '@/store/gateway-switch' import { checkLocalRuntimeUpdate, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs' import { notify, notifyError } from '@/store/notifications' +import { loadPoolLimits } from '@/store/pool-limits' import { $activeGatewayProfile, normalizeProfileKey, @@ -900,6 +901,10 @@ export function useGatewayBoot({ // this a socket dropped during sleep sits closed until the user clicks. window.addEventListener('focus', onFocus) + // Pool limits are main-process state; mirror them once for the Settings + // rows and prewarmProfileBackend's saturation guard. + void loadPoolLimits() + // Keep live pool backends alive while this window is open (the main process // can't observe the direct renderer↔backend WS). No-op for the primary. const keepaliveTimer = setInterval(() => { diff --git a/apps/desktop/src/app/settings/config-settings.tsx b/apps/desktop/src/app/settings/config-settings.tsx index 40a30662da..73cb0d205b 100644 --- a/apps/desktop/src/app/settings/config-settings.tsx +++ b/apps/desktop/src/app/settings/config-settings.tsx @@ -45,6 +45,7 @@ import { import { MemoryConnect } from './memory/connect' import { ProviderConfigPanel } from './memory/provider-config-panel' import { ModelSettings, ModelSettingsSkeleton } from './model-settings' +import { PoolLimitsSetting } from './pool-limits-setting' import { EmptyState, ListRow, SettingsContent, SettingsSkeleton, ToggleRow } from './primitives' import { SettingsProfileScope } from './profile-scope' import { QuickEntrySettings } from './quick-entry-settings' @@ -405,6 +406,7 @@ function ConfigSettingsInner({ label={c.disableF12Title} onChange={setDisableF12} /> + )} diff --git a/apps/desktop/src/app/settings/pool-limits-setting.tsx b/apps/desktop/src/app/settings/pool-limits-setting.tsx new file mode 100644 index 0000000000..4db5ee8824 --- /dev/null +++ b/apps/desktop/src/app/settings/pool-limits-setting.tsx @@ -0,0 +1,113 @@ +import { useStore } from '@nanostores/react' +import { useEffect, useState } from 'react' + +import { ListRow } from '@/app/settings/primitives' +import { Input } from '@/components/ui/input' +import { $poolLimits, loadPoolLimits, savePoolLimits } from '@/store/pool-limits' + +// Bounds imported from main's clamp module so the advertised input ranges +// can never drift from what the pool actually enforces (review note on #92581). +import { POOL_LIMITS_BOUNDS } from '../../../electron/pool-limits' + +const MAX_BACKENDS_MAX = POOL_LIMITS_BOUNDS.maxBackendsMax +const IDLE_MS_MAX = POOL_LIMITS_BOUNDS.idleMsMax + +/** Settings → Advanced: warm-bot-backends count + backend idle timeout. + * Device-local (not profile-scoped): the pool is sized once per machine and + * changes apply live — main evicts/reaps to converge without a restart. */ +export function PoolLimitsSetting() { + const limits = useStore($poolLimits) + const [maxDraft, setMaxDraft] = useState(String(limits.maxBackends)) + const [idleDraft, setIdleDraft] = useState(String(limits.idleMs)) + + useEffect(() => { + void loadPoolLimits() + }, []) + + useEffect(() => { + setMaxDraft(String(limits.maxBackends)) + setIdleDraft(String(limits.idleMs)) + }, [limits]) + + const commitMax = () => { + const parsed = Number(maxDraft) + + if (!Number.isFinite(parsed) || parsed === limits.maxBackends) { + setMaxDraft(String(limits.maxBackends)) + + return + } + + void savePoolLimits({ maxBackends: parsed }) + .then(() => undefined) + .catch(() => setMaxDraft(String($poolLimits.get().maxBackends))) + } + + const commitIdle = () => { + const parsed = Number(idleDraft) + + if (!Number.isFinite(parsed) || parsed === limits.idleMs) { + setIdleDraft(String(limits.idleMs)) + + return + } + + void savePoolLimits({ idleMs: parsed }) + .then(() => undefined) + .catch(() => setIdleDraft(String($poolLimits.get().idleMs))) + } + + return ( + <> + + setMaxDraft(event.target.value)} + onKeyDown={event => { + if (event.key === 'Enter') { + event.currentTarget.blur() + } + }} + type="number" + value={maxDraft} + /> + + } + description="How many bot backends stay running for instant switching. Higher = faster switches, more memory (~60MB per backend). Applies immediately." + title="Warm Bot Backends" + /> + + setIdleDraft(event.target.value)} + onKeyDown={event => { + if (event.key === 'Enter') { + event.currentTarget.blur() + } + }} + type="number" + value={idleDraft} + /> + ms + + } + description="How long an unused bot backend stays warm before it is shut down. Raise this so bots you revisit every few minutes never pay a cold start." + title="Backend Idle Timeout" + /> + + ) +} diff --git a/apps/desktop/src/global.d.ts b/apps/desktop/src/global.d.ts index 8067308df7..0362b8ff12 100644 --- a/apps/desktop/src/global.d.ts +++ b/apps/desktop/src/global.d.ts @@ -1,6 +1,8 @@ import type { GatewayWsUrlResult } from '@hermes/shared' import type { TranslucencyState } from '@hermes/shared/translucency' +import type { PoolLimits } from '../electron/pool-limits' + import type { WakeIndicatorState } from './lib/wake-indicator' import type { PetOverlayBounds, @@ -46,6 +48,14 @@ declare global { // Keepalive: mark a pool profile backend as recently used so the idle // reaper spares it while its chat is active. touchBackend: (profile?: string | null) => Promise<{ ok: boolean }> + // Pool sizing (Settings → Advanced): device-local, live-applied by the + // main process. get resolves the limits currently in force; set applies + // (and persists) new ones, evicting/reaping to converge immediately. + getPoolLimits: () => Promise + setPoolLimits: (limits: { maxBackends?: number; idleMs?: number }) => Promise<{ + ok: boolean + limits: PoolLimits + }> getGatewayWsUrl: (profile?: null | string) => Promise // Open (or focus) a standalone OS window for a single chat session so // the user can work with multiple chats side by side. Returns ok:false diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index b07971d13f..5bf36abaea 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -1584,6 +1584,24 @@ export function reconnectSecondaryGateways({ forceOpenSockets = false }: { force } } +// How many non-primary backends currently hold an open socket. Hover-intent +// prewarming consults this before spawning: a speculative spawn that pushes +// the pool past its cap causes the Electron main to LRU-evict a warm backend +// — often one the user is about to click — turning the prewarm into churn +// (the #91545 evict/respawn cascade). The active gateway's backend is +// primary-routed and never counts toward the pool cap. +export function openSecondaryCount(): number { + let count = 0 + + for (const entry of g.secondaries.values()) { + if (isOpen(entry.gateway)) { + count += 1 + } + } + + return count +} + // Keep the idle reaper from killing a backend we still need: ping every live // secondary. The active one is pinged separately (touchActiveGatewayBackend). export function touchSecondaryGateways(): void { diff --git a/apps/desktop/src/store/pool-limits.ts b/apps/desktop/src/store/pool-limits.ts new file mode 100644 index 0000000000..71bd9e4ade --- /dev/null +++ b/apps/desktop/src/store/pool-limits.ts @@ -0,0 +1,65 @@ +/** + * Pool limits — how many bot backends may stay spawned, and how long an + * unused one survives before it is shut down. + * + * A device-local preference (each machine trades RAM against switching + * speed for itself). The MAIN process is authoritative: it owns the pool + * and the persisted copy, and applies a new max immediately by evicting + * least-recently-used idle backends — no restart. This store mirrors the + * live values for the Settings rows and feeds prewarmProfileBackend's + * saturation guard. + */ + +import { atom } from 'nanostores' + +export interface PoolLimits { + /** Max concurrently spawned non-primary profile backends. */ + maxBackends: number + /** Idle lifetime of an unused pool backend, in milliseconds. */ + idleMs: number +} + +export const POOL_LIMITS_DEFAULTS: PoolLimits = { + maxBackends: 3, + idleMs: 10 * 60_000 +} + +export const $poolLimits = atom({ ...POOL_LIMITS_DEFAULTS }) + +/** Seed from main's authoritative state once at startup; no-op without the + * bridge (web/older builds just keep the defaults for the UI). */ +export async function loadPoolLimits(): Promise { + try { + const limits = await window.hermesDesktop?.getPoolLimits?.() + + if (limits) { + $poolLimits.set(limits) + } + } catch { + // Keep defaults — Settings rows still render and can retry on save. + } +} + +/** Push new limits to main; adopt the post-clamp values it reports. */ +export async function savePoolLimits(next: { maxBackends?: number; idleMs?: number }): Promise { + const current = $poolLimits.get() + + const optimistic: PoolLimits = { + maxBackends: next.maxBackends ?? current.maxBackends, + idleMs: next.idleMs ?? current.idleMs + } + + // Optimistic paint, then honest reconciliation with the clamped result. + $poolLimits.set(optimistic) + + try { + const result = await window.hermesDesktop?.setPoolLimits?.(next) + + if (result?.limits) { + $poolLimits.set(result.limits) + } + } catch { + $poolLimits.set(current) + throw new Error('Applying pool limits failed') + } +} diff --git a/apps/desktop/src/store/profile.test.ts b/apps/desktop/src/store/profile.test.ts index e6825de85a..33c9578427 100644 --- a/apps/desktop/src/store/profile.test.ts +++ b/apps/desktop/src/store/profile.test.ts @@ -9,10 +9,24 @@ import type { ProfileInfo } from '@/types/hermes' const ensureGatewayForProfile = vi.fn(async () => undefined) const ensureGatewayForAgent = vi.fn(async () => undefined) const openGatewayForProfile = vi.fn(async (_profile: string) => undefined) +const openSecondaryCount = vi.fn(() => 0) const $gateway = atom({ id: 'live-socket', connectionState: 'open' }) const resetStarmapGraph = vi.fn() -vi.mock('@/store/gateway', () => ({ $gateway, ensureGatewayForAgent, ensureGatewayForProfile, openGatewayForProfile })) +vi.mock('@/store/gateway', () => ({ + $gateway, + ensureGatewayForAgent, + ensureGatewayForProfile, + openGatewayForProfile, + openSecondaryCount +})) +// The pool-limits atom is profile.ts's live saturation signal — keep the real +// one so tests can move the cap via the store, but stub its IPC bridge. +vi.mock('@/store/pool-limits', async () => { + const { atom } = await import('nanostores') + + return { $poolLimits: atom({ idleMs: 600_000, maxBackends: 3 }) } +}) vi.mock('@/hermes', () => ({ getProfiles: vi.fn(async () => ({ profiles: [] })), setApiRequestProfile: vi.fn() @@ -29,6 +43,8 @@ const { refreshProfiles } = await import('./profile') +const { $poolLimits } = await import('@/store/pool-limits') + const { $connection } = await import('./session') const { invalidateProfileScopedQueries } = await import('@/lib/query-client') const { getProfiles } = await import('@/hermes') @@ -55,6 +71,7 @@ beforeEach(() => { getConnection.mockReset() ensureGatewayForProfile.mockClear() openGatewayForProfile.mockClear() + openSecondaryCount.mockReturnValue(0) $gateway.set({ id: 'live-socket', connectionState: 'open' }) $activeGatewayProfile.set('default') $connection.set(localConn()) @@ -169,6 +186,43 @@ describe('prewarmProfileBackend (hover-intent pool spawn)', () => { expect(() => prewarmProfileBackend('warm-failing')).not.toThrow() }) + + it('skips pre-warm when the pool is saturated (#91545 evict/respawn cascade)', () => { + // Every pool slot occupied: a speculative spawn would LRU-evict a warm + // backend — often the one the user is about to click. Default limit 3, + // 3 open secondaries → the next spawn would exceed the cap. + openSecondaryCount.mockReturnValue(3) + + prewarmProfileBackend('warm-saturated') + + expect(openGatewayForProfile).not.toHaveBeenCalled() + }) + + it('pre-warms while pool slots are free', () => { + openSecondaryCount.mockReturnValue(1) + + prewarmProfileBackend('warm-slot-free') + + expect(openGatewayForProfile).toHaveBeenCalledWith('warm-slot-free') + }) + + it('follows the live pool-limit atom, not a hard-coded cap', () => { + // User raises Warm Bot Backends to 8 in Settings: prewarming must keep + // working well past the old default of 3. + openSecondaryCount.mockReturnValue(5) + $poolLimits.set({ idleMs: 600_000, maxBackends: 8 }) + + prewarmProfileBackend('warm-raised-cap') + + expect(openGatewayForProfile).toHaveBeenCalledWith('warm-raised-cap') + + // And lowering the cap re-engages the guard at the new boundary. + $poolLimits.set({ idleMs: 600_000, maxBackends: 2 }) + + prewarmProfileBackend('warm-lowered-cap') + + expect(openGatewayForProfile).not.toHaveBeenCalledWith('warm-lowered-cap') + }) }) describe('refreshProfiles shared rail list (#49289)', () => { diff --git a/apps/desktop/src/store/profile.ts b/apps/desktop/src/store/profile.ts index 9c9fdaf29a..cf903e120e 100644 --- a/apps/desktop/src/store/profile.ts +++ b/apps/desktop/src/store/profile.ts @@ -21,9 +21,11 @@ import { ensureGatewayForAgent, ensureGatewayForProfile, openGatewayForAgent, - openGatewayForProfile + openGatewayForProfile, + openSecondaryCount } from '@/store/gateway' import { notifyError } from '@/store/notifications' +import { $poolLimits } from '@/store/pool-limits' import { notifyRemoteOverrideAuthFailure } from '@/store/profile-remote-override' import { clearComposerSelectionOwner, setComposerSelectionOwner, setConnection } from '@/store/session' import type { SessionOwnerRoute } from '@/store/session-request-router' @@ -423,6 +425,17 @@ export function prewarmProfileBackend(name: string): void { return } + // Prewarm/cap harmony (#91545): the pool caps spawned backends at the + // configured max, and a spawn over the cap LRU-evicts the warmest idle + // backend. A hover sweep across the rail therefore evicted backends for + // profiles the user was about to click — prewarming caused the exact churn + // it exists to prevent. Skip speculative spawns once every pool slot is + // occupied by an open socket; the real click still spawns on demand, it + // just doesn't get a head start. + if (openSecondaryCount() + 1 > $poolLimits.get().maxBackends) { + return + } + prewarmedAt.set(key, now) openGatewayForProfile(key).catch(() => undefined) }