43e67d872f
Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
198 lines
6.7 KiB
TypeScript
198 lines
6.7 KiB
TypeScript
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
|
|
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
|
|
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'
|
|
|
|
import { DropdownMenu, DropdownMenuContent } from '@/components/ui/dropdown-menu'
|
|
import { $localModelsEnabled } from '@/store/local-models-flag'
|
|
import { $localRuntimeJobs } from '@/store/local-runtime-jobs'
|
|
import {
|
|
$modelVisibilityOpen,
|
|
$visibleModels,
|
|
modelVisibilityKey,
|
|
setModelVisibilityOpen,
|
|
setVisibleModels
|
|
} from '@/store/model-visibility'
|
|
import type { LocalRuntimeJob } from '@/types/hermes'
|
|
|
|
import { ModelCatalogMenu, type ModelMenuController } from './model-catalog-menu'
|
|
|
|
// Radix calls these on open; jsdom doesn't implement them.
|
|
beforeAll(() => {
|
|
Element.prototype.scrollIntoView = vi.fn()
|
|
Element.prototype.hasPointerCapture = vi.fn(() => false)
|
|
Element.prototype.releasePointerCapture = vi.fn()
|
|
})
|
|
|
|
const getGlobalModelOptions = vi.fn()
|
|
|
|
vi.mock('@/hermes', () => ({
|
|
getGlobalModelOptions: (...args: unknown[]) => getGlobalModelOptions(...args),
|
|
// The menu kicks the app-level job poller on mount; echo the store so a
|
|
// poll can't wipe the jobs a test staged (the real backend is authority,
|
|
// and here the store plays that part).
|
|
getLocalModelsJobs: vi.fn(async () => {
|
|
const { $localRuntimeJobs } = await import('@/store/local-runtime-jobs')
|
|
|
|
return { jobs: [...$localRuntimeJobs.get()] }
|
|
}),
|
|
getLocalModelsStatus: vi.fn().mockResolvedValue({ loading: {} }),
|
|
setApiRequestProfile: vi.fn()
|
|
}))
|
|
|
|
beforeEach(() => {
|
|
$visibleModels.set(null)
|
|
$localRuntimeJobs.set([])
|
|
// These suites exercise the local-models rows, which ship behind --local.
|
|
$localModelsEnabled.set(true)
|
|
setModelVisibilityOpen(false)
|
|
getGlobalModelOptions.mockResolvedValue({
|
|
providers: [{ models: ['gemini-3.1-pro', 'gemini-2.5-flash'], name: 'Google', slug: 'google' }]
|
|
})
|
|
})
|
|
|
|
afterEach(() => {
|
|
cleanup()
|
|
vi.clearAllMocks()
|
|
})
|
|
|
|
// A minimal controller — these tests are about the CATALOG's own behaviour
|
|
// (what it lists, what it offers), not about what any host does with a pick.
|
|
function renderMenu() {
|
|
const select = vi.fn()
|
|
|
|
const controller: ModelMenuController = {
|
|
applyPreset: vi.fn(),
|
|
current: { effort: '', fast: false, model: '', provider: '' },
|
|
presetFor: () => ({}),
|
|
select,
|
|
setOptions: vi.fn()
|
|
}
|
|
|
|
const client = new QueryClient({ defaultOptions: { queries: { retry: false } } })
|
|
|
|
render(
|
|
<QueryClientProvider client={client}>
|
|
<DropdownMenu open>
|
|
<DropdownMenuContent>
|
|
<ModelCatalogMenu controller={controller} />
|
|
</DropdownMenuContent>
|
|
</DropdownMenu>
|
|
</QueryClientProvider>
|
|
)
|
|
|
|
return select
|
|
}
|
|
|
|
// Curation is ONE global preference, so it belongs to the catalog rather than
|
|
// to whichever surface mounted it. If a host had to opt in, the composer and
|
|
// the kanban board would end up disagreeing about what "my models" means —
|
|
// which is exactly the drift extracting this component was meant to prevent.
|
|
describe('the catalog owns model curation', () => {
|
|
it('honours the stored Edit Models shortlist', async () => {
|
|
setVisibleModels(new Set([modelVisibilityKey('google', 'gemini-2.5-flash')]))
|
|
|
|
renderMenu()
|
|
|
|
await screen.findByText(/Gemini 2\.5 Flash/i)
|
|
expect(screen.queryByText(/Gemini 3\.1 Pro/i)).toBeNull()
|
|
})
|
|
|
|
it('still finds a hidden model by search — curation narrows the default view, not the catalog', async () => {
|
|
setVisibleModels(new Set([modelVisibilityKey('google', 'gemini-2.5-flash')]))
|
|
|
|
renderMenu()
|
|
await screen.findByText(/Gemini 2\.5 Flash/i)
|
|
|
|
const input = screen.getByRole('textbox', { name: 'Search models' })
|
|
|
|
fireEvent.change(input, { target: { value: 'gemini-3.1' } })
|
|
|
|
await vi.waitFor(() => {
|
|
expect(screen.queryByText(/Gemini 3\.1 Pro/i)).not.toBeNull()
|
|
})
|
|
})
|
|
|
|
it('offers Edit Models without the host wiring it up', async () => {
|
|
renderMenu()
|
|
await screen.findByText(/Gemini 3\.1 Pro/i)
|
|
|
|
fireEvent.click(screen.getByText('Edit models…'))
|
|
|
|
expect($modelVisibilityOpen.get()).toBe(true)
|
|
})
|
|
})
|
|
|
|
describe('in-flight local downloads', () => {
|
|
const DOWNLOAD_JOB: LocalRuntimeJob = {
|
|
job_id: 'dl1',
|
|
kind: 'model-download',
|
|
target: 'Qwen3.8 Flash Next (UD-Q4_K_XL)',
|
|
model_id: 'qwen3.8-flash-next',
|
|
status: 'running',
|
|
phase: 'downloading',
|
|
detail: '',
|
|
total_bytes: 100,
|
|
done_bytes: 41,
|
|
percent: 41,
|
|
error: null
|
|
}
|
|
|
|
it('shows a downloading model as a disabled progress row in its own Local group', async () => {
|
|
// No llamacpp provider in the catalog (first-ever download).
|
|
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
|
renderMenu()
|
|
await screen.findByText(/Gemini 3\.1 Pro/i)
|
|
|
|
const row = screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
|
|
|
|
expect(row).toBeTruthy()
|
|
expect(screen.getByText('41%')).toBeTruthy()
|
|
expect(row.closest('[role="menuitem"]')?.getAttribute('aria-disabled')).toBe('true')
|
|
})
|
|
|
|
it('shows the download inside the Local provider group when it exists', async () => {
|
|
getGlobalModelOptions.mockResolvedValue({
|
|
providers: [
|
|
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
|
|
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
|
|
]
|
|
})
|
|
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
|
renderMenu()
|
|
|
|
await screen.findByText(/Qwen3\.6 27B/i)
|
|
expect(screen.getByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeTruthy()
|
|
// One Local heading — the trailing fallback group must not double up.
|
|
expect(screen.getAllByText('Local').length).toBe(1)
|
|
})
|
|
|
|
it('drops the placeholder row once the download settles', async () => {
|
|
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
|
renderMenu()
|
|
await screen.findByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')
|
|
|
|
$localRuntimeJobs.set([{ ...DOWNLOAD_JOB, status: 'done', phase: 'done' }])
|
|
await waitFor(() => {
|
|
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
|
|
})
|
|
})
|
|
|
|
it('hides the local provider group and download rows without the --local flag (strict)', async () => {
|
|
$localModelsEnabled.set(false)
|
|
getGlobalModelOptions.mockResolvedValue({
|
|
providers: [
|
|
{ models: ['Qwen3.6-27B-UD-Q4_K_XL'], name: 'Local', slug: 'llamacpp' },
|
|
{ models: ['gemini-3.1-pro'], name: 'Google', slug: 'google' }
|
|
]
|
|
})
|
|
$localRuntimeJobs.set([DOWNLOAD_JOB])
|
|
renderMenu()
|
|
|
|
// Staged models exist and a download is running — none of it shows.
|
|
await screen.findByText(/Gemini 3\.1 Pro/i)
|
|
expect(screen.queryByText(/Qwen3\.6 27B/i)).toBeNull()
|
|
expect(screen.queryByText('Qwen3.8 Flash Next (UD-Q4_K_XL)')).toBeNull()
|
|
expect(screen.queryByText('Local')).toBeNull()
|
|
})
|
|
})
|