diff --git a/agent/context_compressor.py b/agent/context_compressor.py index d66bee1cad..03639cf985 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1885,6 +1885,7 @@ class ContextCompressor(ContextEngine): self._cooldown_persist_failed = False self._last_summary_error = None self._last_compress_aborted = False + self._last_compress_refused_would_grow = False self.last_real_prompt_tokens = 0 self.last_compression_rough_tokens = 0 self.last_rough_tokens_when_real_prompt_fit = 0 @@ -2167,6 +2168,7 @@ class ContextCompressor(ContextEngine): self._summary_failure_cooldown_until = 0.0 self._cooldown_persist_failed = False self._last_compress_aborted = False + self._last_compress_refused_would_grow = False self._context_probed = False self._context_probe_persistable = False self.last_real_prompt_tokens = 0 @@ -6927,6 +6929,7 @@ This compaction should PRIORITISE preserving all information related to the focu self._last_aux_model_failure_error = None self._last_aux_model_failure_model = None self._last_compress_aborted = False + self._last_compress_refused_would_grow = False self._last_compression_made_progress = False # NOTE: do NOT reset _last_summary_auth_failure or # _last_summary_network_failure here. These flags are set by diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 97a9e3dbf2..ba142416b3 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3395,6 +3395,17 @@ def compress_context( f"{_rough_in:,}", f"{_rough_out:,}", ) + # Flag the refusal on the compressor state so manual + # /compress feedback can report it honestly. Without this, + # the CLI compared the returned list against its pre-call + # snapshot, saw a difference (durable-snapshot adoption can + # legitimately change the count), and printed + # "✅ Compressed: 8 → 14 messages" directly under the + # refusal warning (Aug 2026 full-surface CLI QA sweep). + try: + agent.context_compressor._last_compress_refused_would_grow = True + except Exception: + pass try: agent._emit_warning( "⚠️ Compression refused: the generated summary " diff --git a/agent/manual_compression_feedback.py b/agent/manual_compression_feedback.py index b2e12d6834..b37361e6e2 100644 --- a/agent/manual_compression_feedback.py +++ b/agent/manual_compression_feedback.py @@ -53,6 +53,11 @@ def summarize_manual_compression( compression_state is not None and getattr(compression_state, "_last_compress_aborted", False) is True ) + refused_would_grow = ( + compression_state is not None + and getattr(compression_state, "_last_compress_refused_would_grow", False) + is True + ) fallback_used = ( compression_state is not None and getattr(compression_state, "_last_summary_fallback_used", False) is True @@ -65,7 +70,12 @@ def summarize_manual_compression( if not isinstance(failure_reason, str) or not failure_reason.strip(): failure_reason = None - if aborted: + if refused_would_grow: + headline = ( + f"Compression refused (summary would grow the conversation): " + f"{before_count} messages preserved" + ) + elif aborted: headline = f"Compression aborted: {before_count} messages preserved" elif fallback_used: headline = ( @@ -78,6 +88,8 @@ def summarize_manual_compression( if noop and after_tokens == before_tokens: token_line = f"Approx request size: ~{before_tokens:,} tokens (unchanged)" + elif refused_would_grow: + token_line = f"Approx request size: ~{before_tokens:,} tokens (unchanged)" else: token_line = ( f"Approx request size: ~{before_tokens:,} → " @@ -85,7 +97,12 @@ def summarize_manual_compression( ) note = None - if aborted: + if refused_would_grow: + note = ( + "The generated summary was larger than what it would replace; " + "no messages were removed." + ) + elif aborted: note = "Summary generation failed; no messages were removed." elif fallback_used: dropped_count = getattr( @@ -113,6 +130,7 @@ def summarize_manual_compression( return { "noop": noop, "aborted": aborted, + "refused_would_grow": refused_would_grow, "fallback_used": fallback_used, "headline": headline, "token_line": token_line, diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index ed1533748b..4f26ffcf2a 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -191,6 +191,21 @@ MEMORY_GUIDANCE = ( "workflows belong in skills, not memory." ) +USER_PROFILE_GUIDANCE = ( + "You have a persistent user profile across sessions. Save durable facts about " + "the user with the memory tool (target='user'): name, role, preferences, " + "corrections, and communication style. The profile is injected into every turn, " + "so keep it compact and focused on facts that will still matter later.\n" + "The built-in memory notes store is disabled — write only to the user profile " + "(target='user'), never target='memory'.\n" + "Prioritize what reduces future user steering — the most valuable entry is one " + "that prevents the user from having to correct or remind you again.\n" + "Write entries as declarative facts, not instructions to yourself. " + "'User prefers concise responses' ✓ — 'Always respond concisely' ✗. " + "Imperative phrasing gets re-read as a directive in later sessions and can " + "cause repeated work or override the user's current request." +) + SESSION_SEARCH_GUIDANCE = ( "When the user references something from a past conversation or you suspect " "relevant cross-session context exists, use session_search to recall it before " diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index 231f36fea3..48b44a25ee 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -36,8 +36,14 @@ Rules for call sites: from __future__ import annotations +import re from typing import Optional, Sequence +#: K3 slug detector — matches ``k3`` as a delimited token (``k3``, +#: ``k3-256k``, ``kimi-k3``, ``kimi-k3-cot``) without matching K2-era names +#: (``kimi-k2.6``). From #76427 by @ruizanthony. +_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)") + # Canonical low→high ordering used for nearest-level clamping. Superset of # hermes_constants.VALID_REASONING_EFFORTS ("none" included so an explicit # disable can be clamped too when a provider publishes it as a level). @@ -124,11 +130,14 @@ SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high") def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]: """Supported effort set for a Moonshot/Kimi model slug. - K3 is served as the bare slug ``k3`` and the ``kimi-k3*`` aliases; its - documented set is low/high/max. Everything earlier speaks low/medium/high. + K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``, + and the ``kimi-k3*`` aliases; its documented set is low/high/max. + Everything earlier speaks low/medium/high. Boundary-matched so K2-era + names (``kimi-k2.6``) never match (detection regex from #76427 by + @ruizanthony). """ m = (model or "").strip().lower().split("/")[-1] - if m == "k3" or m.startswith("kimi-k3"): + if _KIMI_K3_SLUG_RE.search(m): return KIMI_K3_EFFORTS return KIMI_K2_EFFORTS diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 7d5bba76b3..b176a2554f 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -38,6 +38,7 @@ from agent.prompt_builder import ( HERMES_AGENT_HELP_GUIDANCE, KANBAN_GUIDANCE, MEMORY_GUIDANCE, + USER_PROFILE_GUIDANCE, OPENAI_MODEL_EXECUTION_GUIDANCE, PARALLEL_TOOL_CALL_GUIDANCE, PLATFORM_HINTS, @@ -415,8 +416,21 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) # Tool-aware behavioral guidance: only inject when the tools are loaded tool_guidance = [] + # MEMORY_GUIDANCE instructs the model to save facts to the built-in + # MEMORY.md/USER.md stores. With both disabled in config no store is built, + # so the guidance would steer the model at a tool whose every call returns + # "Memory is not available". Defaults to True for the rare code paths that + # build an agent view without going through agent_init. + # When only the user profile store is enabled, the narrower + # USER_PROFILE_GUIDANCE is injected instead — the full block instructs the + # model to write notes to a MEMORY.md store that does not exist. + _mem_enabled = getattr(agent, "_memory_enabled", True) + _profile_enabled = getattr(agent, "_user_profile_enabled", True) if "memory" in agent.valid_tool_names: - tool_guidance.append(MEMORY_GUIDANCE) + if _mem_enabled: + tool_guidance.append(MEMORY_GUIDANCE) + elif _profile_enabled: + tool_guidance.append(USER_PROFILE_GUIDANCE) if "session_search" in agent.valid_tool_names: tool_guidance.append(SESSION_SEARCH_GUIDANCE) if "skill_manage" in agent.valid_tool_names: diff --git a/apps/desktop/DESIGN.md b/apps/desktop/DESIGN.md index 485073200d..8d4512e65f 100644 --- a/apps/desktop/DESIGN.md +++ b/apps/desktop/DESIGN.md @@ -151,6 +151,12 @@ Notes: - SVGs inherit `size-3.5` (`size-3` at `xs`). Don't re-set icon size. - Polymorph with `asChild` when the button must render as a link/Slot. +## Badges — one component + +`src/components/ui/badge.tsx`. Variants: `default` (tinted primary), `muted`, +`warn`, `destructive`, `outline`, `solid` (primary fill — icon-corner counts). +Sizes: `default`, `xs`, `overlay` (titlebar glyph counts). + ## Form controls - **`controlVariants`** (`src/components/ui/control.ts`) is the shared shape for diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 91b3224f5c..85bec53638 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -4886,9 +4886,7 @@ function downloadViaTokenToFile(url, token, ctx, options: any = {}) { parsed, { method: 'GET', - headers: options.bearer - ? { Authorization: `Bearer ${options.bearer}` } - : { 'X-Hermes-Session-Token': token } + headers: options.bearer ? { Authorization: `Bearer ${options.bearer}` } : { 'X-Hermes-Session-Token': token } }, res => { // Headers arrived — the connection phase is done. Drop the idle timeout diff --git a/apps/desktop/src/app/shell/titlebar-controls.tsx b/apps/desktop/src/app/shell/titlebar-controls.tsx index 72e297d9ac..33d8d4e235 100644 --- a/apps/desktop/src/app/shell/titlebar-controls.tsx +++ b/apps/desktop/src/app/shell/titlebar-controls.tsx @@ -5,9 +5,11 @@ import { useLocation, useNavigate } from 'react-router' import { hudTargetSessionId } from '@/app/hud/handoff' import { toggleLayoutEditMode } from '@/components/pane-shell/edit-mode' import { resetLayoutTree } from '@/components/pane-shell/tree/store' +import { Badge } from '@/components/ui/badge' import { Button } from '@/components/ui/button' import { Tip, TipKeybindLabel } from '@/components/ui/tooltip' import { useI18n } from '@/i18n' +import { compactNumber } from '@/lib/format' import { triggerHaptic } from '@/lib/haptics' import { formatModifierToken } from '@/lib/keybinds/combo' import { cn } from '@/lib/utils' @@ -15,11 +17,13 @@ import { $hapticsMuted, toggleHapticsMuted } from '@/store/haptics' import { toggleHud } from '@/store/hud' import { $fileBrowserOpen, + $panesFlipped, $sidebarOpen, toggleFileBrowserOpen, togglePanesFlipped, toggleSidebarOpen } from '@/store/layout' +import { $unreadSessionCount } from '@/store/session-dot-state' import { appViewForPath, isOverlayView } from '../routes' @@ -43,6 +47,8 @@ export interface TitlebarTool { onSelect?: (event?: MouseEvent) => void /** Keybind action id — when set, the tooltip shows the label + keybind hint. */ actionId?: string + /** Overlay count on the glyph (unread sessions). Hidden when 0/undefined. */ + badge?: number title?: string to?: string } @@ -79,6 +85,24 @@ function LayoutGlyph({ modHeld }: { modHeld: boolean }) { ) } +/** Overlay count on a titlebar glyph. Hidden when count is 0/undefined. */ +function withCountBadge(icon: ReactNode, count: number | undefined): ReactNode { + if (!count) { + return icon + } + + return ( + + {icon} + + + {compactNumber(count)} + + + + ) +} + /** Live ⌘/Ctrl tracking — mod-click affordances telegraph themselves (the * layout button morphs into its reset form while the modifier is down). */ function useModifierHeld(): boolean { @@ -109,7 +133,11 @@ export function TitlebarControls({ leftTools = [], tools = [], onOpenSettings }: const modHeld = useModifierHeld() const hapticsMuted = useStore($hapticsMuted) const fileBrowserOpen = useStore($fileBrowserOpen) + const panesFlipped = useStore($panesFlipped) const sidebarOpen = useStore($sidebarOpen) + const unreadCount = useStore($unreadSessionCount) + const unreadBadge = unreadCount > 0 ? unreadCount : undefined + const unreadHint = unreadBadge ? ` · ${t.titlebar.unreadSessions(unreadBadge)}` : '' const toggleHaptics = () => { if (!hapticsMuted) { @@ -130,13 +158,16 @@ export function TitlebarControls({ leftTools = [], tools = [], onOpenSettings }: // show/hide affordances. const leftEdge = { open: sidebarOpen, toggle: toggleSidebarOpen } const rightEdge = { open: fileBrowserOpen, toggle: toggleFileBrowserOpen } + const leftLabel = leftEdge.open ? t.titlebar.hideSidebar : t.titlebar.showSidebar + const rightLabel = rightEdge.open ? t.titlebar.hideRightSidebar : t.titlebar.showRightSidebar const leftToolbarTools: TitlebarTool[] = [ { actionId: 'view.toggleSidebar', + badge: panesFlipped ? undefined : unreadBadge, icon: , id: 'sidebar', - label: leftEdge.open ? t.titlebar.hideSidebar : t.titlebar.showSidebar, + label: `${leftLabel}${panesFlipped ? '' : unreadHint}`, onSelect: () => { triggerHaptic('tap') leftEdge.toggle() @@ -157,9 +188,10 @@ export function TitlebarControls({ leftTools = [], tools = [], onOpenSettings }: const rightSidebarTool: TitlebarTool = { actionId: 'view.toggleRightSidebar', + badge: panesFlipped ? unreadBadge : undefined, icon: , id: 'right-sidebar', - label: rightEdge.open ? t.titlebar.hideRightSidebar : t.titlebar.showRightSidebar, + label: `${rightLabel}${panesFlipped ? unreadHint : ''}`, onSelect: () => { triggerHaptic('tap') rightEdge.toggle() @@ -306,7 +338,7 @@ function TitlebarToolButton({ navigate, tool }: { navigate: ReturnType - {tool.icon} + {withCountBadge(tool.icon, tool.badge)} @@ -332,7 +364,7 @@ function TitlebarToolButton({ navigate, tool }: { navigate: ReturnType - {tool.icon} + {withCountBadge(tool.icon, tool.badge)} ) diff --git a/apps/desktop/src/components/ui/badge.tsx b/apps/desktop/src/components/ui/badge.tsx index c4e46e5236..d3a8a50449 100644 --- a/apps/desktop/src/components/ui/badge.tsx +++ b/apps/desktop/src/components/ui/badge.tsx @@ -15,11 +15,14 @@ const badgeVariants = cva( muted: 'bg-muted text-muted-foreground', warn: 'bg-amber-500/10 text-amber-600 dark:text-amber-300', destructive: 'bg-destructive/10 text-destructive', - outline: 'border border-(--ui-stroke-secondary) text-muted-foreground' + outline: 'border border-(--ui-stroke-secondary) text-muted-foreground', + // Solid fill — icon-corner counts (titlebar unread, etc.). + solid: 'bg-primary text-primary-foreground' }, size: { default: 'px-1.5 py-0.5 text-[0.65rem] [&_svg]:size-3', - xs: 'px-1 py-px text-[0.6rem] [&_svg]:size-2.5' + xs: 'px-1 py-px text-[0.6rem] [&_svg]:size-2.5', + overlay: 'h-2 min-w-2 justify-center rounded-[2px] px-px text-[7px] font-semibold tabular-nums' } }, defaultVariants: { variant: 'default', size: 'default' } diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 5ed5db5686..2280ab4a4c 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -181,6 +181,7 @@ export const ar = defineLocale({ swapSidebarSides: 'تبديل جانبي الأشرطة', hideRightSidebar: 'إخفاء الشريط الأيمن', showRightSidebar: 'إظهار الشريط الأيمن', + unreadSessions: count => (count === 1 ? 'جلسة واحدة غير مقروءة' : `${count} جلسات غير مقروءة`), muteHaptics: 'كتم الاهتزازات', unmuteHaptics: 'تفعيل الاهتزازات', openSettings: 'فتح الإعدادات', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 5e94e4c4f0..f5f0664aa4 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -211,6 +211,7 @@ export const en: Translations = { swapSidebarSides: 'Swap sidebar sides', hideRightSidebar: 'Hide right sidebar', showRightSidebar: 'Show right sidebar', + unreadSessions: count => (count === 1 ? '1 unread session' : `${count} unread sessions`), muteHaptics: 'Mute haptics', unmuteHaptics: 'Unmute haptics', openSettings: 'Open settings', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 7864578edc..034dc6a631 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -212,6 +212,7 @@ export const ja = defineLocale({ swapSidebarSides: 'サイドバーの向きを切り替え', hideRightSidebar: '右サイドバーを非表示', showRightSidebar: '右サイドバーを表示', + unreadSessions: count => (count === 1 ? '未読セッション 1 件' : `未読セッション ${count} 件`), muteHaptics: '触覚フィードバックをオフ', unmuteHaptics: '触覚フィードバックをオン', openSettings: '設定を開く', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index afd3fda150..dcba755c64 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -253,6 +253,7 @@ export interface Translations { swapSidebarSides: string hideRightSidebar: string showRightSidebar: string + unreadSessions: (count: number) => string muteHaptics: string unmuteHaptics: string openSettings: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 23a3232625..c68b3e2f73 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -206,6 +206,7 @@ export const zhHant = defineLocale({ swapSidebarSides: '交換側邊欄位置', hideRightSidebar: '隱藏右側邊欄', showRightSidebar: '顯示右側邊欄', + unreadSessions: count => (count === 1 ? '1 個未讀工作階段' : `${count} 個未讀工作階段`), muteHaptics: '靜音觸感回饋', unmuteHaptics: '開啟觸感回饋', openSettings: '開啟設定', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 613679cd72..44620ac17e 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -206,6 +206,7 @@ export const zh: Translations = { swapSidebarSides: '交换侧边栏位置', hideRightSidebar: '隐藏右侧栏', showRightSidebar: '显示右侧栏', + unreadSessions: count => (count === 1 ? '1 个未读会话' : `${count} 个未读会话`), muteHaptics: '关闭触感反馈', unmuteHaptics: '开启触感反馈', openSettings: '打开设置', diff --git a/apps/desktop/src/lib/svg-image.test.ts b/apps/desktop/src/lib/svg-image.test.ts index 03e0fc33c9..2e81a5958c 100644 --- a/apps/desktop/src/lib/svg-image.test.ts +++ b/apps/desktop/src/lib/svg-image.test.ts @@ -43,8 +43,8 @@ describe('normalizeSvgSize', () => { }) it('leaves an explicit pixel height when only width is a percentage', () => { - const svg = - '' + const svg = '' + const out = normalizeSvgSize(svg) expect(out).toContain('width="500"') diff --git a/apps/desktop/src/store/projects.test.ts b/apps/desktop/src/store/projects.test.ts index d91882032d..5e0e447343 100644 --- a/apps/desktop/src/store/projects.test.ts +++ b/apps/desktop/src/store/projects.test.ts @@ -3,7 +3,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { NO_PROJECT_ID, type SidebarProjectTree } from '@/app/chat/sidebar/projects/workspace-groups' import { $sidebarAgentsGrouped, setSidebarAgentsGrouped } from '@/store/layout' -import { $activeGatewayProfile } from '@/store/profile' +import { $activeGatewayProfile, setShowAllProfiles } from '@/store/profile' import { $currentCwd, $selectedStoredSessionId, $sessions, applyConfiguredDefaultProjectDir } from '@/store/session' import { @@ -21,6 +21,7 @@ import { endSessionMutation, enterProject, exitProjectScope, + fetchProjectSessions, openProjectCreate, pickProjectFolder, projectIdForCwd, @@ -63,6 +64,7 @@ vi.mock('@/lib/desktop-git', async importOriginal => ({ vi.mock('@/hermes', () => ({ getHermesConfig: vi.fn(), getProfiles: vi.fn(), + hermesApi: vi.fn(), setApiRequestProfile: vi.fn(), STARTUP_REQUEST_TIMEOUT_MS: 1000 })) @@ -84,6 +86,16 @@ const getHermesConfig = vi.mocked(hermes.getHermesConfig) const notifications = await import('@/store/notifications') const notify = vi.mocked(notifications.notify) +function deferred() { + let resolve!: (value: T) => void + + const promise = new Promise(done => { + resolve = done + }) + + return { promise, resolve } +} + describe('project scope', () => { beforeEach(() => { window.localStorage.clear() @@ -118,6 +130,50 @@ describe('project scope', () => { }) }) +describe('projects RPC profile forwarding', () => { + beforeEach(() => { + vi.clearAllMocks() + $activeGatewayProfile.set('default') + $activeProjectId.set(null) + $projectTree.set([]) + setShowAllProfiles(false) + }) + + it('forwards the normalized active profile to project read RPCs', async () => { + const request = vi.fn(async () => ({ active_id: null, projects: [], scoped_session_ids: [] })) + const gateway = { connectionState: 'open', request } + activeGateway.mockReturnValue(gateway as never) + gatewayAtom.set(gateway as never) + $activeGatewayProfile.set(' coder ') + + await refreshProjects() + await refreshProjectTree() + await fetchProjectSessions('p_123') + + expect(request).toHaveBeenNthCalledWith(1, 'projects.list', { profile: 'coder' }) + expect(request).toHaveBeenNthCalledWith(2, 'projects.tree', { preview_limit: 3, profile: 'coder' }) + expect(request).toHaveBeenNthCalledWith(3, 'projects.project_sessions', { + profile: 'coder', + project_id: 'p_123' + }) + }) + + it('skips project reads in the all-profiles view rather than forwarding its sentinel', async () => { + const request = vi.fn() + const gateway = { connectionState: 'open', request } + activeGateway.mockReturnValue(gateway as never) + gatewayAtom.set(gateway as never) + setShowAllProfiles(true) + + await refreshProjects() + await refreshProjectTree() + await fetchProjectSessions('p_123') + + expect(request).not.toHaveBeenCalled() + setShowAllProfiles(false) + }) +}) + describe('resolveNewSessionCwd', () => { beforeEach(() => { $projectScope.set(ALL_PROJECTS) @@ -478,6 +534,7 @@ describe('repository discovery policy', () => { expect(scanRepos).not.toHaveBeenCalled() expect(request).toHaveBeenCalledWith('projects.record_repos', { discovery_policy: { enabled: false, exclude_paths: [], roots: [] }, + profile: 'default', repos: [] }) }) @@ -513,24 +570,154 @@ describe('repository discovery policy', () => { exclude_paths: ['/work/vendor'], roots: ['/work'] }, + profile: 'default', repos: [{ label: 'repo', root: '/work/repo' }] }) }) - it('does not scan the local filesystem for remote connections', async () => { + it('does not scan the local filesystem for remote connections but still refreshes the project tree', async () => { isDesktopFsRemoteMode.mockReturnValue(true) const scanRepos = vi.fn() desktopGit.mockReturnValue({ scanRepos } as never) + const request = vi.fn(async (method: string) => + method === 'projects.tree' + ? { active_id: null, projects: [], scoped_session_ids: [] } + : { accepted: false, repos: [] } + ) + gatewayWith(request) + $projectTree.set([{ id: 'seed', label: 'seed', path: null, repos: [], sessionCount: 0 } satisfies SidebarProjectTree]) await scanAndRecordRepos(true) expect(scanRepos).not.toHaveBeenCalled() expect(getHermesConfig).not.toHaveBeenCalled() + // The desktop can't crawl the remote host's filesystem, so it asks the + // host to scan its own discovery roots (`projects.discover_repos` with + // `scan: true`) — repos with zero Hermes sessions must still surface — + // then refreshes the tree to pick up the merged list. Regression for + // #81723: the sidebar used to go silent in remote mode and never + // refresh again. + expect(request).toHaveBeenCalledWith('projects.discover_repos', { profile: 'default', scan: true }) + expect(request).toHaveBeenCalledWith( + 'projects.tree', + expect.objectContaining({ preview_limit: expect.any(Number), profile: 'default' }) + ) + // A successful scan refreshes the tree (here to the empty list the mock + // tree returns), so a later discover-repos call replaces it instead of + // keeping the stale seed. + expect($projectTree.get()).toEqual([]) + }) + + it('surfaces a reject from remote discover_repos without clearing the sidebar', async () => { + // Backend error (RPC `error` frame) rejects the request — the sidebar must + // keep its last known list and flag the failure, not go silently blank. + isDesktopFsRemoteMode.mockReturnValue(true) + desktopGit.mockReturnValue({ scanRepos: vi.fn() } as never) + const request = vi.fn(async (method: string) => { + if (method === 'projects.discover_repos') { + throw new Error('discover_repos failed') + } + if (method === 'projects.tree') { + return { active_id: null, projects: [], scoped_session_ids: [] } + } + return { accepted: false, repos: [] } + }) + gatewayWith(request) + $projectTree.set([{ id: 'seed', label: 'seed', path: null, repos: [], sessionCount: 0 } satisfies SidebarProjectTree]) + + await scanAndRecordRepos(true) + + // The tree refresh must NOT run against a failed remote scan ... + expect(request).not.toHaveBeenCalledWith( + 'projects.tree', + expect.objectContaining({ preview_limit: expect.any(Number) }) + ) + // ... the cached tree is preserved ... + expect($projectTree.get()).toEqual([ + { id: 'seed', label: 'seed', path: null, repos: [], sessionCount: 0 } + ]) + }) + + it('does not treat an error-shaped discover_repos response as a successful refresh', async () => { + // A resolved-but-error-shaped body (`{accepted:false}` / no `repos`) must + // be treated as a failure: keep the old list rather than refreshing into + // the silent, empty sidebar of #81723. + isDesktopFsRemoteMode.mockReturnValue(true) + desktopGit.mockReturnValue({ scanRepos: vi.fn() } as never) + const request = vi.fn(async (method: string) => + method === 'projects.tree' + ? { active_id: null, projects: [], scoped_session_ids: [] } + : { accepted: false } + ) + gatewayWith(request) + $projectTree.set([{ id: 'seed', label: 'seed', path: null, repos: [], sessionCount: 0 } satisfies SidebarProjectTree]) + + await scanAndRecordRepos(true) + + expect(request).not.toHaveBeenCalledWith( + 'projects.tree', + expect.objectContaining({ preview_limit: expect.any(Number) }) + ) + expect($projectTree.get()).toEqual([ + { id: 'seed', label: 'seed', path: null, repos: [], sessionCount: 0 } + ]) + }) + + it('records repos under the profile the scan started with, not one focused mid-scan', async () => { + const { promise: scanResult, resolve: resolveScan } = deferred>() + const { promise: scanStarted, resolve: markScanStarted } = deferred() + + const request = vi.fn(async (method: string) => + method === 'projects.tree' + ? { + active_id: null, + projects: [{ id: 'p_lured', label: 'Lured', path: null, repos: [], sessionCount: 0 }], + scoped_session_ids: [] + } + : { accepted: true, repos: [] } + ) + + gatewayWith(request) + const scanRepos = vi.fn(() => { + markScanStarted() + return scanResult + }) + desktopGit.mockReturnValue({ scanRepos } as never) + getHermesConfig.mockResolvedValue({ + desktop: { + repo_scan_enabled: true, + repo_scan_exclude_paths: [], + repo_scan_roots: ['/work'] + } + }) + $activeGatewayProfile.set('launch') + $projectTree.set([]) + + const pending = scanAndRecordRepos() + await scanStarted + $activeGatewayProfile.set('coder') + resolveScan([{ label: 'repo', root: '/work/repo' }]) + await pending + + expect(request).toHaveBeenCalledWith('projects.record_repos', { + discovery_policy: { enabled: true, exclude_paths: [], roots: ['/work'] }, + profile: 'launch', + repos: [{ label: 'repo', root: '/work/repo' }] + }) + expect(request).not.toHaveBeenCalledWith('projects.record_repos', expect.objectContaining({ profile: 'coder' })) + expect($projectTree.get()).toEqual([]) }) }) describe('project tree profile isolation', () => { - it('does not publish a late response from the previous profile', async () => { + beforeEach(() => { + setShowAllProfiles(false) + $activeGatewayProfile.set('default') + $projects.set([]) + $projectTree.set([]) + }) + + it('does not publish a late response from the previous gateway', async () => { let resolveA: ((value: unknown) => void) | undefined const responseA = new Promise(resolve => { @@ -566,6 +753,84 @@ describe('project tree profile isolation', () => { expect($projectTree.get().map(project => project.id)).toEqual(['profile-b']) }) + + it('does not publish a late projects.list response from the previous profile', async () => { + const { promise: defaultResponse, resolve: resolveDefault } = deferred() + const request = vi.fn((_method: string, params: Record) => + params.profile === 'default' + ? defaultResponse + : Promise.resolve({ + active_id: null, + projects: [{ id: 'profile-b', label: 'Profile B' }] + }) + ) + const gateway = { connectionState: 'open', request } + activeGateway.mockReturnValue(gateway as never) + gatewayAtom.set(gateway as never) + + const pendingDefault = refreshProjects() + $activeGatewayProfile.set('profile-b') + await refreshProjects() + resolveDefault({ + active_id: null, + projects: [{ id: 'profile-a', label: 'Profile A' }] + }) + await pendingDefault + + expect($projects.get().map(project => project.id)).toEqual(['profile-b']) + }) + + it('does not publish a late projects.tree response from the previous profile', async () => { + const { promise: defaultResponse, resolve: resolveDefault } = deferred() + const request = vi.fn((_method: string, params: Record) => + params.profile === 'default' + ? defaultResponse + : Promise.resolve({ + active_id: null, + projects: [{ id: 'profile-b', label: 'Profile B', path: null, repos: [], sessionCount: 0 }], + scoped_session_ids: [] + }) + ) + const gateway = { connectionState: 'open', request } + activeGateway.mockReturnValue(gateway as never) + gatewayAtom.set(gateway as never) + + const pendingDefault = refreshProjectTree() + $activeGatewayProfile.set('profile-b') + await refreshProjectTree() + resolveDefault({ + active_id: null, + projects: [{ id: 'profile-a', label: 'Profile A', path: null, repos: [], sessionCount: 0 }], + scoped_session_ids: [] + }) + await pendingDefault + + expect($projectTree.get().map(project => project.id)).toEqual(['profile-b']) + }) + + it('drops a late hydrated-project response from the previous profile', async () => { + const { promise: defaultResponse, resolve: resolveDefault } = deferred() + const request = vi.fn((_method: string, params: Record) => + params.profile === 'default' + ? defaultResponse + : Promise.resolve({ + project: { id: 'profile-b', label: 'Profile B', path: null, repos: [], sessionCount: 0 } + }) + ) + const gateway = { connectionState: 'open', request } + activeGateway.mockReturnValue(gateway as never) + gatewayAtom.set(gateway as never) + + const pendingDefault = fetchProjectSessions('p_123') + $activeGatewayProfile.set('profile-b') + const profileB = await fetchProjectSessions('p_123') + resolveDefault({ + project: { id: 'profile-a', label: 'Profile A', path: null, repos: [], sessionCount: 0 } + }) + + expect(profileB?.id).toBe('profile-b') + await expect(pendingDefault).resolves.toBeNull() + }) }) describe('tombstone pruning', () => { diff --git a/apps/desktop/src/store/projects.ts b/apps/desktop/src/store/projects.ts index b904b0f34e..eced3fe82b 100644 --- a/apps/desktop/src/store/projects.ts +++ b/apps/desktop/src/store/projects.ts @@ -16,7 +16,13 @@ import { persistentAtom } from '@/lib/persisted' import { $gateway, activeGateway, ensureActiveGatewayOpen } from '@/store/gateway' import { setSidebarAgentsGrouped } from '@/store/layout' import { notify } from '@/store/notifications' -import { $activeGatewayProfile, $profileScope, ALL_PROFILES, requestFreshSession } from '@/store/profile' +import { + $activeGatewayProfile, + $profileScope, + ALL_PROFILES, + normalizeProfileKey, + requestFreshSession +} from '@/store/profile' import { $selectedStoredSessionId, $sessions, @@ -341,6 +347,23 @@ async function gatewayRequest(method: string, params: Record return gateway.request(method, params) } +function projectProfile(): null | string { + const profile = normalizeProfileKey($activeGatewayProfile.get()) + + return $profileScope.get() === ALL_PROFILES || profile === ALL_PROFILES ? null : profile +} + +function projectParams( + params: Record = {}, + profile: null | string = projectProfile() +): Record { + if (!profile) { + throw new Error('Projects are unavailable while viewing all profiles') + } + + return { ...params, profile } +} + async function gatewayRequestOn( gateway: HermesGateway, method: string, @@ -354,15 +377,24 @@ interface ActiveProjectsContext { profile: string } +function stillOnProjectsContext(context: ActiveProjectsContext): boolean { + return activeGateway() === context.gateway && projectProfile() === context.profile +} + async function activeProjectsContext(): Promise { - const profile = $activeGatewayProfile.get() || 'default' + const profile = projectProfile() + + if (!profile) { + throw new Error('Projects are unavailable while viewing all profiles') + } + let gateway = activeGateway() if (!gateway || gateway.connectionState !== 'open') { gateway = await ensureActiveGatewayOpen() } - if (!gateway || gateway !== activeGateway() || profile !== ($activeGatewayProfile.get() || 'default')) { + if (!gateway || gateway !== activeGateway() || profile !== projectProfile()) { throw new Error('Active Hermes profile changed while connecting') } @@ -380,20 +412,24 @@ let projectsRefreshGeneration = 0 // not up yet) leaves the cached atoms intact so the sidebar doesn't flicker. export async function refreshProjects(): Promise { const generation = ++projectsRefreshGeneration - let gateway: HermesGateway | null = null + let context: ActiveProjectsContext | null = null try { - gateway = (await activeProjectsContext()).gateway - const payload = await gatewayRequestOn(gateway, 'projects.list') + context = await activeProjectsContext() + const payload = await gatewayRequestOn( + context.gateway, + 'projects.list', + projectParams({}, context.profile) + ) - if (generation !== projectsRefreshGeneration || activeGateway() !== gateway) { + if (generation !== projectsRefreshGeneration || !stillOnProjectsContext(context)) { return } applyPayload(payload) markProjectsRpcSuccess() } catch (err) { - if (generation === projectsRefreshGeneration && (!gateway || activeGateway() === gateway)) { + if (context && generation === projectsRefreshGeneration && stillOnProjectsContext(context)) { markProjectsRpcFailure(err) } // Backend may not be ready; keep the last known list. @@ -433,26 +469,29 @@ function applyProjectTreePayload(res: ProjectTreePayload): void { } } -async function refreshProjectTreeOn(gateway: HermesGateway): Promise { +async function refreshProjectTreeOn(context: ActiveProjectsContext): Promise { const generation = ++projectTreeRefreshGeneration + const { gateway, profile } = context if (activeGateway() === gateway) { $projectTreeLoading.set(true) } try { - const res = await gatewayRequestOn(gateway, 'projects.tree', { - preview_limit: PROJECT_TREE_PREVIEW_LIMIT - }) + const res = await gatewayRequestOn( + gateway, + 'projects.tree', + projectParams({ preview_limit: PROJECT_TREE_PREVIEW_LIMIT }, profile) + ) - if (generation !== projectTreeRefreshGeneration || activeGateway() !== gateway) { + if (generation !== projectTreeRefreshGeneration || !stillOnProjectsContext(context)) { return } applyProjectTreePayload(res) markProjectsRpcSuccess() } catch (err) { - if (activeGateway() === gateway) { + if (generation === projectTreeRefreshGeneration && stillOnProjectsContext(context)) { markProjectsRpcFailure(err) } } finally { @@ -473,8 +512,7 @@ export async function refreshProjectTree(): Promise { } try { - const { gateway } = await activeProjectsContext() - await refreshProjectTreeOn(gateway) + await refreshProjectTreeOn(await activeProjectsContext()) } catch { // Backend may not be ready; keep the last known tree. } @@ -514,11 +552,22 @@ async function refreshProjectTreeAcrossProfiles(): Promise { // Fully hydrated lanes (repo -> lane -> session rows) for one project, fetched // when the user enters it. Same backend grouping as `projects.tree`, so ids and // membership match exactly. +let projectSessionsRefreshGeneration = 0 + export async function fetchProjectSessions(projectId: string): Promise { + const generation = ++projectSessionsRefreshGeneration + try { - const res = await gatewayRequest<{ project: SidebarProjectTree | null }>('projects.project_sessions', { - project_id: projectId - }) + const context = await activeProjectsContext() + const res = await gatewayRequestOn<{ project: SidebarProjectTree | null }>( + context.gateway, + 'projects.project_sessions', + projectParams({ project_id: projectId }, context.profile) + ) + + if (generation !== projectSessionsRefreshGeneration || !stillOnProjectsContext(context)) { + return null + } return res.project ?? null } catch { @@ -616,6 +665,41 @@ $gateway.subscribe(syncReposScanning) export async function scanAndRecordRepos(force = false): Promise { if (isDesktopFsRemoteMode()) { + // On a remote backend the desktop can't crawl the host filesystem. + // Ask the host to scan its own discovery roots (`projects.discover_repos` + // with `scan: true` — added in #81723) so repos with zero Hermes + // sessions still surface, then refresh the tree so the sidebar picks up + // the merged session-derived + scanned list. + try { + const context = await activeProjectsContext() + const discovered = await gatewayRequestOn<{ + repos?: unknown + discovery_policy?: unknown + }>(context.gateway, 'projects.discover_repos', projectParams({ scan: true }, context.profile)) + + // A resolved response must be the discovery shape. Anything else (an + // error/`accepted:false` body, or a backend that ignored `scan` and + // returned no repo list) means the scan didn't happen — bail out without + // touching the tree so the sidebar keeps its last known list instead of + // being blanked back to the silent, unpopulated state of #81723. + if (discovered?.repos === undefined) { + markProjectsRpcFailure(new Error('projects.discover_repos returned no repo list')) + return + } + + // Remote scan succeeded: refresh the tree so the merged session-derived + + // scanned list surfaces. Skip if the user moved on — a stale scan must + // not publish into the newly focused profile. + if (stillOnProjectsContext(context)) { + await refreshProjectTreeOn(context) + } + } catch (err) { + // Surface the failure (stale backend, RPC error, gateway drop) instead + // of swallowing it: a silent return is exactly the "sidebar goes quiet" + // symptom `scan:true` was meant to fix (#81723). Keep the old list and + // let the sidebar show the error/absent state. + markProjectsRpcFailure(err) + } return } @@ -649,10 +733,11 @@ export async function scanAndRecordRepos(force = false): Promise { state.runningSignature = signature if (!policy.enabled) { - await gatewayRequestOn(context.gateway, 'projects.record_repos', { - discovery_policy: policy, - repos: [] - }) + await gatewayRequestOn( + context.gateway, + 'projects.record_repos', + projectParams({ discovery_policy: policy, repos: [] }, context.profile) + ) } else { scanningGatewayGenerations.set(context.gateway, generation) syncReposScanning() @@ -666,10 +751,11 @@ export async function scanAndRecordRepos(force = false): Promise { return } - await gatewayRequestOn(context.gateway, 'projects.record_repos', { - discovery_policy: policy, - repos - }) + await gatewayRequestOn( + context.gateway, + 'projects.record_repos', + projectParams({ discovery_policy: policy, repos }, context.profile) + ) } if (state.generation !== generation) { @@ -677,10 +763,13 @@ export async function scanAndRecordRepos(force = false): Promise { } state.completedSignature = signature - // Scope-aware on purpose: the scan records into one profile, but folding - // its result back in through the active scope keeps an all-profiles tree - // from being overwritten by the scanned profile's own. - await refreshProjectTree() + // Completion refresh only when the focused profile still matches the one + // the scan was captured under. refreshProjectTree() re-derives the current + // context, so skipping on mismatch keeps a stale scan from publishing into + // the newly focused profile. + if (stillOnProjectsContext(context)) { + await refreshProjectTree() + } } catch { state.completedSignature = undefined } finally { @@ -808,17 +897,20 @@ export async function createProject(input: CreateProjectInput): Promise('projects.create', { - name: input.name, - folders: input.folders ?? [], - primary_path: input.primaryPath, - slug: input.slug, - description: input.description, - icon: input.icon, - color: input.color, - board_slug: input.boardSlug, - use: input.use ?? false - }) + res = await gatewayRequest<{ project: ProjectInfo | null }>( + 'projects.create', + projectParams({ + name: input.name, + folders: input.folders ?? [], + primary_path: input.primaryPath, + slug: input.slug, + description: input.description, + icon: input.icon, + color: input.color, + board_slug: input.boardSlug, + use: input.use ?? false + }) + ) } catch (err) { if (isMissingRpcMethod(err)) { $projectsRpcAvailable.set(false) @@ -891,12 +983,15 @@ export async function updateProject( // Backend treats null/undefined as "leave unchanged"; "" clears (stores NULL). // Map explicit null → "" so "no color"/"no icon" actually clear. await persistOrRollback(snap, () => - gatewayRequest('projects.update', { - id, - ...patch, - ...(patch.color === null && { color: '' }), - ...(patch.icon === null && { icon: '' }) - }) + gatewayRequest( + 'projects.update', + projectParams({ + id, + ...patch, + ...(patch.color === null && { color: '' }), + ...(patch.icon === null && { icon: '' }) + }) + ) ) } @@ -967,7 +1062,10 @@ export async function addProjectFolder( } await persistOrRollback(snap, () => - gatewayRequest('projects.add_folder', { id, path, label: opts.label, is_primary: opts.isPrimary ?? false }) + gatewayRequest( + 'projects.add_folder', + projectParams({ id, path, label: opts.label, is_primary: opts.isPrimary ?? false }) + ) ) reconcileProjects() } @@ -1010,13 +1108,13 @@ export async function deleteProject(id: string): Promise { } await persistOrRollback(snap, async () => { - applyPayload(await gatewayRequest('projects.delete', { id })) + applyPayload(await gatewayRequest('projects.delete', projectParams({ id }))) }) void refreshProjectTree() } export async function setActiveProject(id: null | string): Promise { - const res = await gatewayRequest<{ active_id: null | string }>('projects.set_active', { id }) + const res = await gatewayRequest<{ active_id: null | string }>('projects.set_active', projectParams({ id })) $activeProjectId.set(res.active_id ?? null) } diff --git a/apps/desktop/src/store/session-dot-state.test.ts b/apps/desktop/src/store/session-dot-state.test.ts index 1661c3ba09..52b614ecf4 100644 --- a/apps/desktop/src/store/session-dot-state.test.ts +++ b/apps/desktop/src/store/session-dot-state.test.ts @@ -4,7 +4,13 @@ import { createClientSessionState } from '@/lib/chat-runtime' import type { SessionInfo } from '@/types/hermes' import { $sessions, $unreadFinishedSessionIds, setSessions } from './session' -import { $delegatingSessionIds, $sessionDotStateById, hasLiveTurn, showsRunningArc } from './session-dot-state' +import { + $delegatingSessionIds, + $sessionDotStateById, + hasLiveTurn, + showsRunningArc, + unreadSessionCount +} from './session-dot-state' import { clearAllSessionStates, publishSessionState } from './session-states' import { $unreadWriteGuard } from './session-unread-remote' import { $subagentsBySession, type SubagentProgress } from './subagents' @@ -148,3 +154,20 @@ describe('persisted unread (backend watermark)', () => { expect($sessionDotStateById.get()['s1'] ?? 'idle').not.toBe('unread') }) }) + +describe('unreadSessionCount', () => { + it('counts listed unread rows and skips archived', () => { + expect( + unreadSessionCount({ a: 'unread', b: 'working', c: 'unread' }, [ + { id: 'a' }, + { id: 'b' }, + { archived: true, id: 'c' }, + { id: 'missing' } + ]) + ).toBe(1) + }) + + it('does not count alias keys that are not listed rows', () => { + expect(unreadSessionCount({ tip: 'unread', root: 'unread' }, [{ id: 'tip' }])).toBe(1) + }) +}) diff --git a/apps/desktop/src/store/session-dot-state.ts b/apps/desktop/src/store/session-dot-state.ts index f6f30169ff..58c0f8182c 100644 --- a/apps/desktop/src/store/session-dot-state.ts +++ b/apps/desktop/src/store/session-dot-state.ts @@ -28,7 +28,7 @@ import { computed } from 'nanostores' import { stableArray, stableRecord } from '@/lib/stable-array' import { $backgroundRunningSessionIds } from './composer-status' -import { $sessions, $unreadFinishedSessionIds, lineageAliases } from './session' +import { $cronSessions, $messagingSessions, $sessions, $unreadFinishedSessionIds, lineageAliases } from './session' import { $attentionSessionIds, $draftSessionIds, @@ -182,3 +182,27 @@ export const $sessionDotStateById = computed( return (dotStates = stableRecord(dotStates, next)) } ) + +/** Listed, non-archived rows whose resolved status is unread. Alias keys in + * `$sessionDotStateById` are ignored unless they are themselves a listed row. */ +export function unreadSessionCount( + byId: Readonly>, + ...lists: Array +): number { + let n = 0 + + for (const rows of lists) { + for (const row of rows) { + if (!row.archived && byId[row.id] === 'unread') { + n++ + } + } + } + + return n +} + +export const $unreadSessionCount = computed( + [$sessionDotStateById, $sessions, $cronSessions, $messagingSessions], + (byId, sessions, cron, messaging) => unreadSessionCount(byId, sessions, cron, messaging) +) diff --git a/cli.py b/cli.py index 45b9b17ebc..97d5719fd4 100644 --- a/cli.py +++ b/cli.py @@ -12970,7 +12970,11 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self.agent, "context_compressor", None ), ) - if summary.get("aborted") or summary.get("fallback_used"): + if ( + summary.get("aborted") + or summary.get("fallback_used") + or summary.get("refused_would_grow") + ): icon = "⚠️" else: icon = "🗜️" if summary["noop"] else "✅" diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 57f943eda1..30ed5a8108 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -245,7 +245,15 @@ def _coerce_request_bool(value: Any, default: bool = False) -> bool: _REQUEST_OPTION_MISSING = object() -_REASONING_EFFORTS = frozenset({"none", "minimal", "low", "medium", "high", "xhigh"}) +# Full internal ladder + "none": the API server accepts what /reasoning and +# config.yaml accept (hermes_constants.VALID_REASONING_EFFORTS); wire-level +# clamping to each provider's vocabulary happens downstream in the +# transports/profiles via agent.reasoning_effort. Rejecting "max"/"ultra" +# here made API/browser clients second-class citizens of the ladder +# (#78216's api_server observation). +_REASONING_EFFORTS = frozenset( + {"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"} +) _RUNTIME_AGENT_OVERRIDE_KEYS = ( "api_key", "base_url", diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 3b57335f74..1778c28749 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -88,8 +88,19 @@ class CLICommandsMixin: args = filtered if not args: - # List checkpoints + # List checkpoints — fall back to the cross-project view when the + # current directory has none (#10505, reapply of PR #10633 by + # @nightq). The Aug 2026 QA sweep hit this live: writes landed + # checkpoints under the session cwd (/tmp/qa-repo) while bare + # /rollback searched only TERMINAL_CWD's project and reported + # "No checkpoints found" despite fresh checkpoints existing. checkpoints = mgr.list_checkpoints(cwd) + if not checkpoints: + all_checkpoints = mgr.list_all_checkpoints() + if all_checkpoints: + print(f" No checkpoints for {cwd} — showing all directories.") + print(format_checkpoint_list(all_checkpoints, "all directories")) + return print(format_checkpoint_list(checkpoints, cwd)) return diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index b1f9c67db1..1420fcafaf 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -2517,6 +2517,19 @@ class TelegramAdapter(BasePlatformAdapter): return if generation != self._polling_generation: return + if not self._polling_progress_event.is_set(): + # The first confirmed getUpdates round-trip of this generation + # resolves the "health pending getUpdates progress" line both + # reconnect paths end on. Without it the log stream for + # "reconnected and healthy" is byte-identical to "reconnected + # and hung" — a wedged long-poll is invisible until a user + # notices silence (#90504). + logger.info( + "[%s] Telegram polling confirmed healthy: getUpdates progressing " + "(generation %d)", + self.name, + generation, + ) self._polling_progress_event.set() self._polling_network_error_count = 0 if generation == self._polling_conflict_recovery_generation: diff --git a/tests/agent/test_builtin_memory_disabled_surface.py b/tests/agent/test_builtin_memory_disabled_surface.py new file mode 100644 index 0000000000..d2771bf463 --- /dev/null +++ b/tests/agent/test_builtin_memory_disabled_surface.py @@ -0,0 +1,161 @@ +"""Built-in memory disabled in config must leave no dead surface behind. + +Setting ``memory.memory_enabled: false`` and ``memory.user_profile_enabled: +false`` stops ``agent_init`` from building a ``MemoryStore``, so the ``memory`` +tool dispatches against ``store=None`` and every call comes back "Memory is not +available". Before the fix the tool stayed in the schema and MEMORY_GUIDANCE +stayed in the system prompt, so users running a third-party provider (Hindsight, +Mem0, …) paid for both on every API call with no way to drop them — listing +``memory`` under ``disabled_toolsets`` takes the provider's tools down too. + +These tests exercise the real resolution chain (config on disk → check_fn → +``get_tool_definitions``) against a temp ``HERMES_HOME``, not mocks. +""" + +import pytest +import yaml + +from model_tools import get_tool_definitions + + +@pytest.fixture(autouse=True) +def _clear_caches(): + """check_fn results and tool definitions are both cached; config written by + a test only takes effect once those are dropped.""" + from model_tools import _clear_tool_defs_cache + from tools.registry import invalidate_check_fn_cache + + invalidate_check_fn_cache() + _clear_tool_defs_cache() + yield + invalidate_check_fn_cache() + _clear_tool_defs_cache() + + +def _write_memory_config(home, **memory_section): + home.mkdir(parents=True, exist_ok=True) + (home / "config.yaml").write_text( + yaml.safe_dump({"memory": memory_section}), encoding="utf-8" + ) + + +@pytest.fixture +def hermes_home(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + monkeypatch.setenv("HERMES_HOME", str(home)) + return home + + +def _memory_tool_names(): + tools = get_tool_definitions(enabled_toolsets=["memory"], quiet_mode=True) + return {tool["function"]["name"] for tool in tools} + + +class TestBuiltinMemoryToolAvailability: + def test_tool_hidden_when_both_stores_disabled(self, hermes_home): + _write_memory_config( + hermes_home, memory_enabled=False, user_profile_enabled=False + ) + assert "memory" not in _memory_tool_names() + + def test_tool_present_when_only_user_profile_enabled(self, hermes_home): + _write_memory_config( + hermes_home, memory_enabled=False, user_profile_enabled=True + ) + assert "memory" in _memory_tool_names() + + def test_tool_present_when_only_memory_enabled(self, hermes_home): + _write_memory_config( + hermes_home, memory_enabled=True, user_profile_enabled=False + ) + assert "memory" in _memory_tool_names() + + def test_tool_present_by_default(self, hermes_home): + """No config file at all must not strip a working tool.""" + assert "memory" in _memory_tool_names() + + def test_unreadable_config_fails_open(self, hermes_home, monkeypatch): + """A config read error must not silently remove the tool.""" + from tools import memory_tool as memory_tool_module + + def _boom(): + raise RuntimeError("config unreadable") + + monkeypatch.setattr( + "hermes_cli.config.load_config_readonly", _boom, raising=False + ) + assert memory_tool_module.check_memory_requirements() is True + + +class TestExternalProviderSurvivesBuiltinDisable: + """Dropping the built-in tool must not drop the external provider's tools. + + ``memory_provider_tools_enabled`` short-circuits on the built-in tool being + present, so hiding that tool moves the decision onto the toolset gate. The + provider must still be reachable for every way a caller can ask for memory. + """ + + def test_provider_tools_enabled_when_memory_toolset_requested(self): + from agent.memory_manager import memory_provider_tools_enabled + + assert memory_provider_tools_enabled( + ["memory", "file"], None, memory_tool_present=False + ) + + def test_provider_tools_enabled_for_unrestricted_toolsets(self): + from agent.memory_manager import memory_provider_tools_enabled + + assert memory_provider_tools_enabled(None, None, memory_tool_present=False) + + def test_disabled_toolsets_still_takes_everything_down(self): + """The heavy switch keeps its documented meaning.""" + from agent.memory_manager import memory_provider_tools_enabled + + assert not memory_provider_tools_enabled( + None, ["memory"], memory_tool_present=False + ) + + +class TestInjectionEndToEnd: + """The real ``inject_memory_provider_tools`` with no built-in memory tool.""" + + def test_provider_tools_injected_without_builtin_memory_tool(self): + from types import SimpleNamespace + + from agent.memory_manager import MemoryManager, inject_memory_provider_tools + from agent.memory_provider import MemoryProvider + + class _Provider(MemoryProvider): + @property + def name(self): + return "fake_hindsight" + + def is_available(self): + return True + + def initialize(self, session_id, **kwargs): + pass + + def get_tool_schemas(self): + return [ + { + "name": "hindsight_retain", + "description": "retain", + "parameters": {"type": "object", "properties": {}}, + } + ] + + manager = MemoryManager() + manager.add_provider(_Provider()) + agent = SimpleNamespace( + _memory_manager=manager, + enabled_toolsets=["memory"], + disabled_toolsets=None, + tools=[], + valid_tool_names=set(), + ) + + added = inject_memory_provider_tools(agent) + + assert added == 1 + assert "hindsight_retain" in agent.valid_tool_names diff --git a/tests/agent/test_manual_compression_refusal_feedback.py b/tests/agent/test_manual_compression_refusal_feedback.py new file mode 100644 index 0000000000..6ed514b6b6 --- /dev/null +++ b/tests/agent/test_manual_compression_refusal_feedback.py @@ -0,0 +1,58 @@ +"""Regression tests: manual /compress feedback for a would-grow refusal. + +The anti-growth guard (#83339 / PR #86700) can refuse the commit AFTER the +compressor produced a candidate. The CLI's feedback previously compared the +returned list against its pre-call snapshot; durable-snapshot adoption can +legitimately change that list, so a REFUSED compression printed +"✅ Compressed: 8 → 14 messages" directly under the refusal warning +(Aug 2026 full-surface CLI QA sweep). The refusal must surface as its own +honest headline. +""" + +from types import SimpleNamespace + +from agent.manual_compression_feedback import summarize_manual_compression + + +def _msgs(n): + return [{"role": "user", "content": f"m{n_i}"} for n_i in range(n)] + + +def _state(**flags): + base = { + "_last_compress_aborted": False, + "_last_compress_refused_would_grow": False, + "_last_summary_fallback_used": False, + "_last_summary_error": None, + "_last_summary_dropped_count": 0, + } + base.update(flags) + return SimpleNamespace(**base) + + +class TestWouldGrowRefusalFeedback: + def test_refusal_reports_preserved_not_compressed(self): + summary = summarize_manual_compression( + _msgs(8), _msgs(14), 22025, 22453, + compression_state=_state(_last_compress_refused_would_grow=True), + ) + assert summary["refused_would_grow"] is True + assert "refused" in summary["headline"].lower() + assert "→" not in summary["headline"] # never claims a message rewrite + assert "unchanged" in summary["token_line"] + assert "no messages were removed" in summary["note"] + + def test_normal_compression_unchanged(self): + summary = summarize_manual_compression( + _msgs(10), _msgs(4), 30000, 12000, compression_state=_state(), + ) + assert summary["refused_would_grow"] is False + assert summary["headline"].startswith("Compressed: 10 → 4") + + def test_abort_still_wins_its_own_headline(self): + summary = summarize_manual_compression( + _msgs(6), _msgs(6), 9000, 9000, + compression_state=_state(_last_compress_aborted=True), + ) + assert summary["aborted"] is True + assert summary["headline"].startswith("Compression aborted") diff --git a/tests/agent/test_reasoning_effort_module.py b/tests/agent/test_reasoning_effort_module.py index 6cb4c73d02..dc04406a1f 100644 --- a/tests/agent/test_reasoning_effort_module.py +++ b/tests/agent/test_reasoning_effort_module.py @@ -112,13 +112,15 @@ class TestClampEffort: class TestKimiVocabulary: @pytest.mark.parametrize( - "model", ["k3", "kimi-k3", "kimi-k3-cot", "moonshotai/kimi-k3"] + "model", + ["k3", "kimi-k3", "kimi-k3-cot", "moonshotai/kimi-k3", "k3-256k"], ) def test_k3_slugs(self, model): assert kimi_supported_efforts(model) is KIMI_K3_EFFORTS @pytest.mark.parametrize( - "model", ["kimi-k2.6", "moonshotai/kimi-k2-0905", "kimi-latest", None] + "model", + ["kimi-k2.6", "moonshotai/kimi-k2-0905", "kimi-latest", "mk3000", None], ) def test_k2_era_slugs(self, model): assert kimi_supported_efforts(model) is KIMI_K2_EFFORTS diff --git a/tests/gateway/test_api_server_reasoning_ladder.py b/tests/gateway/test_api_server_reasoning_ladder.py new file mode 100644 index 0000000000..0af8770f8e --- /dev/null +++ b/tests/gateway/test_api_server_reasoning_ladder.py @@ -0,0 +1,33 @@ +"""API-server reasoning-effort request parsing accepts the full ladder. + +_request_reasoning_config() previously whitelisted only none..xhigh, so an +API/browser client sending ``max`` or ``ultra`` (both valid /reasoning and +config.yaml levels) was silently ignored and the session ran at the default +effort. The server now accepts the full internal ladder; per-provider wire +clamping happens downstream in the transports/profiles via +agent.reasoning_effort (#78216's api_server observation). +""" + +from gateway.platforms.api_server import _request_reasoning_config +from hermes_constants import VALID_REASONING_EFFORTS + + +class TestRequestReasoningFullLadder: + def test_every_configurable_level_is_accepted(self): + for level in VALID_REASONING_EFFORTS: + out = _request_reasoning_config({"reasoning_effort": level}) + assert out == {"enabled": True, "effort": level}, level + + def test_none_disables(self): + assert _request_reasoning_config({"reasoning_effort": "none"}) == { + "enabled": False + } + + def test_structured_object_shape(self): + out = _request_reasoning_config( + {"reasoning": {"enabled": True, "effort": "ultra"}} + ) + assert out == {"enabled": True, "effort": "ultra"} + + def test_unknown_level_still_ignored(self): + assert _request_reasoning_config({"reasoning_effort": "turbo"}) is None diff --git a/tests/gateway/test_telegram_polling_health_confirmation.py b/tests/gateway/test_telegram_polling_health_confirmation.py new file mode 100644 index 0000000000..74be6364fe --- /dev/null +++ b/tests/gateway/test_telegram_polling_health_confirmation.py @@ -0,0 +1,107 @@ +"""Regression tests for the Telegram polling health-confirmation log (#90504). + +Both reconnect paths end on ``health pending getUpdates progress`` and +``_record_polling_progress`` used to complete silently, so the log stream for +"reconnected and healthy" was byte-identical to "reconnected and hung". The +first confirmed getUpdates round-trip of each generation now emits an INFO +line, turning the pending line into a resolvable pair whose *absence* after a +reconnect is a reliable hung-poll signature. +""" + +import asyncio +import logging +import sys +from unittest.mock import MagicMock + +def _ensure_telegram_mock(): + if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"): + return + telegram_mod = MagicMock() + telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None) + telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2" + telegram_mod.constants.ChatType.GROUP = "group" + telegram_mod.constants.ChatType.SUPERGROUP = "supergroup" + telegram_mod.constants.ChatType.CHANNEL = "channel" + telegram_mod.constants.ChatType.PRIVATE = "private" + telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {}) + telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {}) + for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"): + sys.modules.setdefault(name, telegram_mod) + sys.modules.setdefault("telegram.error", telegram_mod.error) + + +_ensure_telegram_mock() + +from gateway.config import Platform # noqa: E402 +from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402 + + +def _bare_adapter(): + a = TelegramAdapter.__new__(TelegramAdapter) + a.platform = Platform.TELEGRAM + a._fatal_error_code = None + a._fatal_error_message = None + a._fatal_error_retryable = True + a._polling_teardown_started = False + a._polling_progress_accepting = True + a._polling_generation = 1 + a._polling_progress_event = asyncio.Event() + a._polling_network_error_count = 2 + a._polling_conflict_count = 3 + a._polling_conflict_recovery_generation = None + a._send_path_degraded = True + return a + + +class TestPollingHealthConfirmation: + def test_first_progress_emits_confirmed_healthy(self, caplog): + a = _bare_adapter() + with caplog.at_level(logging.INFO, logger="plugins.platforms.telegram.adapter"): + a._record_polling_progress(1) + rendered = " | ".join(rec.getMessage() for rec in caplog.records) + assert "confirmed healthy" in rendered + assert "generation 1" in rendered + assert a._polling_progress_event.is_set() + + def test_subsequent_progress_is_silent(self, caplog): + """Only the FIRST round-trip of a generation logs — a quiet evening + must not spam one INFO per getUpdates poll.""" + a = _bare_adapter() + a._record_polling_progress(1) # first — logs + with caplog.at_level(logging.INFO, logger="plugins.platforms.telegram.adapter"): + a._record_polling_progress(1) # second — silent + a._record_polling_progress(1) # third — silent + assert not [ + rec for rec in caplog.records if "confirmed healthy" in rec.getMessage() + ] + + def test_new_generation_logs_again(self, caplog): + """A reconnect starts a new generation with a fresh event; its first + progress must re-emit the confirmation so the pending line of THAT + reconnect also resolves.""" + a = _bare_adapter() + a._record_polling_progress(1) + # reconnect: new generation, event reset, counters possibly nonzero + a._polling_generation = 2 + a._polling_progress_event = asyncio.Event() + a._polling_network_error_count = 1 + a._send_path_degraded = True + with caplog.at_level(logging.INFO, logger="plugins.platforms.telegram.adapter"): + a._record_polling_progress(2) + rendered = " | ".join(rec.getMessage() for rec in caplog.records) + assert "confirmed healthy" in rendered + assert "generation 2" in rendered + + def test_stale_generation_progress_stays_silent(self, caplog): + """Progress from an abandoned generation must neither log nor set the + current event (pre-existing guard, pinned here because the log line + must inherit the same generation-scoping).""" + a = _bare_adapter() + a._polling_generation = 2 + a._polling_progress_event = asyncio.Event() + with caplog.at_level(logging.INFO, logger="plugins.platforms.telegram.adapter"): + a._record_polling_progress(1) + assert not [ + rec for rec in caplog.records if "confirmed healthy" in rec.getMessage() + ] + assert not a._polling_progress_event.is_set() diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index cd4836c575..68eb70bcc8 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -910,9 +910,45 @@ class TestBuildSystemPrompt: def test_memory_guidance_when_memory_tool_loaded(self, agent_with_memory_tool): from agent.prompt_builder import MEMORY_GUIDANCE + agent_with_memory_tool._memory_enabled = True prompt = agent_with_memory_tool._build_system_prompt() assert MEMORY_GUIDANCE in prompt + def test_no_memory_guidance_when_both_builtin_stores_disabled( + self, agent_with_memory_tool + ): + """Guidance must follow the stores, not just the tool's presence. + + With both built-in stores off, ``agent_init`` never builds a + ``MemoryStore``, so every memory call returns "Memory is not + available" — telling the model to save facts there is a dead + instruction paid for on every API call. + """ + from agent.prompt_builder import MEMORY_GUIDANCE, USER_PROFILE_GUIDANCE + + agent_with_memory_tool._memory_enabled = False + agent_with_memory_tool._user_profile_enabled = False + prompt = agent_with_memory_tool._build_system_prompt() + assert MEMORY_GUIDANCE not in prompt + assert USER_PROFILE_GUIDANCE not in prompt + + def test_profile_guidance_when_only_user_profile_enabled( + self, agent_with_memory_tool + ): + """USER.md alone gets the narrower profile-only guidance. + + The full MEMORY_GUIDANCE block instructs the model to save notes to a + MEMORY.md store that does not exist in this configuration, so the + profile-specific block is injected instead. + """ + from agent.prompt_builder import MEMORY_GUIDANCE, USER_PROFILE_GUIDANCE + + agent_with_memory_tool._memory_enabled = False + agent_with_memory_tool._user_profile_enabled = True + prompt = agent_with_memory_tool._build_system_prompt() + assert MEMORY_GUIDANCE not in prompt + assert USER_PROFILE_GUIDANCE in prompt + def test_datetime_is_date_only_not_minute_precision(self, agent): diff --git a/tests/tools/test_computer_use.py b/tests/tools/test_computer_use.py index a717086f9b..68ca6b7da4 100644 --- a/tests/tools/test_computer_use.py +++ b/tests/tools/test_computer_use.py @@ -6,6 +6,7 @@ import base64 import json import os import sys +from pathlib import Path from typing import Any, Dict, List, cast from unittest.mock import MagicMock, patch @@ -79,7 +80,7 @@ class TestRegistration: from tools.computer_use import cua_backend driver = tmp_path / "custom-cua-driver" - driver.write_text("#!/bin/sh\nexit 0\n") + driver.write_text("#!/bin/sh\nexit 0\n", encoding="utf-8") driver.chmod(0o755) monkeypatch.setenv("HERMES_CUA_DRIVER_CMD", str(driver)) @@ -2553,6 +2554,55 @@ class TestElementSpillFile: assert "elements_file" not in out +class TestCaptureScreenshotPersistence: + """Image captures expose a bounded file for explicit user delivery.""" + + _PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAgAAAAICAYAAADED76L" + "AAAADUlEQVR4nGNgGAUgAAABCAABgukLHQAAAABJRU5ErkJggg==" + ) + + def _capture(self): + from tools.computer_use.backend import CaptureResult + + return CaptureResult( + mode="vision", + width=8, + height=8, + png_b64=self._PNG_B64, + image_mime_type="image/png", + png_bytes_len=len(base64.b64decode(self._PNG_B64)), + ) + + def test_multimodal_capture_exposes_shareable_screenshot( + self, tmp_path, monkeypatch, + ): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + from tools.computer_use import tool as cu_tool + + monkeypatch.setattr( + cu_tool, "_should_route_through_aux_vision", lambda: False, + ) + out = cu_tool._capture_response(self._capture()) + + screenshot_path = out["meta"]["screenshot_path"] + assert screenshot_path in out["text_summary"] + assert "MEDIA:" not in out["text_summary"] + assert screenshot_path.startswith(str(tmp_path / "cache" / "images")) + assert Path(screenshot_path).read_bytes() == base64.b64decode(self._PNG_B64) + + def test_capture_cache_is_bounded(self, tmp_path, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + from tools.computer_use import tool as cu_tool + + monkeypatch.setattr(cu_tool, "_MAX_CAPTURE_FILES", 2) + for _ in range(3): + assert cu_tool._persist_capture_image(self._capture()) is not None + + captures = list((tmp_path / "cache" / "images").glob("computer_use_*.*")) + assert len(captures) == 2 + + class TestBoundsScaleField: def test_scale_reported_when_spaces_diverge(self, tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(tmp_path)) diff --git a/tests/tools/test_rollback_all_directories.py b/tests/tools/test_rollback_all_directories.py new file mode 100644 index 0000000000..66f4f8c635 --- /dev/null +++ b/tests/tools/test_rollback_all_directories.py @@ -0,0 +1,70 @@ +"""Regression tests for the /rollback all-directories fallback (#10505). + +Bare /rollback searched only TERMINAL_CWD's project; checkpoints created +under a different session cwd were invisible ("No checkpoints found for +/home/user" while checkpoints existed seconds earlier). Reapply of PR +#10633 by @nightq onto the v2 single-store layout. +""" + +import json + +import tools.checkpoint_manager as cm +from tools.checkpoint_manager import CheckpointManager, format_checkpoint_list + + +def _make_store_with_project(tmp_path, workdir: str): + store = tmp_path / "store" + (store / cm._PROJECTS_DIRNAME).mkdir(parents=True) + (store / "HEAD").write_text("ref: refs/heads/main\n") + dir_hash = cm._project_hash(workdir) + meta = {"workdir": workdir, "created_at": 1, "last_touch": 2} + (store / cm._PROJECTS_DIRNAME / f"{dir_hash}.json").write_text(json.dumps(meta)) + return store + + +class TestListAllCheckpoints: + def test_aggregates_projects_and_labels_workdir(self, tmp_path, monkeypatch): + workdir = str(tmp_path / "proj") + store = _make_store_with_project(tmp_path, workdir) + monkeypatch.setattr(cm, "CHECKPOINT_BASE", tmp_path) + monkeypatch.setattr(cm, "_store_path", lambda base=None: store) + + mgr = CheckpointManager(enabled=True) + fake_entries = [ + {"hash": "a" * 40, "short_hash": "aaaaaaa", "timestamp": "2026-08-19T10:00:00", + "reason": "before write_file", "files_changed": 1, "insertions": 2, "deletions": 0}, + ] + monkeypatch.setattr( + CheckpointManager, "list_checkpoints", lambda self, wd: list(fake_entries) + ) + + results = mgr.list_all_checkpoints() + assert len(results) == 1 + assert results[0]["workdir"] == workdir + + def test_empty_store_returns_empty(self, tmp_path, monkeypatch): + monkeypatch.setattr(cm, "CHECKPOINT_BASE", tmp_path) + monkeypatch.setattr(cm, "_store_path", lambda base=None: tmp_path / "missing") + assert CheckpointManager(enabled=True).list_all_checkpoints() == [] + + +class TestFormatAllDirectories: + def test_workdir_label_shown_in_all_directories_view(self): + cps = [{ + "hash": "b" * 40, "short_hash": "bbbbbbb", + "timestamp": "2026-08-19T10:00:00", "reason": "before patch", + "files_changed": 0, "insertions": 0, "deletions": 0, + "workdir": "/tmp/qa-repo", + }] + out = format_checkpoint_list(cps, "all directories") + assert "[qa-repo]" in out + + def test_single_directory_view_unchanged(self): + cps = [{ + "hash": "c" * 40, "short_hash": "ccccccc", + "timestamp": "2026-08-19T10:00:00", "reason": "before patch", + "files_changed": 0, "insertions": 0, "deletions": 0, + }] + out = format_checkpoint_list(cps, "/tmp/qa-repo") + assert "[qa-repo]" not in out + assert "ccccccc" in out diff --git a/tests/tui_gateway/test_projects_rpc.py b/tests/tui_gateway/test_projects_rpc.py index 55fa043635..a2f9314828 100644 --- a/tests/tui_gateway/test_projects_rpc.py +++ b/tests/tui_gateway/test_projects_rpc.py @@ -2,11 +2,14 @@ from __future__ import annotations +import contextlib import os import subprocess +from pathlib import Path import pytest +from hermes_constants import reset_hermes_home_override, set_hermes_home_override import tui_gateway.server as server @@ -336,6 +339,139 @@ def test_scan_time_is_not_treated_as_session_activity(tmp_path): assert active["last_active"] > idle["last_active"] +def test_remote_scan_failure_merges_instead_of_replacing_cache(tmp_path, monkeypatch): + """A backend scan that can't fully walk its roots must NOT wipe the cache. + + `projects.discover_repos` with `scan:true` asks the remote host to scan its + own discovery roots. When one root fails to walk, the scan result is not the + authoritative full universe — the previously cached repos must survive so a + failed remote refresh can't blank the sidebar back to the silent, empty + state of #81723 (regression for MEDIUM: `replace=True` was wiping on every + call regardless of success). + """ + from hermes_cli import projects_db as pdb + import tui_gateway.server as server + + def _git_repo(path): + repo = path + repo.mkdir(parents=True) + subprocess.run(["git", "init", "-q"], cwd=repo, check=True) + return str(repo) + + # A cached repo the partial scan never visits, parked OUTSIDE the scan roots. + seed = _git_repo(tmp_path / "elsewhere" / "seed-repo") + with pdb.connect_closing() as conn: + pdb.record_discovered_repos(conn, [(seed, "seed-repo")]) + seeded = [r["root"] for r in pdb.list_discovered_repos(conn)] + assert seed in seeded + + good = _git_repo(tmp_path / "good-repo") + + # Force one root to fail to walk: the scan becomes non-authoritative, so it + # must merge into the cache, never wipe it. + real_walk = os.walk + bad_root = str(tmp_path / "unwalkable") + + def _flaky_walk(top, *a, **k): + if top == bad_root: + raise OSError("boom") + yield from real_walk(top, *a, **k) + + monkeypatch.setattr(server.os, "walk", _flaky_walk) + + policy = {"enabled": True, "roots": [good, bad_root], "exclude_paths": []} + + with pdb.connect_closing() as conn: + authoritative = server._scan_discovered_repos_remote(conn, policy) + joined = [r["root"] for r in pdb.list_discovered_repos(conn)] + + # The scan found the good repo and merged it, but the failed root means the + # result is not authoritative, so it must NOT have replaced the cache. + assert not authoritative + assert good in joined + # The seeded repo that the partial scan never saw is still cached. + assert seed in joined + + +def test_remote_scan_missing_root_does_not_wipe_cache(tmp_path): + """A configured discovery root missing on disk must NOT wipe the cache. + + ``os.walk`` on a non-existent root silently yields nothing instead of + raising, so a temporarily unavailable root (unmounted volume, moved path) + would otherwise make the scan look like a genuinely empty authoritative + set and DELETE-replace every cached repo that lived under it. The missing + root must contribute nothing, and the scan must merge — never wipe. + """ + from hermes_cli import projects_db as pdb + import tui_gateway.server as server + + def _git_repo(path): + repo = path + repo.mkdir(parents=True) + subprocess.run(["git", "init", "-q"], cwd=repo, check=True) + return str(repo) + + # A cached repo the scan never visits, parked OUTSIDE the scan roots. + seed = _git_repo(tmp_path / "elsewhere" / "seed-repo") + with pdb.connect_closing() as conn: + pdb.record_discovered_repos(conn, [(seed, "seed-repo")]) + seeded = [r["root"] for r in pdb.list_discovered_repos(conn)] + assert seed in seeded + + good = _git_repo(tmp_path / "good-repo") + + # This root is configured but does NOT exist on disk. os.walk on it yields + # nothing silently — without the guard the scan would stay authoritative + # and wipe the cache. + missing_root = str(tmp_path / "missing-root") + + policy = {"enabled": True, "roots": [good, missing_root], "exclude_paths": []} + + with pdb.connect_closing() as conn: + authoritative = server._scan_discovered_repos_remote(conn, policy) + joined = [r["root"] for r in pdb.list_discovered_repos(conn)] + + # The missing root is not authoritative, so the scan must merge, not wipe. + assert not authoritative + assert good in joined + # The seeded repo the missing root would have wiped is still cached. + assert seed in joined + + +def test_remote_scan_full_authoritative_replaces_cache(tmp_path): + """Only a fully-walked scan may replace the stale cache.""" + from hermes_cli import projects_db as pdb + import tui_gateway.server as server + + def _git_repo(path): + repo = path + repo.mkdir(parents=True) + subprocess.run(["git", "init", "-q"], cwd=repo, check=True) + return str(repo) + + # Park the stale repo OUTSIDE the scan root so the authoritative scan no + # longer sees it, and the fresh repo inside the root it walks. + stale = _git_repo(tmp_path / "outside" / "stale-repo") + scandir = tmp_path / "scandir" + scandir.mkdir() + fresh = _git_repo(scandir / "fresh-repo") + + with pdb.connect_closing() as conn: + pdb.record_discovered_repos(conn, [(stale, "stale-repo")]) + + policy = {"enabled": True, "roots": [str(scandir)], "exclude_paths": []} + + with pdb.connect_closing() as conn: + authoritative = server._scan_discovered_repos_remote(conn, policy) + joined = [r["root"] for r in pdb.list_discovered_repos(conn)] + + assert authoritative + assert fresh in joined + # A full, authoritative scan replaced the stale cache: the old repo the + # scan no longer saw is gone from the authoritative set. + assert stale not in joined + + def test_terminal_session_persists_its_launch_cwd(): """A terminal session's cwd IS its workspace, so the row must record it. @@ -460,3 +596,229 @@ def test_nondefault_policy_rejects_stale_or_legacy_results(monkeypatch, tmp_path assert any(item["root"] == str(root) for item in accepted["repos"]) +def _profile_dir(tmp_path: Path, name: str) -> Path: + home = tmp_path / "homes" / name + home.mkdir(parents=True, exist_ok=True) + return home + + +def _bind_profiles(monkeypatch, tmp_path: Path, homes: dict[str, Path]) -> None: + """Resolve profile names to this test's throwaway homes. + + Unmapped names resolve to a path that does not exist, which is how the + gateway detects "not a real profile on this host" and stays on launch. + """ + monkeypatch.setattr( + "hermes_cli.profiles.get_profile_dir", + lambda name: homes.get(name, tmp_path / "homes" / "missing" / name), + ) + + +def _create_project(home: Path, name: str, folder: Path, *, use: bool = False) -> dict: + """Create a project in ``home``'s projects.db via the real RPC.""" + token = set_hermes_home_override(home) + try: + return _call( + "projects.create", {"name": name, "folders": [str(folder)], "use": use} + )["project"] + finally: + reset_hermes_home_override(token) + + +def _create_session(home: Path, session_id: str, cwd: Path) -> None: + """Seed one message-bearing session in ``home``'s state.db.""" + from hermes_state import SessionDB + + db = SessionDB(db_path=home / "state.db") + try: + db.create_session(session_id, "cli", cwd=str(cwd)) + db.append_message(session_id, "user", f"hello from {session_id}") + finally: + db.close() + + +@contextlib.contextmanager +def _serving_launch_profile(launch_home: Path): + """Run the handlers as a backend launched under ``launch_home``.""" + from hermes_state import SessionDB + + token = set_hermes_home_override(launch_home) + prev_db, prev_error = server._db, server._db_error + server._db = SessionDB(db_path=launch_home / "state.db") + server._db_error = None + try: + yield + finally: + server._db.close() + server._db, server._db_error = prev_db, prev_error + reset_hermes_home_override(token) + + +def _cached_repo_labels(home: Path) -> list[str]: + """Labels in ``home``'s discovered-repo cache, read straight off disk.""" + from hermes_cli import projects_db as pdb + + with pdb.connect_closing(home / "projects.db") as conn: + return sorted(str(entry.get("label") or "") for entry in pdb.list_discovered_repos(conn)) + + +def test_projects_reads_are_scoped_to_the_requested_profile(monkeypatch, tmp_path): + """A ``profile`` param reads that profile's projects.db AND its state.db.""" + launch_home = _profile_dir(tmp_path, "launch") + coder_home = _profile_dir(tmp_path, "coder") + launch_repo = tmp_path / "repos" / "launch-repo" + coder_repo = tmp_path / "repos" / "coder-repo" + launch_repo.mkdir(parents=True) + coder_repo.mkdir(parents=True) + _bind_profiles(monkeypatch, tmp_path, {"default": launch_home, "coder": coder_home}) + + launch_project = _create_project(launch_home, "Launch", launch_repo, use=True) + coder_project = _create_project(coder_home, "Coder", coder_repo, use=True) + _create_session(launch_home, "launch-session", launch_repo) + _create_session(coder_home, "coder-session", coder_repo) + + with _serving_launch_profile(launch_home): + launch_listing = _call("projects.list") + coder_listing = _call("projects.list", {"profile": "coder"}) + launch_tree = _call("projects.tree") + coder_tree = _call("projects.tree", {"profile": "coder"}) + coder_sessions = _call( + "projects.project_sessions", + {"profile": "coder", "project_id": coder_project["id"]}, + ) + # The override must not leak: the very next unscoped read is launch again. + launch_again = _call("projects.list") + + assert [p["name"] for p in launch_listing["projects"]] == ["Launch"] + assert [p["name"] for p in coder_listing["projects"]] == ["Coder"] + assert launch_listing["active_id"] == launch_project["id"] + assert coder_listing["active_id"] == coder_project["id"] + assert launch_again == launch_listing + + assert [p["label"] for p in launch_tree["projects"]] == ["Launch"] + assert [p["label"] for p in coder_tree["projects"]] == ["Coder"] + # Session counts prove the SESSION db was swapped too, not just projects.db. + assert launch_tree["projects"][0]["sessionCount"] == 1 + assert coder_tree["projects"][0]["sessionCount"] == 1 + assert launch_tree["scoped_session_ids"] == ["launch-session"] + assert coder_tree["scoped_session_ids"] == ["coder-session"] + + assert coder_sessions["project"]["id"] == coder_project["id"] + assert coder_sessions["project"]["sessionCount"] == 1 + lane = coder_sessions["project"]["repos"][0]["groups"][0] + assert [s["id"] for s in lane["sessions"]] == ["coder-session"] + + +def test_projects_tree_is_scoped_to_the_requested_profile(monkeypatch, tmp_path): + """``projects.tree`` on its own reads the requested profile's stores.""" + launch_home = _profile_dir(tmp_path, "launch") + coder_home = _profile_dir(tmp_path, "coder") + launch_repo = tmp_path / "repos" / "tree-launch" + coder_repo = tmp_path / "repos" / "tree-coder" + launch_repo.mkdir(parents=True) + coder_repo.mkdir(parents=True) + _bind_profiles(monkeypatch, tmp_path, {"default": launch_home, "coder": coder_home}) + + _create_project(launch_home, "Launch", launch_repo, use=True) + _create_project(coder_home, "Coder", coder_repo, use=True) + _create_session(launch_home, "tree-launch-session", launch_repo) + _create_session(coder_home, "tree-coder-session", coder_repo) + + with _serving_launch_profile(launch_home): + coder_tree = _call("projects.tree", {"profile": "coder"}) + launch_tree = _call("projects.tree") + + assert [p["label"] for p in coder_tree["projects"]] == ["Coder"] + assert coder_tree["scoped_session_ids"] == ["tree-coder-session"] + assert [p["label"] for p in launch_tree["projects"]] == ["Launch"] + assert launch_tree["scoped_session_ids"] == ["tree-launch-session"] + + +def test_project_sessions_is_scoped_to_the_requested_profile(monkeypatch, tmp_path): + """``projects.project_sessions`` on its own hydrates from the requested profile.""" + launch_home = _profile_dir(tmp_path, "launch") + coder_home = _profile_dir(tmp_path, "coder") + launch_repo = tmp_path / "repos" / "drill-launch" + coder_repo = tmp_path / "repos" / "drill-coder" + launch_repo.mkdir(parents=True) + coder_repo.mkdir(parents=True) + _bind_profiles(monkeypatch, tmp_path, {"default": launch_home, "coder": coder_home}) + + launch_project = _create_project(launch_home, "Launch", launch_repo, use=True) + coder_project = _create_project(coder_home, "Coder", coder_repo, use=True) + _create_session(launch_home, "drill-launch-session", launch_repo) + _create_session(coder_home, "drill-coder-session", coder_repo) + + with _serving_launch_profile(launch_home): + coder_drill = _call( + "projects.project_sessions", + {"profile": "coder", "project_id": coder_project["id"]}, + ) + launch_drill = _call( + "projects.project_sessions", {"project_id": launch_project["id"]} + ) + + assert coder_drill["project"] is not None + assert coder_drill["project"]["id"] == coder_project["id"] + coder_lane = coder_drill["project"]["repos"][0]["groups"][0] + assert [s["id"] for s in coder_lane["sessions"]] == ["drill-coder-session"] + + assert launch_drill["project"]["id"] == launch_project["id"] + launch_lane = launch_drill["project"]["repos"][0]["groups"][0] + assert [s["id"] for s in launch_lane["sessions"]] == ["drill-launch-session"] + + +def test_record_repos_writes_to_the_requested_profiles_projects_db(monkeypatch, tmp_path): + """The scan cache is per-profile: a scoped write must not land on launch.""" + launch_home = _profile_dir(tmp_path, "launch") + coder_home = _profile_dir(tmp_path, "coder") + launch_repo = tmp_path / "repos" / "launch-scan" + coder_repo = tmp_path / "repos" / "coder-scan" + launch_repo.mkdir(parents=True) + coder_repo.mkdir(parents=True) + _bind_profiles(monkeypatch, tmp_path, {"default": launch_home, "coder": coder_home}) + + with _serving_launch_profile(launch_home): + _call("projects.record_repos", {"repos": [{"root": str(launch_repo), "label": "launch"}]}) + _call( + "projects.record_repos", + {"profile": "coder", "repos": [{"root": str(coder_repo), "label": "coder"}]}, + ) + + launch_repos = _call("projects.discover_repos")["repos"] + coder_repos = _call("projects.discover_repos", {"profile": "coder"})["repos"] + + assert [repo["label"] for repo in launch_repos] == ["launch"] + assert [repo["label"] for repo in coder_repos] == ["coder"] + assert _cached_repo_labels(launch_home) == ["launch"] + assert _cached_repo_labels(coder_home) == ["coder"] + + +def test_projects_without_a_profile_stay_on_the_launch_home(monkeypatch, tmp_path): + """Omitted/blank/unknown profile is a no-op — the pre-scoping behavior.""" + launch_home = _profile_dir(tmp_path, "launch") + coder_home = _profile_dir(tmp_path, "coder") + repo = tmp_path / "repos" / "launch-only" + repo.mkdir(parents=True) + _bind_profiles(monkeypatch, tmp_path, {"default": launch_home, "coder": coder_home}) + + with _serving_launch_profile(launch_home): + created = _call( + "projects.create", {"name": "Launch only", "folders": [str(repo)], "use": True} + )["project"] + _call("projects.record_repos", {"repos": [{"root": str(repo), "label": "only"}]}) + + omitted = _call("projects.list") + blank = _call("projects.list", {"profile": ""}) + unknown = _call("projects.list", {"profile": "not-a-profile"}) + + assert [p["name"] for p in omitted["projects"]] == ["Launch only"] + assert blank == omitted + assert unknown == omitted + assert omitted["active_id"] == created["id"] + + assert _cached_repo_labels(launch_home) == ["only"] + assert not (coder_home / "projects.db").exists() + assert not (Path(os.environ["HERMES_HOME"]) / "projects.db").exists() + + diff --git a/tools/checkpoint_manager.py b/tools/checkpoint_manager.py index 5448632103..7149eebc90 100644 --- a/tools/checkpoint_manager.py +++ b/tools/checkpoint_manager.py @@ -968,6 +968,29 @@ class CheckpointManager: results.append(entry) return results + def list_all_checkpoints(self) -> List[Dict]: + """List checkpoints across every registered project (most recent first). + + Surgical reapply of PR #10633 by @nightq (#10505) onto the v2 + single-store layout: iterate ``projects/.json`` metadata via + ``_list_projects`` instead of the pre-v2 per-shadow-dir scan. Each + entry carries the extra ``workdir`` key so callers can label which + project a checkpoint belongs to. + """ + store = _store_path(CHECKPOINT_BASE) + if not (store / "HEAD").exists(): + return [] + results: List[Dict] = [] + for meta in _list_projects(store): + workdir = meta.get("workdir") or "" + if not workdir: + continue + for entry in self.list_checkpoints(workdir): + entry["workdir"] = workdir + results.append(entry) + results.sort(key=lambda x: x.get("timestamp", ""), reverse=True) + return results + @staticmethod def _parse_shortstat(stat_line: str, entry: Dict) -> None: """Parse git --shortstat output into entry dict.""" @@ -1563,7 +1586,16 @@ def format_checkpoint_list(checkpoints: List[Dict], directory: str) -> str: else: stat = "" - lines.append(f" {i}. {cp['short_hash']} {ts} {cp['reason']}{stat}") + # Label per-project entries when showing the cross-project view + # (workdir key only present on list_all_checkpoints results). + workdir = cp.get("workdir", "") + if workdir and directory == "all directories": + workdir_short = Path(workdir).name or workdir + lines.append( + f" {i}. {cp['short_hash']} {ts} [{workdir_short}] {cp['reason']}{stat}" + ) + else: + lines.append(f" {i}. {cp['short_hash']} {ts} {cp['reason']}{stat}") lines.append("\n /rollback restore to checkpoint N") lines.append(" /rollback diff preview changes since checkpoint N") diff --git a/tools/computer_use/schema.py b/tools/computer_use/schema.py index 388a4778a9..1c0d322e41 100644 --- a/tools/computer_use/schema.py +++ b/tools/computer_use/schema.py @@ -22,7 +22,11 @@ COMPUTER_USE_SCHEMA: Dict[str, Any] = { "Preferred workflow: call with " "action='capture' (mode='som' gives numbered element overlays), " "then click by `element` index for reliability. Pixel coordinates " - "are supported for models trained on them. Works on any window — " + "are supported for models trained on them. Image captures include a " + "shareable `screenshot_path`; when the user asks to receive the image " + "and the current surface supports attachments, deliver that file using " + "the platform's native MEDIA attachment syntax. Do not automatically " + "send screenshots used only for computer control. Works on any window — " "hidden, minimized, or behind another app. Requires cua-driver to " "be installed." ), diff --git a/tools/computer_use/tool.py b/tools/computer_use/tool.py index 951cf9715c..ab2824b8d0 100644 --- a/tools/computer_use/tool.py +++ b/tools/computer_use/tool.py @@ -1195,6 +1195,11 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME or image_dimensions[1] < _MIN_PROVIDER_IMAGE_DIMENSION ) ) + screenshot_path = ( + _persist_capture_image(cap) + if cap.png_b64 and cap.mode != "ax" and not image_too_small + else None + ) # Index only what's actually surfaced in the response — otherwise the # human-readable summary references element indices the model cannot @@ -1209,6 +1214,10 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME ] if bounds_note: summary_lines.append(f" ({bounds_note})") + if screenshot_path: + summary_lines.append( + f" (shareable screenshot saved to {screenshot_path})" + ) if elements_file: summary_lines.append( f" (full element tree with untruncated labels saved to " @@ -1243,6 +1252,7 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME visible_elements=visible_elements, truncated_elements=truncated_elements, elements_file=elements_file, + screenshot_path=screenshot_path, ) if routed is not None: return routed @@ -1278,6 +1288,8 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME payload["truncated_elements"] = truncated_elements if elements_file: payload["elements_file"] = elements_file + if screenshot_path: + payload["screenshot_path"] = screenshot_path if bounds_scale: payload["bounds_scale"] = bounds_scale return json.dumps(payload) @@ -1303,9 +1315,10 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME ], "text_summary": summary, "meta": {"mode": cap.mode, "width": response_width, "height": response_height, - "elements": total_elements, "png_bytes": cap.png_bytes_len, - **({"elements_file": elements_file} if elements_file else {}), - **({"bounds_scale": bounds_scale} if bounds_scale else {})}, + "elements": total_elements, "png_bytes": cap.png_bytes_len, + **({"screenshot_path": screenshot_path} if screenshot_path else {}), + **({"elements_file": elements_file} if elements_file else {}), + **({"bounds_scale": bounds_scale} if bounds_scale else {})}, } # AX-only (or image-missing fallback): text path actually carries the # `elements` array, so the truncation note applies here. @@ -1446,6 +1459,7 @@ def _route_capture_through_aux_vision( visible_elements: Optional[List[UIElement]] = None, truncated_elements: int = 0, elements_file: Optional[str] = None, + screenshot_path: Optional[str] = None, ) -> Optional[str]: """Pre-analyse the captured PNG via ``vision_analyze`` and return a text result. @@ -1559,6 +1573,8 @@ def _route_capture_through_aux_vision( payload["truncated_elements"] = truncated_elements if elements_file: payload["elements_file"] = elements_file + if screenshot_path: + payload["screenshot_path"] = screenshot_path return json.dumps(payload) @@ -1648,6 +1664,53 @@ _MAX_ELEMENT_LABEL_CHARS = 120 # capture of a dense UI can spill; without pruning the cache grows unbounded. _MAX_SPILL_FILES = 20 +# Keep user-shareable capture files bounded independently from the gateway's +# periodic media-cache cleanup. CLI-only sessions may never start the gateway, +# and capture_after can otherwise leave an unbounded screenshot trail. +_MAX_CAPTURE_FILES = 20 + + +def _persist_capture_image(cap: CaptureResult) -> Optional[str]: + """Save a capture in Hermes' media cache and return its absolute path. + + Captures are normally embedded only in the model's tool context. Persisting + a bounded copy gives attachment-capable surfaces a real file to deliver + when the user explicitly asks for the screenshot. This is best-effort: an + unwritable cache must never break computer control. + """ + if not cap.png_b64: + return None + try: + import uuid as _uuid + + from hermes_constants import get_hermes_dir + + raw = base64.b64decode(cap.png_b64, validate=False) + mime = str(cap.image_mime_type or "").lower() + ext = ".jpg" if mime == "image/jpeg" or ( + not mime and cap.png_b64[:8].startswith("/9j/") + ) else ".png" + + cache_dir = get_hermes_dir("cache/images", "image_cache") + cache_dir.mkdir(parents=True, exist_ok=True) + try: + captures = sorted( + cache_dir.glob("computer_use_*.*"), + key=lambda path: path.stat().st_mtime, + ) + keep_before_write = max(0, _MAX_CAPTURE_FILES - 1) + for stale in captures[: max(0, len(captures) - keep_before_write)]: + stale.unlink(missing_ok=True) + except Exception: + pass + + path = cache_dir / f"computer_use_{_uuid.uuid4().hex}{ext}" + path.write_bytes(raw) + return str(path) + except Exception as exc: # pragma: no cover - defensive + logger.debug("computer_use: screenshot persistence failed: %s", exc) + return None + def _spill_elements_to_file(cap: CaptureResult) -> Optional[str]: """Write the FULL element tree (untruncated labels) to a cache file. diff --git a/tools/memory_tool.py b/tools/memory_tool.py index 44effd02c2..c2ba72dc0e 100644 --- a/tools/memory_tool.py +++ b/tools/memory_tool.py @@ -1143,9 +1143,34 @@ def memory_tool( return json.dumps(result, ensure_ascii=False) +def builtin_memory_stores_enabled() -> bool: + """Return whether either built-in store (MEMORY.md / USER.md) is enabled. + + ``agent_init`` only builds a ``MemoryStore`` when at least one of + ``memory.memory_enabled`` / ``memory.user_profile_enabled`` is true, so with + both off the tool dispatches against ``store=None`` and every call fails + with "Memory is not available". + + Fails open when config can't be read: an unreadable config must not strip a + tool that would otherwise work. + """ + try: + from hermes_cli.config import load_config_readonly + + section = (load_config_readonly() or {}).get("memory") + if not isinstance(section, dict): + return True + return bool(section.get("memory_enabled", True)) or bool( + section.get("user_profile_enabled", True) + ) + except Exception: + logger.debug("Could not read memory config for availability", exc_info=True) + return True + + def check_memory_requirements() -> bool: - """Memory tool has no external requirements -- always available.""" - return True + """Available unless both built-in memory stores are disabled in config.""" + return builtin_memory_stores_enabled() def apply_memory_pending(payload: Dict[str, Any], store: "MemoryStore") -> Dict[str, Any]: diff --git a/tui_gateway/methods_config.py b/tui_gateway/methods_config.py index 53ccb5d91c..314b38d904 100644 --- a/tui_gateway/methods_config.py +++ b/tui_gateway/methods_config.py @@ -17,31 +17,40 @@ _profile_scoped = _registry.profile_scoped @method("projects.discover_repos") +@_profile_scoped def _(rid, params: dict) -> dict: """Repos for the desktop overview: scanned-from-disk (cached) ∪ session-derived.""" try: - db = _get_db() - if db is None: - return _ok(rid, {"repos": []}) - from hermes_cli import projects_db as pdb + with _profile_db(params) as db: + if db is None: + return _ok(rid, {"repos": []}) + from hermes_cli import projects_db as pdb - policy = _repo_discovery_policy() - policy_key = _repo_discovery_policy_key(policy) - with pdb.connect_closing() as conn: - pdb.reconcile_discovered_repos_policy( - conn, - policy_key, - preserve_unversioned=_repo_discovery_policy_is_default(policy), - ) - repos = _discover_repos_payload( - db, conn=conn, include_cached=policy["enabled"] - ) - return _ok(rid, {"repos": repos, "discovery_policy": policy}) + policy = _repo_discovery_policy() + policy_key = _repo_discovery_policy_key(policy) + with pdb.connect_closing() as conn: + pdb.reconcile_discovered_repos_policy( + conn, + policy_key, + preserve_unversioned=_repo_discovery_policy_is_default(policy), + ) + # `scan=true` (set by the desktop in remote-gateway mode): run a + # backend-side filesystem scan of the policy roots so repos with + # zero Hermes sessions still surface. The desktop's native scan + # only runs on the local filesystem; on a remote connection it + # must ask the host to scan itself (#81723). + if params.get("scan") and policy["enabled"]: + _scan_discovered_repos_remote(conn, policy) + repos = _discover_repos_payload( + db, conn=conn, include_cached=policy["enabled"] + ) + return _ok(rid, {"repos": repos, "discovery_policy": policy}) except Exception as e: return _err(rid, 5061, str(e)) @method("projects.record_repos") +@_profile_scoped def _(rid, params: dict) -> dict: """Persist git repo roots found by the client's filesystem scan, then return the merged repo list. The native crawl runs on the desktop (local fs); this @@ -88,24 +97,25 @@ def _(rid, params: dict) -> dict: elif not policy["enabled"]: pdb.clear_discovered_repos(conn, policy_key=policy_key) - db = _get_db() - return _ok( - rid, - { - "repos": _discover_repos_payload( - db, include_cached=policy["enabled"] - ) - if db is not None - else [], - "accepted": accepted, - "discovery_policy": policy, - }, - ) + with _profile_db(params) as db: + return _ok( + rid, + { + "repos": _discover_repos_payload( + db, include_cached=policy["enabled"] + ) + if db is not None + else [], + "accepted": accepted, + "discovery_policy": policy, + }, + ) except Exception as e: return _err(rid, 5061, str(e)) @method("projects.tree") +@_profile_scoped def _(rid, params: dict) -> dict: """Authoritative project overview: project -> repo -> lane structure with counts + a few preview sessions per project, plus the flat set of session @@ -113,26 +123,27 @@ def _(rid, params: dict) -> dict: Lanes carry no session rows here; drill-in uses ``projects.project_sessions``. """ try: - db = _get_db() - if db is None: - return _ok(rid, {"projects": [], "active_id": None, "scoped_session_ids": []}) + with _profile_db(params) as db: + if db is None: + return _ok(rid, {"projects": [], "active_id": None, "scoped_session_ids": []}) - tree, active_id = _build_project_tree( - db, - preview_limit=int(params.get("preview_limit") or 3), - hydrate=False, - session_limit=int(params.get("session_limit") or 2000), - include_discovered=True, - ) - return _ok( - rid, - {"projects": tree["projects"], "active_id": active_id, "scoped_session_ids": tree["scoped_session_ids"]}, - ) + tree, active_id = _build_project_tree( + db, + preview_limit=int(params.get("preview_limit") or 3), + hydrate=False, + session_limit=int(params.get("session_limit") or 2000), + include_discovered=True, + ) + return _ok( + rid, + {"projects": tree["projects"], "active_id": active_id, "scoped_session_ids": tree["scoped_session_ids"]}, + ) except Exception as e: return _err(rid, 5061, str(e)) @method("projects.project_sessions") +@_profile_scoped def _(rid, params: dict) -> dict: """Fully hydrated lanes (repo -> lane -> session rows) for one project, built from the same authoritative grouping as ``projects.tree`` so ids and @@ -142,18 +153,18 @@ def _(rid, params: dict) -> dict: if not project_id: return _err(rid, 5063, "project_id required") - db = _get_db() - if db is None: - return _ok(rid, {"project": None}) + with _profile_db(params) as db: + if db is None: + return _ok(rid, {"project": None}) - # Drill-in only needs the entered project (which has sessions), so skip - # the zero-session discovery tier entirely. - tree, _active = _build_project_tree( - db, preview_limit=0, hydrate=True, session_limit=int(params.get("session_limit") or 5000), - include_discovered=False, - ) - proj = next((p for p in tree["projects"] if p["id"] == project_id), None) - return _ok(rid, {"project": proj}) + # Drill-in only needs the entered project (which has sessions), so skip + # the zero-session discovery tier entirely. + tree, _active = _build_project_tree( + db, preview_limit=0, hydrate=True, session_limit=int(params.get("session_limit") or 5000), + include_discovered=False, + ) + proj = next((p for p in tree["projects"] if p["id"] == project_id), None) + return _ok(rid, {"project": proj}) except Exception as e: return _err(rid, 5061, str(e)) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 621032da67..6bb356201c 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -1534,13 +1534,12 @@ def _profile_home(profile: str | None) -> Path | None: def _profile_scoped(handler): - """Bind ``params['profile']``'s HERMES_HOME around a pet RPC handler. + """Bind ``params['profile']``'s HERMES_HOME around a handler. - Pets are per-profile: ``display.pet.*`` lives in the profile's config.yaml and - sprites install under its ``pets/`` dir (both resolve via ``get_hermes_home``). - The desktop sends ``profile`` on pet calls so config + pets dir resolve to the - focused profile even in app-global remote mode, where one backend serves every - profile. No-op for the launch profile (own-profile backends already resolve it). + Pets (config + sprites) and projects (projects.db, discovery policy) both + resolve via ``get_hermes_home``. The desktop sends ``profile`` so a single + backend serving every profile in app-global remote mode still hits the + focused profile's home. No-op for the launch profile. """ def wrapper(rid, params): @@ -12634,13 +12633,14 @@ def _projects_payload(conn) -> dict: def _projects_method(name: str): """Register a projects RPC, injecting (pdb, conn) and unifying error mapping. - Every project CRUD handler opened the per-profile DB, mapped a missing id to - 5062, bad args to 5063, and everything else to 5061. This collapses that - boilerplate so each handler is just its one meaningful operation. + Binds ``params['profile']`` (via ``@_profile_scoped``) so app-global remote + mode reads that profile's ``projects.db``. Missing id maps to 5062, bad args + to 5063, everything else to 5061. """ def decorator(fn): @method(name) + @_profile_scoped def handler(rid, params: dict) -> dict: try: from hermes_cli import projects_db as pdb @@ -12879,6 +12879,89 @@ def _repo_discovery_policy_is_default(policy: dict) -> bool: ) +def _scan_discovered_repos_remote(conn, policy: dict) -> bool: + """Backend-side disk scan of the discovery policy roots. + + The desktop's native repo scan only runs on the local filesystem. On a + remote gateway connection the host must scan its own disk so repos with + zero Hermes sessions still appear in the sidebar (#81723). Mirrors the + desktop's behavior: walk each root (bounded depth), find `.git` + directories, record (root, label) pairs into the discovery cache. + + Best-effort: any failure logs and leaves the cache untouched — the + session-derived repos from `_discover_repos_payload` still surface. + + Returns True when the scan is authoritative (every root was walked to + completion without error and the per-scan cap was not hit). Only then may + the caller treat the result as a full replacement and pass ``replace=True`` + to the cache write — a partial or errored scan must merge, never wipe, so + a failed remote refresh can't blank the previously cached repos into the + silent, unpopulated sidebar of #81723. + """ + from hermes_cli import projects_db as pdb + + roots = policy.get("roots") or [] + excludes = policy.get("exclude_paths") or [] + pairs: list[tuple[str, str | None]] = [] + seen: set[str] = set() + authoritative = True + + def _is_excluded(path: str) -> bool: + return any(path == ex or path.startswith(ex.rstrip("/\\") + os.sep) for ex in excludes if ex) + + for root in roots: + if not os.path.isdir(root): + # `os.walk` on a missing root silently yields nothing instead of + # raising, so a temporarily unavailable root (unmounted volume, + # moved path) would otherwise look like a genuinely empty scan and + # let `authoritative` stay True — letting the replace wipe every + # cached repo that lived under the missing root. A missing root + # contributes nothing and must not be treated as authoritative. + authoritative = False + logger.debug("discover_repos scan root missing, skipping: %s", root) + continue + try: + for dirpath, dirnames, _filenames in os.walk(root): + if _is_excluded(dirpath): + dirnames[:] = [] + continue + # A `.git` directory marks this directory as a repo root. Check + # BEFORE pruning hidden dirs — `.git` is itself hidden, so a + # prune-first order would drop it and never detect any repo. + if ".git" in dirnames: + repo_root = dirpath + if repo_root not in seen: + seen.add(repo_root) + pairs.append((repo_root, os.path.basename(repo_root))) + # Don't descend into the repo's own .git to hunt nested repos. + dirnames[:] = [] + else: + # Not a repo: skip hidden dirs (e.g. .hermes) and node_modules. + dirnames[:] = [d for d in dirnames if not d.startswith(".") and d not in ("node_modules",)] + if len(pairs) >= 500: + break + except Exception: + # A root that can't be walked yields no authoritative set — fall back + # to merging, never replacing, so the prior cache survives. + authoritative = False + logger.debug("discover_repos scan failed for root %s", root, exc_info=True) + if len(pairs) >= 500: + # Cap hit means the walk didn't cover the full roots; the collected + # set must not be treated as the complete authoritative universe. + authoritative = False + break + + if pairs: + try: + pdb.record_discovered_repos( + conn, pairs, replace=authoritative, policy_key=_repo_discovery_policy_key(policy) + ) + except Exception: + logger.debug("discover_repos cache write failed", exc_info=True) + authoritative = False + return authoritative + + def _discover_repos_payload( db, *, conn=None, backfill: bool = True, include_cached: bool = True ) -> list[dict]: diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index 624465821e..cf575508ab 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -240,6 +240,20 @@ memory: write_approval: false # false = write freely (default) | true = require approval ``` +Setting **both** `memory_enabled` and `user_profile_enabled` to `false` turns the +built-in stores off completely: the `memory` tool is dropped from the schema and +its guidance block is dropped from the system prompt, so the model is never told +about a tool it cannot use. An external provider set via `memory.provider` +(Hindsight, Mem0, Honcho, …) is unaffected and keeps its own tools — use this +when you want a third-party memory backend *instead of* the built-in files. +Listing `memory` under `agent.disabled_toolsets` is the heavier switch: it hides +external provider tools too. + +With only `memory_enabled: false` (user profile still on), the tool stays — +it backs the profile store — but the system prompt swaps the full memory +guidance for a narrower profile-only block, so the model is only instructed to +save user-profile facts and never steered at the disabled notes store. + ## Controlling memory writes (`write_approval`) By default the agent saves memory freely — including from the background