From 1abc7ce8f059aadb60466fe9400913c55bd90b92 Mon Sep 17 00:00:00 2001 From: rroverin Date: Wed, 12 Aug 2026 18:32:57 +0200 Subject: [PATCH 001/426] feat(models): add Nemotron 3.5 Lightning 30B-A3B to NVIDIA NIM picker --- hermes_cli/models.py | 1 + 1 file changed, 1 insertion(+) diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 3a809c5f1a..e2dc50c3cb 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -341,6 +341,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # NVIDIA flagship reasoning models "nvidia/nemotron-3-ultra-550b-a55b", "nvidia/nemotron-3-super-120b-a12b", + "nvidia/nemotron-3.5-lightning-30b-a3b", "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", # Third-party agentic models hosted on build.nvidia.com # (map to OpenRouter defaults — users get familiar picks on NIM) From ef9cb96b1a4a6e508211dce5037cb16396841a68 Mon Sep 17 00:00:00 2001 From: yoniebans Date: Thu, 13 Aug 2026 17:26:25 +0200 Subject: [PATCH 002/426] fix(desktop): re-emit session.info when approvals config changes out of band The desktop settings page saves approvals.mode through REST PUT /api/config (and the raw editor through PUT /api/config/raw). Enforcement follows the file immediately, because the approval gate re-reads config per command, but every live session's YOLO/approval indicator repaints only on a session.info event, and the REST save emitted nothing. The indicator showed bypass OFF while approvals.mode=off silently auto-approved every dangerous command, and switching sessions repainted the stale cached per-session state, making the toggle look like it flipped itself back. The gateway /approvals slash command had the same gap. The config.set RPC handler already re-emits session.info to all live sessions after a mode flip; give the other writers the same contract: - tui_gateway/server.py: add broadcast_session_info(), which snapshots _sessions under _sessions_lock and re-emits via _emit_session_info_for_session. Also call it from the /approvals slash mirror when a mode argument was persisted (bare /approvals is read-only). - hermes_cli/web_server.py: after a REST save that actually changed the normalized approvals.mode, call the broadcast through a sys.modules guard (no gateway imported means no sessions to notify). The comparison runs on the in-memory documents (existing vs merged, parsed vs raw): the settings page PUTs the defaulted GET record while disk holds sparse YAML, so a block-level compare would broadcast on every autosave, and re-reading through the config cache after the save could serve the pre-save document on an (mtime_ns, size) key collision. Own-profile saves only: a profile-scoped save targets a different HERMES_HOME than this process's gateway sessions. No broadcast on saves that leave the effective mode unchanged, so settings autosave churn (skin, font, TTS) can't spam session.info. Scope: reaches sessions of the in-process gateway (hermes serve / hermes dashboard, the topologies the desktop app talks to). A spawned tui_gateway.entry child gateway has its own process and _sessions; its TUI statusbar reconciles each turn via the existing session.info emissions. --- hermes_cli/web_server.py | 63 +++++- .../test_web_server_approvals_broadcast.py | 213 ++++++++++++++++++ tui_gateway/server.py | 18 ++ 3 files changed, 293 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_web_server_approvals_broadcast.py diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 701c5662d6..443c58b897 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -7025,6 +7025,7 @@ def _denormalize_config_from_web(config: Dict[str, Any]) -> Dict[str, Any]: @app.put("/api/config") async def update_config(body: ConfigUpdate, profile: Optional[str] = None): def _run(): + approvals_mode_changed = False with _profile_scope(body.profile or profile): # The dashboard form is schema-driven (see CONFIG_SCHEMA). Any root # key absent from the schema — most visibly ``custom_providers``, but @@ -7035,7 +7036,23 @@ async def update_config(body: ConfigUpdate, profile: Optional[str] = None): with _CONFIG_MUTATION_LOCK: existing = read_raw_config() incoming = _denormalize_config_from_web(body.config) - save_config(_deep_merge(existing, incoming)) + merged = _deep_merge(existing, incoming) + # Compare normalized approvals.mode across the in-memory + # documents, not config blocks and not cache re-reads: the + # settings page PUTs the defaulted GET record while disk + # holds sparse YAML, so a block compare is always-unequal + # (every autosave would broadcast), and reloading after the + # save can serve the pre-save cache on an (mtime_ns, size) + # key collision. Only approvals.mode feeds session.info, so + # it is the honest trigger. + approvals_mode_changed = _approval_mode_of(merged) != _approval_mode_of(existing) + save_config(merged) + # REST saves bypass the config.set RPC (which re-emits itself), so + # refresh live sessions' cached approval/YOLO indicators after a mode + # change. Own-profile saves only: a profile-scoped save targets a + # different HERMES_HOME than this process's gateway sessions. + if approvals_mode_changed and not _is_other_profile(body.profile or profile): + _broadcast_gateway_session_info() return {"ok": True} try: @@ -7047,6 +7064,45 @@ async def update_config(body: ConfigUpdate, profile: Optional[str] = None): raise HTTPException(status_code=500, detail="Internal server error") +def _is_other_profile(profile: Optional[str]) -> bool: + """True when ``profile`` names a profile other than this process's own.""" + requested = (profile or "").strip() + return bool(requested) and requested.lower() != "current" + + +def _approval_mode_of(config: Dict[str, Any]) -> str: + """Normalize approvals.mode from an in-memory config document. + + Both sides of the broadcast comparison use in-memory documents (the raw + on-disk dict and the about-to-be-saved dict): re-reading through the + config cache after a save can serve the pre-save document when the + replacement file collides on the (mtime_ns, size) cache key, which would + suppress the broadcast exactly when the mode changed. Absent block or + key normalizes to the same default the approval gate uses. + """ + from tools.approval import _normalize_approval_mode + + approvals = config.get("approvals") + default_mode = (DEFAULT_CONFIG.get("approvals") or {}).get("mode", "manual") + mode = approvals.get("mode", default_mode) if isinstance(approvals, dict) else default_mode + return _normalize_approval_mode(mode) + + +def _broadcast_gateway_session_info() -> None: + """Broadcast session.info on the in-process gateway when it's loaded. + + ``sys.modules`` guard, not an import: gateway never imported means no + live sessions in this process to notify. + """ + server = sys.modules.get("tui_gateway.server") + if server is None: + return + try: + server.broadcast_session_info() + except Exception: + _log.exception("session.info broadcast after config save failed") + + def _catalog_provider_env_metadata() -> dict: """Map provider env vars → desktop card metadata, derived from the catalog. @@ -14408,10 +14464,15 @@ async def update_config_raw(body: RawConfigUpdate, profile: Optional[str] = None parsed = yaml.safe_load(body.yaml_text) if not isinstance(parsed, dict): raise HTTPException(status_code=400, detail="YAML must be a mapping") + approvals_mode_changed = False with _profile_scope(body.profile or profile): # Full-document replacement: the editor owns the whole file; do not # merge omitted sections back from disk (#62723). + approvals_mode_changed = _approval_mode_of(parsed) != _approval_mode_of(read_raw_config()) save_config(parsed, merge_existing=False) + # Same indicator refresh as the schema-driven save above. + if approvals_mode_changed and not _is_other_profile(body.profile or profile): + _broadcast_gateway_session_info() return {"ok": True} try: diff --git a/tests/hermes_cli/test_web_server_approvals_broadcast.py b/tests/hermes_cli/test_web_server_approvals_broadcast.py new file mode 100644 index 0000000000..643654d140 --- /dev/null +++ b/tests/hermes_cli/test_web_server_approvals_broadcast.py @@ -0,0 +1,213 @@ +"""Approvals config saves must re-emit session.info to live gateway sessions. + +Regression for the desktop "YOLO toggle does nothing / flips back off" bug: +the settings page saves ``approvals.mode`` through REST ``PUT /api/config`` +(and the raw editor through ``PUT /api/config/raw``), which wrote config.yaml +and emitted nothing. Enforcement follows the file immediately (the approval +gate re-reads config per command), but every live session's YOLO/approval +indicator repaints only on a ``session.info`` event, so the UI kept showing +stale bypass state, in the dangerous direction. The ``config.set`` RPC path +already re-emits after a mode flip; these tests pin the REST paths to the +same contract. +""" + +import types + +import pytest + + +@pytest.fixture +def client(_isolate_hermes_home): + try: + from starlette.testclient import TestClient + except ImportError: + pytest.skip("fastapi/starlette not installed") + from hermes_cli import web_server + + client = TestClient(web_server.app) + client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN + return client + + +@pytest.fixture +def broadcast_calls(monkeypatch): + """Stub the in-memory gateway module seam and record broadcasts.""" + import sys + + calls = [] + # tests/conftest.py's session-reaper teardown walks tui_gateway.server + # attributes; the stub must carry an empty _sessions to survive it. + stub = types.SimpleNamespace( + broadcast_session_info=lambda: calls.append(True), + _sessions={}, + ) + monkeypatch.setitem(sys.modules, "tui_gateway.server", stub) + return calls + + +class TestApprovalsSaveBroadcast: + def test_get_shaped_record_roundtrip_does_not_broadcast(self, client, broadcast_calls): + """The settings page PUTs the defaulted GET record back verbatim on + every autosave. That must not broadcast: disk holds sparse YAML while + GET returns defaults, so a block-level compare is always-unequal (the + review-caught spam bug). Only an effective mode change may emit.""" + record = client.get("/api/config").json() + assert "approvals" in record + + first = client.put("/api/config", json={"config": record}) + assert first.status_code == 200 + second = client.put("/api/config", json={"config": record}) + assert second.status_code == 200 + assert not broadcast_calls, ( + "autosaving the unmodified GET record broadcast session.info; " + "every settings autosave would walk all live sessions" + ) + + flipped = {**record, "approvals": {**record["approvals"], "mode": "off"}} + resp = client.put("/api/config", json={"config": flipped}) + assert resp.status_code == 200 + assert len(broadcast_calls) == 1, ( + "an actual approvals.mode change in the GET-shaped record must " + "broadcast exactly once" + ) + + def test_approvals_mode_change_broadcasts(self, client, broadcast_calls): + resp = client.put("/api/config", json={"config": {"approvals": {"mode": "off"}}}) + assert resp.status_code == 200 + assert broadcast_calls, ( + "PUT /api/config changed approvals.mode but no session.info " + "broadcast reached the gateway, so live sessions keep painting " + "stale YOLO/approval state" + ) + + def test_non_approvals_change_does_not_broadcast(self, client, broadcast_calls): + resp = client.put("/api/config", json={"config": {"display": {"skin": "mono"}}}) + assert resp.status_code == 200 + assert not broadcast_calls, ( + "a save that never touched approvals must not spam session.info" + ) + + def test_approvals_noop_save_does_not_broadcast(self, client, broadcast_calls): + first = client.put("/api/config", json={"config": {"approvals": {"mode": "off"}}}) + assert first.status_code == 200 + broadcast_calls.clear() + + again = client.put("/api/config", json={"config": {"approvals": {"mode": "off"}}}) + assert again.status_code == 200 + assert not broadcast_calls, ( + "saving an identical approvals block is a no-op and must not " + "re-emit session.info" + ) + + def test_other_profile_save_does_not_broadcast(self, client, broadcast_calls, monkeypatch, tmp_path): + from hermes_cli import web_server + + profile_dir = tmp_path / "profiles" / "other" + profile_dir.mkdir(parents=True) + monkeypatch.setattr(web_server, "_resolve_profile_dir", lambda name: profile_dir) + + resp = client.put( + "/api/config", + json={"config": {"approvals": {"mode": "off"}}, "profile": "other"}, + ) + assert resp.status_code == 200 + assert not broadcast_calls, ( + "a profile-scoped save targets a different HERMES_HOME than this " + "process's gateway sessions; broadcasting our own sessions' " + "unchanged state is wrong" + ) + + def test_gateway_not_imported_is_a_noop(self, client, monkeypatch): + import sys + + monkeypatch.delitem(sys.modules, "tui_gateway.server", raising=False) + resp = client.put("/api/config", json={"config": {"approvals": {"mode": "smart"}}}) + assert resp.status_code == 200 + assert "tui_gateway.server" not in sys.modules, ( + "the broadcast seam must not IMPORT the gateway; a process " + "without one has no sessions to notify" + ) + + def test_raw_save_deleting_approvals_block_broadcasts(self, client, broadcast_calls): + seed = client.put( + "/api/config/raw", + json={"yaml_text": "approvals:\n mode: 'off'\n"}, + ) + assert seed.status_code == 200 + broadcast_calls.clear() + + # Full-document replacement that drops the approvals block entirely: + # effective mode falls back to default (manual), so indicators must + # repaint. + resp = client.put( + "/api/config/raw", + json={"yaml_text": "display:\n skin: default\n"}, + ) + assert resp.status_code == 200 + assert broadcast_calls, ( + "deleting the approvals block changes the effective mode and " + "must broadcast" + ) + + def test_raw_save_approvals_change_broadcasts(self, client, broadcast_calls): + resp = client.put( + "/api/config/raw", + json={"yaml_text": "approvals:\n mode: 'off'\n"}, + ) + assert resp.status_code == 200 + assert broadcast_calls, ( + "PUT /api/config/raw changed approvals but no session.info " + "broadcast reached the gateway" + ) + + def test_raw_save_without_approvals_change_does_not_broadcast(self, client, broadcast_calls): + seed = client.put( + "/api/config/raw", + json={"yaml_text": "approvals:\n mode: manual\ndisplay:\n skin: default\n"}, + ) + assert seed.status_code == 200 + broadcast_calls.clear() + + resp = client.put( + "/api/config/raw", + json={"yaml_text": "approvals:\n mode: manual\ndisplay:\n skin: mono\n"}, + ) + assert resp.status_code == 200 + assert not broadcast_calls + + +class TestGatewayBroadcastHelper: + def test_broadcast_session_info_emits_for_live_sessions(self, _isolate_hermes_home, monkeypatch): + """tui_gateway.server.broadcast_session_info walks _sessions and emits.""" + from tui_gateway import server + + emitted = [] + monkeypatch.setattr( + server, "_emit_session_info_for_session", + lambda sid, sess: emitted.append(sid), + ) + monkeypatch.setattr( + server, "_sessions", + {"s1": {"agent": object()}, "s2": {"agent": object()}}, + ) + + server.broadcast_session_info() + + assert sorted(emitted) == ["s1", "s2"] + + def test_approvals_slash_mirror_broadcasts(self, _isolate_hermes_home, monkeypatch): + """/approvals through the slash worker persists config out of + band; the mirror must repaint live sessions and the bare read-only + form must not.""" + from tui_gateway import server + + calls = [] + monkeypatch.setattr(server, "broadcast_session_info", lambda: calls.append(True)) + + session = {"agent": None} + server._mirror_slash_side_effects("sid1", session, "/approvals off") + assert calls, "/approvals writes approvals.mode and must broadcast" + + calls.clear() + server._mirror_slash_side_effects("sid1", session, "/approvals") + assert not calls, "bare /approvals only reads the mode" diff --git a/tui_gateway/server.py b/tui_gateway/server.py index cbfaf3c948..2077971b17 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -5357,6 +5357,20 @@ def _emit_session_info_for_session(sid: str, session: dict) -> None: pass +def broadcast_session_info() -> None: + """Re-emit ``session.info`` to every live session. + + For approvals-config writers that bypass the ``config.set`` RPC (which + re-emits itself): the REST config saves and the ``/approvals`` slash + mirror. Only reaches sessions in THIS process; a spawned + ``tui_gateway.entry`` child gateway has its own ``_sessions``. + """ + with _sessions_lock: + sessions = list(_sessions.items()) + for sid, sess in sessions: + _emit_session_info_for_session(sid, sess) + + # Tool Args/Result text shipped to the TUI for the verbose trail line. The TUI # renders only a small persisted preview (ui-tui VERBOSE_TRAIL_MAX_CHARS), kept # all session and expanded by default — so shipping more than that is pure pipe @@ -12972,6 +12986,10 @@ def _mirror_slash_side_effects(sid: str, session: dict, command: str) -> str: if name == "model" and arg and agent: result = _apply_model_switch(sid, session, arg) return result.get("warning", "") + elif name == "approvals" and arg: + # The slash worker already persisted the new approvals.mode; the + # bare (read-only) form has no arg and needs no repaint. + broadcast_session_info() elif name == "personality" and arg and agent: pname, new_prompt = _validate_personality(arg, _load_cfg()) # Persist through the single owner so this surface can never From 040ee114aa12e54cfa176f816169007b3dfc02ff Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Mon, 17 Aug 2026 01:33:02 -0600 Subject: [PATCH 003/426] fix(desktop): remove invalid get-windows recovery command --- apps/desktop/scripts/stage-native-deps.mjs | 1 - apps/desktop/scripts/stage-native-deps.test.mjs | 6 +++++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/apps/desktop/scripts/stage-native-deps.mjs b/apps/desktop/scripts/stage-native-deps.mjs index c55a8ed3ab..c811ce296b 100644 --- a/apps/desktop/scripts/stage-native-deps.mjs +++ b/apps/desktop/scripts/stage-native-deps.mjs @@ -518,7 +518,6 @@ export function stageGetWindowsInto( throw new Error( `[stage-native-deps] get-windows has no win32-${arch} prebuilt binding under lib/binding. ` + 'Recover from the checkout root with:\n' + - ' npm install-scripts approve get-windows\n' + ' npm rebuild get-windows' ) } diff --git a/apps/desktop/scripts/stage-native-deps.test.mjs b/apps/desktop/scripts/stage-native-deps.test.mjs index 6392b52a80..d7603f5a75 100644 --- a/apps/desktop/scripts/stage-native-deps.test.mjs +++ b/apps/desktop/scripts/stage-native-deps.test.mjs @@ -505,7 +505,11 @@ test('win32 staging reports the recovery steps when the rebuild hook produces no arch: 'x64', rebuild: () => {} }), - /npm rebuild get-windows/ + (error) => { + assert.match(error.message, /npm rebuild get-windows/) + assert.doesNotMatch(error.message, /npm install-scripts/) + return true + } ) } finally { fs.rmSync(tmp, { recursive: true, force: true }) From 97a04aef6cf3c75beb8c250ff77c119f0df93bcd Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Mon, 17 Aug 2026 11:47:29 -0600 Subject: [PATCH 004/426] fix(desktop): install get-windows binding directly --- apps/desktop/scripts/stage-native-deps.mjs | 67 +++++++++++++------ .../scripts/stage-native-deps.test.mjs | 65 ++++++++++++++++-- 2 files changed, 104 insertions(+), 28 deletions(-) diff --git a/apps/desktop/scripts/stage-native-deps.mjs b/apps/desktop/scripts/stage-native-deps.mjs index c811ce296b..385d031725 100644 --- a/apps/desktop/scripts/stage-native-deps.mjs +++ b/apps/desktop/scripts/stage-native-deps.mjs @@ -437,7 +437,7 @@ const GET_WINDOWS_VERSION = '9.3.0' export function stageGetWindowsInto( srcRoot, destRoot, - { platform = process.platform, arch = process.arch, rebuild } = {} + { platform = process.platform, arch = process.arch, install } = {} ) { // The STAGED_WINDOWS_JS rewrite mirrors this exact version's export surface. // A version bump must fail the build here until the rewrite is re-verified — @@ -495,6 +495,7 @@ export function stageGetWindowsInto( ) : [] let bindingDirs = scanBindingDirs() + let installAttempted = false if (bindingDirs.length === 0 && arch === 'arm64') { // get-windows 9.3.0 publishes win32 prebuilds for ia32/x64 only. // The staged windows.js deliberately fails soft when binding/ is absent, @@ -503,23 +504,25 @@ export function stageGetWindowsInto( '[stage-native-deps] get-windows has no win32-arm64 prebuilt binding; ' + 'staging the fail-soft JS surface without native window enumeration.' ) - } else if (bindingDirs.length === 0 && typeof rebuild === 'function') { + } else if (bindingDirs.length === 0 && typeof install === 'function') { // A plain `npm install` won't re-run an install script for a package // that is already on disk, so every checkout that installed while // get-windows was missing from allowScripts stays bricked even after - // the allowlist is fixed. `npm rebuild` re-runs it. + // the allowlist is fixed. Invoke node-pre-gyp directly: npm treats this + // optional dependency's failed lifecycle as non-fatal and can report a + // successful rebuild without producing the Windows binding. console.log( - '[stage-native-deps] get-windows has no win32 binding; running `npm rebuild get-windows`...' + '[stage-native-deps] get-windows has no win32 binding; running its native installer...' ) - rebuild() + installAttempted = true + install() bindingDirs = scanBindingDirs() } if (bindingDirs.length === 0 && arch !== 'arm64') { - throw new Error( - `[stage-native-deps] get-windows has no win32-${arch} prebuilt binding under lib/binding. ` + - 'Recover from the checkout root with:\n' + - ' npm rebuild get-windows' - ) + const reason = installAttempted + ? `native installer completed without producing a win32-${arch} binding under lib/binding` + : `has no win32-${arch} prebuilt binding under lib/binding` + throw new Error(`[stage-native-deps] get-windows ${reason}`) } for (const dir of bindingDirs) { const dest = join(destRoot, 'lib', 'binding', dir) @@ -541,15 +544,35 @@ export function stageGetWindowsInto( return destRoot } -function rebuildGetWindowsViaNpm() { - const result = spawnSync('npm', ['rebuild', 'get-windows'], { - cwd: resolve(projectRoot, '..', '..'), - stdio: 'inherit', - // npm resolves to npm.cmd on Windows, which needs a shell. - shell: process.platform === 'win32' +export function installGetWindowsNativeBinding( + srcRoot, + { resolveInstaller, spawn = spawnSync } = {} +) { + let installerPath + try { + const resolveNodePreGyp = + resolveInstaller ?? + (() => + require.resolve('@mapbox/node-pre-gyp/bin/node-pre-gyp', { + paths: [srcRoot] + })) + installerPath = resolveNodePreGyp() + } catch (error) { + const detail = error instanceof Error ? error.message : String(error) + throw new Error(`[stage-native-deps] cannot resolve get-windows native installer: ${detail}`) + } + + const result = spawn(process.execPath, [installerPath, 'install', '--fallback-to-build'], { + cwd: srcRoot, + stdio: 'inherit' }) + if (result.error) { + throw new Error( + `[stage-native-deps] get-windows native installer could not start: ${result.error.message}` + ) + } if (result.status !== 0) { - console.warn(`[stage-native-deps] npm rebuild get-windows exited with ${result.status}`) + throw new Error(`[stage-native-deps] get-windows native installer exited with ${result.status}`) } } @@ -584,10 +607,12 @@ export function stageGetWindows( } // Only a win32 host can produce the win32 binding, so a cross-platform pack - // has nothing to gain from the rebuild. - const rebuild = - platform === 'win32' && process.platform === 'win32' ? rebuildGetWindowsViaNpm : undefined - return stageGetWindowsInto(srcRoot, destRoot, { platform, arch, rebuild }) + // has nothing to gain from the native installer. + const install = + platform === 'win32' && process.platform === 'win32' + ? () => installGetWindowsNativeBinding(srcRoot) + : undefined + return stageGetWindowsInto(srcRoot, destRoot, { platform, arch, install }) } // Allow direct CLI invocation: node scripts/stage-native-deps.mjs [platform] [arch] diff --git a/apps/desktop/scripts/stage-native-deps.test.mjs b/apps/desktop/scripts/stage-native-deps.test.mjs index d7603f5a75..51eea1718d 100644 --- a/apps/desktop/scripts/stage-native-deps.test.mjs +++ b/apps/desktop/scripts/stage-native-deps.test.mjs @@ -6,6 +6,7 @@ import { pathToFileURL } from 'node:url' import { test } from 'vitest' import { + installGetWindowsNativeBinding, stageGetWindows, stageGetWindowsInto, stageNodePtyInto, @@ -460,7 +461,7 @@ test('win32-arm64 staging omits incompatible bindings and keeps the fail-soft JS } }) -test('win32 staging self-heals through the rebuild hook when the binding is missing', () => { +test('win32 staging self-heals through the native installer when the binding is missing', () => { const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-')) try { const srcRoot = join(tmp, 'get-windows') @@ -471,7 +472,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss makeFakeGetWindows(srcRoot, { bindings: [] }) let calls = 0 - const rebuild = () => { + const install = () => { calls += 1 makeFakeNode( join(srcRoot, 'lib', 'binding', 'napi-9-win32-unknown-x64', 'node-get-windows.node'), @@ -479,7 +480,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss ) } - stageGetWindowsInto(srcRoot, destRoot, { platform: 'win32', arch: 'x64', rebuild }) + stageGetWindowsInto(srcRoot, destRoot, { platform: 'win32', arch: 'x64', install }) assert.equal(calls, 1) assert.ok( @@ -490,7 +491,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss } }) -test('win32 staging reports the recovery steps when the rebuild hook produces nothing', () => { +test('win32 staging rejects a successful installer that produces no binding', () => { const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-')) try { const srcRoot = join(tmp, 'get-windows') @@ -503,11 +504,11 @@ test('win32 staging reports the recovery steps when the rebuild hook produces no stageGetWindowsInto(srcRoot, destRoot, { platform: 'win32', arch: 'x64', - rebuild: () => {} + install: () => {} }), (error) => { - assert.match(error.message, /npm rebuild get-windows/) - assert.doesNotMatch(error.message, /npm install-scripts/) + assert.match(error.message, /installer completed without producing a win32-x64 binding/) + assert.doesNotMatch(error.message, /npm rebuild/) return true } ) @@ -516,6 +517,56 @@ test('win32 staging reports the recovery steps when the rebuild hook produces no } }) +test('get-windows native install invokes node-pre-gyp directly from the package root', () => { + const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-')) + try { + const srcRoot = join(tmp, 'get-windows') + const installer = join( + srcRoot, + 'node_modules', + '@mapbox', + 'node-pre-gyp', + 'bin', + 'node-pre-gyp' + ) + fs.mkdirSync(path.dirname(installer), { recursive: true }) + fs.writeFileSync( + join(srcRoot, 'node_modules', '@mapbox', 'node-pre-gyp', 'package.json'), + JSON.stringify({ name: '@mapbox/node-pre-gyp', version: '1.0.11' }) + ) + fs.writeFileSync(installer, '') + + const calls = [] + installGetWindowsNativeBinding(srcRoot, { + spawn: (command, args, options) => { + calls.push({ command, args, options }) + return { status: 0 } + } + }) + + assert.deepEqual(calls, [ + { + command: process.execPath, + args: [installer, 'install', '--fallback-to-build'], + options: { cwd: srcRoot, stdio: 'inherit' } + } + ]) + } finally { + fs.rmSync(tmp, { recursive: true, force: true }) + } +}) + +test('get-windows native install surfaces node-pre-gyp failure', () => { + assert.throws( + () => + installGetWindowsNativeBinding('C:\\fake\\get-windows', { + resolveInstaller: () => 'C:\\fake\\node-pre-gyp', + spawn: () => ({ status: 1 }) + }), + /native installer exited with 1/ + ) +}) + test('staging refuses a get-windows version the lib/windows.js rewrite was not verified against', () => { const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-')) try { From 48c1eefa2677419a5b80967f708eb1098e415709 Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Mon, 17 Aug 2026 21:43:52 -0600 Subject: [PATCH 005/426] fix(desktop): repair missing Windows runtime --- apps/desktop/electron/main.ts | 24 ++++++----- .../electron/windows-hermes-path.test.ts | 43 ++++++++++++++----- apps/desktop/electron/windows-hermes-path.ts | 40 +++++++++-------- 3 files changed, 68 insertions(+), 39 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 06a1eb9181..51607c1397 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -3707,16 +3707,20 @@ async function handOffWindowsBootstrapRecovery(reason) { const venvHermes = path.join(venvBin, IS_WINDOWS ? 'hermes.exe' : 'hermes') const venvPython = path.join(venvBin, IS_WINDOWS ? 'python.exe' : 'python') - // Choose the gentle in-place --update when ANY real-install signal is present, - // not just the `hermes.exe` console-script shim. That shim is generated at the - // END of venv setup and is absent in exactly the interrupted/quarantined states - // this recovery exists to heal — gating on it alone forced the destructive - // --repair (full venv recreate) and drove reinstall loops. The venv interpreter - // and the bootstrap-complete marker are present earlier and are better signals. - const haveRealInstall = - fileExists(venvPython) || fileExists(venvHermes) || fileExists(path.join(updateRoot, '.hermes-bootstrap-complete')) - - const updaterArgs = chooseUpdaterArgs(haveRealInstall, branch) + // The updater invokes the venv's Hermes launcher, which in turn requires the + // venv interpreter. A bootstrap-complete marker proves only that setup once + // finished; it can outlive a manually removed or quarantined venv. Sending a + // marker-only install through --update dead-ends at "Could not find the hermes + // CLI" instead of rebuilding the runtime, so only a runnable pair gets the + // gentle update path. Partial or missing runtimes go through full repair. + const updaterArgs = chooseUpdaterArgs( + { + hasBootstrapMarker: fileExists(path.join(updateRoot, '.hermes-bootstrap-complete')), + hasVenvHermes: fileExists(venvHermes), + hasVenvPython: fileExists(venvPython) + }, + branch + ) await releaseBackendLockForUpdate(updateRoot) diff --git a/apps/desktop/electron/windows-hermes-path.test.ts b/apps/desktop/electron/windows-hermes-path.test.ts index 778042b278..76e4d0ac50 100644 --- a/apps/desktop/electron/windows-hermes-path.test.ts +++ b/apps/desktop/electron/windows-hermes-path.test.ts @@ -5,9 +5,9 @@ // 1. buildPathExtCandidates() — PATHEXT extensions must be tried BEFORE the // empty extension, or an extensionless Git-Bash `hermes` shim shadows // the real hermes.cmd/hermes.exe. -// 2. chooseUpdaterArgs() — must gate on haveRealInstall (any real-install -// signal), not just the hermes.exe console-script shim, or healthy -// installs get forced into a destructive --repair. +// 2. chooseUpdaterArgs() — must distinguish a runnable updater from stale +// install provenance. The bootstrap marker can outlive the venv, and a +// partial venv cannot run the updater; those states require --repair. // 3. resolveVenvHermesCommand() — must probe the venv python via // canImportHermesCli() before trusting it, or a broken venv gets // re-selected forever instead of falling through to bootstrap. @@ -45,17 +45,40 @@ test('buildPathExtCandidates: non-Windows only tries the bare name', () => { assert.deepEqual(buildPathExtCandidates(undefined, false), ['']) }) -test('chooseUpdaterArgs: gentle --update when a real-install signal is present', () => { - assert.deepEqual(chooseUpdaterArgs(true, 'main'), ['--update', '--branch', 'main']) +test('chooseUpdaterArgs: gentle --update when both updater runtime files exist', () => { + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: true, hasVenvPython: true }, 'main'), + ['--update', '--branch', 'main'] + ) }) -test('chooseUpdaterArgs: destructive --repair only when NO real-install signal is present', () => { - assert.deepEqual(chooseUpdaterArgs(false, 'main'), ['--repair', '--branch', 'main']) +test('chooseUpdaterArgs: marker-only install uses --repair when the venv is gone', () => { + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: false, hasVenvPython: false }, 'main'), + ['--repair', '--branch', 'main'] + ) }) -test('chooseUpdaterArgs: passes the branch through unchanged in both cases', () => { - assert.deepEqual(chooseUpdaterArgs(true, 'release/1.2'), ['--update', '--branch', 'release/1.2']) - assert.deepEqual(chooseUpdaterArgs(false, 'release/1.2'), ['--repair', '--branch', 'release/1.2']) +test('chooseUpdaterArgs: partial updater runtimes use --repair', () => { + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: false, hasVenvPython: true }, 'main'), + ['--repair', '--branch', 'main'] + ) + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: true, hasVenvPython: false }, 'main'), + ['--repair', '--branch', 'main'] + ) +}) + +test('chooseUpdaterArgs: passes the branch through unchanged in both modes', () => { + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: false, hasVenvHermes: true, hasVenvPython: true }, 'release/1.2'), + ['--update', '--branch', 'release/1.2'] + ) + assert.deepEqual( + chooseUpdaterArgs({ hasBootstrapMarker: false, hasVenvHermes: false, hasVenvPython: false }, 'release/1.2'), + ['--repair', '--branch', 'release/1.2'] + ) }) function makeDeps(overrides: Partial[2]> = {}) { diff --git a/apps/desktop/electron/windows-hermes-path.ts b/apps/desktop/electron/windows-hermes-path.ts index 6f3542853e..f94d4b2c39 100644 --- a/apps/desktop/electron/windows-hermes-path.ts +++ b/apps/desktop/electron/windows-hermes-path.ts @@ -11,12 +11,11 @@ * hermes.cmd/hermes.exe; the shim then failed the --version probe and * the desktop fell through to a spurious bootstrap/repair. The fix: * PATHEXT extensions first, empty extension LAST. - * 2. chooseUpdaterArgs() — handOffWindowsBootstrapRecovery() chose - * --update vs the destructive --repair by checking ONLY - * venv\Scripts\hermes.exe (the console-script shim, written at the END - * of venv setup and absent in interrupted states), so it escalated to a - * full venv recreate even on healthy installs. The fix: gate on ANY - * real-install signal, not just the shim. + * 2. chooseUpdaterArgs() — handOffWindowsBootstrapRecovery() must separate + * install provenance from updater viability. A bootstrap-complete marker + * can outlive a deleted venv, while the updater needs BOTH the venv Python + * and Hermes launcher. Marker-only or partial runtimes must use --repair; + * only a runnable pair can use --update. * 3. resolveVenvHermesCommand() — unwrapWindowsVenvHermesCommand() returned * the venv python with NO runtime probe (bypassing the caller's * --version check too), so a venv broken mid-update (e.g. missing @@ -61,23 +60,26 @@ export function buildPathExtCandidates(pathext: string | undefined, isWindows: b } /** - * Choose the Windows bootstrap-recovery updater invocation: the gentle - * in-place --update when ANY real-install signal is present, the - * destructive --repair (full venv recreate) otherwise. + * Choose the Windows bootstrap-recovery invocation. The gentle in-place + * updater can only start when both pieces of its runtime contract exist: the + * venv Python interpreter and the Hermes launcher that drives `hermes update`. + * A bootstrap-complete marker proves install provenance, not current runtime + * usability, and may remain after the venv is removed or quarantined. * - * haveRealInstall must be computed by the caller from ALL real-install - * signals (venv python interpreter, venv hermes shim, bootstrap-complete - * marker) — gating on just the hermes.exe console-script shim alone is the - * regression this function's callers must avoid: that shim is written at - * the END of venv setup and is absent in exactly the interrupted/quarantined - * states this recovery exists to heal. - * - * @param {boolean} haveRealInstall + * @param {BootstrapRecoverySignals} signals * @param {string} branch * @returns {string[]} updater argv, e.g. ['--update', '--branch', 'main']. */ -export function chooseUpdaterArgs(haveRealInstall: boolean, branch: string): string[] { - return haveRealInstall ? ['--update', '--branch', branch] : ['--repair', '--branch', branch] +export interface BootstrapRecoverySignals { + hasBootstrapMarker: boolean + hasVenvHermes: boolean + hasVenvPython: boolean +} + +export function chooseUpdaterArgs(signals: BootstrapRecoverySignals, branch: string): string[] { + const canRunUpdater = signals.hasVenvHermes && signals.hasVenvPython + + return canRunUpdater ? ['--update', '--branch', branch] : ['--repair', '--branch', branch] } /** From 6a843f95c8e9ae91fa0858cc687b7db6a6d01a53 Mon Sep 17 00:00:00 2001 From: xxxigm Date: Tue, 18 Aug 2026 22:19:14 +0700 Subject: [PATCH 006/426] fix(desktop): let Files panel download remote backend files Remote mode only offered Copy Path, which is a Linux server path and useless on the local machine. Reuse the existing gateway save bridge so a selected file can land on this computer. --- .../src/app/right-sidebar/file-actions.tsx | 12 +++++- .../app/right-sidebar/review/file-tree.tsx | 15 ++++++- apps/desktop/src/i18n/ar.ts | 3 ++ apps/desktop/src/i18n/en.ts | 3 ++ apps/desktop/src/i18n/ja.ts | 3 ++ apps/desktop/src/i18n/types.ts | 3 ++ apps/desktop/src/i18n/zh-hant.ts | 3 ++ apps/desktop/src/i18n/zh.ts | 3 ++ apps/desktop/src/store/file-actions.ts | 39 ++++++++++++++++--- 9 files changed, 77 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/app/right-sidebar/file-actions.tsx b/apps/desktop/src/app/right-sidebar/file-actions.tsx index dbb2ccb65d..e5e9443f32 100644 --- a/apps/desktop/src/app/right-sidebar/file-actions.tsx +++ b/apps/desktop/src/app/right-sidebar/file-actions.tsx @@ -19,11 +19,13 @@ import { cancelInlineRename, closeFileActionDialog, copyFilePath, + downloadRemoteFile, executeFileDelete, executeFileRename, type FileActionTarget, requestFileDelete, revealFile, + shouldOfferRemoteFileDownload, toRelativePath } from '@/store/file-actions' import { notifyError } from '@/store/notifications' @@ -57,8 +59,10 @@ export function FileEntryContextMenu({ children, isDirectory, name, path, relati const { t } = useI18n() const m = t.fileMenu // Reveal / rename / delete need the local filesystem; hide them on a remote - // backend (copy-path still works everywhere). + // backend (copy-path still works everywhere). Download uses the existing + // gateway save bridge so a remote file can land on this machine. const localFs = !isDesktopFsRemoteMode() + const remoteDownload = shouldOfferRemoteFileDownload(isDirectory) const target: FileActionTarget = { isDirectory, name, path } const revealLabel = pickRevealLabel(m.revealFinder, m.revealExplorer, m.revealFileManager) @@ -80,6 +84,12 @@ export function FileEntryContextMenu({ children, isDirectory, name, path, relati {m.copyRelativePath} )} + {remoteDownload && ( + <> + + void downloadRemoteFile(path)}>{m.download} + + )} {localFs && ( <> diff --git a/apps/desktop/src/app/right-sidebar/review/file-tree.tsx b/apps/desktop/src/app/right-sidebar/review/file-tree.tsx index 745839c523..1ae6e21bc6 100644 --- a/apps/desktop/src/app/right-sidebar/review/file-tree.tsx +++ b/apps/desktop/src/app/right-sidebar/review/file-tree.tsx @@ -20,7 +20,14 @@ import { isDesktopFsRemoteMode } from '@/lib/desktop-fs' import { displayPath } from '@/lib/display-path' import { normalizeOrLocalPreviewTarget } from '@/lib/local-preview' import { cn } from '@/lib/utils' -import { $renamingPath, copyFilePath, revealFile, toRelativePath } from '@/store/file-actions' +import { + $renamingPath, + copyFilePath, + downloadRemoteFile, + revealFile, + shouldOfferRemoteFileDownload, + toRelativePath +} from '@/store/file-actions' import { $sidebarWorkspaceNodeOpen, revealFileInTree, toggleWorkspaceNodeCollapsed } from '@/store/layout' import { notifyError } from '@/store/notifications' import { openPreview } from '@/store/preview' @@ -514,6 +521,12 @@ function ReviewFileContextMenu({ {m.copyRelativePath} )} + {shouldOfferRemoteFileDownload(false) && ( + <> + + void downloadRemoteFile(dragPath)}>{m.download} + + )} ) diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 06d7370adf..7749e69d8d 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -46,6 +46,9 @@ export const ar = defineLocale({ revealInSidebar: 'إظهار في شجرة الملفات', copyPath: 'نسخ المسار', copyRelativePath: 'نسخ المسار النسبي', + download: 'تنزيل', + downloadSaved: 'تم الحفظ', + downloadFailed: 'فشل التنزيل', rename: 'إعادة تسمية...', delete: 'حذف', renameTitle: 'إعادة تسمية', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 5b9c56a8c1..962b7338c6 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -52,6 +52,9 @@ export const en: Translations = { revealInSidebar: 'Reveal in filetree', copyPath: 'Copy Path', copyRelativePath: 'Copy Relative Path', + download: 'Download', + downloadSaved: 'Saved', + downloadFailed: 'Download failed', rename: 'Rename…', delete: 'Delete', renameTitle: 'Rename', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 0d55b75067..8d5872788e 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -52,6 +52,9 @@ export const ja = defineLocale({ revealInSidebar: 'ファイルツリーで表示', copyPath: 'パスをコピー', copyRelativePath: '相対パスをコピー', + download: 'ダウンロード', + downloadSaved: '保存しました', + downloadFailed: 'ダウンロードに失敗しました', rename: '名前を変更…', delete: '削除', renameTitle: '名前を変更', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 678e0f88e7..ec4b3e5643 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -98,6 +98,9 @@ export interface Translations { revealInSidebar: string copyPath: string copyRelativePath: string + download: string + downloadSaved: string + downloadFailed: string rename: string delete: string renameTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 490fc29841..f2db8dd923 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -52,6 +52,9 @@ export const zhHant = defineLocale({ revealInSidebar: '在檔案樹中顯示', copyPath: '複製路徑', copyRelativePath: '複製相對路徑', + download: '下載', + downloadSaved: '已儲存', + downloadFailed: '下載失敗', rename: '重新命名…', delete: '刪除', renameTitle: '重新命名', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index dd3b63af90..a04f5a8844 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -52,6 +52,9 @@ export const zh: Translations = { revealInSidebar: '在文件树中显示', copyPath: '复制路径', copyRelativePath: '复制相对路径', + download: '下载', + downloadSaved: '已保存', + downloadFailed: '下载失败', rename: '重命名…', delete: '删除', renameTitle: '重命名', diff --git a/apps/desktop/src/store/file-actions.ts b/apps/desktop/src/store/file-actions.ts index 55f2b0fac3..efd3f219ef 100644 --- a/apps/desktop/src/store/file-actions.ts +++ b/apps/desktop/src/store/file-actions.ts @@ -1,15 +1,23 @@ import { atom } from 'nanostores' import { translateNow } from '@/i18n' -import { copyTextToClipboard, renameDesktopPath, revealDesktopPath, trashDesktopPath } from '@/lib/desktop-fs' +import { + copyTextToClipboard, + isDesktopFsRemoteMode, + renameDesktopPath, + revealDesktopPath, + trashDesktopPath +} from '@/lib/desktop-fs' +import { downloadGatewayMediaFile } from '@/lib/media' import { notify, notifyError } from '@/store/notifications' import { notifyWorkspaceChanged } from '@/store/workspace-events' // Shared file-row actions for BOTH trees (the file browser + the review/git -// tree): reveal, copy path, rename, delete. Rename/delete route through a single -// dialog set (driven by this atom, rendered once by `FileActionDialogs`) instead -// of one dialog per row. After a successful mutation we bump the workspace tick -// so every git-/fs-mirroring surface refreshes. +// tree): reveal, copy path, download (remote), rename, delete. Rename/delete +// route through a single dialog set (driven by this atom, rendered once by +// `FileActionDialogs`) instead of one dialog per row. After a successful +// mutation we bump the workspace tick so every git-/fs-mirroring surface +// refreshes. export interface FileActionTarget { isDirectory: boolean @@ -65,6 +73,27 @@ export async function copyFilePath(path: string): Promise { } } +/** Remote Files panel can list gateway files but Reveal/Rename/Delete are local-only. + * Download is the local-copy affordance. Folders stay out — `/api/fs/download` + * streams a single file. */ +export function shouldOfferRemoteFileDownload(isDirectory: boolean, remote = isDesktopFsRemoteMode()): boolean { + return remote && !isDirectory +} + +export async function downloadRemoteFile(path: string): Promise { + try { + const result = await downloadGatewayMediaFile(path) + + if (result.canceled || !result.saved) { + return + } + + notify({ durationMs: 1500, kind: 'info', message: translateNow('fileMenu.downloadSaved') }) + } catch (error) { + notifyError(error, translateNow('fileMenu.downloadFailed')) + } +} + /** Strip a `relativeTo` prefix to produce a repo/cwd-relative path. */ export function toRelativePath(path: string, relativeTo: string): string { const base = relativeTo.replace(/[\\/]+$/, '') From d28f2ed05b33dd8bf42e8ccd2028f64c0c7ce33b Mon Sep 17 00:00:00 2001 From: xxxigm Date: Tue, 18 Aug 2026 22:19:24 +0700 Subject: [PATCH 007/426] test(desktop): cover remote Files panel download action Lock the remote-file-only menu gate and the save-bridge success, cancel, and error paths. --- apps/desktop/src/store/file-actions.test.ts | 58 +++++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 apps/desktop/src/store/file-actions.test.ts diff --git a/apps/desktop/src/store/file-actions.test.ts b/apps/desktop/src/store/file-actions.test.ts new file mode 100644 index 0000000000..d78780c5d2 --- /dev/null +++ b/apps/desktop/src/store/file-actions.test.ts @@ -0,0 +1,58 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { $notifications, clearNotifications } from '@/store/notifications' + +vi.mock('@/lib/media', () => ({ + downloadGatewayMediaFile: vi.fn() +})) + +const media = await import('@/lib/media') +const downloadGatewayMediaFile = vi.mocked(media.downloadGatewayMediaFile) + +const { downloadRemoteFile, shouldOfferRemoteFileDownload } = await import('./file-actions') + +describe('shouldOfferRemoteFileDownload', () => { + it('is only for files on a remote backend', () => { + expect(shouldOfferRemoteFileDownload(false, true)).toBe(true) + expect(shouldOfferRemoteFileDownload(true, true)).toBe(false) + expect(shouldOfferRemoteFileDownload(false, false)).toBe(false) + expect(shouldOfferRemoteFileDownload(true, false)).toBe(false) + }) +}) + +describe('downloadRemoteFile', () => { + beforeEach(() => { + clearNotifications() + downloadGatewayMediaFile.mockReset() + }) + + afterEach(() => { + clearNotifications() + }) + + it('saves a remote gateway file through the native download bridge', async () => { + downloadGatewayMediaFile.mockResolvedValue({ path: '/Users/me/Downloads/notes.md', saved: true }) + + await downloadRemoteFile('/home/linux/project/notes.md') + + expect(downloadGatewayMediaFile).toHaveBeenCalledWith('/home/linux/project/notes.md') + expect($notifications.get()[0]?.message).toBe('Saved') + }) + + it('stays quiet when the save dialog is canceled', async () => { + downloadGatewayMediaFile.mockResolvedValue({ canceled: true, saved: false }) + + await downloadRemoteFile('/home/linux/project/notes.md') + + expect($notifications.get()).toEqual([]) + }) + + it('toasts when the gateway download fails', async () => { + downloadGatewayMediaFile.mockRejectedValue(new Error('Desktop file download bridge is unavailable')) + + await downloadRemoteFile('/home/linux/project/notes.md') + + expect($notifications.get()[0]?.kind).toBe('error') + expect($notifications.get()[0]?.title).toBe('Download failed') + }) +}) From 246477a80986f2af64fd00181eeb4ed5c5b6330c Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 12:05:28 -0400 Subject: [PATCH 008/426] fix(gateway): /goal no longer lies when state.db init is slow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A fresh state.db init (schema DDL, FTS tables, first config import) measures ~300ms warm on a fast machine. The gateway constructs GoalManager on the event-loop thread, and a cold cache ran that init behind a 0.25s bootstrap grace window: on a slow CI box the /goal set path's waits expired and save_goal silently no-oped — the reply said "Goal set (7-turn budget)..." but nothing persisted, and a fresh GoalManager read back no state (first assertion passes, second fails). Two changes, one per caller shape: - Async callers (_get_goal_manager_for_event, _get_heartbeat_manager_for_event, _post_turn_goal_continuation, and the heartbeat poller) warm the SessionDB cache off-loop through the context-preserving executor before constructing the manager (shared _warm_goals_session_db helper). The loop never blocks and the first write lands at any init duration. A bare to_thread would lose the per-turn profile home override under multiplex; the executor hop keeps it (same pattern as the goal judge path). - Sync callers (heartbeat persistence, _goal_still_active_for_session) cannot await, so the bootstrap windows stay: the call that starts the bootstrap waits a one-time init window (1.5s) instead of the short per-call window (0.25s), giving healthy cold inits room to land while a contended migration still degrades to None with only a bounded one-time stall. The bootstrap thread binds the caller's home as a contextvar override so a multiplexed worker cannot cache the default profile's DB under another profile's key. save_goal and heartbeat save_state now log at WARNING when they drop a write, because the reply has already told the user the state was set. Regression test pins the contract: init past the window, write persists, loop gap under 2s (the flake-policy floor for wall-clock bounds; the slow-init margin grew to match, so the test still tells on-loop from off-loop). Independent diagnosis + measurement by jackulau (#88965 review); the off-loop warm-up shape follows their harness table. Simplify-code review (4-agent) contributed the helper extraction and the poller warm-up. --- gateway/run.py | 35 +++++++ hermes_cli/goals.py | 73 +++++++++++--- hermes_cli/heartbeat.py | 6 ++ hermes_cli/loops.py | 31 +++--- tests/gateway/test_goal_max_turns_config.py | 96 +++++++++++++++++-- tests/gateway/test_loop_command.py | 6 +- .../test_goals_db_bootstrap_off_loop.py | 32 +++++-- tests/hermes_cli/test_loops.py | 6 +- tests/tui_gateway/test_loop_command.py | 6 +- 9 files changed, 232 insertions(+), 59 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 9970c20780..04c80279c8 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -20998,6 +20998,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: return 20 + async def _warm_goals_session_db(self, ctx: str) -> None: + """Warm the goals SessionDB cache off-loop (best-effort). + + A cold cache runs the state.db init on the loop thread behind the + bootstrap windows. That freezes the loop for the init duration. + The executor hop keeps the profile home override alive under + multiplex, so the warm cache belongs to the caller's profile. + On failure the caller falls back to the bootstrap windows, so a + dropped warm-up is a bounded stall, never a crash. + """ + try: + from hermes_cli.goals import _get_session_db as _warm_goals_db + + await self._run_in_executor_with_context(_warm_goals_db) + except Exception as exc: + logger.warning("%s: session DB warm-up failed: %s", ctx, exc) + async def _get_goal_manager_for_event(self, event: "MessageEvent"): """Return a GoalManager bound to the session for this gateway event. @@ -21009,6 +21026,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.debug("goal manager unavailable: %s", exc) return None, None + # Warm the SessionDB cache off-loop. A cold cache freezes the + # loop for the init duration and drops the first write: the + # /goal reply claims the goal was set. + await self._warm_goals_session_db("goal manager") try: # Session lookups on behalf of an internal event must not advance # the user-activity clock that drives idle/daily reset policy @@ -21036,6 +21057,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.debug("heartbeat manager unavailable: %s", exc) return None, None + # Warm the SessionDB cache off-loop. A cold cache can drop the + # first /heartbeat write while the reply claims it was set. + await self._warm_goals_session_db("heartbeat manager") try: # Same reset-policy contract as _get_goal_manager_for_event: # internal events look up the session without touching activity. @@ -21086,6 +21110,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew watch = getattr(self, "_heartbeat_watch", None) if not watch: continue + # Warm the cache off-loop once per poll. A watch can only + # be registered through the warmed /heartbeat command, so + # this covers only the degraded path where that warm-up + # failed. + await self._warm_goals_session_db("heartbeat poll") for quick_key, (source, session_id) in list(watch.items()): try: # Busy sessions coalesce their tick to the next idle poll. @@ -21217,6 +21246,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew max_turns = self._goal_max_turns_from_config() + # Warm the SessionDB cache off-loop. A cold cache runs the + # state.db init on the loop thread at the turn boundary (the + # 2026-08-14 crash-loop seam). A slow init can drop the goal + # read and silently end the goal loop. + await self._warm_goals_session_db("goal continuation") + mgr = GoalManager(session_id=sid, default_max_turns=max_turns) if not mgr.is_active(): return diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index 9a685ff8d3..39870322af 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -666,20 +666,47 @@ _DB_CACHE: Dict[str, Any] = {} _DB_BOOTSTRAP_LOCK = threading.Lock() _DB_BOOTSTRAP_INFLIGHT: Dict[str, threading.Event] = {} -# How long a loop-thread caller waits for the background bootstrap before -# degrading to None. Normal SessionDB init is ~10-100ms, so the common case -# still returns a real DB (no silently dropped goal writes); a contended -# init (locked state.db mid-migration) blows past this and the caller -# degrades, with the loop stalled far under the watchdog's probe window. +# How long a loop-thread caller waits for an ALREADY-RUNNING bootstrap +# before degrading to None. Normal SessionDB init is ~10-100ms, so a call +# that arrives mid-bootstrap usually picks the cached instance up within +# this window. A contended init (locked state.db mid-migration) blows past +# it and the caller degrades. The loop stalls far under the watchdog's +# probe window. _DB_BOOTSTRAP_LOOP_WAIT_S = 0.25 +# The call that STARTS the bootstrap (cold cache, nothing in flight) +# waits this long instead of the short window above. A fresh state.db +# init measures ~300ms warm on a fast machine: schema DDL, FTS table +# creation, and the first hermes_cli.config import (journal-mode +# resolution). It is longer on a slow CI box, and it is well past 0.25s. +# The old window dropped the first /goal write. The response said +# "Goal set" but nothing persisted. The longer window is a bounded +# one-time stall. Only the kick call pays it. Every later call keeps +# the short window, so a contended migration never stalls the loop +# repeatedly. +_DB_BOOTSTRAP_INIT_WAIT_S = 1.5 + def _bootstrap_session_db(home: str, done: threading.Event) -> None: """Construct SessionDB off-loop and populate the cache (worker thread).""" try: + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) from hermes_state import SessionDB - db = SessionDB() + # Bind the caller's home for this thread. The cache key is the + # caller's scoped home, so the constructed SessionDB must point at + # that home's state.db too. Without the override, a multiplexed + # worker thread resolves the process env (the default profile's + # HERMES_HOME). It then caches the wrong profile's DB under this + # profile's key. + token = set_hermes_home_override(home) + try: + db = SessionDB() + finally: + reset_hermes_home_override(token) except Exception as exc: # pragma: no cover logger.debug("GoalManager: background SessionDB() raised (%s)", exc) db = None @@ -704,8 +731,12 @@ def _get_session_db() -> Optional[Any]: seconds — on the gateway's loop thread that starves the loop-liveness watchdog, which hard-exits the process (exit 75) and crash-loops the gateway (enterprise field report, 2026-08-14). On a cache miss with a running - loop we kick a one-shot background bootstrap and return None; every - caller already degrades gracefully on None, and a later call returns the + loop we kick a one-shot background bootstrap and wait a bounded grace + window for it. The kick call waits the one-time init window + (``_DB_BOOTSTRAP_INIT_WAIT_S``), so a healthy cold init completes and + the first write is not dropped. Later calls wait only the short window + (``_DB_BOOTSTRAP_LOOP_WAIT_S``). On timeout we return None. Every + caller degrades gracefully on None, and a later call returns the cached instance. """ try: @@ -745,12 +776,20 @@ def _get_session_db() -> Optional[Any]: name="goals-sessiondb-bootstrap", daemon=True, ).start() - # Grace window: a healthy init finishes in tens of ms, so waiting - # briefly keeps goal/heartbeat persistence working on the very first - # loop-thread call instead of silently dropping it. A contended init - # (the crash-loop scenario) exceeds the window and we degrade to - # None — a bounded stall far below the watchdog's probe timeout. - done.wait(_DB_BOOTSTRAP_LOOP_WAIT_S) + # This call starts the bootstrap, so it pays the one-time + # init cost. Wait long enough for a healthy cold init + # (~300ms warm, more on slow CI) to finish. This keeps the + # first goal/heartbeat write from being silently dropped. + wait = _DB_BOOTSTRAP_INIT_WAIT_S + else: + # Bootstrap already running: brief grace window only. A + # healthy init usually finishes in tens of ms, so this + # still picks the cached instance up. A contended init + # (the crash-loop scenario) exceeds the window and we + # degrade to None. The stall is bounded, far below the + # watchdog's probe timeout. + wait = _DB_BOOTSTRAP_LOOP_WAIT_S + done.wait(wait) return _DB_CACHE.get(home) try: @@ -799,6 +838,12 @@ def save_goal(session_id: str, state: GoalState) -> None: return db = _get_session_db() if db is None: + logger.warning( + "GoalManager: goal for %s not persisted — session DB " + "unavailable (bootstrap window exceeded, in-memory state " + "still active)", + session_id, + ) return try: db.set_meta(_meta_key(session_id), state.to_json()) diff --git a/hermes_cli/heartbeat.py b/hermes_cli/heartbeat.py index bf0df9f5b6..872fbdd806 100644 --- a/hermes_cli/heartbeat.py +++ b/hermes_cli/heartbeat.py @@ -184,6 +184,12 @@ def save_heartbeat(session_id: str, state: HeartbeatState) -> None: return db = _get_session_db() if db is None: + logger.warning( + "HeartbeatManager: heartbeat for %s not persisted — session " + "DB unavailable (bootstrap window exceeded, in-memory state " + "still active)", + session_id, + ) return try: db.set_meta(_meta_key(session_id), state.to_json()) diff --git a/hermes_cli/loops.py b/hermes_cli/loops.py index f496907c2d..87e07a6842 100644 --- a/hermes_cli/loops.py +++ b/hermes_cli/loops.py @@ -369,30 +369,23 @@ def _meta_key(session_id: str) -> str: return f"{_META_PREFIX}{session_id}" -_DB_CACHE: Dict[str, Any] = {} - - def _get_session_db() -> Optional[Any]: - """One SessionDB per HERMES_HOME (same pattern as goals._get_session_db).""" - try: - from hermes_constants import get_hermes_home - from hermes_state import SessionDB + """One SessionDB per HERMES_HOME. - home = str(get_hermes_home()) + Delegates to the goals module's cached SessionDB so goals, loops, + and heartbeats share one connection (same pattern as + ``hermes_cli/heartbeat.py``). The delegation also inherits the + off-loop bootstrap and the window logic: a cold cache on the loop + thread never runs ``SessionDB()`` inline. The previous copy here + did, which froze the loop for the init duration and dropped the + first ``loop:*`` write (the /goal bug class, #88965). + """ + try: + from hermes_cli.goals import _get_session_db as _goals_db except Exception as exc: # pragma: no cover logger.debug("LoopManager: SessionDB bootstrap failed (%s)", exc) return None - - cached = _DB_CACHE.get(home) - if cached is not None: - return cached - try: - db = SessionDB() - except Exception as exc: # pragma: no cover - logger.debug("LoopManager: SessionDB() raised (%s)", exc) - return None - _DB_CACHE[home] = db - return db + return _goals_db() def load_loop(session_id: str) -> Optional[LoopState]: diff --git a/tests/gateway/test_goal_max_turns_config.py b/tests/gateway/test_goal_max_turns_config.py index 9dda0e44d3..f98eddb1d1 100644 --- a/tests/gateway/test_goal_max_turns_config.py +++ b/tests/gateway/test_goal_max_turns_config.py @@ -1,3 +1,6 @@ +import asyncio +import time + import pytest from gateway.config import GatewayConfig, Platform, PlatformConfig @@ -22,15 +25,8 @@ class _FakeSessionStore: return "agent:main:discord:channel:goal-config" -@pytest.mark.asyncio -async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monkeypatch): - """Gateway /goal should honor top-level goals.max_turns from config.yaml.""" - home = tmp_path / ".hermes" - home.mkdir() - (home / "config.yaml").write_text("goals:\n max_turns: 7\n", encoding="utf-8") - monkeypatch.setenv("HERMES_HOME", str(home)) - goals._DB_CACHE.clear() - +def _make_runner() -> GatewayRunner: + """GatewayRunner skeleton for the /goal command path (no adapters).""" runner = object.__new__(GatewayRunner) runner.config = GatewayConfig( platforms={Platform.DISCORD: PlatformConfig(enabled=True, token="token")} @@ -38,8 +34,11 @@ async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monk runner.session_store = _FakeSessionStore() runner.adapters = {} runner._queued_events = {} + return runner - event = MessageEvent( + +def _make_goal_event() -> MessageEvent: + return MessageEvent( text="/goal ship the benchmark", message_type=MessageType.TEXT, source=SessionSource( @@ -51,6 +50,20 @@ async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monk message_id="msg-goal-config", ) + +@pytest.mark.asyncio +async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monkeypatch): + """Gateway /goal should honor top-level goals.max_turns from config.yaml.""" + home = tmp_path / ".hermes" + home.mkdir() + (home / "config.yaml").write_text("goals:\n max_turns: 7\n", encoding="utf-8") + monkeypatch.setenv("HERMES_HOME", str(home)) + goals._DB_CACHE.clear() + + runner = _make_runner() + + event = _make_goal_event() + response = await GatewayRunner._handle_goal_command(runner, event) try: @@ -60,3 +73,66 @@ async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monk assert state.max_turns == 7 finally: goals._DB_CACHE.clear() + + +@pytest.mark.asyncio +async def test_goal_command_slow_db_init_keeps_loop_free_and_persists(tmp_path, monkeypatch): + """A slow state.db init (cold cache, first /goal of the process) + must not freeze the event loop or silently drop the goal write. + The cache is warmed off-loop, so the reply is honest at any init + duration. Review follow-up (#88965): the bootstrap window alone + froze the loop for the init duration and still dropped the write + past ~1.5s.""" + import hermes_state + + # Past the init window: the window-only path expires its wait and + # drops the write, so this test discriminates warm-up from windows. + # The margin is 2.5s so the loop-gap ceiling below can be 2.0s (the + # flake-policy floor) and an on-loop init still exceeds it. + INIT_S = goals._DB_BOOTSTRAP_INIT_WAIT_S + 2.5 + + class _SlowSessionDB(hermes_state.SessionDB): + def __init__(self, *a, **k): + time.sleep(INIT_S) + super().__init__(*a, **k) + + monkeypatch.setattr(hermes_state, "SessionDB", _SlowSessionDB) + + home = tmp_path / ".hermes" + home.mkdir() + (home / "config.yaml").write_text("goals:\n max_turns: 7\n", encoding="utf-8") + monkeypatch.setenv("HERMES_HOME", str(home)) + goals._DB_CACHE.clear() + + runner = _make_runner() + event = _make_goal_event() + + gaps = {"max": 0.0} + stop = asyncio.Event() + + async def _ticker(): + last = time.monotonic() + while not stop.is_set(): + await asyncio.sleep(0.05) + now = time.monotonic() + gaps["max"] = max(gaps["max"], now - last) + last = now + + ticker = asyncio.create_task(_ticker()) + await asyncio.sleep(0.15) + try: + response = await GatewayRunner._handle_goal_command(runner, event) + + assert "⊙ Goal set (7-turn budget): ship the benchmark" in response + state = goals.GoalManager("sid-gateway-goal-config").state + assert state is not None, "goal write must persist even with a slow init" + # 2.0s is the loose-bound floor from the flake policy. The init + # (4.0s) runs off-loop, so an on-loop regression exceeds this + # ceiling and a loaded runner does not. + assert gaps["max"] < 2.0, ( + f"event loop frozen for {gaps['max']:.2f}s while the init ran off-loop" + ) + finally: + stop.set() + await ticker + goals._DB_CACHE.clear() diff --git a/tests/gateway/test_loop_command.py b/tests/gateway/test_loop_command.py index 73d63317f7..f18d1c5d5c 100644 --- a/tests/gateway/test_loop_command.py +++ b/tests/gateway/test_loop_command.py @@ -10,7 +10,7 @@ from gateway.config import GatewayConfig, Platform, PlatformConfig from gateway.platforms.base import MessageEvent, MessageType from gateway.run import GatewayRunner from gateway.session import SessionSource -from hermes_cli import loops +from hermes_cli import goals, loops class _FakeSessionEntry: @@ -33,9 +33,9 @@ def loop_env(tmp_path, monkeypatch): home = tmp_path / ".hermes" home.mkdir() monkeypatch.setenv("HERMES_HOME", str(home)) - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() yield home - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() def _make_runner(): diff --git a/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py b/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py index c77387db34..12b747eab9 100644 --- a/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py +++ b/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py @@ -109,30 +109,48 @@ def test_worker_thread_constructs_inline(monkeypatch): def test_slow_construction_does_not_block_the_loop(monkeypatch): - """A SessionDB whose init blocks (locked-DB migration) must not stall - the event loop past the watchdog probe window: bounded grace wait, - then degrade to None.""" + """A SessionDB whose init blocks (locked-DB migration) must not + stall the event loop. The kick call is bounded by the one-time init + window. Every later call is bounded by the short per-call window. + Both calls degrade to None.""" import hermes_state class _BlockingDB: def __init__(self): - time.sleep(2.0) # simulated contended migration + # Far past both wait windows. The margin keeps the "still + # None" assertions from racing the bootstrap thread on a + # loaded runner (negative-timing race, flake policy). + time.sleep(6.0) # simulated contended migration def get_meta(self, key): return None monkeypatch.setattr(hermes_state, "SessionDB", _BlockingDB) elapsed = None + elapsed2 = None result = "UNSET" + result2 = "UNSET" async def main(): - nonlocal elapsed, result + nonlocal elapsed, elapsed2, result, result2 t0 = time.monotonic() result = goals._get_session_db() elapsed = time.monotonic() - t0 + t0 = time.monotonic() + result2 = goals._get_session_db() + elapsed2 = time.monotonic() - t0 asyncio.run(main()) assert result is None, "contended init must degrade to None" - assert elapsed is not None and elapsed < 1.0, ( - f"loop-thread call blocked for {elapsed:.2f}s — watchdog territory" + # The slack on each ceiling is 2.0s, the loose-bound floor from the + # flake policy. The windows are 1.5s and 0.25s, so the two ceilings + # stay far apart and a loaded runner does not cross either one. + assert elapsed is not None and elapsed < goals._DB_BOOTSTRAP_INIT_WAIT_S + 2.0, ( + f"kick call blocked for {elapsed:.2f}s — past the one-time init window" + ) + # The in-flight call waits the short window, not the init window. + # A contended migration must not stall the loop repeatedly. + assert result2 is None, "contended init must keep degrading to None" + assert elapsed2 is not None and elapsed2 < goals._DB_BOOTSTRAP_LOOP_WAIT_S + 2.0, ( + f"in-flight call blocked for {elapsed2:.2f}s — watchdog territory" ) diff --git a/tests/hermes_cli/test_loops.py b/tests/hermes_cli/test_loops.py index dba5ed39a4..99dc26d4ec 100644 --- a/tests/hermes_cli/test_loops.py +++ b/tests/hermes_cli/test_loops.py @@ -23,11 +23,11 @@ def hermes_home(tmp_path, monkeypatch): monkeypatch.setattr(Path, "home", lambda: tmp_path) monkeypatch.setenv("HERMES_HOME", str(home)) - from hermes_cli import loops + from hermes_cli import goals - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() yield home - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() # ────────────────────────────────────────────────────────────────────── diff --git a/tests/tui_gateway/test_loop_command.py b/tests/tui_gateway/test_loop_command.py index 41522bcc49..f437e26974 100644 --- a/tests/tui_gateway/test_loop_command.py +++ b/tests/tui_gateway/test_loop_command.py @@ -25,11 +25,11 @@ def hermes_home(tmp_path, monkeypatch): monkeypatch.setattr(Path, "home", lambda: tmp_path) monkeypatch.setenv("HERMES_HOME", str(home)) - from hermes_cli import loops + from hermes_cli import goals - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() yield home - loops._DB_CACHE.clear() + goals._DB_CACHE.clear() @pytest.fixture() From 46d8cf0be32efdaeb6591d2558a9824adbe528b5 Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 12:34:52 -0400 Subject: [PATCH 009/426] fix(gateway): /loop paths warm the SessionDB cache off-loop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The loops delegation (previous commit) moved /loop onto the shared bootstrap windows, but the loop-class gateway callers still constructed LoopManager on the loop thread with no warm-up — the same false-ack class as /goal, one sibling over: - The /loop command handler: a cold init past the window made save_loop discard the write while the reply claimed the loop was set. Reproduced with a 2s init: "↻ Loop set" with nothing persisted; with the warm-up the loop persists. - _post_turn_loop_completion: a cold cache at the turn boundary stalled the loop for the init duration and could drop the tick-completion write. - _loop_wakeup_watcher: the scan reads every persisted loop, so a cold cache ran the state.db init on the loop thread before the first read. All three now warm the cache off-loop first (same helper, same reason as the goal paths). save_loop also logs at WARNING when it drops a write, matching save_goal: the reply has already told the user the loop was set. Review findings (PR #88965): the goal and heartbeat command paths warmed off-loop, the loop-class paths did not. --- gateway/run.py | 11 +++++++++++ gateway/slash_commands.py | 4 ++++ hermes_cli/loops.py | 6 ++++++ 3 files changed, 21 insertions(+) diff --git a/gateway/run.py b/gateway/run.py index 04c80279c8..b5d49dbcb3 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -21397,6 +21397,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not sid: return + # Warm the SessionDB cache off-loop. A cold cache at the turn + # boundary stalls the loop for the init duration and can drop + # the tick-completion write (the /goal continuation seam, one + # sibling over). + await self._warm_goals_session_db("loop completion") + mgr = LoopManager(session_id=sid) state = mgr.state if state is None or not state.awaiting_response: @@ -21435,6 +21441,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew list_active_loops, ) + # Warm the cache off-loop once per scan. The scan reads + # every persisted loop, so a cold cache runs the state.db + # init on the loop thread before the first read. + await self._warm_goals_session_db("loop wakeup") + now = time.time() for sid, state in list_active_loops(): if state.awaiting_response or now < state.next_due_at: diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index d66162fb60..78c276dcfc 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -3038,6 +3038,10 @@ class GatewaySlashCommandsMixin: except Exception as exc: logger.debug("loop manager unavailable: %s", exc) return None, None + # Warm the SessionDB cache off-loop. A cold cache drops the first + # /loop write while the reply claims the loop was set (same class + # as the /goal false-ack fix). + await self._warm_goals_session_db("loop manager") try: session_entry = await self.async_session_store.get_or_create_session(event.source) except Exception: diff --git a/hermes_cli/loops.py b/hermes_cli/loops.py index 87e07a6842..7368d85b2e 100644 --- a/hermes_cli/loops.py +++ b/hermes_cli/loops.py @@ -415,6 +415,12 @@ def save_loop(session_id: str, state: LoopState) -> None: return db = _get_session_db() if db is None: + logger.warning( + "LoopManager: loop for %s not persisted — session DB " + "unavailable (bootstrap window exceeded, in-memory state " + "still active)", + session_id, + ) return try: db.set_meta(_meta_key(session_id), state.to_json()) From 6e67841a9a45c8160bc85846001004edbca97386 Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 17:55:29 -0400 Subject: [PATCH 010/426] refactor(goals): share one dropped-write warning across managers Review fixups for #88965. The goal, loop, and heartbeat managers each had a copy of the same WARNING text. The shared _warn_dropped_write helper in goals.py keeps the three logs identical and greppable as one bug class. The _warm_goals_session_db parameter is now label. The old name ctx said context, but the value is a log label. --- gateway/run.py | 4 ++-- hermes_cli/goals.py | 23 +++++++++++++++++------ hermes_cli/heartbeat.py | 9 +++------ hermes_cli/loops.py | 9 +++------ 4 files changed, 25 insertions(+), 20 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index b5d49dbcb3..8895660879 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -20998,7 +20998,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: return 20 - async def _warm_goals_session_db(self, ctx: str) -> None: + async def _warm_goals_session_db(self, label: str) -> None: """Warm the goals SessionDB cache off-loop (best-effort). A cold cache runs the state.db init on the loop thread behind the @@ -21013,7 +21013,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await self._run_in_executor_with_context(_warm_goals_db) except Exception as exc: - logger.warning("%s: session DB warm-up failed: %s", ctx, exc) + logger.warning("%s: session DB warm-up failed: %s", label, exc) async def _get_goal_manager_for_event(self, event: "MessageEvent"): """Return a GoalManager bound to the session for this gateway event. diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index 39870322af..61e952d44c 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -811,6 +811,22 @@ def _get_session_db() -> Optional[Any]: return db +def _warn_dropped_write(manager: str, kind: str, session_id: str) -> None: + """Log a dropped state write at WARNING. + + The reply already told the user that the state was set. A silent + drop makes that reply a lie. One shared message keeps the goal, + loop, and heartbeat logs greppable as one bug class. + """ + logger.warning( + "%s: %s for %s not persisted — session DB unavailable " + "(bootstrap window exceeded, in-memory state still active)", + manager, + kind, + session_id, + ) + + def load_goal(session_id: str) -> Optional[GoalState]: """Load the goal for a session, or None if none exists.""" if not session_id: @@ -838,12 +854,7 @@ def save_goal(session_id: str, state: GoalState) -> None: return db = _get_session_db() if db is None: - logger.warning( - "GoalManager: goal for %s not persisted — session DB " - "unavailable (bootstrap window exceeded, in-memory state " - "still active)", - session_id, - ) + _warn_dropped_write("GoalManager", "goal", session_id) return try: db.set_meta(_meta_key(session_id), state.to_json()) diff --git a/hermes_cli/heartbeat.py b/hermes_cli/heartbeat.py index 872fbdd806..11df286869 100644 --- a/hermes_cli/heartbeat.py +++ b/hermes_cli/heartbeat.py @@ -184,12 +184,9 @@ def save_heartbeat(session_id: str, state: HeartbeatState) -> None: return db = _get_session_db() if db is None: - logger.warning( - "HeartbeatManager: heartbeat for %s not persisted — session " - "DB unavailable (bootstrap window exceeded, in-memory state " - "still active)", - session_id, - ) + from hermes_cli.goals import _warn_dropped_write + + _warn_dropped_write("HeartbeatManager", "heartbeat", session_id) return try: db.set_meta(_meta_key(session_id), state.to_json()) diff --git a/hermes_cli/loops.py b/hermes_cli/loops.py index 7368d85b2e..0ee3c31b1c 100644 --- a/hermes_cli/loops.py +++ b/hermes_cli/loops.py @@ -415,12 +415,9 @@ def save_loop(session_id: str, state: LoopState) -> None: return db = _get_session_db() if db is None: - logger.warning( - "LoopManager: loop for %s not persisted — session DB " - "unavailable (bootstrap window exceeded, in-memory state " - "still active)", - session_id, - ) + from hermes_cli.goals import _warn_dropped_write + + _warn_dropped_write("LoopManager", "loop", session_id) return try: db.set_meta(_meta_key(session_id), state.to_json()) From 9d86ac62b41d2b953aafc7d9c1e75711762f68e5 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 15:57:53 -0700 Subject: [PATCH 011/426] test: shrink the goal-DB timing tests to sub-second wall time MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up on the salvaged #88965 work: - test_goal_command_slow_db_init_still_persists: drop the 4s slow-init loop-gap harness (wall-clock gap assertions on shared CI runners are their own flake class; loop-freeze bounds are already covered in test_goals_db_bootstrap_off_loop.py). The persistence contract keeps its discriminating power by shrinking the monkeypatched init window (0.2s) under a 0.8s slow init — the window-only path still fails it. - test_slow_construction_does_not_block_the_loop: monkeypatch both bootstrap windows down (0.3s/0.05s) and shrink the blocking init from 6s to 1.5s; the two-window contract is what's under test, not the production constants. Adds an elapsed ordering assertion so the kick-vs-in-flight window distinction stays pinned. Combined wall time for the pair: ~10.5s -> ~1.4s. --- tests/gateway/test_goal_max_turns_config.py | 52 +++++++------------ .../test_goals_db_bootstrap_off_loop.py | 34 ++++++++---- 2 files changed, 42 insertions(+), 44 deletions(-) diff --git a/tests/gateway/test_goal_max_turns_config.py b/tests/gateway/test_goal_max_turns_config.py index f98eddb1d1..4e4f8657d9 100644 --- a/tests/gateway/test_goal_max_turns_config.py +++ b/tests/gateway/test_goal_max_turns_config.py @@ -76,22 +76,26 @@ async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monk @pytest.mark.asyncio -async def test_goal_command_slow_db_init_keeps_loop_free_and_persists(tmp_path, monkeypatch): - """A slow state.db init (cold cache, first /goal of the process) - must not freeze the event loop or silently drop the goal write. - The cache is warmed off-loop, so the reply is honest at any init - duration. Review follow-up (#88965): the bootstrap window alone - froze the loop for the init duration and still dropped the write - past ~1.5s.""" +async def test_goal_command_slow_db_init_still_persists(tmp_path, monkeypatch): + """A slow state.db init (cold cache, first /goal of the process) must + not silently drop the goal write: the gateway warms the cache off-loop, + so the reply is honest at any init duration. + + The init is deliberately slower than the (shrunk) bootstrap init + window, so the window-only path would drop the write — this test + discriminates the off-loop warm-up from mere window-widening without + multi-second sleeps. Loop-freeze bounds are covered separately in + tests/hermes_cli/test_goals_db_bootstrap_off_loop.py; no wall-clock + gap assertions here (those are their own flake class). + """ import hermes_state - # Past the init window: the window-only path expires its wait and - # drops the write, so this test discriminates warm-up from windows. - # The margin is 2.5s so the loop-gap ceiling below can be 2.0s (the - # flake-policy floor) and an on-loop init still exceeds it. - INIT_S = goals._DB_BOOTSTRAP_INIT_WAIT_S + 2.5 + monkeypatch.setattr(goals, "_DB_BOOTSTRAP_INIT_WAIT_S", 0.2) + INIT_S = 0.8 # past the shrunk init window - class _SlowSessionDB(hermes_state.SessionDB): + real_session_db = hermes_state.SessionDB + + class _SlowSessionDB(real_session_db): def __init__(self, *a, **k): time.sleep(INIT_S) super().__init__(*a, **k) @@ -107,32 +111,12 @@ async def test_goal_command_slow_db_init_keeps_loop_free_and_persists(tmp_path, runner = _make_runner() event = _make_goal_event() - gaps = {"max": 0.0} - stop = asyncio.Event() - - async def _ticker(): - last = time.monotonic() - while not stop.is_set(): - await asyncio.sleep(0.05) - now = time.monotonic() - gaps["max"] = max(gaps["max"], now - last) - last = now - - ticker = asyncio.create_task(_ticker()) - await asyncio.sleep(0.15) try: response = await GatewayRunner._handle_goal_command(runner, event) assert "⊙ Goal set (7-turn budget): ship the benchmark" in response state = goals.GoalManager("sid-gateway-goal-config").state assert state is not None, "goal write must persist even with a slow init" - # 2.0s is the loose-bound floor from the flake policy. The init - # (4.0s) runs off-loop, so an on-loop regression exceeds this - # ceiling and a loaded runner does not. - assert gaps["max"] < 2.0, ( - f"event loop frozen for {gaps['max']:.2f}s while the init ran off-loop" - ) + assert state.max_turns == 7 finally: - stop.set() - await ticker goals._DB_CACHE.clear() diff --git a/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py b/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py index 12b747eab9..5bc8cae5bf 100644 --- a/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py +++ b/tests/hermes_cli/test_goals_db_bootstrap_off_loop.py @@ -112,15 +112,24 @@ def test_slow_construction_does_not_block_the_loop(monkeypatch): """A SessionDB whose init blocks (locked-DB migration) must not stall the event loop. The kick call is bounded by the one-time init window. Every later call is bounded by the short per-call window. - Both calls degrade to None.""" + Both calls degrade to None. + + The windows are monkeypatched down so the test costs well under a + second of wall time instead of sleeping through the production + 1.5s window plus margins; the two-window CONTRACT is what's under + test, not the production constants. + """ import hermes_state + monkeypatch.setattr(goals, "_DB_BOOTSTRAP_INIT_WAIT_S", 0.3) + monkeypatch.setattr(goals, "_DB_BOOTSTRAP_LOOP_WAIT_S", 0.05) + class _BlockingDB: def __init__(self): - # Far past both wait windows. The margin keeps the "still - # None" assertions from racing the bootstrap thread on a - # loaded runner (negative-timing race, flake policy). - time.sleep(6.0) # simulated contended migration + # Far past both (shrunk) wait windows. The margin keeps the + # "still None" assertions from racing the bootstrap thread on + # a loaded runner (negative-timing race, flake policy). + time.sleep(1.5) # simulated contended migration def get_meta(self, key): return None @@ -142,15 +151,20 @@ def test_slow_construction_does_not_block_the_loop(monkeypatch): asyncio.run(main()) assert result is None, "contended init must degrade to None" - # The slack on each ceiling is 2.0s, the loose-bound floor from the - # flake policy. The windows are 1.5s and 0.25s, so the two ceilings - # stay far apart and a loaded runner does not cross either one. - assert elapsed is not None and elapsed < goals._DB_BOOTSTRAP_INIT_WAIT_S + 2.0, ( + # Each ceiling gets ~1s slack over its (shrunk) window; the 1.5s + # blocking init exceeds both ceilings if it ever runs on-loop, and a + # loaded runner does not cross either one. + assert elapsed is not None and elapsed < 1.0, ( f"kick call blocked for {elapsed:.2f}s — past the one-time init window" ) # The in-flight call waits the short window, not the init window. # A contended migration must not stall the loop repeatedly. assert result2 is None, "contended init must keep degrading to None" - assert elapsed2 is not None and elapsed2 < goals._DB_BOOTSTRAP_LOOP_WAIT_S + 2.0, ( + assert elapsed2 is not None and elapsed2 < 1.0, ( f"in-flight call blocked for {elapsed2:.2f}s — watchdog territory" ) + # The kick call must wait the LONGER window: otherwise a healthy cold + # init loses its grace period and the first write drops again. + assert elapsed > elapsed2, ( + "kick call should wait the one-time init window; in-flight calls the short one" + ) From 97b41f8cf34c8d62da705b1a4e3bba94fa28ea19 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 16:04:01 -0700 Subject: [PATCH 012/426] feat(bot-mode): group chats accept PDFs, files, and drag & drop Completes group/1:1 attachment parity (#88983). PR #89486 covered images; this adds the remaining half: - The composer picker accepts any file type; kind (image/pdf/file) decides the staging RPC. Paste handlers accept non-image files too. - Drag & drop anywhere on the room drops into the active composer (open reply box, else main), with a drop overlay naming the target. - Member turns stage PDFs via pdf.attach (rendered per-page into vision tiles by the gateway) and other files via file.attach; each returned @file: ref is appended to that member's turn prompt so file tools can read the artifact. Failed attaches still degrade to text-only. - Transcript markers distinguish [attached PDF: x] / [attached file: x] / [attached image: x]; room log renders non-image attachments as named chips, pending chips show type icons. Tests: 3 new vm-harness tests (per-kind RPC routing across members, @file: ref injection into turn prompts, transcript labels); 279 total pass. --- .../desktop/src/plugins/hermes-bots/plugin.js | 202 ++++++++++++++---- .../tests/group-chat-attachments.test.mjs | 58 ++++- 2 files changed, 211 insertions(+), 49 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 20c6f3b7eb..d013949e26 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -1929,20 +1929,35 @@ function pickImageFromDevice() { }) } -// ── group-chat attachments: pick + downscale images the room's members see ── +// ── group-chat attachments: pick/paste/drop files the room's members see ──── -/** File objects → [{ name, data }] (data URLs), oversized files skipped with - * a toast. Shared by the picker button and the composer's paste handler. */ -async function filesToGroupImages(files) { +/** Classify a picked file for the group-attachment pipeline. */ +function groupAttachmentKind(file) { + if (/^image\//.test(file.type || '')) { + return 'image' + } + + if (file.type === 'application/pdf' || /\.pdf$/i.test(file.name || '')) { + return 'pdf' + } + + return 'file' +} + +/** File objects → [{ name, data, kind }] (data URLs), oversized files skipped + * with a toast. Images are downscaled; PDFs and other files ride as raw data + * URLs for the gateway's pdf.attach / file.attach staging. Shared by the + * picker button, the composer paste handler, and room drag & drop. */ +async function filesToGroupAttachments(files) { const picked = [] for (const file of [...(files || [])]) { - if (!file || !/^image\//.test(file.type || '')) { + if (!file) { continue } if (file.size > 15_000_000) { - host.notify({ kind: 'error', message: `${file.name || 'image'}: too large (max 15MB).` }) + host.notify({ kind: 'error', message: `${file.name || 'attachment'}: too large (max 15MB).` }) continue } @@ -1953,23 +1968,29 @@ async function filesToGroupImages(files) { reader.readAsDataURL(file) }) - if (data) { - picked.push({ name: file.name || 'pasted image', data: await normalizeGroupAttachment(data) }) + if (!data) { + continue } + + const kind = groupAttachmentKind(file) + picked.push({ + name: file.name || (kind === 'image' ? 'pasted image' : 'attachment'), + data: kind === 'image' ? await normalizeGroupAttachment(data) : data, + kind + }) } return picked } -/** Multi-file variant of pickImageFromDevice for the group composer. Resolves - * to [{ name, data }] (data URLs); oversized files are skipped with a toast. */ -function pickGroupImages() { +/** Multi-file picker for the group composer — any file type; kind decides + * the staging RPC. Resolves to [{ name, data, kind }]. */ +function pickGroupAttachments() { return new Promise(resolve => { const input = document.createElement('input') input.type = 'file' - input.accept = 'image/png,image/jpeg,image/webp,image/gif' input.multiple = true - input.onchange = () => resolve(filesToGroupImages(input.files)) + input.onchange = () => resolve(filesToGroupAttachments(input.files)) input.click() }) } @@ -3675,10 +3696,15 @@ function groupSpeakerLabel(name) { /** Room-log line as a member sees it: `Name (user): …` / `Name: …` / * `Name (you): …`. */ function formatGroupChatLine(entry, viewerName) { - // Attachments are staged into each member's session as real images; the - // transcript line names them so the delta text and the pixels line up. + // Attachments are staged into each member's session as real payloads; the + // transcript line names them so the delta text and the bytes line up. const attached = Array.isArray(entry.images) && entry.images.length - ? ` ${entry.images.map(img => `[attached image: ${img.name || 'image'}]`).join(' ')}` + ? ` ${entry.images + .map(img => { + const label = img.kind === 'pdf' ? 'attached PDF' : img.kind === 'file' ? 'attached file' : 'attached image' + return `[${label}: ${img.name || 'image'}]` + }) + .join(' ')}` : '' if (entry.from.kind === 'user') { @@ -4045,28 +4071,55 @@ async function runGroupChatMemberTurn(group, member, prompt, thread, images) { } // Stage this delta's attachments into the member's session so the model - // receives the actual pixels with the prompt (the same image.attach_bytes - // path the 1:1 chat uses — it also works cross-connection, where the - // member's gateway can't see this machine's files). A failed attach - // degrades that member to text-only; the transcript line still names the - // image so the member knows something was shared. + // receives the actual payload with the prompt — the same attach RPCs the + // 1:1 chat uses (they also work cross-connection, where the member's + // gateway can't see this machine's files). Images queue as vision tiles, + // PDFs render per-page via pdf.attach, and other files materialize in the + // session workspace (their @file: refs are appended to the prompt so the + // member's file tools can read them). A failed attach degrades that + // member to text-only; the transcript line still names the attachment so + // the member knows something was shared. + const fileRefs = [] + for (const img of Array.isArray(images) ? images : []) { if (!img || typeof img.data !== 'string' || !img.data) { continue } try { - await requestForBot(member, 'image.attach_bytes', { - session_id: runtime, - content_base64: img.data, - filename: img.name || 'attachment.png' - }) + if (img.kind === 'pdf') { + await requestForBot(member, 'pdf.attach', { + session_id: runtime, + content_base64: img.data, + filename: img.name || 'attachment.pdf' + }) + } else if (img.kind === 'file') { + const res = await requestForBot(member, 'file.attach', { + session_id: runtime, + data_url: img.data, + name: img.name || 'attachment' + }) + + if (res?.ref_text) { + fileRefs.push(`${img.name || 'attachment'} → ${res.ref_text}`) + } + } else { + await requestForBot(member, 'image.attach_bytes', { + session_id: runtime, + content_base64: img.data, + filename: img.name || 'attachment.png' + }) + } } catch { /* text-only fallback for this member */ } } - await requestForBot(member, 'prompt.submit', { session_id: runtime, text: prompt }) + const turnText = fileRefs.length + ? `${prompt}\n\nAttached files staged in your session workspace:\n${fileRefs.join('\n')}` + : prompt + + await requestForBot(member, 'prompt.submit', { session_id: runtime, text: turnText }) const started = Date.now() let deadline = started + GROUP_TURN_TIMEOUT_MS @@ -8470,16 +8523,34 @@ function GroupChatWorkspace({ group, members, onBack }) { setPendingImages(prev => ({ ...prev, [key]: (prev[key] || []).filter((_, i) => i !== index) })) } - // Ctrl/⌘-V a screenshot into any composer in this room. + // Ctrl/⌘-V a screenshot (or any file) into any composer in this room. const pasteImages = (thread, event) => { - const files = [...(event.clipboardData?.files || [])].filter(f => /^image\//.test(f.type || '')) + const files = [...(event.clipboardData?.files || [])] if (!files.length) { return } event.preventDefault() - void filesToGroupImages(files).then(picked => addImages(thread, picked)) + void filesToGroupAttachments(files).then(picked => addImages(thread, picked)) + } + + // Drag & drop anywhere on the room drops into the ACTIVE composer — the + // open reply box when one owns the composer, else the main (new-thread) + // composer. Matches the 1:1 chat's drop affordance. + const [dragOver, setDragOver] = useState(false) + + const dropFiles = event => { + const files = [...(event.dataTransfer?.files || [])] + + setDragOver(false) + + if (!files.length) { + return + } + + event.preventDefault() + void filesToGroupAttachments(files).then(picked => addImages(replyThread, picked)) } const header = jsxs('div', { @@ -8584,8 +8655,9 @@ function GroupChatWorkspace({ group, members, onBack }) { setOpenThreads(prev => ({ ...prev, [thread]: true })) } - /** Pending-attachment chips + the paperclip picker for one composer - * (thread = null → main). Chips preview the image and X removes it. */ + /** Pending-attachment chips + the picker for one composer (thread = null → + * main). Image chips preview the pixels; PDFs/files show a type icon. + * X removes it. */ const attachmentRow = thread => { const images = imagesFor(thread) @@ -8600,7 +8672,12 @@ function GroupChatWorkspace({ group, members, onBack }) { className: 'flex items-center gap-1 rounded-md border border-(--ui-stroke-secondary) bg-(--ui-bg-secondary,#181818) px-1 py-0.5', children: [ - jsx('img', { src: img.data, alt: '', className: 'size-6 rounded object-cover' }), + img.kind === 'pdf' || img.kind === 'file' + ? jsx(Codicon, { + name: img.kind === 'pdf' ? 'file-pdf' : 'file', + className: 'text-[0.9rem] text-(--ui-text-tertiary)' + }) + : jsx('img', { src: img.data, alt: '', className: 'size-6 rounded object-cover' }), jsx('span', { className: 'max-w-32 truncate text-[0.65rem] text-(--ui-text-tertiary)', children: img.name || 'image' @@ -8624,9 +8701,9 @@ function GroupChatWorkspace({ group, members, onBack }) { variant: 'ghost', size: 'sm', className: 'shrink-0 text-(--ui-text-tertiary) hover:text-foreground', - title: 'Attach images — every responding bot sees them', - onClick: () => void pickGroupImages().then(picked => addImages(thread, picked)), - children: jsx(Codicon, { name: 'device-camera' }) + title: 'Attach files — every responding bot sees them', + onClick: () => void pickGroupAttachments().then(picked => addImages(thread, picked)), + children: jsx(Codicon, { name: 'attach' }) }) // One log entry, rendered exactly as before conversation folding existed. @@ -8727,18 +8804,30 @@ function GroupChatWorkspace({ group, members, onBack }) { 'data-selectable-text': 'true', children: Streamdown ? jsx(Streamdown, { children: entry.text }) : entry.text }), - // User attachments: the images every responding bot was shown. + // User attachments: what every responding bot was + // shown — image previews, or a named chip for + // PDFs/files. Array.isArray(entry.images) && entry.images.length ? jsx('div', { - className: 'mt-1 flex flex-wrap gap-1.5', + className: 'mt-1 flex flex-wrap items-center gap-1.5', children: entry.images.map((img, imgIndex) => - jsx('img', { - src: img.data, - alt: img.name || 'attached image', - title: img.name || 'attached image', - className: - 'max-h-40 max-w-60 rounded-md border border-(--ui-stroke-secondary) object-contain' - }, `${entryKey}:img:${imgIndex}`) + img.kind === 'pdf' || img.kind === 'file' + ? jsxs('div', { + className: + 'flex items-center gap-1 rounded-md border border-(--ui-stroke-secondary) px-1.5 py-1 text-[0.65rem] text-(--ui-text-tertiary)', + title: img.name || 'attached file', + children: [ + jsx(Codicon, { name: img.kind === 'pdf' ? 'file-pdf' : 'file', className: 'text-[0.8rem]' }), + jsx('span', { className: 'max-w-48 truncate', children: img.name || 'attached file' }) + ] + }, `${entryKey}:img:${imgIndex}`) + : jsx('img', { + src: img.data, + alt: img.name || 'attached image', + title: img.name || 'attached image', + className: + 'max-h-40 max-w-60 rounded-md border border-(--ui-stroke-secondary) object-contain' + }, `${entryKey}:img:${imgIndex}`) ) }) : null @@ -8880,8 +8969,29 @@ function GroupChatWorkspace({ group, members, onBack }) { }) return jsxs('div', { - className: 'flex h-full flex-col', + className: 'relative flex h-full flex-col', + onDragOver: event => { + if ([...(event.dataTransfer?.types || [])].includes('Files')) { + event.preventDefault() + setDragOver(true) + } + }, + onDragLeave: event => { + // Only clear when leaving the room container itself, not when the + // cursor moves between its children. + if (!event.currentTarget.contains(event.relatedTarget)) { + setDragOver(false) + } + }, + onDrop: dropFiles, children: [ + dragOver + ? jsx('div', { + className: + 'pointer-events-none absolute inset-0 z-40 flex items-center justify-center border-2 border-dashed border-(--ui-accent,#4f9cf9) text-sm font-medium text-(--ui-accent,#4f9cf9)', + children: replyThread ? 'Drop to attach to this thread reply' : 'Drop to attach — every responding bot sees it' + }, 'dropzone') + : null, header, jsx(ScrollArea, { className: 'min-h-0 flex-1', diff --git a/apps/desktop/src/plugins/hermes-bots/tests/group-chat-attachments.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/group-chat-attachments.test.mjs index bc3cf1a0d6..2653500428 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/group-chat-attachments.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/group-chat-attachments.test.mjs @@ -61,15 +61,19 @@ function load(turnScript) { running: false } } - if (method === 'image.attach_bytes') { + if (method === 'image.attach_bytes' || method === 'pdf.attach' || method === 'file.attach') { const session = resolveSession(null, params.session_id) attaches.push({ + method, profile: session ? session.profile : null, runtime: params.session_id, - filename: params.filename, - data: params.content_base64, + filename: params.filename || params.name, + data: params.content_base64 || params.data_url, order: calls.length // how many prompt.submits happened before this attach }) + if (method === 'file.attach') { + return { attached: true, ref_text: `@file:attachments/${params.name || 'attachment'}` } + } return { attached: true } } if (method === 'prompt.submit') { @@ -206,3 +210,51 @@ test('an invalid attachment is skipped and the member turn still runs text-only' assert.equal(gc.attaches.length, 0) // invalid image skipped, no attach attempted assert.equal(gc.calls.length, 2) // both turns still ran }) + +const PDF = { name: 'spec.pdf', data: 'data:application/pdf;base64,JVBERi0=', kind: 'pdf' } +const DOC = { name: 'notes.txt', data: 'data:text/plain;base64,aGVsbG8=', kind: 'file' } + +test('PDFs route through pdf.attach and files through file.attach, per member', async () => { + const gc = load(() => '(pass)') + + gc.sendToGroupChat('Mixed', MEMBERS, 'review these', null, [IMG, PDF, DOC]) + await drain(gc, 'Mixed') + + // 3 attachments × 2 members, each via its own RPC. + assert.equal(gc.attaches.length, 6) + const byMethod = {} + for (const a of gc.attaches) { + byMethod[a.method] = (byMethod[a.method] || 0) + 1 + } + assert.deepEqual(byMethod, { 'image.attach_bytes': 2, 'pdf.attach': 2, 'file.attach': 2 }) + + const pdfAttach = gc.attaches.find(a => a.method === 'pdf.attach') + assert.equal(pdfAttach.filename, 'spec.pdf') + assert.equal(pdfAttach.data, PDF.data) +}) + +test('file.attach ref_text is appended to the member turn prompt', async () => { + const gc = load(() => '(pass)') + + gc.sendToGroupChat('Refs', MEMBERS, 'read the notes', null, [DOC]) + await drain(gc, 'Refs') + + assert.equal(gc.calls.length, 2) + for (const call of gc.calls) { + assert.ok(call.prompt.includes('Attached files staged in your session workspace:'), 'prompt carries the staging note') + assert.ok(call.prompt.includes('notes.txt → @file:attachments/notes.txt'), 'prompt carries the @file: ref') + } +}) + +test('formatGroupChatLine labels PDFs and files distinctly', () => { + const gc = load(() => '(pass)') + + const line = gc.formatGroupChatLine( + { from: { kind: 'user', name: 'You' }, text: 'here', images: [PDF, DOC, IMG] }, + 'research' + ) + assert.equal( + line, + 'You (user): here [attached PDF: spec.pdf] [attached file: notes.txt] [attached image: screenshot.png]' + ) +}) From 0ee9bc8d1e16daee44c0b2c466659c8c113461ce Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Tue, 18 Aug 2026 17:34:16 -0500 Subject: [PATCH 013/426] fix(agent): give interrupted-turn hidden placeholder a neutral provider-replay sidecar MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bot-mode interrupted member turns with no visible assistant text persisted an empty assistant row (content="" + display_kind="hidden"). The pre-call sanitizer repair_empty_non_final_messages() re-healed that row on every later call (wire copy only), so the loop never converged (#88955). Stamp api_content="[response interrupted]" (the canonical _INTERRUPTED_PLACEHOLDER) on the hidden placeholder instead. display_kind is stripped before sanitization, but api_content is projected back into content for historical assistant rows, so the provider sees a non-empty neutral turn and the sanitizer stops touching the row — while the durable transcript stays hidden and empty. Uses the neutral interruption text, never the _INTERRUPTED_SCAFFOLD_MARKER, which replaying as assistant text caused #81841. Adds regression coverage proving (A) the placeholder carries the replay sidecar, (B) two consecutive projections converge without sanitizer healing, (C) the sanitizer still repairs genuinely-empty unmarked assistants. Refs #88955 --- agent/conversation_loop.py | 13 +++ contributors/emails/andrexibiza@gmail.com | 1 + tests/run_agent/test_steer.py | 127 +++++++++++++++++++++- 3 files changed, 139 insertions(+), 2 deletions(-) create mode 100644 contributors/emails/andrexibiza@gmail.com diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index ea9e1ce264..8a7a4f33fb 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -331,6 +331,19 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text } if not visible: placeholder["display_kind"] = "hidden" + # Keep the transcript hidden and empty, but give the historical + # API projection a non-empty neutral assistant turn so the + # pre-call sanitizer (repair_empty_non_final_messages) does not + # re-heal this row on every later call (#88955). display_kind is + # stripped before sanitization, while api_content is projected + # back into content for historical assistant rows. Use the + # canonical neutral interruption placeholder, never + # _INTERRUPT_SCAFFOLD_MARKER: replaying the scaffold as assistant + # text made the model echo it and self-replicate ghost rows + # (#81841). + from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER + + placeholder["api_content"] = _INTERRUPTED_PLACEHOLDER append_message(messages, placeholder) append_message( messages, diff --git a/contributors/emails/andrexibiza@gmail.com b/contributors/emails/andrexibiza@gmail.com new file mode 100644 index 0000000000..01c4cfb74a --- /dev/null +++ b/contributors/emails/andrexibiza@gmail.com @@ -0,0 +1 @@ +andrexibiza diff --git a/tests/run_agent/test_steer.py b/tests/run_agent/test_steer.py index d020d80eb8..a02c85ad51 100644 --- a/tests/run_agent/test_steer.py +++ b/tests/run_agent/test_steer.py @@ -327,7 +327,9 @@ class TestActiveTurnRedirectCheckpoint: assert placeholder["role"] == "assistant" assert placeholder["display_kind"] == "hidden" assert placeholder.get("content") == "" - assert not placeholder.get("api_content") + # Neutral provider-replay payload (#88955): keeps the row out of the + # re-heal sanitizer loop; the interrupt scaffold is still never here. + assert placeholder.get("api_content") == "[response interrupted]" assert correction["content"] == "New direction." assert ( "[This response was interrupted by a user correction.]" @@ -353,7 +355,8 @@ class TestActiveTurnRedirectCheckpoint: assert placeholder["role"] == "assistant" assert placeholder.get("display_kind") == "hidden" assert placeholder.get("content") == "" - assert not placeholder.get("api_content") + # Neutral provider-replay payload (#88955), NOT the interrupt scaffold. + assert placeholder.get("api_content") == "[response interrupted]" assert correction["role"] == "user" assert correction["content"] == "Stop and do X instead." assert correction["api_content"].startswith( @@ -362,6 +365,126 @@ class TestActiveTurnRedirectCheckpoint: ) +class TestEmptyHiddenAssistantRehealRegression: + """#88955: a no-visible-text redirect persisted an empty + ``display_kind="hidden"`` assistant placeholder that the pre-call sanitizer + re-healed on every later call (wire copy only, so the loop never converged). + The placeholder must carry a neutral provider-replay ``api_content`` so the + historical API projection fills ``content`` and the sanitizer stops + touching the row — while the durable transcript stays hidden and empty.""" + + def test_active_turn_redirect_hidden_placeholder_has_provider_replay_payload(self): + from agent.conversation_loop import _apply_active_turn_redirect + + agent = _bare_agent() + agent._current_streamed_assistant_text = "" + messages = [{"role": "user", "content": "start"}] + + _apply_active_turn_redirect(agent, messages, "Use Postgres instead.") + + placeholder = messages[-2] + correction = messages[-1] + assert placeholder["role"] == "assistant" + assert placeholder["content"] == "" + assert placeholder["display_kind"] == "hidden" + assert placeholder["api_content"] == "[response interrupted]" + # The user correction keeps clean text in content and the interruption + # context only in its own api_content sidecar. + assert correction["role"] == "user" + assert correction["content"] == "Use Postgres instead." + assert ( + "[This response was interrupted by a user correction.]" + in correction["api_content"] + ) + # #81841: the interrupt scaffold must never reach assistant content or + # api_content (API replay substitutes api_content back into content). + assert ( + "[This response was interrupted by a user correction.]" + not in str(placeholder.get("content") or "") + + str(placeholder.get("api_content") or "") + ) + + def test_hidden_redirect_placeholder_does_not_reheal_on_repeated_projection(self): + from agent.agent_runtime_helpers import ( + _msg_has_payload, + repair_empty_non_final_messages, + ) + from agent.conversation_loop import _apply_active_turn_redirect + + agent = _bare_agent() + agent._current_streamed_assistant_text = "" + messages = [{"role": "user", "content": "start"}] + _apply_active_turn_redirect(agent, messages, "Do X instead.") + durable = list(messages) + + def project(rows): + """Mirror the real send-time projection (conversation_loop.py): + api_content -> content for historical user/assistant rows, and the + display/row bookkeeping stripped from every outgoing copy.""" + out = [] + for msg in rows: + api_msg = dict(msg) + _api_content = api_msg.pop("api_content", None) + api_msg.pop("display_kind", None) + api_msg.pop("display_metadata", None) + api_msg.pop("_row_id", None) + if ( + isinstance(_api_content, str) + and _api_content + and msg.get("role") in ("user", "assistant") + ): + api_msg["content"] = _api_content + out.append(api_msg) + return out + + for _pass in range(2): + projected = project(durable) + hidden_assistant = next( + m for m in projected if m.get("role") == "assistant" + ) + # The provider replay sidecar was projected into content, so the + # row already carries payload and the sanitizer has nothing to heal. + assert _msg_has_payload(hidden_assistant) is True + assert hidden_assistant["content"] == "[response interrupted]" + assert "display_kind" not in hidden_assistant + assert "api_content" not in hidden_assistant + + healed = repair_empty_non_final_messages(projected) + healed_assistant = next( + m for m in healed if m.get("role") == "assistant" + ) + assert healed_assistant["content"] == "[response interrupted]" + assert "display_kind" not in healed_assistant + assert "api_content" not in healed_assistant + # Durable transcript is never mutated by projection or sanitizer. + assert durable == messages + assert durable[1]["content"] == "" + assert durable[1]["display_kind"] == "hidden" + assert durable[1]["api_content"] == "[response interrupted]" + + # #81841 scaffold never appears on the assistant wire. + assert ( + "[This response was interrupted by a user correction.]" + not in healed_assistant["content"] + ) + + def test_empty_non_final_sanitizer_still_repairs_unmarked_empty_assistant(self): + """Control: a genuinely empty non-final assistant with no provider-replay + sidecar is still healed — the fix must not disable the generic net.""" + from agent.agent_runtime_helpers import repair_empty_non_final_messages + + rows = [ + {"role": "user", "content": "start"}, + {"role": "assistant", "content": "", "display_kind": "hidden"}, + {"role": "user", "content": "correction"}, + ] + healed = repair_empty_non_final_messages(rows) + assistant = next(m for m in healed if m.get("role") == "assistant") + assert assistant["content"] == "[response interrupted]" + # The durable list is not mutated (wire-copy-only design). + assert rows[1]["content"] == "" + + class TestSteerInjection: def test_appends_to_last_tool_result(self): agent = _bare_agent() From 693c0e1c62aea75f2dd490b66a960917a62a370d Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Tue, 18 Aug 2026 17:34:16 -0500 Subject: [PATCH 014/426] chore: remove case-colliding agent@Agents-Mac-mini.local contributor entries The tracked contributors/emails/agent@Agents-Mac-mini.local and agent@agents-Mac-mini.local differ only by case, which cannot materialize on case-insensitive filesystems and surfaces one of them as perpetually modified in git status, breaking clean checkouts. These entries are stale agent identifiers, not real contributors; remove both. --- contributors/emails/agent@Agents-Mac-mini.local | 1 - contributors/emails/agent@agents-Mac-mini.local | 1 - 2 files changed, 2 deletions(-) delete mode 100644 contributors/emails/agent@Agents-Mac-mini.local delete mode 100644 contributors/emails/agent@agents-Mac-mini.local diff --git a/contributors/emails/agent@Agents-Mac-mini.local b/contributors/emails/agent@Agents-Mac-mini.local deleted file mode 100644 index 73bf022af1..0000000000 --- a/contributors/emails/agent@Agents-Mac-mini.local +++ /dev/null @@ -1 +0,0 @@ -skip-agent diff --git a/contributors/emails/agent@agents-Mac-mini.local b/contributors/emails/agent@agents-Mac-mini.local deleted file mode 100644 index 850e0a4eb0..0000000000 --- a/contributors/emails/agent@agents-Mac-mini.local +++ /dev/null @@ -1 +0,0 @@ -momomojo From 210cdb0ed35d4f7ef0957182312baaaa9e19bfbc Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 16:09:47 -0700 Subject: [PATCH 015/426] fix(agent): legacy hidden redirect placeholders get the neutral wire payload at projection time (#88955) The salvaged writer-side fix stamps api_content on NEW hidden redirect placeholders, but rows persisted before it (content="" + display_kind=hidden, no sidecar) would keep re-triggering repair_empty_non_final_messages on every call forever. Substitute [response interrupted] on the wire copy at the api_content/display_kind projection stage so legacy sessions converge too. Never the interrupt scaffold (#81841). Durable transcript untouched. Regression tests drive run_conversation end-to-end with a spied sanitizer: the projection must leave the sanitizer nothing to heal (its per-turn warning spam is the bug), verified failing via sabotage run against the writer-only fix. Projection-side approach credit: @JoaoMarcos44 (PR #88996). --- agent/conversation_loop.py | 22 +++++- tests/run_agent/test_steer.py | 136 ++++++++++++++++++++++++++++++++++ 2 files changed, 157 insertions(+), 1 deletion(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 8a7a4f33fb..ab8e60b6d0 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -2136,9 +2136,29 @@ def run_conversation( # from every outgoing copy so strict OpenAI-compatible backends # don't reject the request after a model switch or resumed typed # event row enters the live history. - api_msg.pop("display_kind", None) + _display_kind = api_msg.pop("display_kind", None) api_msg.pop("display_metadata", None) + # Legacy hidden redirect placeholders (#88955): rows persisted + # BEFORE the writer-side api_content stamp in + # _apply_active_turn_redirect are content="" with no sidecar. + # Once display_kind is stripped the pre-call sanitizer + # (repair_empty_non_final_messages) would re-heal such a row on + # every call forever, since the durable transcript is never + # mutated. Give the wire copy the same neutral payload here so + # old sessions converge too. Never the interrupt scaffold — + # replaying scaffold bytes as assistant text is #81841. + if ( + _display_kind == "hidden" + and api_msg.get("role") == "assistant" + and not _api_content + and not (api_msg.get("content") or "").strip() + and not api_msg.get("tool_calls") + ): + from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER + + api_msg["content"] = _INTERRUPTED_PLACEHOLDER + # Durable row identity stamped by _rows_to_conversation so the # desktop can address a specific persisted message (reactions). # Bookkeeping, never a provider field — only the chat-completions diff --git a/tests/run_agent/test_steer.py b/tests/run_agent/test_steer.py index a02c85ad51..cd5266ef3a 100644 --- a/tests/run_agent/test_steer.py +++ b/tests/run_agent/test_steer.py @@ -711,3 +711,139 @@ class TestSteerCommandRegistry: if __name__ == "__main__": # pragma: no cover pytest.main([__file__, "-v"]) + + +class TestLegacyHiddenPlaceholderWireSubstitution: + """Projection-side half of #88955: rows persisted BEFORE the writer-side + ``api_content`` stamp are ``content=""`` + ``display_kind="hidden"`` with + no sidecar. The send-time projection must give the WIRE copy the neutral + ``[response interrupted]`` payload so legacy sessions converge instead of + re-healing forever — while the durable row stays hidden and empty.""" + + def _loop_agent(self): + from unittest.mock import MagicMock, patch + + from run_agent import AIAgent + + with ( + patch("run_agent.get_tool_definitions", return_value=[]), + patch("run_agent.check_toolset_requirements", return_value={}), + patch("run_agent.OpenAI"), + ): + agent = AIAgent( + api_key="test-key-1234567890", + base_url="https://openrouter.ai/api/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + agent.client = MagicMock() + agent._cached_system_prompt = "You are helpful." + agent._use_prompt_caching = False + agent.tool_delay = 0 + agent.compression_enabled = False + agent.save_trajectories = False + return agent + + def test_legacy_empty_hidden_assistant_row_gets_neutral_wire_payload(self): + """The projection itself must fill the row — the sanitizer must have + NOTHING left to heal (its per-turn warning spam IS the bug).""" + from unittest.mock import patch + + import agent.agent_runtime_helpers as _arh + + from tests.run_agent.test_run_agent import _mock_response + + agent = self._loop_agent() + agent.client.chat.completions.create.side_effect = [ + _mock_response(content="ok", finish_reason="stop"), + ] + sanitizer_inputs = [] + _real_repair = _arh.repair_empty_non_final_messages + + def _spy_repair(messages, *a, **k): + sanitizer_inputs.append( + [ + (m.get("role"), m.get("content")) + for m in messages + if isinstance(m, dict) + ] + ) + return _real_repair(messages, *a, **k) + # Legacy pre-fix row: no api_content sidecar. + history = [ + {"role": "user", "content": "start"}, + {"role": "assistant", "content": "", "display_kind": "hidden"}, + {"role": "user", "content": "correction", "finish_reason": "stop"}, + {"role": "assistant", "content": "earlier reply", "finish_reason": "stop"}, + ] + + with ( + patch.object(agent, "_flush_messages_to_session_db"), + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object( + _arh, "repair_empty_non_final_messages", side_effect=_spy_repair + ), + ): + agent.run_conversation("next question", conversation_history=history) + + # Precondition: the sanitizer actually ran on this call path. + assert sanitizer_inputs, "sanitizer was never invoked — test is vacuous" + # The projection already filled the legacy row BEFORE sanitization: + # every assistant row the sanitizer saw carried payload, so it healed 0. + for snapshot in sanitizer_inputs: + for role, content in snapshot: + if role == "assistant": + assert (content or "").strip(), ( + "sanitizer still received an empty assistant row — " + "the re-heal loop is back (#88955)" + ) + + wire = agent.client.chat.completions.create.call_args.kwargs["messages"] + wire_assistants = [m for m in wire if m.get("role") == "assistant"] + legacy = wire_assistants[0] + # Substituted on the wire by the projection (not the sanitizer): + assert legacy["content"] == "[response interrupted]" + assert "display_kind" not in legacy + # #81841: never the interrupt scaffold. + assert "[This response was interrupted" not in legacy["content"] + # Durable history untouched. + assert history[1]["content"] == "" + assert history[1]["display_kind"] == "hidden" + assert "api_content" not in history[1] + + def test_hidden_row_with_tool_calls_or_text_is_not_touched(self): + from agent.conversation_loop import _clone_message_for_send # noqa: F401 + from unittest.mock import patch + + from tests.run_agent.test_run_agent import _mock_response + + agent = self._loop_agent() + agent.client.chat.completions.create.side_effect = [ + _mock_response(content="ok", finish_reason="stop"), + ] + history = [ + {"role": "user", "content": "start"}, + { + "role": "assistant", + "content": "visible text", + "display_kind": "hidden", + "finish_reason": "stop", + }, + {"role": "user", "content": "more"}, + {"role": "assistant", "content": "reply", "finish_reason": "stop"}, + ] + + with ( + patch.object(agent, "_flush_messages_to_session_db"), + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + agent.run_conversation("next", conversation_history=history) + + wire = agent.client.chat.completions.create.call_args.kwargs["messages"] + wire_assistants = [m for m in wire if m.get("role") == "assistant"] + assert wire_assistants[0]["content"] == "visible text" From 0c5f195ee238a47b2897f5fdee48cf95dd0c59c1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 16:33:51 -0700 Subject: [PATCH 016/426] docs: document one-click plugin install links (hermes://plugin/install) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The deeplink-driven plugin install flow shipped in #89464 (salvage of #82735 by @serefyarar) had no docs. Adds: - user-guide/features/plugins.md: "One-click install links (Desktop)" section under Managing plugins — link forms (repo/enable/force), the confirm-first dialog contract (never auto-installs, same install-time security scanning as the CLI), hybrid-repo behavior, legacy plugin-agent/plugin-desktop routing, hermes-dev:// in dev builds, and the no-SDK anchor example. Cross-links the MCP "Add to Hermes link" equivalent. - developer-guide/desktop-plugin-sdk.md: "Distributing with an install link" section so plugin authors find the link form next to the packaging docs. --- .../developer-guide/desktop-plugin-sdk.md | 15 ++++++++ website/docs/user-guide/features/plugins.md | 35 +++++++++++++++++++ 2 files changed, 50 insertions(+) diff --git a/website/docs/developer-guide/desktop-plugin-sdk.md b/website/docs/developer-guide/desktop-plugin-sdk.md index 83d4a35971..a111e7e800 100644 --- a/website/docs/developer-guide/desktop-plugin-sdk.md +++ b/website/docs/developer-guide/desktop-plugin-sdk.md @@ -688,6 +688,21 @@ only locally installed packages contribute a desktop half (same rule as the standalone door). ::: +### Distributing with an install link {#install-link} + +Ship your plugin repo (agent half, desktop half, or both) and link to it with +the `hermes://` scheme — a plain anchor on your website or README: + +```html +Install in Hermes +``` + +The user gets a confirmation dialog (repo id, source links, a probe of what +the repo ships) and picks components before anything is installed — deep links +never auto-install. `force=1` replaces an existing install; dev builds use +`hermes-dev://`. Full link reference: +[One-click install links](/user-guide/features/plugins#one-click-install-links-desktop). + ### The Python side Desktop plugins reuse the dashboard plugin backend mount. Put the backend in a diff --git a/website/docs/user-guide/features/plugins.md b/website/docs/user-guide/features/plugins.md index 7c1979482c..14c162f123 100644 --- a/website/docs/user-guide/features/plugins.md +++ b/website/docs/user-guide/features/plugins.md @@ -344,6 +344,41 @@ hermes plugins disable my-plugin # remove from allow-list + add to d hermes plugins capabilities [my-plugin] # declared vs granted capabilities ``` +### One-click install links (Desktop) + +Hermes Desktop registers the `hermes://` URL scheme, so a website, README, or +chat message can link straight to a plugin install: + +``` +hermes://plugin/install?repo=owner/repo # main install link +hermes://plugin/install?repo=owner/repo&enable=1 # enable the agent plugin after install +hermes://plugin/install?repo=owner/repo&force=1 # replace an existing install +``` + +Clicking one opens Hermes and shows a **confirmation dialog** — the repo id, +a "Before you install" note, and GitHub browse + clone links — then +shallow-clones the repo to detect what it ships (an **agent plugin** — +backend Python, a **desktop plugin** — app UI, or both). You pick the +components with checkboxes and confirm. Nothing is installed until you do; +deep links never auto-install, and agent-plugin installs go through the same +[install-time security scanning](#install-time-security-scanning) as +`hermes plugins install`. + +Hybrid repos (agent + desktop halves in one repo) use one link and one +dialog. The same modal is reachable without a link via **Settings → Plugins → +Install from Git**. Legacy `hermes://plugin-agent/…` and +`hermes://plugin-desktop/…` URLs route into the same dialog. In dev builds +(`npm run dev`) the scheme is `hermes-dev://`. + +Websites need no SDK — a normal anchor works: + +```html +Install in Hermes +``` + +MCP servers have the equivalent link form — see +[Add to Hermes link](/reference/mcp-config-reference#add-to-hermes-link). + ### Plugin capabilities and consent Plugins can declare the privileged host surfaces they want in their From 0cc26777bb8a31437c6207f50bffd15f0e5d6b56 Mon Sep 17 00:00:00 2001 From: andyst-dev <150129844+andyst-dev@users.noreply.github.com> Date: Sat, 15 Aug 2026 09:05:25 +0200 Subject: [PATCH 017/426] fix(gateway): persist resume recovery notes Fixes #86580 --- gateway/run.py | 21 ++++++++++++++++++-- tests/gateway/test_restart_resume_pending.py | 13 ++++++++++++ 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 8895660879..ad7824ffc3 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -1228,6 +1228,24 @@ def build_resume_recovery_note( ) +def _prepare_resume_pending_message( + reason: Optional[str], + message: Optional[str], + *, + interactive: bool = True, +) -> tuple[str, str]: + """Return the recovery message and the identical persisted user text. + + Resume turns replace the empty startup event with a recovery note before + entering the agent. Persist that note too, rather than the original empty + event, so the durable transcript matches what the model received. + """ + recovery_message = build_resume_recovery_note( + reason, message or "", interactive=interactive, + ) + return recovery_message, recovery_message + + # Assistant-message fields that must survive transcript replay so multi-turn # reasoning context, prefix-cache hits, and provider-specific echo # requirements all behave the same on the gateway as they do in the CLI. @@ -6121,7 +6139,6 @@ class TurnRunner: if _is_resume_pending: _reason = getattr(_resume_entry, "resume_reason", None) or "restart_timeout" - _persist_user_message_override = ctx.message # The empty-message case is the auto-resume startup turn # synthesized by _schedule_resume_pending_sessions — there is # no NEW user message to address. Guidance is adapter-aware: @@ -6134,7 +6151,7 @@ class TurnRunner: _interactive_resume = bool( getattr(_resume_adapter, "interactive_resume", True) ) - ctx.message = build_resume_recovery_note( + ctx.message, _persist_user_message_override = _prepare_resume_pending_message( _reason, ctx.message, interactive=_interactive_resume, ) elif _has_fresh_tool_tail: diff --git a/tests/gateway/test_restart_resume_pending.py b/tests/gateway/test_restart_resume_pending.py index 1acc357e16..2f602faa21 100644 --- a/tests/gateway/test_restart_resume_pending.py +++ b/tests/gateway/test_restart_resume_pending.py @@ -40,6 +40,7 @@ from gateway.run import ( _coerce_gateway_timestamp, _is_fresh_gateway_interruption, _last_transcript_timestamp, + _prepare_resume_pending_message, _should_clear_resume_pending_after_turn, build_resume_recovery_note, ) @@ -314,6 +315,18 @@ class TestResumePendingSystemNote: assert "already appear in the history" in note + def test_resume_note_is_persisted_instead_of_original_empty_message(self): + """The auto-resume note must not leave an empty row in state.db.""" + message, persisted = _prepare_resume_pending_message( + "restart_timeout", "", interactive=False + ) + + assert message + assert "CONTINUE the interrupted task" in message + assert persisted == message + assert persisted != "" + + def test_resume_pending_fires_without_tool_tail(self): """Key improvement over PR #9934: the restart-resume note fires even when the transcript's last role is NOT ``tool``.""" From c69b6471e677d7ff23b5cef0cbc2924e900b6453 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 16:37:25 -0700 Subject: [PATCH 018/426] fix(gateway): real user text stays clean in the transcript on resume-pending turns (#86580) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The salvaged fix persisted the recovery note unconditionally — correct for the synthesized empty auto-resume turn, but a user who typed real text while resume was pending would get the [System note: ...] scaffold persisted as their own words, leaking scaffolding into the durable transcript (the same class as #81841 on the assistant side). _prepare_resume_pending_message now persists the note only when the original message is blank; real text persists verbatim while the model still receives the wrapped note. Whitespace-only counts as blank. Tests cover all three shapes. --- gateway/run.py | 18 +++++++++++---- tests/gateway/test_restart_resume_pending.py | 24 ++++++++++++++++++++ 2 files changed, 37 insertions(+), 5 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index ad7824ffc3..07952e2085 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -1234,16 +1234,24 @@ def _prepare_resume_pending_message( *, interactive: bool = True, ) -> tuple[str, str]: - """Return the recovery message and the identical persisted user text. + """Return the recovery message and the user text to persist. - Resume turns replace the empty startup event with a recovery note before - entering the agent. Persist that note too, rather than the original empty - event, so the durable transcript matches what the model received. + Resume turns replace the startup event's text with a recovery note before + entering the agent. When the original message is empty (the synthesized + auto-resume turn), persist the note too — persisting the empty string + left a blank user row in state.db that the pre-call sanitizer re-healed + on every later call forever (#86580). When the user sent REAL text while + the resume was pending, keep persisting their clean words: the transcript + stays scaffold-free (the model still receives the wrapped note), and a + non-empty row never trips the sanitizer. """ recovery_message = build_resume_recovery_note( reason, message or "", interactive=interactive, ) - return recovery_message, recovery_message + persist_message = ( + message if isinstance(message, str) and message.strip() else recovery_message + ) + return recovery_message, persist_message # Assistant-message fields that must survive transcript replay so multi-turn diff --git a/tests/gateway/test_restart_resume_pending.py b/tests/gateway/test_restart_resume_pending.py index 2f602faa21..d52b8176d8 100644 --- a/tests/gateway/test_restart_resume_pending.py +++ b/tests/gateway/test_restart_resume_pending.py @@ -326,6 +326,30 @@ class TestResumePendingSystemNote: assert persisted == message assert persisted != "" + def test_whitespace_only_message_also_persists_the_note(self): + """A whitespace-only startup event is as blank as an empty one — + persisting it verbatim would recreate the sanitizer loop (#86580).""" + message, persisted = _prepare_resume_pending_message( + "shutdown_timeout", " ", interactive=True + ) + + assert persisted == message + assert persisted.strip() + + def test_real_user_text_persists_clean_not_the_scaffolded_note(self): + """When the user typed real text while resume was pending, the durable + transcript keeps their clean words; only the MODEL sees the wrapped + recovery note (transcript stays scaffold-free).""" + message, persisted = _prepare_resume_pending_message( + "restart_timeout", "what were we doing?", interactive=True + ) + + assert persisted == "what were we doing?" + assert "[System note:" not in persisted + assert message != persisted + assert "what were we doing?" in message + assert "[System note:" in message + def test_resume_pending_fires_without_tool_tail(self): """Key improvement over PR #9934: the restart-resume note fires From 34d1aed3f1d8a101571582b8c9776564d9772efc Mon Sep 17 00:00:00 2001 From: Calvin Ng <580008+calvinnwq@users.noreply.github.com> Date: Wed, 19 Aug 2026 09:30:59 +1000 Subject: [PATCH 019/426] fix(desktop): hide close affordance on navigation tabs Add a pane-level opt-out for the hover close button and apply it to the persistent Sessions and Bots navigation panes. Keep their existing close handlers and other tab behavior intact.\n\nFixes #89546 --- apps/desktop/src/app/contrib/controller.tsx | 1 + .../pane-shell/tree/renderer/track-model.ts | 2 ++ .../pane-shell/tree/renderer/tree-group.tsx | 1 + apps/desktop/src/components/ui/pane-tab.test.tsx | 15 +++++++++++++++ apps/desktop/src/components/ui/pane-tab.tsx | 5 ++++- apps/desktop/src/plugins/hermes-bots/plugin.js | 2 +- 6 files changed, 24 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index f14f957c33..3fd6009096 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -154,6 +154,7 @@ registry.registerMany([ collapsible: true, dock: { pane: 'workspace', pos: 'left' }, revealAliases: ['chat-sidebar'], + showCloseButton: false, width: `${SIDEBAR_DEFAULT_WIDTH}px`, minWidth: `${SIDEBAR_DEFAULT_WIDTH}px`, maxWidth: `${SIDEBAR_MAX_WIDTH}px` diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts index 57652df418..dd719f16a9 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts @@ -65,6 +65,8 @@ interface PaneChrome extends PaneSizing { /** No Close in the tab menu — the one surface the app can't lose (the * main workspace). Session tiles share `placement: 'main'` but close. */ uncloseable?: boolean + /** Hide the hover ✕ while retaining explicit close behavior for this pane. */ + showCloseButton?: boolean /** Wrap this pane's TAB (e.g. in a domain context menu — a session tile's * pin/branch/rename/archive/delete). The wrapper must render `tab` as its * interactive child; the zone's own strip menu still owns non-tab space. */ diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 638cad093b..9b26cb1392 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -539,6 +539,7 @@ export function TreeGroup({ }} role="tab" selected={isSelected} + showCloseButton={chrome.showCloseButton !== false} style={{ cursor: 'grab' }} > {chrome.tabLead ? ( diff --git a/apps/desktop/src/components/ui/pane-tab.test.tsx b/apps/desktop/src/components/ui/pane-tab.test.tsx index f76ed4142b..2ada280145 100644 --- a/apps/desktop/src/components/ui/pane-tab.test.tsx +++ b/apps/desktop/src/components/ui/pane-tab.test.tsx @@ -124,6 +124,21 @@ describe('PaneTab hover close button', () => { expect(screen.queryByRole('button', { name: 'Close' })).toBeNull() }) + it('can hide the hover ✕ while retaining the close handler', () => { + const onClose = vi.fn() + render( + + tab + + ) + + expect(screen.queryByRole('button', { name: 'Close' })).toBeNull() + const tab = screen.getByText('tab') + fireEvent.pointerDown(tab, { button: 1 }) + fireEvent.pointerUp(tab, { button: 1 }) + expect(onClose).toHaveBeenCalledTimes(1) + }) + it('renders no ✕ on a vertical rail tab (middle/⌘-click only there)', () => { const onClose = vi.fn() render( diff --git a/apps/desktop/src/components/ui/pane-tab.tsx b/apps/desktop/src/components/ui/pane-tab.tsx index f34c282dc0..c08e98d10b 100644 --- a/apps/desktop/src/components/ui/pane-tab.tsx +++ b/apps/desktop/src/components/ui/pane-tab.tsx @@ -58,6 +58,8 @@ interface PaneTabProps extends React.ComponentProps<'div'> { /** Part of a multi-tab selection (⌥/Ctrl-click, Shift-click) — an accent * wash marks every tab that a drag would carry, Chrome-style. */ selected?: boolean + /** Whether a closeable horizontal tab reveals the hover ✕. */ + showCloseButton?: boolean /** Vertical rail form (collapsed sidebar zones). */ vertical?: boolean /** Content-facing edge of a vertical rail — the strip line the active tab cuts. */ @@ -81,6 +83,7 @@ export const PaneTab = React.forwardRef(function P onPointerUp, onClickCapture, selected = false, + showCloseButton = true, vertical = false, side = 'left', children, @@ -159,7 +162,7 @@ export const PaneTab = React.forwardRef(function P )} - {onClose && !vertical && ( + {onClose && showCloseButton && !vertical && ( // Hover ✕, painted OVER the label's right edge as an overlay (no // layout shift, tab width never jumps on hover). The runway is a tiny // transparent→`--tab-face` gradient, so the button melts into the diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index d013949e26..7fdf85737b 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -9887,7 +9887,7 @@ export default { // sessions pane collapses alone without this flag. The zone then keeps // a stranded BOTS tab on screen. The narrow edge overlay mirrors the // zone's tab strip, so the pane stays reachable while collapsed. - data: { placement: 'left', width: '260px', collapsible: true, dock: { pane: 'sessions', pos: 'center', enforce: true } }, + data: { placement: 'left', width: '260px', collapsible: true, showCloseButton: false, dock: { pane: 'sessions', pos: 'center', enforce: true } }, render: () => jsx(BotsPane, {}) }) From 4ac938ddeccee2e9846d5508450639587f1e04ef Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 18 Aug 2026 17:03:47 -0700 Subject: [PATCH 020/426] =?UTF-8?q?feat(desktop):=20sessions/bots=20tabs?= =?UTF-8?q?=20are=20show/hide=20chrome=20=E2=80=94=20no=20close=20gestures?= =?UTF-8?q?,=20right-click=20+=20Cmd-K=20toggles?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Builds on #89551 (@calvinnwq, cherry-picked): his showCloseButton flag hid the hover X; this completes the model so standing chrome can never be closed at all, only shown/hidden (#89546). - hideOnly pane chrome (sessions + Bots): no hover X, no middle/meta click close, no Close verbs in the tab menu, excluded from close-others/right/all sweeps - zone right-click menu gains Show/Hide rows for the strip's chrome tabs (Hide bots / Show sessions, localized in 6 locales) - Cmd-K palette: auto-registered "Toggle tab" rows for every hideOnly pane, on-screen truth semantics, plugin panes included via registry subscription - hides persist across launches (survive the enforced dock re-adopt); reveal intent and Layout reset clear them - last-visible-tab guard: hiding the zone's last shown tab is refused with a toast, so the strip can never become an empty dead zone --- apps/desktop/src/app/contrib/controller.tsx | 62 +++++++++ .../tree/hide-only-strip-tabs.test.ts | 122 ++++++++++++++++++ .../pane-shell/tree/renderer/track-model.ts | 5 + .../pane-shell/tree/renderer/tree-group.tsx | 40 +++++- .../src/components/pane-shell/tree/store.ts | 120 ++++++++++++++++- apps/desktop/src/i18n/ar.ts | 5 + apps/desktop/src/i18n/en.ts | 5 + apps/desktop/src/i18n/ja.ts | 5 + apps/desktop/src/i18n/types.ts | 5 + apps/desktop/src/i18n/zh-hant.ts | 5 + apps/desktop/src/i18n/zh.ts | 5 + .../desktop/src/plugins/hermes-bots/plugin.js | 2 +- 12 files changed, 372 insertions(+), 9 deletions(-) create mode 100644 apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index 3fd6009096..36cb6462ed 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -29,6 +29,7 @@ import { removeTreePane, resetLayoutTree, revealTreePane, + setStripTabHidden, togglePaneVisible, watchContributedPanes } from '@/components/pane-shell/tree/store' @@ -38,6 +39,7 @@ import { Slot } from '@/contrib/react/slot' import { useContributions } from '@/contrib/react/use-contributions' import { registry } from '@/contrib/registry' import { discoverRuntimePlugins } from '@/contrib/runtime-loader' +import { translateNow } from '@/i18n' import { NEW_SESSION_TITLE, sessionTitle as storedSessionTitle } from '@/lib/chat-runtime' import { Download, FileText, LayoutDashboard, PanelBottom, Terminal, Upload, Zap } from '@/lib/icons' import { type KeybindContribution, KEYBINDS_AREA } from '@/lib/keybinds/actions' @@ -155,6 +157,9 @@ registry.registerMany([ dock: { pane: 'workspace', pos: 'left' }, revealAliases: ['chat-sidebar'], showCloseButton: false, + // Standing chrome: no close gestures at all — the tab is shown/hidden + // (zone menu Show/Hide rows + the auto-registered ⌘K toggle below). + hideOnly: true, width: `${SIDEBAR_DEFAULT_WIDTH}px`, minWidth: `${SIDEBAR_DEFAULT_WIDTH}px`, maxWidth: `${SIDEBAR_MAX_WIDTH}px` @@ -673,6 +678,63 @@ registry.register( }) ) +// Hide-only chrome tabs (sessions / Bots) get a ⌘K toggle each — the palette +// door onto the same show/hide the zone menu offers. Auto-registered from the +// panes area so a plugin's hideOnly pane (Bots registers at plugin load, after +// this module runs) gets its row for free; disposers keep it in step when a +// plugin unloads. Registry writes during a subscriber callback are safe (the +// registry snapshots per-area and re-notifies), and re-registering the same +// palette id replaces the row instead of stacking duplicates. +{ + const stripTabToggles = new Map void>() + + const syncStripTabToggles = () => { + const hideOnlyPanes = registry + .getArea('panes') + .filter(c => (c.data as { hideOnly?: boolean } | undefined)?.hideOnly) + const wanted = new Set(hideOnlyPanes.map(c => c.id)) + + for (const [paneId, dispose] of stripTabToggles) { + if (!wanted.has(paneId)) { + dispose() + stripTabToggles.delete(paneId) + } + } + + for (const pane of hideOnlyPanes) { + if (stripTabToggles.has(pane.id)) { + continue + } + + const title = String(pane.title ?? pane.id) + + stripTabToggles.set( + pane.id, + registry.register( + paletteToggle({ + id: `strip-tab.${pane.id}`, + label: translateNow('zones.toggleStripTab', title), + icon: LayoutDashboard, + keywords: [title.toLowerCase(), 'tab', 'pane', 'sidebar', 'show', 'hide'], + // On-screen truth, same contract as the logs toggle above. + get: () => isPaneVisible(pane.id), + set: visible => { + if (visible) { + revealTreePane(pane.id) + } else { + setStripTabHidden(pane.id, true) + } + } + }) + ) + ) + } + } + + syncStripTabToggles() + registry.subscribeArea('panes', syncStripTabToggles) +} + // YOLO (dangerous-command approval bypass) is a status-bar zap and a /yolo // command; ⌘K is the third door onto the SAME store function, so a user who // lives in the palette never has to hunt for the pill. diff --git a/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts b/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts new file mode 100644 index 0000000000..048f67f95b --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts @@ -0,0 +1,122 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { registry } from '@/contrib/registry' + +import { allPaneIds, group, split } from './model' +import { + $hiddenStripTabs, + $hiddenTreePanes, + $layoutTree, + closeAllTreeTabs, + hideOnlyZoneTabs, + isHideOnlyPane, + revealTreePane, + setStripTabHidden, + treeTabCloseTargets +} from './store' + +vi.mock('@/store/notifications', () => ({ notify: vi.fn() })) + +import { notify } from '@/store/notifications' + +const disposers: (() => void)[] = [] + +function registerPane(id: string, data: Record) { + disposers.push(registry.register({ area: 'panes', data, id, render: () => null, title: id })) +} + +/** The SESSIONS | BOTS shape: both hide-only chrome tabs stacked in one zone. */ +function sessionsBotsTree() { + registerPane('sessions', { placement: 'left', hideOnly: true }) + registerPane('hermes-bots:pane', { placement: 'left', hideOnly: true }) + registerPane('workspace', { placement: 'main', uncloseable: true }) + $layoutTree.set( + split('row', [ + group(['sessions', 'hermes-bots:pane'], { active: 'sessions', id: 'g-side' }), + group(['workspace'], { active: 'workspace', id: 'g-main' }) + ]) + ) +} + +beforeEach(() => { + window.localStorage.clear() + $hiddenStripTabs.set(new Set()) + $hiddenTreePanes.set(new Set()) + vi.mocked(notify).mockReset() +}) + +afterEach(() => { + disposers.splice(0).forEach(dispose => dispose()) +}) + +describe('hide-only strip tabs', () => { + it('hides and shows a chrome tab, keeping the pane in the tree', () => { + sessionsBotsTree() + + expect(setStripTabHidden('hermes-bots:pane', true)).toBe(true) + expect($hiddenTreePanes.get()).toContain('hermes-bots:pane') + expect($hiddenStripTabs.get()).toContain('hermes-bots:pane') + // Hidden, not dismissed: the pane stays in the layout tree. + expect(allPaneIds($layoutTree.get()!)).toContain('hermes-bots:pane') + + expect(setStripTabHidden('hermes-bots:pane', false)).toBe(true) + expect($hiddenTreePanes.get()).not.toContain('hermes-bots:pane') + expect($hiddenStripTabs.get()).not.toContain('hermes-bots:pane') + }) + + it('refuses to hide the zone last visible tab', () => { + sessionsBotsTree() + setStripTabHidden('hermes-bots:pane', true) + + // Sessions is now the only visible tab in the zone — the hide is refused + // and the user is told, so the zone can never become an empty dead strip. + expect(setStripTabHidden('sessions', true)).toBe(false) + expect($hiddenTreePanes.get()).not.toContain('sessions') + expect(vi.mocked(notify)).toHaveBeenCalledTimes(1) + }) + + it('persists hides and clears them on reveal', () => { + sessionsBotsTree() + setStripTabHidden('hermes-bots:pane', true) + + const persisted = JSON.parse(window.localStorage.getItem('hermes.desktop.hiddenStripTabs.v1') ?? '[]') + + expect(persisted).toContain('hermes-bots:pane') + + // Reveal intent (⌘K toggle on, a programmatic reveal) beats the hide — + // including the persisted record, so the tab can't pop back hidden on the + // next launch while visibly on screen now. + revealTreePane('hermes-bots:pane') + expect($hiddenTreePanes.get()).not.toContain('hermes-bots:pane') + expect(window.localStorage.getItem('hermes.desktop.hiddenStripTabs.v1')).toBeNull() + }) + + it('lists the zone hide-only tabs with live hidden state for the menu', () => { + sessionsBotsTree() + setStripTabHidden('hermes-bots:pane', true) + + expect(hideOnlyZoneTabs('g-side')).toEqual([ + { hidden: false, id: 'sessions', title: 'sessions' }, + { hidden: true, id: 'hermes-bots:pane', title: 'hermes-bots:pane' } + ]) + expect(hideOnlyZoneTabs('g-main')).toEqual([]) + }) + + it('excludes hide-only tabs from every close verb', () => { + registerPane('sessions', { placement: 'left', hideOnly: true }) + registerPane('hermes-bots:pane', { placement: 'left', hideOnly: true }) + registerPane('session-tile:x', { placement: 'main' }) + $layoutTree.set( + group(['sessions', 'hermes-bots:pane', 'session-tile:x'], { active: 'sessions', id: 'g-mixed' }) + ) + + expect(isHideOnlyPane('sessions')).toBe(true) + // Close-others measured from the tile must not sweep standing chrome. + expect(treeTabCloseTargets('session-tile:x')).toEqual({ all: 1, others: 0, right: 0 }) + // Close-all leaves both chrome tabs in the tree. + closeAllTreeTabs('sessions') + expect(allPaneIds($layoutTree.get()!)).toContain('sessions') + expect(allPaneIds($layoutTree.get()!)).toContain('hermes-bots:pane') + expect(allPaneIds($layoutTree.get()!)).not.toContain('session-tile:x') + }) +}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts index dd719f16a9..785ea7519d 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts @@ -67,6 +67,11 @@ interface PaneChrome extends PaneSizing { uncloseable?: boolean /** Hide the hover ✕ while retaining explicit close behavior for this pane. */ showCloseButton?: boolean + /** Standing chrome tab (sessions / Bots) whose tab shows NO ✕ and no Close + * verbs — it is shown/hidden instead (the zone menu's Show/Hide rows and a + * ⌘K toggle, via `setStripTabHidden`). Close was too destructive for these: + * an accidental ✕ removed Bot Mode until the next launch. */ + hideOnly?: boolean /** Wrap this pane's TAB (e.g. in a domain context menu — a session tile's * pin/branch/rename/archive/delete). The wrapper must render `tab` as its * interactive child; the zone's own strip menu still owns non-tab space. */ diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 9b26cb1392..28967fd802 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -49,6 +49,7 @@ import { closeTabPane, closeTreeTabsToRight, collapseTreePane, + hideOnlyZoneTabs, isCollapsePane, isMainStripPane, isSessionStripPane, @@ -56,6 +57,7 @@ import { reloadTreePane, restoreTreePane, SESSION_TILE_DRAG, + setStripTabHidden, setTreeGroupHeaderHidden, setTreeGroupMinimized, treeTabCloseTargets @@ -129,6 +131,30 @@ function ZoneMenu({ onCloseOthers: () => closeOtherTreeTabs(targetId), onCloseToRight: () => closeTreeTabsToRight(targetId) })} + {(() => { + // Show/hide rows for the zone's hide-only chrome tabs (sessions / + // Bots) — their Close replacement. Resolved when the menu OPENS, + // same no-subscription contract as the close-verb counts above. + const hideOnly = hideOnlyZoneTabs(nodeId) + + if (hideOnly.length === 0) { + return null + } + + return ( + <> + + {hideOnly.map(tab => + renderActionItem(kit, { + icon: tab.hidden ? 'eye' : 'eye-closed', + key: `strip-tab-${tab.id}`, + label: tab.hidden ? t.zones.showStripTab(tab.title) : t.zones.hideStripTab(tab.title), + onSelect: () => setStripTabHidden(tab.id, !tab.hidden) + }) + )} + + ) + })()} {renderActionItem(kit, { icon: headerHidden ? 'eye' : 'eye-closed', @@ -287,11 +313,13 @@ export function TreeGroup({ const targetPane = () => menuPane ?? activeId // Close targets the right-clicked chip (falling back to the active pane); - // only panes that declare `uncloseable` (the main workspace) are exempt. + // panes that declare `uncloseable` (the main workspace) or `hideOnly` + // (sessions / Bots — show/hide replaces Close) are exempt. const closable = () => { const paneId = targetPane() + const chrome = paneChrome(paneFor(paneId)) - return paneChrome(paneFor(paneId)).uncloseable ? undefined : paneId + return chrome.uncloseable || chrome.hideOnly ? undefined : paneId } // The zone hosting the uncloseable workspace never minimizes — collapsing @@ -304,8 +332,12 @@ export function TreeGroup({ // A pane whose store owns Close keeps the gesture even when the pane itself // is uncloseable — the workspace tab empties to a fresh draft rather than - // leaving the tree. - const closeableTab = (paneId: string) => !paneChrome(paneFor(paneId)).uncloseable || panesWithCloser.has(paneId) + // leaving the tree. Hide-only chrome (sessions / Bots) opts out of every + // close gesture: its tabs are shown/hidden (zone menu, ⌘K), never closed — + // an accidental ✕ on standing chrome removed Bot Mode until the next launch. + const closeableTab = (paneId: string) => + !paneChrome(paneFor(paneId)).hideOnly && + (!paneChrome(paneFor(paneId)).uncloseable || panesWithCloser.has(paneId)) // A pane's own live label when it has one, else its registered string. const tabLabel = (paneId: string) => paneChrome(paneFor(paneId)).tabTitle?.() ?? paneFor(paneId)?.title ?? paneId diff --git a/apps/desktop/src/components/pane-shell/tree/store.ts b/apps/desktop/src/components/pane-shell/tree/store.ts index 622323d10f..f1e98f9cdb 100644 --- a/apps/desktop/src/components/pane-shell/tree/store.ts +++ b/apps/desktop/src/components/pane-shell/tree/store.ts @@ -248,6 +248,72 @@ function recalledEdgeWeights(paneId: string): [number, number] | undefined { return validShare(share) ? [1 - share, share] : undefined } +// HIDE-ONLY STRIP TABS (`hideOnly` chrome: sessions / Bots) — standing chrome +// whose tab must never grow a ✕. Show/hide replaces Close for them: the zone +// menu's Show/Hide rows and the auto-registered ⌘K toggles both land here. +// Persisted separately from `$hiddenTreePanes` (whose persistence each side +// binding owns) so a hidden Bots tab stays hidden across launches even though +// dock enforcement re-adopts the pane into the sessions zone every boot. +const HIDDEN_STRIP_TAB_KEY = 'hermes.desktop.hiddenStripTabs.v1' + +export const $hiddenStripTabs = atom>(new Set(readJson(HIDDEN_STRIP_TAB_KEY) ?? [])) + +function saveHiddenStripTabs(next: ReadonlySet) { + $hiddenStripTabs.set(next) + writeJson(HIDDEN_STRIP_TAB_KEY, next.size === 0 ? null : [...next]) +} + +export function isStripTabHidden(paneId: string): boolean { + return $hiddenStripTabs.get().has(paneId) +} + +/** Would hiding `paneId` leave its zone with no visible tab? Hiding the last + * one strands an empty zone (or collapses the whole sidebar with no strip + * left to right-click), so the setter refuses and says why. */ +function isLastShownInGroup(paneId: string): boolean { + const tree = $layoutTree.get() + const group = tree ? findGroupOfPane(tree, paneId) : null + + if (!group) { + return false + } + + const hidden = $hiddenTreePanes.get() + + return !group.panes.some(id => id !== paneId && !hidden.has(id)) +} + +/** Show/hide a hide-only chrome tab (the Close replacement for `hideOnly` + * panes). Returns false when the hide was refused — the zone must keep at + * least one visible tab, so the LAST shown tab can't be hidden. */ +export function setStripTabHidden(paneId: string, hidden: boolean): boolean { + if (hidden && isLastShownInGroup(paneId)) { + notify({ + kind: 'info', + title: translateNow('zones.lastTabKeptTitle'), + message: translateNow('zones.lastTabKeptBody') + }) + + return false + } + + const next = toggledSet($hiddenStripTabs.get(), paneId, hidden) + + if (next) { + saveHiddenStripTabs(next) + } + + setTreePaneHidden(paneId, hidden) + + return true +} + +// Boot hydration: re-apply persisted hides through the same chrome-hidden set +// the strips render from ($hiddenTreePanes starts empty every launch). +for (const paneId of $hiddenStripTabs.get()) { + setTreePaneHidden(paneId, true) +} + const paneClosers: Record void> = {} const paneOpeners: Record void> = {} @@ -409,6 +475,12 @@ const isUncloseablePane = (paneId: string): boolean => (registry.getArea('panes').find(c => c.id === paneId)?.data as { uncloseable?: boolean } | undefined)?.uncloseable ) +/** Hide-only chrome tabs (sessions / Bots): excluded from every close verb — + * Close-others / Close-all sweeping the sessions strip must not take standing + * chrome with it. They hide through `setStripTabHidden` instead. */ +export const isHideOnlyPane = (paneId: string): boolean => + Boolean((registry.getArea('panes').find(c => c.id === paneId)?.data as { hideOnly?: boolean } | undefined)?.hideOnly) + /** A pane that belongs to a CHAT tab strip — the workspace or a session tile. * Chat surfaces only: this gates where a session may DOCK (drops, ⌘T's "+"), * not which zones the generic tab verbs serve — that's `isMainStripPane`. */ @@ -506,8 +578,8 @@ function closeableTreeSiblings(paneId: string): { others: string[]; right: strin const idx = panes.indexOf(paneId) return { - others: panes.filter(id => id !== paneId && !isUncloseablePane(id)), - right: panes.filter((id, i) => i > idx && !isUncloseablePane(id)) + others: panes.filter(id => id !== paneId && !isUncloseablePane(id) && !isHideOnlyPane(id)), + right: panes.filter((id, i) => i > idx && !isUncloseablePane(id) && !isHideOnlyPane(id)) } } @@ -515,7 +587,11 @@ function closeableTreeSiblings(paneId: string): { others: string[]; right: strin export function treeTabCloseTargets(paneId: string): { all: number; others: number; right: number } { const { others, right } = closeableTreeSiblings(paneId) - return { all: others.length + (isUncloseablePane(paneId) ? 0 : 1), others: others.length, right: right.length } + return { + all: others.length + (isUncloseablePane(paneId) || isHideOnlyPane(paneId) ? 0 : 1), + others: others.length, + right: right.length + } } /** @@ -557,7 +633,32 @@ export function closeAllTreeTabs(paneId: string): void { const tree = $layoutTree.get() const panes = (tree ? findGroupOfPane(tree, paneId) : null)?.panes ?? [] - panes.filter(id => !isUncloseablePane(id)).forEach(closeTabPane) + panes.filter(id => !isUncloseablePane(id) && !isHideOnlyPane(id)).forEach(closeTabPane) +} + +/** Hide-only chrome tabs in `groupId` (sessions / Bots), with live hidden + * state — the zone menu's Show/Hide rows. Resolved when the menu OPENS (same + * contract as the close-verb counts), never subscribed from a zone render. */ +export function hideOnlyZoneTabs(groupId: string): { hidden: boolean; id: string; title: string }[] { + const tree = $layoutTree.get() + const group = tree ? findGroup(tree, groupId) : null + + if (!group) { + return [] + } + + const panes = registry.getArea('panes') + const hidden = $hiddenTreePanes.get() + + return group.panes.flatMap(id => { + const pane = panes.find(p => p.id === id) + + if (!(pane?.data as { hideOnly?: boolean } | undefined)?.hideOnly) { + return [] + } + + return [{ hidden: hidden.has(id), id, title: String(pane?.title ?? id) }] + }) } /** Pane ids in the tree under a `${prefix}:` namespace — lets a mirror prune @@ -923,6 +1024,12 @@ export function revealTreePane(paneId: string) { adoptContributedPanes() } + // Reveal beats a hide too: clear the persisted hide-only record, or the + // pane pops back hidden on the next launch even though it's on screen now. + if ($hiddenStripTabs.get().has(paneId)) { + saveHiddenStripTabs(toggledSet($hiddenStripTabs.get(), paneId, false) ?? $hiddenStripTabs.get()) + } + const side = treeSideOfPane(paneId) if (side && $collapsedTreeSides.get().has(side)) { @@ -1796,6 +1903,11 @@ export function resetLayoutTree() { // placement back to the app (user-placed pins cleared). saveDismissed(new Set()) saveUserPlaced(new Set()) + // Hide-only chrome tabs (sessions / Bots) come back too — clear their + // persisted hides through the setter so $hiddenTreePanes agrees. + for (const paneId of [...$hiddenStripTabs.get()]) { + setStripTabHidden(paneId, false) + } $layoutTree.set(defaultTree) markActivePreset('default') // Owners PRE-PLACE their panes into the fresh default (session tiles stack diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 06d7370adf..44413962db 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2327,6 +2327,11 @@ export const ar = defineLocale({ zones: { showHeader: 'إظهار الرأس', hideHeader: 'إخفاء الرأس', + showStripTab: title => `إظهار ${title}`, + hideStripTab: title => `إخفاء ${title}`, + lastTabKeptTitle: 'يبقى آخر تبويب', + lastTabKeptBody: 'تحتاج هذه المنطقة إلى تبويب مرئي واحد على الأقل. أظهر تبويبا آخر أولا، أو اطو الشريط الجانبي بأكمله.', + toggleStripTab: title => `تبديل تبويب ${title}`, minimize: 'تصغير', restore: 'استعادة', closeRunningTitle: 'إغلاق تبويب يعمل؟', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 8fd119411a..25f122255c 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -2958,6 +2958,11 @@ export const en: Translations = { zones: { showHeader: 'Show header', hideHeader: 'Hide header', + showStripTab: title => `Show ${title}`, + hideStripTab: title => `Hide ${title}`, + lastTabKeptTitle: 'Last tab stays', + lastTabKeptBody: 'This zone needs at least one visible tab. Show another tab first, or collapse the whole sidebar.', + toggleStripTab: title => `Toggle ${title} tab`, minimize: 'Minimize', restore: 'Restore', closeRunningTitle: 'Close running tab?', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 0d55b75067..eb1456722e 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2618,6 +2618,11 @@ export const ja = defineLocale({ zones: { showHeader: 'ヘッダーを表示', hideHeader: 'ヘッダーを隠す', + showStripTab: title => `${title} を表示`, + hideStripTab: title => `${title} を隠す`, + lastTabKeptTitle: '最後のタブは残ります', + lastTabKeptBody: 'このゾーンには少なくとも 1 つの表示タブが必要です。先に別のタブを表示するか、サイドバー全体を折りたたんでください。', + toggleStripTab: title => `${title} タブを切り替え`, minimize: '最小化', restore: '復元', reload: '再読み込み', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 67704864b4..e162fa0d9c 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2530,6 +2530,11 @@ export interface Translations { zones: { showHeader: string hideHeader: string + showStripTab: (title: string) => string + hideStripTab: (title: string) => string + lastTabKeptTitle: string + lastTabKeptBody: string + toggleStripTab: (title: string) => string minimize: string restore: string closeRunningTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 490fc29841..3cfe280c28 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2532,6 +2532,11 @@ export const zhHant = defineLocale({ zones: { showHeader: '顯示標題列', hideHeader: '隱藏標題列', + showStripTab: title => `顯示 ${title}`, + hideStripTab: title => `隱藏 ${title}`, + lastTabKeptTitle: '保留最後一個分頁', + lastTabKeptBody: '此區域至少需要一個可見分頁。請先顯示另一個分頁,或收合整個側邊欄。', + toggleStripTab: title => `切換 ${title} 分頁`, minimize: '最小化', restore: '還原', reload: '重新載入', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index dc5e7907fc..e803a2cde6 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3124,6 +3124,11 @@ export const zh: Translations = { zones: { showHeader: '显示标题栏', hideHeader: '隐藏标题栏', + showStripTab: title => `显示 ${title}`, + hideStripTab: title => `隐藏 ${title}`, + lastTabKeptTitle: '保留最后一个标签', + lastTabKeptBody: '该区域至少需要一个可见标签。请先显示另一个标签,或折叠整个侧边栏。', + toggleStripTab: title => `切换 ${title} 标签`, minimize: '最小化', restore: '还原', closeRunningTitle: '关闭正在运行的标签?', diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 7fdf85737b..1f5257f18c 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -9887,7 +9887,7 @@ export default { // sessions pane collapses alone without this flag. The zone then keeps // a stranded BOTS tab on screen. The narrow edge overlay mirrors the // zone's tab strip, so the pane stays reachable while collapsed. - data: { placement: 'left', width: '260px', collapsible: true, showCloseButton: false, dock: { pane: 'sessions', pos: 'center', enforce: true } }, + data: { placement: 'left', width: '260px', collapsible: true, showCloseButton: false, hideOnly: true, dock: { pane: 'sessions', pos: 'center', enforce: true } }, render: () => jsx(BotsPane, {}) }) From 2163f7f8ca82f7c892c7e815dadd80a1486cc194 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Wed, 19 Aug 2026 00:20:52 +0000 Subject: [PATCH 021/426] fmt(js): `npm run fix` on merge (#89580) Co-authored-by: github-actions[bot] --- apps/desktop/src/app/contrib/controller.tsx | 1 + .../components/pane-shell/tree/hide-only-strip-tabs.test.ts | 4 +--- .../src/components/pane-shell/tree/renderer/tree-group.tsx | 3 +-- apps/desktop/src/components/pane-shell/tree/store.ts | 2 ++ apps/desktop/src/i18n/ar.ts | 3 ++- apps/desktop/src/i18n/ja.ts | 3 ++- 6 files changed, 9 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index 36cb6462ed..c48ef1e33a 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -692,6 +692,7 @@ registry.register( const hideOnlyPanes = registry .getArea('panes') .filter(c => (c.data as { hideOnly?: boolean } | undefined)?.hideOnly) + const wanted = new Set(hideOnlyPanes.map(c => c.id)) for (const [paneId, dispose] of stripTabToggles) { diff --git a/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts b/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts index 048f67f95b..309985706d 100644 --- a/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts +++ b/apps/desktop/src/components/pane-shell/tree/hide-only-strip-tabs.test.ts @@ -106,9 +106,7 @@ describe('hide-only strip tabs', () => { registerPane('sessions', { placement: 'left', hideOnly: true }) registerPane('hermes-bots:pane', { placement: 'left', hideOnly: true }) registerPane('session-tile:x', { placement: 'main' }) - $layoutTree.set( - group(['sessions', 'hermes-bots:pane', 'session-tile:x'], { active: 'sessions', id: 'g-mixed' }) - ) + $layoutTree.set(group(['sessions', 'hermes-bots:pane', 'session-tile:x'], { active: 'sessions', id: 'g-mixed' })) expect(isHideOnlyPane('sessions')).toBe(true) // Close-others measured from the tile must not sweep standing chrome. diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 28967fd802..481dfd02e5 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -336,8 +336,7 @@ export function TreeGroup({ // close gesture: its tabs are shown/hidden (zone menu, ⌘K), never closed — // an accidental ✕ on standing chrome removed Bot Mode until the next launch. const closeableTab = (paneId: string) => - !paneChrome(paneFor(paneId)).hideOnly && - (!paneChrome(paneFor(paneId)).uncloseable || panesWithCloser.has(paneId)) + !paneChrome(paneFor(paneId)).hideOnly && (!paneChrome(paneFor(paneId)).uncloseable || panesWithCloser.has(paneId)) // A pane's own live label when it has one, else its registered string. const tabLabel = (paneId: string) => paneChrome(paneFor(paneId)).tabTitle?.() ?? paneFor(paneId)?.title ?? paneId diff --git a/apps/desktop/src/components/pane-shell/tree/store.ts b/apps/desktop/src/components/pane-shell/tree/store.ts index f1e98f9cdb..f1d4f35e97 100644 --- a/apps/desktop/src/components/pane-shell/tree/store.ts +++ b/apps/desktop/src/components/pane-shell/tree/store.ts @@ -1903,11 +1903,13 @@ export function resetLayoutTree() { // placement back to the app (user-placed pins cleared). saveDismissed(new Set()) saveUserPlaced(new Set()) + // Hide-only chrome tabs (sessions / Bots) come back too — clear their // persisted hides through the setter so $hiddenTreePanes agrees. for (const paneId of [...$hiddenStripTabs.get()]) { setStripTabHidden(paneId, false) } + $layoutTree.set(defaultTree) markActivePreset('default') // Owners PRE-PLACE their panes into the fresh default (session tiles stack diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 44413962db..075865128a 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2330,7 +2330,8 @@ export const ar = defineLocale({ showStripTab: title => `إظهار ${title}`, hideStripTab: title => `إخفاء ${title}`, lastTabKeptTitle: 'يبقى آخر تبويب', - lastTabKeptBody: 'تحتاج هذه المنطقة إلى تبويب مرئي واحد على الأقل. أظهر تبويبا آخر أولا، أو اطو الشريط الجانبي بأكمله.', + lastTabKeptBody: + 'تحتاج هذه المنطقة إلى تبويب مرئي واحد على الأقل. أظهر تبويبا آخر أولا، أو اطو الشريط الجانبي بأكمله.', toggleStripTab: title => `تبديل تبويب ${title}`, minimize: 'تصغير', restore: 'استعادة', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index eb1456722e..e54902a919 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2621,7 +2621,8 @@ export const ja = defineLocale({ showStripTab: title => `${title} を表示`, hideStripTab: title => `${title} を隠す`, lastTabKeptTitle: '最後のタブは残ります', - lastTabKeptBody: 'このゾーンには少なくとも 1 つの表示タブが必要です。先に別のタブを表示するか、サイドバー全体を折りたたんでください。', + lastTabKeptBody: + 'このゾーンには少なくとも 1 つの表示タブが必要です。先に別のタブを表示するか、サイドバー全体を折りたたんでください。', toggleStripTab: title => `${title} タブを切り替え`, minimize: '最小化', restore: '復元', From f709bd844561474b3b0826367e07637c9e297e3b Mon Sep 17 00:00:00 2001 From: Soju06 Date: Wed, 12 Aug 2026 05:30:57 +0000 Subject: [PATCH 022/426] fix(aux): translate response_format to output_config.format for anthropic transport Plugin structured completions (plugin_llm.complete_structured) build an OpenAI Chat Completions response_format payload in extra_body. The anthropic_messages transport forwarded it verbatim, and strict Anthropic-compatible gateways reject it with HTTP 400: response_format: OpenAI Chat Completions structured-output shape is not supported. Use output_config.format = {"type": "json_schema", ...} Observed live: every discord-thread-autotitle structured call failed for 2+ days (1,600+ logged errors) once the main provider became an anthropic_messages gateway. Fix: _translate_anthropic_response_format converts - json_schema -> output_config.format = {type: json_schema, schema: S} - json_object -> permissive object schema (SDK 0.87.0 has no schema-less JSON mode) merging into any existing output_config (adaptive-thinking effort coexists) and excluding response_format from the raw extra_body passthrough alongside the existing reasoning exclusion. The async adapter delegates to the sync adapter via asyncio.to_thread and is covered by a test. Non-Anthropic transports are unchanged. --- agent/auxiliary_client.py | 44 +++++- .../agent/test_anthropic_structured_output.py | 147 ++++++++++++++++++ 2 files changed, 189 insertions(+), 2 deletions(-) create mode 100644 tests/agent/test_anthropic_structured_output.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index d1a513c7ec..587b9442d3 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1956,6 +1956,39 @@ class AsyncCodexAuxiliaryClient: self._real_client = sync_wrapper._real_client +def _translate_anthropic_response_format( + anthropic_kwargs: Dict[str, Any], response_format: Any, +) -> None: + """Merge an OpenAI response format into Anthropic ``output_config``.""" + if not isinstance(response_format, dict): + return + + format_type = response_format.get("type") + if format_type == "json_schema": + json_schema = response_format.get("json_schema") + if not isinstance(json_schema, dict) or "schema" not in json_schema: + return + native_format = { + "type": "json_schema", + "schema": json_schema["schema"], + } + elif format_type == "json_object": + # Anthropic SDK 0.87.0 exposes only JSONOutputFormatParam, whose + # required type is ``json_schema``; it has no schema-less JSON mode. + native_format = { + "type": "json_schema", + "schema": {"type": "object"}, + } + else: + return + + output_config = anthropic_kwargs.get("output_config") + if not isinstance(output_config, dict): + output_config = {} + anthropic_kwargs["output_config"] = output_config + output_config["format"] = native_format + + class _AnthropicCompletionsAdapter: """OpenAI-client-compatible adapter for Anthropic Messages API.""" @@ -2056,18 +2089,25 @@ class _AnthropicCompletionsAdapter: # form is the documented Anthropic SDK passthrough for non-standard # request body keys; merge on top of whatever build_anthropic_kwargs # already produced (e.g. fast-mode ``speed``) so call-time settings - # survive. Two exclusions: + # survive. Three exclusions: # - ``reasoning``: the OpenAI-shaped config dict is TRANSLATED into # the native ``thinking`` field above (build_anthropic_kwargs); # forwarding the raw field alongside would double-specify # reasoning and 400 on strict gateways. + # - ``response_format``: the OpenAI structured-output shape is + # TRANSLATED into top-level ``output_config.format`` below; + # forwarding the raw field 400s on strict Anthropic gateways. # - ``_``-prefixed keys: private Hermes plumbing (_reasoning_config # et al.), never wire fields. caller_extra_body = kwargs.get("extra_body") if caller_extra_body and isinstance(caller_extra_body, dict): + _translate_anthropic_response_format( + anthropic_kwargs, caller_extra_body.get("response_format"), + ) passthrough = { k: v for k, v in caller_extra_body.items() - if k != "reasoning" and not str(k).startswith("_") + if k not in {"reasoning", "response_format"} + and not str(k).startswith("_") } if passthrough: existing = anthropic_kwargs.get("extra_body") or {} diff --git a/tests/agent/test_anthropic_structured_output.py b/tests/agent/test_anthropic_structured_output.py new file mode 100644 index 0000000000..eb504b6a55 --- /dev/null +++ b/tests/agent/test_anthropic_structured_output.py @@ -0,0 +1,147 @@ +"""Structured-output translation for Anthropic auxiliary calls.""" + +from __future__ import annotations + +import asyncio +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + + +def _capture_anthropic_kwargs( + extra_body: dict, *, model: str = "claude-sonnet-4-6", async_call: bool = False, +) -> dict: + from agent.auxiliary_client import ( + _AnthropicCompletionsAdapter, + _AsyncAnthropicCompletionsAdapter, + ) + + captured = {} + sync_adapter = _AnthropicCompletionsAdapter( + MagicMock(name="anthropic_client"), model, is_oauth=False, + ) + adapter = ( + _AsyncAnthropicCompletionsAdapter(sync_adapter) + if async_call + else sync_adapter + ) + + def _fake_create(_client, api_kwargs, **_kwargs): + captured.update(api_kwargs) + return SimpleNamespace() + + normalized = SimpleNamespace( + content="ok", + tool_calls=None, + reasoning=None, + finish_reason="stop", + ) + with patch( + "agent.anthropic_adapter.create_anthropic_message", + side_effect=_fake_create, + ), patch("agent.transports.get_transport") as mock_get_transport: + mock_get_transport.return_value.normalize_response.return_value = normalized + call = adapter.create( + model=model, + messages=[{"role": "user", "content": "hi"}], + max_tokens=64, + extra_body=extra_body, + ) + if async_call: + asyncio.run(call) + + return captured + + +def _assert_no_raw_response_format(api_kwargs: dict) -> None: + assert "response_format" not in api_kwargs + assert "response_format" not in api_kwargs.get("extra_body", {}) + + +def test_json_schema_response_format_uses_native_output_config(): + schema = { + "type": "object", + "properties": {"title": {"type": "string"}}, + "required": ["title"], + } + api_kwargs = _capture_anthropic_kwargs({ + "response_format": { + "type": "json_schema", + "json_schema": { + "name": "thread_title", + "schema": schema, + "strict": False, + }, + }, + }) + + assert api_kwargs["output_config"]["format"] == { + "type": "json_schema", + "schema": schema, + } + _assert_no_raw_response_format(api_kwargs) + + +def test_json_object_response_format_uses_permissive_object_schema(): + api_kwargs = _capture_anthropic_kwargs({ + "response_format": {"type": "json_object"}, + }) + + assert api_kwargs["output_config"]["format"] == { + "type": "json_schema", + "schema": {"type": "object"}, + } + _assert_no_raw_response_format(api_kwargs) + + +def test_response_format_merges_with_adaptive_thinking_effort(): + schema = {"type": "object", "properties": {"ok": {"type": "boolean"}}} + api_kwargs = _capture_anthropic_kwargs({ + "reasoning": {"enabled": True, "effort": "high"}, + "response_format": { + "type": "json_schema", + "json_schema": {"schema": schema}, + }, + }) + + assert api_kwargs["output_config"] == { + "effort": "high", + "format": {"type": "json_schema", "schema": schema}, + } + assert api_kwargs["thinking"] == { + "type": "adaptive", + "display": "summarized", + } + _assert_no_raw_response_format(api_kwargs) + + +def test_unrelated_extra_body_keys_still_pass_through(): + api_kwargs = _capture_anthropic_kwargs({ + "response_format": {"type": "json_object"}, + "metadata": {"user_id": "thread-autotitle"}, + "vendor_option": True, + }) + + assert api_kwargs["extra_body"] == { + "metadata": {"user_id": "thread-autotitle"}, + "vendor_option": True, + } + _assert_no_raw_response_format(api_kwargs) + + +def test_async_anthropic_adapter_uses_the_same_translation(): + schema = {"type": "object"} + api_kwargs = _capture_anthropic_kwargs( + { + "response_format": { + "type": "json_schema", + "json_schema": {"schema": schema}, + }, + }, + async_call=True, + ) + + assert api_kwargs["output_config"]["format"] == { + "type": "json_schema", + "schema": schema, + } + _assert_no_raw_response_format(api_kwargs) From 8f2d61e3dea057361f75d1aaea99986e1cded703 Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 20:24:04 -0400 Subject: [PATCH 023/426] fix(aux): translate top-level response_format kwarg on the Anthropic adapter The adapter builds the Messages body from a fixed allow-list of kwargs. A caller that passes response_format as a top-level kwarg (the OpenAI SDK call shape) got it dropped on the floor. The request succeeded, but the schema contract silently became prompt compliance. No in-tree caller uses this shape today. The pin-test makes sure that a future refactor cannot open this leak again. The top-level kwarg gets the same output_config.format translation as the extra_body shape. When a caller sends both shapes, the extra_body value wins because every in-tree caller uses that shape. Pin-test pattern from PR #85626 review follow-up. Co-authored-by: Matt McClean --- agent/auxiliary_client.py | 13 ++++ .../agent/test_anthropic_structured_output.py | 71 ++++++++++++++++++- 2 files changed, 82 insertions(+), 2 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 587b9442d3..8bc5faf871 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2100,6 +2100,19 @@ class _AnthropicCompletionsAdapter: # - ``_``-prefixed keys: private Hermes plumbing (_reasoning_config # et al.), never wire fields. caller_extra_body = kwargs.get("extra_body") + # A top-level ``response_format`` kwarg (the OpenAI SDK's documented + # call shape) must get the same translation as the extra_body form. + # The adapter builds the Messages body from a fixed allow-list of + # kwargs, so before this an unrecognized top-level kwarg was dropped + # on the floor: the request succeeded but the schema contract + # silently became prompt compliance (#85626 review, point 2). When + # both shapes are present, the extra_body form wins — it is the shape + # every in-tree caller uses. + top_level_response_format = kwargs.get("response_format") + if top_level_response_format is not None: + _translate_anthropic_response_format( + anthropic_kwargs, top_level_response_format, + ) if caller_extra_body and isinstance(caller_extra_body, dict): _translate_anthropic_response_format( anthropic_kwargs, caller_extra_body.get("response_format"), diff --git a/tests/agent/test_anthropic_structured_output.py b/tests/agent/test_anthropic_structured_output.py index eb504b6a55..34f92a0593 100644 --- a/tests/agent/test_anthropic_structured_output.py +++ b/tests/agent/test_anthropic_structured_output.py @@ -8,7 +8,8 @@ from unittest.mock import MagicMock, patch def _capture_anthropic_kwargs( - extra_body: dict, *, model: str = "claude-sonnet-4-6", async_call: bool = False, + extra_body: dict | None, *, model: str = "claude-sonnet-4-6", + async_call: bool = False, top_level_kwargs: dict | None = None, ) -> dict: from agent.auxiliary_client import ( _AnthropicCompletionsAdapter, @@ -35,6 +36,9 @@ def _capture_anthropic_kwargs( reasoning=None, finish_reason="stop", ) + call_kwargs = dict(top_level_kwargs or {}) + if extra_body is not None: + call_kwargs["extra_body"] = extra_body with patch( "agent.anthropic_adapter.create_anthropic_message", side_effect=_fake_create, @@ -44,7 +48,7 @@ def _capture_anthropic_kwargs( model=model, messages=[{"role": "user", "content": "hi"}], max_tokens=64, - extra_body=extra_body, + **call_kwargs, ) if async_call: asyncio.run(call) @@ -145,3 +149,66 @@ def test_async_anthropic_adapter_uses_the_same_translation(): "schema": schema, } _assert_no_raw_response_format(api_kwargs) + + +def test_top_level_response_format_kwarg_is_translated_not_dropped(): + """#85626 review, point 2: the top-level kwarg shape must also translate. + + ``client.chat.completions.create(..., response_format=...)`` is the + OpenAI SDK's documented call shape. The adapter builds the Messages body + from a fixed allow-list of kwargs, so before this an unrecognized + top-level kwarg was dropped on the floor: the request succeeded, but the + schema contract silently became prompt compliance. Pin-test pattern from + PR #85626 (Matt McClean), adapted from strip to translate semantics. + """ + schema = { + "type": "object", + "properties": {"title": {"type": "string"}}, + "required": ["title"], + } + api_kwargs = _capture_anthropic_kwargs( + None, + top_level_kwargs={ + "response_format": { + "type": "json_schema", + "json_schema": {"name": "session_title", "schema": schema}, + }, + }, + ) + + assert api_kwargs["output_config"]["format"] == { + "type": "json_schema", + "schema": schema, + } + _assert_no_raw_response_format(api_kwargs) + + +def test_extra_body_response_format_wins_over_top_level_kwarg(): + """When both shapes are present, extra_body wins. + + Every in-tree caller uses the extra_body shape. The top-level kwarg is + the compatibility path, so it must not override an explicit extra_body + value when a caller somehow sends both. + """ + eb_schema = {"type": "object", "properties": {"a": {"type": "string"}}} + top_schema = {"type": "object", "properties": {"b": {"type": "string"}}} + api_kwargs = _capture_anthropic_kwargs( + { + "response_format": { + "type": "json_schema", + "json_schema": {"schema": eb_schema}, + }, + }, + top_level_kwargs={ + "response_format": { + "type": "json_schema", + "json_schema": {"schema": top_schema}, + }, + }, + ) + + assert api_kwargs["output_config"]["format"] == { + "type": "json_schema", + "schema": eb_schema, + } + _assert_no_raw_response_format(api_kwargs) From 3c675019f1249b137fd2baa381f5986cc33036ab Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 20:28:05 -0400 Subject: [PATCH 024/426] fix(aux): retry once without response_format when a provider rejects it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Some providers reject the structured-output request field with a hard 400. The error classifier marks a 400 as non-retryable, so one rejected field failed the whole auxiliary call. Session titles stayed derived forever (#82816), and no fallback fired. Three rejection shapes are covered, from live reports: - vLLM gateways translate response_format into guided_grammar and fail when the grammar backend is absent (compile_grammar_error: No module named 'xgrammar'). - Some OpenAI-compatible endpoints answer "This response_format type is unavailable now". - Anthropic-compatible gateways that predate structured outputs reject the translated field: "output_config: Extra inputs are not permitted". The documented case is the bedrock-mantle Messages endpoint. The fix is reactive, the same pattern as the temperature and max_tokens rungs: when the provider rejects the field, retry once without it. Callers tolerate an unconstrained reply — the title prompt demands bare JSON and _extract_title_text has a loose-JSON fallback — so the call succeeds with prompt compliance instead of failing. The retry only fires when the request carried the field, and both the sync and async paths get the same rung. Closes #82816 --- agent/auxiliary_client.py | 132 ++++++++ .../test_structured_output_rejection_retry.py | 307 ++++++++++++++++++ 2 files changed, 439 insertions(+) create mode 100644 tests/agent/test_structured_output_rejection_retry.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 8bc5faf871..ca6bea2d26 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -4368,6 +4368,71 @@ def _is_unsupported_temperature_error(exc: Exception) -> bool: return _is_unsupported_parameter_error(exc, "temperature") +def _is_structured_output_rejection(exc: Exception) -> bool: + """Detect provider 400s that reject the structured-output request field. + + One predicate covers the field on both wires, because both come from the + same caller-supplied ``response_format``: + + - OpenAI wire: the provider rejects ``response_format`` itself. vLLM + gateways translate the field into ``guided_grammar`` and fail when the + grammar backend is absent (``compile_grammar_error: No module named + 'xgrammar'``, #82816). Other endpoints answer ``This response_format + type is unavailable now``. + - Anthropic wire: the adapter translates ``response_format`` into + ``output_config.format``. Gateways that predate structured outputs + (the documented case is the ``bedrock-mantle`` Messages endpoint) + reject that field: ``output_config: Extra inputs are not permitted``. + + Callers tolerate an unconstrained reply — the title prompt demands bare + JSON and ``_extract_title_text`` has a loose-JSON fallback — so the right + reaction is one retry without the field, not a hard failure. + """ + status = getattr(exc, "status_code", None) + if status is not None and status not in {400, 422}: + return False + err_lower = str(exc).lower() + # vLLM grammar-backend failures name the translated parameter, not ours. + if "guided_grammar" in err_lower or "xgrammar" in err_lower or ( + "compile_grammar_error" in err_lower + ): + return True + if "extra inputs are not permitted" in err_lower and ( + "response_format" in err_lower or "output_config" in err_lower + ): + return True + if "response_format" in err_lower and "unavailable" in err_lower: + return True + return ( + _is_unsupported_parameter_error(exc, "response_format") + or _is_unsupported_parameter_error(exc, "output_config") + ) + + +def _without_structured_output_format(kwargs: dict) -> Optional[dict]: + """Copy *kwargs* without any ``response_format`` request field. + + Removes the top-level kwarg and the ``extra_body`` entry. Returns None + when the kwargs carry no such field, so call sites do not retry a + request that the removal did not change. + """ + changed = False + retry_kwargs = dict(kwargs) + if retry_kwargs.pop("response_format", None) is not None: + changed = True + extra_body = retry_kwargs.get("extra_body") + if isinstance(extra_body, dict) and "response_format" in extra_body: + remaining = { + k: v for k, v in extra_body.items() if k != "response_format" + } + if remaining: + retry_kwargs["extra_body"] = remaining + else: + retry_kwargs.pop("extra_body", None) + changed = True + return retry_kwargs if changed else None + + def _is_model_not_found_error(exc: Exception) -> bool: """Detect "the requested model doesn't exist" errors (404 / invalid model). @@ -9541,6 +9606,39 @@ def _call_llm_impl( first_err = retry_err kwargs = retry_kwargs + if _is_structured_output_rejection(first_err): + retry_kwargs = _without_structured_output_format(kwargs) + if retry_kwargs is not None: + logger.info( + "Auxiliary %s: provider rejected the structured-output " + "format field; retrying once without it (schema " + "enforcement degrades to prompt compliance): %s", + task or "call", first_err, + ) + try: + return _validate_llm_response( + _relay_sync_completion( + client, + retry_kwargs, + provider=resolved_provider, + api_mode=resolved_api_mode, + ), task) + except Exception as retry_err: + # Same contract as the temperature rung: fall through to + # the max_tokens / payment / auth chains below with the + # stripped kwargs; re-raise anything those chains do not + # handle. + if not ( + _is_payment_error(retry_err) + or _is_connection_error(retry_err) + or _is_auth_error(retry_err) + or "max_tokens" in str(retry_err) + or "unsupported_parameter" in str(retry_err) + ): + raise + first_err = retry_err + kwargs = retry_kwargs + err_str = str(first_err) # ZAI vision models (glm-4v-flash etc.) return error code 1210 # ("API 调用参数有误") when max_tokens is passed on multimodal @@ -10253,6 +10351,40 @@ async def _async_call_llm_impl( first_err = retry_err kwargs = retry_kwargs + if _is_structured_output_rejection(first_err): + retry_kwargs = _without_structured_output_format(kwargs) + if retry_kwargs is not None: + logger.info( + "Auxiliary %s (async): provider rejected the " + "structured-output format field; retrying once without " + "it (schema enforcement degrades to prompt " + "compliance): %s", + task or "call", first_err, + ) + try: + return _validate_llm_response( + await _relay_async_completion( + client, + retry_kwargs, + provider=resolved_provider, + api_mode=resolved_api_mode, + ), task) + except Exception as retry_err: + # Same contract as the temperature rung: fall through to + # the max_tokens / payment / auth chains below with the + # stripped kwargs; re-raise anything those chains do not + # handle. + if not ( + _is_payment_error(retry_err) + or _is_connection_error(retry_err) + or _is_auth_error(retry_err) + or "max_tokens" in str(retry_err) + or "unsupported_parameter" in str(retry_err) + ): + raise + first_err = retry_err + kwargs = retry_kwargs + err_str = str(first_err) # ZAI vision models (glm-4v-flash etc.) return error code 1210 # ("API 调用参数有误") when max_tokens is passed on multimodal diff --git a/tests/agent/test_structured_output_rejection_retry.py b/tests/agent/test_structured_output_rejection_retry.py new file mode 100644 index 0000000000..6b0132c755 --- /dev/null +++ b/tests/agent/test_structured_output_rejection_retry.py @@ -0,0 +1,307 @@ +"""Regression tests for the structured-output rejection retry in +``agent.auxiliary_client``. + +Auxiliary callers (title generation, plugin structured completions) send an +OpenAI ``response_format`` request field. Some providers reject the field, or +its Anthropic translation, with a hard 400: + + * vLLM gateways translate ``response_format: json_schema`` into + ``guided_grammar`` and fail when the grammar backend is absent + (``compile_grammar_error: No module named 'xgrammar'``, #82816). + * Some OpenAI-compatible endpoints answer + ``This response_format type is unavailable now`` (#82816). + * Anthropic-compatible gateways that predate structured outputs reject the + translated ``output_config`` field with + ``output_config: Extra inputs are not permitted`` (the documented case is + the ``bedrock-mantle`` Messages endpoint). + +Callers tolerate an unconstrained reply: the title prompt demands bare JSON +and ``_extract_title_text`` has a loose-JSON fallback. The fix is reactive, +like the temperature retry: when the provider rejects the structured-output +field, retry once without it. These tests lock in that behaviour for both +sync and async paths. +""" + +from unittest.mock import patch, MagicMock, AsyncMock + +import pytest + +from agent.auxiliary_client import ( + call_llm, + async_call_llm, + _is_structured_output_rejection, + _without_structured_output_format, +) + + +_TITLE_RESPONSE_FORMAT = { + "type": "json_schema", + "json_schema": { + "name": "session_title", + "strict": True, + "schema": { + "type": "object", + "properties": {"title": {"type": "string"}}, + "required": ["title"], + "additionalProperties": False, + }, + }, +} + + +class TestIsStructuredOutputRejection: + """The detector must match the phrasings providers actually return.""" + + @pytest.mark.parametrize("message", [ + # vLLM guided_grammar / xgrammar (#82816, verbatim from the report) + ( + "Error code: 400 - {'error': {'message': 'guided_grammar " + '\'{"additionalProperties":false}\' has compile_grammar_error: ' + "No module named 'xgrammar'', 'type': 'invalid_request_error'}}" + ), + # Second endpoint from the same report + "HTTP 400: This response_format type is unavailable now", + # Strict Anthropic-wire gateways rejecting the raw OpenAI field + "HTTP 400: response_format: Extra inputs are not permitted", + # Gateways that predate output_config (bedrock-mantle documented case) + "HTTP 400: output_config: Extra inputs are not permitted", + # Generic unsupported-parameter phrasings for both field names + "Unsupported parameter: response_format", + "output_config is not supported", + ]) + def test_matches_real_provider_messages(self, message): + assert _is_structured_output_rejection(RuntimeError(message)) is True + + @pytest.mark.parametrize("message", [ + # Unrelated 400s must NOT trigger a silent schema downgrade + "HTTP 400: Invalid value: 'tool'. Supported values are: 'assistant'", + "HTTP 400: Unsupported parameter: temperature", + "max_tokens is too large for this model", + "Rate limit exceeded", + "Connection reset by peer", + # Alternation errors that happen to mention messages + "messages: Extra inputs are not permitted", + ]) + def test_does_not_match_unrelated_errors(self, message): + assert _is_structured_output_rejection(RuntimeError(message)) is False + + def test_does_not_match_non_400_statuses(self): + exc = RuntimeError("output_config: Extra inputs are not permitted") + exc.status_code = 500 + assert _is_structured_output_rejection(exc) is False + + +class TestWithoutStructuredOutputFormat: + """The kwargs scrubber removes the field on both call shapes.""" + + def test_removes_extra_body_entry_and_keeps_siblings(self): + kwargs = { + "model": "m", + "extra_body": { + "response_format": dict(_TITLE_RESPONSE_FORMAT), + "metadata": {"user_id": "u1"}, + }, + } + result = _without_structured_output_format(kwargs) + assert result is not None + assert result["extra_body"] == {"metadata": {"user_id": "u1"}} + # The input dict is not mutated. + assert "response_format" in kwargs["extra_body"] + + def test_drops_extra_body_entirely_when_it_becomes_empty(self): + kwargs = { + "model": "m", + "extra_body": {"response_format": dict(_TITLE_RESPONSE_FORMAT)}, + } + result = _without_structured_output_format(kwargs) + assert result is not None + assert "extra_body" not in result + + def test_removes_top_level_kwarg(self): + kwargs = {"model": "m", "response_format": dict(_TITLE_RESPONSE_FORMAT)} + result = _without_structured_output_format(kwargs) + assert result is not None + assert "response_format" not in result + + def test_returns_none_when_nothing_to_remove(self): + assert _without_structured_output_format({"model": "m"}) is None + assert _without_structured_output_format( + {"model": "m", "extra_body": {"metadata": {}}} + ) is None + + +def _dummy_response(): + return {"ok": True} + + +class TestCallLlmStructuredOutputRetry: + """``call_llm`` retries once without the field and returns on success.""" + + def _setup(self, first_exc): + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create.side_effect = [ + first_exc, _dummy_response(), + ] + return client + + @pytest.mark.parametrize("error_message", [ + # vLLM guided_grammar (#82816) + "Error code: 400 - guided_grammar has compile_grammar_error: " + "No module named 'xgrammar'", + # Second endpoint flavor from the same report + "HTTP 400: This response_format type is unavailable now", + # Strict gateway that rejects the translated Anthropic field + "HTTP 400: output_config: Extra inputs are not permitted", + ]) + def test_retries_once_without_response_format(self, error_message): + client = self._setup(RuntimeError(error_message)) + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai-codex", "gpt-5.5", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "gpt-5.5")), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + ): + result = call_llm( + task="title_generation", + messages=[{"role": "user", "content": "hi"}], + max_tokens=64, + extra_body={"response_format": dict(_TITLE_RESPONSE_FORMAT)}, + ) + + assert result == {"ok": True} + assert client.chat.completions.create.call_count == 2 + first_kwargs = client.chat.completions.create.call_args_list[0].kwargs + retry_kwargs = client.chat.completions.create.call_args_list[1].kwargs + first_eb = first_kwargs.get("extra_body") or {} + retry_eb = retry_kwargs.get("extra_body") or {} + assert "response_format" in first_eb + assert "response_format" not in retry_eb + assert "response_format" not in retry_kwargs + assert retry_kwargs["model"] == first_kwargs["model"] + + def test_unrelated_400_does_not_strip_response_format(self): + """Unrelated 400s must not silently downgrade the schema contract.""" + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create.side_effect = RuntimeError( + "HTTP 400: Invalid value: 'tool'. Supported values are: 'assistant'" + ) + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai-codex", "gpt-5.5", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "gpt-5.5")), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + patch("agent.auxiliary_client._try_payment_fallback", + return_value=None), + ): + with pytest.raises(RuntimeError, match="Invalid value"): + call_llm( + task="title_generation", + messages=[{"role": "user", "content": "x"}], + max_tokens=64, + extra_body={ + "response_format": dict(_TITLE_RESPONSE_FORMAT), + }, + ) + assert client.chat.completions.create.call_count == 1 + + def test_no_retry_when_no_response_format_was_sent(self): + """A rejection with no field in the request must not loop a retry.""" + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create.side_effect = RuntimeError( + "HTTP 400: output_config: Extra inputs are not permitted" + ) + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai-codex", "gpt-5.5", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "gpt-5.5")), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + patch("agent.auxiliary_client._try_payment_fallback", + return_value=None), + ): + with pytest.raises(RuntimeError): + call_llm( + task="title_generation", + messages=[{"role": "user", "content": "x"}], + max_tokens=64, + ) + assert client.chat.completions.create.call_count == 1 + + +class TestAsyncCallLlmStructuredOutputRetry: + """``async_call_llm`` mirror of the sync retry semantics.""" + + @pytest.mark.asyncio + async def test_async_retries_once_without_response_format(self): + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create = AsyncMock(side_effect=[ + RuntimeError( + "Error code: 400 - guided_grammar has compile_grammar_error: " + "No module named 'xgrammar'" + ), + _dummy_response(), + ]) + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai-codex", "gpt-5.5", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "gpt-5.5")), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + ): + result = await async_call_llm( + task="title_generation", + messages=[{"role": "user", "content": "hi"}], + max_tokens=64, + extra_body={"response_format": dict(_TITLE_RESPONSE_FORMAT)}, + ) + + assert result == {"ok": True} + assert client.chat.completions.create.await_count == 2 + first_kwargs = client.chat.completions.create.call_args_list[0].kwargs + retry_kwargs = client.chat.completions.create.call_args_list[1].kwargs + assert "response_format" in (first_kwargs.get("extra_body") or {}) + assert "response_format" not in (retry_kwargs.get("extra_body") or {}) + assert "response_format" not in retry_kwargs + + @pytest.mark.asyncio + async def test_async_unrelated_400_does_not_retry(self): + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create = AsyncMock( + side_effect=RuntimeError("HTTP 400: Invalid value: 'tool'"), + ) + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai-codex", "gpt-5.5", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "gpt-5.5")), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + patch("agent.auxiliary_client._try_payment_fallback", + return_value=None), + ): + with pytest.raises(RuntimeError, match="Invalid value"): + await async_call_llm( + task="title_generation", + messages=[{"role": "user", "content": "x"}], + max_tokens=64, + extra_body={ + "response_format": dict(_TITLE_RESPONSE_FORMAT), + }, + ) + assert client.chat.completions.create.await_count == 1 From d5a9c2ba6c1e04a177e2e1a1c96ea670b8ab8dff Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 11 Aug 2026 20:25:42 -0400 Subject: [PATCH 025/426] feat(nix): home-manager module, shared with the NixOS module Hermes is an agent for one person. The credentials, the memory, the sessions and the cron jobs all belong to that person. But the only declarative path was a NixOS system service. Issue #9056 asks for the user-level equivalent. 25 public Nix configurations already write one by hand, and several of them copy nix/nixosModules.nix and edit the systemd part. This module is not a second copy of that file. The code that both modules share moves into nix/moduleCommon.nix: - the options - the renderers for config.yaml, .env and the documents - the activation body - the command lines of the processes nixosModules.nix keeps only the parts that need root. Those parts are the service user, stateDir, addToSystemPackages, container mode and tmpfiles. The file goes from 1008 lines to 666. `services.hermes-agent` is now the same option set on both modules. A NixOS example works on Home Manager without a change, and an option added one time appears on both. The Home Manager module is different only where it must be. It uses systemd.user.services on Linux and launchd.agents on Darwin. It uses home.activation and not system.activationScripts. It sets HERMES_HOME directly, with the default ~/.hermes, so an existing directory continues to work. It uses the modes 0600 and 0700, because the state has one user and does not need the group-shared umask of the NixOS module. It does not support container mode, which needs root and the Docker socket. The change also makes four corrections that apply to both modules: - backend.mode runs `hermes serve` or `hermes dashboard`. Both modules had only the gateway. But Hermes Desktop and the web dashboard connect to a different process, so six of the configurations in public repos add a second unit by hand. serve and dashboard are one entry point with one flag of difference, and you can run only one of them. Thus the option is an enum. The NixOS module asserts against container mode with a backend, and does not make a unit that cannot start. - hermesHomeFiles installs files into HERMES_HOME. The `documents` option installs into the working directory, which is correct for AGENTS.md but wrong for SOUL.md and memories/. Hermes reads those files from HERMES_HOME, in agent/prompt_builder.py:2095. A SOUL.md in `documents` made a workspace file that Hermes never loaded as the identity. The documentation said this in prose, but two directory diagrams showed the opposite. This change corrects both. A key in either option can now contain subdirectories. - `documents` needs an explicit `workingDirectory`. The default of that option is bad on both modules. It is the home directory of the user on Home Manager, and ${stateDir}/workspace on NixOS. A user who declares workspace files without a directory therefore gets a place that the user did not select. The place is also different on each module. The modules now refuse that combination. The test is on the priority of the option and not on its value. An option that nothing sets keeps the priority of its own default, and each definition from a user is stronger. Thus a directory with the same text as the default still counts as a selection, and so does a mkDefault. A comparison of values detects neither case. - Each activation writes .env again from a base in the Nix store, and does not add to the file that exists. Thus a second activation cannot put the same secret in the file two times, and a removed environmentFile goes away. environmentFiles keeps the type `listOf str` and not `path`, so Nix cannot copy a sops-nix or agenix path into the Nix store, which all users can read. - HERMES_MANAGED and the .managed marker now hold the name of the system that manages the install. Thus a refusal says "managed by home-manager" and not "managed by NixOS", and `hermes update` gives the Nix guidance for both shapes. The CLI does not print a rebuild command for each system. It names the owner, and the user knows their own tool. A bare `true` and an empty marker still mean NixOS, so this does not change an existing install. Verification. Six new checks, all built: nixos-module evaluates the module with evalModules and the NixOS module list. It asserts both units, one HERMES_HOME, and that the module refuses container mode with a backend. home-manager-module evaluates the module with the homeManagerConfiguration function of home-manager. The process assertions run against systemd units on Linux and launchd agents on Darwin. module-option-parity asserts that each shared option is on both modules, and that the two exclusion lists name only options that exist. env-file-assembly runs the real .env script and checks the contents, the mode, that a second run gives the same bytes, and that a removed file goes away. workspace-files-need-a-directory checks that the module refuses `documents` without a directory, and accepts a directory that has the same text as the default. service-argv runs each command line that the modules build through the real parser of the CLI, with one sentinel flag added, and requires that argparse refuses only the sentinel. `nix flake check` passes, with 21 checks in total. The CLI branches that treat an install as a Nix install move to one helper, is_nix_install_method. Four call sites in main.py, web_server.py, update_cmd.py and doctor.py tested the literal set {"nix", "nixos"}, and each one missed home-manager. recommended_update_command asks the managed state before the code-scoped stamp again, because a managed install can carry a stale stamp that names an update path the managed guard refuses. The metrics contract gets a home-manager bucket, so a Home Manager install does not report as unknown. Each check was mutation-probed. 22 faults were injected, and the checks caught all 22: - a lost --no-open - a backend that runs the gateway - an overwritten config.yaml - documents in the wrong directory - a different HERMES_HOME on the two processes - a lost HERMES_HOME export - a missing backend unit - a removed assertion - an .env file that grows at each activation - an install that reports NixOS - an empty .managed marker - an option on the NixOS module only - a stale entry in an exclusion list - a renamed subcommand - an unknown flag - the workspace-files assertion always passes - the assertion compares values instead of priorities - an off-by-one that lets an untouched default through - the assertion also fires for hermesHomeFiles - a mkDefault no longer counts as a selection - the Home Manager module stops wiring the assertion - the NixOS module stops wiring the assertion The 16 Python tests in tests/hermes_cli/test_managed_install_shapes.py were probed the same way. 8 faults were injected and 8 were caught. These tests fail on this tree. They fail in the same way on the stashed HEAD, and they have no relation to Nix: - test_git_probe_tree_kill.py (2 tests) - test_update_import_guard.py (1 test) - test_telegram_media_read_timeout.py (2 tests) - test_teams.py (a collection error) Closes #9056 # Conflicts: # hermes_cli/main.py # hermes_cli/update_cmd.py # hermes_cli/web_server.py --- flake.lock | 21 + flake.nix | 9 + hermes_cli/config.py | 79 +- hermes_cli/doctor.py | 3 +- hermes_cli/gateway.py | 4 +- hermes_cli/main.py | 3 +- .../hermes.shared_metrics.v2.schema.json | 2 +- .../observability/shared_metrics_contract.py | 1 + hermes_cli/update_cmd.py | 8 +- hermes_cli/web_server.py | 3 +- nix/checks.nix | 492 +++++- nix/homeManagerModules.nix | 261 +++ nix/moduleCommon.nix | 916 ++++++++++ nix/nixosModules.nix | 1522 +++++++---------- .../hermes_cli/test_managed_install_shapes.py | 112 ++ website/docs/getting-started/nix-setup.md | 210 ++- 16 files changed, 2649 insertions(+), 997 deletions(-) create mode 100644 nix/homeManagerModules.nix create mode 100644 nix/moduleCommon.nix create mode 100644 tests/hermes_cli/test_managed_install_shapes.py diff --git a/flake.lock b/flake.lock index 99d5dd57e2..5e59b460ce 100644 --- a/flake.lock +++ b/flake.lock @@ -20,6 +20,26 @@ "type": "github" } }, + "home-manager": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1786487128, + "narHash": "sha256-ad60hrRVhH/bo3Jl1YLzO+QcS3mUNZr9LNJdzhMO2P4=", + "owner": "nix-community", + "repo": "home-manager", + "rev": "f404edbfa4117810c96b97048299242fc50e5362", + "type": "github" + }, + "original": { + "owner": "nix-community", + "repo": "home-manager", + "type": "github" + } + }, "nixpkgs": { "locked": { "lastModified": 1785318670, @@ -105,6 +125,7 @@ "root": { "inputs": { "flake-parts": "flake-parts", + "home-manager": "home-manager", "nixpkgs": "nixpkgs", "npm-lockfile-fix": "npm-lockfile-fix", "pyproject-build-systems": "pyproject-build-systems", diff --git a/flake.nix b/flake.nix index 8715490be1..46e1f3d4ff 100644 --- a/flake.nix +++ b/flake.nix @@ -26,6 +26,14 @@ url = "github:jeslie0/npm-lockfile-fix"; inputs.nixpkgs.follows = "nixpkgs"; }; + # Used only by nix/checks.nix, to evaluate homeManagerModules.default + # against the real Home Manager module system rather than a stub of it. + # Consuming the module does not require this input — import it from your + # own home-manager, exactly as you would any other HM module. + home-manager = { + url = "github:nix-community/home-manager"; + inputs.nixpkgs.follows = "nixpkgs"; + }; }; outputs = @@ -41,6 +49,7 @@ ./nix/packages.nix ./nix/overlays.nix ./nix/nixosModules.nix + ./nix/homeManagerModules.nix ./nix/checks.nix ./nix/devShell.nix ]; diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 5e6fd1942e..15cfeaf354 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -338,10 +338,10 @@ from hermes_cli.default_soul import DEFAULT_SOUL_MD, is_legacy_template_soul # ============================================================================= _MANAGED_TRUE_VALUES = ("true", "1", "yes") -_MANAGED_SYSTEM_NAMES = { - "nix": "NixOS", - "nixos": "NixOS", -} +_NIX_MANAGED_SYSTEMS = {"nixos", "home-manager"} +# Only the NixOS module ever wrote a bare "true" or an empty marker, so both +# legacy signals name that system. +_LEGACY_MANAGED_SYSTEM = "nixos" # The Nix store root. Used by detect_install_method to identify installs # from `nix run` / `nix profile install` (which don't set HERMES_MANAGED). # A module-level constant so tests can patch it without creating files @@ -357,18 +357,30 @@ _IGNORED_MANAGED_VALUES = frozenset({"brew", "homebrew"}) def get_managed_system() -> Optional[str]: """Return the package manager owning this install, if any.""" raw = os.getenv("HERMES_MANAGED", "").strip() + marker = None if raw: - normalized = raw.lower() - if normalized in _IGNORED_MANAGED_VALUES: - return None - if normalized in _MANAGED_TRUE_VALUES: - return "NixOS" - return _MANAGED_SYSTEM_NAMES.get(normalized, raw) + marker = raw.lower() + else: + managed_marker = get_hermes_home() / ".managed" + # An interactive shell reads the marker, because it does not see the + # HERMES_MANAGED variable of the service. A marker with content + # names the system that manages the install. + if managed_marker.exists(): + try: + marker = managed_marker.read_text(encoding="utf-8", errors="replace").strip().lower() + except OSError: + marker = "" - managed_marker = get_hermes_home() / ".managed" - if managed_marker.exists(): - return "NixOS" - return None + if marker is None: + return None + + if marker in _IGNORED_MANAGED_VALUES: + return None + + if marker == "" or marker in _MANAGED_TRUE_VALUES: + return _LEGACY_MANAGED_SYSTEM + + return marker def is_managed() -> bool: @@ -381,6 +393,9 @@ def is_managed() -> bool: return get_managed_system() is not None +# Nix installs arrive by several routes (nix run, nix profile, a system flake, +# home-manager), and the running process cannot tell which one. Thus this text +# names the routes instead of one command. _NIX_UPDATE_MSG = ( "Update Hermes through the Nix source that installed it " "(e.g. nix profile upgrade, or update your flake input and rebuild with nixos-rebuild or home-manager switch)" @@ -390,7 +405,7 @@ _NIX_UPDATE_MSG = ( def get_managed_update_command() -> Optional[str]: """Return the preferred upgrade command for a managed install.""" managed_system = get_managed_system() - if managed_system == "NixOS": + if managed_system in _NIX_MANAGED_SYSTEMS: return _NIX_UPDATE_MSG return None @@ -546,9 +561,19 @@ def stamp_install_method(method: str, project_root: Optional[Path] = None) -> No pass +def is_nix_install_method(method: str) -> bool: + """Return True for every install method that Nix owns. + + The callers that branch on the install method must treat "nix", + "nixos" and "home-manager" the same way. One helper keeps the three + names in one place, so a new Nix shape cannot miss a call site. + """ + return method == "nix" or method in _NIX_MANAGED_SYSTEMS + + def recommended_update_command_for_method(method: str) -> str: """Return the update command or guidance for a given install method.""" - if method in {"nix", "nixos"}: + if is_nix_install_method(method): return _NIX_UPDATE_MSG if method == "docker": return "docker pull nousresearch/hermes-agent:latest" @@ -561,6 +586,9 @@ def recommended_update_command_for_method(method: str) -> str: def recommended_update_command() -> str: """Return the best update command for the current installation.""" + # The managed state wins over the code-scoped stamp. A managed install + # can carry a stale stamp from an earlier install shape, and the stamp + # then names an update path that the managed guard refuses. managed_cmd = get_managed_update_command() if managed_cmd: return managed_cmd @@ -623,17 +651,6 @@ def format_docker_update_message() -> str: def format_managed_message(action: str = "modify this Hermes installation") -> str: """Build a user-facing error for managed installs.""" managed_system = get_managed_system() or "a package manager" - raw = os.getenv("HERMES_MANAGED", "").strip().lower() - - if managed_system == "NixOS": - env_hint = "true" if raw in _MANAGED_TRUE_VALUES else raw or "true" - return ( - f"Cannot {action}: this Hermes installation is managed by NixOS " - f"(HERMES_MANAGED={env_hint}).\n" - "Edit services.hermes-agent.settings in your configuration.nix and run:\n" - " sudo nixos-rebuild switch" - ) - return ( f"Cannot {action}: this Hermes installation is managed by {managed_system}.\n" "Use your package manager to upgrade or reinstall Hermes." @@ -928,16 +945,12 @@ def _ensure_hermes_home_managed(home: Path): """Managed-mode variant: verify dirs exist (activation creates them), seed SOUL.md.""" if not home.is_dir(): raise RuntimeError( - f"HERMES_HOME {home} does not exist. " - "Run 'sudo nixos-rebuild switch' first." + f"HERMES_HOME {home} does not exist." ) for subdir in ("cron", "sessions", "logs", "memories"): d = home / subdir if not d.is_dir(): - raise RuntimeError( - f"{d} does not exist. " - "Run 'sudo nixos-rebuild switch' first." - ) + raise RuntimeError(f"{d} does not exist.") # Curator reports dir is a sub-path of logs/; create it if missing. # In managed mode the activation script may not know about this subdir, # so we mkdir it ourselves (it's inside an already-secured logs/ dir). diff --git a/hermes_cli/doctor.py b/hermes_cli/doctor.py index a75c9c6c00..253fb1c7bc 100644 --- a/hermes_cli/doctor.py +++ b/hermes_cli/doctor.py @@ -16,6 +16,7 @@ from hermes_cli.config import ( get_env_path, get_hermes_home, get_project_root, + is_nix_install_method, recommended_update_command_for_method, ) from hermes_cli.env_loader import load_hermes_dotenv @@ -90,7 +91,7 @@ def _sqlite_upgrade_hint(install_method: str | None = None) -> str: if method == "docker": command = recommended_update_command_for_method(method) action = f"run `{command}`, then recreate all Hermes containers" - elif method in {"nix", "nixos"}: + elif is_nix_install_method(method): # The Nix helper is prose guidance, not a literal shell command. action = recommended_update_command_for_method(method) elif method == "apt": diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index ade328da4f..b1871b15ae 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -7511,7 +7511,7 @@ def _gateway_command_inner(args): # Service management commands if subcmd == "install": if is_managed(): - managed_error("install gateway service (managed by NixOS)") + managed_error("install gateway service") return force = getattr(args, "force", False) system = getattr(args, "system", False) @@ -7623,7 +7623,7 @@ def _gateway_command_inner(args): elif subcmd == "uninstall": if is_managed(): - managed_error("uninstall gateway service (managed by NixOS)") + managed_error("uninstall gateway service") return system = getattr(args, "system", False) if is_termux(): diff --git a/hermes_cli/main.py b/hermes_cli/main.py index 044501e2b6..bb0dc3fcd7 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -9905,6 +9905,7 @@ def cmd_update(args): detect_install_method, format_docker_update_message, is_managed, + is_nix_install_method, managed_error, recommended_update_command_for_method, ) @@ -9924,7 +9925,7 @@ def cmd_update(args): print(format_docker_update_message()) sys.exit(1) - if install_method in {"nix", "nixos", "apt"}: + if is_nix_install_method(install_method) or install_method == "apt": print(recommended_update_command_for_method(install_method)) sys.exit(1) diff --git a/hermes_cli/observability/schemas/hermes.shared_metrics.v2.schema.json b/hermes_cli/observability/schemas/hermes.shared_metrics.v2.schema.json index 0052569679..5d845ce612 100644 --- a/hermes_cli/observability/schemas/hermes.shared_metrics.v2.schema.json +++ b/hermes_cli/observability/schemas/hermes.shared_metrics.v2.schema.json @@ -58,7 +58,7 @@ }, "install_method": { "type": "string", - "enum": ["apt", "docker", "git", "homebrew", "nixos", "pip", "unknown"] + "enum": ["apt", "docker", "git", "home-manager", "homebrew", "nixos", "pip", "unknown"] }, "os_family": { "type": "string", diff --git a/hermes_cli/observability/shared_metrics_contract.py b/hermes_cli/observability/shared_metrics_contract.py index 6a1cec1c63..eb1f858216 100644 --- a/hermes_cli/observability/shared_metrics_contract.py +++ b/hermes_cli/observability/shared_metrics_contract.py @@ -197,6 +197,7 @@ CLIENT_INSTALL_METHODS: frozenset[str] = frozenset({ "apt", "docker", "git", + "home-manager", "homebrew", "nixos", "pip", diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index e05850c7b9..107f6f778c 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -2690,7 +2690,11 @@ def _cmd_update_check(branch: str = "main", *, branch_explicit: bool = False): Installs that can't honor non-default branches (e.g. Docker) surface a one-line notice instead of silently dropping the flag. """ - from hermes_cli.config import detect_install_method, recommended_update_command_for_method + from hermes_cli.config import ( + detect_install_method, + is_nix_install_method, + recommended_update_command_for_method, + ) method = detect_install_method(_m().PROJECT_ROOT) if method == "docker": # Docker can't ``git fetch`` from within the container. Surface the @@ -2701,7 +2705,7 @@ def _cmd_update_check(branch: str = "main", *, branch_explicit: bool = False): print(format_docker_update_message()) sys.exit(1) - if method in {"nix", "nixos", "apt"}: + if is_nix_install_method(method) or method == "apt": print(recommended_update_command_for_method(method)) sys.exit(1) diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 85f6d61ee5..439d8cd66c 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -78,6 +78,7 @@ from hermes_cli.config import ( check_config_version, detect_install_method, format_docker_update_message, + is_nix_install_method, recommended_update_command_for_method, redact_key, write_platform_config_field, @@ -4776,7 +4777,7 @@ async def update_hermes(): "update_command": recommended_update_command_for_method(install_method), } - if install_method in {"nix", "nixos", "apt"}: + if is_nix_install_method(install_method) or install_method == "apt": message = recommended_update_command_for_method(install_method) _record_completed_action("hermes-update", message, exit_code=1) return { diff --git a/nix/checks.nix b/nix/checks.nix index 1cf3421259..a9c32176fc 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -11,6 +11,73 @@ configMergeScript = pkgs.callPackage ./configMergeScript.nix { }; + # ── How the checks evaluate the modules ─────────────────────────── + # The checks evaluate both modules for real. The NixOS module goes + # through lib.evalModules with the NixOS module list. The Home Manager + # module goes through the homeManagerConfiguration function of + # home-manager. The option system rejects a wrong type, an option that + # does not exist, and a broken activation string. Each of these faults + # then stops the check, and not the rebuild of a user. + evalNixosModule = + settings: + inputs.nixpkgs.lib.evalModules { + modules = import "${inputs.nixpkgs}/nixos/modules/module-list.nix" ++ [ + inputs.self.nixosModules.default + { _module.args.lib = inputs.nixpkgs.lib; } + { nixpkgs.hostPlatform = pkgs.stdenv.hostPlatform.system; } + { + system.stateVersion = "24.11"; + boot.loader.grub.enable = false; + fileSystems."/" = { + device = "/dev/null"; + fsType = "ext4"; + }; + } + { services.hermes-agent = settings; } + ]; + }; + + evalHomeModule = + settings: + inputs.home-manager.lib.homeManagerConfiguration { + inherit pkgs; + modules = [ + inputs.self.homeManagerModules.default + { + home = { + username = "hermes-check"; + homeDirectory = "/home/hermes-check"; + stateVersion = "24.11"; + }; + } + { services.hermes-agent = settings; } + ]; + }; + + # The option names that each module defines under + # services.hermes-agent. The internal names that the module system adds + # are not in the list. + moduleOptionNames = + eval: lib.attrNames (lib.filterAttrs (n: _: !lib.hasPrefix "_" n) eval.options.services.hermes-agent); + + # These options belong to one module by design. The check does not + # compare the two lists against each other, because that test only + # detects a change. The important property is that each shared option + # is on both modules. + nixosOnlyOptions = [ + "addToSystemPackages" + "container" + "createUser" + "group" + "stateDir" + "user" + ]; + homeOnlyOptions = [ + "gateway" + "hermesHome" + "installPackage" + ]; + # Auto-generated config key reference — always in sync with Python configKeys = pkgs.runCommand "hermes-config-keys" {} '' set -euo pipefail @@ -74,7 +141,427 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) mkdir -p $out echo "ok" > $out/result ''; + + # ── The Home Manager module ────────────────────────────────────── + # This check evaluates homeManagerModules.default through the real + # module system of home-manager. It runs on each platform. The module + # supports Linux, with systemd user units, and Darwin, with launchd + # agents. Each host checks its own kind of process. + home-manager-module = + let + enabled = evalHomeModule { + enable = true; + gateway.enable = true; + backend.mode = "serve"; + settings.model.default = "test/model"; + environment.HERMES_TEST = "1"; + environmentFiles = [ "/run/secrets/hermes-env" ]; + hermesHomeFiles."SOUL.md" = "test soul"; + # documents needs an explicit workingDirectory. The check + # workspace-files-need-a-directory below asserts that rule. + workingDirectory = "/home/test-user/workspace"; + documents."AGENTS.md" = "test agents"; + mcpServers.demo = { + command = "echo"; + args = [ "hi" ]; + }; + }; + cfg = enabled.config; + + # The gateway and the backend are two processes with one + # HERMES_HOME. + processes = + if pkgs.stdenv.hostPlatform.isDarwin then + lib.mapAttrs (_: agent: { + argv = agent.config.ProgramArguments; + env = agent.config.EnvironmentVariables; + }) (lib.filterAttrs (n: _: lib.hasPrefix "hermes" n) cfg.launchd.agents) + else + lib.mapAttrs (_: unit: { + argv = [ unit.Service.ExecStart ]; + env = unit.Service.Environment; + }) (lib.filterAttrs (n: _: lib.hasPrefix "hermes" n) cfg.systemd.user.services); + + names = lib.attrNames processes; + argvOf = name: lib.concatStringsSep " " (lib.flatten (processes.${name}.argv)); + # The systemd Environment is a list of "K=V" strings. The launchd + # equivalent is an attribute set. Make both into one "K=V K=V" + # string, so that the assertions below are the same on each host. + envOf = + name: + let + env = processes.${name}.env; + in + lib.concatStringsSep " " ( + if lib.isAttrs env then lib.mapAttrsToList (k: v: "${k}=${toString v}") env else env + ); + + activation = cfg.home.activation.hermesAgentSetup.data; + + failures = + lib.optional (names != [ + "hermes-agent" + "hermes-backend" + ]) "expected hermes-agent + hermes-backend processes, got: ${toString names}" + ++ lib.optional ( + !lib.hasInfix "bin/hermes gateway" (argvOf "hermes-agent") + ) "gateway process does not run `hermes gateway`: ${argvOf "hermes-agent"}" + ++ lib.optional ( + !lib.hasInfix "bin/hermes serve" (argvOf "hermes-backend") + ) "backend process does not run `hermes serve`: ${argvOf "hermes-backend"}" + ++ lib.optional ( + !lib.hasInfix "--no-open" (argvOf "hermes-backend") + ) "backend must pass --no-open so a service never opens a browser" + ++ lib.optional ( + lib.any (n: !lib.hasInfix "/home/hermes-check/.hermes" (envOf n)) names + ) "gateway and backend must share one HERMES_HOME" + ++ lib.optional ( + cfg.home.sessionVariables.HERMES_HOME or null != "/home/hermes-check/.hermes" + ) "installPackage must export HERMES_HOME for interactive shells" + ++ lib.optional ( + !lib.hasInfix "hermes-config-merge" activation + ) "activation must deep-merge config.yaml, not overwrite it" + ++ lib.optional ( + !lib.hasInfix "/home/hermes-check/.hermes/SOUL.md" activation + ) "hermesHomeFiles must install into HERMES_HOME" + ++ lib.optional ( + !lib.hasInfix "/home/test-user/workspace/AGENTS.md" activation + ) "documents must install into workingDirectory" + # The CLI reads HERMES_MANAGED to name the rebuild command when + # it refuses to write the configuration. A Home Manager install + # has no nixos-rebuild command. Thus it must not report NixOS. + ++ lib.optional ( + !lib.any (n: lib.hasInfix "HERMES_MANAGED=home-manager" (envOf n)) names + ) "processes must report HERMES_MANAGED=home-manager" + ++ lib.optional ( + !lib.hasInfix "hermes-managed" activation + ) "activation must write a .managed marker naming the managing system"; + in + pkgs.runCommand "hermes-home-manager-module" { } ( + if failures != [ ] then + throw "Home Manager module check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: home-manager module evaluates (${toString (lib.length names)} processes)" + mkdir -p $out + echo "ok" > $out/result + '' + ); + + # ── Workspace files need a chosen directory ────────────────────── + # `documents` goes into workingDirectory. The default of that option + # is bad, and it is different on each module, so the modules refuse + # the two options together. + # + # Home Manager reads the assertions while it builds `config`. Thus a + # refused case throws an error and does not return a list. tryEval + # makes the error into data again. + workspace-files-need-a-directory = + let + accepts = + settings: + let + cfg = (evalHomeModule ({ enable = true; } // settings)).config; + probe = builtins.tryEval (lib.all (a: a.assertion) cfg.assertions); + in + probe.success && probe.value; + + rejects = settings: !(accepts settings); + + # This directory has the same text as the default. A comparison + # of values reads it as untouched, but a comparison of priorities + # sees the definition. This row is the reason that the code tests + # the priority. + sameAsDefault = "/home/hermes-check"; + + cases = [ + { + name = "documents without a directory is refused"; + ok = rejects { documents."AGENTS.md" = "x"; }; + } + { + name = "documents with a directory is accepted"; + ok = accepts { + documents."AGENTS.md" = "x"; + workingDirectory = "/srv/workspace"; + }; + } + { + name = "a directory equal to the default still counts as chosen"; + ok = accepts { + documents."AGENTS.md" = "x"; + workingDirectory = sameAsDefault; + }; + } + { + name = "mkDefault counts as chosen"; + ok = accepts { + documents."AGENTS.md" = "x"; + workingDirectory = lib.mkDefault "/srv/workspace"; + }; + } + { + name = "hermesHomeFiles needs no directory"; + ok = accepts { hermesHomeFiles."SOUL.md" = "x"; }; + } + { + name = "no files at all is accepted"; + ok = accepts { }; + } + ]; + + failed = lib.filter (c: !c.ok) cases; + in + pkgs.runCommand "hermes-workspace-files-need-a-directory" { } ( + if failed != [ ] then + throw "workspace-files rule failed:\n${ + lib.concatMapStringsSep "\n" (c: " - ${c.name}") failed + }" + else + '' + ${lib.concatMapStringsSep "\n" (c: ''echo "PASS: ${c.name}"'') cases} + mkdir -p $out + echo "ok" > $out/result + '' + ); + + # ── The two modules keep the same options ──────────────────────── + # The modules share one option set, in nix/moduleCommon.nix. Thus a + # NixOS example works on Home Manager without a change. This check + # asserts that relation and not the current list of names. An option + # that goes into the shared set must appear on both modules. An + # option for one module must be in that module's exclusion list. + module-option-parity = + let + nixosNames = moduleOptionNames (evalNixosModule { }); + homeNames = moduleOptionNames (evalHomeModule { }); + + sharedFromNixos = lib.subtractLists nixosOnlyOptions nixosNames; + sharedFromHome = lib.subtractLists homeOnlyOptions homeNames; + + missingInHome = lib.subtractLists homeNames sharedFromNixos; + missingInNixos = lib.subtractLists nixosNames sharedFromHome; + + # These two values check the exclusion lists. An entry for an + # option that does not exist makes the check weaker, and gives no + # message. + staleNixosOnly = lib.subtractLists nixosNames nixosOnlyOptions; + staleHomeOnly = lib.subtractLists homeNames homeOnlyOptions; + + failures = + lib.optional ( + missingInHome != [ ] + ) "shared options missing from the Home Manager module: ${toString missingInHome} (add to nix/moduleCommon.nix, or list under nixosOnlyOptions if system-scoped)" + ++ lib.optional ( + missingInNixos != [ ] + ) "shared options missing from the NixOS module: ${toString missingInNixos} (add to nix/moduleCommon.nix, or list under homeOnlyOptions if user-scoped)" + ++ lib.optional ( + staleNixosOnly != [ ] + ) "nixosOnlyOptions names options the NixOS module no longer defines: ${toString staleNixosOnly}" + ++ lib.optional ( + staleHomeOnly != [ ] + ) "homeOnlyOptions names options the Home Manager module no longer defines: ${toString staleHomeOnly}"; + in + pkgs.runCommand "hermes-module-option-parity" { } ( + if failures != [ ] then + throw "Module option parity failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: ${toString (lib.length sharedFromNixos)} shared options present on both modules" + mkdir -p $out + echo "ok" > $out/result + '' + ); } // lib.optionalAttrs pkgs.stdenv.hostPlatform.isLinux { + # ── The NixOS module ───────────────────────────────────────────── + # This check runs on Linux only. The evaluation of a NixOS module + # needs a Linux hostPlatform. + nixos-module = + let + cfg = (evalNixosModule { + enable = true; + backend.mode = "dashboard"; + settings.model.default = "test/model"; + environmentFiles = [ "/run/secrets/hermes-env" ]; + hermesHomeFiles."SOUL.md" = "test soul"; + }).config; + + units = lib.filterAttrs (n: _: lib.hasPrefix "hermes" n) cfg.systemd.services; + names = lib.attrNames units; + execOf = name: units.${name}.serviceConfig.ExecStart; + activation = cfg.system.activationScripts."hermes-agent-setup".text; + + failures = + lib.optional (names != [ + "hermes-agent" + "hermes-backend" + ]) "expected hermes-agent + hermes-backend units, got: ${toString names}" + ++ lib.optional ( + !lib.hasInfix "bin/hermes gateway" (execOf "hermes-agent") + ) "gateway unit does not run `hermes gateway`: ${execOf "hermes-agent"}" + ++ lib.optional ( + !lib.hasInfix "bin/hermes dashboard" (execOf "hermes-backend") + ) "backend unit does not run `hermes dashboard`: ${execOf "hermes-backend"}" + ++ lib.optional ( + units.hermes-agent.environment.HERMES_HOME != units.hermes-backend.environment.HERMES_HOME + ) "gateway and backend must share one HERMES_HOME" + ++ lib.optional ( + !lib.hasInfix "/var/lib/hermes/.hermes/SOUL.md" activation + ) "hermesHomeFiles must install into HERMES_HOME"; + + # You cannot use container mode and the backend together. The + # module says so with an assertion. Without the assertion it + # makes a unit that never starts. + containerConflict = builtins.tryEval ( + lib.deepSeq + (evalNixosModule { + enable = true; + container.enable = true; + backend.mode = "serve"; + }).config.system.build.toplevel.drvPath + true + ); + in + pkgs.runCommand "hermes-nixos-module" { } ( + if failures != [ ] then + throw "NixOS module check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else if containerConflict.success then + throw "NixOS module check failed:\n - an assertion must reject backend.mode with container.enable" + else + '' + echo "PASS: nixos module evaluates (${toString (lib.length names)} units)" + mkdir -p $out + echo "ok" > $out/result + '' + ); + + # ── How .env is built ──────────────────────────────────────────── + # This check runs the real script that both modules use to build + # $HERMES_HOME/.env. The important property is that a second run + # gives the same result. Activation runs at each rebuild. If the + # script added the secrets to the file that exists, the file would + # grow at each rebuild. The script writes the file again from the + # base in the Nix store, which prevents that fault. This check proves + # it. + env-file-assembly = + let + envScript = (import ./moduleCommon.nix { inherit lib; }).mkEnvScript { + inherit pkgs; + environment = { + HERMES_PUBLIC = "visible"; + }; + }; + in + pkgs.runCommand "hermes-env-file-assembly" { } '' + set -e + workdir=$(mktemp -d) + printf 'SECRET_TOKEN=s3cret\n' > "$workdir/secret-a" + printf 'OTHER_TOKEN=t0ken\n' > "$workdir/secret-b" + + echo "=== First activation ===" + ${envScript} "$workdir/.env" 0600 "$workdir/secret-a" "$workdir/secret-b" + first=$(cat "$workdir/.env") + + grep -qx 'HERMES_PUBLIC=visible' "$workdir/.env" || \ + (echo "FAIL: non-secret environment missing"; cat "$workdir/.env"; exit 1) + grep -qx 'SECRET_TOKEN=s3cret' "$workdir/.env" || \ + (echo "FAIL: secret from environmentFile missing"; cat "$workdir/.env"; exit 1) + grep -qx 'OTHER_TOKEN=t0ken' "$workdir/.env" || \ + (echo "FAIL: second environmentFile missing"; cat "$workdir/.env"; exit 1) + echo "PASS: .env contains the declared environment and every secret" + + test "$(stat -c %a "$workdir/.env")" = "600" || \ + (echo "FAIL: .env mode is $(stat -c %a "$workdir/.env"), want 600"; exit 1) + echo "PASS: .env installed with the requested mode" + + echo "=== Re-activation is idempotent ===" + ${envScript} "$workdir/.env" 0600 "$workdir/secret-a" "$workdir/secret-b" + second=$(cat "$workdir/.env") + test "$first" = "$second" || \ + (echo "FAIL: second run changed .env"; diff <(echo "$first") <(echo "$second") || true; exit 1) + + COUNT=$(grep -c '^SECRET_TOKEN=' "$workdir/.env") + test "$COUNT" -eq 1 || \ + (echo "FAIL: secret appears $COUNT times after two activations"; exit 1) + echo "PASS: secrets are not accumulated across activations" + + echo "=== A removed environmentFile disappears ===" + ${envScript} "$workdir/.env" 0600 "$workdir/secret-a" + if grep -q '^OTHER_TOKEN=' "$workdir/.env"; then + echo "FAIL: dropped environmentFile still present in .env"; exit 1 + fi + echo "PASS: .env tracks the declared environmentFiles" + + mkdir -p $out + echo "ok" > $out/result + ''; + + # ── The command lines of the services ──────────────────────────── + # The modules build these command lines. This check runs each one + # through the real parser of the CLI. A subcommand or a flag with a + # new name then fails here, and not as a service that restarts again + # and again after a rebuild. + # + # The method: add one sentinel flag that the CLI does not know, and + # parse without --help. argparse refuses unknown arguments before it + # calls the command, so no process starts and no port is bound. The + # error names each argument that argparse did not accept. If the + # error names only the sentinel, the parser accepts each other flag. + # + # `--help` cannot do this job. It returns before argparse reads the + # remainder of the command line. + service-argv = + let + common = import ./moduleCommon.nix { inherit lib; }; + cfgFor = mode: { + package = hermes-agent; + extraPythonPackages = [ ]; + extraDependencyGroups = [ ]; + extraArgs = [ ]; + backend = { + inherit mode; + host = "127.0.0.1"; + port = 9119; + extraArgs = [ ]; + }; + }; + sentinel = "--hermes-nix-argv-probe"; + probe = argv: lib.escapeShellArgs (argv ++ [ sentinel ]); + in + pkgs.runCommand "hermes-service-argv" { } '' + set -e + export HOME=$(mktemp -d) + + check() { + local label="$1" + shift + local output + output=$("$@" 2>&1) && { + echo "FAIL: $label — the sentinel flag was accepted, so this probe proves nothing" + exit 1 + } + case "$output" in + *"unrecognized arguments: ${sentinel}") + echo "PASS: $label — every flag but the sentinel is recognized" ;; + *"unrecognized arguments"*) + echo "FAIL: $label — the CLI also rejected flags the module passes:" + echo "$output" | tail -3 + exit 1 ;; + *) + echo "FAIL: $label — argv rejected before flag parsing (bad subcommand?):" + echo "$output" | tail -3 + exit 1 ;; + esac + } + + check "gateway" ${probe (common.gatewayArgv (cfgFor "none"))} + check "serve" ${probe (common.backendArgv (cfgFor "serve"))} + check "dashboard" ${probe (common.backendArgv (cfgFor "dashboard"))} + + mkdir -p $out + echo "ok" > $out/result + ''; + # Verify binaries exist and are executable package-contents = pkgs.runCommand "hermes-package-contents" { } '' set -e @@ -288,7 +775,10 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) local label="$1" shift OUTPUT=$(HERMES_MANAGED=true "$@" 2>&1 || true) - echo "$OUTPUT" | grep -q "managed by NixOS" || (echo "FAIL: $label not guarded"; echo "$OUTPUT"; exit 1) + # Case-insensitive: the message names the managing system as the + # identifier it is keyed by, and the display form is not the + # property under test here. + echo "$OUTPUT" | grep -qi "managed by nixos" || (echo "FAIL: $label not guarded"; echo "$OUTPUT"; exit 1) echo "PASS: $label blocked in managed mode" } diff --git a/nix/homeManagerModules.nix b/nix/homeManagerModules.nix new file mode 100644 index 0000000000..55b934672c --- /dev/null +++ b/nix/homeManagerModules.nix @@ -0,0 +1,261 @@ +# nix/homeManagerModules.nix — the Home Manager module for hermes-agent +# +# This module is the user-level equivalent of nixosModules.default. Hermes is +# an agent for one person. The credentials, the memory, the sessions and the +# cron jobs all belong to that person. Thus a user-level module is correct on +# each distribution, and not only on NixOS. +# +# `services.hermes-agent` is the same option set on both modules. All of the +# options except the system-level ones come from nix/moduleCommon.nix, so an +# example from the NixOS documentation works here without a change. Only the +# necessary parts are different: +# +# removed user, group, createUser — Home Manager runs as the user +# removed container.* — it needs root and the Docker socket +# removed UMask 0007 — that mode shares state with a UNIX +# group, but this state has one user +# changed systemd.services -> systemd.user.services or +# launchd.agents +# changed system.activationScripts -> home.activation +# changed addToSystemPackages -> installPackage and +# home.sessionVariables +# changed stateDir (+ "/.hermes") -> hermesHome, set directly +# +# To use the module: +# imports = [ hermes-agent.homeManagerModules.default ]; +# services.hermes-agent = { +# enable = true; +# gateway.enable = true; +# settings.model.default = "anthropic/claude-sonnet-4"; +# environmentFiles = [ config.sops.secrets."hermes/env".path ]; +# }; +# +# CAUTION: Enable linger for the account. Without linger, systemd stops the +# user manager at logout, and both units stop with it. Home Manager cannot +# run `loginctl enable-linger`. On NixOS, set +# users.users..linger = true; +# On other systems, run `loginctl enable-linger ` one time. +{ inputs, ... }: +{ + flake.homeManagerModules.default = + { + config, + lib, + options, + pkgs, + ... + }: + + let + cfg = config.services.hermes-agent; + common = import ./moduleCommon.nix { inherit lib; }; + + effectivePackage = common.effectivePackage cfg; + hermes-agent = inputs.self.packages.${pkgs.stdenv.hostPlatform.system}.default; + + inherit (pkgs.stdenv.hostPlatform) isDarwin isLinux; + + processEnvironment = common.processEnvironment { + inherit (cfg) hermesHome; + # The CLI reads this value and names it when it refuses a + # configuration change. + managedSystem = "home-manager"; + }; + unitPath = lib.makeBinPath (common.processPath { inherit pkgs cfg; }); + + # The systemd unit that the gateway and the backend both start from. + mkUnit = + { + description, + argv, + }: + { + Unit = { + Description = description; + # Do not use network-online.target here. That is a system target. + # A user unit that orders against it has no effect, and systemd + # gives no message. + After = [ "default.target" ]; + }; + Install.WantedBy = [ "default.target" ]; + Service = { + Type = "simple"; + Environment = (lib.mapAttrsToList (k: v: "${k}=${v}") processEnvironment) ++ [ + "PATH=${unitPath}" + ]; + ExecStart = lib.escapeShellArgs argv; + WorkingDirectory = cfg.workingDirectory; + Restart = cfg.restart; + RestartSec = cfg.restartSec; + # This state has one user. Keep it private. The NixOS module uses + # 0007 to share the state with a UNIX group. + UMask = "0077"; + NoNewPrivileges = true; + PrivateTmp = true; + }; + }; + + mkAgent = + { argv, logName }: + { + enable = true; + config = { + Label = "org.nix-community.home.${logName}"; + ProgramArguments = argv; + EnvironmentVariables = processEnvironment // { + PATH = "${unitPath}:/usr/bin:/bin:/usr/sbin:/sbin"; + }; + WorkingDirectory = cfg.workingDirectory; + RunAtLoad = true; + KeepAlive = + if cfg.restart == "always" then + true + else + { + SuccessfulExit = false; + Crashed = true; + }; + ThrottleInterval = cfg.restartSec; + StandardOutPath = "${config.home.homeDirectory}/Library/Logs/${logName}.log"; + StandardErrorPath = "${config.home.homeDirectory}/Library/Logs/${logName}.err.log"; + ProcessType = "Background"; + }; + }; + + in + { + options.services.hermes-agent = + common.sharedOptions { + defaultPackage = hermes-agent; + defaultPackageText = lib.literalExpression "hermes-agent.packages.\${system}.default"; + defaultWorkingDirectory = config.home.homeDirectory; + defaultWorkingDirectoryText = lib.literalExpression "config.home.homeDirectory"; + } + // { + hermesHome = lib.mkOption { + type = lib.types.str; + default = "${config.home.homeDirectory}/.hermes"; + defaultText = lib.literalExpression ''"''${config.home.homeDirectory}/.hermes"''; + description = '' + The value of HERMES_HOME. This state directory holds + config.yaml, .env, auth.json, the sessions, the skills, the + memory and the cron jobs. + + The NixOS module takes a `stateDir` and adds `/.hermes` to it. + This module sets HERMES_HOME directly. Thus an existing + ~/.hermes continues to work, and you can give the directory any + name. + ''; + example = "/home/alice/.hermes-work"; + }; + + installPackage = lib.mkOption { + type = lib.types.bool; + default = true; + description = '' + Add the hermes CLI to home.packages, and export HERMES_HOME + with home.sessionVariables. Interactive shells then use the + same state as the services. + + The equivalent NixOS option, `addToSystemPackages`, exports + HERMES_HOME with environment.variables. That variable applies + to the full system and replaces the HERMES_HOME of each other + user. This module exports the variable for one user session + only, which is the reason to use Home Manager. + ''; + }; + + gateway.enable = lib.mkEnableOption "the messaging gateway service (Telegram, Discord, Slack, ...)"; + }; + + config = lib.mkIf cfg.enable ( + lib.mkMerge [ + + # ── Merge MCP servers into settings ──────────────────────────── + (lib.mkIf (cfg.mcpServers != { }) { + services.hermes-agent.settings.mcp_servers = common.mcpServersToConfig cfg.mcpServers; + }) + + { + assertions = + common.pluginNameAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; + } + ++ common.workspaceFilesAssertions { + inherit cfg; + opt = options.services.hermes-agent.workingDirectory; + optionPath = "services.hermes-agent"; + }; + } + + # ── Packages and interactive-shell environment ───────────────── + (lib.mkIf cfg.installPackage { + home.packages = [ effectivePackage ] ++ cfg.extraPackages; + home.sessionVariables.HERMES_HOME = cfg.hermesHome; + }) + + # ── Activation: directories, config, secrets, documents ──────── + { + # The activation runs after writeBoundary, when the home.file + # symlinks are in place. It also runs after linkGeneration, when + # Home Manager completes the switch. A secret that the activation + # entry of sops-nix writes exists at that point. + home.activation.hermesAgentSetup = + lib.hm.dag.entryAfter + [ + "writeBoundary" + "linkGeneration" + ] + ( + common.mkStateScript { + inherit pkgs cfg; + inherit (cfg) hermesHome workingDirectory; + run = "$DRY_RUN_CMD "; + stateDirs = common.stateSubdirs; + managedSystem = "home-manager"; + # This state has one user. No group needs access to it. + modes = { + config = "0600"; + env = "0600"; + managed = "0600"; + auth = "0600"; + document = "0600"; + }; + } + ); + } + + # ── Linux: systemd user services ─────────────────────────────── + (lib.mkIf (isLinux && cfg.gateway.enable) { + systemd.user.services.hermes-agent = mkUnit { + description = "Hermes Agent Gateway"; + argv = common.gatewayArgv cfg; + }; + }) + + (lib.mkIf (isLinux && cfg.backend.mode != "none") { + systemd.user.services.hermes-backend = mkUnit { + description = common.backendDescription cfg; + argv = common.backendArgv cfg; + }; + }) + + # ── Darwin: launchd agents ───────────────────────────────────── + (lib.mkIf (isDarwin && cfg.gateway.enable) { + launchd.agents.hermes-agent = mkAgent { + argv = common.gatewayArgv cfg; + logName = "hermes-agent"; + }; + }) + + (lib.mkIf (isDarwin && cfg.backend.mode != "none") { + launchd.agents.hermes-backend = mkAgent { + argv = common.backendArgv cfg; + logName = "hermes-backend"; + }; + }) + ] + ); + }; +} diff --git a/nix/moduleCommon.nix b/nix/moduleCommon.nix new file mode 100644 index 0000000000..c021209123 --- /dev/null +++ b/nix/moduleCommon.nix @@ -0,0 +1,916 @@ +# nix/moduleCommon.nix — the code that the NixOS and Home Manager modules share +# +# `services.hermes-agent` is the same option set on both modules. Both modules +# get their options, their renderers for config.yaml, .env and documents, and +# their state setup from this file. A NixOS example works on Home Manager +# without a change. An option added here appears on both modules at once. +# +# Each module keeps only the parts that belong to its own scope: +# +# nixosModules.nix the service user and group, stateDir, +# addToSystemPackages, container mode, tmpfiles, +# system.activationScripts, system systemd units +# homeManagerModules.nix hermesHome, installPackage, home.activation, +# systemd.user.services, launchd.agents +# +# The split is by scope, not by feature. Code that needs root or a system +# identity stays in the NixOS module. All other code is here. +{ lib }: + +let + inherit (lib) + literalExpression + mkOption + types + ; + + # ── Configuration type ────────────────────────────────────────────────── + # More than one module can set `settings = { ... }`. recursiveUpdate joins + # all of the definitions. Without it, only the last definition applies. + deepConfigType = types.mkOptionType { + name = "hermes-config-attrs"; + description = "Hermes YAML config (attrset), merged deeply via lib.recursiveUpdate."; + check = builtins.isAttrs; + merge = _loc: defs: lib.foldl' lib.recursiveUpdate { } (map (d: d.value) defs); + }; + + # ── MCP server submodule ──────────────────────────────────────────────── + mcpServerType = types.submodule { + options = { + # Stdio transport + command = mkOption { + type = types.nullOr types.str; + default = null; + description = "MCP server command (stdio transport)."; + }; + args = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Command-line arguments (stdio transport)."; + }; + env = mkOption { + type = types.attrsOf types.str; + default = { }; + description = "Environment variables for the server process (stdio transport)."; + }; + + # HTTP/StreamableHTTP transport + url = mkOption { + type = types.nullOr types.str; + default = null; + description = "MCP server endpoint URL (HTTP/StreamableHTTP transport)."; + }; + headers = mkOption { + type = types.attrsOf types.str; + default = { }; + description = "HTTP headers, e.g. for authentication (HTTP transport)."; + }; + + # Authentication + auth = mkOption { + type = types.nullOr (types.enum [ "oauth" ]); + default = null; + description = '' + Authentication method. Set to "oauth" for OAuth 2.1 PKCE flow + (remote MCP servers). Tokens are stored in $HERMES_HOME/mcp-tokens/. + ''; + }; + + # Enable/disable + enabled = mkOption { + type = types.bool; + default = true; + description = "Enable or disable this MCP server."; + }; + + # Common options + timeout = mkOption { + type = types.nullOr types.int; + default = null; + description = "Tool call timeout in seconds (default: 120)."; + }; + connect_timeout = mkOption { + type = types.nullOr types.int; + default = null; + description = "Initial connection timeout in seconds (default: 60)."; + }; + + # Tool filtering + tools = mkOption { + type = types.nullOr ( + types.submodule { + options = { + include = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Tool allowlist — only these tools are registered."; + }; + exclude = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Tool blocklist — these tools are hidden."; + }; + }; + } + ); + default = null; + description = "Filter which tools are exposed by this server."; + }; + + # Sampling (server-initiated LLM requests) + sampling = mkOption { + type = types.nullOr ( + types.submodule { + options = { + enabled = mkOption { + type = types.bool; + default = true; + description = "Enable sampling."; + }; + model = mkOption { + type = types.nullOr types.str; + default = null; + description = "Override model for sampling requests."; + }; + max_tokens_cap = mkOption { + type = types.nullOr types.int; + default = null; + description = "Max tokens per request."; + }; + timeout = mkOption { + type = types.nullOr types.int; + default = null; + description = "LLM call timeout in seconds."; + }; + max_rpm = mkOption { + type = types.nullOr types.int; + default = null; + description = "Max requests per minute."; + }; + max_tool_rounds = mkOption { + type = types.nullOr types.int; + default = null; + description = "Max tool-use rounds per sampling request."; + }; + allowed_models = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Models the server is allowed to request."; + }; + log_level = mkOption { + type = types.nullOr ( + types.enum [ + "debug" + "info" + "warning" + ] + ); + default = null; + description = "Audit log level for sampling requests."; + }; + }; + } + ); + default = null; + description = "Sampling configuration for server-initiated LLM requests."; + }; + }; + }; + + # Convert the mcpServers submodules into the shape that config.yaml uses. + mcpServersToConfig = + servers: + lib.mapAttrs ( + _name: srv: + # Stdio transport + lib.optionalAttrs (srv.command != null) { inherit (srv) command args; } + // lib.optionalAttrs (srv.env != { }) { inherit (srv) env; } + # HTTP transport + // lib.optionalAttrs (srv.url != null) { inherit (srv) url; } + // lib.optionalAttrs (srv.headers != { }) { inherit (srv) headers; } + # Auth + // lib.optionalAttrs (srv.auth != null) { inherit (srv) auth; } + # Enable/disable + // { + inherit (srv) enabled; + } + # Common options + // lib.optionalAttrs (srv.timeout != null) { inherit (srv) timeout; } + // lib.optionalAttrs (srv.connect_timeout != null) { inherit (srv) connect_timeout; } + # Tool filtering + // lib.optionalAttrs (srv.tools != null) { + tools = lib.filterAttrs (_: v: v != [ ]) { + inherit (srv.tools) include exclude; + }; + } + # Sampling + // lib.optionalAttrs (srv.sampling != null) { + sampling = lib.filterAttrs (_: v: v != null && v != [ ]) { + inherit (srv.sampling) + enabled + model + max_tokens_cap + timeout + max_rpm + max_tool_rounds + allowed_models + log_level + ; + }; + } + ) servers; + + documentsType = types.attrsOf (types.either types.str types.path); + + # ── The options that both modules share ───────────────────────────────── + # `defaultPackage` and `defaultWorkingDirectory` are different on each + # module, so the caller gives them. All other options are the same. + sharedOptions = + { + defaultPackage, + defaultPackageText, + defaultWorkingDirectory, + defaultWorkingDirectoryText, + }: + { + enable = lib.mkEnableOption "Hermes Agent"; + + # ── Package ──────────────────────────────────────────────────────── + package = mkOption { + type = types.package; + default = defaultPackage; + defaultText = defaultPackageText; + description = "The hermes-agent package to use."; + }; + + workingDirectory = mkOption { + type = types.str; + default = defaultWorkingDirectory; + defaultText = defaultWorkingDirectoryText; + description = '' + The working directory for the agent. The module also writes this + path to config.yaml as `terminal.cwd`. The terminal and file tools + of the agent use that value. + ''; + }; + + # ── Declarative config ───────────────────────────────────────────── + configFile = mkOption { + type = types.nullOr types.path; + default = null; + description = '' + The path to an existing config.yaml. If you set this option, it + replaces the `settings` option. The module installs the file + without a change and overwrites all runtime edits on each + activation. + ''; + }; + + settings = mkOption { + type = deepConfigType; + default = { }; + description = '' + The Hermes configuration, as an attribute set. The module joins the + definitions from all modules and writes the result to config.yaml. + + The merge into the config.yaml on disk is also a deep merge. These + keys replace the keys on disk. The module keeps all other keys, + which includes the keys that `hermes config set` and the settings + panes of the TUI and the desktop app write at runtime. + ''; + example = literalExpression '' + { + model.default = "anthropic/claude-sonnet-4"; + terminal.backend = "local"; + compression = { enabled = true; threshold = 0.85; }; + } + ''; + }; + + # ── Secrets / environment ────────────────────────────────────────── + environmentFiles = mkOption { + # The type is `str` and not `path` for a reason. A Nix path literal + # copies the secret into the Nix store, which all users can read. Use + # a runtime path from sops-nix or agenix instead, for example + # `config.sops.secrets."x".path`. + type = types.listOf types.str; + default = [ ]; + description = '' + The paths to environment files that contain secrets, for example + API keys and tokens. Activation adds the contents of these files to + $HERMES_HOME/.env. Hermes reads that file at each start, with + load_hermes_dotenv(). + + Each activation writes .env again from the start. Thus a secret + file cannot go into .env two times. + ''; + example = literalExpression ''[ config.sops.secrets."hermes/env".path ]''; + }; + + environment = mkOption { + type = types.attrsOf types.str; + default = { }; + description = '' + Environment variables that are not secret. Activation writes them + to $HERMES_HOME/.env. + + CAUTION: Do not put secrets in this option. All users can read the + Nix store. Use environmentFiles for secrets. + ''; + }; + + authFile = mkOption { + type = types.nullOr types.path; + default = null; + description = '' + The path to a file that gives the first contents of auth.json, the + OAuth credentials. The module copies the file only when auth.json + does not exist. Thus a token that Hermes refreshes at runtime stays + after an activation. + ''; + }; + + authFileForceOverwrite = mkOption { + type = types.bool; + default = false; + description = "Always overwrite auth.json from authFile on activation."; + }; + + # ── Documents ────────────────────────────────────────────────────── + documents = mkOption { + type = documentsType; + default = { }; + description = '' + Workspace files. The module installs them into workingDirectory. + Each key is a path relative to that directory, and the module makes + the necessary subdirectories. Each value is a string or a path. + + Use this option for the project context that the agent reads from + its working directory, for example AGENTS.md, notes and checklists. + Hermes reads SOUL.md and memories/ from HERMES_HOME, so put those + files in `hermesHomeFiles`. + + If you set this option, you must also set `workingDirectory`. The + default of that option is different on each module. Thus an unset + default puts these files in a directory that you did not select. + ''; + example = literalExpression '' + { + "AGENTS.md" = ./AGENTS.md; + "notes/oncall.md" = "Page #infra before restarting anything."; + } + ''; + }; + + hermesHomeFiles = mkOption { + type = documentsType; + default = { }; + description = '' + Files that the module installs into HERMES_HOME. Each key is a path + relative to that directory, and the module makes the necessary + subdirectories. Each value is a string or a path. + + Hermes reads SOUL.md and the memory files from HERMES_HOME and not + from the working directory. Declare those files here, or Hermes + does not load them. + ''; + example = literalExpression '' + { + "SOUL.md" = "You are a helpful AI assistant."; + "memories/USER.md" = ./USER.md; + } + ''; + }; + + # ── MCP Servers ──────────────────────────────────────────────────── + mcpServers = mkOption { + type = types.attrsOf mcpServerType; + default = { }; + description = '' + MCP server configurations (merged into settings.mcp_servers). + Each server uses either stdio (command/args) or HTTP (url) transport. + ''; + example = literalExpression '' + { + filesystem = { + command = "npx"; + args = [ "-y" "@modelcontextprotocol/server-filesystem" "/home/user" ]; + }; + remote-api = { + url = "http://my-server:8080/v0/mcp"; + headers = { Authorization = "Bearer ..."; }; + }; + remote-oauth = { + url = "https://mcp.example.com/mcp"; + auth = "oauth"; + }; + } + ''; + }; + + # ── Packages / plugins ───────────────────────────────────────────── + extraPackages = mkOption { + type = types.listOf types.package; + default = [ ]; + description = "More packages on the PATH of the agent. The agent can run these tools."; + }; + + extraPlugins = mkOption { + type = types.listOf types.package; + default = [ ]; + description = '' + Directory-based plugin packages to symlink into the hermes plugins + directory. Each package must contain a plugin.yaml and __init__.py + at its root. Hermes discovers these automatically on startup. + ''; + example = literalExpression '' + [ + (pkgs.fetchFromGitHub { + owner = "stephenschoettler"; + repo = "hermes-lcm"; + name = "hermes-lcm"; + rev = "v0.7.0"; + hash = "sha256-..."; + }) + ] + ''; + }; + + extraPythonPackages = mkOption { + type = types.listOf types.package; + default = [ ]; + description = '' + Python packages to add to PYTHONPATH for entry-point plugin discovery. + These are pip-packaged plugins that register via the + hermes_agent.plugins entry-point group. Each package must be built + with the same Python interpreter as hermes (python312). + ''; + example = literalExpression '' + [ + (pkgs.python312Packages.buildPythonPackage { + pname = "rtk-hermes"; + version = "1.0.0"; + src = pkgs.fetchFromGitHub { + owner = "ogallotti"; + repo = "rtk-hermes"; + rev = "main"; + hash = "sha256-..."; + }; + }) + ] + ''; + }; + + extraDependencyGroups = mkOption { + type = types.listOf types.str; + default = [ ]; + description = '' + Additional pyproject.toml optional-dependency groups to include in + the sealed Python venv. These are resolved by uv alongside core + dependencies — no PYTHONPATH patching or collision risk. + + Use this for optional extras already declared in hermes-agent's + pyproject.toml (e.g. "hindsight", "honcho", "voice"). + Use extraPythonPackages for external packages not in pyproject.toml. + ''; + example = [ "hindsight" ]; + }; + + # ── Service behaviour ────────────────────────────────────────────── + extraArgs = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Extra command-line arguments for `hermes gateway`."; + }; + + restart = mkOption { + type = types.str; + default = "always"; + description = "The systemd Restart= policy. Darwin does not use this option."; + }; + + restartSec = mkOption { + type = types.int; + default = 5; + description = "The systemd RestartSec= value. Darwin does not use this option."; + }; + + # ── The backend: `hermes serve` or `hermes dashboard` ────────────── + # `hermes serve` and `hermes dashboard` are the same entry point, + # hermes_cli.main:cmd_dashboard, with one flag of difference. serve runs + # without a user interface. dashboard also serves the web application. + # Both give the /api/ws and /api/pty sockets that Hermes Desktop + # connects to. They are one process, and you can run only one of them. + # Thus this option is an enum and not two booleans. + # + # The backend does not run the messaging gateway. web_server.py only + # controls an external gateway, with `hermes gateway restart`. It does + # not contain a gateway. + backend = { + mode = mkOption { + type = types.enum [ + "none" + "serve" + "dashboard" + ]; + default = "none"; + description = '' + The backend process to run with the messaging gateway. + + - "none" — no backend + - "serve" — the backend without a user interface. It gives + the /api/ws and /api/pty sockets that Hermes + Desktop connects to. + - "dashboard" — all that "serve" gives, and the browser admin + panel on the same port + + "dashboard" contains all of "serve". + ''; + }; + + host = mkOption { + type = types.str; + default = "127.0.0.1"; + description = '' + The address that the backend binds to. + + An address other than loopback starts the authentication gate of + the dashboard. You must then configure credentials, or a client + cannot connect. The server also refuses each request with a Host + header that is different from the address that the server bound + to. This is a defence against DNS rebinding. Bind to the name or + the address that your clients use. + ''; + }; + + port = mkOption { + type = types.port; + default = 9119; + description = "The port for the backend."; + }; + + extraArgs = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "More command-line arguments for the backend command."; + }; + }; + }; + + # ── Package resolution ────────────────────────────────────────────────── + effectivePackage = + cfg: + if cfg.extraPythonPackages == [ ] && cfg.extraDependencyGroups == [ ] then + cfg.package + else + cfg.package.override { inherit (cfg) extraPythonPackages extraDependencyGroups; }; + + # ── The rendered config.yaml ──────────────────────────────────────────── + # YAML contains JSON, so the output of toJSON is a correct config.yaml. + # terminal.cwd replaces the old MESSAGING_CWD environment variable. The + # order of the recursiveUpdate lets an explicit settings.terminal.cwd + # replace the default value. + mkConfigFiles = + { + pkgs, + cfg, + workingDirectory, + }: + let + generated = pkgs.writeText "hermes-config.yaml" ( + builtins.toJSON (lib.recursiveUpdate { terminal.cwd = workingDirectory; } cfg.settings) + ); + in + { + inherit generated; + effective = if cfg.configFile != null then cfg.configFile else generated; + mergeScript = pkgs.callPackage ./configMergeScript.nix { }; + }; + + # ── Documents ─────────────────────────────────────────────────────────── + # A key can contain subdirectories. The tree has the same shape, so the + # install loop can copy each entry with `install -D`. + mkDocumentTree = + { pkgs, documents }: + pkgs.runCommand "hermes-documents" { } ( + '' + mkdir -p $out + '' + + lib.concatStringsSep "\n" ( + lib.mapAttrsToList ( + name: value: + let + dir = builtins.dirOf name; + mkdir = lib.optionalString (dir != ".") "mkdir -p $out/${dir}"; + in + if builtins.isPath value || lib.isStorePath value then + "${mkdir}\ncp ${value} $out/${name}" + else + "${mkdir}\ncat > $out/${name} <<'HERMES_DOC_EOF'\n${value}\nHERMES_DOC_EOF" + ) documents + ) + ); + + # ── How .env is built ─────────────────────────────────────────────────── + # The values that are not secret come from the Nix store. Activation adds + # the secrets from paths outside the store. This is one command, so it is + # safe in a dry run. A second activation writes .env again and does not add + # the same secrets a second time. + mkEnvScript = + { pkgs, environment }: + let + base = pkgs.writeText "hermes-env-base" ( + lib.concatStringsSep "\n" (lib.mapAttrsToList (k: v: "${k}=${v}") environment) + + lib.optionalString (environment != { }) "\n" + ); + in + pkgs.writeShellScript "hermes-env-merge" '' + set -eu + + dest="$1" + mode="$2" + shift 2 + + install -m "$mode" ${base} "$dest" + for file in "$@"; do + if [ -r "$file" ]; then + printf '\n' >> "$dest" + cat "$file" >> "$dest" + else + echo "hermes-agent: WARNING cannot read environmentFile $file" >&2 + fi + done + ''; + + # ── State setup ───────────────────────────────────────────────────────── + # The activation code that both modules run. It makes the directories and + # installs config.yaml, .env, auth.json, the documents and the plugins. The + # differences between the two modules are only the install flags for the + # owner and the file modes. Thus they are arguments, and not a second copy + # of the script. + # + # run the command prefix ("" on NixOS, "$DRY_RUN_CMD " on + # Home Manager) + # owner "user:group" that owns each file, or null for the user that + # runs the activation + # modes the file mode for each kind of file + mkStateScript = + { + pkgs, + cfg, + hermesHome, + workingDirectory, + # The value to write as terminal.cwd. It is different from + # workingDirectory only in the container mode of NixOS. There the agent + # sees the directory at its mount point in the container, but + # activation writes to the path on the host. + configWorkingDirectory ? workingDirectory, + run ? "", + owner ? null, + modes, + stateDirs ? [ ], + # The module writes this value into the .managed marker. An + # interactive shell reads the marker, because it does not see the + # HERMES_MANAGED variable of the service. The value tells the shell + # which system owns the install and which rebuild command to name. + managedSystem ? "nixos", + }: + let + installFlags = lib.optionalString (owner != null) ( + let + parts = lib.splitString ":" owner; + in + "-o ${lib.head parts} -g ${lib.last parts}" + ); + configFiles = mkConfigFiles { + inherit pkgs cfg; + workingDirectory = configWorkingDirectory; + }; + envScript = mkEnvScript { + inherit pkgs; + inherit (cfg) environment; + }; + documentTree = mkDocumentTree { + inherit pkgs; + inherit (cfg) documents; + }; + homeDocumentTree = mkDocumentTree { + inherit pkgs; + documents = cfg.hermesHomeFiles; + }; + + inst = "${run}install ${installFlags}"; + + installDocuments = + tree: root: docs: + lib.concatStringsSep "\n" ( + lib.mapAttrsToList ( + name: _value: "${inst} -m ${modes.document} -D ${tree}/${name} ${root}/${name}" + ) docs + ); + in + '' + # Directories. The service units and Hermes make most of these + # directories when they first need them. Activation makes them here so + # that the first activation sets the correct owner and mode, and does + # not use the umask. + ${run}mkdir -p ${ + lib.escapeShellArgs ( + [ + hermesHome + workingDirectory + ] + ++ map (d: "${hermesHome}/${d}") stateDirs + ) + } + + # config.yaml: merge the Nix settings into the file on disk. Hermes + # writes this file at runtime. A read-only symlink to the Nix store + # breaks each save from the application. The Nix keys replace the keys + # on disk, and the module keeps all other keys. + ${ + if cfg.configFile != null then + "${inst} -m ${modes.config} -D ${configFiles.effective} ${hermesHome}/config.yaml" + else + '' + ${run}${configFiles.mergeScript} ${configFiles.generated} ${hermesHome}/config.yaml + ${run}chmod ${modes.config} ${hermesHome}/config.yaml + '' + } + + # The managed-mode marker. It makes an interactive shell also refuse to + # change the configuration that Nix owns. + ${inst} -m ${modes.managed} ${pkgs.writeText "hermes-managed" managedSystem} ${hermesHome}/.managed + + ${lib.optionalString (cfg.environment != { } || cfg.environmentFiles != [ ]) '' + ${run}${envScript} ${hermesHome}/.env ${modes.env} ${lib.escapeShellArgs cfg.environmentFiles} + ${lib.optionalString (owner != null) "${run}chown ${owner} ${hermesHome}/.env"} + ''} + + ${lib.optionalString (cfg.authFile != null) ( + if cfg.authFileForceOverwrite then + "${inst} -m ${modes.auth} ${cfg.authFile} ${hermesHome}/auth.json" + else + '' + if [ ! -e ${hermesHome}/auth.json ]; then + ${inst} -m ${modes.auth} ${cfg.authFile} ${hermesHome}/auth.json + fi + '' + )} + + ${installDocuments documentTree workingDirectory cfg.documents} + ${installDocuments homeDocumentTree hermesHome cfg.hermesHomeFiles} + + # Declarative plugins. Activation first deletes the old managed + # symlinks. Thus a plugin that you remove from the configuration also + # goes away from the plugins directory. + ${run}find ${hermesHome}/plugins -maxdepth 1 -type l -name 'nix-managed-*' -delete 2>/dev/null || true + ${lib.concatMapStringsSep "\n" (plugin: '' + if [ ! -f ${plugin}/plugin.yaml ]; then + echo "hermes-agent: ERROR extraPlugins entry '${plugin}' has no plugin.yaml" >&2 + exit 1 + fi + ${run}ln -sfn ${plugin} ${hermesHome}/plugins/nix-managed-${lib.getName plugin} + '') cfg.extraPlugins} + ''; + + # ── Process argv ──────────────────────────────────────────────────────── + gatewayArgv = + cfg: + [ + "${effectivePackage cfg}/bin/hermes" + "gateway" + ] + ++ cfg.extraArgs; + + backendArgv = + cfg: + [ + "${effectivePackage cfg}/bin/hermes" + cfg.backend.mode + "--host" + cfg.backend.host + "--port" + (toString cfg.backend.port) + # CAUTION: A service must not try to open a browser when it starts. + "--no-open" + ] + ++ cfg.backend.extraArgs; + + backendDescription = + cfg: + if cfg.backend.mode == "dashboard" then + "Hermes Agent web dashboard and desktop backend" + else + "Hermes Agent backend for Hermes Desktop"; + + # The environment that each Hermes process needs, from either module. + # + # managedSystem gives the value of HERMES_MANAGED. The CLI reads that + # variable to refuse a configuration change that it cannot keep, and to + # name the correct rebuild command. The answer is different on each module, + # so each module gives its own value. + processEnvironment = + { + hermesHome, + managedSystem ? "true", + }: + { + HERMES_HOME = hermesHome; + HERMES_MANAGED = managedSystem; + }; + + processPath = + { pkgs, cfg }: + [ + (effectivePackage cfg) + pkgs.bash + pkgs.coreutils + pkgs.git + ] + ++ cfg.extraPackages; + + # workingDirectory has a default on both modules, but a bad one. It is the + # home directory of the user on Home Manager, and ${stateDir}/workspace on + # NixOS. A user who declares files without a directory therefore gets a + # place that the user did not select. The place is also different on each + # module. The modules refuse that combination. + # + # The test is on the priority of the option and not on its value. An option + # that nothing sets keeps the priority of its own default, and each + # definition from a user is stronger. Thus a directory that has the same + # text as the default is still a selection, and so is a mkDefault. A + # comparison of values detects neither. + workspaceFilesAssertions = + { + cfg, + opt, + optionPath, + }: + let + untouched = (lib.mkOptionDefault null).priority; # 1500, derived not spelled + in + [ + { + assertion = cfg.documents == { } || opt.highestPrio < untouched; + message = '' + ${optionPath}.documents needs an explicit ${optionPath}.workingDirectory. + + The files go into workingDirectory. The default of that option is + different on each module, so an unset default puts the files in a + directory that you did not select. Set the directory: + + ${optionPath}.workingDirectory = "/path/you/want"; + + To give Hermes an identity and a memory, use + ${optionPath}.hermesHomeFiles instead. Those files go to + HERMES_HOME. Hermes reads SOUL.md and memories/ only from there. + ''; + } + ]; + + # Two plugins with the same name use one nix-managed- symlink. One of + # the plugins then disappears without a message. Both modules assert + # against this condition. + pluginNameAssertions = + { cfg, optionPath }: + let + names = map lib.getName cfg.extraPlugins; + in + [ + { + assertion = (lib.length names) == (lib.length (lib.unique names)); + message = "${optionPath}.extraPlugins: duplicate plugin names detected: ${toString names}. If using fetchFromGitHub, set name = \"plugin-name\" to disambiguate."; + } + ]; + + # The subdirectories of HERMES_HOME that both modules make. + stateSubdirs = [ + "cron" + "sessions" + "logs" + "memories" + "plugins" + ]; +in +{ + inherit + backendArgv + backendDescription + deepConfigType + effectivePackage + gatewayArgv + mcpServerType + mcpServersToConfig + mkConfigFiles + mkDocumentTree + mkEnvScript + mkStateScript + pluginNameAssertions + processEnvironment + processPath + sharedOptions + stateSubdirs + workspaceFilesAssertions + ; +} diff --git a/nix/nixosModules.nix b/nix/nixosModules.nix index df74427c48..38af2ec5ed 100644 --- a/nix/nixosModules.nix +++ b/nix/nixosModules.nix @@ -1,4 +1,10 @@ -# nix/nixosModules.nix — NixOS module for hermes-agent +# nix/nixosModules.nix — the NixOS module for hermes-agent +# +# This module shares its options, its renderers for config.yaml, .env and +# documents, and its state setup with the Home Manager module +# (nix/homeManagerModules.nix). The shared code is in nix/moduleCommon.nix. +# This file holds only the parts that need root: the service user, a system +# state directory, the system PATH, and container mode. # # Two modes: # container.enable = false (default) → native systemd service @@ -19,990 +25,642 @@ # Usage: # services.hermes-agent = { # enable = true; -# settings.model = "anthropic/claude-sonnet-4"; +# settings.model.default = "anthropic/claude-sonnet-4"; # environmentFiles = [ config.sops.secrets."hermes/env".path ]; # }; # -{ inputs, ... }: { - flake.nixosModules.default = { config, lib, pkgs, ... }: +{ inputs, ... }: +{ + flake.nixosModules.default = + { + config, + lib, + options, + pkgs, + ... + }: - let - cfg = config.services.hermes-agent; - effectivePackage = - if cfg.extraPythonPackages == [ ] && cfg.extraDependencyGroups == [ ] - then cfg.package - else cfg.package.override { inherit (cfg) extraPythonPackages extraDependencyGroups; }; - hermes-agent = inputs.self.packages.${pkgs.stdenv.hostPlatform.system}.default; + let + cfg = config.services.hermes-agent; + common = import ./moduleCommon.nix { inherit lib; }; - # Deep-merge config type (from 0xrsydn/nix-hermes-agent) - deepConfigType = lib.types.mkOptionType { - name = "hermes-config-attrs"; - description = "Hermes YAML config (attrset), merged deeply via lib.recursiveUpdate."; - check = builtins.isAttrs; - merge = _loc: defs: lib.foldl' lib.recursiveUpdate { } (map (d: d.value) defs); - }; + effectivePackage = common.effectivePackage cfg; + hermes-agent = inputs.self.packages.${pkgs.stdenv.hostPlatform.system}.default; - # Generate config.yaml from Nix attrset (YAML is a superset of JSON). - # terminal.cwd replaces the deprecated MESSAGING_CWD env var — hermes - # reads it from config.yaml and bridges it to TERMINAL_CWD internally. - # recursiveUpdate: cfg.settings wins, so an explicit - # settings.terminal.cwd overrides the workingDirectory default. - # Container mode uses the in-container mount path. - effectiveWorkDir = if cfg.container.enable then containerWorkDir else cfg.workingDirectory; - configJson = builtins.toJSON ( - lib.recursiveUpdate { terminal.cwd = effectiveWorkDir; } cfg.settings - ); - generatedConfigFile = pkgs.writeText "hermes-config.yaml" configJson; - configFile = if cfg.configFile != null then cfg.configFile else generatedConfigFile; + hermesHome = "${cfg.stateDir}/.hermes"; - configMergeScript = pkgs.callPackage ./configMergeScript.nix { }; + # In container mode, the agent uses the mount path in the container. + effectiveWorkDir = if cfg.container.enable then containerWorkDir else cfg.workingDirectory; - # config.yaml mode: group-writable (0660) when interactive users share this - # HERMES_HOME via addToSystemPackages, so they can save settings through the - # CLI/TUI without hitting EACCES; otherwise group-read-only (0640). Secrets - # (.env) stay 0640 regardless — see below. - configYamlMode = if cfg.addToSystemPackages then "0660" else "0640"; + # config.yaml mode: group-writable (0660) when interactive users share this + # HERMES_HOME via addToSystemPackages, so they can save settings through the + # CLI/TUI without hitting EACCES; otherwise group-read-only (0640). Secrets + # (.env) stay 0640 regardless. + configYamlMode = if cfg.addToSystemPackages then "0660" else "0640"; - # Generate .env from non-secret environment attrset - envFileContent = lib.concatStringsSep "\n" ( - lib.mapAttrsToList (k: v: "${k}=${v}") cfg.environment - ); - # Build documents derivation (from 0xrsydn) - documentDerivation = pkgs.runCommand "hermes-documents" { } ( - '' - mkdir -p $out - '' + lib.concatStringsSep "\n" ( - lib.mapAttrsToList (name: value: - if builtins.isPath value || lib.isStorePath value - then "cp ${value} $out/${name}" - else "cat > $out/${name} <<'HERMES_DOC_EOF'\n${value}\nHERMES_DOC_EOF" - ) cfg.documents - ) - ); + containerName = "hermes-agent"; + containerDataDir = "/data"; # stateDir mount point inside container + containerHomeDir = "/home/hermes"; - containerName = "hermes-agent"; - containerDataDir = "/data"; # stateDir mount point inside container - containerHomeDir = "/home/hermes"; + # ── Container mode helpers ────────────────────────────────────────── + containerBin = + if cfg.container.backend == "docker" then + "${pkgs.docker}/bin/docker" + else + "${pkgs.podman}/bin/podman"; - # ── Container mode helpers ────────────────────────────────────────── - containerBin = if cfg.container.backend == "docker" - then "${pkgs.docker}/bin/docker" - else "${pkgs.podman}/bin/podman"; + # Runs as root inside the container on every start. Provisions the + # hermes user + sudo on first boot (writable layer persists), then + # drops privileges. Supports arbitrary base images (Debian, Alpine, etc). + containerEntrypoint = pkgs.writeShellScript "hermes-container-entrypoint" '' + set -eu - # Runs as root inside the container on every start. Provisions the - # hermes user + sudo on first boot (writable layer persists), then - # drops privileges. Supports arbitrary base images (Debian, Alpine, etc). - containerEntrypoint = pkgs.writeShellScript "hermes-container-entrypoint" '' - set -eu + HERMES_UID="''${HERMES_UID:?HERMES_UID must be set}" + HERMES_GID="''${HERMES_GID:?HERMES_GID must be set}" - HERMES_UID="''${HERMES_UID:?HERMES_UID must be set}" - HERMES_GID="''${HERMES_GID:?HERMES_GID must be set}" - - # ── Group: ensure a group with GID=$HERMES_GID exists ── - # Check by GID (not name) to avoid collisions with pre-existing groups - # (e.g. GID 100 = "users" on Ubuntu) - EXISTING_GROUP=$(getent group "$HERMES_GID" 2>/dev/null | cut -d: -f1 || true) - if [ -n "$EXISTING_GROUP" ]; then - GROUP_NAME="$EXISTING_GROUP" - else - GROUP_NAME="hermes" - if command -v groupadd >/dev/null 2>&1; then - groupadd -g "$HERMES_GID" "$GROUP_NAME" - elif command -v addgroup >/dev/null 2>&1; then - addgroup -g "$HERMES_GID" "$GROUP_NAME" 2>/dev/null || true + # ── Group: ensure a group with GID=$HERMES_GID exists ── + # Check by GID (not name) to avoid collisions with pre-existing groups + # (e.g. GID 100 = "users" on Ubuntu) + EXISTING_GROUP=$(getent group "$HERMES_GID" 2>/dev/null | cut -d: -f1 || true) + if [ -n "$EXISTING_GROUP" ]; then + GROUP_NAME="$EXISTING_GROUP" + else + GROUP_NAME="hermes" + if command -v groupadd >/dev/null 2>&1; then + groupadd -g "$HERMES_GID" "$GROUP_NAME" + elif command -v addgroup >/dev/null 2>&1; then + addgroup -g "$HERMES_GID" "$GROUP_NAME" 2>/dev/null || true + fi fi - fi - # ── User: ensure a user with UID=$HERMES_UID exists ── - PASSWD_ENTRY=$(getent passwd "$HERMES_UID" 2>/dev/null || true) - if [ -n "$PASSWD_ENTRY" ]; then - TARGET_USER=$(echo "$PASSWD_ENTRY" | cut -d: -f1) - TARGET_HOME=$(echo "$PASSWD_ENTRY" | cut -d: -f6) - else - TARGET_USER="hermes" - TARGET_HOME="/home/hermes" - if command -v useradd >/dev/null 2>&1; then - useradd -u "$HERMES_UID" -g "$HERMES_GID" -m -d "$TARGET_HOME" -s /bin/bash "$TARGET_USER" - elif command -v adduser >/dev/null 2>&1; then - adduser -u "$HERMES_UID" -D -h "$TARGET_HOME" -s /bin/sh -G "$GROUP_NAME" "$TARGET_USER" 2>/dev/null || true + # ── User: ensure a user with UID=$HERMES_UID exists ── + PASSWD_ENTRY=$(getent passwd "$HERMES_UID" 2>/dev/null || true) + if [ -n "$PASSWD_ENTRY" ]; then + TARGET_USER=$(echo "$PASSWD_ENTRY" | cut -d: -f1) + TARGET_HOME=$(echo "$PASSWD_ENTRY" | cut -d: -f6) + else + TARGET_USER="hermes" + TARGET_HOME="/home/hermes" + if command -v useradd >/dev/null 2>&1; then + useradd -u "$HERMES_UID" -g "$HERMES_GID" -m -d "$TARGET_HOME" -s /bin/bash "$TARGET_USER" + elif command -v adduser >/dev/null 2>&1; then + adduser -u "$HERMES_UID" -D -h "$TARGET_HOME" -s /bin/sh -G "$GROUP_NAME" "$TARGET_USER" 2>/dev/null || true + fi fi - fi - mkdir -p "$TARGET_HOME" - chown "$HERMES_UID:$HERMES_GID" "$TARGET_HOME" - chmod 0750 "$TARGET_HOME" + mkdir -p "$TARGET_HOME" + chown "$HERMES_UID:$HERMES_GID" "$TARGET_HOME" + chmod 0750 "$TARGET_HOME" - # Ensure HERMES_HOME is owned by the target user. - # Use find instead of chown -R: chown strips the setgid bit (kernel - # behavior), destroying the 2770 permissions the NixOS activation - # script sets for group access by hostUsers. Only touch files with - # wrong ownership so correctly-owned dirs keep their permission bits. - if [ -n "''${HERMES_HOME:-}" ] && [ -d "$HERMES_HOME" ]; then - find "$HERMES_HOME" \! -user "$HERMES_UID" -exec chown "$HERMES_UID:$HERMES_GID" {} + - fi + # Ensure HERMES_HOME is owned by the target user. + # Use find instead of chown -R: chown strips the setgid bit (kernel + # behavior), destroying the 2770 permissions the NixOS activation + # script sets for group access by hostUsers. Only touch files with + # wrong ownership so correctly-owned dirs keep their permission bits. + if [ -n "''${HERMES_HOME:-}" ] && [ -d "$HERMES_HOME" ]; then + find "$HERMES_HOME" \! -user "$HERMES_UID" -exec chown "$HERMES_UID:$HERMES_GID" {} + + fi - # ── Provision apt packages (first boot only, cached in writable layer) ── - # sudo: agent self-modification - # nodejs/npm: writable node so npm i -g works (nix store copies are read-only) - # Node 22 via NodeSource — Ubuntu 24.04 ships Node 18 which is EOL. - # curl: needed for uv installer + NodeSource setup - if [ ! -f /var/lib/hermes-tools-provisioned ] && command -v apt-get >/dev/null 2>&1; then - echo "First boot: provisioning agent tools..." - apt-get update -qq - apt-get install -y -qq sudo curl ca-certificates gnupg - mkdir -p /etc/apt/keyrings - curl -fsSL https://deb.nodesource.com/gpgkey/nodesource-repo.gpg.key \ - | gpg --dearmor -o /etc/apt/keyrings/nodesource.gpg - echo "deb [signed-by=/etc/apt/keyrings/nodesource.gpg] https://deb.nodesource.com/node_22.x nodistro main" \ - > /etc/apt/sources.list.d/nodesource.list - apt-get update -qq - apt-get install -y -qq nodejs - touch /var/lib/hermes-tools-provisioned - fi + # ── Provision apt packages (first boot only, cached in writable layer) ── + # sudo: agent self-modification + # nodejs/npm: writable node so npm i -g works (nix store copies are read-only) + # Node 22 via NodeSource — Ubuntu 24.04 ships Node 18 which is EOL. + # curl: needed for uv installer + NodeSource setup + if [ ! -f /var/lib/hermes-tools-provisioned ] && command -v apt-get >/dev/null 2>&1; then + echo "First boot: provisioning agent tools..." + apt-get update -qq + apt-get install -y -qq sudo curl ca-certificates gnupg + mkdir -p /etc/apt/keyrings + curl -fsSL https://deb.nodesource.com/gpgkey/nodesource-repo.gpg.key \ + | gpg --dearmor -o /etc/apt/keyrings/nodesource.gpg + echo "deb [signed-by=/etc/apt/keyrings/nodesource.gpg] https://deb.nodesource.com/node_22.x nodistro main" \ + > /etc/apt/sources.list.d/nodesource.list + apt-get update -qq + apt-get install -y -qq nodejs + touch /var/lib/hermes-tools-provisioned + fi - if command -v sudo >/dev/null 2>&1 && [ ! -f /etc/sudoers.d/hermes ]; then - mkdir -p /etc/sudoers.d - echo "$TARGET_USER ALL=(ALL) NOPASSWD:ALL" > /etc/sudoers.d/hermes - chmod 0440 /etc/sudoers.d/hermes - fi + if command -v sudo >/dev/null 2>&1 && [ ! -f /etc/sudoers.d/hermes ]; then + mkdir -p /etc/sudoers.d + echo "$TARGET_USER ALL=(ALL) NOPASSWD:ALL" > /etc/sudoers.d/hermes + chmod 0440 /etc/sudoers.d/hermes + fi - # uv (Python manager) — not in Ubuntu repos, retry-safe outside the sentinel - if ! command -v uv >/dev/null 2>&1 && [ ! -x "$TARGET_HOME/.local/bin/uv" ] && command -v curl >/dev/null 2>&1; then - su -s /bin/sh "$TARGET_USER" -c 'curl -LsSf https://astral.sh/uv/install.sh | sh' || true - fi + # uv (Python manager) — not in Ubuntu repos, retry-safe outside the sentinel + if ! command -v uv >/dev/null 2>&1 && [ ! -x "$TARGET_HOME/.local/bin/uv" ] && command -v curl >/dev/null 2>&1; then + su -s /bin/sh "$TARGET_USER" -c 'curl -LsSf https://astral.sh/uv/install.sh | sh' || true + fi - # Python 3.12 venv — gives the agent a writable Python with pip. - # --seed includes pip/setuptools so bare `pip install` works. - _UV_BIN="$TARGET_HOME/.local/bin/uv" - if [ ! -d "$TARGET_HOME/.venv" ] && [ -x "$_UV_BIN" ]; then - su -s /bin/sh "$TARGET_USER" -c " - export PATH=\"\$HOME/.local/bin:\$PATH\" - uv python install 3.12 - uv venv --python 3.12 --seed \"\$HOME/.venv\" - " || true - fi + # Python 3.12 venv — gives the agent a writable Python with pip. + # --seed includes pip/setuptools so bare `pip install` works. + _UV_BIN="$TARGET_HOME/.local/bin/uv" + if [ ! -d "$TARGET_HOME/.venv" ] && [ -x "$_UV_BIN" ]; then + su -s /bin/sh "$TARGET_USER" -c " + export PATH=\"\$HOME/.local/bin:\$PATH\" + uv python install 3.12 + uv venv --python 3.12 --seed \"\$HOME/.venv\" + " || true + fi - # Put the agent venv first on PATH so python/pip resolve to writable copies - if [ -d "$TARGET_HOME/.venv/bin" ]; then - export PATH="$TARGET_HOME/.venv/bin:$PATH" - fi + # Put the agent venv first on PATH so python/pip resolve to writable copies + if [ -d "$TARGET_HOME/.venv/bin" ]; then + export PATH="$TARGET_HOME/.venv/bin:$PATH" + fi - if command -v setpriv >/dev/null 2>&1; then - exec setpriv --reuid="$HERMES_UID" --regid="$HERMES_GID" --init-groups "$@" - elif command -v su >/dev/null 2>&1; then - exec su -s /bin/sh "$TARGET_USER" -c 'exec "$0" "$@"' -- "$@" - else - echo "WARNING: no privilege-drop tool (setpriv/su), running as root" >&2 - exec "$@" - fi - ''; + if command -v setpriv >/dev/null 2>&1; then + exec setpriv --reuid="$HERMES_UID" --regid="$HERMES_GID" --init-groups "$@" + elif command -v su >/dev/null 2>&1; then + exec su -s /bin/sh "$TARGET_USER" -c 'exec "$0" "$@"' -- "$@" + else + echo "WARNING: no privilege-drop tool (setpriv/su), running as root" >&2 + exec "$@" + fi + ''; - # Identity hash — only recreate container when structural config changes. - # Package and entrypoint use stable symlinks (current-package, current-entrypoint) - # so they can update without recreation. Env vars go through $HERMES_HOME/.env. - containerIdentity = builtins.hashString "sha256" (builtins.toJSON { - schema = 4; # bump when identity inputs change (4: Node 18→22 via NodeSource) - image = cfg.container.image; - extraVolumes = cfg.container.extraVolumes; - extraOptions = cfg.container.extraOptions; - }); + # Identity hash — only recreate container when structural config changes. + # Package and entrypoint use stable symlinks (current-package, current-entrypoint) + # so they can update without recreation. Env vars go through $HERMES_HOME/.env. + containerIdentity = builtins.hashString "sha256" ( + builtins.toJSON { + schema = 4; # bump when identity inputs change (4: Node 18→22 via NodeSource) + image = cfg.container.image; + extraVolumes = cfg.container.extraVolumes; + extraOptions = cfg.container.extraOptions; + } + ); - identityFile = "${cfg.stateDir}/.container-identity"; + identityFile = "${cfg.stateDir}/.container-identity"; - # Default: /var/lib/hermes/workspace → /data/workspace. - # Custom paths outside stateDir pass through unchanged (user must add extraVolumes). - containerWorkDir = - if lib.hasPrefix "${cfg.stateDir}/" cfg.workingDirectory - then "${containerDataDir}/${lib.removePrefix "${cfg.stateDir}/" cfg.workingDirectory}" - else cfg.workingDirectory; + # The CLI on the host reads this file, in get_container_exec_info. The + # file tells the CLI to run in the container and not on the host. + containerModeFile = pkgs.writeText "hermes-container-mode" '' + # Written by the NixOS activation script. Do not edit manually. + backend=${cfg.container.backend} + container_name=${containerName} + exec_user=${cfg.user} + hermes_bin=${containerDataDir}/current-package/bin/hermes + ''; - in { - options.services.hermes-agent = with lib; { - enable = mkEnableOption "Hermes Agent gateway service"; + # Default: /var/lib/hermes/workspace → /data/workspace. + # Custom paths outside stateDir pass through unchanged (user must add extraVolumes). + containerWorkDir = + if lib.hasPrefix "${cfg.stateDir}/" cfg.workingDirectory then + "${containerDataDir}/${lib.removePrefix "${cfg.stateDir}/" cfg.workingDirectory}" + else + cfg.workingDirectory; - # ── Package ────────────────────────────────────────────────────────── - package = mkOption { - type = types.package; - default = hermes-agent; - description = "The hermes-agent package to use."; + # The hardening and the environment that the gateway unit and the + # backend unit share. + commonServiceConfig = { + User = cfg.user; + Group = cfg.group; + WorkingDirectory = cfg.workingDirectory; + + Restart = cfg.restart; + RestartSec = cfg.restartSec; + + # Shared-state: files created by the service should be group-writable + # so interactive users in the hermes group can read/write them. + UMask = "0007"; + + # Hardening + NoNewPrivileges = true; + ProtectSystem = "strict"; + ProtectHome = false; + ReadWritePaths = [ + cfg.stateDir + cfg.workingDirectory + ]; + PrivateTmp = true; }; - # ── Service identity ───────────────────────────────────────────────── - user = mkOption { - type = types.str; - default = "hermes"; - description = "System user running the gateway."; - }; + commonUnitEnvironment = { + HOME = cfg.stateDir; + } + // common.processEnvironment { inherit hermesHome; }; - group = mkOption { - type = types.str; - default = "hermes"; - description = "System group running the gateway."; - }; + unitPath = common.processPath { inherit pkgs cfg; }; - createUser = mkOption { - type = types.bool; - default = true; - description = "Create the user/group automatically."; - }; - - # ── Directories ────────────────────────────────────────────────────── - stateDir = mkOption { - type = types.str; - default = "/var/lib/hermes"; - description = "State directory. Contains .hermes/ subdir (HERMES_HOME)."; - }; - - workingDirectory = mkOption { - type = types.str; - default = "${cfg.stateDir}/workspace"; - defaultText = literalExpression ''"''${cfg.stateDir}/workspace"''; - description = "Working directory for the agent."; - }; - - # ── Declarative config ─────────────────────────────────────────────── - configFile = mkOption { - type = types.nullOr types.path; - default = null; - description = '' - Path to an existing config.yaml. If set, takes precedence over - the declarative `settings` option. - ''; - }; - - settings = mkOption { - type = deepConfigType; - default = { }; - description = '' - Declarative Hermes config (attrset). Deep-merged across module - definitions and rendered as config.yaml. - ''; - example = literalExpression '' + in + { + options.services.hermes-agent = + common.sharedOptions { + defaultPackage = hermes-agent; + defaultPackageText = lib.literalExpression "hermes-agent.packages.\${system}.default"; + defaultWorkingDirectory = "${cfg.stateDir}/workspace"; + defaultWorkingDirectoryText = lib.literalExpression ''"''${cfg.stateDir}/workspace"''; + } + // ( + with lib; { - model = "anthropic/claude-sonnet-4"; - terminal.backend = "local"; - compression = { enabled = true; threshold = 0.85; }; - toolsets = [ "all" ]; - } - ''; - }; - - # ── Secrets / environment ──────────────────────────────────────────── - environmentFiles = mkOption { - type = types.listOf types.str; - default = [ ]; - description = '' - Paths to environment files containing secrets (API keys, tokens). - Contents are merged into $HERMES_HOME/.env at activation time. - Hermes reads this file on every startup via load_hermes_dotenv(). - ''; - }; - - environment = mkOption { - type = types.attrsOf types.str; - default = { }; - description = '' - Non-secret environment variables. Merged into $HERMES_HOME/.env - at activation time. Do NOT put secrets here — use environmentFiles. - ''; - }; - - authFile = mkOption { - type = types.nullOr types.path; - default = null; - description = '' - Path to an auth.json seed file (OAuth credentials). - Only copied on first deploy — existing auth.json is preserved. - ''; - }; - - authFileForceOverwrite = mkOption { - type = types.bool; - default = false; - description = "Always overwrite auth.json from authFile on activation."; - }; - - # ── Documents ──────────────────────────────────────────────────────── - documents = mkOption { - type = types.attrsOf (types.either types.str types.path); - default = { }; - description = '' - Workspace files (SOUL.md, USER.md, etc.). Keys are filenames, - values are inline strings or paths. Installed into workingDirectory. - ''; - example = literalExpression '' - { - "SOUL.md" = "You are a helpful AI assistant."; - "USER.md" = ./documents/USER.md; - } - ''; - }; - - # ── MCP Servers ────────────────────────────────────────────────────── - mcpServers = mkOption { - type = types.attrsOf (types.submodule { - options = { - # Stdio transport - command = mkOption { - type = types.nullOr types.str; - default = null; - description = "MCP server command (stdio transport)."; - }; - args = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Command-line arguments (stdio transport)."; - }; - env = mkOption { - type = types.attrsOf types.str; - default = { }; - description = "Environment variables for the server process (stdio transport)."; + # ── Service identity ─────────────────────────────────────────── + user = mkOption { + type = types.str; + default = "hermes"; + description = "System user running the gateway."; }; - # HTTP/StreamableHTTP transport - url = mkOption { - type = types.nullOr types.str; - default = null; - description = "MCP server endpoint URL (HTTP/StreamableHTTP transport)."; - }; - headers = mkOption { - type = types.attrsOf types.str; - default = { }; - description = "HTTP headers, e.g. for authentication (HTTP transport)."; + group = mkOption { + type = types.str; + default = "hermes"; + description = "System group running the gateway."; }; - # Authentication - auth = mkOption { - type = types.nullOr (types.enum [ "oauth" ]); - default = null; + createUser = mkOption { + type = types.bool; + default = true; + description = "Create the user/group automatically."; + }; + + # ── Directories ──────────────────────────────────────────────── + stateDir = mkOption { + type = types.str; + default = "/var/lib/hermes"; + description = "State directory. Contains .hermes/ subdir (HERMES_HOME)."; + }; + + addToSystemPackages = mkOption { + type = types.bool; + default = false; description = '' - Authentication method. Set to "oauth" for OAuth 2.1 PKCE flow - (remote MCP servers). Tokens are stored in $HERMES_HOME/mcp-tokens/. + Add the hermes CLI to environment.systemPackages and export + HERMES_HOME system-wide (via environment.variables) so interactive + shells share state with the gateway service. ''; }; - # Enable/disable - enabled = mkOption { - type = types.bool; - default = true; - description = "Enable or disable this MCP server."; - }; + # ── OCI Container (opt-in) ──────────────────────────────────── + container = { + enable = mkEnableOption "OCI container mode (Ubuntu base, full self-modification support)"; - # Common options - timeout = mkOption { - type = types.nullOr types.int; - default = null; - description = "Tool call timeout in seconds (default: 120)."; - }; - connect_timeout = mkOption { - type = types.nullOr types.int; - default = null; - description = "Initial connection timeout in seconds (default: 60)."; - }; - - # Tool filtering - tools = mkOption { - type = types.nullOr (types.submodule { - options = { - include = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Tool allowlist — only these tools are registered."; - }; - exclude = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Tool blocklist — these tools are hidden."; - }; - }; - }); - default = null; - description = "Filter which tools are exposed by this server."; - }; - - # Sampling (server-initiated LLM requests) - sampling = mkOption { - type = types.nullOr (types.submodule { - options = { - enabled = mkOption { type = types.bool; default = true; description = "Enable sampling."; }; - model = mkOption { type = types.nullOr types.str; default = null; description = "Override model for sampling requests."; }; - max_tokens_cap = mkOption { type = types.nullOr types.int; default = null; description = "Max tokens per request."; }; - timeout = mkOption { type = types.nullOr types.int; default = null; description = "LLM call timeout in seconds."; }; - max_rpm = mkOption { type = types.nullOr types.int; default = null; description = "Max requests per minute."; }; - max_tool_rounds = mkOption { type = types.nullOr types.int; default = null; description = "Max tool-use rounds per sampling request."; }; - allowed_models = mkOption { type = types.listOf types.str; default = [ ]; description = "Models the server is allowed to request."; }; - log_level = mkOption { - type = types.nullOr (types.enum [ "debug" "info" "warning" ]); - default = null; - description = "Audit log level for sampling requests."; - }; - }; - }); - default = null; - description = "Sampling configuration for server-initiated LLM requests."; - }; - }; - }); - default = { }; - description = '' - MCP server configurations (merged into settings.mcp_servers). - Each server uses either stdio (command/args) or HTTP (url) transport. - ''; - example = literalExpression '' - { - filesystem = { - command = "npx"; - args = [ "-y" "@modelcontextprotocol/server-filesystem" "/home/user" ]; - }; - remote-api = { - url = "http://my-server:8080/v0/mcp"; - headers = { Authorization = "Bearer ..."; }; - }; - remote-oauth = { - url = "https://mcp.example.com/mcp"; - auth = "oauth"; - }; - } - ''; - }; - - # ── Service behavior ───────────────────────────────────────────────── - extraArgs = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Extra command-line arguments for `hermes gateway`."; - }; - - extraPackages = mkOption { - type = types.listOf types.package; - default = [ ]; - description = '' - Extra packages available to the agent — terminal commands, skills, - cron jobs, and the service process all see them. - - Implemented via the hermes user's per-user profile - (`/etc/profiles/per-user/${cfg.user}/bin`), which NixOS includes - in PATH for login shells. The packages are also added to the - systemd service PATH for direct process access. - ''; - }; - - extraPlugins = mkOption { - type = types.listOf types.package; - default = [ ]; - description = '' - Directory-based plugin packages to symlink into the hermes plugins - directory. Each package should contain a plugin.yaml and __init__.py - at its root. Hermes discovers these automatically on startup. - ''; - example = literalExpression '' - [ - (pkgs.fetchFromGitHub { - owner = "stephenschoettler"; - repo = "hermes-lcm"; - name = "hermes-lcm"; - rev = "v0.7.0"; - hash = "sha256-..."; - }) - ] - ''; - }; - - extraPythonPackages = mkOption { - type = types.listOf types.package; - default = [ ]; - description = '' - Python packages to add to PYTHONPATH for entry-point plugin discovery. - These are pip-packaged plugins that register via the - hermes_agent.plugins entry-point group. Each package must be built - with the same Python interpreter as hermes (python312). - ''; - example = literalExpression '' - [ - (pkgs.python312Packages.buildPythonPackage { - pname = "rtk-hermes"; - version = "1.0.0"; - src = pkgs.fetchFromGitHub { - owner = "ogallotti"; - repo = "rtk-hermes"; - rev = "main"; - hash = "sha256-..."; + backend = mkOption { + type = types.enum [ + "docker" + "podman" + ]; + default = "docker"; + description = "Container runtime."; }; - }) - ] - ''; - }; - extraDependencyGroups = mkOption { - type = types.listOf types.str; - default = [ ]; - description = '' - Additional pyproject.toml optional-dependency groups to include in - the sealed Python venv. These are resolved by uv alongside core - dependencies — no PYTHONPATH patching or collision risk. + extraVolumes = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Extra volume mounts (host:container:mode format)."; + example = [ "/home/user/projects:/projects:rw" ]; + }; - Use this for optional extras already declared in hermes-agent's - pyproject.toml (e.g. "hindsight", "honcho", "voice"). - Use extraPythonPackages for external packages not in pyproject.toml. - ''; - example = [ "hindsight" ]; - }; + extraOptions = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Extra arguments passed to docker/podman run."; + }; - restart = mkOption { - type = types.str; - default = "always"; - description = "systemd Restart= policy."; - }; + image = mkOption { + type = types.str; + default = "ubuntu:24.04"; + description = "OCI container image. The container pulls this at runtime via Docker/Podman."; + }; - restartSec = mkOption { - type = types.int; - default = 5; - description = "systemd RestartSec= value."; - }; + hostUsers = mkOption { + type = types.listOf types.str; + default = [ ]; + description = '' + Interactive users who get a ~/.hermes symlink to the service + stateDir. These users are automatically added to the hermes group. + ''; + example = [ "sidbin" ]; + }; + }; + } + ); - addToSystemPackages = mkOption { - type = types.bool; - default = false; - description = '' - Add the hermes CLI to environment.systemPackages and export - HERMES_HOME system-wide (via environment.variables) so interactive - shells share state with the gateway service. - ''; - }; + config = lib.mkIf cfg.enable ( + lib.mkMerge [ - # ── OCI Container (opt-in) ────────────────────────────────────────── - container = { - enable = mkEnableOption "OCI container mode (Ubuntu base, full self-modification support)"; + # ── Merge MCP servers into settings ──────────────────────────────── + (lib.mkIf (cfg.mcpServers != { }) { + services.hermes-agent.settings.mcp_servers = common.mcpServersToConfig cfg.mcpServers; + }) - backend = mkOption { - type = types.enum [ "docker" "podman" ]; - default = "docker"; - description = "Container runtime."; - }; + # ── User / group ────────────────────────────────────────────────── + (lib.mkIf cfg.createUser { + users.groups.${cfg.group} = { }; + users.users.${cfg.user} = { + isSystemUser = true; + group = cfg.group; + home = cfg.stateDir; + createHome = true; + shell = pkgs.bashInteractive; + }; + }) - extraVolumes = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Extra volume mounts (host:container:mode format)."; - example = [ "/home/user/projects:/projects:rw" ]; - }; + # ── Host CLI ────────────────────────────────────────────────────── + # Add the hermes CLI to system PATH and export HERMES_HOME system-wide + # so interactive shells share state (sessions, skills, cron) with the + # gateway service instead of creating a separate ~/.hermes/. + (lib.mkIf cfg.addToSystemPackages { + environment.systemPackages = [ effectivePackage ]; + environment.variables.HERMES_HOME = hermesHome; + }) - extraOptions = mkOption { - type = types.listOf types.str; - default = [ ]; - description = "Extra arguments passed to docker/podman run."; - }; + # ── Host user group membership ───────────────────────────────────── + (lib.mkIf (cfg.container.enable && cfg.container.hostUsers != [ ]) { + users.users = lib.genAttrs cfg.container.hostUsers (_user: { + extraGroups = [ cfg.group ]; + }); + }) - image = mkOption { - type = types.str; - default = "ubuntu:24.04"; - description = "OCI container image. The container pulls this at runtime via Docker/Podman."; - }; + # ── Assertions ───────────────────────────────────────────────────── + { + assertions = + common.pluginNameAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; + } + ++ common.workspaceFilesAssertions { + inherit cfg; + opt = options.services.hermes-agent.workingDirectory; + optionPath = "services.hermes-agent"; + } + ++ [ + { + # Container mode runs one command in one container. A second + # process needs its own container and its own ports. This + # module does not do that. + assertion = !(cfg.container.enable && cfg.backend.mode != "none"); + message = "services.hermes-agent: backend.mode is not supported together with container.enable — the container runs the gateway only."; + } + ]; + } - hostUsers = mkOption { - type = types.listOf types.str; - default = [ ]; - description = '' - Interactive users who get a ~/.hermes symlink to the service - stateDir. These users are automatically added to the hermes group. - ''; - example = [ "sidbin" ]; - }; - }; + # ── Per-user profile for extraPackages ─────────────────────────── + # Wire extraPackages into the hermes user's per-user profile so the + # login-shell snapshot (which rebuilds PATH from NixOS profiles) sees + # them. The systemd service PATH also includes them for direct access. + (lib.mkIf (cfg.extraPackages != [ ]) { + # listOf options are merged by the NixOS module system — this appends to + # any packages the operator assigned to this user externally (e.g. when + # createUser = false and the user definition lives elsewhere in the config). + users.users.${cfg.user}.packages = cfg.extraPackages; + }) + + # ── Warnings ────────────────────────────────────────────────────── + (lib.mkIf + (cfg.container.enable && !cfg.addToSystemPackages && cfg.container.hostUsers != [ ]) + { + warnings = [ + '' + services.hermes-agent: container.enable is true and container.hostUsers + is set, but addToSystemPackages is false. Without a host-installed hermes + binary, container routing will not work for interactive users. + Set addToSystemPackages = true or ensure hermes is on PATH. + '' + ]; + } + ) + + # ── Directories ─────────────────────────────────────────────────── + { + systemd.tmpfiles.rules = [ + "d ${cfg.stateDir} 2770 ${cfg.user} ${cfg.group} - -" + "d ${hermesHome} 2770 ${cfg.user} ${cfg.group} - -" + "d ${cfg.stateDir}/home 0750 ${cfg.user} ${cfg.group} - -" + "d ${cfg.workingDirectory} 2770 ${cfg.user} ${cfg.group} - -" + ] + ++ map (d: "d ${hermesHome}/${d} 2770 ${cfg.user} ${cfg.group} - -") common.stateSubdirs; + } + + # ── Activation: link config + auth + documents ──────────────────── + { + system.activationScripts."hermes-agent-setup" = + lib.stringAfter + ( + [ "users" ] ++ lib.optional (config.system.activationScripts ? setupSecrets) "setupSecrets" + ) + '' + # Ensure directories exist (activation runs before tmpfiles) + mkdir -p ${hermesHome} + mkdir -p ${cfg.stateDir}/home + mkdir -p ${cfg.workingDirectory} + chown ${cfg.user}:${cfg.group} ${cfg.stateDir} ${hermesHome} ${cfg.stateDir}/home ${cfg.workingDirectory} + chmod 2770 ${cfg.stateDir} ${hermesHome} ${cfg.workingDirectory} + chmod 0750 ${cfg.stateDir}/home + + # Create subdirs, set setgid + group-writable, migrate existing files. + # Nix-managed .env/.managed stay 0640/0644; config.yaml uses + # configYamlMode (0660 under addToSystemPackages, else 0640). + find ${hermesHome} -maxdepth 1 \ + \( -name "*.db" -o -name "*.db-wal" -o -name "*.db-shm" -o -name "SOUL.md" \) \ + -exec chmod g+rw {} + 2>/dev/null || true + for _subdir in ${lib.concatStringsSep " " common.stateSubdirs}; do + mkdir -p "${hermesHome}/$_subdir" + chown ${cfg.user}:${cfg.group} "${hermesHome}/$_subdir" + chmod 2770 "${hermesHome}/$_subdir" + find "${hermesHome}/$_subdir" -type f \ + -exec chmod g+rw {} + 2>/dev/null || true + done + + ${common.mkStateScript { + inherit pkgs cfg hermesHome; + workingDirectory = cfg.workingDirectory; + configWorkingDirectory = effectiveWorkDir; + owner = "${cfg.user}:${cfg.group}"; + stateDirs = common.stateSubdirs; + modes = { + config = configYamlMode; + env = "0640"; + managed = "0644"; + auth = "0600"; + document = "0640"; + }; + }} + + chown -h ${cfg.user}:${cfg.group} ${hermesHome}/plugins/nix-managed-* 2>/dev/null || true + + # Container mode metadata — tells the host CLI to exec into the + # container instead of running locally. Removed when container mode + # is disabled so the host CLI falls back to native execution. + ${ + if cfg.container.enable then + '' + install -o ${cfg.user} -g ${cfg.group} -m 0644 ${containerModeFile} ${hermesHome}/.container-mode + '' + else + '' + rm -f ${hermesHome}/.container-mode + + # Remove symlink bridge for hostUsers + ${lib.concatStringsSep "\n" ( + map ( + user: + let + userHome = config.users.users.${user}.home; + symlinkPath = "${userHome}/.hermes"; + in + '' + if [ -L "${symlinkPath}" ] && [ "$(readlink "${symlinkPath}")" = "${hermesHome}" ]; then + rm -f "${symlinkPath}" + echo "hermes-agent: removed symlink ${symlinkPath}" + fi + '' + ) cfg.container.hostUsers + )} + '' + } + + # ── Symlink bridge for interactive users ─────────────────────── + # Create ~/.hermes -> stateDir/.hermes for each hostUser so the + # host CLI shares state with the container service. + # Only runs when container mode is enabled. + ${lib.optionalString cfg.container.enable ( + lib.concatStringsSep "\n" ( + map ( + user: + let + userHome = config.users.users.${user}.home; + symlinkPath = "${userHome}/.hermes"; + in + '' + if [ -d "${symlinkPath}" ] && [ ! -L "${symlinkPath}" ]; then + # Real directory — back it up, then create symlink. + # (ln -sfn cannot atomically replace a directory.) + _backup="${symlinkPath}.bak.$(date +%s)" + echo "hermes-agent: backing up existing ${symlinkPath} to $_backup" + mv "${symlinkPath}" "$_backup" + fi + # For everything else (existing symlink, doesn't exist, etc.) + # ln -sfn handles it: replaces symlinks, creates new ones. + ln -sfn "${hermesHome}" "${symlinkPath}" + chown -h ${user}:${cfg.group} "${symlinkPath}" + '' + ) cfg.container.hostUsers + ) + )} + ''; + } + + # ══════════════════════════════════════════════════════════════════ + # MODE A: Native systemd service (default) + # ══════════════════════════════════════════════════════════════════ + (lib.mkIf (!cfg.container.enable) { + systemd.services.hermes-agent = { + description = "Hermes Agent Gateway"; + wantedBy = [ "multi-user.target" ]; + after = [ "network-online.target" ]; + wants = [ "network-online.target" ]; + + # cfg.environment and cfg.environmentFiles are written to + # $HERMES_HOME/.env by the activation script. load_hermes_dotenv() + # reads them at Python startup — no systemd EnvironmentFile needed. + environment = commonUnitEnvironment; + + serviceConfig = commonServiceConfig // { + ExecStart = lib.escapeShellArgs (common.gatewayArgv cfg); + }; + + path = unitPath; + }; + }) + + # ── The backend: hermes serve or hermes dashboard ───────────────── + # This is a different process from the gateway. Both use one + # HERMES_HOME. + (lib.mkIf (!cfg.container.enable && cfg.backend.mode != "none") { + systemd.services.hermes-backend = { + description = common.backendDescription cfg; + wantedBy = [ "multi-user.target" ]; + after = [ "network-online.target" ]; + wants = [ "network-online.target" ]; + + environment = commonUnitEnvironment; + + serviceConfig = commonServiceConfig // { + ExecStart = lib.escapeShellArgs (common.backendArgv cfg); + }; + + path = unitPath; + }; + }) + + # ══════════════════════════════════════════════════════════════════ + # MODE B: OCI container (persistent writable layer) + # ══════════════════════════════════════════════════════════════════ + (lib.mkIf cfg.container.enable { + # Ensure the container runtime is available + virtualisation.docker.enable = lib.mkDefault (cfg.container.backend == "docker"); + + systemd.services.hermes-agent = { + description = "Hermes Agent Gateway (container)"; + wantedBy = [ "multi-user.target" ]; + after = [ + "network-online.target" + ] + ++ lib.optional (cfg.container.backend == "docker") "docker.service"; + wants = [ "network-online.target" ]; + requires = lib.optional (cfg.container.backend == "docker") "docker.service"; + + preStart = '' + # Stable symlinks — container references these, not store paths directly + ln -sfn ${effectivePackage} ${cfg.stateDir}/current-package + ln -sfn ${containerEntrypoint} ${cfg.stateDir}/current-entrypoint + + # GC roots so nix-collect-garbage doesn't remove store paths in use + ${pkgs.nix}/bin/nix-store --add-root ${cfg.stateDir}/.gc-root --indirect -r ${effectivePackage} 2>/dev/null || true + ${pkgs.nix}/bin/nix-store --add-root ${cfg.stateDir}/.gc-root-entrypoint --indirect -r ${containerEntrypoint} 2>/dev/null || true + + # Check if container needs (re)creation + NEED_CREATE=false + if ! ${containerBin} inspect ${containerName} &>/dev/null; then + NEED_CREATE=true + elif [ ! -f ${identityFile} ] || [ "$(cat ${identityFile})" != "${containerIdentity}" ]; then + echo "Container config changed, recreating..." + ${containerBin} rm -f ${containerName} || true + NEED_CREATE=true + fi + + if [ "$NEED_CREATE" = "true" ]; then + # Resolve numeric UID/GID — passed to entrypoint for in-container user setup + HERMES_UID=$(${pkgs.coreutils}/bin/id -u ${cfg.user}) + HERMES_GID=$(${pkgs.coreutils}/bin/id -g ${cfg.user}) + + echo "Creating container..." + ${containerBin} create \ + --name ${containerName} \ + --network=host \ + --entrypoint ${containerDataDir}/current-entrypoint \ + --volume /nix/store:/nix/store:ro \ + --volume ${cfg.stateDir}:${containerDataDir} \ + --volume ${cfg.stateDir}/home:${containerHomeDir} \ + ${lib.concatStringsSep " " (map (v: "--volume ${v}") cfg.container.extraVolumes)} \ + --env HERMES_UID="$HERMES_UID" \ + --env HERMES_GID="$HERMES_GID" \ + --env HERMES_HOME=${containerDataDir}/.hermes \ + --env HERMES_MANAGED=true \ + --env HOME=${containerHomeDir} \ + ${lib.concatStringsSep " " cfg.container.extraOptions} \ + ${cfg.container.image} \ + ${containerDataDir}/current-package/bin/hermes gateway run --replace ${lib.concatStringsSep " " cfg.extraArgs} + + echo "${containerIdentity}" > ${identityFile} + fi + ''; + + script = '' + exec ${containerBin} start -a ${containerName} + ''; + + preStop = '' + ${containerBin} stop -t 10 ${containerName} || true + ''; + + serviceConfig = { + Type = "simple"; + Restart = cfg.restart; + RestartSec = cfg.restartSec; + TimeoutStopSec = 30; + }; + }; + }) + ] + ); }; - - config = lib.mkIf cfg.enable (lib.mkMerge [ - - # ── Merge MCP servers into settings ──────────────────────────────── - (lib.mkIf (cfg.mcpServers != { }) { - services.hermes-agent.settings.mcp_servers = lib.mapAttrs (_name: srv: - # Stdio transport - lib.optionalAttrs (srv.command != null) { inherit (srv) command args; } - // lib.optionalAttrs (srv.env != { }) { inherit (srv) env; } - # HTTP transport - // lib.optionalAttrs (srv.url != null) { inherit (srv) url; } - // lib.optionalAttrs (srv.headers != { }) { inherit (srv) headers; } - # Auth - // lib.optionalAttrs (srv.auth != null) { inherit (srv) auth; } - # Enable/disable - // { inherit (srv) enabled; } - # Common options - // lib.optionalAttrs (srv.timeout != null) { inherit (srv) timeout; } - // lib.optionalAttrs (srv.connect_timeout != null) { inherit (srv) connect_timeout; } - # Tool filtering - // lib.optionalAttrs (srv.tools != null) { - tools = lib.filterAttrs (_: v: v != [ ]) { - inherit (srv.tools) include exclude; - }; - } - # Sampling - // lib.optionalAttrs (srv.sampling != null) { - sampling = lib.filterAttrs (_: v: v != null && v != [ ]) { - inherit (srv.sampling) enabled model max_tokens_cap timeout max_rpm - max_tool_rounds allowed_models log_level; - }; - } - ) cfg.mcpServers; - }) - - # ── User / group ────────────────────────────────────────────────── - (lib.mkIf cfg.createUser { - users.groups.${cfg.group} = { }; - users.users.${cfg.user} = { - isSystemUser = true; - group = cfg.group; - home = cfg.stateDir; - createHome = true; - shell = pkgs.bashInteractive; - }; - }) - - # ── Host CLI ────────────────────────────────────────────────────── - # Add the hermes CLI to system PATH and export HERMES_HOME system-wide - # so interactive shells share state (sessions, skills, cron) with the - # gateway service instead of creating a separate ~/.hermes/. - (lib.mkIf cfg.addToSystemPackages { - environment.systemPackages = [ effectivePackage ]; - environment.variables.HERMES_HOME = "${cfg.stateDir}/.hermes"; - }) - - # ── Host user group membership ───────────────────────────────────── - (lib.mkIf (cfg.container.enable && cfg.container.hostUsers != []) { - users.users = lib.genAttrs cfg.container.hostUsers (user: { - extraGroups = [ cfg.group ]; - }); - }) - - # ── Assertions ───────────────────────────────────────────────────── - { - assertions = let - names = map lib.getName cfg.extraPlugins; - in [{ - assertion = (lib.length names) == (lib.length (lib.unique names)); - message = "services.hermes-agent.extraPlugins: duplicate plugin names detected: ${toString names}. If using fetchFromGitHub, set name = \"plugin-name\" to disambiguate."; - }]; - } - - # ── Warnings ────────────────────────────────────────────────────── - # ── Per-user profile for extraPackages ─────────────────────────── - # Wire extraPackages into the hermes user's per-user profile so the - # login-shell snapshot (which rebuilds PATH from NixOS profiles) sees - # them. The systemd service PATH also includes them for direct access. - (lib.mkIf (cfg.extraPackages != []) { - # listOf options are merged by the NixOS module system — this appends to - # any packages the operator assigned to this user externally (e.g. when - # createUser = false and the user definition lives elsewhere in the config). - users.users.${cfg.user}.packages = cfg.extraPackages; - }) - - (lib.mkIf (cfg.container.enable && !cfg.addToSystemPackages && cfg.container.hostUsers != []) { - warnings = [ - '' - services.hermes-agent: container.enable is true and container.hostUsers - is set, but addToSystemPackages is false. Without a host-installed hermes - binary, container routing will not work for interactive users. - Set addToSystemPackages = true or ensure hermes is on PATH. - '' - ]; - }) - - # ── Directories ─────────────────────────────────────────────────── - { - systemd.tmpfiles.rules = [ - "d ${cfg.stateDir} 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes/cron 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes/sessions 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes/logs 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes/memories 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/.hermes/plugins 2770 ${cfg.user} ${cfg.group} - -" - "d ${cfg.stateDir}/home 0750 ${cfg.user} ${cfg.group} - -" - "d ${cfg.workingDirectory} 2770 ${cfg.user} ${cfg.group} - -" - ]; - } - - # ── Activation: link config + auth + documents ──────────────────── - { - system.activationScripts."hermes-agent-setup" = lib.stringAfter ([ "users" ] ++ lib.optional (config.system.activationScripts ? setupSecrets) "setupSecrets") '' - # Ensure directories exist (activation runs before tmpfiles) - mkdir -p ${cfg.stateDir}/.hermes - mkdir -p ${cfg.stateDir}/home - mkdir -p ${cfg.workingDirectory} - chown ${cfg.user}:${cfg.group} ${cfg.stateDir} ${cfg.stateDir}/.hermes ${cfg.stateDir}/home ${cfg.workingDirectory} - chmod 2770 ${cfg.stateDir} ${cfg.stateDir}/.hermes ${cfg.workingDirectory} - chmod 0750 ${cfg.stateDir}/home - - # Create subdirs, set setgid + group-writable, migrate existing files. - # Nix-managed .env/.managed stay 0640/0644; config.yaml uses - # configYamlMode (0660 under addToSystemPackages, else 0640). - find ${cfg.stateDir}/.hermes -maxdepth 1 \ - \( -name "*.db" -o -name "*.db-wal" -o -name "*.db-shm" -o -name "SOUL.md" \) \ - -exec chmod g+rw {} + 2>/dev/null || true - for _subdir in cron sessions logs memories plugins; do - mkdir -p "${cfg.stateDir}/.hermes/$_subdir" - chown ${cfg.user}:${cfg.group} "${cfg.stateDir}/.hermes/$_subdir" - chmod 2770 "${cfg.stateDir}/.hermes/$_subdir" - find "${cfg.stateDir}/.hermes/$_subdir" -type f \ - -exec chmod g+rw {} + 2>/dev/null || true - done - - # Merge Nix settings into existing config.yaml. - # Preserves user-added keys (skills, streaming, etc.); Nix keys win. - # If configFile is user-provided (not generated), overwrite instead of merge. - # Mode is configYamlMode (0660 under addToSystemPackages so interactive - # hermes-group users can save settings via the CLI/TUI, else 0640). - ${if cfg.configFile != null then '' - install -o ${cfg.user} -g ${cfg.group} -m ${configYamlMode} -D ${configFile} ${cfg.stateDir}/.hermes/config.yaml - '' else '' - ${configMergeScript} ${generatedConfigFile} ${cfg.stateDir}/.hermes/config.yaml - chown ${cfg.user}:${cfg.group} ${cfg.stateDir}/.hermes/config.yaml - chmod ${configYamlMode} ${cfg.stateDir}/.hermes/config.yaml - ''} - - # Managed mode marker (so interactive shells also detect NixOS management) - touch ${cfg.stateDir}/.hermes/.managed - chown ${cfg.user}:${cfg.group} ${cfg.stateDir}/.hermes/.managed - chmod 0644 ${cfg.stateDir}/.hermes/.managed - - # Container mode metadata — tells the host CLI to exec into the - # container instead of running locally. Removed when container mode - # is disabled so the host CLI falls back to native execution. - ${if cfg.container.enable then '' - cat > ${cfg.stateDir}/.hermes/.container-mode <<'HERMES_CONTAINER_MODE_EOF' - # Written by NixOS activation script. Do not edit manually. - backend=${cfg.container.backend} - container_name=${containerName} - exec_user=${cfg.user} - hermes_bin=${containerDataDir}/current-package/bin/hermes - HERMES_CONTAINER_MODE_EOF - chown ${cfg.user}:${cfg.group} ${cfg.stateDir}/.hermes/.container-mode - chmod 0644 ${cfg.stateDir}/.hermes/.container-mode - '' else '' - rm -f ${cfg.stateDir}/.hermes/.container-mode - - # Remove symlink bridge for hostUsers - ${lib.concatStringsSep "\n" (map (user: - let - userHome = config.users.users.${user}.home; - symlinkPath = "${userHome}/.hermes"; - in '' - if [ -L "${symlinkPath}" ] && [ "$(readlink "${symlinkPath}")" = "${cfg.stateDir}/.hermes" ]; then - rm -f "${symlinkPath}" - echo "hermes-agent: removed symlink ${symlinkPath}" - fi - '') cfg.container.hostUsers)} - ''} - - # ── Symlink bridge for interactive users ─────────────────────── - # Create ~/.hermes -> stateDir/.hermes for each hostUser so the - # host CLI shares state with the container service. - # Only runs when container mode is enabled. - ${lib.optionalString cfg.container.enable - (lib.concatStringsSep "\n" (map (user: - let - userHome = config.users.users.${user}.home; - symlinkPath = "${userHome}/.hermes"; - target = "${cfg.stateDir}/.hermes"; - in '' - if [ -d "${symlinkPath}" ] && [ ! -L "${symlinkPath}" ]; then - # Real directory — back it up, then create symlink. - # (ln -sfn cannot atomically replace a directory.) - _backup="${symlinkPath}.bak.$(date +%s)" - echo "hermes-agent: backing up existing ${symlinkPath} to $_backup" - mv "${symlinkPath}" "$_backup" - fi - # For everything else (existing symlink, doesn't exist, etc.) - # ln -sfn handles it: replaces symlinks, creates new ones. - ln -sfn "${target}" "${symlinkPath}" - chown -h ${user}:${cfg.group} "${symlinkPath}" - '') cfg.container.hostUsers))} - - # Seed auth file if provided - ${lib.optionalString (cfg.authFile != null) '' - ${if cfg.authFileForceOverwrite then '' - install -o ${cfg.user} -g ${cfg.group} -m 0600 ${cfg.authFile} ${cfg.stateDir}/.hermes/auth.json - '' else '' - if [ ! -f ${cfg.stateDir}/.hermes/auth.json ]; then - install -o ${cfg.user} -g ${cfg.group} -m 0600 ${cfg.authFile} ${cfg.stateDir}/.hermes/auth.json - fi - ''} - ''} - - # Seed .env from Nix-declared environment + environmentFiles. - # Hermes reads $HERMES_HOME/.env at startup via load_hermes_dotenv(), - # so this is the single source of truth for both native and container mode. - ${lib.optionalString (cfg.environment != {} || cfg.environmentFiles != []) '' - ENV_FILE="${cfg.stateDir}/.hermes/.env" - install -o ${cfg.user} -g ${cfg.group} -m 0640 /dev/null "$ENV_FILE" - cat > "$ENV_FILE" <<'HERMES_NIX_ENV_EOF' - ${envFileContent} - HERMES_NIX_ENV_EOF - ${lib.concatStringsSep "\n" (map (f: '' - if [ -f "${f}" ]; then - echo "" >> "$ENV_FILE" - cat "${f}" >> "$ENV_FILE" - fi - '') cfg.environmentFiles)} - ''} - - # Link documents into workspace - ${lib.concatStringsSep "\n" (lib.mapAttrsToList (name: _value: '' - install -o ${cfg.user} -g ${cfg.group} -m 0640 ${documentDerivation}/${name} ${cfg.workingDirectory}/${name} - '') cfg.documents)} - - # ── Declarative plugins ───────────────────────────────────────── - # Remove stale managed symlinks (plugins removed from config) - find ${cfg.stateDir}/.hermes/plugins -maxdepth 1 -type l -name 'nix-managed-*' -delete 2>/dev/null || true - - ${lib.concatStringsSep "\n" (map (plugin: - let - name = lib.getName plugin; - in '' - if [ ! -f "${plugin}/plugin.yaml" ]; then - echo "ERROR: extraPlugins entry '${plugin}' has no plugin.yaml" >&2 - exit 1 - fi - ln -sfn ${plugin} ${cfg.stateDir}/.hermes/plugins/nix-managed-${name} - chown -h ${cfg.user}:${cfg.group} ${cfg.stateDir}/.hermes/plugins/nix-managed-${name} - '') cfg.extraPlugins)} - ''; - } - - # ══════════════════════════════════════════════════════════════════ - # MODE A: Native systemd service (default) - # ══════════════════════════════════════════════════════════════════ - (lib.mkIf (!cfg.container.enable) { - systemd.services.hermes-agent = { - description = "Hermes Agent Gateway"; - wantedBy = [ "multi-user.target" ]; - after = [ "network-online.target" ]; - wants = [ "network-online.target" ]; - - environment = { - HOME = cfg.stateDir; - HERMES_HOME = "${cfg.stateDir}/.hermes"; - HERMES_MANAGED = "true"; - # Working directory is declared via terminal.cwd in the merged - # config.yaml (see configJson above) — MESSAGING_CWD is deprecated. - }; - - serviceConfig = { - User = cfg.user; - Group = cfg.group; - WorkingDirectory = cfg.workingDirectory; - - # cfg.environment and cfg.environmentFiles are written to - # $HERMES_HOME/.env by the activation script. load_hermes_dotenv() - # reads them at Python startup — no systemd EnvironmentFile needed. - - ExecStart = lib.concatStringsSep " " ([ - "${effectivePackage}/bin/hermes" - "gateway" - ] ++ cfg.extraArgs); - - Restart = cfg.restart; - RestartSec = cfg.restartSec; - - # Shared-state: files created by the gateway should be group-writable - # so interactive users in the hermes group can read/write them. - UMask = "0007"; - - # Hardening - NoNewPrivileges = true; - ProtectSystem = "strict"; - ProtectHome = false; - ReadWritePaths = [ - cfg.stateDir - cfg.workingDirectory - ]; - PrivateTmp = true; - }; - - path = [ - effectivePackage - pkgs.bash - pkgs.coreutils - pkgs.git - ] ++ cfg.extraPackages; - }; - }) - - # ══════════════════════════════════════════════════════════════════ - # MODE B: OCI container (persistent writable layer) - # ══════════════════════════════════════════════════════════════════ - (lib.mkIf cfg.container.enable { - # Ensure the container runtime is available - virtualisation.docker.enable = lib.mkDefault (cfg.container.backend == "docker"); - - systemd.services.hermes-agent = { - description = "Hermes Agent Gateway (container)"; - wantedBy = [ "multi-user.target" ]; - after = [ "network-online.target" ] - ++ lib.optional (cfg.container.backend == "docker") "docker.service"; - wants = [ "network-online.target" ]; - requires = lib.optional (cfg.container.backend == "docker") "docker.service"; - - preStart = '' - # Stable symlinks — container references these, not store paths directly - ln -sfn ${effectivePackage} ${cfg.stateDir}/current-package - ln -sfn ${containerEntrypoint} ${cfg.stateDir}/current-entrypoint - - # GC roots so nix-collect-garbage doesn't remove store paths in use - ${pkgs.nix}/bin/nix-store --add-root ${cfg.stateDir}/.gc-root --indirect -r ${effectivePackage} 2>/dev/null || true - ${pkgs.nix}/bin/nix-store --add-root ${cfg.stateDir}/.gc-root-entrypoint --indirect -r ${containerEntrypoint} 2>/dev/null || true - - # Check if container needs (re)creation - NEED_CREATE=false - if ! ${containerBin} inspect ${containerName} &>/dev/null; then - NEED_CREATE=true - elif [ ! -f ${identityFile} ] || [ "$(cat ${identityFile})" != "${containerIdentity}" ]; then - echo "Container config changed, recreating..." - ${containerBin} rm -f ${containerName} || true - NEED_CREATE=true - fi - - if [ "$NEED_CREATE" = "true" ]; then - # Resolve numeric UID/GID — passed to entrypoint for in-container user setup - HERMES_UID=$(${pkgs.coreutils}/bin/id -u ${cfg.user}) - HERMES_GID=$(${pkgs.coreutils}/bin/id -g ${cfg.user}) - - echo "Creating container..." - ${containerBin} create \ - --name ${containerName} \ - --network=host \ - --entrypoint ${containerDataDir}/current-entrypoint \ - --volume /nix/store:/nix/store:ro \ - --volume ${cfg.stateDir}:${containerDataDir} \ - --volume ${cfg.stateDir}/home:${containerHomeDir} \ - ${lib.concatStringsSep " " (map (v: "--volume ${v}") cfg.container.extraVolumes)} \ - --env HERMES_UID="$HERMES_UID" \ - --env HERMES_GID="$HERMES_GID" \ - --env HERMES_HOME=${containerDataDir}/.hermes \ - --env HERMES_MANAGED=true \ - --env HOME=${containerHomeDir} \ - ${lib.concatStringsSep " " cfg.container.extraOptions} \ - ${cfg.container.image} \ - ${containerDataDir}/current-package/bin/hermes gateway run --replace ${lib.concatStringsSep " " cfg.extraArgs} - - echo "${containerIdentity}" > ${identityFile} - fi - ''; - - script = '' - exec ${containerBin} start -a ${containerName} - ''; - - preStop = '' - ${containerBin} stop -t 10 ${containerName} || true - ''; - - serviceConfig = { - Type = "simple"; - Restart = cfg.restart; - RestartSec = cfg.restartSec; - TimeoutStopSec = 30; - }; - }; - }) - ]); - }; } diff --git a/tests/hermes_cli/test_managed_install_shapes.py b/tests/hermes_cli/test_managed_install_shapes.py new file mode 100644 index 0000000000..89aad936a4 --- /dev/null +++ b/tests/hermes_cli/test_managed_install_shapes.py @@ -0,0 +1,112 @@ +"""Managed-mode detection across the Nix install shapes. + +The NixOS module and the Home Manager module both mark the install as +managed, and the CLI then refuses a configuration change that it cannot +keep. The two modules write different values, and an install from an +earlier version writes an empty marker, so detection must handle all three. +""" + +import os + +import pytest + +from hermes_cli import config as config_mod + + +@pytest.fixture +def hermes_home(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + home.mkdir() + monkeypatch.setenv("HERMES_HOME", str(home)) + monkeypatch.delenv("HERMES_MANAGED", raising=False) + return home + + +@pytest.mark.parametrize( + ("env_value", "expected"), + [ + ("nixos", "nixos"), + ("home-manager", "home-manager"), + ("HOME-MANAGER", "home-manager"), + # An install from an earlier version sets a bare "true". Only the + # NixOS module did that. + ("true", "nixos"), + ("1", "nixos"), + # Homebrew is not a distribution method, so these must not block a + # configuration change. + ("brew", None), + ("homebrew", None), + (None, None), + ], +) +def test_env_var_names_the_managing_system(hermes_home, monkeypatch, env_value, expected): + if env_value is None: + monkeypatch.delenv("HERMES_MANAGED", raising=False) + else: + monkeypatch.setenv("HERMES_MANAGED", env_value) + + assert config_mod.get_managed_system() == expected + assert config_mod.is_managed() is (expected is not None) + + +@pytest.mark.parametrize( + ("marker_text", "expected"), + [ + ("home-manager", "home-manager"), + ("nixos", "nixos"), + # An install from an earlier version has an empty marker. Only the + # NixOS module wrote one. + ("", "nixos"), + ], +) +def test_marker_file_names_the_managing_system( + hermes_home, monkeypatch, marker_text, expected +): + """An interactive shell reads .managed, not the HERMES_MANAGED of the service.""" + (hermes_home / ".managed").write_text(marker_text, encoding="utf-8") + monkeypatch.delenv("HERMES_MANAGED", raising=False) + + assert config_mod.get_managed_system() == expected + assert config_mod.is_managed() is True + + +def test_env_var_wins_over_the_marker(hermes_home, monkeypatch): + (hermes_home / ".managed").write_text("nixos", encoding="utf-8") + monkeypatch.setenv("HERMES_MANAGED", "home-manager") + + assert config_mod.get_managed_system() == "home-manager" + + +@pytest.mark.parametrize("managed_value", ["nixos", "home-manager"]) +def test_managed_install_names_its_system_and_offers_an_update( + hermes_home, monkeypatch, managed_value +): + """The message names the system, so the user knows what owns the install.""" + monkeypatch.setenv("HERMES_MANAGED", managed_value) + + assert managed_value in config_mod.format_managed_message("set model") + assert "set model" in config_mod.format_managed_message("set model") + assert config_mod.get_managed_update_command() + assert config_mod.detect_install_method(config_mod.get_project_root()) == managed_value + # `hermes update` cannot run on a managed install, so the advice must not + # name it. + assert config_mod.recommended_update_command() != "hermes update" + + +def test_unmanaged_install_offers_no_update_command(hermes_home, monkeypatch): + monkeypatch.delenv("HERMES_MANAGED", raising=False) + assert config_mod.get_managed_update_command() is None + + +def test_unreadable_marker_still_reports_managed(hermes_home, monkeypatch): + """A marker we cannot read is still a marker. Fail closed, not open.""" + marker = hermes_home / ".managed" + marker.write_text("home-manager", encoding="utf-8") + marker.chmod(0o000) + monkeypatch.delenv("HERMES_MANAGED", raising=False) + try: + if os.access(marker, os.R_OK): + pytest.skip("running as a user that ignores file modes (for example root)") + assert config_mod.is_managed() is True + finally: + marker.chmod(0o600) diff --git a/website/docs/getting-started/nix-setup.md b/website/docs/getting-started/nix-setup.md index 2919d14acf..43c2f268bd 100644 --- a/website/docs/getting-started/nix-setup.md +++ b/website/docs/getting-started/nix-setup.md @@ -12,11 +12,12 @@ Nix and NixOS are [Tier 2 platforms](./platform-support.md#tier-2). The flake an For a supported setup, use one of the standard [installation](./installation.md) paths - either Docker or an FHS environment. ::: -Hermes Agent ships a Nix flake & a NixOS module. +Hermes Agent ships a Nix flake, a NixOS module, and a Home Manager module. | Level | Who it's for | What you get | |-------|-------------|--------------| | **`nix run` / `nix profile install`** | Any Nix user (macOS, Linux) | Pre-built binary with all deps — then use the standard CLI workflow | +| **Home Manager module** | An agent for one person, on any distribution or on macOS | Declarative configuration and a user service, without root | | **NixOS module (native)** | NixOS server deployments | Declarative config, hardened systemd service, managed secrets | | **NixOS module (container)** | Agents that need self-modification | Everything above, plus a persistent Ubuntu container where the agent can `apt`/`pip`/`npm install` | @@ -84,7 +85,7 @@ hermes setup The flake exports `nixosModules.default` — a full NixOS service module that declaratively manages user creation, directories, config generation, secrets, documents, and service lifecycle. :::note -This module requires NixOS. For non-NixOS systems (macOS, other Linux distros), use `nix profile install` and the standard CLI workflow above. +This module needs NixOS. Hermes is an agent for one person. If you want an agent for one person and not a system service, use the [Home Manager module](#home-manager-module). That module runs on NixOS and on each other system that Home Manager supports. ::: ### Add the Flake Input @@ -287,8 +288,10 @@ Run `nix build .#configKeys && cat result` to see every leaf config key extracte environmentFiles = [ config.sops.secrets."hermes-env".path ]; # ── Documents ────────────────────────────────────────────────────── - documents = { - "USER.md" = ./documents/USER.md; + # USER.md is memory, so it goes to HERMES_HOME. Workspace files use + # `documents`, and that option needs an explicit `workingDirectory`. + hermesHomeFiles = { + "memories/USER.md" = ./documents/USER.md; }; # ── MCP Servers ──────────────────────────────────────────────────── @@ -336,7 +339,9 @@ Quick reference for the most common things Nix users want to customize: | Change the LLM model | `settings.model.default` | `"anthropic/claude-sonnet-4"` | | Use a different provider endpoint | `settings.model.base_url` | `"https://openrouter.ai/api/v1"` | | Add API keys | `environmentFiles` | `[ config.sops.secrets."hermes-env".path ]` | -| Give the agent a personality | `${services.hermes-agent.stateDir}/.hermes/SOUL.md` | manage the file directly | +| Give the agent an identity | `hermesHomeFiles."SOUL.md"` | `"You are a terse ops assistant."` | +| Add project context to the workspace | `documents."AGENTS.md"` | `./documents/AGENTS.md` | +| Run the backend for the desktop app or the dashboard | `backend.mode` | `"serve"` or `"dashboard"` | | Add MCP tool servers | `mcpServers.` | See [MCP Servers](#mcp-servers) | | Enable Discord/Telegram/Slack | `extraDependencyGroups` | `[ "messaging" ]` | | Mount host directories into container | `container.extraVolumes` | `[ "/data:/data:rw" ]` | @@ -416,22 +421,45 @@ The file is only copied if `auth.json` doesn't already exist (unless `authFileFo ## Documents -The `documents` option installs files into the agent's working directory (the `workingDirectory`, which the agent reads as its workspace). Hermes looks for specific filenames by convention: +Hermes reads files from two directories. Thus there are two options. Use the option for the directory that the file must go into. -- **`USER.md`** — context about the user the agent is interacting with. -- Any other files you place here are visible to the agent as workspace files. - -The agent identity file is separate: Hermes loads its primary `SOUL.md` from `$HERMES_HOME/SOUL.md`, which in the NixOS module is `${services.hermes-agent.stateDir}/.hermes/SOUL.md`. Putting `SOUL.md` in `documents` only creates a workspace file and will not replace the main persona file. +`documents` installs into the **working directory** of the agent, which is `workingDirectory`. The agent reads its project context from that workspace: ```nix { - services.hermes-agent.documents = { - "USER.md" = ./documents/USER.md; # path reference, copied from Nix store + services.hermes-agent = { + # documents needs this option. Read the note below. + workingDirectory = "/var/lib/hermes/workspace"; + documents = { + "AGENTS.md" = ./documents/AGENTS.md; # path reference, copied from Nix store + "notes/oncall.md" = "Page #infra before restarting anything."; + }; }; } ``` -Values can be inline strings or path references. Files are installed on every `nixos-rebuild switch`. +:::warning documents needs an explicit workingDirectory +The module refuses `documents` until you set `workingDirectory`. The default of +that option is different on each module. It is your home directory on Home +Manager, and `${stateDir}/workspace` on NixOS. Thus an unset default puts the +files in a directory that you did not select. A directory with the same path as +the default is a correct selection, and it satisfies the rule. +::: + +`hermesHomeFiles` installs into **`HERMES_HOME`**. Hermes reads the identity file and the memory files of the agent from that directory. `SOUL.md` and `memories/` work only from there. A `SOUL.md` in `documents` makes a workspace file. Hermes does not load that file as the identity: + +```nix +{ + services.hermes-agent.hermesHomeFiles = { + "SOUL.md" = "You are a helpful AI assistant."; + "memories/USER.md" = ./documents/USER.md; + }; +} +``` + +Each value is a string or a path. A key in either option can contain subdirectories, and the module makes the parent directories. Each activation installs the files again. + +`hermesHomeFiles` needs no `workingDirectory`, because the module owns the `HERMES_HOME` directory. Most users want `hermesHomeFiles`. --- @@ -554,10 +582,108 @@ When hermes runs via the NixOS module, the following CLI commands are **blocked* This prevents drift between what Nix declares and what's on disk. Detection uses two signals: -1. **`HERMES_MANAGED=true`** environment variable — set by the systemd service, visible to the gateway process -2. **`.managed` marker file** in `HERMES_HOME` — set by the activation script, visible to interactive shells (e.g., `docker exec -it hermes-agent hermes config set ...` is also blocked) +1. **The `HERMES_MANAGED` environment variable.** The service sets it, and the gateway process reads it. +2. **The `.managed` marker file** in `HERMES_HOME`. The activation script writes it, and an interactive shell reads it. Thus the CLI also blocks a command such as `docker exec -it hermes-agent hermes config set ...`. -To change configuration, edit your Nix config and run `sudo nixos-rebuild switch`. +Both signals hold the name of the system that manages the install. Thus the refusal names the correct rebuild command. The NixOS module gives `sudo nixos-rebuild switch`. The Home Manager module gives `home-manager switch`. + +--- + +## Home Manager Module + +The flake also exports `homeManagerModules.default`. Hermes is an agent for one person. The credentials, the memory, the sessions and the cron jobs all belong to that person. Thus a user service is the correct shape on a personal machine. It runs on each distribution that Home Manager supports, and not only on NixOS. + +The option set is the same set that the NixOS module uses. It is `services.hermes-agent`, with the same `settings`, `environmentFiles`, `documents`, `mcpServers`, `extraPlugins` and `backend` options. Each example above works here without a change. Only the necessary parts are different: + +| | NixOS module | Home Manager module | +|---|---|---| +| Runs as | a system user that you declare, with `user`, `group` and `createUser` | you | +| State directory | `stateDir` and `/.hermes` | `hermesHome`, set directly. The default is `~/.hermes`. | +| Service | `systemd.services` | `systemd.user.services` on Linux, `launchd.agents` on macOS | +| CLI on the PATH | `addToSystemPackages`, which exports `HERMES_HOME` for the full system | `installPackage`, which exports it for your session only | +| Container mode | supported | not supported, because it needs root and the Docker socket | + +### Add the Flake Input + +```nix +{ + inputs = { + nixpkgs.url = "github:NixOS/nixpkgs/nixos-unstable"; + home-manager.url = "github:nix-community/home-manager"; + home-manager.inputs.nixpkgs.follows = "nixpkgs"; + hermes-agent.url = "github:NousResearch/hermes-agent"; + }; +} +``` + +Then import the module into your Home Manager configuration. The configuration can be standalone. It can also be under `home-manager.users.` in a NixOS or nix-darwin configuration: + +```nix +{ + imports = [ hermes-agent.homeManagerModules.default ]; + + services.hermes-agent = { + enable = true; + gateway.enable = true; + settings.model.default = "anthropic/claude-sonnet-4"; + environmentFiles = [ config.sops.secrets."hermes-env".path ]; + }; +} +``` + +`home-manager switch` makes `~/.hermes`, writes `config.yaml`, builds `.env` and starts the gateway as a user service. + +:::warning Enable linger, or the service stops at logout +CAUTION: Enable linger for your account. Without linger, systemd stops the user manager when your last session ends, and the gateway stops with it. Home Manager cannot set linger, because linger is a property of the account: + +```nix +# NixOS +users.users.your-username.linger = true; +``` + +```bash +# anywhere else +sudo loginctl enable-linger your-username +``` + +macOS has no equivalent option. A `launchd` agent with `RunAtLoad` starts at login and continues to run. +::: + +### Running the Desktop / Dashboard Backend + +`gateway.enable` runs the messaging gateway for Telegram, Discord, Slack and the other platforms. Hermes Desktop and the web dashboard connect to a *different* process, which is `hermes serve` or `hermes dashboard`. `backend.mode` runs that process with the gateway: + +```nix +{ + services.hermes-agent = { + enable = true; + gateway.enable = true; # messaging platforms + backend.mode = "dashboard"; # + the browser dashboard on 127.0.0.1:9119 + backend.port = 9119; + }; +} +``` + +`serve` runs without a user interface. It gives the `/api/ws` and `/api/pty` sockets that Hermes Desktop connects to, and it does not build the web application. `dashboard` gives all of that, and also serves the browser admin panel. Both processes use one `HERMES_HOME` with the gateway. Thus the sessions, the skills, the memory and the cron jobs are the same for all of them. `backend.mode` works in the same way on the NixOS module, but not in container mode. + +:::warning Binding to an address other than loopback +The default address is `127.0.0.1`. Each other address starts the authentication gate of the dashboard. The server also refuses each request with a `Host` header that is different from the address that the server bound to. This is a defence against DNS rebinding. Bind to the name or the address that your client uses. +::: + +### Verify It Works + +```bash +# Linux +systemctl --user status hermes-agent +journalctl --user -u hermes-agent -f + +# macOS +launchctl list | grep hermes +tail -f ~/Library/Logs/hermes-agent.log + +hermes version +hermes config # shows the configuration that Nix wrote +``` --- @@ -587,7 +713,7 @@ Host Container │ └── mcp-tokens/ (OAuth tokens for MCP servers) ├── home/ ──► /home/hermes (rw) └── workspace/ (agent working directory) - ├── SOUL.md (from documents option) + ├── AGENTS.md (from the documents option) └── (agent-created files) Container writable layer (apt/pip/npm): /usr, /usr/local, /tmp @@ -857,7 +983,8 @@ nix build .#checks.x86_64-linux.config-roundtrip # merge script preserves use | Option | Type | Default | Description | |---|---|---|---| -| `documents` | `attrsOf (either str path)` | `{}` | Workspace files. Keys are filenames, values are inline strings or paths. Installed into `workingDirectory` on activation | +| `documents` | `attrsOf (either str path)` | `{}` | Workspace files. Each key is a path relative to `workingDirectory`. You must set that option to use this one. | +| `hermesHomeFiles` | `attrsOf (either str path)` | `{}` | Files that go into `HERMES_HOME`. `SOUL.md` and `memories/` must be here, or Hermes does not load them. | ### MCP Servers @@ -885,10 +1012,29 @@ nix build .#checks.x86_64-linux.config-roundtrip # merge script preserves use | `extraPlugins` | `listOf package` | `[]` | Directory plugin packages to symlink into `$HERMES_HOME/plugins/`. Each must contain `plugin.yaml` | | `extraPythonPackages` | `listOf package` | `[]` | Python packages added to PYTHONPATH for entry-point plugin discovery. Build with `python312Packages` | | `extraDependencyGroups` | `listOf str` | `[]` | pyproject.toml optional extras to include in the sealed venv (e.g. `["hindsight"]`). Resolved by uv — no collisions | -| `restart` | `str` | `"always"` | systemd `Restart=` policy | -| `restartSec` | `int` | `5` | systemd `RestartSec=` value | +| `restart` | `str` | `"always"` | The systemd `Restart=` policy. macOS does not use it. | +| `restartSec` | `int` | `5` | The systemd `RestartSec=` value. macOS does not use it. | -### Container +### Backend (`hermes serve` / `hermes dashboard`) + +This option runs the process that Hermes Desktop and the web dashboard connect to, with the gateway. You cannot use it with `container.enable`. + +| Option | Type | Default | Description | +|---|---|---|---| +| `backend.mode` | `enum ["none" "serve" "dashboard"]` | `"none"` | `serve` runs without a user interface and gives `/api/ws` and `/api/pty`. `dashboard` also serves the browser panel. | +| `backend.host` | `str` | `"127.0.0.1"` | The address to bind to. Each address other than loopback starts the authentication gate. | +| `backend.port` | `port` | `9119` | The port to bind to | +| `backend.extraArgs` | `listOf str` | `[]` | More arguments for the backend command | + +### Home Manager only + +| Option | Type | Default | Description | +|---|---|---|---| +| `hermesHome` | `str` | `"${config.home.homeDirectory}/.hermes"` | `HERMES_HOME` directly. The NixOS module builds it from `stateDir`. | +| `installPackage` | `bool` | `true` | Add the `hermes` CLI to `home.packages`, and export `HERMES_HOME` for your shells | +| `gateway.enable` | `bool` | `false` | Run the messaging gateway. On the NixOS module the gateway is the service, so that module has no such option. | + +### Container (NixOS only) | Option | Type | Default | Description | |---|---|---|---| @@ -908,6 +1054,7 @@ nix build .#checks.x86_64-linux.config-roundtrip # merge script preserves use ``` /var/lib/hermes/ # stateDir (owned by hermes:hermes, 0750) ├── .hermes/ # HERMES_HOME +│ ├── SOUL.md # from hermesHomeFiles: the agent identity │ ├── config.yaml # Nix-generated (deep-merged each rebuild) │ ├── .managed # Marker: CLI config mutation blocked │ ├── .env # Merged from environment + environmentFiles @@ -922,10 +1069,26 @@ nix build .#checks.x86_64-linux.config-roundtrip # merge script preserves use │ └── logs/ ├── home/ # Agent HOME └── workspace/ # Agent working directory - ├── SOUL.md # From documents option + ├── AGENTS.md # from the documents option └── (agent-created files) ``` +### Home Manager + +``` +~/.hermes/ # hermesHome (HERMES_HOME), 0700 +├── SOUL.md # from hermesHomeFiles +├── config.yaml # written by Nix, merged at each activation +├── .managed # marker: names the system that manages this +├── .env # written again from environment + environmentFiles +├── auth.json # OAuth credentials: seeded, then Hermes owns it +├── memories/ sessions/ skills/ cron/ logs/ plugins/ +└── (runtime state) + +~/ # workingDirectory, your home by default +└── AGENTS.md # from the documents option +``` + ### Container Mode Same layout, mounted into the container: @@ -946,7 +1109,8 @@ Same layout, mounted into the container: cd /etc/nixos && nix flake update hermes-agent # Rebuild -sudo nixos-rebuild switch +sudo nixos-rebuild switch # for the NixOS module +home-manager switch # for the Home Manager module ``` In container mode, the `current-package` symlink is updated and the agent picks up the new binary on restart. No container recreation, no loss of installed packages. From 1dbe469276b11d53c74615da75c048f452638892 Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 18:00:56 -0400 Subject: [PATCH 026/426] refactor(ci): hoist docker detect-changes into the .py file The docker.yml gate held its own copy of the build formula, in shell. classify_changes.py now owns a derived docker lane, and the nix lane in the next commit derives from the same file. Two formulas in two languages drift apart, and one Python function with tests does not. --- .github/actions/detect-changes/action.yml | 3 +++ .github/workflows/docker.yml | 12 ++++++------ scripts/ci/classify_changes.py | 22 ++++++++++++++++------ tests/ci/test_classify_changes.py | 1 + 4 files changed, 26 insertions(+), 12 deletions(-) diff --git a/.github/actions/detect-changes/action.yml b/.github/actions/detect-changes/action.yml index 967e566878..9c8fb223e8 100644 --- a/.github/actions/detect-changes/action.yml +++ b/.github/actions/detect-changes/action.yml @@ -24,6 +24,9 @@ outputs: docker_meta: description: Docker setup and meta files have changed. value: ${{ steps.classify.outputs.docker_meta }} + docker: + description: Files included in the docker image have changed. + value: ${{ steps.classify.outputs.docker }} site: description: Build the Docusaurus docs site. value: ${{ steps.classify.outputs.site }} diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index a39c4150da..8c259fe872 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -52,14 +52,14 @@ jobs: - name: Decide whether to build id: gate env: - # python_prod (not python): the image copies installed code, never - # tests/, so tests-only PRs skip the build. - PYTHON_PROD: ${{ steps.classify.outputs.python_prod }} - FRONTEND: ${{ steps.classify.outputs.frontend }} - DOCKER_META: ${{ steps.classify.outputs.docker_meta }} + # The docker lane derives from python_prod (not python: the image + # copies installed code, never tests/, so tests-only PRs skip the + # build), frontend and docker_meta. classify_changes.py owns the + # formula so this gate and the nix lane cannot drift apart. + DOCKER: ${{ steps.classify.outputs.docker }} run: | set -euo pipefail - if [ "$PYTHON_PROD" = "true" ] || [ "$FRONTEND" = "true" ] || [ "$DOCKER_META" = "true" ]; then + if [ "$DOCKER" = "true" ]; then echo "build=true" >> "$GITHUB_OUTPUT" else echo "build=false" >> "$GITHUB_OUTPUT" diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py index 3e781233bc..c5e9db383b 100644 --- a/scripts/ci/classify_changes.py +++ b/scripts/ci/classify_changes.py @@ -14,6 +14,7 @@ Lanes: test suite. A tests-only PR keeps ``python`` (pytest must run) while skipping those product jobs. * ``docker_meta`` — Dockerfiles etc. +* ``docker`` — any product change + docker meta * ``frontend`` — TS typecheck matrix + desktop build. * ``site`` — Docusaurus + generated skill docs. * ``scan`` — supply-chain scan (Python files, .pth, setup hooks). @@ -127,16 +128,24 @@ def ci_review_files(files: list[str]) -> list[str]: def classify(files: list[str]) -> dict[str, bool]: """Map changed paths to ``{lane: should_run}``.""" files = [f.strip() for f in files if f.strip()] + python = any(not _py_irrelevant(f) for f in files) + python_prod = any(not _py_irrelevant(f) and not _py_test_only(f) for f in files) + frontend = any(f.startswith(_FRONTEND) or f in _ROOT_NPM for f in files) + deps = any(f == "pyproject.toml" for f in files) + npm_lock = any(f.split("/")[-1] == "package-lock.json" for f in files) + docker_meta = any(f.startswith(_DOCKER_META) for f in files) + ret = { - "python": any(not _py_irrelevant(f) for f in files), - "python_prod": any(not _py_irrelevant(f) and not _py_test_only(f) for f in files), - "docker_meta": any(f.startswith(_DOCKER_META) for f in files), - "frontend": any(f.startswith(_FRONTEND) or f in _ROOT_NPM for f in files), + "python": python, + "python_prod": python_prod, + "docker": docker_meta or python_prod or frontend, + "docker_meta": docker_meta, + "frontend": frontend, "site": any(f.startswith(_SITE) for f in files), "scan": any(_is_scan(f) for f in files), - "deps": any(f == "pyproject.toml" for f in files), + "deps": deps, "uv_lock": any(f in ("pyproject.toml", "uv.lock") for f in files), - "npm_lock": any(f.split("/")[-1] == "package-lock.json" for f in files), + "npm_lock": npm_lock, "installer": any(_is_installer(f) for f in files), "mcp_catalog": any(_is_mcp_catalog(f) for f in files), "ci_review": any(_is_ci_review(f) for f in files), @@ -144,6 +153,7 @@ def classify(files: list[str]) -> dict[str, bool]: if not files or any(f.startswith(".github/") for f in files): ret["python"] = True ret["python_prod"] = True + ret["docker"] = True ret["docker_meta"] = True ret["frontend"] = True ret["site"] = True diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py index 94ec212ee6..d1df6cdac2 100644 --- a/tests/ci/test_classify_changes.py +++ b/tests/ci/test_classify_changes.py @@ -25,6 +25,7 @@ DEFAULT = { "python": True, "python_prod": True, "frontend": True, + "docker": True, "docker_meta": True, "site": True, "scan": True, From 00c38728824e7e5d8120a229eb82fc94df44095c Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 18:00:56 -0400 Subject: [PATCH 027/426] feat(ci): add nix flake check as unrequired job The workflow owns its triggers and ci.yml does not call it. A reusable-workflow call holds the caller run in progress for the full build, and GitHub refuses `gh run rerun` on a run that is still in progress. A separate run reruns and cancels on its own. The job restores /nix/store from the GitHub Actions cache and saves from main only. A cache that a PR writes is visible to that PR alone, so a save there spends the quota of the repository and helps no later run. --- .github/actions/detect-changes/action.yml | 3 + .github/workflows/nix.yml | 117 ++++++++++++++++++++++ scripts/ci/classify_changes.py | 21 +++- tests/ci/test_classify_changes.py | 30 +++++- 4 files changed, 168 insertions(+), 3 deletions(-) create mode 100644 .github/workflows/nix.yml diff --git a/.github/actions/detect-changes/action.yml b/.github/actions/detect-changes/action.yml index 9c8fb223e8..b16500346f 100644 --- a/.github/actions/detect-changes/action.yml +++ b/.github/actions/detect-changes/action.yml @@ -27,6 +27,9 @@ outputs: docker: description: Files included in the docker image have changed. value: ${{ steps.classify.outputs.docker }} + nix: + description: Run `nix flake check` (flake inputs, or any product Python change). + value: ${{ steps.classify.outputs.nix }} site: description: Build the Docusaurus docs site. value: ${{ steps.classify.outputs.site }} diff --git a/.github/workflows/nix.yml b/.github/workflows/nix.yml new file mode 100644 index 0000000000..23627cbc9f --- /dev/null +++ b/.github/workflows/nix.yml @@ -0,0 +1,117 @@ +name: Nix flake check + +# Builds every output of the flake: the package, the devShell, and the 21 +# checks under nix/checks.nix — module evaluation, option parity, .env +# assembly, service argv, and the rest. +# +# This workflow owns its triggers and ci.yml does not call it, for the reason +# docker.yml gives: a reusable-workflow call holds the caller run in progress +# for the full build, and GitHub refuses `gh run rerun` on a run that is still +# in progress. One slow advisory job in the CI lane blocks every rerun of the +# fast required jobs beside it. A separate run reruns and cancels on its own. + +on: + pull_request: + push: + branches: [main] + +permissions: + contents: read + +# PR runs collapse to the newest commit. A push to main is never cancelled: +# each one saves the store cache that later PRs restore from, so cancelling a +# merge would leave the next PR to build from nothing. +concurrency: + group: nix-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +jobs: + # A `paths:` filter cannot gate this workflow correctly. The flake packages + # the product, and nine of the checks then run the built binary, so a change + # to hermes_cli/ alone can fail `nix flake check` without touching one file + # under nix/. The `nix` lane therefore follows python_prod as well as the + # flake inputs. On push the classifier fails open and every lane is true. + detect: + name: Detect affected areas + runs-on: ubuntu-latest + timeout-minutes: 10 + outputs: + nix: ${{ steps.classify.outputs.nix }} + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Detect affected areas + id: classify + uses: ./.github/actions/detect-changes + with: + github-token: ${{ github.token }} + + flake-check: + name: nix flake check + needs: [detect] + if: needs.detect.outputs.nix == 'true' + # The build compiles the package and its whole dependency closure, so this + # is minutes and not seconds when the cache misses. + runs-on: ubuntu-latest + timeout-minutes: 60 + steps: + - name: Checkout code + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Install Nix + uses: cachix/install-nix-action@630ae543ea3a38a9a4166f03376c02c50f408342 # v31.11.0 + with: + extra_nix_config: | + experimental-features = nix-command flakes + # A store path that does not substitute is a cache miss and not a + # build failure. Build it here instead. + fallback = true + # Each source archive is fetched one time in a run, and not one + # time for each evaluation. + tarball-ttl = 3600 + + # Restores /nix/store from the GitHub Actions cache. The store holds the + # whole dependency closure, so a hit turns a build of several minutes + # into a short evaluation. + # + # The Magic Nix Cache is not an option here. Its free tier ended in + # February 2025 with the GitHub cache API that it was built on. This + # action uses the current API and needs no account and no secret. + - name: Restore and save the Nix store + uses: nix-community/cache-nix-action@7df957e333c1e5da7721f60227dbba6d06080569 # v7 + with: + # The closure changes when the flake inputs change or when the + # dependencies of the project change. The key hashes both, so an + # edit to the source alone keeps the hit. + primary-key: nix-${{ runner.os }}-${{ hashFiles('flake.lock', 'nix/**', 'pyproject.toml', 'uv.lock') }} + # On a miss, restore the newest store for this runner. Most of the + # closure — Python, node, each transitive library — survives a bump + # of the lockfile, so an old store still removes most of the work. + restore-prefixes-first-match: nix-${{ runner.os }}- + + # Save from main only. A cache that a PR writes is visible to that + # PR alone and never to another branch, so a save there spends the + # 10 GB quota of the repository and helps no later run. A PR still + # restores: it reads the cache that the merge to main wrote. This is + # the same rule that docker.yml applies to `cache-to`. + save: ${{ github.event_name != 'pull_request' }} + + # Collect garbage before the save, so the store stays inside the + # 10 GB quota of the repository. Without a limit the store grows at + # each merge until GitHub removes the entry, and the next PR then + # gets nothing. This number is the size of the store and not the + # size of the compressed archive. + gc-max-store-size-linux: 5G + + # Delete the caches that this key replaces. GitHub removes caches by + # least recent use across the whole repository, so a Nix store that + # is never purged pushes out the caches of the other workflows. + purge: true + purge-prefixes: nix-${{ runner.os }}- + purge-created: 0 + purge-primary-key: never + + - name: nix flake check + # --print-build-logs: a check that fails then prints the assertion + # that failed, and not only the derivation that failed to build. + run: nix flake check --print-build-logs diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py index c5e9db383b..95a9bb13da 100644 --- a/scripts/ci/classify_changes.py +++ b/scripts/ci/classify_changes.py @@ -15,6 +15,7 @@ Lanes: skipping those product jobs. * ``docker_meta`` — Dockerfiles etc. * ``docker`` — any product change + docker meta +* ``nix`` — ``nix flake check``: the flake inputs and any product change. * ``frontend`` — TS typecheck matrix + desktop build. * ``site`` — Docusaurus + generated skill docs. * ``scan`` — supply-chain scan (Python files, .pth, setup hooks). @@ -37,6 +38,10 @@ must never skip one a change could break: or a frontend-only package; an unrecognized path keeps it on. * ``skills/`` (incl. ``SKILL.md``) is python-relevant — the skill-doc tests read that tree, so a doc-looking edit can still break Python. +* ``nix/``, ``flake.nix`` and ``flake.lock`` are the exception the other way: + only the flake reads them, so they skip the Python lanes and run ``nix`` + alone. ``pyproject.toml`` and ``uv.lock`` are flake inputs too, but the + packaging tests read them, so they keep every Python lane. """ from __future__ import annotations @@ -48,6 +53,8 @@ import sys _FRONTEND = ("ui-tui/", "web/", "apps/") # TS typecheck-matrix packages _ROOT_NPM = {"package.json", "package-lock.json"} # shifts every package's tree _DOCKER_META = ("docker/", ".hadolint.yml", "Dockerfile") # docker setup +_NIX_PATHS = ("nix/",) # nix files +_NIX_FILES = {"flake.nix", "flake.lock"} # base nix files _SITE = ("website/", "skills/", "optional-skills/") # docs site + skill pages # Prose/frontend trees that can't touch Python. skills/ is excluded on purpose. _PY_SKIP = ("docs/", "website/") + _FRONTEND @@ -84,8 +91,18 @@ def _is_docs(p: str) -> bool: return p.endswith((".md", ".mdx")) or p.startswith("docs/") or p.startswith("LICENSE") +def _is_nix(p: str) -> bool: + return p.startswith(_NIX_PATHS) or p in _NIX_FILES + + def _py_irrelevant(p: str) -> bool: - return _is_docs(p) or p in _ROOT_NPM or p.startswith(_PY_SKIP) or p.startswith(_DOCKER_META) + return ( + _is_docs(p) + or p in _ROOT_NPM + or p.startswith(_PY_SKIP) + or p.startswith(_DOCKER_META) + or _is_nix(p) + ) def _py_test_only(p: str) -> bool: @@ -149,6 +166,7 @@ def classify(files: list[str]) -> dict[str, bool]: "installer": any(_is_installer(f) for f in files), "mcp_catalog": any(_is_mcp_catalog(f) for f in files), "ci_review": any(_is_ci_review(f) for f in files), + "nix": python_prod or frontend or any(_is_nix(f) for f in files) } if not files or any(f.startswith(".github/") for f in files): ret["python"] = True @@ -162,6 +180,7 @@ def classify(files: list[str]) -> dict[str, bool]: ret["uv_lock"] = True ret["npm_lock"] = True ret["installer"] = True + ret["nix"] = True ret["ci_review"] = True # explicitly skip mcp catalog here. it's not needed unless those files are modified. diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py index d1df6cdac2..7c4bd0415e 100644 --- a/tests/ci/test_classify_changes.py +++ b/tests/ci/test_classify_changes.py @@ -27,6 +27,7 @@ DEFAULT = { "frontend": True, "docker": True, "docker_meta": True, + "nix": True, "site": True, "scan": True, "deps": True, @@ -38,12 +39,20 @@ DEFAULT = { } -def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_lock=False, npm_lock=False, installer=False, mcp_catalog=False, docker_meta=False, ci_review=False, python_prod=None) -> dict[str, bool]: +def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_lock=False, npm_lock=False, installer=False, mcp_catalog=False, docker_meta=False, ci_review=False, python_prod=None, nix=None, docker=None) -> dict[str, bool]: # python_prod tracks python except for tests-only diffs; default it to # python so the majority of cases don't need to spell it out. + # + # docker and nix are derived: both build the product, so both ride on + # python_prod and frontend. The image ships the built web assets, and the + # flake bundles the compiled ui-tui. Pass either explicitly to override. + _python_prod = python if python_prod is None else python_prod + _product = _python_prod or frontend return { "python": python, - "python_prod": python if python_prod is None else python_prod, + "python_prod": _python_prod, + "docker": (docker_meta or _product) if docker is None else docker, + "nix": _product if nix is None else nix, "frontend": frontend, "docker_meta": docker_meta, "site": site, @@ -77,6 +86,23 @@ CASES = { # skill edit must still run Python. "skill md → python + site": (["skills/github/SKILL.md"], _lanes(python=True, site=True)), "dockerfile → docker meta": (["Dockerfile"], _lanes(docker_meta=True)), + # Only the flake reads these, so they run nix alone. No Python test opens + # them, unlike pyproject.toml and uv.lock below. + "nix module → nix only": (["nix/homeManagerModules.nix"], _lanes(nix=True)), + "flake.nix → nix only": (["flake.nix"], _lanes(nix=True)), + "flake.lock → nix only": (["flake.lock"], _lanes(nix=True)), + # A flake-only file must not mask a Python change beside it. + "nix + python → both": (["nix/checks.nix", "agent/x.py"], _lanes(python=True, scan=True)), + # Nine checks run the built binary, so product Python is a nix input even + # when the diff touches no file under nix/. + "product python → nix": (["hermes_cli/config.py"], _lanes(python=True, scan=True)), + # tests/ is not packaged, so the built binary cannot change. + "tests-only → no nix": ( + ["tests/agent/test_foo.py"], + _lanes(python=True, python_prod=False, scan=True), + ), + # Prose cannot change the closure or the binary. + "docs-only → no nix": (["README.md"], _lanes()), # install.ps1 is a shell script Python never imports, but it's also not # provably prose, so python stays on (fail-open) alongside the Windows lane. "install.ps1 → installer": (["scripts/install.ps1"], _lanes(python=True, installer=True)), From 1f234a1033ca15be0aff851657f542af408ab2e7 Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 20:23:07 -0400 Subject: [PATCH 028/426] fix(nix): let the install-method stamp name a home-manager install detect_install_method reads the stamp against an allowlist. The allowlist held "nixos" but not "home-manager", and a stamp that names home-manager gave "unknown". The managed path (step 3) returned the correct name, so the gap was invisible: it appeared only for an install that carries a stamp. An install with the value "unknown" gets "hermes update" as its update guidance. That command is the one command a managed install refuses, so the user gets a dead end. The test for this was also environment-dependent. It called the real get_project_root(), and it passed here only because this worktree carries no stamp. A checkout from the curl installer carries a "git" stamp, and the assertion then failed for the contributor and not for us. The test now detects against a temporary install tree. The new test stamps each managed system and asserts the value that comes back. With the allowlist reverted, the home-manager case fails with "assert 'unknown' == 'home-manager'". The nixos case passes, because that name was already in the allowlist. --- hermes_cli/config.py | 8 +++-- .../hermes_cli/test_managed_install_shapes.py | 29 +++++++++++++++++-- 2 files changed, 33 insertions(+), 4 deletions(-) diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 15cfeaf354..d2d5c65e26 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -425,7 +425,8 @@ def _install_method_project_root(project_root: Optional[Path] = None) -> Path: def detect_install_method(project_root: Optional[Path] = None) -> str: - """Detect how Hermes was installed: 'apt', 'docker', 'nix', 'nixos', 'git', or 'unknown'. + """Detect how Hermes was installed: 'apt', 'docker', 'nix', 'nixos', + 'home-manager', 'git', or 'unknown'. Resolution order: 1. Code-scoped stamp ``/.install_method`` (next to the @@ -473,7 +474,10 @@ def detect_install_method(project_root: Optional[Path] = None) -> str: # generic Debian/Ubuntu APT signal. If another APT-managed distribution is # added, give it a distinct install method or make update-command selection # platform-aware instead of silently reusing Termux's `pkg` command. - supported_methods = {"apt", "docker", "nix", "nixos", "git", "unknown"} + # "home-manager" is here because step 3 can return it. A stamp must name + # every method that this function returns. Without it, the stamp of a + # home-manager install gives "unknown". + supported_methods = {"apt", "docker", "nix", "nixos", "home-manager", "git", "unknown"} # 1. Code-scoped stamp — authoritative, immune to shared $HERMES_HOME. try: diff --git a/tests/hermes_cli/test_managed_install_shapes.py b/tests/hermes_cli/test_managed_install_shapes.py index 89aad936a4..8a94b8c539 100644 --- a/tests/hermes_cli/test_managed_install_shapes.py +++ b/tests/hermes_cli/test_managed_install_shapes.py @@ -79,20 +79,45 @@ def test_env_var_wins_over_the_marker(hermes_home, monkeypatch): @pytest.mark.parametrize("managed_value", ["nixos", "home-manager"]) def test_managed_install_names_its_system_and_offers_an_update( - hermes_home, monkeypatch, managed_value + hermes_home, monkeypatch, tmp_path, managed_value ): """The message names the system, so the user knows what owns the install.""" monkeypatch.setenv("HERMES_MANAGED", managed_value) + # This test uses an install tree of its own. The real checkout can carry + # a stamp from the install shape of the contributor. A stamp answers + # first, and detection never reaches the managed state under test. + install_tree = tmp_path / "install" + install_tree.mkdir() + assert managed_value in config_mod.format_managed_message("set model") assert "set model" in config_mod.format_managed_message("set model") assert config_mod.get_managed_update_command() - assert config_mod.detect_install_method(config_mod.get_project_root()) == managed_value + assert config_mod.detect_install_method(install_tree) == managed_value # `hermes update` cannot run on a managed install, so the advice must not # name it. assert config_mod.recommended_update_command() != "hermes update" +@pytest.mark.parametrize("managed_value", ["nixos", "home-manager"]) +def test_a_stamp_can_name_every_managed_system( + hermes_home, monkeypatch, tmp_path, managed_value +): + """A stamp must give back every value that detection can return. + + Detection reads the stamp against an allowlist. A managed system that is + absent from that allowlist gives "unknown". The update guidance then + names a command that the managed guard refuses. + """ + monkeypatch.delenv("HERMES_MANAGED", raising=False) + install_tree = tmp_path / "install" + install_tree.mkdir() + + config_mod.stamp_install_method(managed_value, project_root=install_tree) + + assert config_mod.detect_install_method(install_tree) == managed_value + + def test_unmanaged_install_offers_no_update_command(hermes_home, monkeypatch): monkeypatch.delenv("HERMES_MANAGED", raising=False) assert config_mod.get_managed_update_command() is None From bd8b658a63346318491752fa2439d40bd6aadb8d Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 15:49:39 -0400 Subject: [PATCH 029/426] feat(clarify): accept a questions batch in the clarify tool core The clarify tool gets an optional questions parameter (2-5 independent questions, issue #18450). Batch-capable platform callbacks receive the normalized list in one call and reply with per-question answers. Legacy callbacks are looped one question at a time. The loop stops on timeout so the user is not asked the remaining questions after they walk away. Locked answers survive a timeout: the result carries them plus a timed_out flag, and unanswered entries have an empty user_response. The single-question path is byte-identical to the previous behavior. --- tests/tools/test_clarify_tool.py | 288 +++++++++++++++++++++++++++++++ tools/clarify_tool.py | 247 +++++++++++++++++++++++++- 2 files changed, 531 insertions(+), 4 deletions(-) diff --git a/tests/tools/test_clarify_tool.py b/tests/tools/test_clarify_tool.py index de721981df..7c40b63902 100644 --- a/tests/tools/test_clarify_tool.py +++ b/tests/tools/test_clarify_tool.py @@ -360,3 +360,291 @@ class TestRegistryMultiSelectPassThrough: )) assert seen["multi"] is False assert result["user_response"] == "a" + + +class TestClarifyBatchValidation: + """Validation of the `questions` batch parameter (issue #18450).""" + + def test_batch_takes_precedence_over_question(self): + """When both are present, `questions` wins and `question` is ignored.""" + seen = {} + + def cb(question, choices, multi_select=False, questions=None): + seen["questions"] = questions + return {"answers": {"q0": "blue"}} + + result = json.loads(clarify_tool( + "ignored single question", + questions=[{"question": "What color?"}], + callback=cb, + )) + assert "responses" in result + assert len(result["responses"]) == 1 + assert result["responses"][0]["question"] == "What color?" + assert seen["questions"][0]["question"] == "What color?" + + def test_batch_rejects_more_than_five(self): + result = json.loads(clarify_tool( + "", + questions=[{"question": f"Q{i}?"} for i in range(6)], + callback=lambda *a, **k: "", + )) + assert "error" in result + + def test_batch_rejects_blank_question_text(self): + result = json.loads(clarify_tool( + "", + questions=[{"question": "Real?"}, {"question": " "}], + callback=lambda *a, **k: "", + )) + assert "error" in result + + def test_batch_rejects_non_list(self): + result = json.loads(clarify_tool( + "", questions={"question": "Q?"}, callback=lambda *a, **k: "", + )) + assert "error" in result + + def test_batch_empty_list_falls_back_to_single_question(self): + """An empty questions array degrades to the single-question path.""" + def cb(question, choices): + assert question == "Single?" + return "yes" + + result = json.loads(clarify_tool("Single?", questions=[], callback=cb)) + assert result["user_response"] == "yes" + assert "responses" not in result + + def test_batch_choices_flattened_capped_and_labelled_per_question(self): + """Each question gets the full choice pipeline: flatten, cap, label.""" + seen = {} + + def cb(question, choices, multi_select=False, questions=None): + seen["questions"] = questions + return {"answers": {"q0": "a", "q1": "Loose layout"}} + + clarify_tool( + "", + questions=[ + {"question": "Pick letter", "choices": ["a", "b", "c", "d", "e", "f"]}, + {"question": "Pick layout", "choices": [ + {"description": "Loose layout"}, "Tight", + ]}, + ], + callback=cb, + ) + q0, q1 = seen["questions"] + assert len(q0["choices"]) == MAX_CHOICES + assert q0["choices"][0] == "a (Recommended)" + assert q1["choices"] == ["Loose layout (Recommended)", "Tight"] + + def test_batch_internal_ids_are_stable_and_model_id_echoed(self): + """Wire ids are q0..qN. A model-supplied id only shows in results.""" + seen = {} + + def cb(question, choices, multi_select=False, questions=None): + seen["questions"] = questions + return {"answers": {"q0": "A", "q1": "B"}} + + result = json.loads(clarify_tool( + "", + questions=[ + {"id": "approach", "question": "Which approach?"}, + {"question": "Timeline?"}, + ], + callback=cb, + )) + assert [q["qid"] for q in seen["questions"]] == ["q0", "q1"] + assert result["responses"][0]["id"] == "approach" + assert "id" not in result["responses"][1] + + def test_batch_multi_select_needs_choices(self): + """multi_select is only honored when the question has choices.""" + seen = {} + + def cb(question, choices, multi_select=False, questions=None): + seen["questions"] = questions + return {"answers": {"q0": "free text"}} + + clarify_tool( + "", + questions=[{"question": "Thoughts?", "multi_select": True}], + callback=cb, + ) + assert seen["questions"][0]["multi_select"] is False + + +class TestClarifyBatchDispatch: + """Batch-capable callbacks get the list once. Legacy callbacks loop.""" + + def test_batch_callback_receives_list_once(self): + calls = [] + + def cb(question, choices, multi_select=False, questions=None): + calls.append(questions) + return {"answers": {"q0": "x", "q1": "y"}} + + result = json.loads(clarify_tool( + "", + questions=[{"question": "One?"}, {"question": "Two?"}], + callback=cb, + )) + assert len(calls) == 1 + assert [r["user_response"] for r in result["responses"]] == ["x", "y"] + + def test_batch_callback_json_string_response(self): + """A _block-style bridge returns the answers as a JSON string.""" + def cb(question, choices, multi_select=False, questions=None): + return json.dumps({"answers": {"q0": "picked"}}) + + result = json.loads(clarify_tool( + "", questions=[{"question": "One?"}], callback=cb, + )) + assert result["responses"][0]["user_response"] == "picked" + + def test_batch_recommended_label_stripped_per_question(self): + def cb(question, choices, multi_select=False, questions=None): + return {"answers": {"q0": questions[0]["choices"][0]}} + + result = json.loads(clarify_tool( + "", + questions=[{"question": "Pick", "choices": ["Rebase", "Merge"]}], + callback=cb, + )) + assert result["responses"][0]["user_response"] == "Rebase" + assert result["responses"][0]["choices_offered"] == ["Rebase", "Merge"] + + def test_batch_multi_select_answer_parsed_to_list(self): + def cb(question, choices, multi_select=False, questions=None): + return {"answers": {"q0": '["red", "blue"]'}} + + result = json.loads(clarify_tool( + "", + questions=[{ + "question": "Colors?", + "choices": ["red", "blue", "green"], + "multi_select": True, + }], + callback=cb, + )) + assert result["responses"][0]["user_response"] == ["red", "blue"] + + def test_batch_timed_out_flag_passthrough_with_partials(self): + """Timeout keeps the locked answers and sets the top-level flag.""" + def cb(question, choices, multi_select=False, questions=None): + return {"answers": {"q0": "kept"}, "timed_out": True} + + result = json.loads(clarify_tool( + "", + questions=[{"question": "One?"}, {"question": "Two?"}], + callback=cb, + )) + assert result["timed_out"] is True + assert result["responses"][0]["user_response"] == "kept" + assert result["responses"][1]["user_response"] == "" + + def test_batch_empty_response_is_skip_not_timeout(self): + """A cancel-all resolves every answer empty with no timed_out flag.""" + def cb(question, choices, multi_select=False, questions=None): + return "" + + result = json.loads(clarify_tool( + "", questions=[{"question": "One?"}], callback=cb, + )) + assert result["responses"][0]["user_response"] == "" + assert "timed_out" not in result + + def test_legacy_callback_gets_sequential_calls_in_order(self): + """A callback without `questions` support is looped per question.""" + calls = [] + + def legacy_cb(question, choices, multi_select=False): + calls.append((question, tuple(choices or []) or None, multi_select)) + return f"answer to {question}" + + result = json.loads(clarify_tool( + "", + questions=[ + {"question": "One?", "choices": ["a", "b"]}, + {"question": "Two?"}, + ], + callback=legacy_cb, + )) + assert [c[0] for c in calls] == ["One?", "Two?"] + assert calls[0][1] == ("a (Recommended)", "b") + assert calls[1][1] is None + assert [r["user_response"] for r in result["responses"]] == [ + "answer to One?", "answer to Two?", + ] + assert "timed_out" not in result + + def test_legacy_loop_aborts_on_timeout_and_keeps_partials(self): + """The loop stops on the first timeout. Collected answers survive.""" + from tools.clarify_tool import TIMEOUT_RESPONSE + calls = [] + + def legacy_cb(question, choices): + calls.append(question) + if len(calls) == 2: + return TIMEOUT_RESPONSE + return "answered" + + result = json.loads(clarify_tool( + "", + questions=[ + {"question": "One?"}, {"question": "Two?"}, {"question": "Three?"}, + ], + callback=legacy_cb, + )) + assert calls == ["One?", "Two?"] + assert result["timed_out"] is True + assert [r["user_response"] for r in result["responses"]] == [ + "answered", "", "", + ] + + def test_legacy_loop_skip_continues(self): + """An explicit empty answer is a skip. The loop continues.""" + calls = [] + + def legacy_cb(question, choices): + calls.append(question) + return "" if len(calls) == 1 else "second" + + result = json.loads(clarify_tool( + "", + questions=[{"question": "One?"}, {"question": "Two?"}], + callback=legacy_cb, + )) + assert calls == ["One?", "Two?"] + assert [r["user_response"] for r in result["responses"]] == ["", "second"] + assert "timed_out" not in result + + def test_single_question_result_shape_unchanged(self): + """No `questions` arg keeps the historic result keys exactly.""" + def cb(question, choices): + return "blue" + + result = json.loads(clarify_tool( + "Color?", choices=["red", "blue"], callback=cb, + )) + assert set(result.keys()) == {"question", "choices_offered", "user_response"} + + +class TestRegistryBatchPassThrough: + """The registered handler forwards `questions` from tool args.""" + + def test_handler_passes_questions(self): + from tools.registry import registry + entry = registry.get_entry("clarify") + seen = {} + + def cb(question, choices, multi_select=False, questions=None): + seen["questions"] = questions + return {"answers": {"q0": "yes"}} + + result = json.loads(entry.handler( + {"questions": [{"question": "Go?"}]}, + callback=cb, + )) + assert seen["questions"][0]["question"] == "Go?" + assert result["responses"][0]["user_response"] == "yes" diff --git a/tools/clarify_tool.py b/tools/clarify_tool.py index d4bc8abeb1..d90930306e 100644 --- a/tools/clarify_tool.py +++ b/tools/clarify_tool.py @@ -15,13 +15,25 @@ a thin dispatcher that delegates to a platform-provided callback. """ import json -from typing import List, Optional, Callable +from typing import Dict, List, Optional, Callable # Maximum number of predefined choices the agent can offer. # A 5th "Other (type your answer)" option is always appended by the UI. MAX_CHOICES = 4 +# Maximum number of independent questions in one batch clarify call. +MAX_QUESTIONS = 5 + +# Canonical timeout sentinel returned to the agent when the user never +# answers. The CLI has always returned this exact text; the batch fallback +# loop also recognises it (alongside ``None``) as "the user walked away", +# which aborts the remaining questions instead of pestering one by one. +TIMEOUT_RESPONSE = ( + "The user did not provide a response within the time limit. " + "Use your best judgement to make the choice and proceed." +) + # Suffix appended to the first choice so the user can see, at a glance, which # option the agent actually recommends. Applied here rather than per-surface so # CLI, TUI, desktop, and messaging adapters all render the same label. @@ -150,10 +162,175 @@ def _parse_multi_select_response(raw_response) -> List[str]: return [s.strip() for s in raw.split(",") if s.strip()] +# ============================================================================= +# Batch (multi-question) support — issue #18450 +# ============================================================================= + +def _normalize_questions(questions) -> tuple: + """Validate and normalize the ``questions`` batch parameter. + + Returns ``(normalized, error)`` where exactly one is non-None, except the + empty-list case which returns ``(None, None)`` — an empty array is not an + error, it just means "no batch here" and the caller falls back to the + single-question path. + + Each normalized entry carries: + - ``qid``: stable wire id (``q0``..``qN``, index order). Surfaces key + their per-question answers by this; a model-supplied ``id`` is NOT + used on the wire (it's unvalidated text) and only echoed in results. + - ``id``: the model's optional identifier, or None. + - ``question``: stripped question text. + - ``choices``: decorated choice list (recommended label applied), or + None for open-ended. + - ``choices_offered``: the bare list as offered, for the result JSON. + - ``multi_select``: honored only when choices exist. + """ + if not isinstance(questions, list): + return None, "questions must be an array of question objects." + if not questions: + return None, None + if len(questions) > MAX_QUESTIONS: + return None, f"questions supports at most {MAX_QUESTIONS} items." + + normalized = [] + for index, item in enumerate(questions): + if isinstance(item, str): + # Tolerate bare-string items: LLMs sometimes send ["Q1?", "Q2?"]. + item = {"question": item} + if not isinstance(item, dict): + return None, f"questions[{index}] must be an object with a 'question'." + + text = str(item.get("question") or "").strip() + if not text: + return None, f"questions[{index}].question must be non-empty text." + + choices = item.get("choices") + if choices is not None: + if not isinstance(choices, list): + return None, f"questions[{index}].choices must be a list." + choices = [s for s in (_flatten_choice(c) for c in choices) if s] + if len(choices) > MAX_CHOICES: + choices = choices[:MAX_CHOICES] + if not choices: + choices = None + + model_id = str(item.get("id") or "").strip() or None + + normalized.append({ + "qid": f"q{index}", + "id": model_id, + "question": text, + "choices": mark_recommended(list(choices)) if choices else None, + "choices_offered": list(choices) if choices else None, + "multi_select": bool(item.get("multi_select")) and bool(choices), + }) + + return normalized, None + + +def _callback_accepts_questions(callback) -> bool: + """True when the platform callback understands the ``questions`` kwarg. + + Same signature-inspection approach as ``_invoke_callback`` (never a + TypeError retry — that would re-prompt the user on an internal bug). + """ + import inspect + + try: + params = inspect.signature(callback).parameters + return "questions" in params or any( + p.kind == inspect.Parameter.VAR_KEYWORD for p in params.values() + ) + except (TypeError, ValueError): + return False + + +def _clean_batch_answer(entry: dict, raw) -> object: + """Strip presentation from one locked answer (label, multi-select JSON).""" + if entry["multi_select"]: + return [strip_recommended(r) for r in _parse_multi_select_response(raw)] + return strip_recommended(raw) + + +def _batch_result(normalized: List[dict], answers: dict, timed_out: bool) -> str: + """Assemble the batch result JSON from per-qid answers. + + Unanswered questions surface as empty ``user_response`` — with the + top-level ``timed_out`` flag (present only when true) telling the agent + whether those blanks are deliberate skips or the user walking away. + """ + responses = [] + for entry in normalized: + row = {} + if entry["id"]: + row["id"] = entry["id"] + row["question"] = entry["question"] + row["choices_offered"] = entry["choices_offered"] + raw = answers.get(entry["qid"]) + row["user_response"] = _clean_batch_answer(entry, raw) if raw else "" + responses.append(row) + + result: Dict[str, object] = {"responses": responses} + if timed_out: + result["timed_out"] = True + return json.dumps(result, ensure_ascii=False) + + +def _run_batch(normalized: List[dict], callback, question: str) -> str: + """Dispatch a validated batch to the platform callback. + + Batch-capable callbacks (a ``questions`` kwarg, detected by signature) + get the whole list once and reply with ``{"answers": {qid: raw}}`` plus + an optional ``timed_out`` flag — as a dict or a JSON string (the + tui_gateway ``_block`` bridge can only carry strings). + + Legacy callbacks are looped one question at a time (messaging adapters, + older plugins). An explicit empty answer is a skip and the loop + continues; a timeout (``None`` or the ``TIMEOUT_RESPONSE`` sentinel) + means the user walked away, so the loop aborts instead of pestering + them with the remaining questions. Answers collected before the abort + are kept either way. + """ + if _callback_accepts_questions(callback): + raw = callback(question, None, questions=normalized) + + answers: dict = {} + timed_out = False + if raw is None or (isinstance(raw, str) and raw.strip() == TIMEOUT_RESPONSE): + timed_out = True + elif isinstance(raw, dict): + answers = dict(raw.get("answers") or {}) + timed_out = bool(raw.get("timed_out")) + elif isinstance(raw, str) and raw.strip(): + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + parsed = None + if isinstance(parsed, dict): + answers = dict(parsed.get("answers") or {}) + timed_out = bool(parsed.get("timed_out")) + # Any other falsy/unparseable reply is a cancel-all: every answer + # empty, no timeout flag (mirrors the single-question skip). + return _batch_result(normalized, answers, timed_out) + + answers = {} + timed_out = False + for entry in normalized: + raw = _invoke_callback( + callback, entry["question"], entry["choices"], entry["multi_select"], + ) + if raw is None or (isinstance(raw, str) and raw.strip() == TIMEOUT_RESPONSE): + timed_out = True + break + answers[entry["qid"]] = raw + return _batch_result(normalized, answers, timed_out) + + def clarify_tool( question: str, choices: Optional[List[str]] = None, multi_select: bool = False, + questions: Optional[List[dict]] = None, callback: Optional[Callable] = None, ) -> str: """ @@ -167,16 +344,40 @@ def clarify_tool( (checkboxes). The ``user_response`` in the output JSON will be a list of strings instead of a single string. Has no effect when ``choices`` is omitted. + questions: Up to 5 independent questions asked as one batch + (issue #18450). Each item: ``{id?, question, choices?, + multi_select?}``. When present (non-empty), the single + ``question``/``choices``/``multi_select`` parameters + are ignored and the result JSON is ``{"responses": + [...]}`` (plus ``"timed_out": true`` when the user + stopped answering partway). callback: Platform-provided function that handles the actual UI interaction. Signature: ``callback(question, choices, multi_select=False) -> str``. - The optional ``multi_select`` keyword is passed so the - platform can render checkboxes instead of radio buttons. + Batch-capable platforms additionally accept a + ``questions`` keyword and receive the normalized list + in one call; platforms without it are looped one + question at a time. Injected by the agent runner (cli.py / gateway). Returns: - JSON string with the user's response. + JSON string with the user's response(s). """ + if questions is not None: + normalized, error = _normalize_questions(questions) + if error: + return tool_error(error) + if normalized: + if callback is None: + return tool_error( + "Clarify tool is not available in this execution context." + ) + try: + return _run_batch(normalized, callback, str(question or "").strip()) + except Exception as exc: + return tool_error(f"Failed to get user input: {exc}") + # Empty questions array → fall through to the single-question path. + if not question or not question.strip(): return tool_error("Question text is required.") @@ -296,6 +497,43 @@ CLARIFY_SCHEMA = { "Has no effect when choices is omitted (open-ended question)." ), }, + "questions": { + "type": "array", + "maxItems": MAX_QUESTIONS, + "description": ( + "Ask 2-5 INDEPENDENT questions in one call instead of " + "several sequential clarify calls — the user answers them " + "on one form, in any order. Each item has its own " + "question/choices/multi_select (same rules as the " + "top-level parameters); optional `id` is echoed back in " + "the matching response. When set, the top-level question/" + "choices are ignored and the result is {responses: [...]}, " + "with `timed_out: true` added if the user stopped part-way " + "(unanswered entries have an empty user_response). Only " + "batch questions that are truly independent — if one " + "answer would change another question, ask separately." + ), + "items": { + "type": "object", + "properties": { + "id": { + "type": "string", + "description": ( + "Optional short identifier echoed in the " + "matching response (e.g. 'approach')." + ), + }, + "question": {"type": "string"}, + "choices": { + "type": "array", + "items": {"type": "string"}, + "maxItems": MAX_CHOICES, + }, + "multi_select": {"type": "boolean"}, + }, + "required": ["question"], + }, + }, }, "required": ["question"], }, @@ -313,6 +551,7 @@ registry.register( question=args.get("question", ""), choices=args.get("choices"), multi_select=args.get("multi_select", False), + questions=args.get("questions"), callback=kw.get("callback")), check_fn=check_clarify_requirements, emoji="❓", From 879d6a4c78e9f9df7ef4e2946c8e222cdfda1e2b Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 16:00:19 -0400 Subject: [PATCH 030/426] feat(tui_gateway): batch clarify bridge with per-question locks One clarify.request carries the question list (qid, question, choices, multi_select per entry). clarify.respond gains an optional question_id: each respond locks one answer, a repeat respond overwrites it, and the batch resolves when every question is locked. A respond without question_id keeps its existing meaning (cancel the whole prompt). Locked answers survive the deadline: a timed-out batch returns the partial answer map with a timed_out flag instead of an empty string. The reconnect replay snapshot also carries the locked answers, so a reattached client restores its per-question state. Both agent-side clarify dispatch sites forward the questions arg. --- agent/agent_runtime_helpers.py | 1 + agent/tool_executor.py | 1 + tests/tui_gateway/test_protocol.py | 204 +++++++++++++++++++++++++++++ tui_gateway/server.py | 124 +++++++++++++++--- 4 files changed, 314 insertions(+), 16 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 96f7bb821a..5d87894a65 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -3200,6 +3200,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i question=next_args.get("question", ""), choices=next_args.get("choices"), multi_select=next_args.get("multi_select", False), + questions=next_args.get("questions"), callback=agent.clarify_callback, ), next_args, diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 381f1000e9..4fedb6dc7d 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2120,6 +2120,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe question=next_args.get("question", ""), choices=next_args.get("choices"), multi_select=next_args.get("multi_select", False), + questions=next_args.get("questions"), callback=agent.clarify_callback, ) function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( diff --git a/tests/tui_gateway/test_protocol.py b/tests/tui_gateway/test_protocol.py index 7b27ae6253..c2cc809591 100644 --- a/tests/tui_gateway/test_protocol.py +++ b/tests/tui_gateway/test_protocol.py @@ -354,6 +354,210 @@ def test_late_prompt_response_is_idempotent(server, method, value_key): assert response["result"] == {"status": "expired"} +# ── clarify batch (multi-question) bridge ──────────────────────────── + + +def _drain_batch_block(server, qids, timeout=5, payload=None): + """Run a batch _block on a worker thread and return (thread, result box, + emitted request payload). The caller resolves questions via + handle_request and then joins.""" + box = {} + + def run(): + box["answer"] = server._block( + "clarify.request", + "s1", + dict(payload or {"questions": [{"qid": q, "question": q} for q in qids]}), + timeout=timeout, + batch_qids=list(qids), + ) + + thread = threading.Thread(target=run, daemon=True) + thread.start() + # Wait for the request to be registered so respond calls can find it. + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + with server._prompt_lock: + if server._batch_clarify: + rid = next(iter(server._batch_clarify)) + return thread, box, rid + time.sleep(0.01) + raise AssertionError("batch clarify request never registered") + + +def test_clarify_batch_resolves_when_all_questions_locked(capture): + server, buf = capture + thread, box, rid = _drain_batch_block(server, ["q0", "q1"]) + + first = server.handle_request({ + "id": "a1", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q1", "answer": "beta"}, + }) + assert first["result"]["status"] == "ok" + assert first["result"]["remaining"] == ["q0"] + assert thread.is_alive() # one question left — still blocking + + second = server.handle_request({ + "id": "a2", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "alpha"}, + }) + assert second["result"]["status"] == "ok" + assert second["result"]["remaining"] == [] + + thread.join(timeout=5) + assert not thread.is_alive() + assert json.loads(box["answer"]) == {"answers": {"q0": "alpha", "q1": "beta"}} + + +def test_clarify_batch_answer_update_overwrites_before_completion(server): + thread, box, rid = _drain_batch_block(server, ["q0", "q1"]) + + server.handle_request({ + "id": "a1", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "first"}, + }) + server.handle_request({ + "id": "a2", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "changed"}, + }) + server.handle_request({ + "id": "a3", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q1", "answer": "done"}, + }) + + thread.join(timeout=5) + assert json.loads(box["answer"])["answers"]["q0"] == "changed" + + +def test_clarify_batch_empty_answer_is_a_locked_skip(server): + """Skipping one question locks an empty answer — it counts toward + completion instead of leaving the batch waiting.""" + thread, box, rid = _drain_batch_block(server, ["q0", "q1"]) + + server.handle_request({ + "id": "a1", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": ""}, + }) + server.handle_request({ + "id": "a2", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q1", "answer": "kept"}, + }) + + thread.join(timeout=5) + assert json.loads(box["answer"]) == {"answers": {"q0": "", "q1": "kept"}} + + +def test_clarify_batch_unknown_question_id_rejected(server): + thread, box, rid = _drain_batch_block(server, ["q0"]) + + response = server.handle_request({ + "id": "bad", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q9", "answer": "x"}, + }) + assert response["error"]["code"] == 4002 + + server.handle_request({ + "id": "ok", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "fine"}, + }) + thread.join(timeout=5) + + +def test_clarify_batch_timeout_keeps_locked_answers(capture): + """Locked answers survive the deadline: the tool sees the partials plus + timed_out instead of an empty string.""" + server, buf = capture + thread, box, rid = _drain_batch_block(server, ["q0", "q1"], timeout=1) + + server.handle_request({ + "id": "a1", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "kept"}, + }) + + thread.join(timeout=10) + assert not thread.is_alive() + result = json.loads(box["answer"]) + assert result == {"answers": {"q0": "kept"}, "timed_out": True} + # The expire notification still fires for the un-finished batch. + messages = [json.loads(line) for line in buf.getvalue().splitlines()] + assert any(m["params"]["type"] == "clarify.expire" for m in messages) + + +def test_clarify_batch_cancel_all_returns_empty(server): + """A respond without question_id cancels the whole batch (Esc path).""" + thread, box, rid = _drain_batch_block(server, ["q0", "q1"]) + + server.handle_request({ + "id": "cancel", "method": "clarify.respond", + "params": {"request_id": rid, "answer": ""}, + }) + + thread.join(timeout=5) + assert box["answer"] == "" + + +def test_clarify_batch_late_question_respond_is_idempotent(server): + response = server.handle_request({ + "id": "late", "method": "clarify.respond", + "params": {"request_id": "gone", "question_id": "q0", "answer": "x"}, + }) + assert response["result"] == {"status": "expired"} + + +def test_clarify_batch_state_cleared_after_resolution(server): + thread, box, rid = _drain_batch_block(server, ["q0"]) + server.handle_request({ + "id": "a", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "x"}, + }) + thread.join(timeout=5) + with server._prompt_lock: + assert rid not in server._batch_clarify + assert rid not in server._pending + + +def test_clarify_block_helper_builds_batch_payload(capture): + """_clarify_block forwards only wire fields (qid/question/choices/ + multi_select) — the tool-side normalized entries carry extra keys the + renderer must not see.""" + server, buf = capture + normalized = [ + { + "qid": "q0", "id": "approach", "question": "Which?", + "choices": ["a (Recommended)", "b"], "choices_offered": ["a", "b"], + "multi_select": False, + }, + ] + + box = {} + + def run(): + box["answer"] = server._clarify_block("s1", "", None, questions=normalized) + + thread = threading.Thread(target=run, daemon=True) + thread.start() + deadline = time.monotonic() + 2 + rid = None + while time.monotonic() < deadline and rid is None: + with server._prompt_lock: + rid = next(iter(server._batch_clarify), None) + time.sleep(0.01) + assert rid + + server.handle_request({ + "id": "a", "method": "clarify.respond", + "params": {"request_id": rid, "question_id": "q0", "answer": "a"}, + }) + thread.join(timeout=5) + + messages = [json.loads(line) for line in buf.getvalue().splitlines()] + request = messages[0]["params"] + assert request["type"] == "clarify.request" + sent = request["payload"]["questions"][0] + assert set(sent) == {"qid", "question", "choices", "multi_select"} + assert "id" not in sent and "choices_offered" not in sent + + def test_approval_pending_replays_unresolved_requests(server, monkeypatch): from tools import approval diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 70213607c7..d024c7e411 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -145,6 +145,10 @@ _methods: dict[str, callable] = {} _pending: dict[str, tuple[str, threading.Event]] = {} _pending_prompt_payloads: dict[str, tuple[str, dict]] = {} _answers: dict[str, str] = {} +# Batch clarify accumulators: rid → {"qids": [...], "answers": {qid: answer}}. +# Written by clarify.respond (per-question lock, update-in-place), read out by +# _block on resolution/timeout so locked answers survive the deadline. +_batch_clarify: dict[str, dict] = {} _db = None _db_error: str | None = None _stdout_lock = threading.Lock() @@ -1952,7 +1956,14 @@ def _pending_clarify_request_payload(sid: str) -> dict | None: continue event, prompt_payload = _pending_prompt_payloads.get(rid, ("", {})) if event == "clarify.request": - return dict(prompt_payload) + snapshot = dict(prompt_payload) + # Batch clarify: replay the answers locked so far, so a + # reconnecting client restores its per-question ✓ state + # instead of presenting every question as unanswered. + batch = _batch_clarify.get(rid) + if batch is not None and batch["answers"]: + snapshot["answers"] = dict(batch["answers"]) + return snapshot return None @@ -3469,16 +3480,28 @@ def _enable_gateway_prompts() -> None: # ── Blocking prompt factory ────────────────────────────────────────── -def _block(event: str, sid: str, payload: dict, timeout: float | None = 300) -> str: +def _block( + event: str, + sid: str, + payload: dict, + timeout: float | None = 300, + batch_qids: list[str] | None = None, +) -> str: rid = uuid.uuid4().hex[:8] ev = threading.Event() with _prompt_lock: _pending[rid] = (sid, ev) payload["request_id"] = rid _pending_prompt_payloads[rid] = (event, dict(payload)) + if batch_qids: + # Multi-question clarify: per-question answers accumulate here + # (update-in-place until every qid is locked). Locked answers + # survive a timeout — see the batch read-out below. + _batch_clarify[rid] = {"qids": list(batch_qids), "answers": {}} answered = False answer = "" answer_present = False + batch_answers: dict | None = None try: _emit(event, sid, payload) # Natural Event semantics: None → wait forever (clarify configured with @@ -3491,6 +3514,27 @@ def _block(event: str, sid: str, payload: dict, timeout: float | None = 300) -> _pending_prompt_payloads.pop(rid, None) answer_present = rid in _answers answer = _answers.pop(rid, "") + batch_state = _batch_clarify.pop(rid, None) + if batch_state is not None: + batch_answers = dict(batch_state["answers"]) + + if batch_qids is not None: + # Cancel-all (respond with no question_id) resolves via _answers with + # an empty string — that stays a plain cancel, not a partial result. + if answer_present: + return answer + result: dict[str, object] = {"answers": batch_answers or {}} + if not answered: + # Deadline hit: keep whatever was locked, tell the tool the rest + # are absences (not skips), and still fire the expire + # notification so live cards tear down. + result["timed_out"] = True + _emit( + f"{event.removesuffix('.request')}.expire", + sid, + {"request_id": rid}, + ) + return json.dumps(result, ensure_ascii=False) # Emit an `.expire` notification on timeout for every blocking request type # whose `*.respond` handler tolerates a late reply (allow_expired=True). @@ -3529,6 +3573,50 @@ def _clarify_timeout_seconds() -> float | None: return 300 +def _clarify_block(sid: str, q, c, multi_select=False, questions=None) -> str: + """Bridge the clarify tool callback onto _block. + + Single-question calls keep the exact historical payload shape (older + renderers never see a new field). Batch calls emit one clarify.request + carrying the question list — only wire fields (qid/question/choices/ + multi_select) are forwarded; the tool-side normalized entries also carry + result-assembly keys (id, choices_offered) the renderer must not see. + The tool decodes the JSON reply via its batch answer parser. + """ + if questions: + wire = [ + { + "qid": entry["qid"], + "question": entry["question"], + "choices": entry["choices"], + "multi_select": bool(entry["multi_select"]), + } + for entry in questions + ] + return _block( + "clarify.request", + sid, + {"questions": wire}, + timeout=_clarify_timeout_seconds(), + batch_qids=[entry["qid"] for entry in questions], + ) + # multi_select is a pass-through hint: renderers with checkbox + # support can honor it; older renderers ignore the extra field + # and stay single-select (a single answer still parses as a + # one-element list on the tool side). Only emitted when True so + # single-select payloads keep the exact pre-multi-select shape. + return _block( + "clarify.request", + sid, + ( + {"question": q, "choices": c, "multi_select": True} + if multi_select + else {"question": q, "choices": c} + ), + timeout=_clarify_timeout_seconds(), + ) + + def _clear_pending(sid: str | None = None) -> None: """Release pending prompts with an empty answer. @@ -6176,20 +6264,8 @@ def _agent_cbs(sid: str) -> dict: "notice_clear_callback": lambda key: _emit( "notification.clear", sid, {"key": key} ), - "clarify_callback": lambda q, c, multi_select=False: _block( - "clarify.request", - sid, - # multi_select is a pass-through hint: renderers with checkbox - # support can honor it; older renderers ignore the extra field - # and stay single-select (a single answer still parses as a - # one-element list on the tool side). Only emitted when True so - # single-select payloads keep the exact pre-multi-select shape. - ( - {"question": q, "choices": c, "multi_select": True} - if multi_select - else {"question": q, "choices": c} - ), - timeout=_clarify_timeout_seconds(), + "clarify_callback": lambda q, c, multi_select=False, questions=None: ( + _clarify_block(sid, q, c, multi_select=multi_select, questions=questions) ), # read_terminal tool (desktop GUI): same blocking bridge as clarify — the # renderer answers terminal.read.respond with the serialized buffer. @@ -11711,6 +11787,7 @@ def _stage_session_file_attachment( def _respond(rid, params, key, *, allow_expired=False): r = params.get("request_id", "") + question_id = str(params.get("question_id") or "") with _prompt_lock: entry = _pending.get(r) if not entry: @@ -11718,6 +11795,21 @@ def _respond(rid, params, key, *, allow_expired=False): return _ok(rid, {"status": "expired"}) return _err(rid, 4009, f"no pending {key} request") _, ev = entry + batch = _batch_clarify.get(r) + if batch is not None and question_id: + # Per-question lock (multi-question clarify). Update-in-place is + # deliberate: a locked answer stays editable until the batch + # completes, and completion is exactly "every qid locked" — the + # final lock is the Confirm-and-continue click. + if question_id not in batch["qids"]: + return _err(rid, 4002, f"unknown question_id {question_id!r}") + batch["answers"][question_id] = params.get(key, "") + remaining = [ + qid for qid in batch["qids"] if qid not in batch["answers"] + ] + if not remaining: + ev.set() + return _ok(rid, {"status": "ok", "remaining": remaining}) _answers[r] = params.get(key, "") ev.set() return _ok(rid, {"status": "ok"}) From e0d8a2eb89f6f5744dcfbf3a703a50ed5f3ec85f Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 18 Aug 2026 16:12:58 -0400 Subject: [PATCH 031/426] feat(desktop): multi-question clarify card with per-question locks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The clarify card renders every batch question at once. Answers stage locally per question; the footer button locks the staged answer with a clarify.respond keyed by question_id. Locked answers stay editable — a new pick un-locks the row and a re-lock overwrites server-side. When exactly one question is unanswered the button relabels to Confirm and continue, and that final lock completes the batch. Skip cancels the whole batch (no question_id). Reconnect replay seeds the locked map so a reattached window restores its earlier state. The settled card lists every question with its answer; blank answers render as Skipped. Single-question cards are untouched. --- .../hooks/use-message-stream/gateway-event.ts | 61 ++- .../assistant-ui/clarify-tool.test.tsx | 183 ++++++- .../components/assistant-ui/clarify-tool.tsx | 448 +++++++++++++++++- apps/desktop/src/i18n/ar.ts | 5 +- apps/desktop/src/i18n/en.ts | 3 + apps/desktop/src/i18n/ja.ts | 3 + apps/desktop/src/i18n/types.ts | 3 + apps/desktop/src/i18n/zh-hant.ts | 3 + apps/desktop/src/i18n/zh.ts | 3 + apps/desktop/src/lib/chat-messages.ts | 4 + apps/desktop/src/store/clarify.test.ts | 48 ++ apps/desktop/src/store/clarify.ts | 54 +++ 12 files changed, 807 insertions(+), 11 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 8ce770e3b2..a4d00bb0a7 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -25,7 +25,7 @@ import { invalidateSlashCompletions } from '@/lib/slash-completion-cache' import { type AgentNoticePayload, clearAgentNotice, nativeNoticeInput, showAgentNotice } from '@/store/agent-notices' import { reconcileApprovalModeForProfile } from '@/store/approval-mode' import { billingCtaLabel, clearBillingBlock, runBillingRecovery, setBillingBlock } from '@/store/billing-block' -import { clearClarifyRequest, normalizeChoices, setClarifyRequest, warnDroppedChoices } from '@/store/clarify' +import { clearClarifyRequest, normalizeChoices, normalizeQuestions, setClarifyRequest, warnDroppedChoices } from '@/store/clarify' import { setSessionCompacting } from '@/store/compaction' import { refreshBackgroundProcesses } from '@/store/composer-status' import { $gateway, activeGatewayConnectionId } from '@/store/gateway' @@ -1165,8 +1165,65 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { const rawChoices = payload?.choices const choices = normalizeChoices(rawChoices) const multiSelect = payload?.multi_select === true + // Batch (multi-question) clarify: `questions` replaces question/choices + // on the wire. `answers` rides along only on reconnect replay, carrying + // the per-question locks the server already accepted. + const questions = normalizeQuestions(payload?.questions) + const lockedAnswers = + typeof payload?.answers === 'object' && payload?.answers !== null + ? Object.fromEntries( + Object.entries(payload.answers as Record).filter( + (entry): entry is [string, string] => typeof entry[1] === 'string' + ) + ) + : undefined - if (requestId && question) { + if (requestId && questions.length > 0) { + setClarifyRequest({ + choices: null, + lockedAnswers, + multiSelect: false, + question: '', + questions, + requestId, + sessionId: sessionId ?? null + }) + + if (sessionId) { + // Same hydration-race guard as the single-question path below: the + // form mounts from the tool row, so upsert a stable one keyed by + // the request id in case tool.start was missed. + upsertToolCall( + sessionId, + { + args: { + questions: questions.map(q => ({ + choices: q.choices ?? undefined, + multi_select: q.multiSelect || undefined, + question: q.question + })) + }, + name: 'clarify', + tool_id: requestId + }, + 'running', + event.type, + occurredAt + ) + updateSessionState(sessionId, state => ({ ...state, needsInput: true })) + + if (sessionId === activeSessionIdRef.current) { + requestScrollToBottom() + } + } + + dispatchNativeNotification({ + body: questions.map(q => q.question).join(' · '), + kind: 'input', + sessionId, + title: translateNow('notifications.native.inputTitle') + }) + } else if (requestId && question) { if (rawChoices != null && choices.length === 0) { warnDroppedChoices('gateway', question, rawChoices) } diff --git a/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx b/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx index a0d616633c..e78158a2eb 100644 --- a/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx +++ b/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx @@ -9,7 +9,7 @@ import { clearClarifyRequest, setClarifyRequest } from '@/store/clarify' import { $gateway } from '@/store/gateway' import { $activeSessionId } from '@/store/session' -import { ClarifyTool, readClarifyResult } from './clarify-tool' +import { ClarifyTool, readClarifyBatchResult, readClarifyResult } from './clarify-tool' // The live pending card only renders while its message is running. Force that so // keyboard-navigation tests can exercise ClarifyToolPending directly. @@ -448,3 +448,184 @@ describe('ClarifyTool pending marker', () => { expect(document.querySelector('[data-clarify-choices]')).toBeNull() }) }) + +// ─── Batch (multi-question) clarify ───────────────────────────────────────── + +function batchArgs(): { questions: { question: string; choices?: string[] }[] } { + return { + questions: [ + { choices: ['red', 'blue'], question: 'Color?' }, + { question: 'Name?' } + ] + } +} + +function liveBatchProps(): ToolCallMessagePartProps { + const args = batchArgs() + + return { + addResult: vi.fn(), + args, + argsText: JSON.stringify(args), + isError: false, + respondToApproval: vi.fn(), + result: undefined, + resume: vi.fn(), + status: { type: 'running' }, + toolCallId: 'clarify-batch', + toolName: 'clarify', + type: 'tool-call' + } +} + +function renderLiveBatch(lockedAnswers?: Record) { + const request = vi.fn().mockResolvedValue({ ok: true, remaining: [] }) + + $activeSessionId.set('session-1') + $gateway.set({ request } as never) + setClarifyRequest({ + choices: null, + lockedAnswers, + multiSelect: false, + question: '', + questions: [ + { choices: ['red', 'blue'], multiSelect: false, qid: 'q0', question: 'Color?' }, + { choices: null, multiSelect: false, qid: 'q1', question: 'Name?' } + ], + requestId: 'request-batch', + sessionId: 'session-1' + }) + renderClarify() + + return request +} + +describe('readClarifyBatchResult', () => { + it('parses responses with string and list answers plus timed_out', () => { + const parsed = readClarifyBatchResult( + JSON.stringify({ + responses: [ + { question: 'Color?', user_response: 'red' }, + { question: 'Tools?', user_response: ['a', 'b'] }, + { question: 'Name?', user_response: '' } + ], + timed_out: true + }) + ) + + expect(parsed.timedOut).toBe(true) + expect(parsed.responses).toHaveLength(3) + expect(parsed.responses[1]?.answer).toEqual(['a', 'b']) + expect(parsed.responses[2]?.answer).toBe('') + }) + + it('returns empty responses for single-question payloads', () => { + expect(readClarifyBatchResult({ question: 'Q?', user_response: 'a' }).responses).toEqual([]) + }) +}) + +describe('ClarifyTool batch card', () => { + it('renders every question at once', () => { + renderLiveBatch() + + expect(screen.getByText('Color?')).toBeTruthy() + expect(screen.getByText('Name?')).toBeTruthy() + expect(screen.getByText('0 of 2 answered')).toBeTruthy() + }) + + it('locks a picked choice via Continue, keyed by qid', async () => { + const request = renderLiveBatch() + + fireEvent.click(screen.getByRole('button', { name: /red/ })) + fireEvent.submit(document.querySelector('form') as HTMLFormElement) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: 'red', + question_id: 'q0', + request_id: 'request-batch' + }) + }) + + await waitFor(() => { + expect(screen.getByText('1 of 2 answered')).toBeTruthy() + }) + }) + + it('answers in any order: free-text question first', async () => { + const request = renderLiveBatch() + + const nameBox = screen.getByPlaceholderText('Type your answer…') + fireEvent.change(nameBox, { target: { value: 'packet' } }) + fireEvent.submit(document.querySelector('form') as HTMLFormElement) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: 'packet', + question_id: 'q1', + request_id: 'request-batch' + }) + }) + }) + + it('relabels the button to Confirm and continue when one question remains', async () => { + renderLiveBatch({ q1: 'already locked' }) + + expect(screen.getByText('1 of 2 answered')).toBeTruthy() + expect(screen.getByRole('button', { name: /Confirm and continue/ })).toBeTruthy() + }) + + it('re-staging a locked answer un-locks it and a re-lock overwrites', async () => { + const request = renderLiveBatch({ q0: 'red' }) + + // q0 arrived locked from replay. Picking blue un-locks it locally… + fireEvent.click(screen.getByRole('button', { name: /blue/ })) + expect(screen.getByText('0 of 2 answered')).toBeTruthy() + + // …and Continue re-locks with the new answer. + fireEvent.submit(document.querySelector('form') as HTMLFormElement) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: 'blue', + question_id: 'q0', + request_id: 'request-batch' + }) + }) + }) + + it('Skip cancels the whole batch without a question_id', async () => { + const request = renderLiveBatch() + + fireEvent.click(screen.getByRole('button', { name: 'Skip' })) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: '', + request_id: 'request-batch' + }) + }) + }) + + it('renders the settled batch with all questions and answers', () => { + renderClarify( + + ) + + expect(screen.getByText('Color?')).toBeTruthy() + expect(screen.getByText('red')).toBeTruthy() + expect(screen.getByText('Name?')).toBeTruthy() + expect(screen.getByText('Skipped')).toBeTruthy() + }) +}) diff --git a/apps/desktop/src/components/assistant-ui/clarify-tool.tsx b/apps/desktop/src/components/assistant-ui/clarify-tool.tsx index 27dd58fa53..b3584ccb4d 100644 --- a/apps/desktop/src/components/assistant-ui/clarify-tool.tsx +++ b/apps/desktop/src/components/assistant-ui/clarify-tool.tsx @@ -27,6 +27,8 @@ import { CircleLetterA, Loader2, MessageQuestion } from '@/lib/icons' import { cn } from '@/lib/utils' import { bareChoice, + type ClarifyQuestion, + type ClarifyRequest, clearClarifyRequest, normalizeChoices, RECOMMENDED_LABEL, @@ -43,6 +45,7 @@ interface ClarifyArgs { question?: string choices?: string[] | null multiSelect?: boolean + questions?: { question: string; choices?: string[] | null; multiSelect?: boolean }[] } interface ClarifyResult { @@ -72,13 +75,74 @@ function readClarifyArgs(args: unknown): ClarifyArgs { warnDroppedChoices('tool_args', question, rawChoices) } + // Batch form: tool args carry the model's questions array. Entries are + // normalized leniently here (qid comes from the gateway request, not args). + let questions: ClarifyArgs['questions'] + + if (Array.isArray(row.questions)) { + const parsed = row.questions + .map(entry => { + const item = parseMaybeObject(entry) + const text = stringField(item, 'question') + + if (!text) { + return null + } + + const itemChoices = normalizeChoices(item.choices) + + return { + choices: itemChoices.length > 0 ? itemChoices : null, + multiSelect: item.multi_select === true && itemChoices.length > 0, + question: text + } + }) + .filter((entry): entry is NonNullable => entry !== null) + + if (parsed.length > 0) { + questions = parsed + } + } + return { question, choices: choices.length > 0 ? choices : null, - multiSelect: row.multi_select === true + multiSelect: row.multi_select === true, + questions } } +interface ClarifyBatchResponse { + id?: string + question?: string + answer?: string | string[] +} + +/** Parse batch clarify tool JSON (`responses` array + optional timed_out). */ +export function readClarifyBatchResult(result: unknown): { + responses: ClarifyBatchResponse[] + timedOut: boolean +} { + const row = parseMaybeObject(result) + + if (!Array.isArray(row.responses)) { + return { responses: [], timedOut: false } + } + + const responses = row.responses.map((entry): ClarifyBatchResponse => { + const item = parseMaybeObject(entry) + const answer = item.user_response + + return { + answer: Array.isArray(answer) ? answer.map(String) : typeof answer === 'string' ? answer : undefined, + id: stringField(item, 'id'), + question: stringField(item, 'question') + } + }) + + return { responses, timedOut: row.timed_out === true } +} + /** Parse clarify tool JSON (`question` + `user_response`). */ export function readClarifyResult(result: unknown): ClarifyResult { const row = parseMaybeObject(result) @@ -236,7 +300,17 @@ function ClarifyToolLive(props: ToolCallMessagePartProps) { return } -function ClarifyToolSettled({ args, result }: ToolCallMessagePartProps) { +function ClarifyToolSettled(props: ToolCallMessagePartProps) { + const batch = readClarifyBatchResult(props.result) + + if (batch.responses.length > 0) { + return + } + + return +} + +function ClarifyToolSingleSettled({ args, result }: ToolCallMessagePartProps) { const { t } = useI18n() const copy = t.assistant.clarify const fromArgs = useMemo(() => readClarifyArgs(args), [args]) @@ -303,19 +377,36 @@ function ClarifyToolSettled({ args, result }: ToolCallMessagePartProps) { ) } -function ClarifyToolPending({ args }: ToolCallMessagePartProps) { - const { t } = useI18n() - const copy = t.assistant.clarify +function ClarifyToolPending(props: ToolCallMessagePartProps) { // The tool row is in whichever session's transcript rendered it — read THAT // session's clarify (primary or tile), not the globally-active one. const sessionId = useStore(useSessionView().$runtimeId) const $request = useMemo(() => sessionClarifyRequest(sessionId), [sessionId]) const request = useStore($request) + const fromArgs = useMemo(() => readClarifyArgs(props.args), [props.args]) + + // Batch: the gateway request carries qid-keyed questions. Args alone can't + // drive the form (no qids to respond with), so batch waits for the request. + if (request?.questions?.length || fromArgs.questions) { + return + } + + return +} + +function ClarifyToolSinglePending({ + fromArgs, + request +}: { + fromArgs: ClarifyArgs + request: ClarifyRequest | null +}) { + const { t } = useI18n() + const copy = t.assistant.clarify const gateway = useStore($gateway) - const fromArgs = useMemo(() => readClarifyArgs(args), [args]) const matchingRequest = useMemo(() => { - if (!request) { + if (!request || request.questions?.length) { return null } @@ -699,3 +790,346 @@ function ClarifyToolPending({ args }: ToolCallMessagePartProps) { ) } + +// ─── Batch (multi-question) clarify ───────────────────────────────────────── + +/** Settled batch card: every question with its locked (or absent) answer. */ +function ClarifyToolBatchSettled({ responses }: { responses: { question?: string; answer?: string | string[] }[] }) { + const { t } = useI18n() + const copy = t.assistant.clarify + + return ( + + {responses.map((row, index) => { + const answer = Array.isArray(row.answer) ? row.answer.join(', ') : (row.answer ?? '') + const blank = !answer.trim() + + return ( +
+ {row.question ? ( + + + {row.question} + + + ) : null} + +

+ {blank ? copy.skipped : answer} +

+
+
+ ) + })} +
+ ) +} + +/** One question's interactive block inside the live batch card. */ +function BatchQuestionBlock({ + disabled, + locked, + onDraft, + onToggle, + question, + staged +}: { + disabled: boolean + locked: boolean + onDraft: (value: string) => void + onToggle: (choice: string) => void + question: ClarifyQuestion + staged: { choices: string[]; draft: string } +}) { + const { t } = useI18n() + const copy = t.assistant.clarify + const choices = question.choices ?? [] + + return ( +
+
+ + {question.question} + + {locked ? ( + + ✓ {copy.answeredBadge} + + ) : null} +
+ + {choices.length > 0 ? ( +
+ {choices.map((choice, index) => ( + onToggle(choice)} + selected={staged.choices.includes(choice)} + /> + ))} +