From c8bbde777033b2a523c9e0a3a4942ef7194e6987 Mon Sep 17 00:00:00 2001 From: Brin Shadewater Date: Sat, 8 Aug 2026 17:38:41 -0700 Subject: [PATCH 001/117] feat: allow configured background review tools Profiles can now grant narrowly scoped tools to the background review runtime whitelist while unrelated tools remain denied. Document the configuration and cover it with a real-config regression test. Agent: codex --- agent/background_review.py | 42 ++++++++++++- ...t_background_review_toolset_restriction.py | 63 ++++++++++++++++++- website/docs/user-guide/features/memory.md | 19 ++++++ 3 files changed, 121 insertions(+), 3 deletions(-) diff --git a/agent/background_review.py b/agent/background_review.py index 72f8fe3a29..79848f479e 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -1563,6 +1563,37 @@ def _run_review_in_thread( # deny message below names that substitute so one denial # redirects the model instead of a storm. review_whitelist |= {"read_file", "search_files"} + # Profile-configured opt-in tools (#44672, salvage #82146 by + # @BrinShadewater): ``auxiliary.background_review.extra_tools`` + # admits named parent tools to the review whitelist — e.g. a + # human-gated proposal tool or a memory-provider write surface. + # Default-empty; a listed tool must already exist in the parent's + # inherited schema (the whitelist can only admit, never advertise), + # and everything unlisted stays denied. Read from task_cfg (the + # auxiliary.background_review block already loaded for this spawn) + # so no extra config I/O happens per review. + configured_extra_tools: set = set() + try: + _extra_raw = _background_review_task_config(task_cfg).get( + "extra_tools", [] + ) + if isinstance(_extra_raw, list): + configured_extra_tools = { + name.strip() + for name in _extra_raw + if isinstance(name, str) and name.strip() + } + review_whitelist |= configured_extra_tools + except Exception: + logger.debug( + "background_review extra_tools parse failed", exc_info=True + ) + _extra_deny_note = ( + " Configured extra tools also allowed: " + + ", ".join(sorted(configured_extra_tools)) + "." + if configured_extra_tools + else "" + ) set_thread_tool_whitelist( review_whitelist, deny_msg_fmt=( @@ -1570,7 +1601,8 @@ def _run_review_in_thread( "{tool_name}. Allowed here: skill_view/skills_list/" "read_file/search_files to read, " "skill_manage(action='patch'|...) to change skills, and " - "memory for notes. Do not retry {tool_name}." + "memory for notes." + _extra_deny_note + + " Do not retry {tool_name}." ), ) try: @@ -1598,6 +1630,14 @@ def _run_review_in_thread( + "\n\nYou can only call memory and skill " "management tools. Other tools will be denied " "at runtime — do not attempt them." + + ( + " Exception — these configured tools are " + "also allowed: " + + ", ".join(sorted(configured_extra_tools)) + + "." + if configured_extra_tools + else "" + ) ), conversation_history=_review_history, ) diff --git a/tests/run_agent/test_background_review_toolset_restriction.py b/tests/run_agent/test_background_review_toolset_restriction.py index bc45bdfba2..07e8e37aff 100644 --- a/tests/run_agent/test_background_review_toolset_restriction.py +++ b/tests/run_agent/test_background_review_toolset_restriction.py @@ -202,7 +202,66 @@ def test_read_file_outside_review_does_not_mark(tmp_path): assert not _background_review_has_read(target) - - +def test_background_review_whitelist_includes_configured_extra_tools( + tmp_path, monkeypatch +): + """A profile may opt a specific proposal tool into background review. + + The review fork inherits the parent's full tool schema for cache parity, + but runtime dispatch remains denied unless the tool is also present in the + thread-local whitelist. This config hook lets profiles grant a narrowly + scoped, human-gated proposal tool without enabling unrelated side effects. + """ + hermes_home = tmp_path / ".hermes" + hermes_home.mkdir() + (hermes_home / "config.yaml").write_text( + "auxiliary:\n" + " background_review:\n" + " extra_tools:\n" + " - propose_shared_memory\n", + encoding="utf-8", + ) + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + + import run_agent + from hermes_cli import config as config_module + from hermes_cli import plugins as _plugins + + config_module._LOAD_CONFIG_CACHE.clear() + config_module._RAW_CONFIG_CACHE.clear() + + captured = {} + + def _capture_whitelist(whitelist, deny_msg_fmt=None): + captured["whitelist"] = set(whitelist) + + def _capture_run_conversation(self, *, user_message, **kwargs): + captured["review_prompt"] = user_message + return {"final_response": "Nothing to save."} + + agent = _make_agent_stub(run_agent.AIAgent) + + def _no_init(self, *args, **kwargs): + return None + + with patch.object(run_agent.AIAgent, "__init__", _no_init), \ + patch.object( + run_agent.AIAgent, + "run_conversation", + _capture_run_conversation, + ), \ + patch.object(run_agent.AIAgent, "shutdown_memory_provider", lambda self: None), \ + patch.object(run_agent.AIAgent, "close", lambda self: None), \ + patch.object(_plugins, "set_thread_tool_whitelist", _capture_whitelist), \ + patch("threading.Thread", _SyncThread): + agent._spawn_background_review( + messages_snapshot=[], + review_memory=True, + review_skills=False, + ) + + assert "propose_shared_memory" in captured["whitelist"] + assert "terminal" not in captured["whitelist"] + assert "propose_shared_memory" in captured["review_prompt"] diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index 205f3e493b..b8c79dbc77 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -350,6 +350,25 @@ Fork usage is persisted in `session_model_usage` with `task='background_review'` and a completion line is written to `agent.log` (`Background review complete: thread=bg-review calls=… in=… out=… result=…`). +### Allowing a narrowly scoped extra review tool (`extra_tools`) + +Background review can use memory, skill-management, and read-only file tools +by default. If a profile provides another tool that is safe for unattended +review, opt it in by name: + +```yaml +auxiliary: + background_review: + extra_tools: + - propose_shared_memory +``` + +The tool must already be available to the parent agent; this setting only adds +it to the review fork's runtime whitelist. It does not enable arbitrary tools, +and tools not listed here remain denied. Keep the list narrow and prefer tools +that stage a proposal for human review rather than applying external or +destructive changes directly. The default is an empty list. + ## Controlling skill writes (`skills.write_approval`) Skills use the same on/off gate, but the review UX differs because a From 0ebce1de810579e40a4468d1d242f5ecea6de2a0 Mon Sep 17 00:00:00 2001 From: Brin Shadewater Date: Sat, 8 Aug 2026 18:39:19 -0700 Subject: [PATCH 002/117] chore: map contributor email Agent: codex --- contributors/emails/brin@shadewaterlabs.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/brin@shadewaterlabs.com diff --git a/contributors/emails/brin@shadewaterlabs.com b/contributors/emails/brin@shadewaterlabs.com new file mode 100644 index 0000000000..78e862d3b3 --- /dev/null +++ b/contributors/emails/brin@shadewaterlabs.com @@ -0,0 +1,2 @@ +BrinShadewater +# PR #82146 From 6cb6aeb16891eeb29437d898e42e4102f5d05b2c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:52:29 -0700 Subject: [PATCH 003/117] =?UTF-8?q?feat(desktop):=20real-profile=20browsin?= =?UTF-8?q?g=20toggle=20in=20Capabilities=20=E2=86=92=20Tools=20=E2=86=92?= =?UTF-8?q?=20Browser?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Users reported no GUI switch for browser.use_real_profile — the only desktop home was the generic Settings → Config editor, which nobody found. The Browser toolset detail pane now renders a 'Use My Real Browser Profile' ToggleRow above the backend/provider matrix. - new BrowserRealProfilePanel: reads the shared profile-scoped config record cache, optimistic write-through, rollback on failure - saveHermesConfigRecord: capability-scoped PUT /api/config counterpart of getHermesConfigRecord, so the Capabilities scope selector writes the profile it points at (possibly another gateway) - i18n: en/ja/zh/zh-hant keys (ar inherits en via defineLocale) - docs: browser.md desktop pointer corrected to the real location Live E2E on the built app over CDP: clicking the switch flipped browser.use_real_profile true→false→true in the sandbox HERMES_HOME config.yaml, GET reflected it, no layout glitches (screenshots in PR). --- apps/desktop/src/api/config.ts | 12 ++ .../browser-real-profile-panel.test.tsx | 109 ++++++++++++++++++ .../settings/browser-real-profile-panel.tsx | 91 +++++++++++++++ apps/desktop/src/app/skills/index.tsx | 5 + apps/desktop/src/i18n/en.ts | 10 ++ apps/desktop/src/i18n/ja.ts | 10 ++ apps/desktop/src/i18n/types.ts | 9 ++ apps/desktop/src/i18n/zh-hant.ts | 10 ++ apps/desktop/src/i18n/zh.ts | 10 ++ website/docs/user-guide/features/browser.md | 4 +- 10 files changed, 269 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/settings/browser-real-profile-panel.test.tsx create mode 100644 apps/desktop/src/app/settings/browser-real-profile-panel.tsx diff --git a/apps/desktop/src/api/config.ts b/apps/desktop/src/api/config.ts index 38fd4945d6..2906c34d87 100644 --- a/apps/desktop/src/api/config.ts +++ b/apps/desktop/src/api/config.ts @@ -99,6 +99,18 @@ export function saveHermesConfig(config: HermesConfigRecord, profile?: null | st }) } +/** Capability-scoped counterpart of saveHermesConfig — writes the config of + * the profile/connection the Capabilities scope selector points at (possibly + * on another registered gateway), mirroring getHermesConfigRecord. */ +export function saveHermesConfigRecord(config: HermesConfigRecord, profile?: ProfileScope): Promise<{ ok: boolean }> { + return window.hermesDesktop.api<{ ok: boolean }>({ + ...capabilityScoped(profile), + path: '/api/config', + method: 'PUT', + body: { config } + }) +} + export function getEnvVars(profile?: null | string): Promise> { return hermesApi>({ ...profileScoped(profile), diff --git a/apps/desktop/src/app/settings/browser-real-profile-panel.test.tsx b/apps/desktop/src/app/settings/browser-real-profile-panel.test.tsx new file mode 100644 index 0000000000..703b96e501 --- /dev/null +++ b/apps/desktop/src/app/settings/browser-real-profile-panel.test.tsx @@ -0,0 +1,109 @@ +// @vitest-environment jsdom +import { act, cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { BrowserRealProfilePanel } from './browser-real-profile-panel' + +const mocks = vi.hoisted(() => ({ + cache: vi.fn(), + loadedConfig: {} as Record, + notify: vi.fn(), + notifyError: vi.fn(), + save: vi.fn() +})) + +vi.mock('@/hermes', () => ({ + saveHermesConfigRecord: (config: Record, profile?: unknown) => mocks.save(config, profile) +})) + +vi.mock('@/i18n', () => ({ + useI18n: () => ({ + t: { + settings: { + toolsets: { + browserRealProfile: { + label: 'Use My Real Browser Profile', + description: 'Copies your default browser profile into a managed snapshot.', + enabledTitle: 'Real-profile browsing on', + enabledMessage: 'New sessions use the snapshot.', + disabledTitle: 'Real-profile browsing off', + disabledMessage: 'Snapshot will be deleted.', + failedSave: 'Could not save the real-profile setting' + } + } + } + } + }) +})) + +vi.mock('@/store/notifications', () => ({ + notify: (...args: unknown[]) => mocks.notify(...args), + notifyError: (...args: unknown[]) => mocks.notifyError(...args) +})) + +vi.mock('../hooks/use-config-record', () => ({ + hermesConfigCacheWriter: () => (config: Record) => mocks.cache(config), + useHermesConfigRecord: () => ({ data: mocks.loadedConfig }) +})) + +describe('BrowserRealProfilePanel', () => { + beforeEach(() => { + mocks.loadedConfig = { browser: { allow_private_urls: false }, model: { provider: 'nous' } } + mocks.save.mockResolvedValue({ ok: true }) + }) + + afterEach(() => { + cleanup() + vi.clearAllMocks() + }) + + it('renders off for a config without the key and turns it on', async () => { + render() + const toggle = screen.getByRole('switch', { name: 'Use My Real Browser Profile' }) + + expect(toggle).toHaveProperty('ariaChecked', 'false') + + await act(async () => { + fireEvent.click(toggle) + }) + + // Saves the WHOLE merged record with only use_real_profile added — sibling + // browser keys survive. + expect(mocks.save).toHaveBeenCalledWith( + { + browser: { allow_private_urls: false, use_real_profile: true }, + model: { provider: 'nous' } + }, + undefined + ) + expect(mocks.cache).toHaveBeenCalledWith(mocks.save.mock.calls[0][0]) + expect(mocks.notify).toHaveBeenCalled() + }) + + it('turns an enabled toggle off', async () => { + mocks.loadedConfig = { browser: { use_real_profile: true } } + render() + const toggle = screen.getByRole('switch', { name: 'Use My Real Browser Profile' }) + + expect(toggle).toHaveProperty('ariaChecked', 'true') + + await act(async () => { + fireEvent.click(toggle) + }) + + expect(mocks.save).toHaveBeenCalledWith({ browser: { use_real_profile: false } }, undefined) + }) + + it('rolls the optimistic cache write back when the save fails', async () => { + mocks.save.mockRejectedValue(new Error('boom')) + render() + + await act(async () => { + fireEvent.click(screen.getByRole('switch', { name: 'Use My Real Browser Profile' })) + }) + + // Last cache write restores the original record. + expect(mocks.cache).toHaveBeenLastCalledWith(mocks.loadedConfig) + expect(mocks.notifyError).toHaveBeenCalled() + }) +}) diff --git a/apps/desktop/src/app/settings/browser-real-profile-panel.tsx b/apps/desktop/src/app/settings/browser-real-profile-panel.tsx new file mode 100644 index 0000000000..72ca7234dc --- /dev/null +++ b/apps/desktop/src/app/settings/browser-real-profile-panel.tsx @@ -0,0 +1,91 @@ +import { useCallback, useState } from 'react' + +import { type ProfileScope, saveHermesConfigRecord } from '@/hermes' +import { useI18n } from '@/i18n' +import { notify, notifyError } from '@/store/notifications' + +import { hermesConfigCacheWriter, useHermesConfigRecord } from '../hooks/use-config-record' + +import { ToggleRow } from './primitives' + +interface BrowserRealProfilePanelProps { + /** Capabilities profile-scope override — the toggle reads/writes THIS + * profile's config.yaml instead of the app-wide active one. */ + profile?: ProfileScope +} + +function readUseRealProfile(record: Record | undefined): boolean { + const browser = record?.browser + + if (browser && typeof browser === 'object' && !Array.isArray(browser)) { + return Boolean((browser as Record).use_real_profile) + } + + return false +} + +/** + * The `browser.use_real_profile` consent toggle, rendered at the top of the + * Capabilities → Tools → Browser detail pane (above the backend/provider + * matrix). This is the GUI home of the real-profile browsing switch: without + * it the only desktop path was the generic Settings → Config editor, which + * users reasonably never found ("no toggle in the browser section"). + * + * Semantics mirror the config comment: turning it ON consents to snapshotting + * the default browser's profile (cookies/logins) into a Hermes-owned copy; + * turning it OFF deletes the snapshot store on next use. The toggle writes + * config.yaml through the same deep-merging PUT /api/config every other + * settings surface uses — applies to new sessions. + */ +export function BrowserRealProfilePanel({ profile }: BrowserRealProfilePanelProps) { + const { t } = useI18n() + const copy = t.settings.toolsets.browserRealProfile + const { data: config } = useHermesConfigRecord(profile) + const setConfig = hermesConfigCacheWriter(profile) + const [busy, setBusy] = useState(false) + + const enabled = readUseRealProfile(config) + + const toggle = useCallback( + async (on: boolean) => { + if (!config) { + return + } + + const browser = + config.browser && typeof config.browser === 'object' && !Array.isArray(config.browser) + ? (config.browser as Record) + : {} + + const next = { ...config, browser: { ...browser, use_real_profile: on } } + + setBusy(true) + setConfig(next) + + try { + await saveHermesConfigRecord(next, profile) + notify({ + kind: 'info', + title: on ? copy.enabledTitle : copy.disabledTitle, + message: on ? copy.enabledMessage : copy.disabledMessage + }) + } catch (err) { + setConfig(config) + notifyError(err, copy.failedSave) + } finally { + setBusy(false) + } + }, + [config, copy, profile, setConfig] + ) + + return ( + void toggle(on)} + /> + ) +} diff --git a/apps/desktop/src/app/skills/index.tsx b/apps/desktop/src/app/skills/index.tsx index 464d84e928..fc24be9157 100644 --- a/apps/desktop/src/app/skills/index.tsx +++ b/apps/desktop/src/app/skills/index.tsx @@ -55,6 +55,7 @@ import { import { PanelEmpty, PanelPill } from '../overlays/panel' import { PageSearchShell } from '../page-search-shell' import { SETTINGS_ROUTE } from '../routes' +import { BrowserRealProfilePanel } from '../settings/browser-real-profile-panel' import { ComputerUsePanel } from '../settings/computer-use-panel' import { asText, includesQuery, prettyName, toolNames, toolsetDisplayLabel } from '../settings/helpers' import { TerminalBackendPanel } from '../settings/terminal-backend-panel' @@ -1166,6 +1167,10 @@ function ToolsetDetail({ )} {toolset.name === 'computer_use' && } + {/* Real-profile consent toggle ABOVE the backend/provider matrix — the + config option users kept missing because its only GUI home was the + generic Settings → Config editor. */} + {toolset.name === 'browser' && } {toolset.name === 'terminal' && } `Terminal commands now run via ${backend}. Applies to new sessions.`, failedSelect: backend => `Failed to select ${backend}`, needsSetupHint: 'You can select this backend now — commands will fail until setup is complete.' + }, + browserRealProfile: { + label: 'Use My Real Browser Profile', + description: + "Copies your default browser's logins and cookies into a managed snapshot the agent browses with. Your live profile is never opened directly. Applies to new sessions.", + enabledTitle: 'Real-profile browsing on', + enabledMessage: 'New sessions will browse with a snapshot of your default browser profile.', + disabledTitle: 'Real-profile browsing off', + disabledMessage: 'The profile snapshot will be deleted; new sessions use a clean browser.', + failedSave: 'Could not save the real-profile setting' } } }, diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 59a8a2992b..376e80fcdc 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -1140,6 +1140,16 @@ export const ja = defineLocale({ selectedMessage: backend => `ターミナルコマンドは ${backend} で実行されます。新しいセッションに適用されます。`, failedSelect: backend => `${backend} の選択に失敗しました`, needsSetupHint: 'このバックエンドは今すぐ選択できますが、セットアップが完了するまでコマンドは失敗します。' + }, + browserRealProfile: { + label: '実際のブラウザプロファイルを使用', + description: + '既定ブラウザのログイン情報と Cookie を管理されたスナップショットにコピーし、エージェントはそれを使ってブラウジングします。実際のプロファイルが直接開かれることはありません。新しいセッションに適用されます。', + enabledTitle: '実プロファイルブラウジング:オン', + enabledMessage: '新しいセッションは既定ブラウザプロファイルのスナップショットでブラウジングします。', + disabledTitle: '実プロファイルブラウジング:オフ', + disabledMessage: 'プロファイルのスナップショットは削除され、新しいセッションはクリーンなブラウザを使用します。', + failedSave: '実プロファイル設定を保存できませんでした' } } }, diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 00ec2eabaa..015fa30aee 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -1092,6 +1092,15 @@ export interface Translations { failedSelect: (backend: string) => string needsSetupHint: string } + browserRealProfile: { + label: string + description: string + enabledTitle: string + enabledMessage: string + disabledTitle: string + disabledMessage: string + failedSave: string + } } } diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 48329d11e9..2558187111 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -1099,6 +1099,16 @@ export const zhHant = defineLocale({ selectedMessage: backend => `終端命令現在透過 ${backend} 執行。將套用於新工作階段。`, failedSelect: backend => `選擇 ${backend} 失敗`, needsSetupHint: '現在即可選擇此後端——但在完成設定前命令將會失敗。' + }, + browserRealProfile: { + label: '使用我的真實瀏覽器設定檔', + description: + '將預設瀏覽器的登入資訊與 Cookie 複製到受管理的快照中,代理使用該快照進行瀏覽。絕不會直接開啟你的真實設定檔。將套用於新工作階段。', + enabledTitle: '真實設定檔瀏覽:已開啟', + enabledMessage: '新工作階段將使用預設瀏覽器設定檔的快照進行瀏覽。', + disabledTitle: '真實設定檔瀏覽:已關閉', + disabledMessage: '設定檔快照將被刪除;新工作階段使用乾淨的瀏覽器。', + failedSave: '無法儲存真實設定檔設定' } } }, diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 14c1231726..c646707b6c 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -1438,6 +1438,16 @@ export const zh: Translations = { selectedMessage: backend => `终端命令现在通过 ${backend} 运行。将应用于新会话。`, failedSelect: backend => `选择 ${backend} 失败`, needsSetupHint: '现在即可选择此后端——但在完成设置前命令将会失败。' + }, + browserRealProfile: { + label: '使用我的真实浏览器配置文件', + description: + '将默认浏览器的登录信息和 Cookie 复制到托管快照中,代理使用该快照进行浏览。绝不会直接打开你的真实配置文件。将应用于新会话。', + enabledTitle: '真实配置文件浏览:已开启', + enabledMessage: '新会话将使用默认浏览器配置文件的快照进行浏览。', + disabledTitle: '真实配置文件浏览:已关闭', + disabledMessage: '配置文件快照将被删除;新会话使用干净的浏览器。', + failedSave: '无法保存真实配置文件设置' } } }, diff --git a/website/docs/user-guide/features/browser.md b/website/docs/user-guide/features/browser.md index af7d5f5cb6..628521ed1f 100644 --- a/website/docs/user-guide/features/browser.md +++ b/website/docs/user-guide/features/browser.md @@ -236,7 +236,9 @@ to fully quit the browser — it won't loop or kill again on its own. - **Security framing:** this is a consent-gated convenience, not an isolation boundary. A page the agent visits runs with your real logins, so only enable it when you want the agent acting as you. Off by default. -- **Desktop:** toggle it in **Settings → Browser → Use My Real Browser Profile**. +- **Desktop:** toggle it in **Capabilities → Tools → Browser → Use My Real + Browser Profile** (the switch sits above the backend options), or in + Settings → Config under the `browser` section. ### Camofox local mode From 04ef14e31fe5e8a5e60e177ac71b0826945e23e4 Mon Sep 17 00:00:00 2001 From: icocode Date: Mon, 17 Aug 2026 00:16:45 +0800 Subject: [PATCH 004/117] =?UTF-8?q?fix(providers):=20add=20missing=20Qwen?= =?UTF-8?q?=20Cloud=20(alibaba)=20models=20=E2=80=94=20qwen3.8-max,=20qwen?= =?UTF-8?q?3.6-flash,=20glm-5.2,=20deepseek-v4-pro/flash-0731?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/models.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 1411022fea..597e2926b9 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -641,16 +641,22 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # to https://dashscope-intl.aliyuncs.com/compatible-mode/v1 (OpenAI-compat) # or https://dashscope-intl.aliyuncs.com/apps/anthropic (Anthropic-compat). "alibaba": [ + # Qwen 千问系列 (DashScope / Qwen Cloud) + "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "qwen3.6-flash", "kimi-k2.5", "qwen3.5-plus", "qwen3-coder-plus", "qwen3-coder-next", - # Third-party models available on coding-intl + # Third-party models available on coding-intl / DashScope + "glm-5.2", "glm-5", "glm-4.7", + "deepseek-v4-pro", + "deepseek-v4-flash-0731", "MiniMax-M2.5", ], # Alibaba DashScope (China) — same platform as alibaba, domestic endpoint From 2215fb0e3517a3a2e0cfc74cd04cb7f28eddf81d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:42:08 -0700 Subject: [PATCH 005/117] fix(providers): mirror new Qwen Cloud models onto alibaba-cn Follow-up to the #87808 salvage: the domestic alibaba-cn picker list gets the same five additions (same DashScope catalog, per models.dev). --- contributors/emails/icocode@users.noreply.github.com | 1 + hermes_cli/models.py | 5 +++++ 2 files changed, 6 insertions(+) create mode 100644 contributors/emails/icocode@users.noreply.github.com diff --git a/contributors/emails/icocode@users.noreply.github.com b/contributors/emails/icocode@users.noreply.github.com new file mode 100644 index 0000000000..5d5e32b543 --- /dev/null +++ b/contributors/emails/icocode@users.noreply.github.com @@ -0,0 +1 @@ +icocode diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 597e2926b9..fb4a4dd83a 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -662,15 +662,20 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # Alibaba DashScope (China) — same platform as alibaba, domestic endpoint # (dashscope.aliyuncs.com); same catalog as the international tier. "alibaba-cn": [ + "qwen3.8-max", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "qwen3.6-flash", "kimi-k2.5", "qwen3.5-plus", "qwen3-coder-plus", "qwen3-coder-next", + "glm-5.2", "glm-5", "glm-4.7", + "deepseek-v4-pro", + "deepseek-v4-flash-0731", "MiniMax-M2.5", ], # Alibaba Coding Plan — same platform as alibaba (DashScope coding-intl), From 86a2fdc6347b0468ae611851ddc236c4119ffea2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:59:06 -0700 Subject: [PATCH 006/117] feat(tui): status rule shows cache-hit %, latency, t/s and honors display.status_bar.fields MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extends PR #98250's classic-CLI status-bar upgrades to the Ink TUI: - tui_gateway/server.py _get_usage() now emits cache_hit_pct, avg_latency_s, avg_tps (reads the same per-call deque history from agent/conversation_loop.py; keys omitted when no data — Codex app-server has no latency, zero cache reads show no %) - StatusRule renders the three read-outs as width-budgeted tail segments (breakpoints 96/104/110 cols, lowest priority — they shed first on narrow terminals) - display.status_bar.fields (the SAME key the classic CLI honors) filters TUI segments too: cache_hit, latency, tps, duration, compressions, bg_tasks, bg_subagents, voice, battery, title, context_pct, context_detail - values ride the existing usage payload/ticker; constants between events so the usage==last dedup keeps suppressing repaints - 3 new server tests, 5 new TUI tests; full ui-tui suite 1727 green --- tests/test_tui_gateway_server.py | 45 ++++++++++ tui_gateway/server.py | 33 +++++++ .../__tests__/appChromeStatusRule.test.tsx | 57 ++++++++++++ ui-tui/src/__tests__/statusRule.test.ts | 13 ++- ui-tui/src/app/interfaces.ts | 4 + ui-tui/src/app/uiStore.ts | 1 + ui-tui/src/app/useConfigSync.ts | 15 ++++ ui-tui/src/components/appChrome.tsx | 89 +++++++++++++++---- ui-tui/src/components/appLayout.tsx | 1 + ui-tui/src/gatewayTypes.ts | 7 ++ ui-tui/src/types.ts | 6 ++ website/docs/user-guide/configuration.md | 1 + 12 files changed, 256 insertions(+), 16 deletions(-) diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index c9e65bc002..6166c5e57f 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -18802,6 +18802,51 @@ class _BareAgent: model = "x" +def test_get_usage_perf_readouts_present(): + """cache_hit_pct / avg_latency_s / avg_tps mirror the classic CLI bar.""" + from collections import deque + + class _PerfAgent: + model = "x" + session_prompt_tokens = 27_873 + session_cache_read_tokens = 24_369 + _api_latency_history = deque([2.1, 4.3], maxlen=10) + _api_output_history = deque([130, 190], maxlen=10) + + usage = server._get_usage(_PerfAgent()) + assert usage["cache_hit_pct"] == 87 + assert usage["avg_latency_s"] == 3.2 + assert usage["avg_tps"] == 50.0 # true throughput sum(out)/sum(lat), not mean of ratios + + +def test_get_usage_perf_readouts_omitted_without_data(): + """Zero cache reads / empty history omit the keys — never fabricate 0s.""" + + class _ColdAgent: + model = "x" + session_prompt_tokens = 100 + session_cache_read_tokens = 0 + + usage = server._get_usage(_ColdAgent()) + assert "cache_hit_pct" not in usage + assert "avg_latency_s" not in usage + assert "avg_tps" not in usage + + +def test_get_usage_perf_readouts_guard_negative_latency(): + """Odd provider timings (negative durations seen in logs) are dropped.""" + from collections import deque + + class _WeirdAgent: + model = "x" + _api_latency_history = deque([-0.8], maxlen=10) + _api_output_history = deque([100], maxlen=10) + + usage = server._get_usage(_WeirdAgent()) + assert "avg_latency_s" not in usage + assert "avg_tps" not in usage + + def test_get_usage_includes_active_subagents(monkeypatch): import tools.async_delegation as ad_mod monkeypatch.setattr(ad_mod, "active_count", lambda: 4) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 887994458a..0536cb056b 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -7156,6 +7156,39 @@ def _get_usage(agent) -> dict: usage["context_max"] = ctx_max usage["context_percent"] = max(0, min(100, round(last_prompt / ctx_max * 100))) usage["compressions"] = getattr(comp, "compression_count", 0) or 0 + # Cache-hit ratio + rolling latency/throughput for the TUI status bar. + # Mirrors the classic CLI bar (cli.py _get_status_bar_snapshot / PR #98250): + # hit = session_cache_read_tokens / session_prompt_tokens + # (CanonicalUsage.prompt_tokens = input + cache_read + cache_write) + # latency/tps read the deque(maxlen=10) history maintained per API call in + # agent/conversation_loop.py. Values are omitted (not fabricated) when no + # data exists — e.g. Codex app-server reports no latency, and a session + # with zero cache reads shows no hit% rather than an alarming 0. + try: + _prompt_total = int(getattr(agent, "session_prompt_tokens", 0) or 0) + _cache_read = int(getattr(agent, "session_cache_read_tokens", 0) or 0) + if _prompt_total > 0 and _cache_read > 0: + usage["cache_hit_pct"] = max(0, min(100, round(_cache_read / _prompt_total * 100))) + except Exception: + pass + try: + _lhist = list(getattr(agent, "_api_latency_history", []) or []) + _ohist = list(getattr(agent, "_api_output_history", []) or []) + _n = min(len(_lhist), len(_ohist)) + if _n: + _lhist = _lhist[-_n:] + _ohist = _ohist[-_n:] + _avg_lat = sum(_lhist) / _n + _total_lat = sum(_lhist) + _avg_vel = (sum(_ohist) / _total_lat) if _total_lat > 0 else None + # Guard NaN/negative/absurd values from odd provider timings. + if _avg_lat == _avg_lat and 0 < _avg_lat < 1e6: + usage["avg_latency_s"] = round(float(_avg_lat), 1) + if _avg_vel is not None and _avg_vel == _avg_vel and 0 < _avg_vel < 1e6: + usage["avg_tps"] = round(float(_avg_vel), 1) + except Exception: + # A status-bar readout must never break usage reporting. + pass # Live count of background/async subagents still running (delegate_task # batches + background single delegations). Mirrors the classic CLI status # bar's ⛓ indicator; sourced from the same async_delegation registry. diff --git a/ui-tui/src/__tests__/appChromeStatusRule.test.tsx b/ui-tui/src/__tests__/appChromeStatusRule.test.tsx index 07fc468768..017ce92371 100644 --- a/ui-tui/src/__tests__/appChromeStatusRule.test.tsx +++ b/ui-tui/src/__tests__/appChromeStatusRule.test.tsx @@ -491,3 +491,60 @@ describe('StatusRule idle-since read-out', () => { expect(findComponentByName(element, 'IdleSince')).toBeNull() }) }) + + +describe('StatusRule perf read-outs (cache hit / latency / tps)', () => { + const perfUsage = { + ...baseProps.usage, + avg_latency_s: 3.2, + avg_tps: 50.4, + cache_hit_pct: 87, + calls: 4, + input: 1000, + output: 500 + } + + it('renders all three segments on a wide terminal', () => { + const element = StatusRule({ ...baseProps, cols: 160, usage: perfUsage }) + const rendered = textContent(element) + + expect(rendered).toContain('◎ 87%') + expect(rendered).toContain('◷ 3.2s') + expect(rendered).toContain('↑ 50 t/s') + }) + + it('self-hides when the server omits the keys', () => { + const element = StatusRule({ ...baseProps, cols: 160 }) + const rendered = textContent(element) + + expect(rendered).not.toContain('◎') + expect(rendered).not.toContain('◷') + expect(rendered).not.toContain('t/s') + }) + + it('honors the display.status_bar.fields visibility filter', () => { + const element = StatusRule({ + ...baseProps, + cols: 160, + statusBarFields: new Set(['model', 'context_pct', 'cache_hit']), + usage: perfUsage + }) + + const rendered = textContent(element) + + expect(rendered).toContain('◎ 87%') + expect(rendered).not.toContain('◷') + expect(rendered).not.toContain('t/s') + }) + + it('hides the session title badge when the fields filter omits title', () => { + const element = StatusRule({ + ...baseProps, + cols: 160, + sessionTitle: 'weekly-digest', + statusBarFields: new Set(['model', 'context_pct']) + }) + + expect(textContent(element)).not.toContain('weekly-digest') + }) +}) diff --git a/ui-tui/src/__tests__/statusRule.test.ts b/ui-tui/src/__tests__/statusRule.test.ts index 1a5334eb13..3b4b25db4e 100644 --- a/ui-tui/src/__tests__/statusRule.test.ts +++ b/ui-tui/src/__tests__/statusRule.test.ts @@ -69,10 +69,21 @@ describe('statusBarSegments', () => { compressions: true, voice: true, bg: true, - subagents: true + subagents: true, + cacheHit: true, + latency: true, + tps: true } satisfies StatusBarSegments) }) + it('sheds cache/latency/tps read-outs first as the terminal narrows', () => { + // 96/104/110-col breakpoints: these are the lowest-priority perf + // read-outs, so they disappear before any pre-existing segment. + expect(statusBarSegments(108)).toMatchObject({ cacheHit: true, latency: true, tps: false }) + expect(statusBarSegments(100)).toMatchObject({ cacheHit: true, latency: false, tps: false }) + expect(statusBarSegments(94)).toMatchObject({ cacheHit: false, latency: false, tps: false, subagents: true }) + }) + it('collapses the context bar to a token count on narrow terminals', () => { const s = statusBarSegments(60) diff --git a/ui-tui/src/app/interfaces.ts b/ui-tui/src/app/interfaces.ts index 69689b74a0..f0c8ef0c2b 100644 --- a/ui-tui/src/app/interfaces.ts +++ b/ui-tui/src/app/interfaces.ts @@ -343,6 +343,10 @@ export interface UiState { sid: null | string status: string statusBar: StatusBarMode + // display.status_bar.fields — visibility filter for status-rule segments, + // shared with the classic CLI bar. null = user has not customized (show + // the default set). + statusBarFields: null | ReadonlySet streaming: boolean theme: Theme // `display.timestamps` — dim [HH:MM] labels on user/assistant transcript diff --git a/ui-tui/src/app/uiStore.ts b/ui-tui/src/app/uiStore.ts index 581d576a48..e924708c06 100644 --- a/ui-tui/src/app/uiStore.ts +++ b/ui-tui/src/app/uiStore.ts @@ -32,6 +32,7 @@ const buildUiState = (): UiState => ({ sid: null, status: 'summoning hermes…', statusBar: 'top', + statusBarFields: null, streaming: true, timestamps: false, // Last session's resolved theme paints frame one (flash-free boot, like diff --git a/ui-tui/src/app/useConfigSync.ts b/ui-tui/src/app/useConfigSync.ts index cf471c4633..32e5b4f462 100644 --- a/ui-tui/src/app/useConfigSync.ts +++ b/ui-tui/src/app/useConfigSync.ts @@ -28,6 +28,20 @@ const STATUSBAR_ALIAS: Record = { export const normalizeStatusBar = (raw: unknown): StatusBarMode => raw === false ? 'off' : typeof raw === 'string' ? (STATUSBAR_ALIAS[raw.trim().toLowerCase()] ?? 'top') : 'top' +// `display.status_bar.fields` — the SAME key the classic CLI bar honors +// (PR #98250). A non-empty list filters status-rule segments; missing/empty/ +// malformed = null (user hasn't customized → show the default set). Unknown +// names pass through harmlessly — the renderer only tests membership. +export const normalizeStatusBarFields = (raw: unknown): null | ReadonlySet => { + if (!Array.isArray(raw) || raw.length === 0) { + return null + } + + const cleaned = raw.map(v => String(v).trim().toLowerCase()).filter(Boolean) + + return cleaned.length ? new Set(cleaned) : null +} + const BUSY_MODES = new Set(['interrupt', 'queue', 'steer']) // TUI defaults to `queue` even though the framework default @@ -289,6 +303,7 @@ export const applyDisplay = ( sections: resolveSections(d.sections), showReasoning: !!d.show_reasoning, statusBar: normalizeStatusBar(d.tui_statusbar), + statusBarFields: normalizeStatusBarFields(d.status_bar?.fields), streaming: d.streaming !== false, // The SAME key that stamps [HH:MM] on classic-CLI labels (#41531) — // no separate TUI knob. diff --git a/ui-tui/src/components/appChrome.tsx b/ui-tui/src/components/appChrome.tsx index 708391eea6..c989678f2a 100644 --- a/ui-tui/src/components/appChrome.tsx +++ b/ui-tui/src/components/appChrome.tsx @@ -291,10 +291,13 @@ export function statusRuleWidths(cols: number, cwdLabel: string, minLeftContent export interface StatusBarSegments { bar: boolean bg: boolean + cacheHit: boolean compactCtx: boolean compressions: boolean duration: boolean + latency: boolean subagents: boolean + tps: boolean voice: boolean } @@ -308,7 +311,10 @@ export function statusBarSegments(cols: number): StatusBarSegments { compressions: w >= 80, voice: w >= 84, bg: w >= 88, - subagents: w >= 92 + subagents: w >= 92, + cacheHit: w >= 96, + latency: w >= 104, + tps: w >= 110 } } @@ -470,6 +476,7 @@ export function StatusRule({ cols, busy, status, + statusBarFields = null, statusColor, model, modelFast, @@ -491,21 +498,28 @@ export function StatusRule({ const barColor = ctxBarColor(pct, t) const segs = statusBarSegments(cols) + // display.status_bar.fields visibility gate (same key + names as the + // classic CLI bar). null = user hasn't customized → everything shows. + const ok = (name: string) => statusBarFields === null || statusBarFields.has(name) + // On narrow terminals the context read-out collapses to a bare token count // (`12k tok`) and the visual fill bar is dropped entirely. - const ctxLabel = usage.context_max - ? segs.compactCtx - ? `${fmtK(usage.context_used ?? 0)} tok` - : `${fmtK(usage.context_used ?? 0)}/${fmtK(usage.context_max)}` - : usage.total > 0 - ? `${fmtK(usage.total)} tok` + const ctxLabel = + ok('context_detail') || ok('context_pct') + ? usage.context_max + ? segs.compactCtx + ? `${fmtK(usage.context_used ?? 0)} tok` + : `${fmtK(usage.context_used ?? 0)}/${fmtK(usage.context_max)}` + : usage.total > 0 + ? `${fmtK(usage.total)} tok` + : '' : '' - const bar = !segs.compactCtx && usage.context_max ? ctxBar(pct) : '' + const bar = !segs.compactCtx && usage.context_max && ok('context_pct') ? ctxBar(pct) : '' const modelText = modelLabel(model, modelReasoningEffort, modelFast) // Battery read-out — the first (pinned) status-bar element when enabled. - const showBattery = !!battery && battery.available && battery.percent != null + const showBattery = !!battery && battery.available && battery.percent != null && ok('battery') const batteryText = showBattery ? batteryLabel(battery!) : '' const batteryColorVal = showBattery ? batteryColor(battery!, t) : '' const batteryWidth = showBattery ? stringWidth(`${batteryText} │ `) : 0 @@ -541,7 +555,7 @@ export function StatusRule({ stringWidth(modelText) + (ctxLabel ? stringWidth(' │ ') + stringWidth(ctxLabel) : 0) - const rightLabel = sessionTitle ? ` ${sessionTitle} ` : cwdLabel + const rightLabel = sessionTitle && ok('title') ? ` ${sessionTitle} ` : cwdLabel const { leftWidth, rightWidth, separatorWidth } = statusRuleWidths(cols, rightLabel, essentialWidth) // Whole-segment progressive disclosure for the tail: a segment renders only @@ -575,7 +589,7 @@ export function StatusRule({ : '' const showBar = !!bar && fits(SEP + stringWidth(`[${bar}] ${pct != null ? `${pct}%` : ''}`)) - const showDuration = segs.duration && !!sessionStartedAt && fits(SEP + MAX_DURATION_WIDTH) + const showDuration = segs.duration && ok('duration') && !!sessionStartedAt && fits(SEP + MAX_DURATION_WIDTH) // Idle clock — time since the last final agent response. Hidden while busy // (the FaceTicker's elapsed tail covers the live turn) and before the first @@ -583,12 +597,26 @@ export function StatusRule({ const showIdle = segs.duration && !busy && lastTurnEndedAt != null && fits(SEP + stringWidth('✓ ') + MAX_DURATION_WIDTH) - const showCompressions = segs.compressions && compressions > 0 && fits(SEP + stringWidth(`cmp ${compressions}`)) - const showVoice = segs.voice && !!voiceLabel && fits(SEP + stringWidth(voiceLabel)) + const showCompressions = + segs.compressions && ok('compressions') && compressions > 0 && fits(SEP + stringWidth(`cmp ${compressions}`)) + + // Cache-hit % + rolling latency / tokens-per-sec — mirrored from the classic + // CLI bar (PR #98250). The server omits the keys when no data exists (zero + // cache reads, Codex app-server with no latency), so these self-hide. + const cacheHitText = typeof usage.cache_hit_pct === 'number' ? `◎ ${usage.cache_hit_pct}%` : '' + const showCacheHit = segs.cacheHit && ok('cache_hit') && !!cacheHitText && fits(SEP + stringWidth(cacheHitText)) + const latencyText = typeof usage.avg_latency_s === 'number' ? `◷ ${usage.avg_latency_s.toFixed(1)}s` : '' + const showLatency = segs.latency && ok('latency') && !!latencyText && fits(SEP + stringWidth(latencyText)) + const tpsText = typeof usage.avg_tps === 'number' ? `↑ ${Math.round(usage.avg_tps)} t/s` : '' + const showTps = segs.tps && ok('tps') && !!tpsText && fits(SEP + stringWidth(tpsText)) + + const showVoice = segs.voice && ok('voice') && !!voiceLabel && fits(SEP + stringWidth(voiceLabel)) const showSessionCount = !!sessionCountText && fits(SEP + stringWidth(sessionCountText)) - const showBg = segs.bg && bgCount > 0 && fits(SEP + stringWidth(`${bgCount} bg`)) + const showBg = segs.bg && ok('bg_tasks') && bgCount > 0 && fits(SEP + stringWidth(`${bgCount} bg`)) const subagentCount = typeof usage.active_subagents === 'number' ? usage.active_subagents : 0 - const showSubagents = segs.subagents && subagentCount > 0 && fits(SEP + stringWidth(`⛓ ${subagentCount}`)) + + const showSubagents = + segs.subagents && ok('bg_subagents') && subagentCount > 0 && fits(SEP + stringWidth(`⛓ ${subagentCount}`)) // Parked-background reassurance: a top-level delegate_task runs in the // background, so the turn ends (idle) while the subagent keeps working and its @@ -705,6 +733,34 @@ export function StatusRule({ ) : null} + {showCacheHit ? ( + + {' │ '} + = 70 + ? t.color.statusGood + : usage.cache_hit_pct! >= 40 + ? t.color.statusWarn + : t.color.muted + } + > + {cacheHitText} + + + ) : null} + {showLatency ? ( + + {' │ '} + {latencyText} + + ) : null} + {showTps ? ( + + {' │ '} + {tpsText} + + ) : null} {showVoice ? ( statusColor: string t: Theme turnStartedAt?: null | number diff --git a/ui-tui/src/components/appLayout.tsx b/ui-tui/src/components/appLayout.tsx index f358097699..47eef82a26 100644 --- a/ui-tui/src/components/appLayout.tsx +++ b/ui-tui/src/components/appLayout.tsx @@ -506,6 +506,7 @@ const StatusRulePane = memo(function StatusRulePane({ sessionStartedAt={status.sessionStartedAt} sessionTitle={status.sessionTitle} status={ui.status} + statusBarFields={ui.statusBarFields} statusColor={status.statusColor} t={ui.theme} turnStartedAt={status.turnStartedAt} diff --git a/ui-tui/src/gatewayTypes.ts b/ui-tui/src/gatewayTypes.ts index 73ddee0a17..bc08637c14 100644 --- a/ui-tui/src/gatewayTypes.ts +++ b/ui-tui/src/gatewayTypes.ts @@ -88,6 +88,10 @@ export interface ConfigDisplayConfig { sections?: Record show_cost?: boolean show_reasoning?: boolean + /** CLI/TUI status-bar field visibility filter (shared with the classic + * CLI bar — see display.status_bar.fields in configuration docs). + * Raw YAML: callers must runtime-validate entries. */ + status_bar?: { fields?: unknown } streaming?: boolean thinking_mode?: string /** Show [HH:MM] timestamps on transcript rows — same key the classic CLI @@ -270,6 +274,9 @@ export interface SessionUndoResponse { export interface SessionUsageResponse { active_subagents?: number + avg_latency_s?: number + avg_tps?: number + cache_hit_pct?: number cache_read?: number cache_write?: number calls?: number diff --git a/ui-tui/src/types.ts b/ui-tui/src/types.ts index 4da7cb5882..1803402bb5 100644 --- a/ui-tui/src/types.ts +++ b/ui-tui/src/types.ts @@ -207,6 +207,12 @@ export interface SessionInfo { export interface Usage { active_subagents?: number + /** Rolling mean API latency over the last 10 calls (seconds). */ + avg_latency_s?: number + /** Rolling output tokens/sec over the last 10 calls. */ + avg_tps?: number + /** Session prompt-cache hit ratio (cache_read / prompt tokens, %). */ + cache_hit_pct?: number calls: number compressions?: number context_max?: number diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 27ff51bd54..539bd0bf8d 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1989,6 +1989,7 @@ Notes: - Narrow terminals still drop wide-mode-only fields (`context_detail`, `cache_hit`, `latency`, `tps`, `prompt_elapsed`, `idle_since`) regardless of config (`cache_hit` also shows in the medium ≥52-col tier). - `latency`/`tps` stay hidden until API calls have been recorded (e.g. the Codex app-server backend reports no latency). - `battery` and `title` visibility here compose with their own toggles (`/battery`, `/title`) — both must be on for the segment to show. +- The same key also filters the **Ink TUI** status rule (`hermes tui`), where `cache_hit`, `latency`, and `tps` render as width-budgeted tail segments (◎ / ◷ / ↑) on terminals ≥96/104/110 columns respectively. - Display-only: no effect on prompt caching or request payloads. Changes take effect on the next session start. ### Runtime-metadata footer (gateway only) From 792dbea7773a27162861920dfa95a5174b25ac35 Mon Sep 17 00:00:00 2001 From: Patrickk Date: Sat, 29 Aug 2026 18:28:12 -0700 Subject: [PATCH 007/117] fix(cron): forward request_overrides into scheduled-job agents Salvaged from #56876 (cron half only; the delegation half is superseded by #98237). run_job's ephemeral AIAgent constructor passed api_key / base_url / provider / api_mode from the resolved runtime but dropped request_overrides, so cron jobs on custom providers silently lost extra_body / extra_headers request settings. --- cron/scheduler.py | 1 + tests/cron/test_cron_request_overrides.py | 60 +++++++++++++++++++++++ 2 files changed, 61 insertions(+) create mode 100644 tests/cron/test_cron_request_overrides.py diff --git a/cron/scheduler.py b/cron/scheduler.py index 4b5af389ea..c69850f5d0 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -6406,6 +6406,7 @@ def run_job( provider=runtime.get("provider"), requested_provider=runtime.get("requested_provider"), api_mode=runtime.get("api_mode"), + request_overrides=runtime.get("request_overrides"), acp_command=runtime.get("command"), acp_args=runtime.get("args"), max_iterations=max_iterations, diff --git a/tests/cron/test_cron_request_overrides.py b/tests/cron/test_cron_request_overrides.py new file mode 100644 index 0000000000..dec012dd44 --- /dev/null +++ b/tests/cron/test_cron_request_overrides.py @@ -0,0 +1,60 @@ +"""Standalone regression test for cron runtime request_overrides forwarding. + +Split out of tests/cron/test_scheduler.py as an independent, upgrade-safe file +(registered in the local-patch ledger's allowed_untracked list) so the local +request_overrides forwarding hotfix no longer collides with upstream inserting +new tests around its anchor in the large test_scheduler.py module. + +The functional change under test lives in cron/scheduler.py (run_job forwards +runtime['request_overrides'] into the ephemeral AIAgent); that hunk stays in the +source patch. Only this test moved here. +""" + +from unittest.mock import patch, MagicMock + +from cron.scheduler import run_job + + +class TestRunJobRequestOverrides: + def test_run_job_forwards_runtime_request_overrides_to_agent(self, tmp_path): + # runtime_provider may resolve provider-specific request_overrides; + # cron must pass them into the ephemeral AIAgent or scheduled jobs + # regress to SDK-default request settings. + job = { + "id": "request-overrides-job", + "name": "request-overrides", + "prompt": "hello", + } + fake_db = MagicMock() + overrides = { + "extra_headers": { + "User-Agent": "codex_cli_rs/0.138.0 (Windows 10.0.26100; x86_64)" + } + } + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._resolve_origin", return_value=None), \ + patch("dotenv.load_dotenv"), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "test-key", + "base_url": "https://example.invalid/v1", + "provider": "custom", + "api_mode": "codex_responses", + "request_overrides": overrides, + }, + ), \ + patch("run_agent.AIAgent") as mock_agent_cls: + mock_agent = MagicMock() + mock_agent.run_conversation.return_value = {"final_response": "ok"} + mock_agent_cls.return_value = mock_agent + + success, _output, final_response, error = run_job(job) + + assert success is True + assert error is None + assert final_response == "ok" + kwargs = mock_agent_cls.call_args.kwargs + assert kwargs["request_overrides"] == overrides From 1fa3edcb627f37916101748d3e1ecf96ad8a4bc8 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:28:41 -0700 Subject: [PATCH 008/117] chore(attribution): map bsbofmusic noreply email --- contributors/emails/bsbofmusic@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/bsbofmusic@users.noreply.github.com diff --git a/contributors/emails/bsbofmusic@users.noreply.github.com b/contributors/emails/bsbofmusic@users.noreply.github.com new file mode 100644 index 0000000000..0b81e779fa --- /dev/null +++ b/contributors/emails/bsbofmusic@users.noreply.github.com @@ -0,0 +1 @@ +bsbofmusic From 556777ddb1734356c1ec319f1e89172705f22df4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:29:10 -0700 Subject: [PATCH 009/117] docs(cron): note request settings carry into scheduled runs --- website/docs/user-guide/features/cron.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/website/docs/user-guide/features/cron.md b/website/docs/user-guide/features/cron.md index 74be2ab3f2..31f5325b3a 100644 --- a/website/docs/user-guide/features/cron.md +++ b/website/docs/user-guide/features/cron.md @@ -28,6 +28,8 @@ All of this is available to Hermes itself through the `cronjob` tool, so you can - **`cron.model` / `cron.model_provider`** — a cron-fleet default: every unpinned job runs on this model, independent of your chat model. Set it once (`hermes config set cron.model `) and switching your chat model with `hermes model` or `/model` never touches your cron fleet. - **Global default** — only when neither of the above is set does a job follow `hermes model`. In this case Hermes **snapshots** the provider and model at creation, and if the global default later changes the job **fails closed**: it skips the run, makes no inference call, and alerts you **once** — the job stays skipped (and silent) on subsequent ticks until you act or the config is restored (#44585). For recurring or otherwise repeatable jobs, pin the provider/model explicitly (`hermes cron edit --provider --model `) to proceed. A consumed finite one-shot cannot be updated; create a new future one-shot with an explicit provider and model instead. This prevents an unattended job from silently inheriting a switch to a paid provider/model. Setting `cron.model` (or a per-job pin) is the deliberate way to route cron spend, and the drift guard does not engage for an axis covered by it. Operators who instead want unpinned jobs to track the changing global default can [disable the drift guard](#letting-unpinned-jobs-track-global-defaults). +Whichever provider a job resolves to, its provider-specific request settings (e.g. `request_overrides` such as `extra_body`/`extra_headers` for custom providers) carry into the scheduled run just like an interactive session. + `hermes setup --portal` is the lowest-friction option for unattended runs since OAuth refresh is automatic. See [Nous Portal](/integrations/nous-portal). ::: From 863aac9012fcdfae88fb3e729b60ced8170674ad Mon Sep 17 00:00:00 2001 From: CharZhou <17255546+CharZhou@users.noreply.github.com> Date: Mon, 20 Jul 2026 08:56:38 +0800 Subject: [PATCH 010/117] fix: preserve named custom provider request_overrides in gateway and /model switches Carry provider-derived request_overrides through runtime resolution, fallback projection, session /model state, restart rehydration, and turn-route merge so named custom providers keep extra_body and related overrides. --- gateway/run.py | 33 ++- gateway/slash_commands.py | 2 + hermes_cli/model_switch.py | 2 + .../test_custom_provider_request_overrides.py | 220 ++++++++++++++++++ .../test_model_command_request_overrides.py | 94 ++++++++ 5 files changed, 349 insertions(+), 2 deletions(-) create mode 100644 tests/gateway/test_custom_provider_request_overrides.py create mode 100644 tests/gateway/test_model_command_request_overrides.py diff --git a/gateway/run.py b/gateway/run.py index 00b62aeb3e..1dcf8edb54 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -2969,6 +2969,7 @@ def _resolve_runtime_agent_kwargs() -> dict: "command": runtime.get("command"), "args": list(runtime.get("args") or []), "credential_pool": runtime.get("credential_pool"), + "request_overrides": dict(runtime.get("request_overrides") or {}), "max_tokens": max_tokens, } @@ -3109,9 +3110,23 @@ def _resolve_runtime_agent_kwargs_for_provider(provider: str) -> dict: "command": runtime.get("command"), "args": list(runtime.get("args") or []), "credential_pool": runtime.get("credential_pool"), + "request_overrides": dict(runtime.get("request_overrides") or {}), } +def _deep_merge_request_overrides(base: Optional[dict], override: Optional[dict]) -> dict: + """Merge request_overrides dicts, deep-merging nested dictionaries.""" + from hermes_cli.config import _deep_merge + + base_dict = dict(base or {}) + override_dict = dict(override or {}) + if not base_dict: + return override_dict + if not override_dict: + return base_dict + return _deep_merge(base_dict, override_dict) + + def _credential_pool_for_provider(provider: Optional[str]): """Return the live credential pool for a provider id (e.g. ``custom:hyper``).""" if not provider or not str(provider).strip(): @@ -3167,6 +3182,7 @@ def _try_resolve_fallback_provider() -> dict | None: "command": runtime.get("command"), "args": list(runtime.get("args") or []), "credential_pool": runtime.get("credential_pool"), + "request_overrides": dict(runtime.get("request_overrides") or {}), "model": entry.get("model"), } except Exception as fb_exc: @@ -8459,6 +8475,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "credential_pool": runtime_kwargs.get("credential_pool"), "max_tokens": runtime_kwargs.get("max_tokens"), } + base_request_overrides = dict(runtime_kwargs.get("request_overrides") or {}) route = { "model": model, "runtime": runtime, @@ -8475,14 +8492,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew service_tier = getattr(self, "_service_tier", None) if not service_tier: - route["request_overrides"] = {} + route["request_overrides"] = base_request_overrides return route try: overrides = resolve_fast_mode_overrides(route["model"]) except Exception: overrides = None - route["request_overrides"] = overrides or {} + route["request_overrides"] = _deep_merge_request_overrides( + base_request_overrides, + overrides or {}, + ) return route def _sync_session_model_from_agent(self, session_id: str, agent: Any) -> None: @@ -27409,6 +27429,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew override["api_key"] = runtime.get("api_key") override["api_mode"] = runtime.get("api_mode") override["credential_pool"] = runtime.get("credential_pool") + override["request_overrides"] = dict( + runtime.get("request_overrides") or {} + ) if not override.get("base_url"): override["base_url"] = runtime.get("base_url") except Exception: @@ -27443,6 +27466,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew val = override.get(key) if val is not None: runtime_kwargs[key] = val + override_request_overrides = override.get("request_overrides") + if isinstance(override_request_overrides, dict): + runtime_kwargs["request_overrides"] = _deep_merge_request_overrides( + runtime_kwargs.get("request_overrides"), + override_request_overrides, + ) if ( runtime_kwargs.get("api_key") and runtime_kwargs.get("credential_pool") is None diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 6aa7de68a8..3279f40439 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -2009,6 +2009,7 @@ class GatewaySlashCommandsMixin: "api_key": result.api_key, "base_url": result.base_url, "api_mode": result.api_mode, + "request_overrides": dict(result.request_overrides or {}), } # Write-through the non-secret parts to the session @@ -2321,6 +2322,7 @@ class GatewaySlashCommandsMixin: "api_key": result.api_key, "base_url": result.base_url, "api_mode": result.api_mode, + "request_overrides": dict(result.request_overrides or {}), } if one_turn: if not hasattr(self, "_pending_one_turn_model_restores"): diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index d4689df874..dc25def3b3 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -620,6 +620,7 @@ class ModelSwitchResult: api_key: str = "" base_url: str = "" api_mode: str = "" + request_overrides: Optional[dict] = None error_message: str = "" warning_message: str = "" provider_label: str = "" @@ -2192,6 +2193,7 @@ def switch_model( api_key=api_key, base_url=base_url, api_mode=api_mode, + request_overrides=dict(request_overrides or {}), warning_message=" | ".join(warnings) if warnings else "", provider_label=provider_label, resolved_via_alias=resolved_alias, diff --git a/tests/gateway/test_custom_provider_request_overrides.py b/tests/gateway/test_custom_provider_request_overrides.py new file mode 100644 index 0000000000..6b40087123 --- /dev/null +++ b/tests/gateway/test_custom_provider_request_overrides.py @@ -0,0 +1,220 @@ +"""Regression tests for gateway preservation of provider-derived request_overrides. + +Named custom providers can return request_overrides (for example +``extra_body.text.verbosity`` for OpenAI Responses). The gateway must preserve +those overrides on the runtime path and merge fast-mode overrides on top rather +than replacing them with an empty dict. +""" + +from __future__ import annotations + +import asyncio +import sys +import threading +import types +from types import SimpleNamespace +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +import gateway.run as gateway_run +from gateway.config import Platform +from gateway.session import SessionSource + + +class _CapturingAgent: + last_init = None + + def __init__(self, *args, **kwargs): + type(self).last_init = dict(kwargs) + self.tools = [] + self.request_overrides = dict(kwargs.get("request_overrides") or {}) + + def run_conversation(self, user_message: str, conversation_history=None, task_id=None): + return { + "final_response": "ok", + "messages": [], + "api_calls": 1, + } + + +def _install_fake_agent(monkeypatch): + fake_run_agent = types.ModuleType("run_agent") + fake_run_agent.AIAgent = _CapturingAgent + monkeypatch.setitem(sys.modules, "run_agent", fake_run_agent) + + +def _make_runner(): + runner = object.__new__(gateway_run.GatewayRunner) + runner.adapters = {} + runner.session_store = None + runner.config = None + runner._voice_mode = {} + runner._ephemeral_system_prompt = "" + runner._prefill_messages = [] + runner._reasoning_config = None + runner._show_reasoning = False + runner._provider_routing = {} + runner._fallback_model = None + runner._service_tier = None + runner._running_agents = {} + runner._running_agents_ts = {} + runner._background_tasks = set() + runner._session_db = None + runner._session_model_overrides = {} + runner._session_reasoning_overrides = {} + runner._pending_model_notes = {} + runner._pending_approvals = {} + runner._agent_cache = {} + runner._agent_cache_lock = threading.Lock() + runner._get_or_create_gateway_honcho = lambda session_key: (None, None) + runner.hooks = MagicMock() + runner.hooks.emit = AsyncMock() + runner.hooks.loaded_hooks = [] + return runner + + +def _make_source() -> SessionSource: + return SessionSource( + platform=Platform.FEISHU, + chat_id="ou_test", + chat_type="dm", + user_id="user-1", + user_name="tester", + ) + + +def test_resolve_runtime_agent_kwargs_preserves_request_overrides(monkeypatch): + monkeypatch.setattr( + "hermes_cli.runtime_provider.resolve_runtime_provider", + lambda: { + "api_key": "***", + "base_url": "https://example.test/v1", + "provider": "custom", + "api_mode": "codex_responses", + "command": None, + "args": [], + "credential_pool": None, + "request_overrides": { + "extra_body": {"text": {"verbosity": "low"}}, + }, + }, + ) + + result = gateway_run._resolve_runtime_agent_kwargs() + + assert result["request_overrides"] == { + "extra_body": {"text": {"verbosity": "low"}}, + } + + +def test_turn_route_preserves_provider_request_overrides_without_fast_mode(): + runner = _make_runner() + runner._service_tier = None + runtime_kwargs = { + "api_key": "***", + "base_url": "https://example.test/v1", + "provider": "custom", + "api_mode": "codex_responses", + "command": None, + "args": [], + "credential_pool": None, + "request_overrides": { + "extra_body": {"text": {"verbosity": "low"}}, + }, + } + + route = gateway_run.GatewayRunner._resolve_turn_agent_config( + runner, + "hi", + "gpt-5.4", + runtime_kwargs, + ) + + assert route["request_overrides"] == { + "extra_body": {"text": {"verbosity": "low"}}, + } + + +def test_turn_route_merges_fast_mode_with_provider_request_overrides(): + runner = _make_runner() + runner._service_tier = "priority" + runtime_kwargs = { + "api_key": "***", + "base_url": "https://example.test/v1", + "provider": "custom", + "api_mode": "codex_responses", + "command": None, + "args": [], + "credential_pool": None, + "request_overrides": { + "extra_body": {"text": {"verbosity": "low"}}, + }, + } + + with patch( + "hermes_cli.models.resolve_fast_mode_overrides", + return_value={"service_tier": "priority"}, + ): + route = gateway_run.GatewayRunner._resolve_turn_agent_config( + runner, + "hi", + "gpt-5.4", + runtime_kwargs, + ) + + assert route["request_overrides"] == { + "extra_body": {"text": {"verbosity": "low"}}, + "service_tier": "priority", + } + + +@pytest.mark.asyncio +async def test_run_agent_preserves_provider_request_overrides_on_gateway_path(monkeypatch): + monkeypatch.setattr(gateway_run, "_load_gateway_config", lambda: {}) + monkeypatch.setattr(gateway_run, "load_dotenv", lambda *args, **kwargs: None) + monkeypatch.setattr(gateway_run, "_load_gateway_runtime_config", lambda: {}) + monkeypatch.setattr(gateway_run, "_resolve_gateway_model", lambda config=None: "gpt-5.4") + monkeypatch.setattr( + gateway_run, + "_resolve_runtime_agent_kwargs", + lambda: { + "provider": "custom", + "api_mode": "codex_responses", + "base_url": "https://example.test/v1", + "api_key": "***", + "request_overrides": { + "extra_body": {"text": {"verbosity": "low"}}, + }, + }, + ) + _install_fake_agent(monkeypatch) + + import hermes_cli.tools_config as tools_config + + monkeypatch.setattr(tools_config, "_get_platform_tools", lambda user_config, platform_key: {"core"}) + + runner = _make_runner() + source = _make_source() + session_key = "agent:main:feishu:dm:ou_test" + + runner.session_store = SimpleNamespace( + get_or_create_session=lambda _source: SimpleNamespace(session_id="session-1"), + load_transcript=lambda _session_id: [], + ) + + _CapturingAgent.last_init = None + result = await runner._run_agent( + message="hi", + context_prompt="", + history=[], + source=source, + session_id="session-1", + session_key=session_key, + ) + + assert result["final_response"] == "ok" + assert _CapturingAgent.last_init is not None + assert _CapturingAgent.last_init["request_overrides"] == { + "extra_body": {"text": {"verbosity": "low"}}, + } diff --git a/tests/gateway/test_model_command_request_overrides.py b/tests/gateway/test_model_command_request_overrides.py new file mode 100644 index 0000000000..f37b88d7e6 --- /dev/null +++ b/tests/gateway/test_model_command_request_overrides.py @@ -0,0 +1,94 @@ +"""Regression tests for gateway /model preserving named-custom request_overrides.""" + +import pytest + +from gateway.config import Platform +from gateway.platforms.base import MessageEvent, MessageType +from gateway.run import GatewayRunner +from gateway.session import SessionSource + + +def _make_runner(): + runner = object.__new__(GatewayRunner) + runner.adapters = {} + runner._voice_mode = {} + runner._session_model_overrides = {} + runner._pending_model_notes = {} + runner._agent_cache = {} + runner._agent_cache_lock = None + runner._session_db = None + runner._evict_cached_agent = lambda _session_key: None + runner.session_store = None + return runner + + +def _make_event(text="/model"): + return MessageEvent( + text=text, + message_type=MessageType.TEXT, + source=SessionSource( + platform=Platform.FEISHU, + chat_id="ou_test", + chat_type="dm", + user_id="user-1", + ), + ) + + +@pytest.mark.asyncio +async def test_handle_model_command_stores_request_overrides_for_named_custom_provider( + tmp_path, + monkeypatch, +): + import gateway.run as gateway_run + from hermes_cli.model_switch import ModelSwitchResult + + hermes_home = tmp_path / ".hermes" + hermes_home.mkdir() + (hermes_home / "config.yaml").write_text( + """ +model: + default: gpt-5.4 + provider: openai-codex +providers: {} +custom_providers: + - name: Local (127.0.0.1:4141) + base_url: http://127.0.0.1:4141/v1 + model: rotator-openrouter-coding + extra_body: + text: + verbosity: low +""".lstrip(), + encoding="utf-8", + ) + + monkeypatch.setattr(gateway_run, "_hermes_home", hermes_home) + monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {}) + monkeypatch.setattr( + "hermes_cli.model_switch.switch_model", + lambda **kw: ModelSwitchResult( + success=True, + new_model="rotator-openrouter-coding", + target_provider="custom:local-(127.0.0.1:4141)", + provider_changed=True, + api_key="no-key-required", + base_url="http://127.0.0.1:4141/v1", + api_mode="codex_responses", + request_overrides={ + "extra_body": {"text": {"verbosity": "low"}}, + }, + provider_label="Local (127.0.0.1:4141)", + is_global=False, + ), + ) + + runner = _make_runner() + event = _make_event("/model rotator-openrouter-coding --provider custom:local-(127.0.0.1:4141)") + + result = await runner._handle_model_command(event) + + assert result is not None + session_key = runner._session_key_for_source(event.source) + assert runner._session_model_overrides[session_key]["request_overrides"] == { + "extra_body": {"text": {"verbosity": "low"}}, + } From a9b696c671ddd65c67d71d7b3d7d170d5a1753ae Mon Sep 17 00:00:00 2001 From: CharZhou <17255546+CharZhou@users.noreply.github.com> Date: Thu, 20 Aug 2026 09:44:22 +0800 Subject: [PATCH 011/117] fix(model): initialize switch request overrides --- hermes_cli/model_switch.py | 1 + 1 file changed, 1 insertion(+) diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index dc25def3b3..f26e3f19ec 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -1497,6 +1497,7 @@ def switch_model( from hermes_cli.runtime_provider import resolve_runtime_provider resolved_alias = "" + request_overrides: dict = {} new_model = raw_input.strip() target_provider = current_provider resolved_moa_preset = False From d2af9900431b3dcabfb1d36760f6c8b1d21949ad Mon Sep 17 00:00:00 2001 From: Jack <4762467+apollo-orbit-dev@users.noreply.github.com> Date: Sat, 27 Jun 2026 12:07:50 -0500 Subject: [PATCH 012/117] fix(agent): carry request_overrides through in-place /model switch (TUI/CLI) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third in the series. The gateway rebuild path (previous two commits) carries a custom provider's `request_overrides` (`extra_body`, e.g. `chat_template_kwargs`) into the agent, but the *in-place* live switch used by the TUI dashboard and the CLI — `agent.switch_model()` -> `agent_runtime_helpers.switch_model()` — swapped model/provider/base_url/api_key without ever updating `request_overrides`. So a `/model` switch to a thinking-enabled custom provider in the TUI/CLI kept the previous provider's `extra_body`. `switch_model()` now re-derives the switched-to provider's `request_overrides` (via `_get_named_custom_provider`) and applies it in place, preserving non-provider overrides (`service_tier`/`speed` from `/fast`). Logic factored into `_apply_switched_provider_request_overrides` for testability. Adds tests/agent/test_switch_model_request_overrides.py. --- agent/agent_runtime_helpers.py | 30 +++++++++++ .../test_switch_model_request_overrides.py | 52 +++++++++++++++++++ 2 files changed, 82 insertions(+) create mode 100644 tests/agent/test_switch_model_request_overrides.py diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index dd30d99ed5..20699283e5 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2848,6 +2848,29 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo return client +def _apply_switched_provider_request_overrides(agent, new_provider): + """Re-derive the switched-to provider's ``request_overrides`` onto a live agent. + + A ``custom_providers`` entry can carry an ``extra_body`` (e.g. + ``chat_template_kwargs`` to toggle a local model's thinking). The gateway + rebuild path carries this via ``request_overrides``; an *in-place* swap + (CLI / TUI ``/model``) must re-derive it for the new provider, otherwise the + previous provider's ``extra_body`` lingers. Non-provider overrides + (``service_tier`` / ``speed`` from ``/fast``) are preserved. + """ + from hermes_cli.runtime_provider import ( + _get_named_custom_provider, + _custom_provider_request_overrides, + ) + cp = _get_named_custom_provider(new_provider) + new_ro = _custom_provider_request_overrides(cp) if cp else None + overrides = dict(getattr(agent, "request_overrides", {}) or {}) + overrides.pop("extra_body", None) + if new_ro and new_ro.get("extra_body"): + overrides["extra_body"] = new_ro["extra_body"] + agent.request_overrides = overrides + + def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mode=''): """Switch the model/provider in-place for a live agent. @@ -3278,6 +3301,13 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo agent._fallback_chain = fallback_chain agent._fallback_model = fallback_chain[0] if fallback_chain else None + # Apply the switched-to provider's request_overrides (custom_providers + # extra_body, e.g. chat_template_kwargs). See helper for rationale. + try: + _apply_switched_provider_request_overrides(agent, new_provider) + except Exception: + logger.debug("switch_model: request_overrides re-derivation failed", exc_info=True) + logger.info( "Model switched in-place: %s (%s) -> %s (%s)", old_model, old_provider, new_model, new_provider, diff --git a/tests/agent/test_switch_model_request_overrides.py b/tests/agent/test_switch_model_request_overrides.py new file mode 100644 index 0000000000..30543e896d --- /dev/null +++ b/tests/agent/test_switch_model_request_overrides.py @@ -0,0 +1,52 @@ +"""Regression tests for the in-place /model switch (CLI/TUI) carrying a custom +provider's request_overrides (extra_body) — _apply_switched_provider_request_overrides. + +Before the fix, agent_runtime_helpers.switch_model() swapped model/provider/ +base_url/api_key in place but never touched request_overrides, so a /model +switch to a thinking-enabled custom provider in the TUI/CLI kept the old +provider's extra_body. +""" + +import agent.agent_runtime_helpers as arh + + +class _Agent: + pass + + +def test_switch_applies_new_provider_extra_body(monkeypatch): + a = _Agent() + a.request_overrides = {"service_tier": "priority"} # pre-existing /fast override + monkeypatch.setattr( + "hermes_cli.runtime_provider._get_named_custom_provider", + lambda name: {"name": "main-think", + "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}}, + ) + arh._apply_switched_provider_request_overrides(a, "custom:main-think") + assert a.request_overrides["extra_body"] == {"chat_template_kwargs": {"enable_thinking": True}} + assert a.request_overrides["service_tier"] == "priority" # preserved + + +def test_switch_to_noncustom_clears_stale_extra_body(monkeypatch): + a = _Agent() + a.request_overrides = { + "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}, + "service_tier": "priority", + } + monkeypatch.setattr( + "hermes_cli.runtime_provider._get_named_custom_provider", lambda name: None + ) + arh._apply_switched_provider_request_overrides(a, "anthropic") + assert "extra_body" not in a.request_overrides # stale extra_body cleared + assert a.request_overrides["service_tier"] == "priority" # preserved + + +def test_switch_from_none_overrides(monkeypatch): + a = _Agent() + a.request_overrides = None + monkeypatch.setattr( + "hermes_cli.runtime_provider._get_named_custom_provider", + lambda name: {"name": "main", "extra_body": {"chat_template_kwargs": {"enable_thinking": False}}}, + ) + arh._apply_switched_provider_request_overrides(a, "custom:main") + assert a.request_overrides == {"extra_body": {"chat_template_kwargs": {"enable_thinking": False}}} From 2f469d7e1e4b062c9ad8a481c9f1cffb10cd81eb Mon Sep 17 00:00:00 2001 From: Heng Cai Date: Sat, 29 Aug 2026 18:33:48 -0700 Subject: [PATCH 013/117] fix(gateway): merge instead of overwrite agent.request_overrides on reused turns Preserve initialization-time request overrides (custom-provider extra_body merged at agent construction) while replacing only the previous turn's routing overrides during the per-turn agent refresh. This keeps custom-provider extra_body settings without leaving stale fast-mode service_tier or speed values on cached agents. Reimplemented from PR #52432 at the code's current location (the per-turn refresh moved into the TurnRunner path since the original patch), keeping the original merge semantics: snapshot this turn's route overrides in agent._gateway_turn_request_overrides, evict only unchanged previous-turn keys, then layer the new turn overrides on top. Salvaged-from: #52432 Co-authored-by: Heng Cai --- gateway/run.py | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/gateway/run.py b/gateway/run.py index 1dcf8edb54..b84c12b062 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -5998,7 +5998,23 @@ class TurnRunner: agent.event_callback = ctx._event_callback_sync agent.reasoning_config = reasoning_config agent.service_tier = self._runner._service_tier - agent.request_overrides = turn_route.get("request_overrides") or {} + # Merge, never overwrite: init-time request overrides (e.g. a custom + # provider's extra_body merged at agent construction) must survive + # every reused-agent turn. Drop only the PREVIOUS turn's routing + # overrides (fast-mode service_tier/speed) before layering this + # turn's route overrides on top, so stale per-turn values never + # linger while construction-time values persist. + request_overrides = dict(getattr(agent, "request_overrides", {}) or {}) + previous_turn_overrides = dict( + getattr(agent, "_gateway_turn_request_overrides", {}) or {} + ) + for key, value in previous_turn_overrides.items(): + if request_overrides.get(key) == value: + request_overrides.pop(key, None) + turn_request_overrides = dict(turn_route.get("request_overrides") or {}) + request_overrides.update(turn_request_overrides) + agent.request_overrides = request_overrides + agent._gateway_turn_request_overrides = turn_request_overrides # Must-deliver notes for THIS turn ride the current user message # (api_content sidecar), never the system prompt: staged by # _handle_message_with_agent (auto-reset note, first-contact From fc00e36c6b38a2ba2b782db3467afd856d95f9e1 Mon Sep 17 00:00:00 2001 From: Jack <4762467+apollo-orbit-dev@users.noreply.github.com> Date: Sat, 27 Jun 2026 12:07:50 -0500 Subject: [PATCH 014/117] fix(gateway): preserve custom-provider request_overrides on agent turns A `custom_providers` entry can carry an `extra_body` (e.g. `chat_template_kwargs` to toggle a local vLLM model's thinking). `resolve_runtime_provider()` correctly surfaces it as `request_overrides` on the resolved runtime dict, but the gateway never plumbed it through to the per-turn agent: - `_resolve_runtime_agent_kwargs()` rebuilt the runtime dict from a fixed key whitelist that omitted `request_overrides`. - `_resolve_turn_agent_config()` rebuilt `runtime` from the same whitelist and set `route["request_overrides"]` solely from `/fast` service-tier overrides (`{}` otherwise). - The per-turn `agent.request_overrides = turn_route.get(...)` assignment then clobbered the value `_merge_custom_provider_extra_body()` applied at agent construction. Net: on the gateway, a custom provider's configured `extra_body` never reached the model -- only `/fast` overrides survived. The CLI/TUI path (which does not go through `_resolve_turn_agent_config`) and the auxiliary client (which sends `extra_body` directly) were unaffected. Fix: carry `request_overrides` through the runtime resolvers (`_resolve_runtime_agent_kwargs`, `_try_resolve_fallback_provider`) and merge the provider overrides into the per-turn route, layering any `/fast` service-tier overrides on top (top-level keys, no collision with `extra_body`). Adds tests/gateway/test_turn_request_overrides.py. Known follow-up: the mid-session `/model`-switch override path (`_session_model_overrides` / `ModelSwitchResult`) does not yet carry `request_overrides`. --- gateway/run.py | 19 ++++ tests/gateway/test_turn_request_overrides.py | 91 ++++++++++++++++++++ 2 files changed, 110 insertions(+) create mode 100644 tests/gateway/test_turn_request_overrides.py diff --git a/gateway/run.py b/gateway/run.py index b84c12b062..ffd8264c12 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -2971,6 +2971,11 @@ def _resolve_runtime_agent_kwargs() -> dict: "credential_pool": runtime.get("credential_pool"), "request_overrides": dict(runtime.get("request_overrides") or {}), "max_tokens": max_tokens, + # Per-provider request_overrides (e.g. a custom_providers ``extra_body`` + # carrying ``chat_template_kwargs``) resolved by resolve_runtime_provider(). + # Must flow through to the per-turn route or the provider's configured + # request body never reaches the model on the gateway path. + "request_overrides": runtime.get("request_overrides"), } @@ -3184,6 +3189,7 @@ def _try_resolve_fallback_provider() -> dict | None: "credential_pool": runtime.get("credential_pool"), "request_overrides": dict(runtime.get("request_overrides") or {}), "model": entry.get("model"), + "request_overrides": runtime.get("request_overrides"), } except Exception as fb_exc: logger.debug("Fallback entry %s failed: %s", entry.get("provider"), fb_exc) @@ -8477,6 +8483,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew enabled and the model supports Priority Processing / Anthropic fast mode, attach `request_overrides` so the API call is marked accordingly. + + Per-provider ``request_overrides`` resolved by + ``resolve_runtime_provider`` (e.g. a ``custom_providers`` ``extra_body`` + carrying ``chat_template_kwargs``) are preserved here and merged *under* + the fast-mode overrides, so a provider's configured request body still + reaches the model on the gateway turn path. """ from hermes_cli.models import resolve_fast_mode_overrides @@ -8506,6 +8518,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ), } + # Provider-level request_overrides (e.g. a custom_providers extra_body) + # resolved upstream by resolve_runtime_provider(). These were being + # dropped by the runtime whitelist above, so a custom provider's + # configured extra_body (chat_template_kwargs, etc.) never reached the + # model on the gateway path -- only /fast service-tier overrides did. service_tier = getattr(self, "_service_tier", None) if not service_tier: route["request_overrides"] = base_request_overrides @@ -8515,6 +8532,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew overrides = resolve_fast_mode_overrides(route["model"]) except Exception: overrides = None + # Fast-mode overrides (service_tier / speed) are top-level keys and do + # not collide with extra_body; deep-merge them over the provider overrides. route["request_overrides"] = _deep_merge_request_overrides( base_request_overrides, overrides or {}, diff --git a/tests/gateway/test_turn_request_overrides.py b/tests/gateway/test_turn_request_overrides.py new file mode 100644 index 0000000000..e65568d644 --- /dev/null +++ b/tests/gateway/test_turn_request_overrides.py @@ -0,0 +1,91 @@ +"""Regression tests: the gateway must preserve a custom provider's +``request_overrides`` on per-turn agent config. + +A ``custom_providers`` entry can carry an ``extra_body`` (e.g. +``chat_template_kwargs`` to toggle a local model's thinking). +``resolve_runtime_provider`` surfaces it as ``request_overrides`` on the +resolved runtime dict, but the gateway used to rebuild the runtime from a +fixed key whitelist that omitted it -- so the provider's configured +``extra_body`` never reached the model on the gateway path, and only +``/fast`` service-tier overrides survived. +""" + +import pytest + +from gateway.run import GatewayRunner + + +PROVIDER_OVERRIDES = {"extra_body": {"chat_template_kwargs": {"enable_thinking": True}}} + + +def _runtime_kwargs(**extra): + base = { + "api_key": "no-key-required", + "base_url": "http://10.0.0.1:8000/v1", + "provider": "custom", + "api_mode": "chat_completions", + "command": None, + "args": [], + "credential_pool": None, + "max_tokens": None, + } + base.update(extra) + return base + + +def _runner(service_tier=None): + runner = object.__new__(GatewayRunner) + runner._service_tier = service_tier + return runner + + +def test_provider_request_overrides_preserved_without_service_tier(): + """No /fast: the provider's extra_body must pass straight through.""" + runner = _runner(service_tier=None) + rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES) + route = runner._resolve_turn_agent_config("hi", "main", rk) + assert route["request_overrides"] == PROVIDER_OVERRIDES + # A copy, not an alias into runtime_kwargs. + assert route["request_overrides"] is not rk["request_overrides"] + + +def test_provider_request_overrides_merged_under_fast_mode(monkeypatch): + """/fast active: provider extra_body AND the service-tier marker both survive.""" + monkeypatch.setattr( + "hermes_cli.models.resolve_fast_mode_overrides", + lambda model_id: {"service_tier": "priority"}, + ) + runner = _runner(service_tier="priority") + rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES) + route = runner._resolve_turn_agent_config("hi", "main", rk) + assert route["request_overrides"]["extra_body"] == PROVIDER_OVERRIDES["extra_body"] + assert route["request_overrides"]["service_tier"] == "priority" + + +def test_no_provider_overrides_yields_empty(): + """Regression: absent provider overrides, behaviour is unchanged ({}).""" + runner = _runner(service_tier=None) + route = runner._resolve_turn_agent_config("hi", "main", _runtime_kwargs()) + assert route["request_overrides"] == {} + + +def test_resolve_runtime_agent_kwargs_carries_request_overrides(monkeypatch): + """The module-level runtime resolver must not drop request_overrides.""" + import gateway.run as gateway_run + + fake_runtime = { + "api_key": "k", + "base_url": "http://10.0.0.1:8000/v1", + "provider": "custom", + "api_mode": "chat_completions", + "request_overrides": PROVIDER_OVERRIDES, + } + monkeypatch.setattr( + "hermes_cli.runtime_provider.resolve_runtime_provider", + lambda *a, **k: dict(fake_runtime), + ) + monkeypatch.setattr( + "hermes_cli.runtime_provider._get_model_config", lambda: {} + ) + rk = gateway_run._resolve_runtime_agent_kwargs() + assert rk["request_overrides"] == PROVIDER_OVERRIDES From 5d238be2ca4ec5abdd8785902f7d6c7e11bfde9f Mon Sep 17 00:00:00 2001 From: Jack <4762467+apollo-orbit-dev@users.noreply.github.com> Date: Sat, 27 Jun 2026 12:07:50 -0500 Subject: [PATCH 015/117] fix(gateway): carry request_overrides through /model session overrides Follow-up to the previous commit (which fixed the default/fallback provider path). A mid-session `/model` switch stores a per-session override bundle in `_session_model_overrides` that omitted `request_overrides`, and the two consumers (`_resolve_session_agent_runtime` fast path and `_apply_session_model_override`) only copied provider/api_key/base_url/api_mode. So switching *to* a custom provider via `/model` did not apply its `extra_body`. - `ModelSwitchResult` gains a `request_overrides` field, derived for the switched provider via `_get_named_custom_provider` / `_custom_provider_request_overrides` (the same overrides `resolve_runtime_provider` surfaces for the default path). - Both `/model` override-storage sites in slash_commands.py persist it. - Both consumers apply it; `_apply_session_model_override` also clears a stale value when switching to a provider that has none. Extends tests/gateway/test_turn_request_overrides.py (3 new cases). --- gateway/run.py | 17 ++++--- hermes_cli/model_switch.py | 16 +++++++ tests/gateway/test_turn_request_overrides.py | 49 ++++++++++++++++++++ 3 files changed, 76 insertions(+), 6 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index ffd8264c12..df62736448 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -8348,6 +8348,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "api_mode": override.get("api_mode"), "max_tokens": override.get("max_tokens"), "credential_pool": override.get("credential_pool"), + "request_overrides": override.get("request_overrides"), } if override_runtime.get("api_key"): if override_runtime.get("credential_pool") is None: @@ -27501,12 +27502,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew val = override.get(key) if val is not None: runtime_kwargs[key] = val - override_request_overrides = override.get("request_overrides") - if isinstance(override_request_overrides, dict): - runtime_kwargs["request_overrides"] = _deep_merge_request_overrides( - runtime_kwargs.get("request_overrides"), - override_request_overrides, - ) + # request_overrides reflects the switched-to provider; apply whenever + # the override recorded it (even as None) so switching to a provider + # without configured overrides clears a stale value left by the + # default provider's runtime resolution. + if "request_overrides" in override: + override_request_overrides = override.get("request_overrides") + if isinstance(override_request_overrides, dict) and override_request_overrides: + runtime_kwargs["request_overrides"] = dict(override_request_overrides) + else: + runtime_kwargs["request_overrides"] = override_request_overrides if ( runtime_kwargs.get("api_key") and runtime_kwargs.get("credential_pool") is None diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index f26e3f19ec..c50df9e104 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -2185,6 +2185,22 @@ def switch_model( if hermes_warn: warnings.append(hermes_warn) + # Carry the switched provider's request_overrides (e.g. a custom_providers + # ``extra_body`` such as chat_template_kwargs) so a ``/model`` switch to a + # custom provider applies it on the gateway, matching the default-provider + # path. resolve_runtime_provider surfaces these for named custom providers. + request_overrides = None + try: + from hermes_cli.runtime_provider import ( + _get_named_custom_provider, + _custom_provider_request_overrides, + ) + _cp_for_ro = _get_named_custom_provider(target_provider) + if _cp_for_ro: + request_overrides = _custom_provider_request_overrides(_cp_for_ro) or None + except Exception: + request_overrides = None + # --- Build result --- return ModelSwitchResult( success=True, diff --git a/tests/gateway/test_turn_request_overrides.py b/tests/gateway/test_turn_request_overrides.py index e65568d644..c985125176 100644 --- a/tests/gateway/test_turn_request_overrides.py +++ b/tests/gateway/test_turn_request_overrides.py @@ -89,3 +89,52 @@ def test_resolve_runtime_agent_kwargs_carries_request_overrides(monkeypatch): ) rk = gateway_run._resolve_runtime_agent_kwargs() assert rk["request_overrides"] == PROVIDER_OVERRIDES + + +# --- /model session-override follow-up: request_overrides must survive a switch --- + +def test_session_override_applies_request_overrides(): + """A /model switch to a custom provider carries its extra_body into runtime.""" + runner = object.__new__(GatewayRunner) + runner._session_model_overrides = { + "sess1": { + "model": "thinkmodel", + "provider": "custom", + "api_key": "k", + "base_url": "http://10.0.0.1:8000/v1", + "api_mode": "chat_completions", + "request_overrides": PROVIDER_OVERRIDES, + } + } + rk = _runtime_kwargs() # default resolution carried no overrides + model, out = runner._apply_session_model_override("sess1", "oldmodel", rk) + assert model == "thinkmodel" + assert out["request_overrides"] == PROVIDER_OVERRIDES + + +def test_session_override_clears_stale_request_overrides(): + """Switching to a provider with no overrides clears a stale value.""" + runner = object.__new__(GatewayRunner) + runner._session_model_overrides = { + "sess1": { + "model": "plain", + "provider": "openrouter", + "api_key": "k", + "base_url": "https://openrouter.ai/api/v1", + "api_mode": "chat_completions", + "request_overrides": None, + } + } + rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES) # stale, from default + _, out = runner._apply_session_model_override("sess1", "old", rk) + assert out.get("request_overrides") is None + + +def test_session_override_absent_is_noop(): + """No override for the session leaves runtime_kwargs untouched.""" + runner = object.__new__(GatewayRunner) + runner._session_model_overrides = {} + rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES) + model, out = runner._apply_session_model_override("nope", "keepme", rk) + assert model == "keepme" + assert out["request_overrides"] == PROVIDER_OVERRIDES From b10b27e6f9f2c7a864c68ed5e78c11a91219dbea Mon Sep 17 00:00:00 2001 From: Jack <4762467+apollo-orbit-dev@users.noreply.github.com> Date: Wed, 15 Jul 2026 08:17:47 -0500 Subject: [PATCH 016/117] fix(agent): match switched-to custom provider by model+base_url, not name Addresses the hermes-sweeper review on #53765. The in-place /model switch helper (_apply_switched_provider_request_overrides) derived a custom provider's extra_body by provider *name* only, while build-time matching in agent_init._merge_custom_provider_extra_body matches by provider key, base_url, AND model. So a different model selected at the same named endpoint could inherit an extra_body configured for another model. Reuse the shared agent_init._custom_provider_extra_body_for_agent matcher (provider key + base_url + model), sourcing custom_providers from the init-time agent._custom_providers cache (fresh-load fallback if absent). A stale extra_body is always cleared when no entry matches; non-provider overrides (service_tier / speed from /fast) are preserved. Tests: add nonmatching-model and endpoint-mismatch regressions; update the existing switch tests onto the model/base_url-aware matcher. --- agent/agent_runtime_helpers.py | 44 +++++-- .../test_switch_model_request_overrides.py | 107 ++++++++++++++---- 2 files changed, 119 insertions(+), 32 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 20699283e5..7b5b9b081a 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2854,20 +2854,42 @@ def _apply_switched_provider_request_overrides(agent, new_provider): A ``custom_providers`` entry can carry an ``extra_body`` (e.g. ``chat_template_kwargs`` to toggle a local model's thinking). The gateway rebuild path carries this via ``request_overrides``; an *in-place* swap - (CLI / TUI ``/model``) must re-derive it for the new provider, otherwise the - previous provider's ``extra_body`` lingers. Non-provider overrides - (``service_tier`` / ``speed`` from ``/fast``) are preserved. + (CLI / TUI ``/model``) must re-derive it for the switched-to provider, + otherwise the previous provider's ``extra_body`` lingers. + + The switched-to entry is matched by **provider key, base_url, and model** — + the same condition ``agent_init._merge_custom_provider_extra_body`` applies + at build time — via the shared ``_custom_provider_extra_body_for_agent`` + matcher. Matching by name alone would let a *different* model selected at the + same named endpoint inherit an ``extra_body`` configured for another model. + A stale ``extra_body`` is always cleared when the switched-to provider/model + resolves none; non-provider overrides (``service_tier`` / ``speed`` from + ``/fast``) are preserved. """ - from hermes_cli.runtime_provider import ( - _get_named_custom_provider, - _custom_provider_request_overrides, + from agent.agent_init import _custom_provider_extra_body_for_agent + + # Prefer the init-time cache (agent_init stores ``agent._custom_providers`` + # right where it runs its own _merge_custom_provider_extra_body); fall back + # to a fresh load only if a caller built the agent without it. + custom_providers = getattr(agent, "_custom_providers", None) + if custom_providers is None: + try: + from hermes_cli.config import load_config, get_compatible_custom_providers + custom_providers = get_compatible_custom_providers(load_config()) + except Exception: + custom_providers = [] + + new_extra_body = _custom_provider_extra_body_for_agent( + provider=new_provider, + model=getattr(agent, "model", "") or "", + base_url=getattr(agent, "base_url", "") or "", + custom_providers=custom_providers or [], ) - cp = _get_named_custom_provider(new_provider) - new_ro = _custom_provider_request_overrides(cp) if cp else None + overrides = dict(getattr(agent, "request_overrides", {}) or {}) - overrides.pop("extra_body", None) - if new_ro and new_ro.get("extra_body"): - overrides["extra_body"] = new_ro["extra_body"] + overrides.pop("extra_body", None) # always drop the previous provider's extra_body + if new_extra_body: + overrides["extra_body"] = dict(new_extra_body) agent.request_overrides = overrides diff --git a/tests/agent/test_switch_model_request_overrides.py b/tests/agent/test_switch_model_request_overrides.py index 30543e896d..512b71f9ed 100644 --- a/tests/agent/test_switch_model_request_overrides.py +++ b/tests/agent/test_switch_model_request_overrides.py @@ -5,6 +5,11 @@ Before the fix, agent_runtime_helpers.switch_model() swapped model/provider/ base_url/api_key in place but never touched request_overrides, so a /model switch to a thinking-enabled custom provider in the TUI/CLI kept the old provider's extra_body. + +The switched-to entry is matched by provider key + base_url + model (the same +condition agent_init._merge_custom_provider_extra_body uses at build time), so a +*different* model selected at the same named endpoint does not inherit an +extra_body configured for another model. """ import agent.agent_runtime_helpers as arh @@ -14,39 +19,99 @@ class _Agent: pass -def test_switch_applies_new_provider_extra_body(monkeypatch): +# Two entries share the same named endpoint / base_url but pin different models — +# the exact case a name-only match got wrong. +CUSTOM_PROVIDERS = [ + { + "name": "main-think", + "base_url": "http://10.0.0.1:8000/v1", + "model": "think-model", + "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}, + }, + { + "name": "main-plain", + "base_url": "http://10.0.0.1:8000/v1", + "model": "plain-model", + "extra_body": {"chat_template_kwargs": {"enable_thinking": False}}, + }, +] + + +def _agent(*, model, base_url, request_overrides, custom_providers=CUSTOM_PROVIDERS): a = _Agent() - a.request_overrides = {"service_tier": "priority"} # pre-existing /fast override - monkeypatch.setattr( - "hermes_cli.runtime_provider._get_named_custom_provider", - lambda name: {"name": "main-think", - "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}}, + # switch_model() sets these on the live agent before calling the helper. + a.model = model + a.base_url = base_url + a.provider = "custom" + a.request_overrides = request_overrides + a._custom_providers = custom_providers # init-time cache the helper reads + return a + + +def test_switch_applies_matched_provider_extra_body(): + """Switching to the matching provider+model applies its extra_body and + preserves non-provider overrides (service_tier/speed from /fast).""" + a = _agent( + model="think-model", + base_url="http://10.0.0.1:8000/v1", + request_overrides={"service_tier": "priority"}, ) arh._apply_switched_provider_request_overrides(a, "custom:main-think") assert a.request_overrides["extra_body"] == {"chat_template_kwargs": {"enable_thinking": True}} assert a.request_overrides["service_tier"] == "priority" # preserved -def test_switch_to_noncustom_clears_stale_extra_body(monkeypatch): - a = _Agent() - a.request_overrides = { - "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}, - "service_tier": "priority", - } - monkeypatch.setattr( - "hermes_cli.runtime_provider._get_named_custom_provider", lambda name: None +def test_switch_to_noncustom_clears_stale_extra_body(): + """Switching to a built-in provider clears the previous provider's extra_body.""" + a = _agent( + model="claude-x", + base_url="https://api.anthropic.com", + request_overrides={ + "extra_body": {"chat_template_kwargs": {"enable_thinking": True}}, + "service_tier": "priority", + }, ) arh._apply_switched_provider_request_overrides(a, "anthropic") assert "extra_body" not in a.request_overrides # stale extra_body cleared assert a.request_overrides["service_tier"] == "priority" # preserved -def test_switch_from_none_overrides(monkeypatch): - a = _Agent() - a.request_overrides = None - monkeypatch.setattr( - "hermes_cli.runtime_provider._get_named_custom_provider", - lambda name: {"name": "main", "extra_body": {"chat_template_kwargs": {"enable_thinking": False}}}, +def test_switch_from_none_overrides(): + """A None request_overrides is handled and gets the matched extra_body.""" + a = _agent( + model="plain-model", + base_url="http://10.0.0.1:8000/v1", + request_overrides=None, ) - arh._apply_switched_provider_request_overrides(a, "custom:main") + arh._apply_switched_provider_request_overrides(a, "custom:main-plain") assert a.request_overrides == {"extra_body": {"chat_template_kwargs": {"enable_thinking": False}}} + + +def test_switch_to_different_model_same_endpoint_does_not_inherit(): + """Review regression: selecting a *different* model while naming a custom + provider must NOT inherit that provider's extra_body when the models differ. + + 'main-think' pins 'think-model'. Selecting 'plain-model' under + custom:main-think must not carry enable_thinking=True — the model-aware + matcher rejects the mismatch and the stale extra_body is cleared. (A + name-only match would have wrongly carried it over.) + """ + a = _agent( + model="plain-model", # differs from main-think's pinned 'think-model' + base_url="http://10.0.0.1:8000/v1", + request_overrides={"extra_body": {"chat_template_kwargs": {"enable_thinking": True}}}, + ) + arh._apply_switched_provider_request_overrides(a, "custom:main-think") + assert "extra_body" not in a.request_overrides # not inherited; stale cleared + + +def test_switch_endpoint_mismatch_does_not_inherit(): + """A matching provider *name* but a different base_url must not match either + (endpoint identity is part of the condition).""" + a = _agent( + model="think-model", + base_url="http://10.9.9.9:8000/v1", # different endpoint than the entry + request_overrides={"extra_body": {"chat_template_kwargs": {"enable_thinking": True}}}, + ) + arh._apply_switched_provider_request_overrides(a, "custom:main-think") + assert "extra_body" not in a.request_overrides # base_url mismatch -> cleared From 1859f95799418053683d126f29af251ba2fbb48c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:40:36 -0700 Subject: [PATCH 017/117] test(gateway): reused-agent merge-not-overwrite regression via real _run_agent Port the PR #52432 regression (fast turn then normal turn on a cached gateway agent) onto the current _run_agent harness: the original test's host file context no longer exists on main after the TurnRunner extraction, so the scenario is re-expressed with the existing _CapturingAgent fixture. Asserts init-time custom-provider extra_body survives both a /fast turn (service_tier layered on top) and the following normal turn (only the stale fast-mode key drops). Salvaged-from: #52432 Co-authored-by: Heng Cai --- .../test_custom_provider_request_overrides.py | 87 +++++++++++++++++++ 1 file changed, 87 insertions(+) diff --git a/tests/gateway/test_custom_provider_request_overrides.py b/tests/gateway/test_custom_provider_request_overrides.py index 6b40087123..4d9d172a3c 100644 --- a/tests/gateway/test_custom_provider_request_overrides.py +++ b/tests/gateway/test_custom_provider_request_overrides.py @@ -218,3 +218,90 @@ async def test_run_agent_preserves_provider_request_overrides_on_gateway_path(mo assert _CapturingAgent.last_init["request_overrides"] == { "extra_body": {"text": {"verbosity": "low"}}, } + +@pytest.mark.asyncio +async def test_reused_agent_turn_merges_request_overrides_not_overwrite(monkeypatch): + """Merge-not-overwrite regression (salvaged from PR #52432). + + A cached/reused gateway agent must keep its init-time request_overrides + (custom-provider extra_body) across turns: a /fast turn layers + service_tier ON TOP, and the following normal turn drops only the stale + fast-mode key while the provider extra_body survives. + """ + monkeypatch.setattr(gateway_run, "_load_gateway_config", lambda: {}) + monkeypatch.setattr(gateway_run, "load_dotenv", lambda *args, **kwargs: None) + monkeypatch.setattr(gateway_run, "_load_gateway_runtime_config", lambda: {}) + monkeypatch.setattr(gateway_run, "_resolve_gateway_model", lambda config=None: "gpt-5.4") + monkeypatch.setattr( + gateway_run, + "_resolve_runtime_agent_kwargs", + lambda: { + "provider": "custom", + "api_mode": "codex_responses", + "base_url": "https://example.test/v1", + "api_key": "***", + "request_overrides": { + "extra_body": {"text": {"verbosity": "low"}}, + }, + }, + ) + _install_fake_agent(monkeypatch) + + import hermes_cli.tools_config as tools_config + + monkeypatch.setattr(tools_config, "_get_platform_tools", lambda user_config, platform_key: {"core"}) + + runner = _make_runner() + source = _make_source() + session_key = "agent:main:feishu:dm:ou_test" + + runner.session_store = SimpleNamespace( + get_or_create_session=lambda _source: SimpleNamespace(session_id="session-1"), + load_transcript=lambda _session_id: [], + ) + + seen_agents = [] + orig_init = _CapturingAgent.__init__ + + def _tracking_init(self, *args, **kwargs): + orig_init(self, *args, **kwargs) + seen_agents.append(self) + + monkeypatch.setattr(_CapturingAgent, "__init__", _tracking_init) + + async def run_turn(): + return await runner._run_agent( + message="hi", + context_prompt="", + history=[], + source=source, + session_id="session-1", + session_key=session_key, + ) + + # Turn 1: /fast active — provider extra_body AND service_tier both present. + # The turn path re-resolves the tier per session, so stub the resolver. + tier_box = {"tier": "priority"} + runner._resolve_session_service_tier = lambda *a, **k: tier_box["tier"] + with patch( + "hermes_cli.models.resolve_fast_mode_overrides", + return_value={"service_tier": "priority"}, + ): + result = await run_turn() + assert result["final_response"] == "ok" + assert len(seen_agents) == 1 + agent = seen_agents[0] + assert agent.request_overrides == { + "extra_body": {"text": {"verbosity": "low"}}, + "service_tier": "priority", + } + + # Turn 2: back to normal — the SAME cached agent must drop only the stale + # fast-mode key; the init-time provider extra_body survives the refresh. + tier_box["tier"] = None + result = await run_turn() + assert result["final_response"] == "ok" + assert len(seen_agents) == 1, "agent should be reused from the gateway cache" + assert agent.request_overrides == { + "extra_body": {"text": {"verbosity": "low"}}, + } From 5a59ba82ddec952c963cb167666ff3c48c264d0a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:53:14 -0700 Subject: [PATCH 018/117] docs(providers): note extra_body survives gateway turns and /model switches; map gitabtion attribution --- contributors/emails/abtion@outlook.com | 1 + website/docs/integrations/providers.md | 2 ++ 2 files changed, 3 insertions(+) create mode 100644 contributors/emails/abtion@outlook.com diff --git a/contributors/emails/abtion@outlook.com b/contributors/emails/abtion@outlook.com new file mode 100644 index 0000000000..aa2f1ed4ed --- /dev/null +++ b/contributors/emails/abtion@outlook.com @@ -0,0 +1 @@ +gitabtion diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 4fe52265e3..0d4569accd 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -1351,6 +1351,8 @@ extra_body: enable_thinking: false ``` +The configured `extra_body` follows the provider everywhere: it is merged at agent construction, **survives every gateway turn** (including turns where `/fast` layers `service_tier`/`speed` overrides on top — those merge over your `extra_body` rather than replacing it), and is **re-derived on `/model` switches** — switching to a named custom provider applies its `extra_body`, and switching away clears it so it never leaks to another provider. + The `hermes model` → Custom Endpoint wizard now prompts for the API mode explicitly and persists your answer to `config.yaml` (as `transport` on the provider entry). URL-based auto-detection (e.g. `/anthropic` paths → `anthropic_messages`) still happens as a fallback when the field is left blank. **Native vision for custom-provider models.** If your custom endpoint serves a vision-capable model that isn't in models.dev, set `model.supports_vision: true` so Hermes routes attached images natively (as `image_url` parts) instead of pre-processing them through `vision_analyze`. Single knob — no need to also set `agent.image_input_mode: native`. From 91d60d2f9ea004eb937727d4d83fd264558366b2 Mon Sep 17 00:00:00 2001 From: RelaxJonh <92573950+RelaxJonh@users.noreply.github.com> Date: Fri, 31 Jul 2026 09:18:27 +0700 Subject: [PATCH 019/117] fix(fallback): re-resolve extra_body when activating fallback provider (#75091) `try_activate_fallback()` re-resolved `reasoning_config` for the new fallback provider (fix for #21256), but never re-resolved `extra_body`. The primary provider's `extra_body` (e.g. `reasoning_effort: "none"`) rode along onto the fallback provider, which is a different API that may reject those fields. Example: primary has `extra_body: {reasoning_effort: "none"}`, fallback is OpenRouter. After failover, every request to OpenRouter carries both the stray top-level `reasoning_effort` AND the nested `reasoning` object, and OpenRouter rejects the pair: HTTP 400: "reasoning_effort" and "reasoning.effort" are both provided The fallback is dead precisely when it is needed. Fix: after swapping provider/model/base_url, clear the primary's extra_body from request_overrides, then re-resolve from the fallback provider's config using the existing _merge_custom_provider_extra_body helper. Same pattern as the reasoning_config re-resolution above. --- agent/chat_completion_helpers.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 033699ebbe..2880bf17a5 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -2842,6 +2842,31 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool ) # Keep whatever reasoning_config was active — don't break the fallback swap. + # Re-resolve extra_body for the fallback provider (Closes #75091). + # The primary's provider-specific extra_body (e.g. reasoning_effort) + # must not ride along onto the fallback provider, which is a different + # API that may reject those fields. Clear the stale extra_body first, + # then re-resolve from the fallback provider's config. + try: + from agent.agent_init import _merge_custom_provider_extra_body + _custom_providers = getattr(agent, "_custom_providers", None) or [] + # Strip the primary's extra_body so it doesn't contaminate the + # fallback (existing_extra_body in _merge wins on conflict). + _overrides = dict(getattr(agent, "request_overrides", {}) or {}) + _overrides.pop("extra_body", None) + agent.request_overrides = _overrides + _merge_custom_provider_extra_body(agent, _custom_providers) + logger.info( + "Fallback %s: extra_body resolved: %s", + agent.model, + (getattr(agent, "request_overrides", {}) or {}).get("extra_body"), + ) + except Exception as _eb_err: + logger.debug( + "Failed to resolve extra_body for fallback %s; keeping current: %s", + agent.model, _eb_err, + ) + # Keep the prompt's self-identity in sync with the model actually # answering, so "what model are you?" doesn't report the primary. rewrite_prompt_model_identity(agent, fb_model, fb_provider) From 131501229a97ad9635a0dc6a363bb2558bc4e577 Mon Sep 17 00:00:00 2001 From: Adam Fortuna Date: Tue, 7 Jul 2026 09:21:48 +0200 Subject: [PATCH 020/117] fix(runtime): restore request_overrides after transport recovery Include request_overrides in primary runtime snapshots so transport recovery restores request-level model parameters. --- agent/agent_init.py | 1 + agent/agent_runtime_helpers.py | 2 + .../run_agent/test_primary_runtime_restore.py | 44 ++++++++++++++++++- 3 files changed, 46 insertions(+), 1 deletion(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index 49a29f1e13..63e8f4c436 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -3105,6 +3105,7 @@ def init_agent( "base_url": agent.base_url, "api_mode": agent.api_mode, "api_key": getattr(agent, "api_key", ""), + "request_overrides": dict(getattr(agent, "request_overrides", {}) or {}), "client_kwargs": dict(agent._client_kwargs), "use_prompt_caching": agent._use_prompt_caching, "use_native_cache_layout": agent._use_native_cache_layout, diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 7b5b9b081a..c518935ce7 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -1486,6 +1486,7 @@ def try_recover_primary_transport( agent._transport_cache.clear() agent.api_key = rt["api_key"] agent._reasoning_echo_flag = rt.get("reasoning_echo_flag", False) + agent.request_overrides = dict(rt.get("request_overrides") or {}) if agent.api_mode == "anthropic_messages": from agent.anthropic_adapter import build_anthropic_client @@ -1748,6 +1749,7 @@ def restore_primary_runtime(agent) -> bool: agent._transport_cache.clear() agent.api_key = rt["api_key"] agent._reasoning_echo_flag = rt.get("reasoning_echo_flag", False) + agent.request_overrides = dict(rt.get("request_overrides") or {}) agent._client_kwargs = dict(rt["client_kwargs"]) agent._use_prompt_caching = rt["use_prompt_caching"] # Default to native layout when the restored snapshot predates the diff --git a/tests/run_agent/test_primary_runtime_restore.py b/tests/run_agent/test_primary_runtime_restore.py index a3344f4f65..cc5bcfb5b5 100644 --- a/tests/run_agent/test_primary_runtime_restore.py +++ b/tests/run_agent/test_primary_runtime_restore.py @@ -30,7 +30,12 @@ def _make_tool_defs(*names: str) -> list: ] -def _make_agent(fallback_model=None, provider="custom", base_url="https://my-llm.example.com/v1"): +def _make_agent( + fallback_model=None, + provider="custom", + base_url="https://my-llm.example.com/v1", + request_overrides=None, +): """Create a minimal AIAgent with optional fallback config.""" with ( patch("run_agent.get_tool_definitions", return_value=_make_tool_defs("web_search")), @@ -45,6 +50,7 @@ def _make_agent(fallback_model=None, provider="custom", base_url="https://my-llm "agent.context_compressor.get_model_context_length", return_value=200_000, ), + patch("agent.anthropic_adapter.build_anthropic_client", return_value=MagicMock()), ): agent = AIAgent( api_key="test-key-12345678", @@ -54,6 +60,7 @@ def _make_agent(fallback_model=None, provider="custom", base_url="https://my-llm skip_context_files=True, skip_memory=True, fallback_model=fallback_model, + request_overrides=dict(request_overrides or {}), ) agent.client = MagicMock() return agent @@ -83,6 +90,13 @@ class TestPrimaryRuntimeSnapshot: assert "client_kwargs" in rt assert "compressor_context_length" in rt + def test_snapshot_includes_request_overrides(self): + overrides = {"extra_body": {"reasoning": {"effort": "medium"}}} + agent = _make_agent(request_overrides=overrides) + rt = agent._primary_runtime + assert rt["request_overrides"] == overrides + assert rt["request_overrides"] is not agent.request_overrides + def test_snapshot_includes_compressor_state(self): agent = _make_agent() rt = agent._primary_runtime @@ -280,6 +294,19 @@ class TestRestorePrimaryRuntime: assert agent._use_prompt_caching == original_caching + def test_restores_request_overrides(self): + original_overrides = {"extra_body": {"reasoning": {"effort": "medium"}}} + agent = _make_agent(request_overrides=original_overrides) + agent._fallback_activated = True + agent.request_overrides = {"extra_body": {"fallback_only": True}} + + with patch("run_agent.OpenAI", return_value=MagicMock()): + result = agent._restore_primary_runtime() + + assert result is True + assert agent.request_overrides == original_overrides + assert agent.request_overrides is not agent._primary_runtime["request_overrides"] + def test_restore_skips_cross_provider_pool_entry(self): """Restore must not swap in a fallback provider credential for the primary runtime.""" @@ -551,6 +578,21 @@ class TestTryRecoverPrimaryTransport: assert result is True + def test_recovery_restores_request_overrides(self): + original_overrides = {"extra_body": {"reasoning": {"effort": "medium"}}} + agent = _make_agent(provider="custom", request_overrides=original_overrides) + error = _make_transport_error("ReadTimeout") + agent.request_overrides = {"extra_body": {"fallback_only": True}} + + with patch("run_agent.OpenAI", return_value=MagicMock()), \ + patch("time.sleep"): + result = agent._try_recover_primary_transport( + error, retry_count=3, max_retries=3, + ) + + assert result is True + assert agent.request_overrides == original_overrides + assert agent.request_overrides is not agent._primary_runtime["request_overrides"] From 3b3ad958d7ecdd4ebbbe304c5695853699d77b79 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:38:31 -0700 Subject: [PATCH 021/117] fix(runtime): key-scoped fallback extra_body re-resolution + request_overrides in switch_model snapshot MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up hardening on the two cherry-picked contributor commits: - try_activate_fallback: replace the blanket request_overrides.pop('extra_body') with KEY-SCOPED removal — only keys the OLD provider's custom_providers entry contributed (value unchanged since the init-time merge) are dropped. Caller/profile-provided extra_body keys survive the swap, matching the caller-over-provider precedence in agent_init._merge_custom_provider_extra_body. The fallback provider's own extra_body is then merged back in. - switch_model: the live _primary_runtime snapshot it rebuilds now carries request_overrides, so a post-switch transport recovery or fallback restore reinstates the switched-to identity's overrides instead of dropping them. - Tests: activation-level stale-key removal + caller-override preservation (test_provider_fallback.py), switch-then-recover / switch-then-restore (test_primary_runtime_restore.py). Cache-safety: none of these paths mutate past context or rebuild the system prompt — only outbound request kwargs change. Fixes #75091 --- agent/agent_runtime_helpers.py | 5 ++ agent/chat_completion_helpers.py | 50 +++++++++-- .../run_agent/test_primary_runtime_restore.py | 76 ++++++++++++++++ tests/run_agent/test_provider_fallback.py | 88 +++++++++++++++++++ 4 files changed, 210 insertions(+), 9 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index c518935ce7..74dc486da2 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -3286,6 +3286,11 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo "use_native_cache_layout": agent._use_native_cache_layout, "reasoning_config": dict(agent.reasoning_config) if getattr(agent, "reasoning_config", None) else None, "reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False), + # Request-level overrides (extra_body etc.) must travel with the + # switched-to identity; without this, a post-switch transport + # recovery or fallback restore would resurrect the PRE-switch + # overrides via the stale init-time snapshot (#75091 seam). + "request_overrides": dict(getattr(agent, "request_overrides", {}) or {}), "compressor_model": getattr(_cc, "model", agent.model) if _cc else agent.model, "compressor_base_url": getattr(_cc, "base_url", agent.base_url) if _cc else agent.base_url, "compressor_api_key": getattr(_cc, "api_key", "") if _cc else "", diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 2880bf17a5..42f186184c 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -2674,6 +2674,7 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool old_model = agent.model old_provider = agent.provider + old_base_url = agent.base_url # Clear the per-config context_length override so the fallback # model's actual context window is resolved instead of inheriting @@ -2843,18 +2844,49 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool # Keep whatever reasoning_config was active — don't break the fallback swap. # Re-resolve extra_body for the fallback provider (Closes #75091). - # The primary's provider-specific extra_body (e.g. reasoning_effort) - # must not ride along onto the fallback provider, which is a different - # API that may reject those fields. Clear the stale extra_body first, - # then re-resolve from the fallback provider's config. + # The OLD provider's custom_providers-contributed extra_body (e.g. a + # vendor-specific reasoning toggle) must not ride along onto the + # fallback provider, which is a different API that may reject those + # fields. Removal is KEY-SCOPED: only keys the old provider's + # custom_providers entry contributed (value unchanged since init) + # are dropped; the fallback provider's own extra_body is then merged + # back in. Caller/profile-provided extra_body keys + # (request_overrides passed at init, which win over provider config + # per _merge_custom_provider_extra_body precedence) MUST survive the + # swap untouched. try: - from agent.agent_init import _merge_custom_provider_extra_body + from agent.agent_init import ( + _custom_provider_extra_body_for_agent, + _merge_custom_provider_extra_body, + ) _custom_providers = getattr(agent, "_custom_providers", None) or [] - # Strip the primary's extra_body so it doesn't contaminate the - # fallback (existing_extra_body in _merge wins on conflict). + # What did the OLD provider's config contribute? + _old_provider_eb = _custom_provider_extra_body_for_agent( + provider=old_provider, + model=old_model, + base_url=old_base_url, + custom_providers=_custom_providers, + ) or {} _overrides = dict(getattr(agent, "request_overrides", {}) or {}) - _overrides.pop("extra_body", None) - agent.request_overrides = _overrides + _existing_eb = _overrides.get("extra_body") + if isinstance(_existing_eb, dict) and _old_provider_eb: + _scrubbed = dict(_existing_eb) + for _k, _v in _old_provider_eb.items(): + # Drop only keys the old provider contributed: the value + # must still match what its config injected — a caller + # override of the same key would have won at init and + # differ, so it survives. Keys the new provider + # redefines are re-added with the NEW provider's value + # by the merge below. + if _k in _scrubbed and _scrubbed[_k] == _v: + _scrubbed.pop(_k) + if _scrubbed: + _overrides["extra_body"] = _scrubbed + else: + _overrides.pop("extra_body", None) + agent.request_overrides = _overrides + # Merge in the fallback provider's own extra_body (existing + # caller-provided keys win on conflict inside the merge helper). _merge_custom_provider_extra_body(agent, _custom_providers) logger.info( "Fallback %s: extra_body resolved: %s", diff --git a/tests/run_agent/test_primary_runtime_restore.py b/tests/run_agent/test_primary_runtime_restore.py index cc5bcfb5b5..72ad60dff3 100644 --- a/tests/run_agent/test_primary_runtime_restore.py +++ b/tests/run_agent/test_primary_runtime_restore.py @@ -790,3 +790,79 @@ class TestRateLimitCooldown: # second call should not have extended the cooldown assert second_cooldown == first_cooldown + + +# ============================================================================= +# request_overrides travels through the switch_model snapshot (#75091 seam) +# ============================================================================= + + +class TestSwitchModelRequestOverridesSnapshot: + """switch_model rebuilds _primary_runtime; it must carry request_overrides + so a post-switch transport recovery or fallback restore reinstates the + switched-to identity's overrides, not a stale or empty set.""" + + def _switch(self, agent, **kwargs): + from agent.agent_runtime_helpers import switch_model + + with ( + patch("run_agent.OpenAI", return_value=MagicMock()), + patch( + "agent.model_metadata.get_model_context_length", + return_value=128_000, + ), + ): + switch_model( + agent, + new_model=kwargs.get("new_model", "gpt-4o"), + new_provider=kwargs.get("new_provider", "openai"), + base_url=kwargs.get("base_url", "https://api.openai.com/v1"), + api_key=kwargs.get("api_key", "sk-test-1234567890"), + ) + + def test_switch_snapshot_carries_request_overrides(self): + overrides = {"extra_body": {"reasoning": {"effort": "high"}}} + agent = _make_agent(request_overrides=overrides) + self._switch(agent) + rt = agent._primary_runtime + assert rt.get("request_overrides") == overrides + assert rt["request_overrides"] is not agent.request_overrides + + def test_switch_then_recover_restores_current_overrides(self): + """After /model switch, a transport recovery must reinstate the + overrides that were live at switch time — not drop them.""" + overrides = {"extra_body": {"reasoning": {"effort": "high"}}} + agent = _make_agent(provider="custom", request_overrides=overrides) + self._switch( + agent, + new_model="local-model", + new_provider="custom", + base_url="https://my-llm.example.com/v1", + ) + # A fallback activation mid-turn clobbers the live overrides… + agent.request_overrides = {"extra_body": {"fallback_only": True}} + error = _make_transport_error("ReadTimeout") + with patch("run_agent.OpenAI", return_value=MagicMock()), \ + patch("time.sleep"): + result = agent._try_recover_primary_transport( + error, retry_count=3, max_retries=3, + ) + assert result is True + # …and recovery restores the switch-time snapshot. + assert agent.request_overrides == overrides + + def test_switch_then_restore_restores_current_overrides(self): + overrides = {"extra_body": {"reasoning": {"effort": "high"}}} + agent = _make_agent(provider="custom", request_overrides=overrides) + self._switch( + agent, + new_model="local-model", + new_provider="custom", + base_url="https://my-llm.example.com/v1", + ) + agent._fallback_activated = True + agent.request_overrides = {"extra_body": {"fallback_only": True}} + with patch("run_agent.OpenAI", return_value=MagicMock()): + result = agent._restore_primary_runtime() + assert result is True + assert agent.request_overrides == overrides diff --git a/tests/run_agent/test_provider_fallback.py b/tests/run_agent/test_provider_fallback.py index eaffe66d32..cf6c4f5475 100644 --- a/tests/run_agent/test_provider_fallback.py +++ b/tests/run_agent/test_provider_fallback.py @@ -414,3 +414,91 @@ class TestFallbackChainDedup: assert called == [("xai", "grok-4.5")] assert agent.provider == "xai" assert agent.model == "grok-4.5" + + +# ── extra_body re-resolution on fallback activation (#75091) ───────────── + + +class TestFallbackExtraBodyReResolution: + """Fallback activation must re-resolve extra_body key-scoped. + + The old provider's custom_providers-contributed extra_body keys are + stale on the new backend and must be dropped; caller-provided + request_overrides keys must survive; the fallback provider's own + extra_body must be merged in (salvage of #75139). + """ + + OLD_URL = "https://old-llm.example.com/v1" + FB_URL = "https://fb-llm.example.com/v1" + + def _agent_with_custom_providers(self, caller_extra_body=None): + agent = _make_agent( + fallback_model={ + "provider": "custom:fbprov", + "model": "fb-model", + "base_url": self.FB_URL, + }, + ) + agent.provider = "custom" + agent.model = "old-model" + agent.base_url = self.OLD_URL + agent._custom_providers = [ + { + "name": "oldprov", + "base_url": self.OLD_URL, + "extra_body": {"enable_thinking": True, "old_only": 1}, + }, + { + "provider_key": "fbprov", + "base_url": self.FB_URL, + "extra_body": {"top_k": 20}, + }, + ] + # Simulate the init-time merge: provider extra_body + caller keys + # (caller wins on conflict — agent_init._merge_custom_provider_extra_body). + merged = {"enable_thinking": True, "old_only": 1} + merged.update(caller_extra_body or {}) + agent.request_overrides = {"extra_body": merged} + return agent + + def _activate(self, agent): + with patch( + "agent.auxiliary_client.resolve_provider_client", + return_value=(_mock_client(base_url=self.FB_URL), "fb-model"), + ), patch( + "agent.model_metadata.get_model_context_length", + return_value=128_000, + ): + assert agent._try_activate_fallback() is True + + def test_stale_provider_keys_removed_and_new_provider_merged(self): + agent = self._agent_with_custom_providers() + self._activate(agent) + eb = agent.request_overrides.get("extra_body") or {} + # Old provider's contributed keys are gone. + assert "enable_thinking" not in eb + assert "old_only" not in eb + # Fallback provider's own extra_body is applied. + assert eb.get("top_k") == 20 + + def test_caller_override_keys_survive_fallback(self): + agent = self._agent_with_custom_providers( + caller_extra_body={"reasoning": {"effort": "high"}, "enable_thinking": False}, + ) + self._activate(agent) + eb = agent.request_overrides.get("extra_body") or {} + # Pure caller key survives untouched. + assert eb.get("reasoning") == {"effort": "high"} + # Caller redefined a key the old provider also set (caller won at + # init: False != True) — the caller's value must survive key-scoped + # removal. + assert eb.get("enable_thinking") is False + # But the key the old provider alone contributed is dropped. + assert "old_only" not in eb + assert eb.get("top_k") == 20 + + def test_non_extra_body_overrides_untouched(self): + agent = self._agent_with_custom_providers() + agent.request_overrides["temperature"] = 0.2 + self._activate(agent) + assert agent.request_overrides.get("temperature") == 0.2 From ac5186ed825671105a074869c0244b73a2244a63 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:39:33 -0700 Subject: [PATCH 022/117] chore(release): map adamfortuna1324@gmail.com to 0xAdamFortuna --- contributors/emails/adamfortuna1324@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/adamfortuna1324@gmail.com diff --git a/contributors/emails/adamfortuna1324@gmail.com b/contributors/emails/adamfortuna1324@gmail.com new file mode 100644 index 0000000000..f4c628862f --- /dev/null +++ b/contributors/emails/adamfortuna1324@gmail.com @@ -0,0 +1 @@ +0xAdamFortuna From d3bfd2e9b19ef2d9f6001be2707922a21a24e0c3 Mon Sep 17 00:00:00 2001 From: fabiantax Date: Thu, 20 Aug 2026 18:50:37 +0200 Subject: [PATCH 023/117] feat(delegation): forward delegation.request_overrides on direct-endpoint branch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The direct base_url branch of _resolve_delegation_credentials returned no request_overrides key, so a direct OpenRouter delegation (provider=custom, base_url=openrouter.ai/api/v1) could not pass routing hints to its children. The named-provider branch already forwards runtime request_overrides; this gives the direct branch the same contract, honouring delegation.request_overrides from config (dict → forwarded, anything else → None). Primary use: extra_body.provider = {"sort": "throughput"} so delegation children route to the fastest OpenRouter provider for their model, per the fab-swarm throughput work (#901). --- .../tools/test_delegate_request_overrides.py | 52 +++++++++++++++++++ tools/delegate_tool.py | 18 +++++++ 2 files changed, 70 insertions(+) create mode 100644 tests/tools/test_delegate_request_overrides.py diff --git a/tests/tools/test_delegate_request_overrides.py b/tests/tools/test_delegate_request_overrides.py new file mode 100644 index 0000000000..3fa77cc61a --- /dev/null +++ b/tests/tools/test_delegate_request_overrides.py @@ -0,0 +1,52 @@ +"""Regression tests for delegation.request_overrides on the direct-endpoint branch. + +The direct base_url branch of _resolve_delegation_credentials (delegation.base_url +set, provider=custom) used to drop delegation.request_overrides on the floor — +the named-provider branch forwards runtime request_overrides (delegate_tool.py +"request_overrides": dict(runtime.get("request_overrides") or {})), but the +direct branch returned no key. That made it impossible to give delegation +children OpenRouter routing hints (extra_body.provider = {"sort": "throughput"}) +when delegating straight to openrouter.ai/api/v1 via base_url+api_key. +""" + +import pytest + +from tools.delegate_tool import _resolve_delegation_credentials + + +def _cfg(**overrides): + cfg = { + "model": "deepseek/deepseek-v4-flash-0731", + "base_url": "https://openrouter.ai/api/v1", + "api_key": "test-key-1234567890", + } + cfg.update(overrides) + return cfg + + +def test_direct_branch_forwards_request_overrides(): + """delegation.request_overrides flows through the direct-endpoint branch.""" + cfg = _cfg( + request_overrides={ + "extra_body": {"provider": {"sort": "throughput"}}, + } + ) + creds = _resolve_delegation_credentials(cfg, parent_agent=None) + assert creds["request_overrides"] == { + "extra_body": {"provider": {"sort": "throughput"}}, + } + + +def test_direct_branch_absent_request_overrides_stays_none(): + """No delegation.request_overrides → None, preserving the old contract.""" + creds = _resolve_delegation_credentials(_cfg(), parent_agent=None) + assert creds["request_overrides"] is None + + +def test_direct_branch_non_dict_request_overrides_stays_none(): + """Garbage in config (string/list) must not crash or forward junk.""" + for bad in ("throughput", ["extra_body"], 42): + creds = _resolve_delegation_credentials( + _cfg(request_overrides=bad), parent_agent=None + ) + assert creds["request_overrides"] is None diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 2c6765098b..1de913bbdf 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -4494,6 +4494,24 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: _is_native_sdk_provider = _provider_lower in _NATIVE_SDK_PROVIDERS if configured_base_url and not _is_native_sdk_provider: + # delegation.request_overrides: same semantics as the named-provider + # branch below (runtime.get("request_overrides")) — a dict merged into + # the child's API kwargs by the transport's profile path. Keys are + # top-level kwargs (e.g. service_tier); an "extra_body" sub-dict is + # merged into extra_body. This is how a direct-endpoint delegation + # (provider=custom) forwards OpenRouter routing hints such as + # extra_body.provider = {"sort": "throughput"} to its children — + # the child's CustomProfile does not emit provider preferences, and + # the parent-inheritance path is deliberately cleared when + # delegation.provider/base_url overrides the parent (see the + # provider-preference clearing in _build_child_agent). + configured_request_overrides = cfg.get("request_overrides") + request_overrides = ( + dict(configured_request_overrides) + if isinstance(configured_request_overrides, dict) + else None + ) + # When delegation.api_key is not set, return None so _build_child_agent # falls back to the parent agent's API key via the credential inheritance # path (effective_api_key = override_api_key or parent_api_key). This From bacb90fe20592a0bf479ddd5d4d3bd2fb95db827 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:38:24 -0700 Subject: [PATCH 024/117] feat(delegation): honor delegation.request_overrides on all three resolution branches with explicit-over-runtime merge precedence MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Completes the #90953 salvage on post-#98237 main: - New _merge_request_overrides helper defines the precedence contract: explicit delegation.request_overrides merges OVER runtime/parent-derived overrides — explicit top-level keys win; extra_body is deep-merged one level so runtime extra_body keys survive unless redefined. Inputs are copy.deepcopy'd so transport-side mutation can't leak into config or the provider runtime cache. - Direct base_url branch: explicit key now merges over the #98237 provider-alongside-base_url runtime overrides instead of being a separate return shape; max_output_tokens preserved. - Named-provider branch and parent-inherit branch now honor the key too, so delegation.request_overrides never silently no-ops. - _build_child_agent honors override_request_overrides whenever set (previously only when override_provider was set), enabling the inherit branch's merged value to reach the child. - DEFAULT_CONFIG: delegation.request_overrides entry with comment. - Tests: expanded tests/tools/test_delegate_request_overrides.py — deep-copy proofs, explicit-over-runtime precedence on the provider-alongside-base_url path, named-provider branch, inherit branch, and merge-helper unit tests. - Docs: configuration.md delegation section + features/delegation.md document the key, precedence, and example YAML (OpenRouter extra_body.provider.sort). --- hermes_cli/config_defaults.py | 9 + .../tools/test_delegate_request_overrides.py | 224 +++++++++++++++++- tools/delegate_tool.py | 121 ++++++++-- website/docs/user-guide/configuration.md | 17 ++ .../docs/user-guide/features/delegation.md | 15 +- 5 files changed, 353 insertions(+), 33 deletions(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 49b11d168f..b5ecdbfa2e 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2077,6 +2077,15 @@ DEFAULT_CONFIG = { # "codex_responses", or "anthropic_messages". Empty = auto-detect # from URL (e.g. /anthropic suffix → anthropic_messages). Set this # explicitly for non-standard endpoints the heuristic can't detect. + # Per-child request settings sent on every delegation API call, on all + # three resolution branches (direct base_url, named provider, and + # parent-inherit). Top-level keys are API kwargs (e.g. service_tier); + # an "extra_body" sub-dict is merged into the request's extra_body — + # e.g. {"extra_body": {"provider": {"sort": "throughput"}}} routes + # OpenRouter delegation children to the fastest provider. Precedence: + # these explicit values merge OVER runtime/parent-derived overrides + # (explicit keys win; extra_body deep-merged one level). + "request_overrides": {}, # When delegate_task narrows child toolsets explicitly, preserve any # MCP toolsets the parent already has enabled. On by default so # narrowing (e.g. toolsets=["web","browser"]) expresses "I want these diff --git a/tests/tools/test_delegate_request_overrides.py b/tests/tools/test_delegate_request_overrides.py index 3fa77cc61a..c66fdfbdfe 100644 --- a/tests/tools/test_delegate_request_overrides.py +++ b/tests/tools/test_delegate_request_overrides.py @@ -1,17 +1,26 @@ -"""Regression tests for delegation.request_overrides on the direct-endpoint branch. +"""Regression tests for the delegation.request_overrides config key. -The direct base_url branch of _resolve_delegation_credentials (delegation.base_url -set, provider=custom) used to drop delegation.request_overrides on the floor — -the named-provider branch forwards runtime request_overrides (delegate_tool.py -"request_overrides": dict(runtime.get("request_overrides") or {})), but the -direct branch returned no key. That made it impossible to give delegation -children OpenRouter routing hints (extra_body.provider = {"sort": "throughput"}) -when delegating straight to openrouter.ai/api/v1 via base_url+api_key. +PR #90953 (salvage): ``delegation.request_overrides`` is an explicit dict of +per-child request settings that must be honored on EVERY resolution branch of +``_resolve_delegation_credentials``: + +1. direct base_url (provider=custom) — the branch the original PR fixed, +2. named provider (delegation.provider set, no base_url), +3. parent-inherit (neither provider nor base_url). + +Precedence contract (post-#98237): explicit config values merge OVER +runtime/parent-derived overrides — top-level explicit keys win; the +``extra_body`` sub-dict is deep-merged one level so runtime extra_body keys +survive unless the explicit key redefines them. Nested values are deep-copied +so transport-side mutation cannot leak back into config. """ -import pytest +from unittest.mock import MagicMock, patch -from tools.delegate_tool import _resolve_delegation_credentials +from tools.delegate_tool import ( + _merge_request_overrides, + _resolve_delegation_credentials, +) def _cfg(**overrides): @@ -24,6 +33,20 @@ def _cfg(**overrides): return cfg +def _parent(**attrs): + parent = MagicMock() + parent._delegate_depth = 0 + # MagicMock attributes are MagicMocks (non-dict) by default; set real + # values for the ones the resolution path inspects. + parent.request_overrides = attrs.pop("request_overrides", None) + for k, v in attrs.items(): + setattr(parent, k, v) + return parent + + +# ── Branch 1: direct base_url ────────────────────────────────────────────── + + def test_direct_branch_forwards_request_overrides(): """delegation.request_overrides flows through the direct-endpoint branch.""" cfg = _cfg( @@ -35,6 +58,8 @@ def test_direct_branch_forwards_request_overrides(): assert creds["request_overrides"] == { "extra_body": {"provider": {"sort": "throughput"}}, } + # Shape parity with the named-provider branch: max_output_tokens present. + assert "max_output_tokens" in creds def test_direct_branch_absent_request_overrides_stays_none(): @@ -50,3 +75,182 @@ def test_direct_branch_non_dict_request_overrides_stays_none(): _cfg(request_overrides=bad), parent_agent=None ) assert creds["request_overrides"] is None + + +def test_direct_branch_deep_copies_nested_extra_body(): + """Transport-side mutation of the child's overrides must not leak back + into the config dict (copy.deepcopy, not a shallow dict()).""" + source = {"extra_body": {"provider": {"sort": "throughput"}}} + cfg = _cfg(request_overrides=source) + creds = _resolve_delegation_credentials(cfg, parent_agent=None) + creds["request_overrides"]["extra_body"]["provider"]["sort"] = "mutated" + creds["request_overrides"]["extra_body"]["injected"] = True + assert source == {"extra_body": {"provider": {"sort": "throughput"}}} + + +@patch("hermes_cli.runtime_provider.resolve_runtime_provider") +def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolve): + """Precedence on the provider-alongside-base_url path (#98237 interplay): + explicit delegation.request_overrides merges OVER the named provider's + runtime overrides — runtime extra_body keys survive unless redefined, + explicit top-level keys win, and max_output_tokens is preserved.""" + mock_resolve.return_value = { + "provider": "custom", + "base_url": "https://provider-default.example/v1", + "api_key": "provider-key", + "api_mode": "chat_completions", + "request_overrides": { + "service_tier": "default", + "extra_body": {"thinking": {"type": "disabled"}, "provider": {"sort": "price"}}, + }, + "max_output_tokens": 8192, + } + cfg = _cfg( + provider="mimo", + request_overrides={ + "service_tier": "flex", + "extra_body": {"provider": {"sort": "throughput"}}, + }, + ) + creds = _resolve_delegation_credentials(cfg, parent_agent=None) + assert creds["request_overrides"] == { + # explicit top-level key wins + "service_tier": "flex", + "extra_body": { + # runtime extra_body key survives (not redefined) + "thinking": {"type": "disabled"}, + # explicit extra_body key wins over runtime's + "provider": {"sort": "throughput"}, + }, + } + assert creds["max_output_tokens"] == 8192 + + +# ── Branch 2: named provider (no base_url) ───────────────────────────────── + + +@patch("hermes_cli.runtime_provider.resolve_runtime_provider") +def test_named_provider_branch_honors_explicit_key(mock_resolve): + """The named-provider branch merges the explicit key over the provider's + runtime overrides — the config key never silently no-ops.""" + mock_resolve.return_value = { + "provider": "openrouter", + "base_url": "https://openrouter.ai/api/v1", + "api_key": "runtime-key", + "api_mode": "chat_completions", + "request_overrides": {"extra_body": {"reasoning": {"enabled": True}}}, + "max_output_tokens": 4096, + } + cfg = { + "model": "deepseek/deepseek-v4-flash-0731", + "provider": "openrouter", + "request_overrides": {"extra_body": {"provider": {"sort": "throughput"}}}, + } + creds = _resolve_delegation_credentials(cfg, parent_agent=None) + assert creds["request_overrides"] == { + "extra_body": { + "reasoning": {"enabled": True}, + "provider": {"sort": "throughput"}, + } + } + + +@patch("hermes_cli.runtime_provider.resolve_runtime_provider") +def test_named_provider_branch_without_explicit_key_unchanged(mock_resolve): + """Without the config key the named-provider branch behaves as before.""" + mock_resolve.return_value = { + "provider": "openrouter", + "base_url": "https://openrouter.ai/api/v1", + "api_key": "runtime-key", + "api_mode": "chat_completions", + "request_overrides": {"extra_body": {"reasoning": {"enabled": True}}}, + "max_output_tokens": 4096, + } + cfg = {"model": "m", "provider": "openrouter"} + creds = _resolve_delegation_credentials(cfg, parent_agent=None) + assert creds["request_overrides"] == {"extra_body": {"reasoning": {"enabled": True}}} + + +# ── Branch 3: parent-inherit (no provider, no base_url) ──────────────────── + + +def test_inherit_branch_honors_explicit_key_over_parent(): + """Pure-inherit setups still apply delegation.request_overrides, merged + over the parent agent's own request_overrides.""" + parent = _parent( + request_overrides={ + "service_tier": "default", + "extra_body": {"thinking": {"type": "disabled"}}, + } + ) + cfg = { + "model": "", + "provider": "", + "request_overrides": {"extra_body": {"provider": {"sort": "throughput"}}}, + } + creds = _resolve_delegation_credentials(cfg, parent) + assert creds["request_overrides"] == { + "service_tier": "default", + "extra_body": { + "thinking": {"type": "disabled"}, + "provider": {"sort": "throughput"}, + }, + } + + +def test_inherit_branch_without_key_stays_none(): + """No explicit key and no parent overrides → None (old contract: the + child's construction path falls back to the parent's request_overrides).""" + parent = _parent(request_overrides=None) + creds = _resolve_delegation_credentials({"model": "", "provider": ""}, parent) + assert creds["request_overrides"] is None + + +def test_inherit_branch_deep_copies_parent_overrides(): + """Parent's nested overrides must be deep-copied on the inherit branch.""" + parent_overrides = {"extra_body": {"thinking": {"type": "disabled"}}} + parent = _parent(request_overrides=parent_overrides) + cfg = { + "model": "", + "provider": "", + "request_overrides": {"extra_body": {"provider": {"sort": "throughput"}}}, + } + creds = _resolve_delegation_credentials(cfg, parent) + creds["request_overrides"]["extra_body"]["thinking"]["type"] = "mutated" + assert parent_overrides == {"extra_body": {"thinking": {"type": "disabled"}}} + + +# ── Merge helper unit tests ──────────────────────────────────────────────── + + +def test_merge_helper_both_none(): + assert _merge_request_overrides(None, None) is None + assert _merge_request_overrides({}, {}) is None + assert _merge_request_overrides("junk", 42) is None + + +def test_merge_helper_explicit_only(): + assert _merge_request_overrides(None, {"a": 1}) == {"a": 1} + + +def test_merge_helper_runtime_only(): + assert _merge_request_overrides({"a": 1}, None) == {"a": 1} + + +def test_merge_helper_explicit_top_level_wins(): + assert _merge_request_overrides({"a": 1, "b": 2}, {"a": 9}) == {"a": 9, "b": 2} + + +def test_merge_helper_extra_body_one_level_merge(): + merged = _merge_request_overrides( + {"extra_body": {"keep": 1, "clash": "runtime"}}, + {"extra_body": {"clash": "explicit", "new": 2}}, + ) + assert merged == {"extra_body": {"keep": 1, "clash": "explicit", "new": 2}} + + +def test_merge_helper_non_dict_runtime_extra_body_replaced(): + merged = _merge_request_overrides( + {"extra_body": "junk"}, {"extra_body": {"a": 1}} + ) + assert merged == {"extra_body": {"a": 1}} diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 1de913bbdf..f38e35b124 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -1994,9 +1994,17 @@ def _build_child_agent( provider_require_parameters=child_provider_require_parameters, provider_data_collection=child_provider_data_collection, request_overrides=( - dict(override_request_overrides or {}) - if override_provider - else dict(getattr(parent_agent, "request_overrides", {}) or {}) + # override_request_overrides is honored whenever set — + # including the inherit branch (override_provider=None), + # where _resolve_delegation_credentials already merged + # delegation.request_overrides OVER the parent's values. + dict(override_request_overrides) + if override_request_overrides is not None + else ( + {} + if override_provider + else dict(getattr(parent_agent, "request_overrides", {}) or {}) + ) ), openrouter_min_coding_score=child_openrouter_min_coding_score, tool_progress_callback=child_progress_cb, @@ -4456,6 +4464,43 @@ def _resolve_child_credential_pool( return None +def _merge_request_overrides(runtime_overrides, explicit_overrides): + """Merge explicit ``delegation.request_overrides`` over runtime-derived ones. + + Precedence contract: the explicit config key WINS over runtime-derived + (provider-catalog or parent-inherited) overrides. Top-level keys from the + explicit dict replace same-named runtime keys; the ``extra_body`` sub-dict + is deep-merged ONE level — runtime ``extra_body`` keys survive unless the + explicit dict redefines that exact key. This keeps provider personality + (e.g. ``thinking: {type: disabled}``) intact while letting users layer + routing hints (e.g. ``extra_body.provider = {"sort": "throughput"}``) on + top. + + Both inputs are deep-copied (``copy.deepcopy``) so transport-side mutation + of the child's request kwargs can never leak back into the loaded config + dict or the provider runtime cache. + + Returns ``None`` when both sides are empty/non-dict. + """ + import copy as _copy + + runtime_overrides = runtime_overrides if isinstance(runtime_overrides, dict) else None + explicit_overrides = explicit_overrides if isinstance(explicit_overrides, dict) else None + if not runtime_overrides and not explicit_overrides: + return None + merged = _copy.deepcopy(runtime_overrides) if runtime_overrides else {} + explicit = _copy.deepcopy(explicit_overrides) if explicit_overrides else {} + runtime_extra = merged.get("extra_body") + explicit_extra = explicit.pop("extra_body", None) + merged.update(explicit) + if isinstance(runtime_extra, dict) and isinstance(explicit_extra, dict): + runtime_extra.update(explicit_extra) + merged["extra_body"] = runtime_extra + elif explicit_extra is not None: + merged["extra_body"] = explicit_extra + return merged or None + + def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: """Resolve credentials for subagent delegation. @@ -4483,6 +4528,18 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: configured_api_key = str(cfg.get("api_key") or "").strip() or None configured_api_mode = str(cfg.get("api_mode") or "").strip().lower() or None + # delegation.request_overrides: explicit per-child request settings from + # config. Honored on EVERY resolution branch (direct base_url, named + # provider, and parent-inherit) so the key never silently no-ops. + # Precedence: explicit merges OVER runtime/parent-derived overrides via + # _merge_request_overrides (top-level explicit keys win; extra_body is + # deep-merged one level). Non-dict values are ignored. + explicit_request_overrides = ( + cfg.get("request_overrides") + if isinstance(cfg.get("request_overrides"), dict) + else None + ) + # Native-SDK providers (Bedrock, Vertex, Google GenAI) speak their own # wire protocol — they cannot be reached via OpenAI chat_completions against # a base_url. For these, always fall through to resolve_runtime_provider() @@ -4494,23 +4551,23 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: _is_native_sdk_provider = _provider_lower in _NATIVE_SDK_PROVIDERS if configured_base_url and not _is_native_sdk_provider: - # delegation.request_overrides: same semantics as the named-provider - # branch below (runtime.get("request_overrides")) — a dict merged into - # the child's API kwargs by the transport's profile path. Keys are - # top-level kwargs (e.g. service_tier); an "extra_body" sub-dict is - # merged into extra_body. This is how a direct-endpoint delegation - # (provider=custom) forwards OpenRouter routing hints such as - # extra_body.provider = {"sort": "throughput"} to its children — - # the child's CustomProfile does not emit provider preferences, and - # the parent-inheritance path is deliberately cleared when - # delegation.provider/base_url overrides the parent (see the + # delegation.request_overrides: an explicit dict of per-child request + # settings merged into the child's API kwargs by the transport's + # profile path. Keys are top-level kwargs (e.g. service_tier); an + # "extra_body" sub-dict is merged into extra_body. This is how a + # direct-endpoint delegation (provider=custom) forwards OpenRouter + # routing hints such as extra_body.provider = {"sort": "throughput"} + # to its children — the child's CustomProfile does not emit provider + # preferences, and the parent-inheritance path is deliberately cleared + # when delegation.provider/base_url overrides the parent (see the # provider-preference clearing in _build_child_agent). - configured_request_overrides = cfg.get("request_overrides") - request_overrides = ( - dict(configured_request_overrides) - if isinstance(configured_request_overrides, dict) - else None - ) + # + # Precedence: explicit delegation.request_overrides MERGES OVER any + # runtime-derived overrides (see _merge_request_overrides) — top-level + # explicit keys win; extra_body is deep-merged one level so runtime + # extra_body keys survive unless the explicit key redefines them. + # (explicit_request_overrides is parsed once at the top of this + # function and applied to every branch.) # When delegation.api_key is not set, return None so _build_child_agent # falls back to the parent agent's API key via the credential inheritance @@ -4577,6 +4634,12 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: exc, ) + # Explicit delegation.request_overrides merges OVER the runtime-derived + # overrides (explicit wins; extra_body deep-merged one level). + request_overrides = _merge_request_overrides( + request_overrides, explicit_request_overrides + ) + return { "model": configured_model, "provider": provider, @@ -4588,14 +4651,22 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: } if not configured_provider: - # No provider override — child inherits everything from parent + # No provider override — child inherits everything from parent. + # delegation.request_overrides still applies: merge the explicit key + # OVER the parent's own request_overrides so the config key works even + # in pure-inherit setups (never a silent no-op). None when neither + # side has values → _build_child_agent falls back to the parent's + # request_overrides unchanged. return { "model": configured_model, "provider": None, "base_url": None, "api_key": None, "api_mode": None, - "request_overrides": None, + "request_overrides": _merge_request_overrides( + getattr(parent_agent, "request_overrides", None), + explicit_request_overrides, + ), "max_output_tokens": None, } @@ -4639,7 +4710,13 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: "base_url": runtime.get("base_url"), "api_key": api_key, "api_mode": runtime.get("api_mode"), - "request_overrides": dict(runtime.get("request_overrides") or {}), + # Explicit delegation.request_overrides merges OVER the named + # provider's runtime overrides (explicit wins; extra_body deep-merged + # one level) — same precedence as the direct-base_url branch above. + "request_overrides": _merge_request_overrides( + runtime.get("request_overrides"), explicit_request_overrides + ) + or {}, "max_output_tokens": runtime.get("max_output_tokens"), "command": runtime.get("command"), "args": list(runtime.get("args") or []), diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 539bd0bf8d..8fe27bd397 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2568,6 +2568,10 @@ delegation: # base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider) # api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY) # api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect. + # request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches). + # extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints: + # provider: + # sort: throughput max_concurrent_children: 3 # Parallel children per batch (floor 1, no ceiling). Also via DELEGATION_MAX_CONCURRENT_CHILDREN env var. worktree_isolation: false # Give each child its own git worktree branched from HEAD (local backend + git repos only; inspired by Muse Code). See Subagent Delegation → Worktree Isolation. max_spawn_depth: 1 # Delegation tree depth cap (1-3, clamped). 1 = flat (default): parent spawns leaves that cannot delegate. 2 = orchestrator children can spawn leaf grandchildren. 3 = three levels. @@ -2578,6 +2582,19 @@ delegation: **Direct endpoint override:** If you want the obvious custom-endpoint path, set `delegation.base_url`, `delegation.api_key`, and `delegation.model`. That sends subagents directly to that OpenAI-compatible endpoint and takes precedence over `delegation.provider`. If `delegation.api_key` is omitted, Hermes falls back to `OPENAI_API_KEY` only. When `delegation.provider` is set alongside `delegation.base_url`, the explicit endpoint and key still win, but that provider's request settings (`extra_body` overrides and max output tokens from your `custom_providers` entry) are carried into the subagent. +**Per-child request settings (`request_overrides`):** `delegation.request_overrides` is a dict of request settings sent on every subagent API call. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. It is honored on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so the key always takes effect. Precedence: explicit `request_overrides` values merge **over** any runtime- or parent-derived overrides — top-level explicit keys win, and `extra_body` is deep-merged one level so runtime `extra_body` keys (e.g. a provider's `thinking: {type: disabled}` personality) survive unless your key redefines them. The canonical use case is OpenRouter routing hints for delegation children: + +```yaml +delegation: + model: "deepseek/deepseek-v4-flash-0731" + base_url: "https://openrouter.ai/api/v1" + api_key: "sk-or-..." + request_overrides: + extra_body: + provider: + sort: throughput # route children to the fastest OpenRouter provider +``` + **Wire protocol (`api_mode`):** Hermes auto-detects the wire protocol from `delegation.base_url` (e.g. paths ending in `/anthropic` → `anthropic_messages`; Codex / native Anthropic / Kimi-coding hostnames keep their existing detection). For endpoints the heuristic can't classify — for example Azure AI Foundry, MiniMax, Zhipu GLM, or LiteLLM proxies fronting an Anthropic-shaped backend — set `delegation.api_mode` explicitly to one of `chat_completions`, `codex_responses`, or `anthropic_messages`. Leave it empty (the default) to keep auto-detection. The delegation provider uses the same credential resolution as CLI/gateway startup. All configured providers are supported: `openrouter`, `nous`, `copilot`, `zai`, `kimi-coding`, `minimax`, `minimax-cn`. When a provider is set, the system automatically resolves the correct base URL, API key, and API mode — no manual credential wiring needed. diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index 1266eeb69b..52db42d895 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -188,7 +188,7 @@ delegation: provider: "openrouter" # optional: route children to a different provider ``` -Resolution order: `delegation.base_url` (direct endpoint) takes precedence, then `delegation.provider` (full credential bundle resolved via the runtime provider system), and when neither is set children inherit the parent's provider and credentials; `delegation.model` applies in all cases, and when it is empty children inherit the parent's model. Setting `delegation.provider` alongside `delegation.base_url` keeps the explicit endpoint but carries that provider's request overrides and max output tokens into the child. +Resolution order: `delegation.base_url` (direct endpoint) takes precedence, then `delegation.provider` (full credential bundle resolved via the runtime provider system), and when neither is set children inherit the parent's provider and credentials; `delegation.model` applies in all cases, and when it is empty children inherit the parent's model. Setting `delegation.provider` alongside `delegation.base_url` keeps the explicit endpoint but carries that provider's request overrides and max output tokens into the child. An explicit `delegation.request_overrides` dict is honored on every branch and merges over those runtime-derived values (see [Configuration](#configuration) below). Note that the pin is global: `delegate_task` has no per-task model parameter, so every child in a batch runs on the configured delegation model. For quality-sensitive subtasks that need a stronger model, either leave `delegation.model` unset for that session or hand the task to the [kanban board](kanban.md#per-task-model-override), which does support a per-task model override. @@ -512,10 +512,23 @@ delegation: base_url: "http://localhost:1234/v1" api_key: "local-key" # api_mode: "anthropic_messages" # Optional. Wire protocol override for base_url ("chat_completions", "codex_responses", or "anthropic_messages"). Empty = auto-detect from URL (e.g. /anthropic suffix). Set explicitly for endpoints the heuristic can't classify (Azure AI Foundry, MiniMax, Zhipu GLM, LiteLLM proxies, …). + +# Send per-child request settings on every subagent API call — e.g. OpenRouter +# routing hints when delegating straight to openrouter.ai via base_url: +delegation: + model: "deepseek/deepseek-v4-flash-0731" + base_url: "https://openrouter.ai/api/v1" + api_key: "sk-or-..." + request_overrides: + extra_body: + provider: + sort: throughput # children route to the fastest OpenRouter provider ``` When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). +`delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. + :::tip The agent handles delegation automatically based on the task complexity. You don't need to explicitly ask it to delegate — it will do so when it makes sense. ::: From 9107b891c9d76cef33c2a2d776860e025d4b204a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:38:34 -0700 Subject: [PATCH 025/117] chore(attribution): map fabiantax@hotmail.com -> fabiantax (PR #90953 salvage) --- contributors/emails/fabiantax@hotmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/fabiantax@hotmail.com diff --git a/contributors/emails/fabiantax@hotmail.com b/contributors/emails/fabiantax@hotmail.com new file mode 100644 index 0000000000..64934a826d --- /dev/null +++ b/contributors/emails/fabiantax@hotmail.com @@ -0,0 +1 @@ +fabiantax From d6ace971d476e1bc1f9b478c6b600b145cf691cb Mon Sep 17 00:00:00 2001 From: nftpoetrist <264138787+nftpoetrist@users.noreply.github.com> Date: Thu, 27 Aug 2026 10:04:02 +0300 Subject: [PATCH 026/117] docs(computer-use): remove the deleted typed browser-page route MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #95620 removed computer_use's cua_browser_* actions entirely (browser_route.py, the action enum, and their dispatch/escalation hint) — computer_use is desktop-only now, and page content goes through the separate browser_navigate/browser_click/... toolset (or browser_exec under the Browser Use CLI backend). The skill doc never caught up: it still taught the model to call cua_browser_state/cua_browser_prepare/etc. and to escalate to a "page" rung that _enrich_escalation can no longer recommend. Following that guidance fails schema validation on the first call. - SKILL.md: replace the "Typed browser page rung" section (including the existing_profile authorization walkthrough, which described a model-facing action parameter that no longer exists anywhere in schema.py/tool.py) with a short pointer to the current browser toolset; drop "page" from the escalation.recommended union and the ladder step that referenced it. - website/docs/user-guide/features/computer-use.md: grant_existing_profile and the rest of the permission-mode config are still real and current (cua_backend.py still reads them for the runtime launch grant) — only reworded the one sentence naming cua_browser_prepare as the mechanism that consumes the grant, since driving a signed-in browser window now goes through the standard capture/click/type actions instead. - Regenerated the mirrored skill doc via website/scripts/generate-skill-docs.py rather than hand-editing it, per its own header. Kept the diff scoped to computer-use only. --- .../computer-use/SKILL.md | 75 +++------------- .../docs/user-guide/features/computer-use.md | 10 +-- .../autonomous-ai-agents-computer-use.md | 90 ++++++------------- 3 files changed, 44 insertions(+), 131 deletions(-) diff --git a/skills/autonomous-ai-agents/computer-use/SKILL.md b/skills/autonomous-ai-agents/computer-use/SKILL.md index f098ce8c20..9a8e7e98d4 100644 --- a/skills/autonomous-ai-agents/computer-use/SKILL.md +++ b/skills/autonomous-ai-agents/computer-use/SKILL.md @@ -115,7 +115,7 @@ Returned fields (present when the driver supports them): - `effect`: `"confirmed"` (driver read the result back — done), `"unverifiable"` (delivered, but confirm it yourself by re-capturing), or `"suspected_noop"` (ran but almost certainly did nothing). -- `escalation`: `{recommended: "px" | "foreground" | "page", reason}` — present +- `escalation`: `{recommended: "px" | "foreground", reason}` — present only when there's a next rung to try. - `code`: a structured refusal like `"background_unavailable"` or `"foreground_unsupported"`. @@ -131,10 +131,7 @@ Walk it in order: 3. **Pixel, background.** After `effect:"suspected_noop"` or a structured refusal recommends `"px"` (or a `degraded` capture has no elements), click by `coordinate=[x,y]` instead of `element`. -4. **Typed page.** When `escalation.recommended == "page"` and the exact - browser-page contract below is available, use the namespaced typed route - before native foreground. This is not the legacy `page` workflow. -5. **Foreground.** After `effect:"suspected_noop"`, +4. **Foreground.** After `effect:"suspected_noop"`, `code:"background_unavailable"`, or a verified pixel no-op, re-issue the SAME action with `delivery_mode="foreground"`. This briefly raises the window and restores focus after; pair with `bring_to_front=True` @@ -142,7 +139,7 @@ Walk it in order: (it's a visible focus change) and is only appropriate when the user isn't actively working. Classic cases: Electron/Chromium consent dialogs (e.g. tldraw offline's "Run Script"), DirectInput games, raw-input canvases. -6. **Keystrokes verified-lost on a KDE/Qt editor → use the app's own I/O.** +5. **Keystrokes verified-lost on a KDE/Qt editor → use the app's own I/O.** Some Qt text components (KTextEditor: Kate, KWrite, KDevelop) discard SYNTHETIC X keystrokes entirely — foreground `type` reports ok ("Typed N characters into the focused widget", `effect:"unverifiable"`) @@ -170,62 +167,16 @@ NOT conclude "cua-driver can't drive this app" — climb the ladder. If action schema lacks that property; choose another verified rung without inferring support from the executable's reported version. -## Typed browser page rung +## Page content is a separate toolset -For page content in a supported GUI browser, the same `computer_use` tool -exposes namespaced `cua_browser_*` actions. They do not collide with other -browser tools. The contract is capability-based: - -1. Discover the exact native browser `(pid, window_id)` with `list_windows` or - native capture, then call `cua_browser_state` with both values. -2. Continue only when it returns `status:"ok"`, `binding_quality:"exact"`, and - `mutation_allowed:true`. Select an opaque `tab_id` from that response. -3. Call `cua_browser_state` with the `tab_id` for a fresh `semantic_v2` - snapshot. Use only refs from that newest snapshot and only for their - declared actions. -4. Use the matching namespaced action (`cua_browser_click`, - `cua_browser_type`, `cua_browser_navigate`, or `cua_browser_pointer`). - Trusted input is the default. `input_route="dom_event"` is an explicit - trust downgrade; never choose it silently after a refusal. -5. Every mutation invalidates refs. Take a fresh state snapshot before another - typed action. Never chain actions from remembered refs. - -`cua_browser_prepare` is a separate approved setup action. Driver-owned -`isolated_new`/`isolated_named` profiles require explicit `allow_launch=true`. -An `existing_profile` is decided by cua-driver's immutable permission mode. -Prefer `isolated_new` unless the task genuinely needs the user's signed-in -session — attaching to an existing profile exposes its live pages, cookies, -and storage over the browser protocol. - -Authorization paths for `existing_profile`: - -1. **Config grant (standard and unrestricted modes).** When - `computer_use.grant_existing_profile: true` is set, the runtime is - launched pre-authorized in standard mode (`--grant existing-profile`) and - Hermes applies the same host-side floor in unrestricted mode. If it is not - set, both modes fail closed. Tell the user to flip that config key and - restart the session if they want this; do not retry or work around it. -2. **Bounded manifest.** When `computer_use.permission_mode: bounded` is - configured with a reviewed `capability_manifest`, prepares inside the - manifest's scope succeed without prompts and everything else fails closed. - -Explicit Hermes YOLO (`--yolo`, `/yolo`, or `approvals.mode: off`) launches an -unrestricted runtime with no runtime Cua approval prompts, but it does not -substitute for `grant_existing_profile: true`. - -These settings belong to runtime launch. The agent cannot add or change them -after the runtime starts. Without the applicable grant or bounded manifest, -`existing_profile` fails closed. Report the refusal and name the config key; -do not retry, downgrade trust, or work around it. - -Every MCP transport owns a private lifecycle session inside the runtime. The -public session name only labels cursor identity and session-scoped state. It -does not select, share, or keep a runtime alive. - -Use the native capture/AX/pixel/foreground ladder for browser chrome, browser -permission UI, OS prompts, native dialogs, extension surfaces, unsupported -engines, and any typed route that cannot prove exact binding or mutation -permission. `cua_browser_dialog` covers page JavaScript dialogs only. +`computer_use` is desktop-only: it does not expose a typed route for browser +page content (no `cua_browser_*` actions). For reading or acting on a page's +DOM — navigation, clicking a link by text, typed input into a form field — +use the separate `browser_navigate`/`browser_click`/`browser_type`/`browser_snapshot` +tools (or `browser_exec` when the Browser Use CLI backend is active); their +own schemas document the current contract. Reserve `computer_use` for browser +*chrome* (the address bar, permission prompts, extension popups, native +dialogs) and anything else on screen that isn't page content. ### Key shortcuts vary per platform @@ -331,7 +282,7 @@ in your conversation context. | `cua-driver not installed` | Run `hermes computer-use install`, or `hermes tools` and enable Computer Use | | Captures consistently return empty / "no on-screen window" | On Linux: DISPLAY may not be set (X11) or you're on pure Wayland — ask the user to run `hermes computer-use doctor`. On Windows: you may be in Session 0 (SSH session) instead of the interactive desktop — see the cua-driver `WINDOWS.md` deep-dive | | Element index stale ("Element N not in cache") | SOM indices are only valid until the next `capture`. Re-capture before clicking. The wrapper carries opaque `element_token`s for stale-detection; you'll see an explicit error rather than a wrong click | -| Click had no effect | Read the structured verdict. `effect:"unverifiable"` → fresh capture/state before retry, even with an escalation hint. `effect:"suspected_noop"` or a structured refusal → climb the recommended ladder: coordinate (px), typed page route when exact, then foreground. Browser chrome/native prompts remain native. Don't conclude the app is undrivable | +| Click had no effect | Read the structured verdict. `effect:"unverifiable"` → fresh capture/state before retry, even with an escalation hint. `effect:"suspected_noop"` or a structured refusal → climb the recommended ladder: coordinate (px), then foreground. Browser chrome/native prompts remain native; page content is a separate toolset. Don't conclude the app is undrivable | | Type text disappears into a terminal emulator | cua-driver detects terminals (Ghostty, iTerm2, Terminal.app, Windows Terminal, mintty, etc.) and routes through key-event synthesis — should "just work" on a recent cua-driver. If it doesn't, ask the user to run `hermes computer-use doctor` | | `blocked pattern in type text` | You tried to `type` a shell command matching the dangerous-pattern block list (`curl ... \| bash`, `sudo rm -rf`, etc.). Break the command up or reconsider | | Anything else weird | **First action: ask the user to run `hermes computer-use doctor`.** It runs the cua-driver `health_report` MCP tool and prints a structured per-check matrix. Their output tells you (and them) exactly what's wrong | diff --git a/website/docs/user-guide/features/computer-use.md b/website/docs/user-guide/features/computer-use.md index d43f0584f7..12a00200dd 100644 --- a/website/docs/user-guide/features/computer-use.md +++ b/website/docs/user-guide/features/computer-use.md @@ -122,11 +122,11 @@ computer_use: ``` Hermes then launches the cua-driver runtime with the trusted-launcher grant -(`--grant existing-profile`), and -`cua_browser_prepare` with an existing profile succeeds against the exact -`(pid, window_id)` the agent proves. Leave it `false` (the default) and -existing-profile attachment fails closed; driver-owned isolated profiles work -either way and are what the agent prefers. +(`--grant existing-profile`), and the standard capture/click/type actions +succeed against the exact `(pid, window_id)` of that signed-in window. Leave +it `false` (the default) and existing-profile attachment fails closed; +driver-owned isolated profiles work either way and are what the agent +prefers. ### Bounded mode for repeatable automation diff --git a/website/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-computer-use.md b/website/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-computer-use.md index 40cfe3ff6f..0e5f22a117 100644 --- a/website/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-computer-use.md +++ b/website/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-computer-use.md @@ -1,14 +1,14 @@ --- -title: "Computer Use — Drive the desktop in the background without stealing focus" +title: "Computer Use — Drive the desktop background-first; escalate on signal" sidebar_label: "Computer Use" -description: "Drive the desktop in the background without stealing focus" +description: "Drive the desktop background-first; escalate on signal" --- {/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */} # Computer Use -Drive the desktop in the background without stealing focus. +Drive the desktop background-first; escalate on signal. ## Skill metadata @@ -131,7 +131,7 @@ Returned fields (present when the driver supports them): - `effect`: `"confirmed"` (driver read the result back — done), `"unverifiable"` (delivered, but confirm it yourself by re-capturing), or `"suspected_noop"` (ran but almost certainly did nothing). -- `escalation`: `{recommended: "px" | "foreground" | "page", reason}` — present +- `escalation`: `{recommended: "px" | "foreground", reason}` — present only when there's a next rung to try. - `code`: a structured refusal like `"background_unavailable"` or `"foreground_unsupported"`. @@ -147,10 +147,7 @@ Walk it in order: 3. **Pixel, background.** After `effect:"suspected_noop"` or a structured refusal recommends `"px"` (or a `degraded` capture has no elements), click by `coordinate=[x,y]` instead of `element`. -4. **Typed page.** When `escalation.recommended == "page"` and the exact - browser-page contract below is available, use the namespaced typed route - before native foreground. This is not the legacy `page` workflow. -5. **Foreground.** After `effect:"suspected_noop"`, +4. **Foreground.** After `effect:"suspected_noop"`, `code:"background_unavailable"`, or a verified pixel no-op, re-issue the SAME action with `delivery_mode="foreground"`. This briefly raises the window and restores focus after; pair with `bring_to_front=True` @@ -158,6 +155,17 @@ Walk it in order: (it's a visible focus change) and is only appropriate when the user isn't actively working. Classic cases: Electron/Chromium consent dialogs (e.g. tldraw offline's "Run Script"), DirectInput games, raw-input canvases. +5. **Keystrokes verified-lost on a KDE/Qt editor → use the app's own I/O.** + Some Qt text components (KTextEditor: Kate, KWrite, KDevelop) discard + SYNTHETIC X keystrokes entirely — foreground `type` reports ok + ("Typed N characters into the focused widget", `effect:"unverifiable"`) + but a fresh AX capture shows the text never arrived, and raw XTest fails + identically (proven live, Aug 2026 — it is the toolkit, not the driver; + the same foreground route works on kcalc/Chrome). After ONE such + verified-lost round trip, stop retrying input rungs: write the file with + terminal/file tools and let the editor reload it, or drive the app's + DBus/CLI interface. Never loop the ladder against a surface that + verifiably swallows synthetic input. ``` computer_use(action="click", element=7) @@ -175,62 +183,16 @@ NOT conclude "cua-driver can't drive this app" — climb the ladder. If action schema lacks that property; choose another verified rung without inferring support from the executable's reported version. -## Typed browser page rung +## Page content is a separate toolset -For page content in a supported GUI browser, the same `computer_use` tool -exposes namespaced `cua_browser_*` actions. They do not collide with other -browser tools. The contract is capability-based: - -1. Discover the exact native browser `(pid, window_id)` with `list_windows` or - native capture, then call `cua_browser_state` with both values. -2. Continue only when it returns `status:"ok"`, `binding_quality:"exact"`, and - `mutation_allowed:true`. Select an opaque `tab_id` from that response. -3. Call `cua_browser_state` with the `tab_id` for a fresh `semantic_v2` - snapshot. Use only refs from that newest snapshot and only for their - declared actions. -4. Use the matching namespaced action (`cua_browser_click`, - `cua_browser_type`, `cua_browser_navigate`, or `cua_browser_pointer`). - Trusted input is the default. `input_route="dom_event"` is an explicit - trust downgrade; never choose it silently after a refusal. -5. Every mutation invalidates refs. Take a fresh state snapshot before another - typed action. Never chain actions from remembered refs. - -`cua_browser_prepare` is a separate approved setup action. Driver-owned -`isolated_new`/`isolated_named` profiles require explicit `allow_launch=true`. -An `existing_profile` is decided by cua-driver's immutable permission mode. -Prefer `isolated_new` unless the task genuinely needs the user's signed-in -session. Attaching to an existing profile exposes its live pages, cookies, -and storage over the browser protocol. - -Authorization paths for `existing_profile`: - -1. **Config grant (standard and unrestricted modes).** When - `computer_use.grant_existing_profile: true` is set, the runtime is - launched pre-authorized in standard mode (`--grant existing-profile`) and - Hermes applies the same host-side floor in unrestricted mode. If it is not - set, both modes fail closed. Tell the user to set that config key and - restart the session if they want this. Do not retry or work around it. -2. **Bounded manifest.** When `computer_use.permission_mode: bounded` is - configured with a reviewed `capability_manifest`, prepares inside the - manifest's scope succeed without prompts and everything else fails closed. - -Explicit Hermes YOLO (`--yolo`, `/yolo`, or `approvals.mode: off`) launches an -unrestricted runtime with no runtime Cua approval prompts, but it does not -substitute for `grant_existing_profile: true`. - -These settings belong to runtime launch. The agent cannot add or change them -after the runtime starts. Without the applicable grant or bounded manifest, -`existing_profile` fails closed. Report the refusal and name the config key; -do not retry, downgrade trust, or work around it. - -Every MCP transport owns a private lifecycle session inside the runtime. The -public session name only labels cursor identity and session-scoped state. It -does not select, share, or keep a runtime alive. - -Use the native capture/AX/pixel/foreground ladder for browser chrome, browser -permission UI, OS prompts, native dialogs, extension surfaces, unsupported -engines, and any typed route that cannot prove exact binding or mutation -permission. `cua_browser_dialog` covers page JavaScript dialogs only. +`computer_use` is desktop-only: it does not expose a typed route for browser +page content (no `cua_browser_*` actions). For reading or acting on a page's +DOM — navigation, clicking a link by text, typed input into a form field — +use the separate `browser_navigate`/`browser_click`/`browser_type`/`browser_snapshot` +tools (or `browser_exec` when the Browser Use CLI backend is active); their +own schemas document the current contract. Reserve `computer_use` for browser +*chrome* (the address bar, permission prompts, extension popups, native +dialogs) and anything else on screen that isn't page content. ### Key shortcuts vary per platform @@ -336,7 +298,7 @@ in your conversation context. | `cua-driver not installed` | Run `hermes computer-use install`, or `hermes tools` and enable Computer Use | | Captures consistently return empty / "no on-screen window" | On Linux: DISPLAY may not be set (X11) or you're on pure Wayland — ask the user to run `hermes computer-use doctor`. On Windows: you may be in Session 0 (SSH session) instead of the interactive desktop — see the cua-driver `WINDOWS.md` deep-dive | | Element index stale ("Element N not in cache") | SOM indices are only valid until the next `capture`. Re-capture before clicking. The wrapper carries opaque `element_token`s for stale-detection; you'll see an explicit error rather than a wrong click | -| Click had no effect | Read the structured verdict. `effect:"unverifiable"` → fresh capture/state before retry, even with an escalation hint. `effect:"suspected_noop"` or a structured refusal → climb the recommended ladder: coordinate (px), typed page route when exact, then foreground. Browser chrome/native prompts remain native. Don't conclude the app is undrivable | +| Click had no effect | Read the structured verdict. `effect:"unverifiable"` → fresh capture/state before retry, even with an escalation hint. `effect:"suspected_noop"` or a structured refusal → climb the recommended ladder: coordinate (px), then foreground. Browser chrome/native prompts remain native; page content is a separate toolset. Don't conclude the app is undrivable | | Type text disappears into a terminal emulator | cua-driver detects terminals (Ghostty, iTerm2, Terminal.app, Windows Terminal, mintty, etc.) and routes through key-event synthesis — should "just work" on a recent cua-driver. If it doesn't, ask the user to run `hermes computer-use doctor` | | `blocked pattern in type text` | You tried to `type` a shell command matching the dangerous-pattern block list (`curl ... \| bash`, `sudo rm -rf`, etc.). Break the command up or reconsider | | Anything else weird | **First action: ask the user to run `hermes computer-use doctor`.** It runs the cua-driver `health_report` MCP tool and prints a structured per-check matrix. Their output tells you (and them) exactly what's wrong | From 56a5eb09ef100c39e2b85e83b71fe6144e24f522 Mon Sep 17 00:00:00 2001 From: Hermes <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:39:24 -0700 Subject: [PATCH 027/117] docs(computer-use): drop remaining existing-profile grant references The grant_existing_profile key was removed in PR #98057; sweep the mode table, opt-in section, runtime-lifecycle notes, config example, and CLI reference that still documented it. --- website/docs/reference/cli-commands.md | 2 +- .../docs/user-guide/features/computer-use.md | 47 ++++++------------- 2 files changed, 16 insertions(+), 33 deletions(-) diff --git a/website/docs/reference/cli-commands.md b/website/docs/reference/cli-commands.md index 6ba9a5e950..8a0fe76233 100644 --- a/website/docs/reference/cli-commands.md +++ b/website/docs/reference/cli-commands.md @@ -1485,7 +1485,7 @@ Registering raw Cua MCP tools is an alternative when you need Cua's low-level tool vocabulary. `cua-driver skills install` detects Hermes and links Cua's skill pack into the Hermes skills directory automatically. -Permission mode, capability-manifest approval, and the existing-profile grant +Permission mode and capability-manifest approval belong to runtime launch. In bounded mode Hermes passes Cua's canonical `--capability-manifest` and `--approve-capability-manifest` flags. Every MCP transport owns a private lifecycle session inside its runtime. Public session diff --git a/website/docs/user-guide/features/computer-use.md b/website/docs/user-guide/features/computer-use.md index 12a00200dd..d1d2983c3f 100644 --- a/website/docs/user-guide/features/computer-use.md +++ b/website/docs/user-guide/features/computer-use.md @@ -99,34 +99,19 @@ or add `computer_use` to your enabled toolsets in `~/.hermes/config.yaml`. ## Permission modes and logged-in browser profiles Hermes maps its existing approval UX onto cua-driver's immutable runtime -modes. Permission mode, capability manifest approval, and the existing-profile -grant are launch settings. They cannot change after the runtime starts: +modes. Permission mode and capability manifest approval are launch settings. +They cannot change after the runtime starts: -| Hermes session | cua-driver mode | Human intervention | `existing_profile` | -|---|---|---|---| -| Manual or smart approvals (default) | `standard` | Normal Hermes approvals; Cua stops at its protected boundary | Refuses unless `computer_use.grant_existing_profile: true` (one-time config opt-in) | -| `computer_use.permission_mode: bounded` + reviewed manifest | private `bounded` daemon | You review and approve the capability manifest once, at launch | Allowed only within the manifest's declared profiles/origins/tools; everything else fails closed | -| `--yolo`, `/yolo`, or `approvals.mode: off` | private `unrestricted` daemon | One explicit Hermes risk acceptance; no runtime Cua prompts | Refuses unless `computer_use.grant_existing_profile: true`; YOLO does not substitute for this grant | +| Hermes session | cua-driver mode | Human intervention | +|---|---|---| +| Manual or smart approvals (default) | `standard` | Normal Hermes approvals; Cua stops at its protected boundary | +| `computer_use.permission_mode: bounded` + reviewed manifest | private `bounded` daemon | You review and approve the capability manifest once, at launch | +| `--yolo`, `/yolo`, or `approvals.mode: off` | private `unrestricted` daemon | One explicit Hermes risk acceptance; no runtime Cua prompts | -### Attaching to your signed-in browser - -The agent can drive a Chrome/Edge window you already have open — including a -signed-in profile — **without restarting the browser, copying the profile, or -touching your tabs**. Because DevTools access exposes that profile's live -pages, cookies, and storage, cua-driver requires an explicit human grant that -ordinary tool approval cannot substitute for. You opt in once, in config.yaml: - -```yaml -computer_use: - grant_existing_profile: true -``` - -Hermes then launches the cua-driver runtime with the trusted-launcher grant -(`--grant existing-profile`), and the standard capture/click/type actions -succeed against the exact `(pid, window_id)` of that signed-in window. Leave -it `false` (the default) and existing-profile attachment fails closed; -driver-owned isolated profiles work either way and are what the agent -prefers. +Browser work — including pages in a signed-in profile — goes through the +`browser` toolset (`browser_exec`), not `computer_use`. The former +`computer_use.grant_existing_profile` opt-in was removed along with the typed +browser route; a leftover key in config.yaml is ignored. ### Bounded mode for repeatable automation @@ -167,14 +152,13 @@ public session name is only a label for cursor identity and session-scoped state. It does not select, share, or keep a runtime alive. Turning `/yolo` off, resetting or closing the Hermes session, cancellation cleanup, or process exit closes that transport session. Hermes also stops private runtimes that it -launched for bounded, unrestricted, or existing-profile access. One Hermes -conversation cannot change another runtime's mode or grants. On macOS, a -standard runtime with an existing-profile grant uses a fresh CuaDriver.app -daemon on a private socket. Bounded and unrestricted modes use a private +launched for bounded or unrestricted access. One Hermes +conversation cannot change another runtime's mode or grants. Bounded and +unrestricted modes use a private embedded service under the Hermes host identity. `smart` approval remains `standard`: an LLM classification cannot stand in for -a reviewed manifest or a launch-time grant. +a reviewed manifest.
@@ -444,7 +428,6 @@ Permission mode and manifest (see computer_use: permission_mode: standard # standard (default) | bounded capability_manifest: "" # capability manifest path, required for bounded - grant_existing_profile: false # opt-in: attach in standard or unrestricted mode ``` Override the driver binary path (tests / CI / local builds): From 5c6e5e7ea3a8de511f64b17271fe9b1e1c96b1bb Mon Sep 17 00:00:00 2001 From: webtecnica <75556242+webtecnica@users.noreply.github.com> Date: Sun, 19 Jul 2026 00:29:26 -0300 Subject: [PATCH 028/117] feat(cli): add /plan command (#67264) Generate a structured execution plan without executing tools. Uses _pending_agent_seed injection (same pattern as /moa). --- cli.py | 2 ++ hermes_cli/commands.py | 2 ++ 2 files changed, 4 insertions(+) diff --git a/cli.py b/cli.py index f2502538e4..bcf71903eb 100644 --- a/cli.py +++ b/cli.py @@ -12862,6 +12862,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._handle_review_command(cmd_original) elif canonical == "loop": self._handle_loop_command(cmd_original) + elif canonical == "plan": + self._handle_plan_command(cmd_original) elif canonical == "moa": # /moa is one-shot sugar only: run a single prompt through the # default MoA preset, then restore the prior model. To *switch* to a diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index b8e388c3fc..0957dbcd98 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -227,6 +227,8 @@ COMMAND_REGISTRY: list[CommandDef] = [ aliases=("proactive",), args_hint="[interval] [--times N] [--until ] | status | pause | resume | stop", argument_mode="mixed", busy_policy="dispatch", busy_handler="loop"), + CommandDef("plan", "Write a markdown implementation plan to .hermes/plans/ without executing anything", "Session", + args_hint="[task]"), CommandDef("moa", "Run one prompt through the default Mixture of Agents preset, then restore your model", "Session", args_hint="", busy_policy="reject", busy_handler="moa"), CommandDef("subgoal", "Add or manage extra criteria on the active goal", "Session", From 0f3fcacd3f92d02d163789b4c3441b6388f8753d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:46:26 -0700 Subject: [PATCH 029/117] feat: /plan graduates from bundled skill to built-in command on every surface MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The bundled plan skill's auto-generated slash command fell off the capped Telegram/Discord command menus for most installs (skills are the only tier trimmed at the platform caps, alphabetically — 'plan' sat past the cutoff at index 57 of 82 bundled skills). Converting it to a first-class CommandDef gives it a guaranteed core-tier menu slot on every platform. - agent/plan_prompt.py: build_plan_prompt() — plan-mode rules + authoring craft distilled from the retired skill; prompt-injection pattern like /learn and /init (no engine, no model-tool footprint, cache-safe). - CLI: _handle_plan_command mixin handler (pending-input injection). - Gateway: /plan branch rewrites event.text and falls through (role alternation preserved). - TUI: command.dispatch branch ('plan' was already in _PENDING_INPUT_COMMANDS). - Removed skills/software-development/plan/ + docs pages (EN + zh-Hans), catalog rows, sidebar entry, related_skills references. - PROTECTED_BUILTIN_SKILLS is now empty (mechanism kept); dependent curator/usage tests moved to monkeypatched sentinels. Salvages #67292 by @webtecnica (credit: first /plan command submission, issue #67264); reworked from inline planning prompt to the prompt-injection pattern with workspace-saved plans. Closes #67264, closes #36821 (empty /plan infers task from conversation context). --- agent/plan_prompt.py | 103 +++++ gateway/run.py | 28 ++ hermes_cli/cli_commands_mixin.py | 27 ++ hermes_cli/tips.py | 2 +- .../software-development/grill-me/SKILL.md | 2 +- .../subagent-driven-development/SKILL.md | 2 +- .../research/research-paper-writing/SKILL.md | 2 +- .../hermes-agent-skill-authoring/SKILL.md | 2 +- skills/software-development/plan/SKILL.md | 338 ----------------- .../requesting-code-review/SKILL.md | 2 +- .../simplify-code/SKILL.md | 2 +- skills/software-development/spike/SKILL.md | 2 +- .../systematic-debugging/SKILL.md | 2 +- .../test-driven-development/SKILL.md | 2 +- tests/agent/test_curator.py | 12 +- tests/agent/test_plan_prompt.py | 84 +++++ .../hermes_cli/test_curator_pin_visibility.py | 16 +- tests/tools/test_skill_usage.py | 5 +- tools/skill_usage.py | 11 +- tui_gateway/methods_tools.py | 8 + website/docs/reference/skills-catalog.md | 1 - website/docs/reference/slash-commands.md | 4 +- website/docs/user-guide/features/curator.md | 2 +- website/docs/user-guide/features/skills.md | 4 +- .../research-research-paper-writing.md | 2 +- ...evelopment-hermes-agent-skill-authoring.md | 2 +- .../software-development-plan.md | 356 ------------------ ...ware-development-requesting-code-review.md | 2 +- .../software-development-simplify-code.md | 2 +- .../software-development-spike.md | 2 +- ...ftware-development-systematic-debugging.md | 2 +- ...are-development-test-driven-development.md | 2 +- .../software-development-grill-me.md | 2 +- ...development-subagent-driven-development.md | 2 +- .../current/reference/skills-catalog.md | 1 - .../current/reference/slash-commands.md | 2 +- .../current/user-guide/features/skills.md | 4 +- .../research-research-paper-writing.md | 2 +- .../software-development-plan.md | 76 ---- .../software-development-spike.md | 2 +- website/sidebars.ts | 1 - 41 files changed, 307 insertions(+), 818 deletions(-) create mode 100644 agent/plan_prompt.py delete mode 100644 skills/software-development/plan/SKILL.md create mode 100644 tests/agent/test_plan_prompt.py delete mode 100644 website/docs/user-guide/skills/bundled/software-development/software-development-plan.md delete mode 100644 website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-plan.md diff --git a/agent/plan_prompt.py b/agent/plan_prompt.py new file mode 100644 index 0000000000..0678371d50 --- /dev/null +++ b/agent/plan_prompt.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +"""``/plan`` — build the plan-mode prompt that turns the user's request into a +saved markdown implementation plan, with no execution. + +``/plan`` used to be a bundled skill (``skills/software-development/plan``) +whose auto-generated slash command fell off the capped Telegram/Discord command +menus for most installs (skills are the only tier trimmed at the platform +caps, alphabetically — ``plan`` sat past the cutoff). It is now a first-class +built-in: this module builds ONE prompt that instructs the live agent to + + 1. Stay in planning mode for the turn — read-only inspection is allowed, + but no implementation, no mutating commands, no side effects. + 2. Write a concrete, bite-sized, TDD-shaped markdown plan under + ``.hermes/plans/`` in the active workspace via ``write_file``. + +There is no engine and no model-tool footprint: the agent does the work with +its existing toolset, so this works identically on local, Docker, and remote +terminal backends. Every surface (CLI ``/plan``, gateway ``/plan``, TUI +``/plan``) calls :func:`build_plan_prompt` and feeds the result to the agent +as a normal turn — same pattern as ``/learn`` and ``/init``, preserving +prompt-cache invariants (no system-prompt or history mutation). +""" + +from __future__ import annotations + +# The plan-mode ground rules + authoring craft, distilled from the retired +# bundled skill (v2.0.0, writing-craft adapted from obra/superpowers). +# Embedded in the prompt so the agent plans the way a maintainer would. +_PLAN_MODE_RULES = """\ +For this turn, you are in PLAN MODE — planning only. + +- Do not implement code. +- Do not edit project files except the plan markdown file itself. +- Do not run mutating terminal commands, commit, push, or perform external + actions. +- You may inspect the repo or other context with read-only commands/tools + when needed. +- Your deliverable is a markdown plan saved inside the active workspace under + `.hermes/plans/YYYY-MM-DD_HHMMSS-.md` (create the directory if + needed; Hermes file tools are backend-aware, so this relative path keeps + the plan with the workspace on local, docker, ssh, modal, and daytona + backends). If the runtime provides a specific target path, use that exact + path instead. +""" + +_PLAN_CRAFT = """\ +Write the plan for an implementer with zero context for the codebase and +questionable taste. A good plan makes implementation obvious — if someone has +to guess, the plan is incomplete. + +Structure (include the sections that are relevant): +- Goal — one sentence. +- Current context / assumptions. +- Architecture / proposed approach — 2-3 sentences. +- Step-by-step tasks. Each task is bite-sized (2-5 minutes of focused work), + names exact file paths (`src/models/user.py`, not "the model file"), + includes complete copy-pasteable code where code is needed, and exact + commands with expected output for verification. +- Tests / validation — for code tasks, follow the TDD cycle per task: write + the failing test, run it to verify failure, implement minimally, run to + verify pass, commit. +- Risks, tradeoffs, and open questions. + +Principles: DRY, YAGNI, TDD, frequent commits. Avoid vague tasks ("add +authentication"), incomplete code ("add validation here"), and unverifiable +steps ("test it works" — instead: the exact command and its expected output). + +Interaction style: +- If the request is clear enough, write the plan directly. +- If it is genuinely underspecified, ask a brief clarifying question instead + of guessing. +- After saving the plan, reply briefly with what you planned and the saved + path, and offer to execute it (e.g. via subagent-driven development) — + but do not start executing in this turn. +""" + + +def build_plan_prompt(task: str = "") -> str: + """Build the plan-mode prompt for the live agent. + + Args: + task: What to plan. Empty → infer the task from the current + conversation context (mirrors the retired skill's behavior and + issue #36821's "plan from context" expectation). + """ + task = (task or "").strip() + if task: + task_block = f"Task to plan:\n{task}\n" + else: + task_block = ( + "No explicit task was given with /plan — infer the task from the " + "current conversation context (the thing we have been discussing " + "or working toward). If the conversation does not imply a task, " + "ask a brief clarifying question.\n" + ) + return ( + "[/plan — plan mode]\n\n" + + _PLAN_MODE_RULES + + "\n" + + task_block + + "\n" + + _PLAN_CRAFT + ) diff --git a/gateway/run.py b/gateway/run.py index df62736448..24abe13daf 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -18457,6 +18457,34 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: return "Could not start /learn — please try again." + if canonical == "plan": + # /plan: rewrite the turn to the plan-mode prompt and fall + # through to normal agent processing (same fall-through as /learn + # so role alternation is preserved). The live agent inspects the + # workspace with read-only tools and saves the markdown plan + # under .hermes/plans/ via write_file. No engine, works on any + # backend. + from agent.plan_prompt import build_plan_prompt + + _plan_task = event.get_command_args().strip() + _ack = ( + f"Planning: {_plan_task[:80]}{'…' if len(_plan_task) > 80 else ''}" + if _plan_task + else "Planning from this conversation's context…" + ) + try: + adapter = self._adapter_for_source(source) + if adapter: + _ack_meta = self._thread_metadata_for_source(source) + await adapter.send(str(source.chat_id), _ack, metadata=_ack_meta) + except Exception: + logger.debug("plan ack send failed", exc_info=True) + try: + event.text = build_plan_prompt(_plan_task) + # fall through to agent processing + except Exception: + return "Could not start /plan — please try again." + if canonical == "init": # /init: rewrite the turn to a guidance-laden prompt and fall # through to normal agent processing (same fall-through as /learn diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 833317d17f..52a416fc93 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -2201,6 +2201,33 @@ class CLICommandsMixin: else: # pragma: no cover - defensive (no live input loop) print(" /learn needs an active chat session to run.") + def _handle_plan_command(self, cmd: str): + """Handle /plan — write a markdown implementation plan, no execution. + + Mirrors /learn: build the plan-mode prompt and inject it onto the + agent's input queue as a normal user turn. The live agent inspects + the workspace with read-only tools and saves the plan under + ``.hermes/plans/`` via ``write_file``. No engine, no model-tool + footprint, works on any terminal backend, and preserves prompt-cache + invariants (no system prompt or history mutation). + """ + from agent.plan_prompt import build_plan_prompt + + # Everything after the command word is the task to plan (optional — + # empty infers the task from conversation context). + parts = cmd.strip().split(None, 1) + task = parts[1].strip() if len(parts) > 1 else "" + + msg = build_plan_prompt(task) + if task: + print(f"\n📋 Planning: {task[:80]}{'...' if len(task) > 80 else ''}") + else: + print("\n📋 Planning from this conversation's context...") + if hasattr(self, "_pending_input"): + self._pending_input.put(msg) + else: # pragma: no cover - defensive (no live input loop) + print(" /plan needs an active chat session to run.") + def _handle_init_command(self, cmd: str): """Handle /init — generate or update AGENTS.md from a project scan. diff --git a/hermes_cli/tips.py b/hermes_cli/tips.py index 1723365d3b..0a8f89b4df 100644 --- a/hermes_cli/tips.py +++ b/hermes_cli/tips.py @@ -175,7 +175,7 @@ TIPS = [ "Skills can restrict to specific OS platforms — some only load on macOS or Linux.", "skills.external_dirs in config.yaml lets you load skills from custom directories.", "The agent can create its own skills as procedural memory using skill_manage.", - "The plan skill saves markdown plans under .hermes/plans/ in the active workspace.", + "/plan writes a markdown implementation plan to .hermes/plans/ without executing anything.", # --- Cron & Scheduling --- "Cron jobs can attach skills: hermes cron add --skill blogwatcher \"Check for new posts\".", diff --git a/optional-skills/software-development/grill-me/SKILL.md b/optional-skills/software-development/grill-me/SKILL.md index 1b723f9c51..18d9d06ac9 100644 --- a/optional-skills/software-development/grill-me/SKILL.md +++ b/optional-skills/software-development/grill-me/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [planning, adversarial, interview, decision-tree, pre-implementation, review, alignment] - related_skills: [plan, requesting-code-review, subagent-driven-development, test-driven-development] + related_skills: [requesting-code-review, subagent-driven-development, test-driven-development] --- # Grill Me diff --git a/optional-skills/software-development/subagent-driven-development/SKILL.md b/optional-skills/software-development/subagent-driven-development/SKILL.md index 3a1469f360..57904fd5b4 100644 --- a/optional-skills/software-development/subagent-driven-development/SKILL.md +++ b/optional-skills/software-development/subagent-driven-development/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [delegation, subagent, implementation, workflow, parallel] - related_skills: [plan, requesting-code-review, test-driven-development] + related_skills: [requesting-code-review, test-driven-development] --- # Subagent-Driven Development diff --git a/skills/research/research-paper-writing/SKILL.md b/skills/research/research-paper-writing/SKILL.md index d230ff6ca2..4e8d89c02e 100644 --- a/skills/research/research-paper-writing/SKILL.md +++ b/skills/research/research-paper-writing/SKILL.md @@ -11,7 +11,7 @@ metadata: hermes: tags: [Research, Paper Writing, Experiments, ML, AI, NeurIPS, ICML, ICLR, ACL, AAAI, COLM, LaTeX, Citations, Statistical Analysis] category: research - related_skills: [arxiv, subagent-driven-development, plan] + related_skills: [arxiv, subagent-driven-development] requires_toolsets: [terminal, files] --- diff --git a/skills/software-development/hermes-agent-skill-authoring/SKILL.md b/skills/software-development/hermes-agent-skill-authoring/SKILL.md index c199a0c589..e979f1b481 100644 --- a/skills/software-development/hermes-agent-skill-authoring/SKILL.md +++ b/skills/software-development/hermes-agent-skill-authoring/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [skills, authoring, hermes-agent, conventions, skill-md] - related_skills: [plan, requesting-code-review] + related_skills: [requesting-code-review] --- # Authoring Hermes-Agent Skills (in-repo) diff --git a/skills/software-development/plan/SKILL.md b/skills/software-development/plan/SKILL.md deleted file mode 100644 index e97eb6abcd..0000000000 --- a/skills/software-development/plan/SKILL.md +++ /dev/null @@ -1,338 +0,0 @@ ---- -name: plan -description: Write a markdown plan to .hermes/plans/; no execution. -version: 2.0.0 -author: Hermes Agent (writing-craft adapted from obra/superpowers) -license: MIT -platforms: [linux, macos, windows] -metadata: - hermes: - tags: [planning, plan-mode, implementation, workflow, design, documentation] - related_skills: [subagent-driven-development, test-driven-development, requesting-code-review] ---- - -# Plan Mode - -Use this skill when the user wants a plan instead of execution. - -## Core behavior - -For this turn, you are planning only. - -- Do not implement code. -- Do not edit project files except the plan markdown file. -- Do not run mutating terminal commands, commit, push, or perform external actions. -- You may inspect the repo or other context with read-only commands/tools when needed. -- Your deliverable is a markdown plan saved inside the active workspace under `.hermes/plans/`. - -## Output requirements - -Write a markdown plan that is concrete and actionable. - -Include, when relevant: -- Goal -- Current context / assumptions -- Proposed approach -- Step-by-step plan -- Files likely to change -- Tests / validation -- Risks, tradeoffs, and open questions - -If the task is code-related, include exact file paths, likely test targets, and verification steps. - -## Save location - -Save the plan with `write_file` under: -- `.hermes/plans/YYYY-MM-DD_HHMMSS-.md` - -Treat that as relative to the active working directory / backend workspace. Hermes file tools are backend-aware, so using this relative path keeps the plan with the workspace on local, docker, ssh, modal, and daytona backends. - -If the runtime provides a specific target path, use that exact path. -If not, create a sensible timestamped filename yourself under `.hermes/plans/`. - -## Interaction style - -- If the request is clear enough, write the plan directly. -- If no explicit instruction accompanies `/plan`, infer the task from the current conversation context. -- If it is genuinely underspecified, ask a brief clarifying question instead of guessing. -- After saving the plan, reply briefly with what you planned and the saved path. - ---- - -# Writing the Plan Well - -The rest of this skill is the craft of authoring a *good* implementation plan — the content that goes inside the markdown file above. - -## Overview - -Write comprehensive implementation plans assuming the implementer has zero context for the codebase and questionable taste. Document everything they need: which files to touch, complete code, testing commands, docs to check, how to verify. Give them bite-sized tasks. DRY. YAGNI. TDD. Frequent commits. - -Assume the implementer is a skilled developer but knows almost nothing about the toolset or problem domain. Assume they don't know good test design very well. - -**Core principle:** A good plan makes implementation obvious. If someone has to guess, the plan is incomplete. - -## When a Full Implementation Plan Helps - -**Always use before:** -- Implementing multi-step features -- Breaking down complex requirements -- Delegating to subagents via subagent-driven-development - -**Don't skip when:** -- Feature seems simple (assumptions cause bugs) -- You plan to implement it yourself (future you needs guidance) -- Working alone (documentation matters) - -## Bite-Sized Task Granularity - -**Each task = 2-5 minutes of focused work.** - -Every step is one action: -- "Write the failing test" — step -- "Run it to make sure it fails" — step -- "Implement the minimal code to make the test pass" — step -- "Run the tests and make sure they pass" — step -- "Commit" — step - -**Too big:** -```markdown -### Task 1: Build authentication system -[50 lines of code across 5 files] -``` - -**Right size:** -```markdown -### Task 1: Create User model with email field -[10 lines, 1 file] - -### Task 2: Add password hash field to User -[8 lines, 1 file] - -### Task 3: Create password hashing utility -[15 lines, 1 file] -``` - -## Plan Document Structure - -### Header (Required) - -Every plan MUST start with: - -```markdown -# [Feature Name] Implementation Plan - -> **For Hermes:** Use subagent-driven-development skill to implement this plan task-by-task. - -**Goal:** [One sentence describing what this builds] - -**Architecture:** [2-3 sentences about approach] - -**Tech Stack:** [Key technologies/libraries] - ---- -``` - -### Task Structure - -Each task follows this format: - -````markdown -### Task N: [Descriptive Name] - -**Objective:** What this task accomplishes (one sentence) - -**Files:** -- Create: `exact/path/to/new_file.py` -- Modify: `exact/path/to/existing.py:45-67` (line numbers if known) -- Test: `tests/path/to/test_file.py` - -**Step 1: Write failing test** - -```python -def test_specific_behavior(): - result = function(input) - assert result == expected -``` - -**Step 2: Run test to verify failure** - -Run: `pytest tests/path/test.py::test_specific_behavior -v` -Expected: FAIL — "function not defined" - -**Step 3: Write minimal implementation** - -```python -def function(input): - return expected -``` - -**Step 4: Run test to verify pass** - -Run: `pytest tests/path/test.py::test_specific_behavior -v` -Expected: PASS - -**Step 5: Commit** - -```bash -git add tests/path/test.py src/path/file.py -git commit -m "feat: add specific feature" -``` -```` - -## Writing Process - -### Step 1: Understand Requirements - -Read and understand: -- Feature requirements -- Design documents or user description -- Acceptance criteria -- Constraints - -### Step 2: Explore the Codebase - -Use Hermes tools to understand the project: - -```python -# Understand project structure -search_files("*.py", target="files", path="src/") - -# Look at similar features -search_files("similar_pattern", path="src/", file_glob="*.py") - -# Check existing tests -search_files("*.py", target="files", path="tests/") - -# Read key files -read_file("src/app.py") -``` - -### Step 3: Design Approach - -Decide: -- Architecture pattern -- File organization -- Dependencies needed -- Testing strategy - -### Step 4: Write Tasks - -Create tasks in order: -1. Setup/infrastructure -2. Core functionality (TDD for each) -3. Edge cases -4. Integration -5. Cleanup/documentation - -### Step 5: Add Complete Details - -For each task, include: -- **Exact file paths** (not "the config file" but `src/config/settings.py`) -- **Complete code examples** (not "add validation" but the actual code) -- **Exact commands** with expected output -- **Verification steps** that prove the task works - -### Step 6: Review the Plan - -Check: -- [ ] Tasks are sequential and logical -- [ ] Each task is bite-sized (2-5 min) -- [ ] File paths are exact -- [ ] Code examples are complete (copy-pasteable) -- [ ] Commands are exact with expected output -- [ ] No missing context -- [ ] DRY, YAGNI, TDD principles applied - -## Principles - -### DRY (Don't Repeat Yourself) - -**Bad:** Copy-paste validation in 3 places -**Good:** Extract validation function, use everywhere - -### YAGNI (You Aren't Gonna Need It) - -**Bad:** Add "flexibility" for future requirements -**Good:** Implement only what's needed now - -```python -# Bad — YAGNI violation -class User: - def __init__(self, name, email): - self.name = name - self.email = email - self.preferences = {} # Not needed yet! - self.metadata = {} # Not needed yet! - -# Good — YAGNI -class User: - def __init__(self, name, email): - self.name = name - self.email = email -``` - -### TDD (Test-Driven Development) - -Every task that produces code should include the full TDD cycle: -1. Write failing test -2. Run to verify failure -3. Write minimal code -4. Run to verify pass - -See `test-driven-development` skill for details. - -### Frequent Commits - -Commit after every task: -```bash -git add [files] -git commit -m "type: description" -``` - -## Common Mistakes - -### Vague Tasks - -**Bad:** "Add authentication" -**Good:** "Create User model with email and password_hash fields" - -### Incomplete Code - -**Bad:** "Step 1: Add validation function" -**Good:** "Step 1: Add validation function" followed by the complete function code - -### Missing Verification - -**Bad:** "Step 3: Test it works" -**Good:** "Step 3: Run `pytest tests/test_auth.py -v`, expected: 3 passed" - -### Missing File Paths - -**Bad:** "Create the model file" -**Good:** "Create: `src/models/user.py`" - -## Execution Handoff - -After saving the plan, offer the execution approach: - -**"Plan complete and saved. Ready to execute using subagent-driven-development — I'll dispatch a fresh subagent per task with two-stage review (spec compliance then code quality). Shall I proceed?"** - -When executing, use the `subagent-driven-development` skill: -- Fresh `delegate_task` per task with full context -- Spec compliance review after each task -- Code quality review after spec passes -- Proceed only when both reviews approve - -## Remember - -``` -Bite-sized tasks (2-5 min each) -Exact file paths -Complete code (copy-pasteable) -Exact commands with expected output -Verification steps -DRY, YAGNI, TDD -Frequent commits -``` - -**A good plan makes implementation obvious.** diff --git a/skills/software-development/requesting-code-review/SKILL.md b/skills/software-development/requesting-code-review/SKILL.md index ad861e9ff0..0a543cee2b 100644 --- a/skills/software-development/requesting-code-review/SKILL.md +++ b/skills/software-development/requesting-code-review/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [code-review, security, verification, quality, pre-commit, auto-fix] - related_skills: [subagent-driven-development, plan, test-driven-development, github-code-review] + related_skills: [subagent-driven-development, test-driven-development, github-code-review] --- # Pre-Commit Code Verification diff --git a/skills/software-development/simplify-code/SKILL.md b/skills/software-development/simplify-code/SKILL.md index b1ca84fdc4..0dcfae3084 100644 --- a/skills/software-development/simplify-code/SKILL.md +++ b/skills/software-development/simplify-code/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [code-review, cleanup, refactor, delegation, subagent, parallel, simplify] - related_skills: [requesting-code-review, test-driven-development, plan] + related_skills: [requesting-code-review, test-driven-development] --- # Simplify Code — Parallel Review & Cleanup diff --git a/skills/software-development/spike/SKILL.md b/skills/software-development/spike/SKILL.md index 94ca054dd2..cd2d97fb14 100644 --- a/skills/software-development/spike/SKILL.md +++ b/skills/software-development/spike/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [spike, prototype, experiment, feasibility, throwaway, exploration, research, planning, mvp, proof-of-concept] - related_skills: [sketch, subagent-driven-development, plan] + related_skills: [sketch, subagent-driven-development] --- # Spike diff --git a/skills/software-development/systematic-debugging/SKILL.md b/skills/software-development/systematic-debugging/SKILL.md index 7ff990e278..275746e7bb 100644 --- a/skills/software-development/systematic-debugging/SKILL.md +++ b/skills/software-development/systematic-debugging/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [debugging, troubleshooting, problem-solving, root-cause, investigation] - related_skills: [test-driven-development, plan, subagent-driven-development] + related_skills: [test-driven-development, subagent-driven-development] --- # Systematic Debugging diff --git a/skills/software-development/test-driven-development/SKILL.md b/skills/software-development/test-driven-development/SKILL.md index 67fd061ea7..0979f7222a 100644 --- a/skills/software-development/test-driven-development/SKILL.md +++ b/skills/software-development/test-driven-development/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [testing, tdd, development, quality, red-green-refactor] - related_skills: [systematic-debugging, plan, subagent-driven-development] + related_skills: [systematic-debugging, subagent-driven-development] --- # Test-Driven Development (TDD) diff --git a/tests/agent/test_curator.py b/tests/agent/test_curator.py index a7defb7338..4b8a46c9bb 100644 --- a/tests/agent/test_curator.py +++ b/tests/agent/test_curator.py @@ -331,13 +331,17 @@ def _disable_prune_builtins(curator_env, monkeypatch): def test_protected_builtin_never_archived_even_when_stale(curator_env, monkeypatch): - """A protected built-in (e.g. `plan`) is never archived, even when it is a - stale bundled skill under prune_builtins — it backs a load-bearing slash - command and must survive every curator pass.""" + """A protected built-in is never archived, even when it is a stale + bundled skill under prune_builtins — it backs a load-bearing UX path and + must survive every curator pass. + + The shipped set is currently empty (``plan`` graduated to a built-in + command), so the mechanism is exercised with a sentinel name.""" u = curator_env["usage"] c = curator_env["curator"] skills_dir = curator_env["home"] / "skills" - name = next(iter(u.PROTECTED_BUILTIN_SKILLS)) # the real protected name(s) + name = "sentinel-protected-skill" + monkeypatch.setattr(u, "PROTECTED_BUILTIN_SKILLS", {name}) _write_skill(skills_dir, name) (skills_dir / ".bundled_manifest").write_text(f"{name}:abc\n", encoding="utf-8") _enable_prune_builtins(curator_env, monkeypatch) diff --git a/tests/agent/test_plan_prompt.py b/tests/agent/test_plan_prompt.py new file mode 100644 index 0000000000..63cf357ce9 --- /dev/null +++ b/tests/agent/test_plan_prompt.py @@ -0,0 +1,84 @@ +"""Tests for the built-in /plan command (formerly the bundled `plan` skill). + +Covers the shared prompt builder (agent.plan_prompt.build_plan_prompt) and the +registry wiring that makes /plan a first-class command on every surface — +CLI, gateway messengers, and TUI. The skill-to-builtin move exists precisely +so /plan survives the Telegram/Discord command-menu caps that trimmed it as +an alphabetical skill entry. +""" + +from agent.plan_prompt import build_plan_prompt + + +class TestBuildPlanPrompt: + def test_task_is_included_verbatim(self): + task = "migrate the auth provider to OIDC with zero downtime" + prompt = build_plan_prompt(task) + assert task in prompt + + def test_empty_task_infers_from_conversation(self): + prompt = build_plan_prompt("") + assert "infer the task from the current conversation context" in prompt + # Whitespace-only behaves the same. + assert "infer the task" in build_plan_prompt(" ") + + def test_plan_mode_ground_rules_always_present(self): + for arg in ("", "build a REST API"): + prompt = build_plan_prompt(arg) + assert "PLAN MODE" in prompt + assert "Do not implement code" in prompt + assert "Do not run mutating terminal commands" in prompt + assert "read-only" in prompt + + def test_save_location_contract(self): + prompt = build_plan_prompt("anything") + assert ".hermes/plans/" in prompt + assert "YYYY-MM-DD_HHMMSS-.md" in prompt + + def test_authoring_craft_travels_with_every_prompt(self): + prompt = build_plan_prompt("x") + assert "bite-sized" in prompt + assert "exact file paths" in prompt.lower() + assert "TDD" in prompt + assert "YAGNI" in prompt + + def test_no_execution_handoff_in_same_turn(self): + prompt = build_plan_prompt("x") + assert "do not start executing in this turn" in prompt + + +class TestPlanRegistryWiring: + def test_plan_is_registered_and_resolves(self): + from hermes_cli.commands import resolve_command + + cmd = resolve_command("plan") + assert cmd is not None + assert cmd.name == "plan" + + def test_plan_is_not_cli_only(self): + # /plan must reach messaging gateways — the whole point of the + # builtin conversion is menu visibility on Telegram/Discord. + from hermes_cli.commands import resolve_command + + assert not resolve_command("plan").cli_only + + def test_plan_reaches_gateway_dispatch(self): + from hermes_cli.commands import GATEWAY_KNOWN_COMMANDS + + assert "plan" in GATEWAY_KNOWN_COMMANDS + + def test_plan_in_telegram_bot_commands(self): + from hermes_cli.commands import telegram_bot_commands + + names = {n for n, _ in telegram_bot_commands()} + assert "plan" in names + + def test_no_bundled_plan_skill_remains(self): + # The bundled skill was removed with the builtin conversion; a + # leftover copy would collide with the core command at scan time + # (scan_skill_commands skips core-colliding skill slugs with a + # warning, so the skill would be silently unreachable). + from pathlib import Path + + repo_root = Path(__file__).resolve().parents[2] + assert not (repo_root / "skills" / "software-development" / "plan").exists() diff --git a/tests/hermes_cli/test_curator_pin_visibility.py b/tests/hermes_cli/test_curator_pin_visibility.py index 7e62740adf..a00b17102c 100644 --- a/tests/hermes_cli/test_curator_pin_visibility.py +++ b/tests/hermes_cli/test_curator_pin_visibility.py @@ -82,18 +82,22 @@ def pin_env(tmp_path, monkeypatch): # --------------------------------------------------------------------------- -def test_pin_fails_loudly_when_write_does_not_land(pin_env, capsys): +def test_pin_fails_loudly_when_write_does_not_land(pin_env, capsys, monkeypatch): """A skill that passes `is_agent_created()` but fails `is_curation_eligible()` must produce a NONZERO exit and an explanatory error — not a success message over a silent no-write. Real trigger: PROTECTED_BUILTIN_SKILLS blocks by NAME. A user's own skill - literally named ``plan`` is not in the bundled manifest, so - ``is_agent_created()`` says True — but ``is_protected_builtin()`` makes it - ineligible, and ``set_pinned()`` silently no-ops through - ``_mutate(require_curation_eligible=True)``.""" + whose name collides with a protected entry is not in the bundled + manifest, so ``is_agent_created()`` says True — but + ``is_protected_builtin()`` makes it ineligible, and ``set_pinned()`` + silently no-ops through ``_mutate(require_curation_eligible=True)``. + + The shipped set is currently empty (``plan`` graduated to a built-in + command), so the collision is staged with a monkeypatched sentinel.""" env = pin_env - name = "plan" # collides with the protected built-in name + name = "sentinel-protected-skill" # collides with the (patched) protected name + monkeypatch.setattr(env["usage"], "PROTECTED_BUILTIN_SKILLS", {name}) _make_skill(env["skills"], name) diff --git a/tests/tools/test_skill_usage.py b/tests/tools/test_skill_usage.py index 05a4bcef8a..87f22703de 100644 --- a/tests/tools/test_skill_usage.py +++ b/tests/tools/test_skill_usage.py @@ -539,7 +539,10 @@ def test_adopt_refuses_skills_the_user_does_not_own(skills_home, monkeypatch, ki json.dumps({"installed": {name: {}}}), encoding="utf-8", ) elif kind == "protected": - name = sorted(skill_usage.PROTECTED_BUILTIN_SKILLS)[0] + # Shipped set is currently empty (plan graduated to a built-in + # command) — stage a sentinel to exercise the mechanism. + name = "sentinel-protected-skill" + monkeypatch.setattr(skill_usage, "PROTECTED_BUILTIN_SKILLS", {name}) _write_skill(skills_dir, name) else: name = "no-such-skill" diff --git a/tools/skill_usage.py b/tools/skill_usage.py index 5fd6425cc7..2bd076388e 100644 --- a/tools/skill_usage.py +++ b/tools/skill_usage.py @@ -57,15 +57,14 @@ _VALID_STATES = {STATE_ACTIVE, STATE_STALE, STATE_ARCHIVED} # Load-bearing bundled built-ins the curator must NEVER archive or consolidate, # regardless of ``curator.prune_builtins``, pin state, or LLM judgment. These -# back advertised UX paths (e.g. ``plan`` powers the ``/plan`` slash-command -# flow and is referenced in tips/docs/fresh-profile seeding); silently archiving -# one turns its slash command into "Unknown command" with no signal to the user. +# back advertised UX paths; silently archiving one turns its slash command +# into "Unknown command" with no signal to the user. # Protection is by skill ``name`` (frontmatter ``name:``), matching the keys used # throughout this module. Keep this list tiny and intentional — it is not a # substitute for ``curator.prune_builtins: false``, which exempts ALL built-ins. -PROTECTED_BUILTIN_SKILLS: Set[str] = { - "plan", -} +# (``plan`` used to live here; it is now a first-class built-in command with +# no skill on disk, so the set is currently empty.) +PROTECTED_BUILTIN_SKILLS: Set[str] = set() def is_protected_builtin(skill_name: str) -> bool: diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index 80d94deee4..d31adb5242 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -622,6 +622,14 @@ def _(rid, params: dict) -> dict: from agent.learn_prompt import build_learn_prompt return _ok(rid, {"type": "send", "message": build_learn_prompt(arg)}) + if name == "plan": + # Plan mode: build the plan-mode prompt and submit it as a normal + # agent turn (same pattern as /learn). The live agent inspects the + # workspace read-only and saves the markdown plan under + # .hermes/plans/ via write_file. Works on any backend. + from agent.plan_prompt import build_plan_prompt + + return _ok(rid, {"type": "send", "message": build_plan_prompt(arg)}) if name == "init": # Generate-or-update AGENTS.md: build the guidance-laden prompt and # submit it as a normal agent turn (same pattern as /learn). The live diff --git a/website/docs/reference/skills-catalog.md b/website/docs/reference/skills-catalog.md index 419788352d..a5bf58c88b 100644 --- a/website/docs/reference/skills-catalog.md +++ b/website/docs/reference/skills-catalog.md @@ -149,7 +149,6 @@ If a skill is missing from this list but present in the repo, the catalog is reg | [`hermes-agent-skill-authoring`](/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring) | Author in-repo SKILL.md files: frontmatter and structure. | `software-development/hermes-agent-skill-authoring` | | [`inspecting-hermes-desktop-dom`](/docs/user-guide/skills/bundled/software-development/software-development-inspecting-hermes-desktop-dom) | Read the live Hermes desktop DOM/CSS over CDP. | `software-development/inspecting-hermes-desktop-dom` | | [`node-inspect-debugger`](/docs/user-guide/skills/bundled/software-development/software-development-node-inspect-debugger) | Debug Node.js via --inspect + Chrome DevTools Protocol CLI. | `software-development/node-inspect-debugger` | -| [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan) | Write a markdown plan to .hermes/plans/; no execution. | `software-development/plan` | | [`python-debugpy`](/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy) | Debug Python: pdb REPL + debugpy remote (DAP). | `software-development/python-debugpy` | | [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review) | Pre-commit review: security scan, quality gates, auto-fix. | `software-development/requesting-code-review` | | [`simplify-code`](/docs/user-guide/skills/bundled/software-development/software-development-simplify-code) | Parallel 4-agent cleanup of recent code changes. | `software-development/simplify-code` | diff --git a/website/docs/reference/slash-commands.md b/website/docs/reference/slash-commands.md index 608eb80b15..1dcc916782 100644 --- a/website/docs/reference/slash-commands.md +++ b/website/docs/reference/slash-commands.md @@ -11,7 +11,7 @@ Hermes has two slash-command surfaces, both driven by a central `COMMAND_REGISTR - **Interactive CLI slash commands** — dispatched by `cli.py`, with autocomplete from the registry - **Messaging slash commands** — dispatched by `gateway/run.py`, with help text and platform menus generated from the registry -Installed skills are also exposed as dynamic slash commands on both surfaces. That includes bundled skills like `/plan`, which opens plan mode and saves markdown plans under `.hermes/plans/` relative to the active workspace/backend working directory. +Installed skills are also exposed as dynamic slash commands on both surfaces. (`/plan` used to be one of these; it is now a built-in command — see the Session table below.) ## Permissions and admin/user split @@ -108,6 +108,7 @@ Type `/` in the CLI to open the autocomplete menu. Built-in commands are case-in | `/memory [pending\|approve\|reject\|approval]` | Review pending memory writes staged by the write-approval gate (`memory.write_approval`) and toggle the gate. See [Controlling memory writes](/user-guide/features/memory#controlling-memory-writes-write_approval). | | `/bundles` | List configured skill bundles — `/` slash aliases that preload several skills at once. Configure under `bundles:` in `~/.hermes/config.yaml`. See [Skill Bundles](/user-guide/features/skills#skill-bundles). | | `/learn ` | Distill a reusable skill from anything you describe — a directory, a URL, the workflow you just walked the agent through, or pasted notes. Open-ended: the agent gathers the sources with its own tools and authors a `SKILL.md` following the house authoring standards. Works in the CLI, the messaging gateway, the TUI, and the dashboard Skills page. | +| `/plan [task]` | Write a markdown implementation plan to `.hermes/plans/` in the active workspace — planning only, no execution. Empty argument infers the task from the conversation. (Formerly the bundled `plan` skill; now built-in so it survives the Telegram/Discord command-menu caps.) | | `/init [notes]` | Generate or update `AGENTS.md` project instructions from a repo scan (port of Codex `/init`). The agent inspects manifests, layout, and toolchain configs with its read-only tools, then writes a concise `AGENTS.md` — or, if one exists, merge-updates it preserving your content. Optional notes steer the emphasis. Works in the CLI, the messaging gateway, and the TUI. | | `/cron` | Manage scheduled tasks (list, add/create, edit, pause, resume, run, remove) | | `/suggestions [accept\|dismiss N\|catalog\|clear]` (alias: `/suggest`) | Review suggested automations. Use `/suggestions` to list pending suggestions, `/suggestions accept ` to create the proposed automation, `/suggestions dismiss ` to reject one, `/suggestions catalog` to add curated starter automations, and `/suggestions clear` to clear resolved suggestion records. Accepted jobs preserve the current surface as the delivery origin. | @@ -268,6 +269,7 @@ The messaging gateway supports the following built-in commands inside Telegram, | `/egress [status]` | Show Docker egress proxy status. | | `/init [notes]` | Generate or update `AGENTS.md` from a repo scan. | | `/learn ` | Distill a reusable skill from anything you describe. | +| `/plan [task]` | Write a markdown implementation plan to `.hermes/plans/`; no execution. | | `/bundles` | List configured skill bundles (`/` aliases that preload several skills). | | `/reload-skills` (alias: `/reload_skills`) | Re-scan `~/.hermes/skills/` for newly installed or removed skills. | | `/footer [on\|off\|status]` | Toggle the runtime-metadata footer on final replies (shows model, context %, and cwd). | diff --git a/website/docs/user-guide/features/curator.md b/website/docs/user-guide/features/curator.md index e6d8617f16..66dd545cc7 100644 --- a/website/docs/user-guide/features/curator.md +++ b/website/docs/user-guide/features/curator.md @@ -315,7 +315,7 @@ Skills named in any cron job's `skills:` list are protected the same way for **a Only **agent-created** skills can be pinned — `hermes curator pin` refuses on bundled and hub-installed skills with an explanatory message if you try. Hub-installed skills are never subject to curator mutation. Bundled built-in skills are only touched when `curator.prune_builtins: true` (the default), and even then only archived after `archive_after_days` of non-use — never patched, consolidated, or deleted. Set `curator.prune_builtins: false` to exempt bundled skills entirely. -A small set of **protected built-ins** is hardcoded as never-archivable and never-consolidatable, regardless of `curator.prune_builtins`, pin state, or LLM judgment. These back load-bearing UX — for example, `plan` powers the `/plan` slash-command flow — so silently archiving one would turn its slash command into an "Unknown command" error with no signal to you. Protected built-ins are filtered out of the curator's candidate list entirely, so the consolidation pass never sees them. +A small set of **protected built-ins** can be hardcoded as never-archivable and never-consolidatable, regardless of `curator.prune_builtins`, pin state, or LLM judgment. These back load-bearing UX, so silently archiving one would turn its slash command into an "Unknown command" error with no signal to you. (The set is currently empty — `plan`, its original member, graduated to a built-in `/plan` command with no skill on disk.) Protected built-ins are filtered out of the curator's candidate list entirely, so the consolidation pass never sees them. If you want a stronger guarantee than "no deletion" — for instance, freezing a skill's content entirely while the agent still reads it — edit `~/.hermes/skills//SKILL.md` directly with your editor. The pin guards tool-driven deletion, not your own filesystem access. diff --git a/website/docs/user-guide/features/skills.md b/website/docs/user-guide/features/skills.md index ec71205d0c..105c7da7a2 100644 --- a/website/docs/user-guide/features/skills.md +++ b/website/docs/user-guide/features/skills.md @@ -56,7 +56,7 @@ Every installed skill is automatically available as a slash command: /gif-search funny cats /axolotl help me fine-tune Llama 3 on my dataset /github-pr-workflow create a PR for the auth refactor -/plan design a rollout for migrating our auth provider +/songsee analyze the frequency spread of this mix # Just the skill name loads it and lets the agent ask what you need: /excalidraw @@ -82,7 +82,7 @@ that happen to start with `/` (like file paths) are never swallowed: For combinations you use repeatedly, prefer a [skill bundle](#skill-bundles) — same effect under one short command. -The bundled `plan` skill is a good example. Running `/plan [request]` loads the skill's instructions, telling Hermes to inspect context if needed, write a markdown implementation plan instead of executing the task, and save the result under `.hermes/plans/` relative to the active workspace/backend working directory. +(Plan mode works the same way but is a built-in command now: `/plan [request]` tells Hermes to inspect context if needed, write a markdown implementation plan instead of executing the task, and save the result under `.hermes/plans/` relative to the active workspace/backend working directory.) You can also interact with skills through natural conversation: diff --git a/website/docs/user-guide/skills/bundled/research/research-research-paper-writing.md b/website/docs/user-guide/skills/bundled/research/research-research-paper-writing.md index d7677cac8e..3514a6d73f 100644 --- a/website/docs/user-guide/skills/bundled/research/research-research-paper-writing.md +++ b/website/docs/user-guide/skills/bundled/research/research-research-paper-writing.md @@ -22,7 +22,7 @@ Write ML papers for NeurIPS/ICML/ICLR: design→submit. | Dependencies | `semanticscholar`, `arxiv`, `habanero`, `requests`, `scipy`, `numpy`, `matplotlib`, `SciencePlots` | | Platforms | linux, macos | | Tags | `Research`, `Paper Writing`, `Experiments`, `ML`, `AI`, `NeurIPS`, `ICML`, `ICLR`, `ACL`, `AAAI`, `COLM`, `LaTeX`, `Citations`, `Statistical Analysis` | -| Related skills | [`arxiv`](/docs/user-guide/skills/bundled/research/research-arxiv), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan) | +| Related skills | [`arxiv`](/docs/user-guide/skills/bundled/research/research-arxiv), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), `plan` (now the built-in `/plan` command) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring.md b/website/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring.md index 8953ec3d86..8a3eceb19d 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring.md @@ -21,7 +21,7 @@ Author in-repo SKILL.md files: frontmatter and structure. | License | MIT | | Platforms | linux, macos, windows | | Tags | `skills`, `authoring`, `hermes-agent`, `conventions`, `skill-md` | -| Related skills | [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review) | +| Related skills | `plan` (now the built-in `/plan` command), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-plan.md b/website/docs/user-guide/skills/bundled/software-development/software-development-plan.md deleted file mode 100644 index 2368cfc6d2..0000000000 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-plan.md +++ /dev/null @@ -1,356 +0,0 @@ ---- -title: "Plan — Write a markdown plan to .hermes/plans/; no execution" -sidebar_label: "Plan" -description: "Write a markdown plan to .hermes/plans/; no execution" ---- - -{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */} - -# Plan - -Write a markdown plan to .hermes/plans/; no execution. - -## Skill metadata - -| | | -|---|---| -| Source | Bundled (installed by default) | -| Path | `skills/software-development/plan` | -| Version | `2.0.0` | -| Author | Hermes Agent (writing-craft adapted from obra/superpowers) | -| License | MIT | -| Platforms | linux, macos, windows | -| Tags | `planning`, `plan-mode`, `implementation`, `workflow`, `design`, `documentation` | -| Related skills | [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review) | - -## Reference: full SKILL.md - -:::info -The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active. -::: - -# Plan Mode - -Use this skill when the user wants a plan instead of execution. - -## Core behavior - -For this turn, you are planning only. - -- Do not implement code. -- Do not edit project files except the plan markdown file. -- Do not run mutating terminal commands, commit, push, or perform external actions. -- You may inspect the repo or other context with read-only commands/tools when needed. -- Your deliverable is a markdown plan saved inside the active workspace under `.hermes/plans/`. - -## Output requirements - -Write a markdown plan that is concrete and actionable. - -Include, when relevant: -- Goal -- Current context / assumptions -- Proposed approach -- Step-by-step plan -- Files likely to change -- Tests / validation -- Risks, tradeoffs, and open questions - -If the task is code-related, include exact file paths, likely test targets, and verification steps. - -## Save location - -Save the plan with `write_file` under: -- `.hermes/plans/YYYY-MM-DD_HHMMSS-.md` - -Treat that as relative to the active working directory / backend workspace. Hermes file tools are backend-aware, so using this relative path keeps the plan with the workspace on local, docker, ssh, modal, and daytona backends. - -If the runtime provides a specific target path, use that exact path. -If not, create a sensible timestamped filename yourself under `.hermes/plans/`. - -## Interaction style - -- If the request is clear enough, write the plan directly. -- If no explicit instruction accompanies `/plan`, infer the task from the current conversation context. -- If it is genuinely underspecified, ask a brief clarifying question instead of guessing. -- After saving the plan, reply briefly with what you planned and the saved path. - ---- - -# Writing the Plan Well - -The rest of this skill is the craft of authoring a *good* implementation plan — the content that goes inside the markdown file above. - -## Overview - -Write comprehensive implementation plans assuming the implementer has zero context for the codebase and questionable taste. Document everything they need: which files to touch, complete code, testing commands, docs to check, how to verify. Give them bite-sized tasks. DRY. YAGNI. TDD. Frequent commits. - -Assume the implementer is a skilled developer but knows almost nothing about the toolset or problem domain. Assume they don't know good test design very well. - -**Core principle:** A good plan makes implementation obvious. If someone has to guess, the plan is incomplete. - -## When a Full Implementation Plan Helps - -**Always use before:** -- Implementing multi-step features -- Breaking down complex requirements -- Delegating to subagents via subagent-driven-development - -**Don't skip when:** -- Feature seems simple (assumptions cause bugs) -- You plan to implement it yourself (future you needs guidance) -- Working alone (documentation matters) - -## Bite-Sized Task Granularity - -**Each task = 2-5 minutes of focused work.** - -Every step is one action: -- "Write the failing test" — step -- "Run it to make sure it fails" — step -- "Implement the minimal code to make the test pass" — step -- "Run the tests and make sure they pass" — step -- "Commit" — step - -**Too big:** -```markdown -### Task 1: Build authentication system -[50 lines of code across 5 files] -``` - -**Right size:** -```markdown -### Task 1: Create User model with email field -[10 lines, 1 file] - -### Task 2: Add password hash field to User -[8 lines, 1 file] - -### Task 3: Create password hashing utility -[15 lines, 1 file] -``` - -## Plan Document Structure - -### Header (Required) - -Every plan MUST start with: - -```markdown -# [Feature Name] Implementation Plan - -> **For Hermes:** Use subagent-driven-development skill to implement this plan task-by-task. - -**Goal:** [One sentence describing what this builds] - -**Architecture:** [2-3 sentences about approach] - -**Tech Stack:** [Key technologies/libraries] - ---- -``` - -### Task Structure - -Each task follows this format: - -````markdown -### Task N: [Descriptive Name] - -**Objective:** What this task accomplishes (one sentence) - -**Files:** -- Create: `exact/path/to/new_file.py` -- Modify: `exact/path/to/existing.py:45-67` (line numbers if known) -- Test: `tests/path/to/test_file.py` - -**Step 1: Write failing test** - -```python -def test_specific_behavior(): - result = function(input) - assert result == expected -``` - -**Step 2: Run test to verify failure** - -Run: `pytest tests/path/test.py::test_specific_behavior -v` -Expected: FAIL — "function not defined" - -**Step 3: Write minimal implementation** - -```python -def function(input): - return expected -``` - -**Step 4: Run test to verify pass** - -Run: `pytest tests/path/test.py::test_specific_behavior -v` -Expected: PASS - -**Step 5: Commit** - -```bash -git add tests/path/test.py src/path/file.py -git commit -m "feat: add specific feature" -``` -```` - -## Writing Process - -### Step 1: Understand Requirements - -Read and understand: -- Feature requirements -- Design documents or user description -- Acceptance criteria -- Constraints - -### Step 2: Explore the Codebase - -Use Hermes tools to understand the project: - -```python -# Understand project structure -search_files("*.py", target="files", path="src/") - -# Look at similar features -search_files("similar_pattern", path="src/", file_glob="*.py") - -# Check existing tests -search_files("*.py", target="files", path="tests/") - -# Read key files -read_file("src/app.py") -``` - -### Step 3: Design Approach - -Decide: -- Architecture pattern -- File organization -- Dependencies needed -- Testing strategy - -### Step 4: Write Tasks - -Create tasks in order: -1. Setup/infrastructure -2. Core functionality (TDD for each) -3. Edge cases -4. Integration -5. Cleanup/documentation - -### Step 5: Add Complete Details - -For each task, include: -- **Exact file paths** (not "the config file" but `src/config/settings.py`) -- **Complete code examples** (not "add validation" but the actual code) -- **Exact commands** with expected output -- **Verification steps** that prove the task works - -### Step 6: Review the Plan - -Check: -- [ ] Tasks are sequential and logical -- [ ] Each task is bite-sized (2-5 min) -- [ ] File paths are exact -- [ ] Code examples are complete (copy-pasteable) -- [ ] Commands are exact with expected output -- [ ] No missing context -- [ ] DRY, YAGNI, TDD principles applied - -## Principles - -### DRY (Don't Repeat Yourself) - -**Bad:** Copy-paste validation in 3 places -**Good:** Extract validation function, use everywhere - -### YAGNI (You Aren't Gonna Need It) - -**Bad:** Add "flexibility" for future requirements -**Good:** Implement only what's needed now - -```python -# Bad — YAGNI violation -class User: - def __init__(self, name, email): - self.name = name - self.email = email - self.preferences = {} # Not needed yet! - self.metadata = {} # Not needed yet! - -# Good — YAGNI -class User: - def __init__(self, name, email): - self.name = name - self.email = email -``` - -### TDD (Test-Driven Development) - -Every task that produces code should include the full TDD cycle: -1. Write failing test -2. Run to verify failure -3. Write minimal code -4. Run to verify pass - -See `test-driven-development` skill for details. - -### Frequent Commits - -Commit after every task: -```bash -git add [files] -git commit -m "type: description" -``` - -## Common Mistakes - -### Vague Tasks - -**Bad:** "Add authentication" -**Good:** "Create User model with email and password_hash fields" - -### Incomplete Code - -**Bad:** "Step 1: Add validation function" -**Good:** "Step 1: Add validation function" followed by the complete function code - -### Missing Verification - -**Bad:** "Step 3: Test it works" -**Good:** "Step 3: Run `pytest tests/test_auth.py -v`, expected: 3 passed" - -### Missing File Paths - -**Bad:** "Create the model file" -**Good:** "Create: `src/models/user.py`" - -## Execution Handoff - -After saving the plan, offer the execution approach: - -**"Plan complete and saved. Ready to execute using subagent-driven-development — I'll dispatch a fresh subagent per task with two-stage review (spec compliance then code quality). Shall I proceed?"** - -When executing, use the `subagent-driven-development` skill: -- Fresh `delegate_task` per task with full context -- Spec compliance review after each task -- Code quality review after spec passes -- Proceed only when both reviews approve - -## Remember - -``` -Bite-sized tasks (2-5 min each) -Exact file paths -Complete code (copy-pasteable) -Exact commands with expected output -Verification steps -DRY, YAGNI, TDD -Frequent commits -``` - -**A good plan makes implementation obvious.** diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review.md b/website/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review.md index 66100e83fd..bf6f13b70d 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review.md @@ -21,7 +21,7 @@ Pre-commit review: security scan, quality gates, auto-fix. | License | MIT | | Platforms | linux, macos, windows | | Tags | `code-review`, `security`, `verification`, `quality`, `pre-commit`, `auto-fix` | -| Related skills | [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), [`github-code-review`](/docs/user-guide/skills/bundled/github/github-github-code-review) | +| Related skills | [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), `plan` (now the built-in `/plan` command), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), [`github-code-review`](/docs/user-guide/skills/bundled/github/github-github-code-review) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-simplify-code.md b/website/docs/user-guide/skills/bundled/software-development/software-development-simplify-code.md index 49e3d0143d..57a1534692 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-simplify-code.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-simplify-code.md @@ -21,7 +21,7 @@ Parallel 4-agent cleanup of recent code changes. | License | MIT | | Platforms | linux, macos, windows | | Tags | `code-review`, `cleanup`, `refactor`, `delegation`, `subagent`, `parallel`, `simplify` | -| Related skills | [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan) | +| Related skills | [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), `plan` (now the built-in `/plan` command) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-spike.md b/website/docs/user-guide/skills/bundled/software-development/software-development-spike.md index 56c0954b69..470e984504 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-spike.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-spike.md @@ -21,7 +21,7 @@ Throwaway experiments to validate an idea before build. | License | MIT | | Platforms | linux, macos, windows | | Tags | `spike`, `prototype`, `experiment`, `feasibility`, `throwaway`, `exploration`, `research`, `planning`, `mvp`, `proof-of-concept` | -| Related skills | [`sketch`](/docs/user-guide/skills/bundled/creative/creative-sketch), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan) | +| Related skills | [`sketch`](/docs/user-guide/skills/bundled/creative/creative-sketch), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), `plan` (now the built-in `/plan` command) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging.md b/website/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging.md index 09c9e3c268..a3f9f288aa 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging.md @@ -21,7 +21,7 @@ description: "4-phase root cause debugging: understand bugs before fixing" | License | MIT | | Platforms | linux, macos, windows | | Tags | `debugging`, `troubleshooting`, `problem-solving`, `root-cause`, `investigation` | -| Related skills | [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development) | +| Related skills | [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development), `plan` (now the built-in `/plan` command), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development.md b/website/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development.md index bdf88a6620..4ef912ec90 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development.md @@ -21,7 +21,7 @@ TDD: enforce RED-GREEN-REFACTOR, tests before code. | License | MIT | | Platforms | linux, macos, windows | | Tags | `testing`, `tdd`, `development`, `quality`, `red-green-refactor` | -| Related skills | [`systematic-debugging`](/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging), [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development) | +| Related skills | [`systematic-debugging`](/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging), `plan` (now the built-in `/plan` command), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/optional/software-development/software-development-grill-me.md b/website/docs/user-guide/skills/optional/software-development/software-development-grill-me.md index fef5deb6ee..8adeabcea0 100644 --- a/website/docs/user-guide/skills/optional/software-development/software-development-grill-me.md +++ b/website/docs/user-guide/skills/optional/software-development/software-development-grill-me.md @@ -21,7 +21,7 @@ Adversarial plan interview before implementation. | License | MIT | | Platforms | linux, macos, windows | | Tags | `planning`, `adversarial`, `interview`, `decision-tree`, `pre-implementation`, `review`, `alignment` | -| Related skills | [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development) | +| Related skills | `plan` (now the built-in `/plan` command), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`subagent-driven-development`](/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development) | ## Reference: full SKILL.md diff --git a/website/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development.md b/website/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development.md index e773edd76c..5d884d63e7 100644 --- a/website/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development.md +++ b/website/docs/user-guide/skills/optional/software-development/software-development-subagent-driven-development.md @@ -21,7 +21,7 @@ Execute plans via delegate_task subagents (2-stage review). | License | MIT | | Platforms | linux, macos, windows | | Tags | `delegation`, `subagent`, `implementation`, `workflow`, `parallel` | -| Related skills | [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development) | +| Related skills | `plan` (now the built-in `/plan` command), [`requesting-code-review`](/docs/user-guide/skills/bundled/software-development/software-development-requesting-code-review), [`test-driven-development`](/docs/user-guide/skills/bundled/software-development/software-development-test-driven-development) | ## Reference: full SKILL.md diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/skills-catalog.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/skills-catalog.md index 10962e16d1..618ff58c8b 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/skills-catalog.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/skills-catalog.md @@ -150,7 +150,6 @@ Hermes 在执行 `hermes update` 时也会同步内置技能,但同步清单 | [`dogfood`](/user-guide/skills/bundled/software-development/software-development-dogfood) | Web 应用探索性 QA:发现 bug、收集证据、生成报告。 | `software-development/dogfood` | | [`hermes-agent-skill-authoring`](/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring) | 编写仓库内 SKILL.md:frontmatter、验证器、结构规范。 | `software-development/hermes-agent-skill-authoring` | | [`node-inspect-debugger`](/user-guide/skills/bundled/software-development/software-development-node-inspect-debugger) | 通过 --inspect + Chrome DevTools Protocol CLI 调试 Node.js。 | `software-development/node-inspect-debugger` | -| [`plan`](/user-guide/skills/bundled/software-development/software-development-plan) | 计划模式:将 Markdown 计划写入 `.hermes/plans/`,不执行。 | `software-development/plan` | | [`python-debugpy`](/user-guide/skills/bundled/software-development/software-development-python-debugpy) | 调试 Python:pdb REPL + debugpy 远程调试(DAP)。 | `software-development/python-debugpy` | | [`requesting-code-review`](/user-guide/skills/bundled/software-development/software-development-requesting-code-review) | 提交前审查:安全扫描、质量门控、自动修复。 | `software-development/requesting-code-review` | | [`spike`](/user-guide/skills/bundled/software-development/software-development-spike) | 一次性实验,在正式构建前验证想法。 | `software-development/spike` | diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/slash-commands.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/slash-commands.md index f40215a40c..8bb77fa52b 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/slash-commands.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/reference/slash-commands.md @@ -11,7 +11,7 @@ Hermes 有两个斜杠命令入口,均由 `hermes_cli/commands.py` 中的中 - **交互式 CLI 斜杠命令** — 由 `cli.py` 分发,支持从注册表自动补全 - **消息平台斜杠命令** — 由 `gateway/run.py` 分发,帮助文本和平台菜单均从注册表生成 -已安装的 skill(技能)也会在两个入口以动态斜杠命令的形式暴露。这包括内置 skill,如 `/plan`,它会打开计划模式并将 markdown 计划保存在活动工作区/后端工作目录下的 `.hermes/plans/` 中。 +已安装的 skill(技能)也会在两个入口以动态斜杠命令的形式暴露。(`/plan` 曾是其中之一;它现在是内置命令 — 见下方 Session 表。) ## 权限与管理员/用户分级 diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/skills.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/skills.md index 78ac16c5d0..f2b2650084 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/skills.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/skills.md @@ -26,13 +26,13 @@ Skills 是 agent 在需要时可以加载的按需知识文档。它们遵循** /gif-search funny cats /axolotl help me fine-tune Llama 3 on my dataset /github-pr-workflow create a PR for the auth refactor -/plan design a rollout for migrating our auth provider +/songsee analyze the frequency spread of this mix # 只输入 skill 名称即可加载它,并让 agent 询问你的需求: /excalidraw ``` -捆绑的 `plan` skill 是一个很好的示例。运行 `/plan [request]` 会加载该 skill 的指令,告知 Hermes 在需要时检查上下文、编写 markdown 实现计划而非直接执行任务,并将结果保存在相对于当前工作区/后端工作目录的 `.hermes/plans/` 下。 +(计划模式的工作方式相同,但现在是内置命令:`/plan [request]` 告知 Hermes 在需要时检查上下文、编写 markdown 实现计划而非直接执行任务,并将结果保存在相对于当前工作区/后端工作目录的 `.hermes/plans/` 下。) 你也可以通过自然对话与 skills 交互: diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/research/research-research-paper-writing.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/research/research-research-paper-writing.md index 035f3c42a7..633bb1453a 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/research/research-research-paper-writing.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/research/research-research-paper-writing.md @@ -22,7 +22,7 @@ description: "为 NeurIPS/ICML/ICLR 撰写 ML 论文:设计→投稿" | 依赖项 | `semanticscholar`, `arxiv`, `habanero`, `requests`, `scipy`, `numpy`, `matplotlib`, `SciencePlots` | | 平台 | linux, macos | | 标签 | `Research`, `Paper Writing`, `Experiments`, `ML`, `AI`, `NeurIPS`, `ICML`, `ICLR`, `ACL`, `AAAI`, `COLM`, `LaTeX`, `Citations`, `Statistical Analysis` | -| 相关 skill | [`arxiv`](/user-guide/skills/bundled/research/research-arxiv), [`subagent-driven-development`](/user-guide/skills/bundled/software-development/software-development-subagent-driven-development), [`plan`](/user-guide/skills/bundled/software-development/software-development-plan) | +| 相关 skill | [`arxiv`](/user-guide/skills/bundled/research/research-arxiv), [`subagent-driven-development`](/user-guide/skills/bundled/software-development/software-development-subagent-driven-development), `plan`(现为内置 `/plan` 命令) | ## 参考:完整 SKILL.md diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-plan.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-plan.md deleted file mode 100644 index fc5bce2f41..0000000000 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-plan.md +++ /dev/null @@ -1,76 +0,0 @@ ---- -title: "Plan — Plan 模式:将 Markdown 计划写入" -sidebar_label: "Plan" -description: "Plan 模式:将 Markdown 计划写入" ---- - -{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */} - -# Plan - -Plan 模式:将 Markdown 计划写入 .hermes/plans/,不执行任何操作。 - -## Skill 元数据 - -| | | -|---|---| -| 来源 | 内置(默认安装) | -| 路径 | `skills/software-development/plan` | -| 版本 | `1.0.0` | -| 作者 | Hermes Agent | -| 许可证 | MIT | -| 平台 | linux, macos, windows | -| 标签 | `planning`, `plan-mode`, `implementation`, `workflow` | -| 相关 skill | [`writing-plans`](/user-guide/skills/bundled/software-development/software-development-writing-plans), [`subagent-driven-development`](/user-guide/skills/bundled/software-development/software-development-subagent-driven-development) | - -## 参考:完整 SKILL.md - -:::info -以下是 Hermes 在触发此 skill 时加载的完整 skill 定义。这是 skill 激活时 agent 所看到的指令内容。 -::: - -# Plan 模式 - -当用户需要计划而非执行时,使用此 skill。 - -## 核心行为 - -在本轮中,你仅进行规划。 - -- 不实现代码。 -- 不编辑项目文件,计划 Markdown 文件除外。 -- 不运行有副作用的终端命令,不提交、不推送,不执行外部操作。 -- 必要时可使用只读命令/工具检查仓库或其他上下文。 -- 你的交付物是保存在活跃工作区 `.hermes/plans/` 目录下的 Markdown 计划文件。 - -## 输出要求 - -编写一份具体且可操作的 Markdown 计划。 - -在相关时包含以下内容: -- 目标 -- 当前上下文 / 假设 -- 建议方案 -- 分步计划 -- 可能变更的文件 -- 测试 / 验证 -- 风险、权衡与待解问题 - -如果任务与代码相关,请包含精确的文件路径、可能的测试目标以及验证步骤。 - -## 保存位置 - -使用 `write_file` 将计划保存至: -- `.hermes/plans/YYYY-MM-DD_HHMMSS-.md` - -将该路径视为相对于活跃工作目录 / 后端工作区的路径。Hermes 文件工具具备后端感知能力,使用此相对路径可确保计划文件在 local、docker、ssh、modal 和 daytona 后端上均与工作区保持一致。 - -如果运行时提供了具体的目标路径,则使用该精确路径。 -如果没有,则自行在 `.hermes/plans/` 下创建一个合理的带时间戳的文件名。 - -## 交互风格 - -- 如果请求足够清晰,直接编写计划。 -- 如果 `/plan` 没有附带明确指令,则从当前对话上下文中推断任务。 -- 如果任务确实描述不足,提出简短的澄清问题,而非凭空猜测。 -- 保存计划后,简要回复你所规划的内容及保存路径。 \ No newline at end of file diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-spike.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-spike.md index e5486edd0d..6e4c54988d 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-spike.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/skills/bundled/software-development/software-development-spike.md @@ -21,7 +21,7 @@ description: "在构建前验证想法的一次性实验" | 许可证 | MIT | | 平台 | linux, macos, windows | | 标签 | `spike`, `prototype`, `experiment`, `feasibility`, `throwaway`, `exploration`, `research`, `planning`, `mvp`, `proof-of-concept` | -| 相关 skill | [`sketch`](/user-guide/skills/bundled/creative/creative-sketch)、[`writing-plans`](/user-guide/skills/bundled/software-development/software-development-writing-plans)、[`subagent-driven-development`](/user-guide/skills/bundled/software-development/software-development-subagent-driven-development)、[`plan`](/user-guide/skills/bundled/software-development/software-development-plan) | +| 相关 skill | [`sketch`](/user-guide/skills/bundled/creative/creative-sketch)、[`writing-plans`](/user-guide/skills/bundled/software-development/software-development-writing-plans)、[`subagent-driven-development`](/user-guide/skills/bundled/software-development/software-development-subagent-driven-development)、`plan`(现为内置 `/plan` 命令) | ## 参考:完整 SKILL.md diff --git a/website/sidebars.ts b/website/sidebars.ts index f8eff725d5..8288c1cfde 100644 --- a/website/sidebars.ts +++ b/website/sidebars.ts @@ -324,7 +324,6 @@ const sidebars: SidebarsConfig = { 'user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring', 'user-guide/skills/bundled/software-development/software-development-inspecting-hermes-desktop-dom', 'user-guide/skills/bundled/software-development/software-development-node-inspect-debugger', - 'user-guide/skills/bundled/software-development/software-development-plan', 'user-guide/skills/bundled/software-development/software-development-python-debugpy', 'user-guide/skills/bundled/software-development/software-development-requesting-code-review', 'user-guide/skills/bundled/software-development/software-development-simplify-code', From 83f4524b4205e0d04f7fa773a4b0288e7aee83ff Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:52:52 -0700 Subject: [PATCH 030/117] feat(discord): expose /plan in the native slash-command picker Text-message /plan already works on Discord via the gateway fall-through; this makes it discoverable in the / picker alongside /steer and /compress. --- plugins/platforms/discord/adapter.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index fcd846b52e..3fb770a5e3 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -5915,6 +5915,11 @@ class DiscordAdapter(BasePlatformAdapter): async def slash_steer(interaction: discord.Interaction, prompt: str): await self._run_simple_slash(interaction, f"/steer {prompt}".strip()) + @tree.command(name="plan", description="Write a markdown implementation plan (no execution)") + @discord.app_commands.describe(task="What to plan. Leave empty to infer from the conversation.") + async def slash_plan(interaction: discord.Interaction, task: str = ""): + await self._run_simple_slash(interaction, f"/plan {task}".strip()) + @tree.command(name="compress", description="Compress conversation context") async def slash_compress(interaction: discord.Interaction): await self._run_simple_slash(interaction, "/compress") From 60a664519cf9028a0a1a5f68e79b74beceb275ad Mon Sep 17 00:00:00 2001 From: LOGIN-TB <7146963+LOGIN-TB@users.noreply.github.com> Date: Sun, 9 Aug 2026 14:25:24 +0200 Subject: [PATCH 031/117] fix(telegram): prioritize dynamic skill menu commands --- hermes_cli/commands.py | 52 ++++++++++++++++++++----- tests/hermes_cli/test_commands.py | 64 +++++++++++++++++++++++++++++++ 2 files changed, 106 insertions(+), 10 deletions(-) diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index 0957dbcd98..943204ec5b 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -836,6 +836,8 @@ def _telegram_command_menu_config() -> dict[str, Any]: raw_priority = menu_cfg.get("priority") if isinstance(raw_priority, list): priority = [str(item) for item in raw_priority if str(item).strip()] + elif isinstance(raw_priority, str) and raw_priority.strip(): + priority = [raw_priority] else: priority = [] @@ -1091,6 +1093,32 @@ def _collect_gateway_skill_entries( # any clamp-induced renames. skill_triples = _clamp_command_names(skill_triples, reserved_names) + # Telegram's configured command-menu priority applies to dynamic skills as + # well as core commands. Reorder before trimming so a prioritized skill can + # claim a scarce remaining BotCommand slot instead of losing to the + # alphabetical default order. + if platform == "telegram": + priority = { + name: index + for index, name in enumerate(_telegram_effective_priority()) + } + skill_triples = [ + entry + for original_index, entry in sorted( + enumerate(skill_triples), + key=lambda item: ( + 0, + priority[item[1][0]], + item[0], + ) + if item[1][0] in priority + else ( + 1, + item[0], + ), + ) + ] + # Skills fill remaining slots — only tier that gets trimmed remaining = max(0, max_slots - len(all_entries)) hidden_count = max(0, len(skill_triples) - remaining) @@ -1112,7 +1140,12 @@ def telegram_menu_commands(max_commands: int = 100) -> tuple[list[tuple[str, str 2. Plugin slash commands (take precedence over skills) 3. Built-in skill commands (fill remaining slots, alphabetical) - Skills are the only tier that gets trimmed when the cap is hit. + Core, plugin, and skill tiers keep their existing relative order unless a + command is named in ``platforms.telegram.extra.command_menu.priority``. + Explicit priority is applied to the combined candidate list before the Bot + API cap, so a prioritized dynamic command can displace an unprioritized core + command when the core tier already fills the menu. + User-installed hub skills are excluded — accessible via /skills. Skills disabled for the ``"telegram"`` platform (via ``hermes skills config``) are excluded from the menu entirely. @@ -1121,22 +1154,21 @@ def telegram_menu_commands(max_commands: int = 100) -> tuple[list[tuple[str, str (menu_commands, hidden_count) where hidden_count is the number of commands omitted due to the cap. """ - core_commands = _prioritize_telegram_menu_commands(list(telegram_bot_commands())) + core_commands = list(telegram_bot_commands()) reserved_names = {n for n, _ in core_commands} - all_commands = list(core_commands) - hidden_core_count = max(0, len(all_commands) - max_commands) - - remaining_slots = max(0, max_commands - len(all_commands)) entries, hidden_count = _collect_gateway_skill_entries( platform="telegram", - max_slots=remaining_slots, + max_slots=max_commands, reserved_names=reserved_names, desc_limit=40, sanitize_name=_sanitize_telegram_name, ) - # Drop the cmd_key — Telegram only needs (name, desc) pairs. - all_commands.extend((n, d) for n, d, _k in entries) - return all_commands[:max_commands], hidden_count + hidden_core_count + # Drop the cmd_key — Telegram only needs (name, desc) pairs. Apply the + # configured priority across all tiers before enforcing the global cap. + all_commands = core_commands + [(n, d) for n, d, _k in entries] + all_commands = _prioritize_telegram_menu_commands(all_commands) + overflow_count = max(0, len(all_commands) - max_commands) + return all_commands[:max_commands], hidden_count + overflow_count def discord_skill_commands( diff --git a/tests/hermes_cli/test_commands.py b/tests/hermes_cli/test_commands.py index e7fb446a21..a4823bfc2c 100644 --- a/tests/hermes_cli/test_commands.py +++ b/tests/hermes_cli/test_commands.py @@ -807,6 +807,70 @@ class TestTelegramMenuCommands: # No empty string in menu names assert "" not in menu_names + def test_configured_priority_promotes_skill_into_last_menu_slot( + self, tmp_path, monkeypatch + ): + """A prioritized dynamic skill must not be trimmed behind alphabetical peers.""" + from unittest.mock import patch + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + local_dir = tmp_path / "skills" + local_dir.mkdir() + fake_cmds = { + "/aaa-skill": { + "name": "aaa-skill", + "description": "Alphabetically first", + "skill_md_path": f"{local_dir}/aaa-skill/SKILL.md", + "skill_dir": f"{local_dir}/aaa-skill", + }, + "/gym": { + "name": "gym", + "description": "GymPilot", + "skill_md_path": f"{local_dir}/gym/SKILL.md", + "skill_dir": f"{local_dir}/gym", + }, + } + core_count = len(telegram_bot_commands()) + + with ( + patch("agent.skill_commands.get_skill_commands", return_value=fake_cmds), + patch("tools.skills_tool.SKILLS_DIR", local_dir), + patch( + "hermes_cli.commands._telegram_effective_priority", + return_value=("gym",), + ), + ): + menu, hidden = telegram_menu_commands(max_commands=core_count) + + menu_names = [name for name, _description in menu] + assert len(menu_names) == core_count + assert menu_names[0] == "gym" + assert "aaa_skill" not in menu_names + assert hidden == 2 + + def test_scalar_configured_priority_is_accepted_as_one_command(self): + """The config CLI's scalar value form must work for a single priority.""" + from unittest.mock import patch + from hermes_cli.commands import _telegram_effective_priority + + raw_config = { + "platforms": { + "telegram": { + "extra": { + "command_menu": { + "priority": "gym", + "priority_mode": "prepend", + } + } + } + } + } + + with patch("hermes_cli.config.read_raw_config", return_value=raw_config): + priority = _telegram_effective_priority() + + assert priority[0] == "gym" + # --------------------------------------------------------------------------- # Backward-compat aliases From 21b503fb184b70783d68cf39416e8a5df9b32fab Mon Sep 17 00:00:00 2001 From: LOGIN-TB <7146963+LOGIN-TB@users.noreply.github.com> Date: Sun, 9 Aug 2026 14:52:39 +0200 Subject: [PATCH 032/117] fix(telegram): rank complete menu candidate set --- hermes_cli/commands.py | 196 +++++++++++++++++------------- tests/hermes_cli/test_commands.py | 149 +++++++++++++++++++++-- 2 files changed, 253 insertions(+), 92 deletions(-) diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index 943204ec5b..23af46886d 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -16,7 +16,7 @@ import re import shutil import subprocess import time -from collections.abc import Callable, Mapping +from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass, field from typing import Any, Dict, Optional, Tuple @@ -716,7 +716,7 @@ def _iter_plugin_command_entries() -> list[tuple[str, str, str]]: return entries -def telegram_bot_commands() -> list[tuple[str, str]]: +def telegram_bot_commands(*, include_plugins: bool = True) -> list[tuple[str, str]]: """Return (command_name, description) pairs for Telegram setMyCommands. Telegram command names cannot contain hyphens, so they are replaced with @@ -728,7 +728,9 @@ def telegram_bot_commands() -> list[tuple[str, str]]: without a payload, making them discoverable via autocomplete. Plugin-registered slash commands that require arguments are **excluded** - because plugins may not provide a no-arg usage fallback. + because plugins may not provide a no-arg usage fallback. Callers that need + source metadata can pass ``include_plugins=False`` and collect plugins via + :func:`_collect_gateway_skill_entries` instead. """ overrides = _resolve_config_gates() result: list[tuple[str, str]] = [] @@ -741,12 +743,13 @@ def telegram_bot_commands() -> list[tuple[str, str]]: tg_name = _sanitize_telegram_name(cmd.name) if tg_name: result.append((tg_name, cmd.description)) - for name, description, args_hint in _iter_plugin_command_entries(): - if _requires_argument(args_hint): - continue - tg_name = _sanitize_telegram_name(name) - if tg_name: - result.append((tg_name, description)) + if include_plugins: + for name, description, args_hint in _iter_plugin_command_entries(): + if _requires_argument(args_hint): + continue + tg_name = _sanitize_telegram_name(name) + if tg_name: + result.append((tg_name, description)) return result @@ -882,24 +885,54 @@ def _telegram_effective_priority() -> tuple[str, ...]: def _prioritize_telegram_menu_commands( commands: list[tuple[str, str]], ) -> list[tuple[str, str]]: - priority = { - name: index - for index, name in enumerate(_telegram_effective_priority()) - } + candidates = [(name, desc, "core", name) for name, desc in commands] + return [(name, desc) for name, desc, _source, _raw_name in _prioritize_telegram_menu_candidates(candidates)] + + +def _prioritize_telegram_menu_candidates( + candidates: list[tuple[str, str, str, str]], +) -> list[tuple[str, str, str, str]]: + """Order Telegram candidates while keeping default priority core-only. + + Candidate tuples contain ``(final_name, description, source, raw_name)``. + ``raw_name`` preserves the pre-clamp command name so an explicitly + configured long command remains addressable after Telegram name clamping. + """ + menu_cfg = _telegram_command_menu_config() + configured = _dedupe_sanitized_names(menu_cfg["priority"]) + defaults = _dedupe_sanitized_names(_TELEGRAM_MENU_PRIORITY) + configured_rank = {name: index for index, name in enumerate(configured)} + default_rank = {name: index for index, name in enumerate(defaults)} + priority_mode = menu_cfg["priority_mode"] + + def _rank(candidate: tuple[str, str, str, str], stable_index: int) -> tuple[int, int, int]: + final_name, _desc, source, raw_name = candidate + configured_index = configured_rank.get(raw_name) + if configured_index is None: + configured_index = configured_rank.get(final_name) + default_index = default_rank.get(final_name) if source == "core" else None + + if priority_mode == "replace": + if configured_index is not None: + return (0, configured_index, stable_index) + return (1, 0, stable_index) + if priority_mode == "append": + if default_index is not None: + return (0, default_index, stable_index) + if configured_index is not None: + return (1, configured_index, stable_index) + return (2, 0, stable_index) + if configured_index is not None: + return (0, configured_index, stable_index) + if default_index is not None: + return (1, default_index, stable_index) + return (2, 0, stable_index) + return [ - command - for _index, command in sorted( - enumerate(commands), - key=lambda item: ( - 0, - priority[item[1][0]], - item[0], - ) - if item[1][0] in priority - else ( - 1, - item[0], - ), + candidate + for stable_index, candidate in sorted( + enumerate(candidates), + key=lambda item: _rank(item[1], item[0]), ) ] @@ -931,7 +964,7 @@ def _sanitize_telegram_name(raw: str) -> str: def _clamp_command_names( - entries: list[tuple[str, ...]], + entries: Sequence[tuple[str, ...]], reserved: set[str], ) -> list[tuple[str, ...]]: """Enforce 32-char command name limit with collision avoidance. @@ -979,11 +1012,11 @@ _clamp_telegram_names = _clamp_command_names def _collect_gateway_skill_entries( platform: str, - max_slots: int, + max_slots: int | None, reserved_names: set[str], desc_limit: int = 100, sanitize_name: "Callable[[str], str] | None" = None, -) -> tuple[list[tuple[str, str, str]], int]: +) -> tuple[list[tuple[str, str, str, str]], int]: """Collect plugin + skill entries for a gateway platform. Priority order: @@ -998,7 +1031,8 @@ def _collect_gateway_skill_entries( platform: Platform identifier for per-platform skill filtering (``"telegram"``, ``"discord"``, etc.). max_slots: Maximum number of entries to return (remaining slots after - built-in/core commands). + built-in/core commands), or ``None`` to return every eligible + plugin and skill candidate for a caller that applies a global cap. reserved_names: Names already taken by built-in commands. Mutated in-place as new names are added. desc_limit: Max description length (40 for Telegram, 100 for Discord). @@ -1007,34 +1041,41 @@ def _collect_gateway_skill_entries( empty string to signal "skip this entry". Returns: - ``(entries, hidden_count)`` where *entries* is a list of - ``(name, description, cmd_key)`` triples and *hidden_count* is the - number of skill entries dropped due to the cap. ``cmd_key`` is the - original ``/skill-name`` key from :func:`get_skill_commands`. + ``(entries, hidden_count)`` where *entries* contains + ``(name, description, cmd_key, raw_name)`` tuples. ``cmd_key`` is the + original skill key (empty for plugins); ``raw_name`` is the sanitized + pre-clamp name used for configured priority matching. """ - all_entries: list[tuple[str, str, str]] = [] + all_entries: list[tuple[str, str, str, str]] = [] # --- Tier 1: Plugin slash commands (never trimmed) --------------------- - plugin_pairs: list[tuple[str, str]] = [] + plugin_pairs: list[tuple[str, str, str]] = [] try: from hermes_cli.plugins import get_plugin_commands plugin_cmds = get_plugin_commands() for cmd_name in sorted(plugin_cmds): + if platform == "telegram": + args_hint = str(plugin_cmds[cmd_name].get("args_hint") or "").strip() + if _requires_argument(args_hint): + continue name = sanitize_name(cmd_name) if sanitize_name else cmd_name if not name: continue desc = plugin_cmds[cmd_name].get("description", "Plugin command") if len(desc) > desc_limit: desc = desc[:desc_limit - 3] + "..." - plugin_pairs.append((name, desc)) + plugin_pairs.append((name, desc, name)) except Exception: pass - plugin_pairs = _clamp_command_names(plugin_pairs, reserved_names) - reserved_names.update(n for n, _ in plugin_pairs) - # Plugins have no cmd_key — use empty string as placeholder - for n, d in plugin_pairs: - all_entries.append((n, d, "")) + plugin_pairs = [ + (name, desc, raw_name) + for name, desc, raw_name in _clamp_command_names(plugin_pairs, reserved_names) + ] + reserved_names.update(n for n, _d, _raw_name in plugin_pairs) + # Plugins have no cmd_key — use empty string as placeholder. + for name, desc, raw_name in plugin_pairs: + all_entries.append((name, desc, "", raw_name)) # --- Tier 2: Built-in skill commands (trimmed at cap) ----------------- _platform_disabled: set[str] = set() @@ -1044,7 +1085,7 @@ def _collect_gateway_skill_entries( except Exception: pass - skill_triples: list[tuple[str, str, str]] = [] + skill_entries: list[tuple[str, str, str, str]] = [] try: from agent.skill_commands import get_skill_commands from tools.skills_tool import SKILLS_DIR @@ -1085,45 +1126,26 @@ def _collect_gateway_skill_entries( desc = info.get("description", "") if len(desc) > desc_limit: desc = desc[:desc_limit - 3] + "..." - skill_triples.append((name, desc, cmd_key)) + skill_entries.append((name, desc, cmd_key, name)) except Exception: pass - # Clamp names; cmd_key is passed through as extra payload so it survives - # any clamp-induced renames. - skill_triples = _clamp_command_names(skill_triples, reserved_names) + # Clamp names; cmd_key and raw_name survive any clamp-induced rename. + skill_entries = [ + (name, desc, cmd_key, raw_name) + for name, desc, cmd_key, raw_name in _clamp_command_names( + skill_entries, reserved_names + ) + ] - # Telegram's configured command-menu priority applies to dynamic skills as - # well as core commands. Reorder before trimming so a prioritized skill can - # claim a scarce remaining BotCommand slot instead of losing to the - # alphabetical default order. - if platform == "telegram": - priority = { - name: index - for index, name in enumerate(_telegram_effective_priority()) - } - skill_triples = [ - entry - for original_index, entry in sorted( - enumerate(skill_triples), - key=lambda item: ( - 0, - priority[item[1][0]], - item[0], - ) - if item[1][0] in priority - else ( - 1, - item[0], - ), - ) - ] + if max_slots is None: + return all_entries + skill_entries, 0 # Skills fill remaining slots — only tier that gets trimmed remaining = max(0, max_slots - len(all_entries)) - hidden_count = max(0, len(skill_triples) - remaining) - for n, d, k in skill_triples[:remaining]: - all_entries.append((n, d, k)) + hidden_count = max(0, len(skill_entries) - remaining) + for name, desc, cmd_key, raw_name in skill_entries[:remaining]: + all_entries.append((name, desc, cmd_key, raw_name)) return all_entries[:max_slots], hidden_count @@ -1154,21 +1176,24 @@ def telegram_menu_commands(max_commands: int = 100) -> tuple[list[tuple[str, str (menu_commands, hidden_count) where hidden_count is the number of commands omitted due to the cap. """ - core_commands = list(telegram_bot_commands()) + core_commands = list(telegram_bot_commands(include_plugins=False)) reserved_names = {n for n, _ in core_commands} entries, hidden_count = _collect_gateway_skill_entries( platform="telegram", - max_slots=max_commands, + max_slots=None, reserved_names=reserved_names, desc_limit=40, sanitize_name=_sanitize_telegram_name, ) - # Drop the cmd_key — Telegram only needs (name, desc) pairs. Apply the - # configured priority across all tiers before enforcing the global cap. - all_commands = core_commands + [(n, d) for n, d, _k in entries] - all_commands = _prioritize_telegram_menu_commands(all_commands) - overflow_count = max(0, len(all_commands) - max_commands) - return all_commands[:max_commands], hidden_count + overflow_count + candidates = [(name, desc, "core", name) for name, desc in core_commands] + for name, desc, cmd_key, raw_name in entries: + source = "skill" if cmd_key else "plugin" + candidates.append((name, desc, source, raw_name)) + + candidates = _prioritize_telegram_menu_candidates(candidates) + overflow_count = max(0, len(candidates) - max_commands) + menu = [(name, desc) for name, desc, _source, _raw_name in candidates[:max_commands]] + return menu, hidden_count + overflow_count def discord_skill_commands( @@ -1193,12 +1218,15 @@ def discord_skill_commands( ``(discord_name, description, cmd_key)`` triples. ``cmd_key`` is the original ``/skill-name`` key needed for the slash handler callback. """ - return _collect_gateway_skill_entries( + entries, hidden_count = _collect_gateway_skill_entries( platform="discord", max_slots=max_slots, reserved_names=set(reserved_names), # copy — don't mutate caller's set desc_limit=100, ) + return [ + (name, desc, cmd_key) for name, desc, cmd_key, _raw_name in entries + ], hidden_count def discord_skill_commands_by_category( diff --git a/tests/hermes_cli/test_commands.py b/tests/hermes_cli/test_commands.py index a4823bfc2c..9c70a7c6f3 100644 --- a/tests/hermes_cli/test_commands.py +++ b/tests/hermes_cli/test_commands.py @@ -830,23 +830,156 @@ class TestTelegramMenuCommands: "skill_dir": f"{local_dir}/gym", }, } - core_count = len(telegram_bot_commands()) + fake_plugins = { + "plugin-one": {"description": "Plugin one"}, + "plugin-two": {"description": "Plugin two"}, + } + fake_core = [ + ("core_one", "Core one"), + ("core_two", "Core two"), + ] + menu_cfg = {"max_commands": 2, "priority_mode": "prepend", "priority": ["gym"]} with ( + patch("hermes_cli.commands.telegram_bot_commands", return_value=fake_core), patch("agent.skill_commands.get_skill_commands", return_value=fake_cmds), + patch("hermes_cli.plugins.get_plugin_commands", return_value=fake_plugins), patch("tools.skills_tool.SKILLS_DIR", local_dir), - patch( - "hermes_cli.commands._telegram_effective_priority", - return_value=("gym",), - ), + patch("hermes_cli.commands._telegram_command_menu_config", return_value=menu_cfg), ): - menu, hidden = telegram_menu_commands(max_commands=core_count) + menu, hidden = telegram_menu_commands(max_commands=len(fake_core)) menu_names = [name for name, _description in menu] - assert len(menu_names) == core_count + assert len(menu_names) == len(fake_core) assert menu_names[0] == "gym" assert "aaa_skill" not in menu_names - assert hidden == 2 + assert hidden == 4 + + def test_default_core_priority_does_not_promote_same_named_skill( + self, tmp_path, monkeypatch + ): + """Built-in defaults must not elevate a dynamic command across tiers.""" + from unittest.mock import patch + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + local_dir = tmp_path / "skills" + local_dir.mkdir() + fake_cmds = { + "/aaa-skill": { + "name": "aaa-skill", + "description": "Alphabetically first skill", + "skill_md_path": f"{local_dir}/aaa-skill/SKILL.md", + "skill_dir": f"{local_dir}/aaa-skill", + }, + "/platforms": { + "name": "platforms", + "description": "Dynamic platforms skill", + "skill_md_path": f"{local_dir}/platforms/SKILL.md", + "skill_dir": f"{local_dir}/platforms", + }, + } + fake_core = [("core_one", "Core one")] + menu_cfg = {"max_commands": 2, "priority_mode": "prepend", "priority": []} + + with ( + patch("hermes_cli.commands.telegram_bot_commands", return_value=fake_core), + patch("agent.skill_commands.get_skill_commands", return_value=fake_cmds), + patch("tools.skills_tool.SKILLS_DIR", local_dir), + patch("hermes_cli.commands._telegram_command_menu_config", return_value=menu_cfg), + ): + menu, hidden = telegram_menu_commands(max_commands=2) + + assert menu == fake_core + [("aaa_skill", "Alphabetically first skill")] + assert hidden == 1 + + def test_long_configured_skill_priority_survives_telegram_clamping( + self, tmp_path, monkeypatch + ): + """Priority matches the original skill key after its menu name is clamped.""" + from unittest.mock import patch + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + local_dir = tmp_path / "skills" + local_dir.mkdir() + long_name = "x" * 40 + fake_cmds = { + f"/{long_name}": { + "name": long_name, + "description": "Long prioritized skill", + "skill_md_path": f"{local_dir}/{long_name}/SKILL.md", + "skill_dir": f"{local_dir}/{long_name}", + }, + } + fake_core = [("core_one", "Core one")] + menu_cfg = { + "max_commands": 1, + "priority_mode": "replace", + "priority": [long_name], + } + + with ( + patch("hermes_cli.commands.telegram_bot_commands", return_value=fake_core), + patch("agent.skill_commands.get_skill_commands", return_value=fake_cmds), + patch("tools.skills_tool.SKILLS_DIR", local_dir), + patch("hermes_cli.commands._telegram_command_menu_config", return_value=menu_cfg), + ): + menu, hidden = telegram_menu_commands(max_commands=1) + + assert menu == [("x" * 32, "Long prioritized skill")] + assert hidden == 1 + + def test_long_configured_plugin_priority_survives_telegram_clamping( + self, tmp_path, monkeypatch + ): + """Plugin priority also matches the original name after clamping.""" + from unittest.mock import patch + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + local_dir = tmp_path / "skills" + local_dir.mkdir() + long_name = "p" * 40 + fake_plugins = {long_name: {"description": "Long prioritized plugin"}} + menu_cfg = { + "max_commands": 1, + "priority_mode": "replace", + "priority": [long_name], + } + + with ( + patch("hermes_cli.plugins.get_plugin_commands", return_value=fake_plugins), + patch("agent.skill_commands.get_skill_commands", return_value={}), + patch("tools.skills_tool.SKILLS_DIR", local_dir), + patch("hermes_cli.commands._telegram_command_menu_config", return_value=menu_cfg), + ): + menu, hidden = telegram_menu_commands(max_commands=1) + + assert menu == [("p" * 32, "Long prioritized plugin")] + assert hidden > 0 + + def test_argument_requiring_plugin_is_excluded_from_telegram_menu( + self, tmp_path, monkeypatch + ): + """Telegram omits plugins that cannot be invoked without a payload.""" + from unittest.mock import patch + + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + local_dir = tmp_path / "skills" + local_dir.mkdir() + fake_plugins = { + "no-arg": {"description": "No argument"}, + "needs-arg": {"description": "Needs argument", "args_hint": ""}, + } + + with ( + patch("hermes_cli.plugins.get_plugin_commands", return_value=fake_plugins), + patch("agent.skill_commands.get_skill_commands", return_value={}), + patch("tools.skills_tool.SKILLS_DIR", local_dir), + ): + menu, _hidden = telegram_menu_commands(max_commands=100) + + menu_names = {name for name, _description in menu} + assert "no_arg" in menu_names + assert "needs_arg" not in menu_names def test_scalar_configured_priority_is_accepted_as_one_command(self): """The config CLI's scalar value form must work for a single priority.""" From 3e6229ec00ff2b53e92c343e0c270be75f0d7792 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:56:01 -0700 Subject: [PATCH 033/117] docs: priority list now guarantees skill commands a Telegram menu slot --- cli-config.yaml.example | 3 +++ website/docs/user-guide/messaging/telegram.md | 5 ++++- 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 5b3a73acd9..622f4fa766 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -1225,8 +1225,11 @@ platform_toolsets: # # append = Hermes defaults first, then user priority # # replace = only the list below defines priority # priority_mode: prepend +# # Priority is applied across core + plugin + skill commands before +# # the cap, so a listed skill command always keeps a menu slot. # priority: # - my_plugin_command +# - my-important-skill # slack: # extra: # # Render live tool calls as Slack-native plan/task cards. This explicit diff --git a/website/docs/user-guide/messaging/telegram.md b/website/docs/user-guide/messaging/telegram.md index 9aa51384bb..5f3dfb3e6d 100644 --- a/website/docs/user-guide/messaging/telegram.md +++ b/website/docs/user-guide/messaging/telegram.md @@ -83,7 +83,7 @@ Notes: Hermes registers its command menu automatically when the Telegram gateway starts. The menu is built from the central slash-command registry plus eligible plugin/skill commands, then capped so Telegram accepts the payload reliably. The default cap is 60 commands — enough to keep all built-in commands plus common skill commands visible. -If you have local or plugin commands that should stay visible in Telegram's `/` picker, prioritize them in `~/.hermes/config.yaml`: +If you have skill, plugin, or built-in commands that should stay visible in Telegram's `/` picker, prioritize them in `~/.hermes/config.yaml`: ```yaml platforms: @@ -94,6 +94,7 @@ platforms: priority_mode: prepend # prepend | append | replace priority: - my_plugin_command + - songsee # skill commands work here too ``` `priority_mode` controls how your list combines with Hermes' built-in priority list: @@ -102,6 +103,8 @@ platforms: - `append`: keep Hermes defaults first, then your commands - `replace`: use only your list for priority ordering +Priority is applied to the **combined** candidate list (core commands, plugin commands, and skill commands) before the cap is enforced — so a prioritized skill command is guaranteed a menu slot even when core commands alone would fill the menu. Previously skills were always trimmed first and alphabetically, so late-alphabet skills could never appear regardless of `priority`. + Telegram allows up to 100 BotCommands, but large command payloads can fail. Hermes defaults to 60 for reliability and clamps configured values to `1..100`; use `/commands` for the full command list. ## Step 3: Privacy Mode (Critical for Groups) From ab9d85287df8f544cedb3214a03b1e12f638c259 Mon Sep 17 00:00:00 2001 From: Drexuxux Date: Wed, 24 Jun 2026 03:46:34 +0300 Subject: [PATCH 034/117] fix(cron): accept documented "every