From 90e30bdd277786c368792f8d8542182224619f5e Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Wed, 19 Aug 2026 13:30:35 -0500 Subject: [PATCH 1/7] refactor(desktop): one script runner into the preview guest page, not a tour-only one --- .../src/app/chat/right-rail/preview-pane.tsx | 10 ++--- .../chat/right-rail/preview-script-runner.ts | 38 +++++++++++++++++++ .../chat/right-rail/preview-tour-runner.ts | 38 ------------------- .../src/app/chat/right-rail/preview-tour.ts | 4 +- 4 files changed, 45 insertions(+), 45 deletions(-) create mode 100644 apps/desktop/src/app/chat/right-rail/preview-script-runner.ts delete mode 100644 apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts diff --git a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx index 7d6150b2f3..e2f236cf2f 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx +++ b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx @@ -30,7 +30,7 @@ import { previewConsoleState } from './preview-console-store' import { LocalFilePreview, PreviewEmptyState } from './preview-file' import { PREVIEW_BROWSER_ATTR, registerPreviewNav } from './preview-nav' import { registerPreviewPageReader } from './preview-reader' -import { registerPreviewTourRunner } from './preview-tour-runner' +import { registerPreviewScriptRunner } from './preview-script-runner' type PreviewWebview = HTMLElement & { canGoBack?: () => boolean @@ -455,15 +455,15 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest = }) }, [isWebPreview, tabId]) - // Publish the TOUR runner for this tab (the tour tool, surface='preview'): - // runs injected driver.js actions inside the guest page so the agent can - // give guided walkthroughs of whatever web app is open here. + // Publish the SCRIPT runner for this tab: the one channel into the guest + // page, shared by the tour tool (injected driver.js walkthroughs) and the + // act_preview tool (clicking, typing, scrolling the page the user sees). useEffect(() => { if (!isWebPreview || !tabId) { return } - return registerPreviewTourRunner(tabId, async code => { + return registerPreviewScriptRunner(tabId, async code => { const webview = webviewRef.current if (!webview?.executeJavaScript) { diff --git a/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts b/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts new file mode 100644 index 0000000000..69a0efa2a4 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts @@ -0,0 +1,38 @@ +/** + * PREVIEW SCRIPT RUNNER REGISTRY — the one way anything in the app reaches into + * the preview pane's guest page, the script analog of preview-nav's handle + * registry. + * + * A live browser pane registers its webview's `executeJavaScript` here, keyed + * by tab id; `activePreviewScriptRunner` resolves the ACTIVE tab from the + * store. Both guest-page features ride it — the tour tool (preview-tour.ts) + * and the interaction tool (preview-act.ts) — so their heavy payloads stay out + * of the pane component's static import graph and only load when used. + */ + +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + +/** Runs JS source in the pane's guest page, resolving its completion value. */ +export type PreviewScriptRunner = (code: string) => Promise + +const runners = new Map() + +/** Register a live preview's script runner; returns an idempotent unregister. */ +export function registerPreviewScriptRunner(tabId: string, runner: PreviewScriptRunner): () => void { + runners.set(tabId, runner) + + return () => { + if (runners.get(tabId) === runner) { + runners.delete(tabId) + } + } +} + +/** The ACTIVE preview tab's script runner. Null = no live page behind it. */ +export function activePreviewScriptRunner(): PreviewScriptRunner | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && runners.get(tab.id)) || null +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts b/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts deleted file mode 100644 index 84a15c8869..0000000000 --- a/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts +++ /dev/null @@ -1,38 +0,0 @@ -/** - * PREVIEW TOUR RUNNER REGISTRY — the tour tool's script bridge into the - * preview pane, the tour analog of preview-nav's handle registry. - * - * A live browser pane registers a script runner (its webview's - * `executeJavaScript`) here, keyed by tab id; `activeTourRunner` resolves the - * ACTIVE tab from the store. Kept separate from preview-tour.ts so the pane - * component's static import stays tiny — the engine source + driver.js - * payload only load when a tour actually runs (gateway-event dynamic-imports - * preview-tour.ts). - */ - -import { $rightRailActiveTabId } from '@/store/layout' -import { $previewTabs } from '@/store/preview' - -/** Runs JS source in the pane's guest page, resolving its completion value. */ -export type TourScriptRunner = (code: string) => Promise - -const runners = new Map() - -/** Register a live preview's script runner; returns an idempotent unregister. */ -export function registerPreviewTourRunner(tabId: string, runner: TourScriptRunner): () => void { - runners.set(tabId, runner) - - return () => { - if (runners.get(tabId) === runner) { - runners.delete(tabId) - } - } -} - -/** The ACTIVE preview tab's script runner. Null = no live page to tour. */ -export function activeTourRunner(): TourScriptRunner | null { - const tabs = $previewTabs.get() - const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] - - return (tab && runners.get(tab.id)) || null -} diff --git a/apps/desktop/src/app/chat/right-rail/preview-tour.ts b/apps/desktop/src/app/chat/right-rail/preview-tour.ts index 12ad67dc71..a033a7c1f8 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-tour.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-tour.ts @@ -21,7 +21,7 @@ import driverIife from 'driver.js/dist/driver.js.iife.js?raw' import { collectTourTargets } from '@/lib/tour/collect-targets' import { runTourEngine, type TourAction, type TourResult } from '@/lib/tour/engine' -import { activeTourRunner } from './preview-tour-runner' +import { activePreviewScriptRunner } from './preview-script-runner' /** Build the idempotent inject-and-run script for one tour action. */ function buildTourScript(action: TourAction): string { @@ -51,7 +51,7 @@ function buildTourScript(action: TourAction): string { /** Run one tour action in the ACTIVE preview tab's page. */ export async function runPreviewTour(action: TourAction): Promise { - const run = activeTourRunner() + const run = activePreviewScriptRunner() if (!run) { return { error: 'No live page is open in the preview pane — open one first.', success: false } From c57581cd0d0de29292ad12c4218e4ae399fadcc6 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Wed, 19 Aug 2026 13:30:39 -0500 Subject: [PATCH 2/7] =?UTF-8?q?feat(tools):=20drive=5Fpreview=20and=20anno?= =?UTF-8?q?tate=5Fpreview=20=E2=80=94=20the=20agent=20can=20use=20the=20pa?= =?UTF-8?q?ge=20it=20opened?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The in-app browser was a one-way mirror. open_preview put a page in the pane and read_preview read its text back, but nothing could touch it. A click meant falling back to the browser_* tools, which drive a separate Chromium the user cannot see — so "log into this and pull my invoices" happened in a different browser from the one on screen, with none of the sessions the user is already signed into. Four pieces, and they only make sense together: · an in-page engine that inventories what is interactable and performs the verb, injected as source because it has to run inside the guest page; · the preview.act.request bridge from the gateway into the pane; · drive_preview, for acting: elements, click, type, scroll, press, and the pane's own back/forward/reload; · annotate_preview, for marking without acting. Those last two started as one tool doing two unrelated jobs. Leaving a mark is not an action — it outlives the turn that drew it — so it gets its own verb, and the interaction verb gets a name that says what it does. Gating is the existing surface rule: desktop_ui folds in on session source: 'desktop', and the bridge refuses to act for a background session, so a turn running behind the user's back cannot reach into the page they are working in. Two details worth a reviewer's attention. Typing assigns through the prototype's value setter, because React shadows value with its own accessor and ignores an input event whose value it believes it already wrote — a plain el.value = … types into a field that snaps back on the next render. And clicking replays the pointer/mouse pair before activation, because frameworks bind to mousedown as often as to click. --- agent/agent_init.py | 2 + agent/agent_runtime_helpers.py | 33 +- agent/tool_executor.py | 51 +++ .../app/chat/right-rail/preview-act.test.ts | 127 ++++++ .../src/app/chat/right-rail/preview-act.ts | 98 +++++ .../src/app/chat/right-rail/preview-nav.ts | 12 + .../gateway-event/desktop-bridge.ts | 45 ++ apps/desktop/src/lib/chat-messages/types.ts | 9 + .../src/lib/preview-act/act-in-page.test.ts | 319 ++++++++++++++ .../src/lib/preview-act/act-in-page.ts | 414 ++++++++++++++++++ run_agent.py | 2 + tests/run_agent/test_run_agent.py | 10 + tests/tools/test_annotate_preview_tool.py | 88 ++++ tests/tools/test_drive_preview_tool.py | 117 +++++ .../tui_gateway/test_gui_surface_toolsets.py | 2 + tools/annotate_preview_tool.py | 151 +++++++ tools/drive_preview_tool.py | 211 +++++++++ toolsets.py | 2 +- tui_gateway/methods_prompt.py | 9 + tui_gateway/server.py | 15 + website/docs/reference/tools-reference.md | 4 +- website/docs/reference/toolsets-reference.md | 2 +- 22 files changed, 1719 insertions(+), 4 deletions(-) create mode 100644 apps/desktop/src/app/chat/right-rail/preview-act.test.ts create mode 100644 apps/desktop/src/app/chat/right-rail/preview-act.ts create mode 100644 apps/desktop/src/lib/preview-act/act-in-page.test.ts create mode 100644 apps/desktop/src/lib/preview-act/act-in-page.ts create mode 100644 tests/tools/test_annotate_preview_tool.py create mode 100644 tests/tools/test_drive_preview_tool.py create mode 100644 tools/annotate_preview_tool.py create mode 100644 tools/drive_preview_tool.py diff --git a/agent/agent_init.py b/agent/agent_init.py index 4e097d6dfc..b839b35576 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -546,6 +546,7 @@ def init_agent( clarify_callback: callable = None, read_terminal_callback: callable = None, read_preview_callback: callable = None, + drive_preview_callback: callable = None, read_window_below_callback: callable = None, setup_mcp_callback: callable = None, tour_callback: callable = None, @@ -843,6 +844,7 @@ def init_agent( agent.clarify_callback = clarify_callback agent.read_terminal_callback = read_terminal_callback agent.read_preview_callback = read_preview_callback + agent.drive_preview_callback = drive_preview_callback agent.read_window_below_callback = read_window_below_callback agent.setup_mcp_callback = setup_mcp_callback agent.tour_callback = tour_callback diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 983c95b5d6..2744c1b72d 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -99,7 +99,7 @@ def _ra(): AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset( - {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} + {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} ) @@ -3234,6 +3234,37 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i ), next_args, ) + elif function_name == "drive_preview": + def _execute(next_args: dict) -> Any: + from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool + return _finish_agent_tool( + _drive_preview_tool( + action=next_args.get("action", ""), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + text=next_args.get("text"), + key=next_args.get("key"), + submit=next_args.get("submit"), + amount=next_args.get("amount"), + to=next_args.get("to"), + limit=next_args.get("max"), + callback=getattr(agent, "drive_preview_callback", None), + ), + next_args, + ) + elif function_name == "annotate_preview": + def _execute(next_args: dict) -> Any: + from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool + return _finish_agent_tool( + _annotate_preview_tool( + action=next_args.get("action", "add"), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + label=next_args.get("label"), + callback=getattr(agent, "drive_preview_callback", None), + ), + next_args, + ) elif function_name == "read_window_below": def _execute(next_args: dict) -> Any: from tools.read_window_tool import read_window_below_tool as _read_window_below_tool diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 4a828477c3..e7bb9126db 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2202,6 +2202,57 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('read_preview', function_args, tool_duration, result=function_result)}") + elif function_name == "drive_preview": + def _execute(next_args: dict) -> Any: + from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool + return _drive_preview_tool( + action=next_args.get("action", ""), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + text=next_args.get("text"), + key=next_args.get("key"), + submit=next_args.get("submit"), + amount=next_args.get("amount"), + to=next_args.get("to"), + limit=next_args.get("max"), + callback=getattr(agent, "drive_preview_callback", None), + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('drive_preview', function_args, tool_duration, result=function_result)}") + elif function_name == "annotate_preview": + def _execute(next_args: dict) -> Any: + from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool + return _annotate_preview_tool( + action=next_args.get("action", "add"), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + label=next_args.get("label"), + callback=getattr(agent, "drive_preview_callback", None), + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('annotate_preview', function_args, tool_duration, result=function_result)}") elif function_name == "read_window_below": def _execute(next_args: dict) -> Any: from tools.read_window_tool import read_window_below_tool as _read_window_below_tool diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.test.ts b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts new file mode 100644 index 0000000000..b53e046342 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts @@ -0,0 +1,127 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { $rightRailActiveTabId } from '@/store/layout' +import { closeRightRail, openPreview, type PreviewTarget } from '@/store/preview' + +import { actOnActivePreview } from './preview-act' +import { registerPreviewNav } from './preview-nav' +import { registerPreviewScriptRunner } from './preview-script-runner' + +function urlTarget(url: string): PreviewTarget { + return { kind: 'url', label: 'Browser', source: url, url } +} + +describe('actOnActivePreview (act_preview tool)', () => { + // URL targets share the singleton Browser tab id, so anything a test + // registers would answer the next one. + let cleanups: Array<() => void> = [] + + const openBrowserTab = () => { + openPreview(urlTarget('https://example.com'), 'tool-result') + + return $rightRailActiveTabId.get()! + } + + const withRunner = (runner: (code: string) => Promise) => + cleanups.push(registerPreviewScriptRunner(openBrowserTab(), runner)) + + beforeEach(() => { + vi.useRealTimers() + + for (const cleanup of cleanups) { + cleanup() + } + + cleanups = [] + closeRightRail() + window.localStorage.clear() + }) + + it('tells the agent to open a page when no live pane is behind the tab', async () => { + const result = await actOnActivePreview({ kind: 'elements' }) + + expect(result.success).toBe(false) + expect(result.error).toContain('open_preview') + }) + + it('injects the engine and returns the page’s answer', async () => { + let injected = '' + + withRunner(async code => { + injected = code + + return JSON.stringify({ acted: 'clicked button "Save"', success: true }) + }) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + expect(result).toMatchObject({ acted: 'clicked button "Save"', success: true }) + // Self-contained payload: the engine source and the action travel together, + // and the holder keeps refs alive across calls on the same page. + expect(injected).toContain('__hermesActHolder') + expect(injected).toContain('"ref":"@e1"') + }) + + it('re-inventories after a mutating action so the next ref is current', async () => { + const actions: string[] = [] + + withRunner(code => { + // Stand in for the guest page: run the script's own settle/rescan shape + // by answering each act() call in order. + actions.push(...(code.match(/"kind":"(\w+)"/g) ?? [])) + + return Promise.resolve( + JSON.stringify({ + acted: 'clicked', + elements: [{ label: 'Log out', ref: '@e1', role: 'button', selector: '#out' }], + success: true, + url: 'https://example.com/app' + }) + ) + }) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + expect(result.elements?.[0].label).toBe('Log out') + expect(result.url).toBe('https://example.com/app') + }) + + it('does not pay the settle delay for a plain inventory', async () => { + let injected = '' + withRunner(async code => { + injected = code + + return JSON.stringify({ elements: [], success: true }) + }) + + await actOnActivePreview({ kind: 'elements' }) + + expect(injected).toContain('0 <= 0') + }) + + it('reports a page that answers with nothing', async () => { + withRunner(async () => '') + + expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).error).toContain('did not answer') + }) + + it('routes history verbs to the pane instead of the guest page', async () => { + const back = vi.fn() + const runner = vi.fn() + + const tabId = openBrowserTab() + cleanups.push(registerPreviewNav(tabId, { back, forward: vi.fn(), reload: vi.fn() })) + cleanups.push(registerPreviewScriptRunner(tabId, runner)) + + const result = await actOnActivePreview({ kind: 'back' }) + + expect(back).toHaveBeenCalledOnce() + expect(runner).not.toHaveBeenCalled() + expect(result.success).toBe(true) + expect(result.note).toContain('elements') + }) + + it('reports history verbs with no pane to drive', async () => { + expect((await actOnActivePreview({ kind: 'reload' })).error).toContain('open_preview') + }) +}) diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.ts b/apps/desktop/src/app/chat/right-rail/preview-act.ts new file mode 100644 index 0000000000..495ce45f0a --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-act.ts @@ -0,0 +1,98 @@ +/** + * PREVIEW ACT — performs the agent's interactions inside the preview pane's + * guest page, so `act_preview` drives whatever web app is open in the in-app + * browser. + * + * The guest page is out-of-process; nothing here can touch its DOM directly. + * Instead each action injects the engine SOURCE over `executeJavaScript` (see + * lib/preview-act/act-in-page.ts's self-containment contract), parked on a + * window global alongside the ref holder so '@e5' still resolves on the next + * call. Injection is idempotent and vanishes with the page — a navigation + * drops the refs, which the engine reports rather than clicking the wrong node. + * + * A mutating action settles briefly and returns a fresh inventory, so the + * click → re-read → click loop costs one round trip instead of two. + * + * Dynamic-imported by the gateway event handler so the engine payload stays + * out of the boot path. + */ + +import { actInPage, type PreviewActAction, type PreviewActResult } from '@/lib/preview-act/act-in-page' + +import { activePreviewNav, type PreviewNavHandle } from './preview-nav' +import { activePreviewScriptRunner } from './preview-script-runner' + +/** Verbs the pane owns; a guest page cannot drive its own history. */ +const NAV_ACTIONS: readonly (keyof PreviewNavHandle)[] = ['back', 'forward', 'reload'] + +/** How long a click/type is given to land before the page is re-inventoried. + * Long enough for a framework re-render, short enough not to stall the turn. */ +const SETTLE_MS = 400 + +const NOTHING_OPEN = 'No live page is open in the in-app browser — open one with open_preview first.' + +/** Build the idempotent inject-and-run script for one action. */ +function buildActScript(action: PreviewActAction, settleMs: number): string { + return `(function () { + var w = window; + if (!w.__hermesAct) { + w.__hermesActHolder = {}; + w.__hermesAct = (${actInPage.toString()}); + } + var act = function (a) { return w.__hermesAct(document, w.__hermesActHolder, a); }; + var result = act(${JSON.stringify(action)}); + if (!result.success || ${settleMs} <= 0) { + return Promise.resolve(JSON.stringify(result)); + } + // Re-inventory after the page has had a moment to react, so the agent's next + // ref is drawn from the DOM its own click produced. + return new Promise(function (resolve) { + setTimeout(function () { + var after = act({ kind: 'elements' }); + result.elements = after.elements; + result.url = after.url; + result.title = after.title; + resolve(JSON.stringify(result)); + }, ${settleMs}); + }); +})()` +} + +/** Run one action against the ACTIVE preview tab's page. `kind` is a bare + * string: the verb arrives off the wire, and the history ones never reach + * the in-page engine. */ +export async function actOnActivePreview( + action: Omit & { kind: string } +): Promise { + const nav = NAV_ACTIONS.find(verb => verb === action.kind) + + if (nav) { + const handle = activePreviewNav() + + if (!handle) { + return { error: NOTHING_OPEN, success: false } + } + + handle[nav]() + + // Navigation is fire-and-forget through the webview; the new document has + // its own refs, so the agent has to re-inventory either way. + return { acted: nav, note: 'Page is loading — call elements to see what is on it.', success: true } + } + + const run = activePreviewScriptRunner() + + if (!run) { + return { error: NOTHING_OPEN, success: false } + } + + const typed = action as PreviewActAction + const settle = typed.kind === 'elements' ? 0 : SETTLE_MS + const raw = await run(buildActScript(typed, settle)) + + if (typeof raw !== 'string' || !raw) { + return { error: 'The page did not answer the action.', success: false } + } + + return JSON.parse(raw) as PreviewActResult +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-nav.ts b/apps/desktop/src/app/chat/right-rail/preview-nav.ts index 954c1a5e9b..5e03db3005 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-nav.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-nav.ts @@ -9,6 +9,9 @@ * sitting in Hermes' own DOM, where `activeElement` is authoritative. */ +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + /** Marks a live browser pane so a gesture can find the one holding focus. */ export const PREVIEW_BROWSER_ATTR = 'data-preview-browser' @@ -31,6 +34,15 @@ export function registerPreviewNav(tabId: string, handle: PreviewNavHandle): () } } +/** The ACTIVE preview tab's commands, for callers with no focus to key off — + * the agent's drive_preview, which runs while focus is in the composer. */ +export function activePreviewNav(): PreviewNavHandle | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && handles.get(tab.id)) || null +} + /** Run `command` on the browser pane holding DOM focus. False = focus is * elsewhere in the app, so the caller falls back to the app-level meaning. */ export function commandFocusedPreview(command: keyof PreviewNavHandle): boolean { diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts index dbb0508510..1a8cee9d02 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts @@ -2,6 +2,7 @@ import { readActivePreview } from '@/app/chat/right-rail/preview-reader' import { writeAgentTerminalChunk } from '@/app/right-sidebar/terminal/agent-terminal-stream' import { readActiveTerminal } from '@/app/right-sidebar/terminal/buffer' import { closeAgentTerminalByProc } from '@/app/right-sidebar/terminal/terminals' +import type { PreviewActAction } from '@/lib/preview-act/act-in-page' import type { TourAction, TourStep } from '@/lib/tour' import { $gateway } from '@/store/gateway' import { applyDesktopLayoutPreset, revealDesktopPane } from '@/store/pane-focus' @@ -55,6 +56,50 @@ export function handleDesktopBridgeEvent(ctx: GatewayEventContext): boolean { return true } + if (event.type === 'preview.act.request') { + // act_preview tool: click/type/scroll/press inside the guest page, or + // drive the pane's history. Dynamic import keeps the injected engine off + // the boot path. Active session only: a background turn must never reach + // into the page the user is working in (desktop AGENTS.md: offer, don't + // hijack). + const requestId = typeof payload?.request_id === 'string' ? payload.request_id : '' + + if (requestId) { + const answer = (result: unknown) => + $gateway.get()?.request('preview.act.respond', { + request_id: requestId, + text: result ? JSON.stringify(result) : '' + }) + + if (isActiveEvent) { + void import('@/app/chat/right-rail/preview-act') + .then(({ actOnActivePreview }) => + actOnActivePreview({ + amount: payload?.amount, + key: payload?.key, + kind: payload?.action ?? '', + max: payload?.max, + ref: payload?.ref, + selector: payload?.selector, + submit: payload?.submit, + text: payload?.text, + to: payload?.to as PreviewActAction['to'] + }) + ) + .then(answer, error => + answer({ error: error instanceof Error ? error.message : String(error), success: false }) + ) + } else { + void answer({ + error: 'The in-app browser only takes actions in the session the user is looking at.', + success: false + }) + } + } + + return true + } + if (event.type === 'window.read.request') { // read_window_below tool: main owns native window enumeration, so ask // it over IPC and answer. Empty text = unavailable (no bridge, or diff --git a/apps/desktop/src/lib/chat-messages/types.ts b/apps/desktop/src/lib/chat-messages/types.ts index 2974b8d56a..0706a73085 100644 --- a/apps/desktop/src/lib/chat-messages/types.ts +++ b/apps/desktop/src/lib/chat-messages/types.ts @@ -121,6 +121,15 @@ export type GatewayEventPayload = { side?: string steps?: unknown step_index?: number + // preview.act.request (drive_preview tool — agent clicking/typing/scrolling in + // the in-app browser). `action` names the verb and `selector` is shared with + // tour above; `ref` addresses an element from the last inventory. + ref?: string + submit?: boolean + key?: string + amount?: number + to?: string + max?: number // message.reaction (agent reacting via the react_to_message tool) — the // durable messages.id, that row's full reaction list after the write, and // the row's role so a live (not-yet-round-tripped) message can be matched. diff --git a/apps/desktop/src/lib/preview-act/act-in-page.test.ts b/apps/desktop/src/lib/preview-act/act-in-page.test.ts new file mode 100644 index 0000000000..aa1985fb7a --- /dev/null +++ b/apps/desktop/src/lib/preview-act/act-in-page.test.ts @@ -0,0 +1,319 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { actInPage, type PreviewActHolder } from './act-in-page' + +/** jsdom lays nothing out, so every rect is 0×0 and the engine's visibility + * check would reject the whole page. Give elements a plausible box and let + * `display: none` (which jsdom DOES compute) carry the hiding. */ +function layOutTheDocument() { + vi.spyOn(Element.prototype, 'getBoundingClientRect').mockImplementation(function (this: Element) { + const hidden = getComputedStyle(this).display === 'none' + const size = hidden ? 0 : 40 + + return { bottom: size, height: size, left: 0, right: size, top: 0, width: size, x: 0, y: 0 } as DOMRect + }) +} + +function page(html: string): PreviewActHolder { + document.body.innerHTML = html + + return {} +} + +/** Take the inventory the agent would take before acting. */ +function inventory(holder: PreviewActHolder) { + return actInPage(document, holder, { kind: 'elements' }) +} + +beforeEach(() => { + vi.restoreAllMocks() + layOutTheDocument() + Element.prototype.scrollIntoView = vi.fn() +}) + +describe('elements', () => { + it('numbers the interactive nodes with browser_*-style refs', () => { + const holder = page(` + + Help + +

Not interactive

+ `) + + const result = inventory(holder) + + expect(result.success).toBe(true) + expect(result.elements?.map(e => [e.ref, e.label])).toEqual([ + ['@e1', 'Save'], + ['@e2', 'Help'], + ['@e3', 'Your name'] + ]) + }) + + it('reports role, current value, and disabled state', () => { + const holder = page(` + + + `) + + const [email, go] = inventory(holder).elements! + + expect(email).toMatchObject({ label: 'Email', role: 'input:email', value: 'a@b.co' }) + expect(go).toMatchObject({ disabled: true, label: 'Go' }) + expect(email.disabled).toBeUndefined() + }) + + it('skips hidden controls and unlabelled ones', () => { + const holder = page(` + + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Real']) + }) + + it('prefers an identity selector so the agent can re-find the node later', () => { + const holder = page(` +
+
+ `) + + const [byTestId, positional] = inventory(holder).elements! + + expect(byTestId.selector).toBe('[data-testid="submit"]') + expect(document.querySelector(positional.selector)).toBe(document.querySelectorAll('button')[1]) + }) + + it('honours the cap', () => { + const holder = page(Array.from({ length: 10 }, (_, i) => ``).join('')) + + expect(inventory(holder).elements).toHaveLength(10) + expect(actInPage(document, holder, { kind: 'elements', max: 3 }).elements).toHaveLength(3) + }) +}) + +describe('click', () => { + it('activates the element a ref points at', () => { + const holder = page('') + inventory(holder) + + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + const result = actInPage(document, holder, { kind: 'click', ref: '@e1' }) + + expect(result.success).toBe(true) + expect(result.acted).toContain('Save') + expect(clicked).toHaveBeenCalledOnce() + }) + + it('replays the pointer/mouse sequence frameworks bind to', () => { + const holder = page('') + inventory(holder) + + const seen: string[] = [] + + for (const type of ['pointerdown', 'mousedown', 'mouseup', 'pointerup', 'click']) { + document.getElementById('save')!.addEventListener(type, () => seen.push(type)) + } + + actInPage(document, holder, { kind: 'click', ref: '@e1' }) + + expect(seen).toContain('mousedown') + expect(seen).toContain('mouseup') + expect(seen.at(-1)).toBe('click') + }) + + it('takes a raw CSS selector when no ref fits', () => { + const holder = page('') + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + expect(actInPage(document, holder, { kind: 'click', selector: '#save' }).success).toBe(true) + expect(clicked).toHaveBeenCalledOnce() + }) + + it('refuses a disabled control instead of silently doing nothing', () => { + const holder = page('') + inventory(holder) + + const result = actInPage(document, holder, { kind: 'click', ref: '@e1' }) + + expect(result.success).toBe(false) + expect(result.error).toContain('disabled') + }) + + it('reports the live url so a navigation is visible to the agent', () => { + const holder = page('') + inventory(holder) + + expect(actInPage(document, holder, { kind: 'click', ref: '@e1' }).url).toBe(document.location.href) + }) +}) + +describe('stale refs', () => { + it('names an unknown ref rather than clicking whatever sits at that index', () => { + const holder = page('') + inventory(holder) + + const result = actInPage(document, holder, { kind: 'click', ref: '@e9' }) + + expect(result.success).toBe(false) + expect(result.error).toContain('elements') + }) + + it('catches a node that was removed after the snapshot', () => { + const holder = page('') + inventory(holder) + document.getElementById('save')!.remove() + + expect(actInPage(document, holder, { kind: 'click', ref: '@e1' }).error).toContain('removed') + }) + + it('invalidates every ref when the page navigated under them', () => { + const holder = page('') + inventory(holder) + holder.url = 'https://elsewhere.example/other' + + expect(actInPage(document, holder, { kind: 'click', ref: '@e1' }).error).toContain('navigated') + }) + + it('asks for a target when given neither', () => { + expect(actInPage(document, page(''), { kind: 'click' }).error).toContain('selector') + }) + + it('reports a selector that matches nothing', () => { + expect(actInPage(document, page(''), { kind: 'click', selector: '#nope' }).error).toContain('No element') + }) +}) + +describe('type', () => { + it('enters text and fires the events a controlled input listens for', () => { + const holder = page('') + inventory(holder) + + const input = document.getElementById('who') as HTMLInputElement + const events: string[] = [] + input.addEventListener('input', () => events.push('input')) + input.addEventListener('change', () => events.push('change')) + + const result = actInPage(document, holder, { kind: 'type', ref: '@e1', text: 'Brooklyn' }) + + expect(result.success).toBe(true) + expect(input.value).toBe('Brooklyn') + expect(events).toEqual(['input', 'change']) + }) + + it('bypasses the own-property shadow React installs on tracked inputs', () => { + const holder = page('') + inventory(holder) + + // React defines its own `value` accessor on the node to track what it last + // wrote, and ignores an input event that agrees with it. Writing through + // that shadow is exactly how typed text snaps back on the next render. + const input = document.getElementById('who') as HTMLInputElement + const nativeValue = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')! + const shadowWrites: string[] = [] + + Object.defineProperty(input, 'value', { + configurable: true, + get: () => nativeValue.get!.call(input), + set: (next: string) => { + shadowWrites.push(next) + } + }) + + actInPage(document, holder, { kind: 'type', ref: '@e1', text: 'Brooklyn' }) + + expect(shadowWrites).toEqual([]) + expect(nativeValue.get!.call(input)).toBe('Brooklyn') + }) + + it('writes into a contenteditable host', () => { + const holder = page('
') + inventory(holder) + + actInPage(document, holder, { kind: 'type', ref: '@e1', text: 'hello' }) + + expect(document.getElementById('editor')!.textContent).toBe('hello') + }) + + it('submits the owning form when asked', () => { + const holder = page('
') + inventory(holder) + + const form = document.getElementById('f') as HTMLFormElement + form.requestSubmit = vi.fn() + + const result = actInPage(document, holder, { kind: 'type', ref: '@e1', submit: true, text: 'cats' }) + + expect(form.requestSubmit).toHaveBeenCalledOnce() + expect(result.acted).toContain('submitted') + }) + + it('refuses a target that has no text to type into', () => { + const holder = page('') + inventory(holder) + + expect(actInPage(document, holder, { kind: 'type', ref: '@e1', text: 'x' }).error).toContain('not a text field') + }) +}) + +describe('press', () => { + it('sends the key to the target', () => { + const holder = page('') + inventory(holder) + + const keys: string[] = [] + document.getElementById('q')!.addEventListener('keydown', e => keys.push((e as KeyboardEvent).key)) + + expect(actInPage(document, holder, { key: 'Enter', kind: 'press', ref: '@e1' }).success).toBe(true) + expect(keys).toEqual(['Enter']) + }) + + it('needs a key', () => { + const holder = page('') + inventory(holder) + + expect(actInPage(document, holder, { kind: 'press', ref: '@e1' }).error).toContain('key') + }) +}) + +describe('scroll', () => { + it('scrolls the page by about a screen when given no distance', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + const result = actInPage(document, holder, { kind: 'scroll' }) + + expect(result.success).toBe(true) + expect(scrollBy).toHaveBeenCalledWith({ behavior: 'auto', top: Math.round(window.innerHeight * 0.9) }) + }) + + it('jumps to the bottom', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + actInPage(document, holder, { kind: 'scroll', to: 'bottom' }) + + const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions] + + expect(options.top).toBeGreaterThan(window.innerHeight) + }) + + it('scrolls a ref’d container instead of the page', () => { + const holder = page('
') + inventory(holder) + + const list = document.getElementById('list') as HTMLElement + list.scrollBy = vi.fn() + const pageScroll = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + const result = actInPage(document, holder, { amount: 200, kind: 'scroll', ref: '@e1' }) + + expect(list.scrollBy).toHaveBeenCalledWith({ behavior: 'auto', top: 200 }) + expect(pageScroll).not.toHaveBeenCalled() + expect(result.acted).toContain('Results') + }) +}) diff --git a/apps/desktop/src/lib/preview-act/act-in-page.ts b/apps/desktop/src/lib/preview-act/act-in-page.ts new file mode 100644 index 0000000000..f664f58299 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/act-in-page.ts @@ -0,0 +1,414 @@ +/** + * PREVIEW ACT ENGINE — the one function that performs an interaction inside a + * page, so the agent can drive the in-app browser instead of only reading it. + * + * `elements` hands back a numbered inventory of what can be interacted with + * (`@e1`, `@e2`, … — the same ref shape the browser_* tools use, so a model + * that knows one knows the other) and parks the matching nodes on a holder. + * Every other verb resolves its target from that holder, falling back to a raw + * CSS selector. + * + * Injected into the preview webview as source (`actInPage.toString()`), so it + * MUST stay self-contained: no imports, no closure references, no renderer + * globals — everything arrives as a parameter. The structural types below + * erase at compile time, so the stringified function stays plain JS. + */ + +/** One interactable node, as the agent sees it. */ +export interface PreviewElement { + /** Present and true only when the control is non-interactive right now. */ + disabled?: boolean + /** Human-readable label (aria-label, text, placeholder, value …). */ + label: string + /** Stable-for-this-snapshot handle: '@e1', '@e2', … */ + ref: string + /** Explicit ARIA role, else the tag name. */ + role: string + /** CSS selector that resolves back to this node, for re-finding it later. */ + selector: string + /** Current value of a form control, truncated. */ + value?: string +} + +/** A normalized action. `kind` is the verb; the rest is per-verb payload. */ +export interface PreviewActAction { + /** scroll distance in px. Defaults to ~90% of the viewport height. */ + amount?: number + key?: string + kind: 'click' | 'elements' | 'press' | 'scroll' | 'type' + /** Cap on the returned inventory. */ + max?: number + ref?: string + selector?: string + /** type: press Enter (and submit the owning form) after entering text. */ + submit?: boolean + text?: string + to?: 'bottom' | 'top' +} + +export interface PreviewActResult { + /** What the action landed on, for the agent's own log. */ + acted?: string + elements?: PreviewElement[] + error?: string + note?: string + success: boolean + title?: string + /** Live document URL after the action — a change means it navigated. */ + url?: string +} + +/** Where the surface keeps the last snapshot between actions (a window global + * in the preview page), so '@e5' still means something on the next call. */ +export interface PreviewActHolder { + nodes?: Element[] + /** URL the snapshot was taken on; a navigation invalidates every ref. */ + url?: string +} + +const ACT_MAX_ELEMENTS = 120 + +/** Run one action against `doc`, resolving refs through `holder`. Self-contained. */ +export function actInPage(doc: Document, holder: PreviewActHolder, action: PreviewActAction): PreviewActResult { + const win = doc.defaultView + const here = doc.location ? doc.location.href : '' + + const cssEscape = (value: string) => + typeof CSS !== 'undefined' && CSS.escape ? CSS.escape(value) : value.replace(/["\\]/g, '\\$&') + + const clamp = (text: string, max: number) => (text.length > max ? text.slice(0, max - 1) + '…' : text) + + const labelOf = (el: Element): string => { + const aria = el.getAttribute('aria-label') + + if (aria) { + return clamp(aria, 80) + } + + const labelledBy = el.getAttribute('aria-labelledby') + const labelled = labelledBy ? doc.getElementById(labelledBy) : null + const text = ((labelled || el).textContent || '').trim().replace(/\s+/g, ' ') + + if (text) { + return clamp(text, 80) + } + + for (const attr of ['placeholder', 'title', 'alt', 'name', 'value']) { + const value = el.getAttribute(attr) + + if (value) { + return clamp(value, 80) + } + } + + return '' + } + + /** Identity-first selector, positional fallback — the caller re-finds nodes + * with this once the refs have gone stale. */ + const selectorFor = (el: Element): string => { + if (el.id) { + return '#' + cssEscape(el.id) + } + + const testId = el.getAttribute('data-testid') + + if (testId) { + return '[data-testid="' + cssEscape(testId) + '"]' + } + + const path: string[] = [] + let node: Element | null = el + + while (node && node !== doc.body && path.length < 8) { + if (node.id) { + path.unshift('#' + cssEscape(node.id)) + + break + } + + const parent: Element | null = node.parentElement + const index = parent ? Array.prototype.indexOf.call(parent.children, node) : -1 + + path.unshift(node.tagName.toLowerCase() + (index >= 0 ? ':nth-child(' + (index + 1) + ')' : '')) + node = parent + } + + return path.join(' > ') + } + + const visible = (el: Element): boolean => { + const style = win && win.getComputedStyle ? win.getComputedStyle(el) : null + + if (style && (style.display === 'none' || style.visibility === 'hidden' || style.opacity === '0')) { + return false + } + + const rect = el.getBoundingClientRect() + + // Off-screen is fine (we scroll to it); collapsed to nothing is not. + return rect.width >= 1 && rect.height >= 1 + } + + const valueOf = (el: Element): string => { + const control = el as HTMLInputElement + + if (typeof control.value === 'string' && control.value) { + return clamp(control.value, 60) + } + + if (control.checked !== undefined && (el.tagName === 'INPUT' || el.getAttribute('role') === 'checkbox')) { + return control.checked ? 'checked' : 'unchecked' + } + + return '' + } + + const collect = (max: number): PreviewElement[] => { + const nodes: Element[] = [] + const elements: PreviewElement[] = [] + + const candidates = doc.querySelectorAll( + 'a[href], button, input:not([type="hidden"]), select, textarea, summary, label[for], ' + + '[role="button"], [role="link"], [role="checkbox"], [role="radio"], [role="tab"], ' + + '[role="menuitem"], [role="switch"], [role="option"], [role="combobox"], [role="searchbox"], ' + + '[role="textbox"], [contenteditable=""], [contenteditable="true"], [onclick], ' + + '[tabindex]:not([tabindex="-1"])' + ) + + for (const el of candidates) { + if (elements.length >= max || nodes.indexOf(el) !== -1 || !visible(el)) { + continue + } + + const tag = el.tagName.toLowerCase() + const role = el.getAttribute('role') || (tag === 'input' ? 'input:' + ((el as HTMLInputElement).type || 'text') : tag) + const label = labelOf(el) + const value = valueOf(el) + + // A control with neither a label nor a value is not addressable in prose + // — the agent could not tell it apart from its unlabelled neighbours. + if (!label && !value) { + continue + } + + const entry: PreviewElement = { + label, + ref: '@e' + (elements.length + 1), + role, + selector: selectorFor(el) + } + + if ((el as HTMLInputElement).disabled) { + entry.disabled = true + } + + if (value) { + entry.value = value + } + + nodes.push(el) + elements.push(entry) + } + + holder.nodes = nodes + holder.url = here + + return elements + } + + /** Resolve the action's target: a ref from the last snapshot, else a selector. */ + const resolve = (): { el?: Element; error?: string } => { + const ref = (action.ref || '').trim() + + if (ref) { + if (holder.url !== here) { + return { error: 'The page navigated since the last snapshot, so ' + ref + ' no longer points anywhere. Call elements again.' } + } + + const index = Number(ref.replace(/^@e/, '')) - 1 + const el = holder.nodes && holder.nodes[index] + + if (!el) { + return { error: 'Unknown element ' + ref + '. Call elements to get current refs.' } + } + + if (!doc.contains(el)) { + return { error: ref + ' has been removed from the page since the last snapshot. Call elements again.' } + } + + return { el } + } + + const selector = (action.selector || '').trim() + + if (!selector) { + return { error: 'Pass a ref from elements, or a CSS selector.' } + } + + let el: Element | null = null + + try { + el = doc.querySelector(selector) + } catch { + return { error: 'Not a valid CSS selector: ' + selector } + } + + return el ? { el } : { error: 'No element matches ' + selector + '.' } + } + + const describe = (el: Element) => { + const label = labelOf(el) + + return el.tagName.toLowerCase() + (label ? ' "' + label + '"' : '') + } + + const answer = (result: PreviewActResult): PreviewActResult => ({ + ...result, + title: doc.title || '', + url: doc.location ? doc.location.href : '' + }) + + const fail = (error: string) => answer({ error, success: false }) + + /** Center of `el` in viewport coords, for pointer events that read position. */ + const pointAt = (el: Element) => { + const rect = el.getBoundingClientRect() + + return { clientX: Math.round(rect.left + rect.width / 2), clientY: Math.round(rect.top + rect.height / 2) } + } + + if (action.kind === 'elements') { + const elements = collect(Math.max(1, Math.min(action.max || ACT_MAX_ELEMENTS, ACT_MAX_ELEMENTS))) + + return answer({ + elements, + note: elements.length ? undefined : 'No interactive elements found — the page may still be loading.', + success: true + }) + } + + if (action.kind === 'scroll') { + const target = action.ref || action.selector ? resolve() : {} + + if (target.error) { + return fail(target.error) + } + + const scroller = (target.el as HTMLElement | undefined) || null + const page = Math.round((win ? win.innerHeight : 800) * 0.9) + const by = action.to === 'top' ? -1e7 : action.to === 'bottom' ? 1e7 : (action.amount ?? page) + + if (scroller) { + scroller.scrollBy({ behavior: 'auto', top: by }) + } else if (win) { + win.scrollBy({ behavior: 'auto', top: by }) + } + + return answer({ acted: scroller ? 'scrolled ' + describe(scroller) : 'scrolled the page', success: true }) + } + + const target = resolve() + + if (target.error || !target.el) { + return fail(target.error || 'No target.') + } + + const el = target.el as HTMLElement + + if ((el as HTMLInputElement).disabled) { + return fail(describe(el) + ' is disabled.') + } + + if (el.scrollIntoView) { + el.scrollIntoView({ block: 'center', inline: 'nearest' }) + } + + if (action.kind === 'click') { + const init = { bubbles: true, cancelable: true, ...pointAt(el) } + + // Frameworks bind to the pointer/mouse pair as often as to click itself, so + // replay the whole sequence; el.click() then runs the native activation + // (following a link, toggling a checkbox, submitting a form) that a bare + // synthetic MouseEvent would leave to chance. + if (typeof PointerEvent === 'function') { + el.dispatchEvent(new PointerEvent('pointerdown', init)) + el.dispatchEvent(new PointerEvent('pointerup', init)) + } + + el.dispatchEvent(new MouseEvent('mousedown', init)) + el.dispatchEvent(new MouseEvent('mouseup', init)) + el.click() + + return answer({ acted: 'clicked ' + describe(el), success: true }) + } + + if (action.kind === 'type') { + const text = action.text ?? '' + + const editable = el.isContentEditable || (el.getAttribute('contenteditable') ?? 'false') !== 'false' + + el.focus() + + if (editable) { + el.textContent = text + } else if (el.tagName === 'INPUT' || el.tagName === 'TEXTAREA') { + // Assign through the prototype's setter: React (and anything else that + // tracks the DOM value) shadows `value` with its own accessor and ignores + // an input event whose value it thinks it already wrote, so a plain + // `el.value = …` types into a field that snaps back on the next render. + const proto = el.tagName === 'TEXTAREA' ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype + const setter = Object.getOwnPropertyDescriptor(proto, 'value')?.set + + if (setter) { + setter.call(el, text) + } else { + ;(el as HTMLInputElement).value = text + } + } else if (el.tagName === 'SELECT') { + return fail(describe(el) + ' is a dropdown — click it and click the option you want.') + } else { + return fail(describe(el) + ' is not a text field.') + } + + el.dispatchEvent(new Event('input', { bubbles: true })) + el.dispatchEvent(new Event('change', { bubbles: true })) + + if (action.submit) { + const enter = { bubbles: true, cancelable: true, code: 'Enter', key: 'Enter' } + + el.dispatchEvent(new KeyboardEvent('keydown', enter)) + el.dispatchEvent(new KeyboardEvent('keyup', enter)) + + const form = (el as HTMLInputElement).form + + if (form) { + if (form.requestSubmit) { + form.requestSubmit() + } else { + form.submit() + } + } + } + + return answer({ acted: 'typed into ' + describe(el) + (action.submit ? ' and submitted' : ''), success: true }) + } + + if (action.kind === 'press') { + const key = action.key || '' + + if (!key) { + return fail('Pass the key to press, e.g. "Enter" or "Escape".') + } + + const init = { bubbles: true, cancelable: true, code: key.length === 1 ? 'Key' + key.toUpperCase() : key, key } + + el.focus() + el.dispatchEvent(new KeyboardEvent('keydown', init)) + el.dispatchEvent(new KeyboardEvent('keyup', init)) + + return answer({ acted: 'pressed ' + key + ' on ' + describe(el), success: true }) + } + + return fail('Unknown action: ' + String(action.kind)) +} diff --git a/run_agent.py b/run_agent.py index a54bc9cf8f..d3588f8f48 100644 --- a/run_agent.py +++ b/run_agent.py @@ -470,6 +470,7 @@ class AIAgent: clarify_callback: callable = None, read_terminal_callback: callable = None, read_preview_callback: callable = None, + drive_preview_callback: callable = None, read_window_below_callback: callable = None, setup_mcp_callback: callable = None, tour_callback: callable = None, @@ -560,6 +561,7 @@ class AIAgent: clarify_callback=clarify_callback, read_terminal_callback=read_terminal_callback, read_preview_callback=read_preview_callback, + drive_preview_callback=drive_preview_callback, read_window_below_callback=read_window_below_callback, setup_mcp_callback=setup_mcp_callback, tour_callback=tour_callback, diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index a72da553be..4688e125b6 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -2421,6 +2421,8 @@ class TestAgentRuntimePostHookOwnershipSync: ("clarify", {"question": "Continue?"}), ("read_terminal", {}), ("read_preview", {}), + ("drive_preview", {"action": "elements"}), + ("annotate_preview", {"action": "clear"}), ("read_window_below", {}), ("setup_mcp", {"server": "linear", "action": "install"}), ("tour", {"action": "stop"}), @@ -2467,6 +2469,14 @@ class TestAgentRuntimePostHookOwnershipSync: "tools.read_preview_tool.read_preview_tool", lambda **kwargs: '{"ok":true}', ) + monkeypatch.setattr( + "tools.drive_preview_tool.drive_preview_tool", + lambda **kwargs: '{"ok":true}', + ) + monkeypatch.setattr( + "tools.annotate_preview_tool.annotate_preview_tool", + lambda **kwargs: '{"ok":true}', + ) monkeypatch.setattr( "tools.read_window_tool.read_window_below_tool", lambda **kwargs: '{"ok":true}', diff --git a/tests/tools/test_annotate_preview_tool.py b/tests/tools/test_annotate_preview_tool.py new file mode 100644 index 0000000000..bb6621a1c8 --- /dev/null +++ b/tests/tools/test_annotate_preview_tool.py @@ -0,0 +1,88 @@ +"""Tests for the GUI-surface ``annotate_preview`` tool.""" + +import json + +import pytest + +from tools import annotate_preview_tool as an +from tools.registry import registry + + +def test_lives_in_the_gui_surface_toolset(monkeypatch): + """Scoped by toolset, not by the backend's env — same as its siblings.""" + monkeypatch.delenv("HERMES_DESKTOP", raising=False) + entry = registry.get_entry("annotate_preview") + + assert entry is not None + assert entry.toolset == "desktop_ui" + assert entry.check_fn is None + + +def test_requires_callback(): + result = json.loads(an.annotate_preview_tool(ref="@e1", callback=None)) + assert "desktop" in result["error"] + + +def test_rejects_an_unknown_action(): + result = json.loads(an.annotate_preview_tool(action="scribble", callback=lambda _p: "{}")) + assert "action must be one of" in result["error"] + + +@pytest.mark.parametrize("verb", ["add", "remove"]) +def test_marking_one_thing_needs_a_target(verb): + """A mark with nothing to attach to is a mistake worth naming early.""" + result = json.loads(an.annotate_preview_tool(action=verb, callback=lambda _p: "{}")) + assert "ref" in result["error"] + + +def test_speaks_the_renderer_s_verbs(): + """The overlay knows pin/unpin; the model gets words it can guess at.""" + sent = [] + + def cb(payload): + sent.append(payload) + return json.dumps({"acted": "pinned Buy now", "success": True}) + + an.annotate_preview_tool(action="add", ref="@e4", label="cheapest", callback=cb) + an.annotate_preview_tool(action="remove", ref="@e4", callback=cb) + + assert sent[0] == {"action": "pin", "ref": "@e4", "text": "cheapest"} + assert sent[1] == {"action": "unpin", "ref": "@e4"} + + +def test_clear_sends_no_target_so_the_overlay_drops_them_all(): + """`clear` IS an untargeted unpin — a ref left on it would take down one.""" + sent = [] + + def cb(payload): + sent.append(payload) + return json.dumps({"acted": "cleared every pin", "success": True}) + + an.annotate_preview_tool(action="clear", ref="@e4", selector="a", callback=cb) + + assert sent == [{"action": "unpin"}] + + +def test_defaults_to_adding(): + sent = [] + + def cb(payload): + sent.append(payload) + return json.dumps({"success": True}) + + an.annotate_preview_tool(ref="@e2", callback=cb) + + assert sent[0]["action"] == "pin" + + +def test_passes_the_renderer_s_answer_straight_through(): + out = json.loads( + an.annotate_preview_tool(ref="@e1", callback=lambda _p: json.dumps({"acted": "pinned Save", "success": True})) + ) + + assert out == {"acted": "pinned Save", "success": True} + + +def test_reports_a_silent_bridge(): + result = json.loads(an.annotate_preview_tool(ref="@e1", callback=lambda _p: "")) + assert "open_preview" in result["error"] diff --git a/tests/tools/test_drive_preview_tool.py b/tests/tools/test_drive_preview_tool.py new file mode 100644 index 0000000000..e7c0a06db3 --- /dev/null +++ b/tests/tools/test_drive_preview_tool.py @@ -0,0 +1,117 @@ +"""Tests for the GUI-surface ``drive_preview`` tool.""" + +import json + +from tools import drive_preview_tool as ap +from tools.registry import registry + + +def test_lives_in_the_gui_surface_toolset(monkeypatch): + """Mirrors read_preview: scoped by toolset, not by the backend's env.""" + monkeypatch.delenv("HERMES_DESKTOP", raising=False) + entry = registry.get_entry("drive_preview") + + assert entry is not None + assert entry.toolset == "desktop_ui" + assert entry.check_fn is None + + +def test_requires_callback(): + """Outside the desktop GUI there is no bridge — a clear error, no crash.""" + result = json.loads(ap.drive_preview_tool(action="elements", callback=None)) + assert "desktop" in result["error"] + + +def test_rejects_an_unknown_action(): + result = json.loads(ap.drive_preview_tool(action="teleport", callback=lambda _p: "{}")) + assert "action must be one of" in result["error"] + + +def test_interaction_verbs_need_a_target(): + """A click with nowhere to land is a mistake worth naming before the bridge.""" + for verb in ("click", "type", "press"): + result = json.loads(ap.drive_preview_tool(action=verb, text="x", key="Enter", callback=lambda _p: "{}")) + assert "ref" in result["error"], verb + + +def test_type_needs_text_and_press_needs_a_key(): + calls = [] + + def cb(payload): + calls.append(payload) + return json.dumps({"success": True}) + + assert "text" in json.loads(ap.drive_preview_tool(action="type", ref="@e1", callback=cb))["error"] + assert "key" in json.loads(ap.drive_preview_tool(action="press", ref="@e1", callback=cb))["error"] + assert calls == [] + + +def test_typing_an_empty_string_is_allowed(): + """Clearing a field is a real intent — only a missing `text` is an error.""" + seen = {} + ap.drive_preview_tool(action="type", ref="@e1", text="", callback=lambda p: seen.update(p) or "{}") + + assert seen["text"] == "" + + +def test_scroll_needs_no_target_and_validates_its_destination(): + seen = {} + + def cb(payload): + seen.clear() + seen.update(payload) + return json.dumps({"success": True}) + + ap.drive_preview_tool(action="scroll", callback=cb) + assert seen == {"action": "scroll"} + + result = json.loads(ap.drive_preview_tool(action="scroll", to="sideways", callback=cb)) + assert "to must be one of" in result["error"] + + +def test_payload_forwards_only_what_was_given(): + seen = {} + ap.drive_preview_tool( + action="type", + ref="@e5", + text="hunter2", + submit=True, + callback=lambda p: seen.update(p) or json.dumps({"success": True}), + ) + + assert seen == {"action": "type", "ref": "@e5", "text": "hunter2", "submit": True} + + +def test_numeric_arguments_are_validated(): + result = json.loads(ap.drive_preview_tool(action="scroll", amount="lots", callback=lambda _p: "{}")) + assert "integers" in result["error"] + + +def test_empty_answer_means_nothing_open(): + result = json.loads(ap.drive_preview_tool(action="elements", callback=lambda _p: "")) + assert "open_preview" in result["error"] + + +def test_passes_the_renderer_answer_through(): + payload = { + "success": True, + "acted": 'clicked button "Sign in"', + "url": "https://example.com/app", + "elements": [{"ref": "@e1", "role": "button", "label": "Log out", "selector": "#out"}], + } + result = json.loads(ap.drive_preview_tool(action="click", ref="@e2", callback=lambda _p: json.dumps(payload))) + + assert result == payload + + +def test_wraps_non_json_text(): + result = json.loads(ap.drive_preview_tool(action="elements", callback=lambda _p: "plain words")) + assert result == {"text": "plain words"} + + +def test_callback_failure_is_reported(): + def _boom(_payload): + raise RuntimeError("renderer went away") + + result = json.loads(ap.drive_preview_tool(action="elements", callback=_boom)) + assert "renderer went away" in result["error"] diff --git a/tests/tui_gateway/test_gui_surface_toolsets.py b/tests/tui_gateway/test_gui_surface_toolsets.py index ceebe004a3..6942a67f3d 100644 --- a/tests/tui_gateway/test_gui_surface_toolsets.py +++ b/tests/tui_gateway/test_gui_surface_toolsets.py @@ -18,7 +18,9 @@ import tui_gateway.server as server from toolsets import TOOLSETS, resolve_toolset GUI_TOOLS = { + "annotate_preview", "close_preview", + "drive_preview", "close_terminal", "focus_pane", "open_preview", diff --git a/tools/annotate_preview_tool.py b/tools/annotate_preview_tool.py new file mode 100644 index 0000000000..59919394e3 --- /dev/null +++ b/tools/annotate_preview_tool.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python3 +"""Leave a mark on the page in the Hermes desktop GUI's in-app browser. + +``drive_preview`` already draws every move it makes — the field it can reach, +a box round its target, the cursor going there — but those are transients: +each one stands for a single action and retires itself. That is right for +narrating a click and no use at all for holding a finding on screen. + +This is the deliberate one. An annotation outlines an element — or, with +``hold``, the entire visible field at once — and stays until the agent takes it +down, so it can show the user what it found, flag the fields +it is about to fill, or keep its place while it works elsewhere on the page. +Named for TouchDesigner's Annotate — the labelled box you drop around part of a +network to call it out. + +Annotations are bound to elements, not coordinates: they ride scrolls and +reflows, and they go when their element does, so a navigation clears them +without the agent having to. + +Rides the same ``preview.act`` bridge as ``drive_preview`` rather than opening +a second channel — the renderer already resolves ``@e`` refs and owns the +overlay, so this is one more verb on a wire that exists. + +Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for +desktop-sourced sessions. +""" + +import json +from typing import Callable, Optional + +from tools.registry import registry, tool_error + +ACTIONS = ("add", "hold", "remove", "clear") + +# Verbs the renderer knows, keyed by ours. `clear` is `unpin` with nothing to +# aim at, which the overlay reads as "all of them". +WIRE = {"add": "pin", "hold": "hold", "remove": "unpin", "clear": "unpin"} + + +def annotate_preview_tool( + action: str = "add", + ref: Optional[str] = None, + selector: Optional[str] = None, + label: Optional[str] = None, + callback: Optional[Callable] = None, +) -> str: + """Put one annotation up, take one down, or clear them all.""" + if callback is None: + return tool_error("annotate_preview is only available in the Hermes desktop app.") + + verb = (action or "add").strip().lower() + if verb not in ACTIONS: + return tool_error(f"action must be one of: {', '.join(ACTIONS)}.") + + if verb in ("add", "remove") and not (ref or selector): + return tool_error( + f"{verb} needs a ref from drive_preview action='elements' " + "(e.g. '@e5') or a CSS selector." + ) + + payload = { + name: val + for name, val in ( + ("action", WIRE[verb]), + ("ref", None if verb in ("clear", "hold") else ref), + ("selector", None if verb in ("clear", "hold") else selector), + ("text", label), + ) + if val is not None + } + + try: + raw = callback(payload) + except Exception as exc: + return tool_error(f"Failed to annotate the in-app browser: {exc}") + + if not raw: + return tool_error( + "The annotation timed out, or no GUI window answered. " + "Open a page with open_preview first." + ) + + try: + return json.dumps(json.loads(raw), ensure_ascii=False) + except (TypeError, ValueError): + return json.dumps({"text": str(raw)}, ensure_ascii=False) + + +ANNOTATE_PREVIEW_SCHEMA = { + "name": "annotate_preview", + "description": ( + "Draw a lasting mark on the page open in the in-app browser / preview " + "pane of the Hermes desktop GUI. Everything drive_preview draws as it " + "works fades on its own; an annotation STAYS until you remove it, so " + "this is how you point at something. Use it to show the user what you " + "found ('here are the three cheapest'), flag what you are about to " + "change before you change it, or keep your place while you work " + "elsewhere on the page. Address elements by the same '@e5' refs " + "drive_preview action='elements' hands back. action='add' outlines an " + "element and gives it an optional short label; 'hold' freezes the WHOLE " + "visible field at once — every element the page offers, outlined and " + "named — which is the " + "picture drive_preview flashes as it works, made to stay; 'remove' " + "takes one down; 'clear' takes them all down. Annotations follow their element " + "as the page scrolls and disappear if it does, so a navigation clears " + "them for you. Keep labels to a word or two — they are drawn on the " + "page, not read aloud." + ), + "parameters": { + "type": "object", + "properties": { + "action": { + "type": "string", + "enum": list(ACTIONS), + "description": ( + "'add' marks one element, 'hold' freezes the whole visible " + "field, 'remove' takes one down, 'clear' takes them all " + "down. Defaults to 'add'." + ), + }, + "ref": { + "type": "string", + "description": "Element reference from drive_preview action='elements' (e.g. '@e5').", + }, + "selector": { + "type": "string", + "description": "CSS selector, as a fallback when no ref fits. Prefer ref.", + }, + "label": { + "type": "string", + "description": "Short caption drawn on the mark, e.g. 'cheapest'. Optional.", + }, + }, + "required": [], + }, +} + + +registry.register( + name="annotate_preview", + toolset="desktop_ui", + schema=ANNOTATE_PREVIEW_SCHEMA, + handler=lambda args, **kw: annotate_preview_tool( + action=args.get("action", "add"), + ref=args.get("ref"), + selector=args.get("selector"), + label=args.get("label"), + callback=kw.get("callback"), + ), + emoji="🔖", +) diff --git a/tools/drive_preview_tool.py b/tools/drive_preview_tool.py new file mode 100644 index 0000000000..7bdb1207a5 --- /dev/null +++ b/tools/drive_preview_tool.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +"""Interact with the in-app browser / preview pane in the Hermes desktop GUI. + +``open_preview`` shows a page and ``read_preview`` reads it; this tool is the +third leg — clicking, typing, scrolling, and history — so the agent can drive +the same page the user is looking at instead of narrating from the outside. + +Elements are addressed by ``@e1``-style refs from ``action="elements"``, the +same shape the ``browser_*`` tools use, so what the model knows about one +transfers to the other. Refs are only valid until the page navigates; the +renderer says so rather than clicking whatever now sits at that index. + +Round-trips through the gateway's blocking-prompt bridge like ``read_preview``: +tui_gateway emits ``preview.act.request``, the renderer injects the interaction +engine into the pane's webview and answers ``preview.act.respond`` with the +outcome plus a fresh inventory. This module is just schema + a thin dispatcher +over the platform-injected callback. + +Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for +desktop-sourced sessions. +""" + +import json +from typing import Callable, Optional + +from tools.registry import registry, tool_error + +ACTIONS = ( + "elements", + "click", + "hover", + "type", + "scroll", + "press", + "strobe", + "back", + "forward", + "reload", +) +SCROLL_TO = ("top", "bottom") + +# Verbs that need something to act on — a ref from the last inventory, or a +# raw CSS selector. `scroll` is deliberately absent: bare, it scrolls the page. +NEEDS_TARGET = ("click", "hover", "type", "press") + + +def drive_preview_tool( + action: str = "", + ref: Optional[str] = None, + selector: Optional[str] = None, + text: Optional[str] = None, + key: Optional[str] = None, + submit: Optional[bool] = None, + amount: Optional[int] = None, + to: Optional[str] = None, + limit: Optional[int] = None, + callback: Optional[Callable] = None, +) -> str: + """Dispatch one interaction to the desktop renderer and return its outcome.""" + if callback is None: + return tool_error("drive_preview is only available in the Hermes desktop app.") + + verb = (action or "").strip().lower() + if verb not in ACTIONS: + return tool_error(f"action must be one of: {', '.join(ACTIONS)}.") + + if verb in NEEDS_TARGET and not (ref or selector): + return tool_error( + f"{verb} needs a ref from action='elements' (e.g. '@e5') or a CSS selector." + ) + + if verb == "type" and text is None: + return tool_error("type needs the text to enter.") + + if verb == "press" and not key: + return tool_error("press needs a key, e.g. 'Enter' or 'Escape'.") + + if to is not None and to not in SCROLL_TO: + return tool_error(f"to must be one of: {', '.join(SCROLL_TO)}.") + + try: + payload = { + name: val + for name, val in ( + ("action", verb), + ("ref", ref), + ("selector", selector), + ("text", text), + ("key", key), + ("submit", submit), + ("to", to), + ("amount", None if amount is None else int(amount)), + ("max", None if limit is None else int(limit)), + ) + if val is not None + } + except (TypeError, ValueError): + return tool_error("amount and max must be integers.") + + try: + raw = callback(payload) + except Exception as exc: + return tool_error(f"Failed to act on the in-app browser: {exc}") + + if not raw: + return tool_error( + "The action timed out, or no GUI window answered. " + "Open a page with open_preview first." + ) + + # The renderer answers with a JSON object; pass it through, else wrap it. + try: + return json.dumps(json.loads(raw), ensure_ascii=False) + except (TypeError, ValueError): + return json.dumps({"text": str(raw)}, ensure_ascii=False) + + +ACT_PREVIEW_SCHEMA = { + "name": "drive_preview", + "description": ( + "Interact with the page open in the in-app browser / preview pane of " + "the Hermes desktop GUI — the pane open_preview opens beside this " + "chat. This is how you USE a web app the user is looking at: log in, " + "fill a form, click through a flow, page a long document. ALWAYS call " + "action='elements' first to get the current inventory of clickable and " + "typable things — each carries a ref like '@e5' plus its role, label, " + "and value — then act with that ref instead of guessing a selector. " + "Every action answers with a refreshed inventory and the live url/" + "title, so a click that navigated is visible immediately and you can " + "chain the next step without re-reading. Refs are invalidated by a " + "navigation; when told they are stale, call elements again. The mouse " + "and keyboard are real: the pointer travels to its target and the page " + "sees genuine input, so hover menus open and hover-only controls work. " + "Actions: 'elements' (inventory), 'click', 'hover' (move the pointer " + "onto something and leave it there — use it to open a dropdown or " + "reveal a tooltip before clicking inside it), 'type' (set a field's " + "text; submit=true also presses Enter and submits the form), 'scroll' " + "(the page, or a ref'd scrollable), 'press' (a named key), 'strobe' " + "(touch nothing — just rattle the highlight through the page again; " + "'elements' already does this once, so reach for it only when asked to " + "flick, flash, or bounce around the page some more, and note one call " + "runs a whole multi-second burst, so never loop it per element), and " + "'back'/'forward'/'reload' for history. The pane draws every move as " + "it happens so the user can follow along; those marks fade on their " + "own, and annotate_preview is how you leave one up on purpose. Use " + "read_preview when you only " + "need the page's text, and the browser_* tools when the work belongs " + "in a separate automated browser rather than the user's own pane." + ), + "parameters": { + "type": "object", + "properties": { + "action": { + "type": "string", + "enum": list(ACTIONS), + "description": "What to do. Start with 'elements'.", + }, + "ref": { + "type": "string", + "description": "Element reference from the last elements call (e.g. '@e5').", + }, + "selector": { + "type": "string", + "description": "CSS selector, as a fallback when no ref fits. Prefer ref.", + }, + "text": {"type": "string", "description": "For 'type': the text to enter."}, + "submit": { + "type": "boolean", + "description": "For 'type': press Enter and submit the owning form afterwards.", + }, + "key": { + "type": "string", + "description": "For 'press': the key name, e.g. 'Enter', 'Escape', 'ArrowDown'.", + }, + "amount": { + "type": "integer", + "description": "For 'scroll': pixels to scroll (negative scrolls up). Defaults to about one screen.", + }, + "to": { + "type": "string", + "enum": list(SCROLL_TO), + "description": "For 'scroll': jump to the top or bottom instead of a distance.", + }, + "max": { + "type": "integer", + "description": "For 'elements': cap the inventory. Defaults to the per-call maximum.", + }, + }, + "required": ["action"], + }, +} + + +registry.register( + name="drive_preview", + toolset="desktop_ui", + schema=ACT_PREVIEW_SCHEMA, + handler=lambda args, **kw: drive_preview_tool( + action=args.get("action", ""), + ref=args.get("ref"), + selector=args.get("selector"), + text=args.get("text"), + key=args.get("key"), + submit=args.get("submit"), + amount=args.get("amount"), + to=args.get("to"), + limit=args.get("max"), + callback=kw.get("callback"), + ), + emoji="🖱️", +) diff --git a/toolsets.py b/toolsets.py index 7c5d032ad8..23eacf088a 100644 --- a/toolsets.py +++ b/toolsets.py @@ -276,7 +276,7 @@ TOOLSETS = { "description": "Desktop GUI affordances — in-app terminal/browser panes, pane focus, reactions (GUI sessions only)", "tools": [ "read_terminal", "close_terminal", - "open_preview", "close_preview", "read_preview", + "open_preview", "close_preview", "read_preview", "drive_preview", "annotate_preview", "read_window_below", "focus_pane", "react_to_message", "setup_mcp", "tour", diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index 115fd14047..6cae7b49e0 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -1423,6 +1423,15 @@ def _(rid, params: dict) -> dict: return _respond(rid, params, "text", allow_expired=True) +@method("preview.act.respond") +def _(rid, params: dict) -> dict: + # `text` is a JSON string with the interaction's outcome (drive_preview + # tool) — what it acted on, the live url/title, and a refreshed element + # inventory. allow_expired=True for the same reason as preview.read: the + # settle-and-rescan can lose the race with the tool's bounded wait. + return _respond(rid, params, "text", allow_expired=True) + + @method("window.read.respond") def _(rid, params: dict) -> dict: # `text` is a JSON string describing the OS window underneath the Hermes diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 6bb356201c..e8e76e0f29 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3548,6 +3548,7 @@ def _block( "clarify.request", "terminal.read.request", "preview.read.request", + "preview.act.request", "window.read.request", "mcp.setup.request", "tour.request", @@ -6367,6 +6368,20 @@ def _agent_cbs(sid: str) -> dict: {k: v for k, v in (("start", start), ("count", count)) if v is not None}, timeout=45, ), + # drive_preview tool (desktop GUI): the renderer injects the interaction + # engine into the preview pane's webview (or drives the pane's history) + # and answers preview.act.respond with the outcome plus a refreshed + # element inventory. Same budget as the preview read, which it ends + # with — a click on a slow page pays for the settle and the re-scan. + # annotate_preview rides this same callback: it resolves a target + # through the same engine and differs only in the verb it sends, so it + # needs a tool of its own but not a channel of its own. + "drive_preview_callback": lambda payload: _block( + "preview.act.request", + sid, + dict(payload), + timeout=45, + ), # read_window_below tool (desktop GUI): the renderer asks its main # process (which owns native window enumeration) which OS window sits # directly underneath the Hermes window, and answers diff --git a/website/docs/reference/tools-reference.md b/website/docs/reference/tools-reference.md index 721ffb91ac..9ccfe87e25 100644 --- a/website/docs/reference/tools-reference.md +++ b/website/docs/reference/tools-reference.md @@ -8,7 +8,7 @@ description: "Authoritative reference for Hermes built-in tools, grouped by tool This page documents Hermes' built-in tools, grouped by toolset. Availability varies by platform, credentials, and enabled toolsets. -**Quick counts (current registry):** ~84 tools — 10 browser tools (core) + 2 CDP-gated browser tools, 4 file tools, 4 Home Assistant tools, 2 terminal tools (`terminal`, `process`), 9 desktop-GUI tools (`read_terminal`, `close_terminal`, `open_preview`, `close_preview`, `read_preview`, `read_window_below`, `focus_pane`, `react_to_message`, `tour` — desktop-app sessions only), 2 web tools, 5 Feishu tools, 7 Spotify tools (registered by the bundled `spotify` plugin), 5 Yuanbao tools, 12 kanban tools (registered when the kanban dispatcher spawns the agent), 3 project tools (desktop/GUI sessions), 2 Discord tools, 3 video tools (`video_generate`, `xai_video_edit`, `xai_video_extend`), and a handful of standalone tools (`memory`, `clarify`, `delegate_task`, `execute_code`, `cronjob`, `session_search`, `skill_view`/`skill_manage`/`skills_list`, `text_to_speech`, `image_generate`, `vision_analyze`, `video_analyze`, `todo`, `computer_use`, `x_search`). +**Quick counts (current registry):** ~86 tools — 10 browser tools (core) + 2 CDP-gated browser tools, 4 file tools, 4 Home Assistant tools, 2 terminal tools (`terminal`, `process`), 11 desktop-GUI tools (`read_terminal`, `close_terminal`, `open_preview`, `close_preview`, `read_preview`, `drive_preview`, `annotate_preview`, `read_window_below`, `focus_pane`, `react_to_message`, `tour` — desktop-app sessions only), 2 web tools, 5 Feishu tools, 7 Spotify tools (registered by the bundled `spotify` plugin), 5 Yuanbao tools, 12 kanban tools (registered when the kanban dispatcher spawns the agent), 3 project tools (desktop/GUI sessions), 2 Discord tools, 3 video tools (`video_generate`, `xai_video_edit`, `xai_video_extend`), and a handful of standalone tools (`memory`, `clarify`, `delegate_task`, `execute_code`, `cronjob`, `session_search`, `skill_view`/`skill_manage`/`skills_list`, `text_to_speech`, `image_generate`, `vision_analyze`, `video_analyze`, `todo`, `computer_use`, `x_search`). :::tip MCP Tools In addition to built-in tools, Hermes can load tools dynamically from MCP servers. MCP tools appear with the prefix `mcp____` (e.g., `mcp__github__create_issue` for the `github` MCP server). See [MCP Integration](/user-guide/features/mcp) for configuration. @@ -199,6 +199,8 @@ messaging, and cron sessions. | `open_preview` | Open a web URL, localhost dev-server URL, or file path in the preview pane beside the chat in the Hermes desktop app. | — | | `close_preview` | Close the preview pane beside the chat, or one tab inside it. Omit `url` to close the whole pane; pass a URL or file path to close that tab. | — | | `read_preview` | Read what's currently shown in the preview pane of the Hermes desktop GUI — the in-app Browser's page text (URL + title + rendered text, pageable with `start`/`count`), or a file/artifact tab's identity. | — | +| `drive_preview` | Interact with the page open in the in-app browser: `elements` inventories what's clickable and typable (each with an `@e1`-style ref, role, label, and value), then `click`, `hover`, `type`, `scroll`, and `press` act on a ref, and `back`/`forward`/`reload` drive the pane's history. The pointer and keyboard are real input, so hover menus open. Every action answers with the live URL and a refreshed inventory. | — | +| `annotate_preview` | Outline an element in the in-app browser and leave the mark up until it's removed — the deliberate counterpart to the transient cues `drive_preview` draws as it works. `add` marks a ref with an optional short label, `remove` takes one down, `clear` takes them all. Marks follow their element and vanish with it, so a navigation clears them. | — | | `read_window_below` | Identify the OS window directly underneath the Hermes desktop window — app name, title, bounds (metadata only, never pixels). On macOS, other apps' titles appear only when Screen Recording is already granted; the tool never prompts for it. | — | | `focus_pane` | Reveal and focus a pane in the Hermes desktop app (chat, files, terminal, review, sessions). | — | | `react_to_message` | React to a message with a single emoji, iMessage-tapback style. Opt-in via Settings → Appearance (`display.message_reactions`). | — | diff --git a/website/docs/reference/toolsets-reference.md b/website/docs/reference/toolsets-reference.md index f71c2eeb1e..5904f1a9f7 100644 --- a/website/docs/reference/toolsets-reference.md +++ b/website/docs/reference/toolsets-reference.md @@ -71,7 +71,7 @@ Or in-session: | `video_gen` | `video_generate`, `xai_video_edit`, `xai_video_extend` | Text-to-video and image-to-video via plugin-registered backends (xAI Grok-Imagine, FAL.ai Veo 3.1 / Pixverse v6 / Kling O3). Pass `image_url` to animate an image; omit it for text-to-video. `xai_video_edit` / `xai_video_extend` are provider-specific edit/extend tools, gated on xAI Imagine credentials. | | `kanban` | `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_block`, `kanban_comment`, `kanban_complete`, `kanban_create`, `kanban_heartbeat`, `kanban_link`, `kanban_list`, `kanban_request_changes`, `kanban_request_review`, `kanban_show`, `kanban_unblock` | Multi-agent coordination tools. Registered for dispatcher-spawned task workers (`HERMES_KANBAN_TASK`) and for profiles that explicitly list the `kanban` toolset by name (the `all`/`*` wildcard does **not** enable it). Workers mark tasks done, request first-class review, block, heartbeat, comment, and create/link follow-up tasks; orchestrator profiles additionally get board-routing tools like list/unblock. `delegate_task` children are not Kanban run owners: their schema strips/disables this toolset and runtime guards reject direct board mutations, even if parent `HERMES_KANBAN_*` env vars are present. | | `memory` | `memory` | Persistent cross-session memory management. | -| `desktop_ui` | `close_preview`, `close_terminal`, `focus_pane`, `open_preview`, `react_to_message`, `read_preview`, `read_terminal`, `read_window_below`, `tour` | Affordances that act on the Hermes desktop app itself — read/close the embedded terminal pane, open/read/close the in-app browser, identify the OS window behind the app, reveal a pane, react to a message, run a guided tour (highlight + narrate UI elements in the app or the preview pane). Enabled for sessions whose source is the desktop app, whichever backend it's connected to (local, SSH, URL, or Hermes Cloud). Never present on CLI, TUI, messaging, or cron sessions. | +| `desktop_ui` | `annotate_preview`, `close_preview`, `close_terminal`, `drive_preview`, `focus_pane`, `open_preview`, `react_to_message`, `read_preview`, `read_terminal`, `read_window_below`, `tour` | Affordances that act on the Hermes desktop app itself — read/close the embedded terminal pane, open, read, close, interact with, and annotate the in-app browser, identify the OS window behind the app, reveal a pane, react to a message, run a guided tour (highlight + narrate UI elements in the app or the preview pane). Enabled for sessions whose source is the desktop app, whichever backend it's connected to (local, SSH, URL, or Hermes Cloud). Never present on CLI, TUI, messaging, or cron sessions. | | `project` | `project_create`, `project_list`, `project_switch` | Create and switch desktop [Projects](../user-guide/cli.md) (named, multi-folder workspaces). GUI / desktop sessions only. | | `safe` | `image_generate`, `vision_analyze`, `web_extract`, `web_search` (via `includes`) | Read-only research + media generation. No file writes, no terminal, no code execution. | | `search` | `web_search` | Web search only (without extract). | From c358a6a570d18e4f461ebd178f43e32ad5d6c499 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 20 Aug 2026 02:32:56 -0500 Subject: [PATCH 3/7] feat(desktop): drive the preview with real input, not synthetic events A dispatched MouseEvent is untrusted, so hover menus never opened and any control that gates on isTrusted ignored it. The pane now sends input through the webview itself: the pointer travels to its target and the page cannot tell it from a hand. --- .../src/app/chat/right-rail/preview-drive.ts | 138 ++++++++++++++++++ .../src/app/chat/right-rail/preview-input.ts | 56 +++++++ .../src/app/chat/right-rail/preview-pane.tsx | 30 +++- 3 files changed, 223 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/chat/right-rail/preview-drive.ts create mode 100644 apps/desktop/src/app/chat/right-rail/preview-input.ts diff --git a/apps/desktop/src/app/chat/right-rail/preview-drive.ts b/apps/desktop/src/app/chat/right-rail/preview-drive.ts new file mode 100644 index 0000000000..18b9dbf24e --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-drive.ts @@ -0,0 +1,138 @@ +/** + * PREVIEW DRIVE — the agent's hand on the mouse and keyboard. + * + * Everything here goes out through `sendInputEvent`, so the guest page receives + * ordinary trusted input: the pointer really moves across it, `:hover` really + * matches, focus really lands, and a menu that only exists while hovered is + * actually open by the time the click arrives. That last part is the whole + * point — a synthetic `el.click()` on a dropdown item fires at a node the page + * never rendered, which is why script-driven clicking quietly misses. + * + * The pointer is walked to its target in steps rather than teleported, for the + * same reason: a page learns it is hovered from a stream of moves, and one + * move from nowhere to the target skips every element in between. Fourteen + * steps over ~280ms is what Playwright's action cursor settles on too. + */ + +import type { PreviewInputHandle } from './preview-input' + +export interface DrivePoint { + x: number + y: number +} + +/** The pointer JUMPS. These few steps over a couple of frames exist only so the + * page receives a move stream at all — `:hover`, `pointerenter` and the menus + * that hang off them need more than a single teleport to resolve — not so the + * travel can be watched. A machine does not sweep a mouse across a screen. */ +const GLIDE_STEPS = 3 +const GLIDE_MS = 36 + +/** Gap between keystrokes. Fast enough not to pad the turn, slow enough that a + * page debouncing its input handler still sees them as separate. */ +const KEY_MS = 7 + +/** Notches in a wheel gesture, and how long the gesture takes. Sent as a stream + * rather than one big delta so scroll-linked effects — sticky headers, reveal + * animations, infinite-scroll loaders — get the same series of events a hand + * would produce, and so it looks like momentum rather than a jump cut. */ +const WHEEL_STEPS = 12 +const WHEEL_MS = 240 + +/** Where the pointer was left, so the next glide starts from there instead of + * jumping. Module-level because the pointer is a property of the pane, not of + * any one action. The origin is a placeholder, not a position — see below. */ +let pointer: DrivePoint = { x: 0, y: 0 } +let placed = false + +/** Whether the pointer has ever been sent anywhere. Until it has, its notional + * spot is the top-left corner, which is a real coordinate a caller can wheel + * or click at by accident — so callers that only need it to be SOMEWHERE + * sensible have to know the difference. */ +export function pointerPlaced(): boolean { + return placed +} + +const wait = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)) + +/** Decelerating, like a hand arriving at a target rather than a linear sweep. */ +const easeOut = (t: number) => 1 - Math.pow(1 - t, 3) + +/** Walk the pointer to `to`, letting the page hover everything on the way. */ +export async function glideTo(input: PreviewInputHandle, to: DrivePoint): Promise { + const from = pointer + + for (let step = 1; step <= GLIDE_STEPS; step++) { + const progress = easeOut(step / GLIDE_STEPS) + + input.send({ + type: 'mouseMove', + x: Math.round(from.x + (to.x - from.x) * progress), + y: Math.round(from.y + (to.y - from.y) * progress) + }) + await wait(GLIDE_MS / GLIDE_STEPS) + } + + pointer = to + placed = true +} + +/** Press and release at the pointer's current spot. `clicks` of 3 selects the + * text under it, which is how a field gets cleared without a modifier key. */ +export async function clickAt(input: PreviewInputHandle, clicks = 1): Promise { + for (let click = 1; click <= clicks; click++) { + input.send({ button: 'left', clickCount: click, type: 'mouseDown', x: pointer.x, y: pointer.y }) + input.send({ button: 'left', clickCount: click, type: 'mouseUp', x: pointer.x, y: pointer.y }) + await wait(KEY_MS) + } +} + +/** Wheel `down` pixels' worth of notches at the pointer's current spot; negative + * scrolls up. Electron's wheel delta is the legacy `wheelDelta` sign — positive + * moves the content down, i.e. scrolls UP — so it is the inverse of the number + * a caller asks for, and of DOM `WheelEvent.deltaY`. */ +export async function wheelBy(input: PreviewInputHandle, down: number): Promise { + let sent = 0 + + for (let step = 1; step <= WHEEL_STEPS; step++) { + const so_far = Math.round(down * easeOut(step / WHEEL_STEPS)) + const notch = so_far - sent + + sent = so_far + + if (notch) { + input.send({ deltaX: 0, deltaY: -notch, type: 'mouseWheel', x: pointer.x, y: pointer.y }) + } + + await wait(WHEEL_MS / WHEEL_STEPS) + } +} + +/** Send one key the long way round, so a page watching any of the three sees it. */ +export async function pressKey(input: PreviewInputHandle, key: string): Promise { + input.send({ keyCode: key, type: 'keyDown' }) + input.send({ keyCode: key, type: 'char' }) + input.send({ keyCode: key, type: 'keyUp' }) + await wait(KEY_MS) +} + +/** Select-all inside whatever has focus, which for a focused field is that + * field's own text and nothing else. This replaced a triple-click: a triple + * click is a POINTER gesture, so it selects whatever paragraph sits under the + * cursor whenever the target turns out not to be a field, and the agent was + * leaving pages with their body text highlighted. There is no `char` phase — + * a chord is not text entry, and sending one types a literal 'a'. */ +export async function selectAll(input: PreviewInputHandle): Promise { + const chord = ['control', 'meta'] + + input.send({ keyCode: 'a', modifiers: chord, type: 'keyDown' }) + input.send({ keyCode: 'a', modifiers: chord, type: 'keyUp' }) + await wait(KEY_MS) +} + +/** Type `text` a character at a time into whatever currently has focus. */ +export async function typeText(input: PreviewInputHandle, text: string): Promise { + for (const character of text) { + await pressKey(input, character) + } +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-input.ts b/apps/desktop/src/app/chat/right-rail/preview-input.ts new file mode 100644 index 0000000000..c0f1dbb5a0 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-input.ts @@ -0,0 +1,56 @@ +/** + * PREVIEW INPUT REGISTRY — real input into the preview pane's guest page, the + * difference between the agent DRIVING the browser and merely poking its DOM. + * + * `executeJavaScript` can only ever dispatch synthetic events: `isTrusted` is + * false, the browser's own hover target never moves, `:hover` rules never + * match, and hover-gated menus never open — so a click lands on a dropdown item + * that was never rendered. `sendInputEvent` goes in through Chromium's input + * pipeline instead, producing the same events a hand on the mouse would. + * + * It has to be called on the `` ELEMENT. Sending to the embedder's + * webContents does not reach a guest (electron/electron#20333), which is why + * this is a per-pane registry rather than something main could do. + * + * Coordinates are relative to the webview, and the webview IS the guest + * viewport — so a rect the act engine measured inside the page needs no + * conversion on the way back out. + */ + +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + +/** The subset of Electron's input events the agent needs to drive a page. */ +export type PreviewInputEvent = + | { button: 'left'; clickCount: number; type: 'mouseDown' | 'mouseUp'; x: number; y: number } + | { deltaX: number; deltaY: number; type: 'mouseWheel'; x: number; y: number } + | { keyCode: string; modifiers?: string[]; type: 'char' | 'keyDown' | 'keyUp' } + | { type: 'mouseMove'; x: number; y: number } + +export interface PreviewInputHandle { + /** Give the guest keyboard focus, so key events reach its active element. */ + focus: () => void + send: (event: PreviewInputEvent) => void +} + +const handles = new Map() + +/** Register a live pane's input channel; returns an idempotent unregister. */ +export function registerPreviewInput(tabId: string, handle: PreviewInputHandle): () => void { + handles.set(tabId, handle) + + return () => { + if (handles.get(tabId) === handle) { + handles.delete(tabId) + } + } +} + +/** The ACTIVE preview tab's input channel. Null = nothing real to drive, and + * the caller falls back to synthesizing events inside the page. */ +export function activePreviewInput(): PreviewInputHandle | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && handles.get(tab.id)) || null +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx index e2f236cf2f..076fc4a94f 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx +++ b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx @@ -28,6 +28,7 @@ import { import { type ConsoleEntry } from './preview-console-state' import { previewConsoleState } from './preview-console-store' import { LocalFilePreview, PreviewEmptyState } from './preview-file' +import { registerPreviewInput, type PreviewInputEvent } from './preview-input' import { PREVIEW_BROWSER_ATTR, registerPreviewNav } from './preview-nav' import { registerPreviewPageReader } from './preview-reader' import { registerPreviewScriptRunner } from './preview-script-runner' @@ -53,6 +54,7 @@ type PreviewWebview = HTMLElement & { reloadIgnoringCache?: () => void replaceMisspelling?: (word: string) => void selectAll?: () => void + sendInputEvent?: (event: PreviewInputEvent) => void } /** The raw Chromium params riding the webview tag's `context-menu` event. */ @@ -457,7 +459,7 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest = // Publish the SCRIPT runner for this tab: the one channel into the guest // page, shared by the tour tool (injected driver.js walkthroughs) and the - // act_preview tool (clicking, typing, scrolling the page the user sees). + // drive_preview tool (clicking, typing, scrolling the page the user sees). useEffect(() => { if (!isWebPreview || !tabId) { return @@ -474,6 +476,32 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest = }) }, [isWebPreview, tabId]) + // Publish the INPUT channel for this tab. Same idea as the script runner, but + // it carries real Chromium input rather than script — the agent's clicks and + // keystrokes arrive as trusted events, so the page hovers, focuses and reacts + // exactly as it would under a human hand. + useEffect(() => { + if (!isWebPreview || isRemoteHtml || !tabId) { + return + } + + return registerPreviewInput(tabId, { + focus: () => webviewRef.current?.focus?.(), + send: event => { + const webview = webviewRef.current + + // Never optional-chain this call away: a missing method would make every + // agent click a silent no-op that still reports success, because the + // overlay and the read-back both run on the separate script channel. + if (typeof webview?.sendInputEvent !== 'function') { + throw new Error('preview webview cannot take input events') + } + + webview.sendInputEvent(event) + } + }) + }, [isRemoteHtml, isWebPreview, tabId]) + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) useEffect(() => { if (!consoleOpen) { From f262e684d80119c8aeeaeae087b83bf0e4961f18 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 20 Aug 2026 02:32:56 -0500 Subject: [PATCH 4/7] feat(desktop): an overlay that shows what the agent is doing to the page MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Driving someone's browser invisibly is unnerving, and this browser is the one they are signed into. The pane now draws the field the agent can reach, a box round what it is touching, a cursor that goes there, and a wipe over text it just read — one cursor primitive and one mark primitive, in a closed shadow root so the agent's own inventory cannot see them. Marks carry the same handle the agent addresses them by, so the word on screen and the word in the transcript are the same string. The point is supervision rather than decoration: a person glancing at the pane can tell what is about to happen to their live session, and stop it. It also has to cover the waiting. The agent flashes through a click in under a second and then sits idle for the twenty to a hundred seconds the model spends deciding what to do next, which is most of the wall clock of any task — so the surface used to look broken during the part where it was working hardest. A think stage runs off the $busy edge, sparsely flashing elements from the field the last action left behind, and rest stops it. It guards itself: started before there is an overlay or a field, it idles until there is one, so it can be raised on the turn boundary without knowing whether anything has been inventoried yet. read_preview had the same hole from the other side. Reading is the cheapest thing the agent does — hundredths of a second between two model round trips — so paging through a document left the pane dark for twenty seconds immediately after the one moment that showed anything. It draws a top-to-bottom wipe over the text it took, and that is the one stage allowed to be a wipe: reading is the only thing the agent does to a page in an order a person could follow. Both go through preview-nudge, which says a single stage to an overlay the page already has rather than re-shipping the engine to narrate. On a page the agent never acted on it is a no-op, which is the honest answer — chrome there would be a lie about what it did. Everything respects prefers-reduced-motion. --- .../app/chat/right-rail/preview-act.test.ts | 189 ++- .../src/app/chat/right-rail/preview-act.ts | 529 +++++++- .../src/app/chat/right-rail/preview-mind.ts | 29 + .../src/app/chat/right-rail/preview-nudge.ts | 38 + .../src/app/chat/right-rail/preview-pane.tsx | 6 +- .../src/app/chat/right-rail/preview-reader.ts | 9 + .../gateway-event/desktop-bridge.ts | 36 +- .../src/lib/preview-act/act-in-page.test.ts | 134 ++- .../src/lib/preview-act/act-in-page.ts | 234 +++- .../src/lib/preview-act/watch-in-page.test.ts | 363 ++++++ .../src/lib/preview-act/watch-in-page.ts | 1063 +++++++++++++++++ 11 files changed, 2574 insertions(+), 56 deletions(-) create mode 100644 apps/desktop/src/app/chat/right-rail/preview-mind.ts create mode 100644 apps/desktop/src/app/chat/right-rail/preview-nudge.ts create mode 100644 apps/desktop/src/lib/preview-act/watch-in-page.test.ts create mode 100644 apps/desktop/src/lib/preview-act/watch-in-page.ts diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.test.ts b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts index b53e046342..2b68b1eb99 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-act.test.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts @@ -4,6 +4,7 @@ import { $rightRailActiveTabId } from '@/store/layout' import { closeRightRail, openPreview, type PreviewTarget } from '@/store/preview' import { actOnActivePreview } from './preview-act' +import { registerPreviewInput } from './preview-input' import { registerPreviewNav } from './preview-nav' import { registerPreviewScriptRunner } from './preview-script-runner' @@ -11,7 +12,7 @@ function urlTarget(url: string): PreviewTarget { return { kind: 'url', label: 'Browser', source: url, url } } -describe('actOnActivePreview (act_preview tool)', () => { +describe('actOnActivePreview (drive_preview tool)', () => { // URL targets share the singleton Browser tab id, so anything a test // registers would answer the next one. let cleanups: Array<() => void> = [] @@ -105,6 +106,192 @@ describe('actOnActivePreview (act_preview tool)', () => { expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).error).toContain('did not answer') }) + /** A pane that answers the locate trip with a fixed on-screen point, and the + * read-back trip with an empty inventory. Returns the input spy. */ + const withDrivenPane = () => { + const tabId = openBrowserTab() + const send = vi.fn() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 120, y: 80 }, success: true }) + : // `hit` is the page's witness that the real pointerdown arrived. + JSON.stringify({ elements: [], hit: { tag: 'BUTTON', trusted: true }, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send })) + + return send + } + + const sentTypes = (send: ReturnType) => send.mock.calls.map(([event]) => event.type) + + it('clicks with real input, walking the pointer to where the page said', async () => { + const send = withDrivenPane() + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + const types = sentTypes(send) + + // Stepped rather than teleported: a page learns it is hovered from a stream + // of moves, and one jump to the target skips everything in between. + expect(types.filter(type => type === 'mouseMove').length).toBeGreaterThan(1) + expect(types).toContain('mouseDown') + expect(types).toContain('mouseUp') + expect(send.mock.calls.map(([event]) => event).find(event => event.type === 'mouseDown')).toMatchObject({ + x: 120, + y: 80 + }) + expect(result.acted).toBe('clicked button "Save"') + }) + + it('types by pressing keys, after selecting whatever the field held', async () => { + const send = withDrivenPane() + + await actOnActivePreview({ kind: 'type', ref: '@e1', submit: true, text: 'hi' }) + + const events = send.mock.calls.map(([event]) => event) + const chars = events.filter(event => event.type === 'char').map(event => event.keyCode) + + // One click to focus, then select-all by keyboard — NOT the triple-click + // this used to do. A triple-click is a pointer gesture, so it grabs the + // paragraph under the cursor whenever the target turns out not to be a + // field, and the agent was leaving pages with their body text highlighted. + expect(events.filter(event => event.type === 'mouseDown').map(event => event.clickCount)).toEqual([1]) + expect(events.filter(event => event.type === 'keyDown' && event.keyCode === 'a')[0]).toMatchObject({ + modifiers: ['control', 'meta'] + }) + // The chord must not send a `char` phase, or select-all types a literal 'a'. + expect(chars).toEqual(['h', 'i', 'Enter']) + }) + + it('hovers by walking the pointer over and leaving it there', async () => { + const send = withDrivenPane() + + const result = await actOnActivePreview({ kind: 'hover', ref: '@e1' }) + const types = sentTypes(send) + + expect(types).toContain('mouseMove') + // The whole request is "be on it" — a click here would open the dropdown the + // agent was trying to reveal, or worse, activate it. + expect(types).not.toContain('mouseDown') + expect(result.acted).toBe('hovered over button "Save"') + }) + + // The witness only speaks for verbs that put the button down. Demanding one + // from a key press reported every press as a failure. + it('does not expect a click witness from a verb that never clicks', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at textbox "Search"', point: { x: 40, y: 20 }, success: true }) + : JSON.stringify({ elements: [], hit: null, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + for (const action of [ + { key: 'Escape', kind: 'press', ref: '@e1' }, + { kind: 'hover', ref: '@e1' } + ]) { + expect(await actOnActivePreview(action)).toMatchObject({ success: true }) + } + }) + + it('scrolls by wheeling for real, not by scripting the page', async () => { + const tabId = openBrowserTab() + const send = vi.fn() + let scripted = false + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => { + scripted ||= code.includes('"kind":"scroll"') + + return code.includes('scrollHeight') + ? JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 4_000, success: true }) + : JSON.stringify({ elements: [], success: true }) + }) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send })) + + await actOnActivePreview({ kind: 'scroll' }) + + const wheels = send.mock.calls.map(([event]) => event).filter(event => event.type === 'mouseWheel') + + // A stream of notches, not one delta: scroll-linked headers and lazy loaders + // only react to the events, so a scripted scrollBy leaves them asleep. + expect(wheels.length).toBeGreaterThan(1) + // Electron's wheel delta is wheelDelta-signed, so scrolling DOWN is negative. + expect(wheels.every(event => event.deltaY < 0)).toBe(true) + expect(scripted).toBe(false) + }) + + it('says so plainly when the page has nothing to scroll', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async () => + JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 0, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + expect(await actOnActivePreview({ kind: 'scroll' })).toMatchObject({ note: expect.stringContaining('nothing to scroll') }) + }) + + it('fails loudly when the pointer input never reaches the page', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true }) + : JSON.stringify({ elements: [], hit: null, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + // Everything else about this action travels on the script channel and would + // report success whether or not a single event landed. + expect(result.success).toBe(false) + expect(result.error).toContain('never reached the page') + }) + + it('says so when the overlay itself swallowed the click', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true }) + : JSON.stringify({ elements: [], hit: { tag: 'HERMES-WATCH', trusted: true }, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).note).toContain('overlay intercepted') + }) + + it('falls back to scripted events when the pane exposes no input channel', async () => { + let injected = '' + + withRunner(async code => { + injected = code + + return JSON.stringify({ acted: 'clicked', success: true }) + }) + + await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + // The one-trip shape: the engine both acts and re-reads, no locate handshake. + expect(injected).toContain('"kind":"click"') + expect(injected).not.toContain('"kind":"locate"') + }) + it('routes history verbs to the pane instead of the guest page', async () => { const back = vi.fn() const runner = vi.fn() diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.ts b/apps/desktop/src/app/chat/right-rail/preview-act.ts index 495ce45f0a..bcef505fbc 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-act.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-act.ts @@ -1,14 +1,23 @@ /** * PREVIEW ACT — performs the agent's interactions inside the preview pane's - * guest page, so `act_preview` drives whatever web app is open in the in-app + * guest page, so `drive_preview` drives whatever web app is open in the in-app * browser. * * The guest page is out-of-process; nothing here can touch its DOM directly. - * Instead each action injects the engine SOURCE over `executeJavaScript` (see - * lib/preview-act/act-in-page.ts's self-containment contract), parked on a - * window global alongside the ref holder so '@e5' still resolves on the next - * call. Injection is idempotent and vanishes with the page — a navigation - * drops the refs, which the engine reports rather than clicking the wrong node. + * Two separate channels reach it, and the split matters: + * + * - `executeJavaScript` injects the engine SOURCE (see + * lib/preview-act/act-in-page.ts's self-containment contract) to RESOLVE + * and READ — turn '@e5' into a node, measure it, inventory the page. It is + * parked on a window global alongside the ref holder so refs survive + * between calls, and it vanishes with the page, so a navigation drops the + * refs and the engine reports that instead of clicking the wrong node. + * - `sendInputEvent` (preview-drive.ts) does the ACTING, as real Chromium + * input. Script can only dispatch synthetic events, which the page can tell + * apart and which never move the browser's own hover or focus target. + * + * Panes with no input channel — a remote HTML preview — fall back to the + * engine's synthetic events, which is worse but still works. * * A mutating action settles briefly and returns a fresh inventory, so the * click → re-read → click loop costs one round trip instead of two. @@ -18,46 +27,461 @@ */ import { actInPage, type PreviewActAction, type PreviewActResult } from '@/lib/preview-act/act-in-page' +import { watchInPage } from '@/lib/preview-act/watch-in-page' +import { clickAt, glideTo, pointerPlaced, pressKey, selectAll, typeText, wheelBy } from './preview-drive' +import { activePreviewInput, type PreviewInputHandle } from './preview-input' import { activePreviewNav, type PreviewNavHandle } from './preview-nav' -import { activePreviewScriptRunner } from './preview-script-runner' +import { activePreviewScriptRunner, type PreviewScriptRunner } from './preview-script-runner' /** Verbs the pane owns; a guest page cannot drive its own history. */ const NAV_ACTIONS: readonly (keyof PreviewNavHandle)[] = ['back', 'forward', 'reload'] /** How long a click/type is given to land before the page is re-inventoried. - * Long enough for a framework re-render, short enough not to stall the turn. */ -const SETTLE_MS = 400 + * Long enough for a framework re-render, short enough not to stall the turn. + * A render that misses this window is not lost — the NEXT action re-reads the + * page anyway, so the cost of being early is a slightly stale inventory, not a + * wrong one. That trade is worth several hundred ms on every single step. */ +const SETTLE_MS = 220 + +/** Cap on one round trip into the guest page. A click that starts a navigation + * tears the document down mid-settle, so the injected promise dies with it and + * `executeJavaScript` never settles — without this the tool eats its whole + * 45s bridge deadline for what was actually a successful click. */ +const ACT_TIMEOUT_MS = 8_000 + +/** Verbs that get the look-then-act treatment: locate the target, walk the + * pointer there, then send real input. Actions with no single target (scroll, + * elements) are absent and skip it. */ +const DRIVEN: readonly string[] = ['click', 'hover', 'press', 'type'] + +/** Verbs that put the mouse button down, and so are the only ones the pointerdown + * witness can speak to. `press` and `hover` never click, so demanding a witness + * from them would report every one of them as a failure. */ +const CLICKS: readonly string[] = ['click', 'type'] const NOTHING_OPEN = 'No live page is open in the in-app browser — open one with open_preview first.' -/** Build the idempotent inject-and-run script for one action. */ -function buildActScript(action: PreviewActAction, settleMs: number): string { +const NAVIGATED = 'The page stopped answering right after — it is probably navigating. Call elements to see where you landed.' + +/** A fingerprint of the overlay's source, so the guest page can tell that the + * code it is running has changed underneath it. + * + * It needs one because the overlay memoizes its own chrome: the cursor, the + * lock frame and the scan line are built ONCE per page and then reused, so an + * edit to any of their styles lands in the injected source, gets injected, and + * changes nothing — the layers it would have styled were built by the previous + * version and are still on the page. In dev that reads as "HMR didn't pick up + * my CSS"; in production it is a long-lived tab pinned to whichever build first + * touched it, across an app upgrade. + * + * Content-addressed rather than a length or a version literal: the change that + * exposed this was one hex colour for another, which is byte-for-byte the same + * length, and a hand-maintained version is a step someone forgets. */ +const WATCH_TAG = (() => { + const source = watchInPage.toString() + let hash = 5381 + + for (let i = 0; i < source.length; i++) { + hash = ((hash << 5) + hash + source.charCodeAt(i)) | 0 + } + + return hash +})() + +/** Injected once per page: the engine, the overlay, and the two helpers every + * script below shares. Idempotent — re-running it on a live page is a no-op. */ +const preamble = () => ` var w = window; + // Reassigned on every trip rather than cached behind a guard: the source is in + // this payload either way, so the guard saved nothing and pinned a long-lived + // tab to whichever build first touched it. The holder is the exception — it + // carries the aimed element from the locate trip to the act trip. + w.__hermesActHolder = w.__hermesActHolder || {}; + w.__hermesAct = (${actInPage.toString()}); + w.__hermesWatch_fn = (${watchInPage.toString()}); + w.__hermesWatchTag = ${WATCH_TAG}; + var holder = w.__hermesActHolder; + var act = function (a) { return w.__hermesAct(document, holder, a); }; + var watch = function (stage, label) { + try { w.__hermesWatch_fn(document, holder, stage, label); } catch (err) {} + }; + var wait = function (ms) { return new Promise(function (resolve) { setTimeout(resolve, ms); }); }; + // Sit out a smooth scroll so the pointer is aimed at a target that has stopped + // moving. A scroll that never started fires no scrollend, so only wait on one + // we can actually see happening; capture catches nested scrollers, whose + // scrollend does not bubble. + var restAfterScroll = function () { + return new Promise(function (resolve) { + var sc = document.scrollingElement || document.documentElement; + var x0 = sc ? sc.scrollLeft : 0; + var y0 = sc ? sc.scrollTop : 0; + requestAnimationFrame(function () { + if (!sc || (sc.scrollLeft === x0 && sc.scrollTop === y0)) { return resolve(); } + var done = false; + var finish = function () { + if (done) { return; } + done = true; + document.removeEventListener('scrollend', finish, true); + clearTimeout(timer); + resolve(); + }; + var timer = setTimeout(finish, 350); + document.addEventListener('scrollend', finish, true); + }); + }); + };` + +/** Bring the target on screen, mark it, and report where it came to rest. */ +function buildLocateScript(action: PreviewActAction, focus: boolean): string { + const locate = { focus, kind: 'locate', ref: action.ref, selector: action.selector } + return `(function () { - var w = window; - if (!w.__hermesAct) { - w.__hermesActHolder = {}; - w.__hermesAct = (${actInPage.toString()}); - } - var act = function (a) { return w.__hermesAct(document, w.__hermesActHolder, a); }; +${preamble()} + var locate = ${JSON.stringify(locate)}; + var found = act(locate); + if (!found.success) { return Promise.resolve(JSON.stringify(found)); } + watch('aim'); + // Arm a witness for the real input that is about to arrive. Without it a + // click that never reached the page is indistinguishable from one the page + // ignored, and the agent would report success either way. + w.__hermesHit = null; + document.addEventListener('pointerdown', function (e) { + w.__hermesHit = { tag: e.target ? e.target.tagName : '?', trusted: e.isTrusted === true }; + }, { capture: true, once: true }); + // Measure again once the scroll has stopped: real input is aimed at a fixed + // viewport coordinate, so it has to be where the target ENDS UP. + return restAfterScroll().then(function () { + var settled = act(locate); + var best = settled.success ? settled : found; + var at = best.point; + // Last line of defence before the pointer is sent somewhere real. An element + // that is still outside the viewport after we scrolled to it is hidden, not + // placed, and aiming at it would drive the cursor off into a corner and + // click whatever happens to be under that coordinate. + if (at && (at.x < 0 || at.y < 0 || at.x > window.innerWidth || at.y > window.innerHeight)) { + return JSON.stringify({ + error: 'That element is still off-screen after scrolling to it, so it is hidden rather than clickable. Call elements again for what is really on the page.', + success: false + }); + } + return JSON.stringify(best); + }); +})()` +} + +/** Put up a mark that outlives the action that made it. Every other cue on the + * overlay retires on a timer, which is right for narrating a click and no use + * at all for holding a finding on screen while the agent keeps working. */ +function buildPinScript(action: PreviewActAction, label: string): string { + const locate = { kind: 'locate', ref: action.ref, selector: action.selector } + + return `(function () { +${preamble()} + var found = act(${JSON.stringify(locate)}); + if (!found.success) { return JSON.stringify(found); } + watch('pin', ${JSON.stringify(label)}); + return JSON.stringify({ acted: 'pinned ' + String(found.acted || 'it').replace(/^looking at /, ''), success: true }); +})()` +} + +/** Freeze the whole visible field as marks. Needs a fresh inventory first, so + * the overlay has both the field to draw and the refs to number it by. */ +function buildHoldScript(): string { + return `(function () { +${preamble()} + var found = act({ kind: 'elements' }); + if (!found.success) { return JSON.stringify(found); } + watch('hold'); + return JSON.stringify({ acted: 'held the field', elements: found.elements, title: found.title, url: found.url, success: true }); +})()` +} + +/** Rattle through the field one box at a time. Needs a fresh inventory for the + * overlay to pick from, and nothing else — the page is never touched. */ +function buildStrobeScript(): string { + return `(function () { +${preamble()} + var found = act({ kind: 'elements' }); + if (!found.success) { return JSON.stringify(found); } + watch('strobe'); + return JSON.stringify({ acted: 'strobed the field', elements: found.elements, title: found.title, url: found.url, success: true }); +})()` +} + +/** Take one mark down, or — with nothing to aim at — all of them. */ +function buildUnpinScript(action: PreviewActAction): string { + const locate = { kind: 'locate', ref: action.ref, selector: action.selector } + const one = !!(action.ref || action.selector) + + return `(function () { +${preamble()} +${ + one + ? ` var found = act(${JSON.stringify(locate)}); + if (!found.success) { return JSON.stringify(found); } + var gone = 'unpinned ' + String(found.acted || 'it').replace(/^looking at /, '');` + : ` holder.aimed = null; + var gone = 'cleared every pin';` +} + watch('unpin'); + return JSON.stringify({ acted: gone, success: true }); +})()` +} + +/** Mark the hit, let the page react, and hand back a fresh inventory. */ +function buildFinishScript(settleMs: number): string { + return `(function () { +${preamble()} + watch('strike'); + return wait(${settleMs}).then(function () { + // A rescan that throws must still answer — an unresolved promise here costs + // the whole bridge deadline. + try { + var out = act({ kind: 'elements' }); + watch('sweep'); + out.hit = w.__hermesHit || null; + return JSON.stringify(out); + } catch (err) { + return JSON.stringify({ note: 'The page changed before it could be re-read: ' + err, success: true }); + } + }); +})()` +} + +/** The no-real-input fallback: the engine both acts and reads, in one trip. + * Both reads mark the field — the explicit `elements` call and the re-read + * every other verb does on its way out — so the page is marked after every + * single action rather than only when the agent asks for an inventory outright. + * They mark it differently, though, and the difference is what each one MEANS: + * an outright `elements` is the agent reading the whole page, which is the + * moment worth showing, so it strobes. The re-read after a click is + * housekeeping, and strobing there would put five seconds of noise between + * every step of a task, so it sweeps. The two never collide: an `elements` call + * settles in 0ms and returns before the re-read. */ +function buildScriptedScript(action: PreviewActAction, settleMs: number): string { + return `(function () { +${preamble()} var result = act(${JSON.stringify(action)}); - if (!result.success || ${settleMs} <= 0) { - return Promise.resolve(JSON.stringify(result)); + ${ + action.kind === 'elements' + ? "if (result.success) { watch('strobe'); }" + : '' } - // Re-inventory after the page has had a moment to react, so the agent's next - // ref is drawn from the DOM its own click produced. - return new Promise(function (resolve) { - setTimeout(function () { + if (!result.success || ${settleMs} <= 0) { return Promise.resolve(JSON.stringify(result)); } + return wait(${settleMs}).then(function () { + try { var after = act({ kind: 'elements' }); + watch('sweep'); result.elements = after.elements; result.url = after.url; result.title = after.title; - resolve(JSON.stringify(result)); - }, ${settleMs}); + } catch (err) { + result.note = 'The page changed before it could be re-read: ' + err; + } + return JSON.stringify(result); }); })()` } +/** The outcome of one round trip into the page. `silent` is its own case on + * purpose: a page that stops answering mid-action is navigating, whereas one + * that answers with nothing is broken, and the agent needs to hear the + * difference. */ +type Trip = { error: string; kind: 'failed' } | { kind: 'answered'; result: PreviewActResult } | { kind: 'silent' } + +async function runJson(run: PreviewScriptRunner, code: string): Promise { + const raw = await Promise.race([ + run(code).catch((error: unknown) => new Error(String(error))), + new Promise(resolve => setTimeout(resolve, ACT_TIMEOUT_MS)) + ]) + + if (raw === undefined) { + return { kind: 'silent' } + } + + if (raw instanceof Error) { + return { error: 'The page rejected the action: ' + raw.message, kind: 'failed' } + } + + if (typeof raw !== 'string' || !raw) { + return { error: 'The page did not answer the action.', kind: 'failed' } + } + + return { kind: 'answered', result: JSON.parse(raw) as PreviewActResult } +} + +/** Past tense of the verb the agent asked for, against what it actually hit. */ +function describeDone(action: PreviewActAction, target: string): string { + if (action.kind === 'type') { + return 'typed into ' + target + (action.submit ? ' and submitted' : '') + } + + if (action.kind === 'press') { + return 'pressed ' + (action.key || '') + ' on ' + target + } + + if (action.kind === 'hover') { + return 'hovered over ' + target + } + + return 'clicked ' + target +} + +/** Look at the target, walk the pointer over, and act on it for real. */ +async function driveAction( + run: PreviewScriptRunner, + input: PreviewInputHandle, + action: PreviewActAction +): Promise { + // A key press must not be preceded by a click — that would activate the + // control rather than type into it — so the page hands it focus instead. + const trip = await runJson(run, buildLocateScript(action, action.kind === 'press')) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + if (trip.kind === 'silent') { + return { acted: action.kind, note: NAVIGATED, success: true } + } + + const found = trip.result + + if (!found.success) { + return found + } + + if (!found.point) { + return { error: 'Could not work out where that element is on screen.', success: false } + } + + await glideTo(input, found.point) + + if (action.kind === 'click') { + await clickAt(input) + } else if (action.kind === 'type') { + if (found.typable === false) { + return { + error: `${String(found.acted || 'That').replace(/^looking at /, '')} is not a text field, so typing into it would only select the text under the pointer. Click it if it opens one, then type into that.`, + success: false + } + } + + input.focus() + await clickAt(input) + // Select-all inside the now-focused field, so typing replaces what is there + // the way it would for a person. NOT a triple-click: that is a pointer + // gesture and selects the paragraph under the cursor whenever the target + // turns out not to be a field. + await selectAll(input) + await typeText(input, action.text ?? '') + + if (action.submit) { + await pressKey(input, 'Enter') + } + } else if (action.kind === 'press') { + input.focus() + await pressKey(input, action.key || 'Enter') + } + // hover is the glide and nothing else — the pointer is already sitting on the + // target, which is the whole request. + + const target = String(found.acted || '').replace(/^looking at /, '') + const after = await runJson(run, buildFinishScript(SETTLE_MS)) + const acted = describeDone(action, target) + + // The action itself already happened as real input, so a page that will not + // answer the read-back is a page that navigated — never a failed click. + if (after.kind !== 'answered') { + return { acted, note: NAVIGATED, success: true } + } + + const { hit, ...result } = after.result as PreviewActResult & { hit?: { tag: string; trusted: boolean } | null } + + // The witness the locate trip armed. No record means the input never reached + // the document, which the agent must hear about — every other signal here + // travels on the script channel and would report success regardless. + if (!hit && CLICKS.indexOf(action.kind) !== -1) { + return { + ...result, + error: 'The pointer input never reached the page, so nothing was ' + acted.split(' ')[0] + '.', + success: false + } + } + + return { ...result, acted, note: hitNote(hit), success: true } +} + +/** Flag a click the overlay intercepted, which would otherwise look like a page + * that simply ignored it. */ +function hitNote(hit?: { tag: string; trusted: boolean } | null): string | undefined { + return hit && hit.tag === 'HERMES-WATCH' + ? 'The action overlay intercepted the click instead of the page.' + : undefined +} + +/** How far a screenful is, whether there is anywhere to go, and a spot to wheel + * over — the renderer cannot see the guest viewport to work any of it out. */ +function buildScrollAnchorScript(): string { + return `(function () { +${preamble()} + var sc = document.scrollingElement || document.documentElement; + var track = sc.clientHeight || window.innerHeight; + return Promise.resolve(JSON.stringify({ + page: Math.round(window.innerHeight * 0.9), + point: { x: Math.round(window.innerWidth / 2), y: Math.round(track / 2) }, + span: sc.scrollHeight - track, + success: true + })); +})()` +} + +/** Scroll the page the way a hand does — wheel it. The scripted path calls + * `scrollBy`, which moves the page without the page ever seeing an input + * event, so scroll-linked headers, reveal animations and infinite-scroll + * loaders all stay asleep through a scroll that looked like it happened. */ +async function driveScroll( + run: PreviewScriptRunner, + input: PreviewInputHandle, + action: PreviewActAction, +): Promise { + const far = action.amount ?? 0 + const trip = await runJson( + run, + buildScrollAnchorScript() + ) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + if (trip.kind === 'silent') { + return { acted: 'scrolled', note: NAVIGATED, success: true } + } + + const anchor = trip.result as PreviewActResult & { page?: number; span?: number } + + if (!anchor.span) { + return { ...anchor, acted: 'scrolled the page', note: 'The page has nothing to scroll — it all fits already.' } + } + + // A person does not move the mouse to scroll; the wheel turns wherever their + // hand already is. Only send it somewhere if it has never been anywhere. + if (!pointerPlaced() && anchor.point) { + await glideTo(input, anchor.point) + } + + await wheelBy(input, action.amount ?? anchor.page ?? 600) + + const after = await runJson(run, buildFinishScript(SETTLE_MS)) + + if (after.kind !== 'answered') { + return { acted: 'scrolled the page', note: NAVIGATED, success: true } + } + + return { ...after.result, acted: 'scrolled the page', success: true } +} + /** Run one action against the ACTIVE preview tab's page. `kind` is a bare * string: the verb arrives off the wire, and the history ones never reach * the in-page engine. */ @@ -87,12 +511,57 @@ export async function actOnActivePreview( } const typed = action as PreviewActAction - const settle = typed.kind === 'elements' ? 0 : SETTLE_MS - const raw = await run(buildActScript(typed, settle)) - if (typeof raw !== 'string' || !raw) { - return { error: 'The page did not answer the action.', success: false } + // Annotation, not interaction: nothing is clicked, nothing settles, and the + // page is not re-read, so these skip the whole act-then-inventory path. + if (typed.kind === 'pin' || typed.kind === 'unpin' || typed.kind === 'hold' || typed.kind === 'strobe') { + const mark = + typed.kind === 'hold' + ? buildHoldScript() + : typed.kind === 'strobe' + ? buildStrobeScript() + : typed.kind === 'pin' + ? buildPinScript(typed, typed.text || '') + : buildUnpinScript(typed) + const trip = await runJson(run, mark) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + return trip.kind === 'answered' ? trip.result : { acted: typed.kind, note: NAVIGATED, success: true } } - return JSON.parse(raw) as PreviewActResult + const input = activePreviewInput() + + if (input && DRIVEN.indexOf(typed.kind) !== -1) { + return driveAction(run, input, typed) + } + + // A plain page scroll is a wheel gesture. Jumping to an end is not — no hand + // wheels to the bottom of a long article — so `to` stays a scripted glide. + const plain = typed.kind === 'scroll' && !typed.to && !typed.ref && !typed.selector + + if (plain && input) { + return driveScroll(run, input, typed) + } + + const settle = typed.kind === 'elements' ? 0 : SETTLE_MS + const scripted = await runJson(run, buildScriptedScript(typed, settle)) + + if (scripted.kind === 'failed') { + return { error: scripted.error, success: false } + } + + // The action almost certainly landed — a page that stops answering right + // after a click is one that navigated. Say so instead of failing it. + return scripted.kind === 'silent' ? { acted: typed.kind, note: NAVIGATED, success: true } : scripted.result +} + +// Self-accept so an edit here, or to the in-page sources this module +// stringifies, doesn't reload the whole renderer out from under a live session. +// The bridge re-requests this module per action in dev, so it picks the change +// up without one (see loadPreviewEngine in gateway-event/desktop-bridge.ts). +if (import.meta.hot) { + import.meta.hot.accept() } diff --git a/apps/desktop/src/app/chat/right-rail/preview-mind.ts b/apps/desktop/src/app/chat/right-rail/preview-mind.ts new file mode 100644 index 0000000000..3f7a1856c4 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-mind.ts @@ -0,0 +1,29 @@ +/** + * IDLE PULSE — keeps the preview overlay alive while the model reasons. + * + * Everything else the overlay draws is narration of an action, so the pane goes + * dark for the gap between actions — which is most of the time the agent spends + * on a page, and the part where it is deciding what to do with what it just + * read. A page that stops responding the moment the agent is thinking hardest + * reads as the agent having wandered off. + * + * This is deliberately NOT part of the act pipeline: a turn boundary only has + * to poke the overlay the last action left behind. See preview-nudge.ts. + */ + +import { $busy } from '@/store/session' + +import { nudgeOverlay } from './preview-nudge' + +// Module-level, matching how review.ts and coding-status.ts watch this edge. It +// is inert without a live pane, so there is nothing to mount or tear down. +let running = $busy.get() + +$busy.subscribe(busy => { + if (busy === running) { + return + } + + running = busy + nudgeOverlay(busy ? 'think' : 'rest') +}) diff --git a/apps/desktop/src/app/chat/right-rail/preview-nudge.ts b/apps/desktop/src/app/chat/right-rail/preview-nudge.ts new file mode 100644 index 0000000000..8a8f419987 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-nudge.ts @@ -0,0 +1,38 @@ +/** + * OVERLAY NUDGE — the one way to say a stage to an overlay that is ALREADY on + * the page. + * + * The act pipeline (preview-act.ts) injects the whole engine because it has to: + * it needs to resolve a ref, measure a node and read the page back. The stages + * that only narrate — the idle pulse, the read wipe — need none of that. They + * are one function call against something the last action already left on the + * page, so re-shipping the engine to make it would put the payload on the wire + * on every turn boundary. + * + * A page the agent has never acted on has no overlay, and the nudge is a no-op + * there. That is the right answer rather than a gap: chrome on a page the agent + * never touched would be a lie about what it did. + */ + +import type { WatchStage } from '@/lib/preview-act/watch-in-page' + +import { activePreviewScriptRunner } from './preview-script-runner' + +/** Run one stage against the active pane's overlay, if it has one. */ +export function nudgeOverlay(stage: WatchStage): void { + const run = activePreviewScriptRunner() + + if (!run) { + return + } + + void run(`(function () { + var w = window; + var fn = w.__hermesWatch_fn; + if (!fn) { return 'cold'; } + try { fn(document, w.__hermesActHolder || {}, ${JSON.stringify(stage)}); } catch (err) {} + return 'ok'; +})()`).catch(() => { + // The page navigated out from under us. The next action re-injects. + }) +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx index 076fc4a94f..817cec638a 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx +++ b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx @@ -1,3 +1,7 @@ +// Side-effect import: watches the turn edge so the overlay keeps a pulse while +// the model reasons. Lives here because the pane is what makes it reachable. +import './preview-mind' + import { useStore } from '@nanostores/react' import type { PointerEvent as ReactPointerEvent } from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' @@ -28,7 +32,7 @@ import { import { type ConsoleEntry } from './preview-console-state' import { previewConsoleState } from './preview-console-store' import { LocalFilePreview, PreviewEmptyState } from './preview-file' -import { registerPreviewInput, type PreviewInputEvent } from './preview-input' +import { type PreviewInputEvent, registerPreviewInput } from './preview-input' import { PREVIEW_BROWSER_ATTR, registerPreviewNav } from './preview-nav' import { registerPreviewPageReader } from './preview-reader' import { registerPreviewScriptRunner } from './preview-script-runner' diff --git a/apps/desktop/src/app/chat/right-rail/preview-reader.ts b/apps/desktop/src/app/chat/right-rail/preview-reader.ts index 7f48c8cbb1..97a372db18 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-reader.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-reader.ts @@ -15,6 +15,8 @@ import { $rightRailActiveTabId } from '@/store/layout' import { $previewTabs } from '@/store/preview' +import { nudgeOverlay } from './preview-nudge' + export interface PreviewReadOptions { /** Characters to return from `start` (capped at PREVIEW_READ_MAX_CHARS). */ count?: number @@ -89,6 +91,13 @@ export async function readActivePreview(opts: PreviewReadOptions = {}): Promise< try { const page = await reader() + // Say it on the page. Reading is by far the cheapest thing the agent + // does — a few hundredths of a second against a model round trip either + // side of it — so a run of reads used to leave the pane dark for the + // twenty seconds it took to page through a document, immediately after + // the one moment that showed anything. + nudgeOverlay('read') + return windowText( { kind: target.kind, path: target.path, title: page.title || target.label, url: page.url || target.url }, page.text, diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts index 1a8cee9d02..227a4eb128 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts @@ -11,6 +11,34 @@ import { setMessages } from '@/store/session' import type { GatewayEventContext } from './types' +/** The preview engine, loaded on demand so ~25KB of page-injectable source stays + * off the boot path. + * + * In dev that lazy chunk is also a trap. The browser caches a dynamic import by + * URL for the life of the page, so this bridge would hand every action to + * whichever build of the engine loaded first, and no edit to it — or to the + * overlay whose source it stringifies into the page — would reach the guest + * until the whole window reloaded. Asking for a fresh copy is more reliable + * than trusting hot-update propagation to reach a module nothing statically + * imports; the dev server stamps the dependency URLs it has invalidated, so a + * fresh engine pulls a fresh overlay down with it. + * + * The literal path is what a bare specifier can't be here, and it has to track + * this module's real location — hence the fall back to the static import, which + * is also the only branch production keeps, `import.meta.hot` being stripped + * there along with everything it guards. */ +const loadPreviewEngine = () => { + const stable = () => import('@/app/chat/right-rail/preview-act') + + if (!import.meta.hot) { + return stable().then(mod => mod.actOnActivePreview) + } + + return import(/* @vite-ignore */ '/src/app/chat/right-rail/preview-act.ts?hot=' + Date.now()) + .catch(stable) + .then(mod => mod.actOnActivePreview as Awaited>['actOnActivePreview']) +} + /** Desktop-surface bridge events: read-back requests the agent blocks on * (terminal/preview/window), agent terminal streaming, pane reveal, and * message reactions. */ @@ -57,7 +85,7 @@ export function handleDesktopBridgeEvent(ctx: GatewayEventContext): boolean { } if (event.type === 'preview.act.request') { - // act_preview tool: click/type/scroll/press inside the guest page, or + // drive_preview tool: click/type/scroll/press inside the guest page, or // drive the pane's history. Dynamic import keeps the injected engine off // the boot path. Active session only: a background turn must never reach // into the page the user is working in (desktop AGENTS.md: offer, don't @@ -72,9 +100,9 @@ export function handleDesktopBridgeEvent(ctx: GatewayEventContext): boolean { }) if (isActiveEvent) { - void import('@/app/chat/right-rail/preview-act') - .then(({ actOnActivePreview }) => - actOnActivePreview({ + void loadPreviewEngine() + .then(run => + run({ amount: payload?.amount, key: payload?.key, kind: payload?.action ?? '', diff --git a/apps/desktop/src/lib/preview-act/act-in-page.test.ts b/apps/desktop/src/lib/preview-act/act-in-page.test.ts index aa1985fb7a..5fafdb91e8 100644 --- a/apps/desktop/src/lib/preview-act/act-in-page.test.ts +++ b/apps/desktop/src/lib/preview-act/act-in-page.test.ts @@ -3,14 +3,17 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import { actInPage, type PreviewActHolder } from './act-in-page' /** jsdom lays nothing out, so every rect is 0×0 and the engine's visibility - * check would reject the whole page. Give elements a plausible box and let + * check would reject the whole page. Give elements a plausible box, honouring + * an explicit width/height so a test can lay out a 1px one, and let * `display: none` (which jsdom DOES compute) carry the hiding. */ function layOutTheDocument() { vi.spyOn(Element.prototype, 'getBoundingClientRect').mockImplementation(function (this: Element) { - const hidden = getComputedStyle(this).display === 'none' - const size = hidden ? 0 : 40 + const style = getComputedStyle(this) + const width = style.display === 'none' ? 0 : parseFloat(style.width) || 40 + const height = style.display === 'none' ? 0 : parseFloat(style.height) || 40 + const left = parseFloat(style.left) || 0 - return { bottom: size, height: size, left: 0, right: size, top: 0, width: size, x: 0, y: 0 } as DOMRect + return { bottom: height, height, left, right: left + width, top: 0, width, x: left, y: 0 } as DOMRect }) } @@ -29,6 +32,33 @@ beforeEach(() => { vi.restoreAllMocks() layOutTheDocument() Element.prototype.scrollIntoView = vi.fn() + // jsdom has no hit-testing at all, so the engine skips the occlusion check + // here by default. Tests that stand one in must not leak it into the next. + delete (document as Document & { elementFromPoint?: unknown }).elementFromPoint +}) + +describe('self-containment', () => { + // The pane injects `actInPage.toString()` into the guest page, where module + // scope does not exist. A single free identifier (a module-level constant, an + // imported helper) is a ReferenceError on every call, and Electron reports it + // only as "Script failed to execute" — so evaluate the source the way the + // guest does and run a real action through it. + it('runs after being stringified and eval’d with no module scope', () => { + const page = document.createElement('div') + page.innerHTML = '' + document.body.replaceChildren(page) + + const injected = new Function('return (' + actInPage.toString() + ')')() as typeof actInPage + const holder: PreviewActHolder = {} + + expect(injected(document, holder, { kind: 'elements' }).elements?.[0].label).toBe('Save') + + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + expect(injected(document, holder, { kind: 'click', ref: '@e1' }).success).toBe(true) + expect(clicked).toHaveBeenCalledOnce() + }) }) describe('elements', () => { @@ -73,6 +103,82 @@ describe('elements', () => { expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Real']) }) + // The screen-reader-only recipe is a 1px box parked at the document origin, + // and it passed a `width >= 1` check. Skip links and CSS-only menu toggles are + // built this way and come FIRST in the document, so they landed on @e1 — the + // agent would aim at one and the pointer would fly to the top-left corner and + // click nothing. + it('skips the visually-hidden controls that sit at the document origin', () => { + const holder = page(` + Jump to content + + Skip navigation + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('skips a control parked off the left edge, which no scroll brings back', () => { + const holder = page(` + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + // Wikipedia and friends are full of these: decorative chrome and collapsed + // menus that are perfectly solid boxes as far as layout is concerned, but + // that the page has already declared are not for anyone to interact with. + it('skips controls the page marked aria-hidden or inert', () => { + const holder = page(` + +
+ + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('skips a control buried under another layer, which would take the click instead', () => { + const holder = page(` + + + `) + const wall = document.createElement('div') + document.body.append(wall) + + // Stand in for the hit-testing jsdom does not do: every element reports + // itself except the buried one, which reports the sheet lying over it. + document.elementFromPoint = (x: number) => (x === 120 ? wall : document.getElementById('real')) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('keeps a control whose own child is what the hit test lands on', () => { + const holder = page(' Home') + + document.elementFromPoint = () => document.getElementById('icon') + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Home']) + }) + + // The overlay draws far more than the agent is told about, so the two lists + // are collected separately: the field is every visible control, the inventory + // is the describable subset that refs point into. + it('parks the whole interactive field for the overlay, not just the inventory', () => { + const holder = page(` + + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + expect(holder.nodes).toEqual([document.getElementById('named')]) + expect(holder.field).toEqual([document.getElementById('named'), document.getElementById('mystery')]) + }) + it('prefers an identity selector so the agent can re-find the node later', () => { const holder = page(`
@@ -288,7 +394,23 @@ describe('scroll', () => { const result = actInPage(document, holder, { kind: 'scroll' }) expect(result.success).toBe(true) - expect(scrollBy).toHaveBeenCalledWith({ behavior: 'auto', top: Math.round(window.innerHeight * 0.9) }) + expect(scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: Math.round(window.innerHeight * 0.9) }) + }) + + it('drops the animation for a reader who asked for reduced motion', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + // Passing 'smooth' explicitly overrides the OS preference, so the engine has + // to opt back out itself rather than leaving it to the browser. + vi.stubGlobal('matchMedia', () => ({ matches: true })) + + actInPage(document, holder, { kind: 'scroll' }) + vi.unstubAllGlobals() + + const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions] + + expect(options.behavior).toBe('auto') }) it('jumps to the bottom', () => { @@ -312,7 +434,7 @@ describe('scroll', () => { const result = actInPage(document, holder, { amount: 200, kind: 'scroll', ref: '@e1' }) - expect(list.scrollBy).toHaveBeenCalledWith({ behavior: 'auto', top: 200 }) + expect(list.scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: 200 }) expect(pageScroll).not.toHaveBeenCalled() expect(result.acted).toContain('Results') }) diff --git a/apps/desktop/src/lib/preview-act/act-in-page.ts b/apps/desktop/src/lib/preview-act/act-in-page.ts index f664f58299..0d78fccf2a 100644 --- a/apps/desktop/src/lib/preview-act/act-in-page.ts +++ b/apps/desktop/src/lib/preview-act/act-in-page.ts @@ -35,7 +35,24 @@ export interface PreviewActAction { /** scroll distance in px. Defaults to ~90% of the viewport height. */ amount?: number key?: string - kind: 'click' | 'elements' | 'press' | 'scroll' | 'type' + /** `pin`/`unpin`/`hold` never reach the engine — they resolve their targets + * through `locate`/`elements` and then talk to the overlay — but they arrive + * on the same wire. */ + kind: + | 'click' + | 'elements' + | 'hold' + | 'hover' + | 'locate' + | 'pin' + | 'press' + | 'scroll' + | 'strobe' + | 'type' + | 'unpin' + /** locate: also give the target keyboard focus, for a key press that must not + * be preceded by a click (which would activate the control instead). */ + focus?: boolean /** Cap on the returned inventory. */ max?: number ref?: string @@ -52,8 +69,12 @@ export interface PreviewActResult { elements?: PreviewElement[] error?: string note?: string + /** Viewport centre of a located target, for aiming real pointer input at it. */ + point?: { x: number; y: number } success: boolean title?: string + /** locate: whether the target actually takes typed text. */ + typable?: boolean /** Live document URL after the action — a change means it navigated. */ url?: string } @@ -61,18 +82,34 @@ export interface PreviewActResult { /** Where the surface keeps the last snapshot between actions (a window global * in the preview page), so '@e5' still means something on the next call. */ export interface PreviewActHolder { + /** Target of the action in flight, for the watch overlay to draw onto. */ + aimed?: Element | null + /** The on-screen subset, for the overlay to outline. Diverges from `nodes` in + * both directions: it drops what is below the fold, and it is not capped at + * the inventory's size. */ + field?: Element[] nodes?: Element[] /** URL the snapshot was taken on; a navigation invalidates every ref. */ url?: string } -const ACT_MAX_ELEMENTS = 120 - /** Run one action against `doc`, resolving refs through `holder`. Self-contained. */ export function actInPage(doc: Document, holder: PreviewActHolder, action: PreviewActAction): PreviewActResult { + // Declared inside, not at module scope: this body is stringified and eval'd + // in the guest page, where a module-level constant is simply not defined. + const maxElements = 120 + // How many nodes the OVERLAY may draw, as opposed to how many the agent gets + // told about. Far higher, because an extra mark costs one rect read where an + // extra inventory row costs tokens on every single call. + const maxMarks = 600 const win = doc.defaultView const here = doc.location ? doc.location.href : '' + // Passing 'smooth' explicitly OVERRIDES the user's OS-level reduce-motion + // setting, so ask before animating over someone who opted out. + const still = !!(win && win.matchMedia && win.matchMedia('(prefers-reduced-motion: reduce)').matches) + const glide: ScrollBehavior = still ? 'auto' : 'smooth' + const cssEscape = (value: string) => typeof CSS !== 'undefined' && CSS.escape ? CSS.escape(value) : value.replace(/["\\]/g, '\\$&') @@ -137,17 +174,118 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ return path.join(' > ') } - const visible = (el: Element): boolean => { + /** On screen right now, and worth drawing a box around. The field is strictly + * viewport-bound: a mark past the fold is one nobody can see, it spends the + * budget that should have gone to what IS on screen, and the ones parked far + * off to the side were landing as stray labels in the corner. */ + const onScreen = (el: Element): boolean => { + if (!win) { + return false + } + + const box = el.getBoundingClientRect() + + if (box.right <= 0 || box.bottom <= 0 || box.left >= win.innerWidth || box.top >= win.innerHeight) { + return false + } + + // Page-sized wrappers. Outlining one draws a rectangle around the whole + // screen, which says nothing and frames everything inside it as if it + // mattered. Full-width banners are fine — it takes both dimensions. + return !(box.width >= win.innerWidth * 0.95 && box.height >= win.innerHeight * 0.9) + } + + /** Painted at all, ancestors included. */ + const shown = (el: Element): boolean => { + // Folds in display/visibility/opacity/content-visibility inherited from an + // ANCESTOR, which this element's own computed style does not report: inside + // a parent at opacity 0, every child still reads back opacity 1. + const seen = (el as Element & { checkVisibility?: (opts: object) => boolean }).checkVisibility + + if (typeof seen === 'function' && !seen.call(el, { checkOpacity: true, checkVisibilityCSS: true })) { + return false + } + const style = win && win.getComputedStyle ? win.getComputedStyle(el) : null - if (style && (style.display === 'none' || style.visibility === 'hidden' || style.opacity === '0')) { + if (style) { + if (style.display === 'none' || style.visibility === 'hidden' || style.opacity === '0') { + return false + } + + // The screen-reader-only recipe, both spellings: the deprecated `clip` and + // the modern `clip-path: inset(50%)`. Neither exists for any purpose other + // than hiding something visually while keeping it focusable. + if ((style.clip && style.clip !== 'auto') || (style.clipPath || '').indexOf('inset(50%') !== -1) { + return false + } + } + + return true + } + + const visible = (el: Element): boolean => { + if (!shown(el)) { + return false + } + + // Things the page itself declares are not for interacting with, neither of + // which shows up in a computed style or a bounding box: an aria-hidden + // subtree is invisible to every other assistive client, and `inert` cannot + // be clicked at all. Disabled controls are deliberately NOT filtered here — + // they stay in the inventory so the agent is told a button is disabled + // rather than hunting for one that appears not to exist. + if (el.closest('[aria-hidden="true"], [inert]')) { return false } const rect = el.getBoundingClientRect() - // Off-screen is fine (we scroll to it); collapsed to nothing is not. - return rect.width >= 1 && rect.height >= 1 + // Below the fold is fine — we scroll to it. Off to the LEFT or ABOVE is the + // other half of the sr-only trick (`left: -9999px`, `top: -9999px`) and + // scrolling never brings those back. + if (rect.right <= 0 || rect.bottom <= 0) { + return false + } + + // Bigger than the 1px box sr-only collapses to. Nothing a person can aim at + // is 2px across, and admitting those is what put the cursor in the corner: + // they cluster at the document origin, so they sort to the FRONT of the + // inventory and the agent reaches for one as @e1. + if (rect.width < 3 || rect.height < 3) { + return false + } + + // Occlusion, and the last filter for a reason: everything above is cheap + // and this one forces layout. If the middle of the element is on screen, + // whatever the browser reports at that point has to BE this element — its + // own descendant (an icon inside a link) or the box it paints inside are + // fine, anything else means it is buried under a sticky header, a cookie + // wall, or a transparent layer the page put on top. Those are exactly the + // targets the agent aims at and then misses. Elements below the fold cannot + // be hit-tested from here, so they are taken on trust and land back in this + // function after the scroll. + const midX = rect.left + rect.width / 2 + const midY = rect.top + rect.height / 2 + const under = (doc as Document & { elementFromPoint?: (x: number, y: number) => Element | null }) + .elementFromPoint + + if ( + typeof under === 'function' && + win && + midX >= 0 && + midY >= 0 && + midX < win.innerWidth && + midY < win.innerHeight + ) { + const over = under.call(doc, midX, midY) + + if (!over || !(over === el || el.contains(over) || over.contains(el))) { + return false + } + } + + return true } const valueOf = (el: Element): string => { @@ -166,6 +304,7 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ const collect = (max: number): PreviewElement[] => { const nodes: Element[] = [] + const field: Element[] = [] const elements: PreviewElement[] = [] const candidates = doc.querySelectorAll( @@ -177,7 +316,22 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ ) for (const el of candidates) { - if (elements.length >= max || nodes.indexOf(el) !== -1 || !visible(el)) { + if (elements.length >= max && field.length >= maxMarks) { + break + } + + if (!visible(el)) { + continue + } + + // Drawable from here on. The two lists diverge deliberately: the field is + // what is on screen to be outlined, the inventory is what the agent can + // name and reach — which includes things below the fold it will scroll to. + if (field.length < maxMarks && onScreen(el)) { + field.push(el) + } + + if (elements.length >= max) { continue } @@ -212,6 +366,7 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ } holder.nodes = nodes + holder.field = field holder.url = here return elements @@ -279,7 +434,7 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ } if (action.kind === 'elements') { - const elements = collect(Math.max(1, Math.min(action.max || ACT_MAX_ELEMENTS, ACT_MAX_ELEMENTS))) + const elements = collect(Math.max(1, Math.min(action.max || maxElements, maxElements))) return answer({ elements, @@ -300,9 +455,9 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ const by = action.to === 'top' ? -1e7 : action.to === 'bottom' ? 1e7 : (action.amount ?? page) if (scroller) { - scroller.scrollBy({ behavior: 'auto', top: by }) + scroller.scrollBy({ behavior: glide, top: by }) } else if (win) { - win.scrollBy({ behavior: 'auto', top: by }) + win.scrollBy({ behavior: glide, top: by }) } return answer({ acted: scroller ? 'scrolled ' + describe(scroller) : 'scrolled the page', success: true }) @@ -316,12 +471,63 @@ export function actInPage(doc: Document, holder: PreviewActHolder, action: Previ const el = target.el as HTMLElement - if ((el as HTMLInputElement).disabled) { - return fail(describe(el) + ' is disabled.') + // Park the target where the watch overlay can find it, before anything can + // fail — a ring around the thing that turned out to be disabled is exactly + // the feedback someone watching wants. + holder.aimed = el + + // 'locate' is the look-before-you-act half of an action: it brings the target + // on screen and names it, so the overlay has something to draw while the real + // verb is still a beat away. + if (action.kind === 'locate') { + if (el.scrollIntoView) { + el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' }) + } + + if (action.focus) { + el.focus() + } + + const spot = pointAt(el) + const tag = el.tagName + + return answer({ + acted: 'looking at ' + describe(el), + point: { x: spot.clientX, y: spot.clientY }, + success: true, + // Real typing starts with a triple-click to clear the field. On anything + // that is not a field that gesture selects the paragraph under it + // instead, which is how the agent ended up highlighting whole pages. + typable: tag === 'TEXTAREA' || tag === 'INPUT' || el.isContentEditable === true + }) } + // Instant, unlike the `scroll` verb above. Getting a target on screen is + // plumbing for the click that follows, not something anyone asked to watch — + // and every millisecond of it is time the caller spends waiting for the page + // to stop moving before it can aim real input at a fixed coordinate. if (el.scrollIntoView) { - el.scrollIntoView({ block: 'center', inline: 'nearest' }) + el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' }) + } + + // Hovering a disabled control is allowed — a disabled button with a tooltip + // explaining WHY is exactly the thing worth hovering. + if (action.kind === 'hover') { + const init = { bubbles: true, cancelable: true, ...pointAt(el) } + + if (typeof PointerEvent === 'function') { + el.dispatchEvent(new PointerEvent('pointerover', init)) + el.dispatchEvent(new PointerEvent('pointermove', init)) + } + + el.dispatchEvent(new MouseEvent('mouseover', init)) + el.dispatchEvent(new MouseEvent('mousemove', init)) + + return answer({ acted: 'hovered over ' + describe(el), success: true }) + } + + if ((el as HTMLInputElement).disabled) { + return fail(describe(el) + ' is disabled.') } if (action.kind === 'click') { diff --git a/apps/desktop/src/lib/preview-act/watch-in-page.test.ts b/apps/desktop/src/lib/preview-act/watch-in-page.test.ts new file mode 100644 index 0000000000..b51c62b7f0 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/watch-in-page.test.ts @@ -0,0 +1,363 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { type WatchHolder, watchInPage } from './watch-in-page' + +/** jsdom has no layout engine, so every rect is zero and nothing would be + * positioned. Give elements a plausible box the way the act-engine tests do. */ +beforeEach(() => { + // Never run the callback: the tracking loop re-arms itself every frame, so a + // stub that invokes synchronously recurses until the stack gives out. The + // first pass happens on the direct call, which is the one worth asserting on. + vi.stubGlobal('requestAnimationFrame', () => 1) + vi.stubGlobal('cancelAnimationFrame', () => {}) + + Element.prototype.getBoundingClientRect = function () { + return { bottom: 140, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 100, width: 200, x: 100, y: 100 } + } + + document.body.replaceChildren() + // The host hangs off documentElement, so clearing body does not reach it. + document.querySelector('hermes-watch')?.remove() + delete (window as unknown as Record).__hermesWatch +}) + +/** The overlay lives in a closed shadow root, so a test can only see the host. */ +const host = () => document.querySelector('hermes-watch') + +/** …except through the parts the engine parks on the window for its own reuse, + * which is the only way to inspect what the overlay actually drew. */ +const drawn = () => + (window as unknown as { __hermesWatch: { parts: Record & { shadow: ShadowRoot } } }) + .__hermesWatch.parts + +/** Everything the overlay drew for this action. The cursor is the only fixed + * layer — every box and pin is a mark that comes and goes. */ +const marks = () => [...drawn().shadow.children].filter(child => child !== drawn().pointer) + +describe('watchInPage', () => { + it('mounts one host on documentElement and reuses it across stages', () => { + const target = document.createElement('button') + document.body.append(target) + + const holder: WatchHolder = { aimed: target } + + watchInPage(document, holder, 'aim') + watchInPage(document, holder, 'strike') + + expect(document.querySelectorAll('hermes-watch')).toHaveLength(1) + expect(host()?.parentElement).toBe(document.documentElement) + }) + + it('stays out of the page, out of the way, and out of the a11y tree', () => { + const target = document.createElement('button') + document.body.append(target) + + watchInPage(document, { aimed: target }, 'aim') + + const node = host() as HTMLElement + + // pointer-events keeps real clicks reaching the page; aria-hidden keeps a + // screen reader from announcing decoration. + expect(node.style.pointerEvents).toBe('none') + expect(node.getAttribute('aria-hidden')).toBe('true') + // The closed shadow root is what keeps the overlay out of the act engine's + // own querySelectorAll inventory — there is nothing to exclude by hand. + expect(node.shadowRoot).toBeNull() + }) + + it('does nothing when there is no target to draw on', () => { + watchInPage(document, {}, 'aim') + watchInPage(document, { aimed: document.createElement('div') }, 'strike') + + expect(host()).toBeNull() + }) + + // The cursor is placed, never interpolated. Beyond reading as a machine, a + // transition here was an outright bug on the first paint: an unset transform + // IS the origin, so the browser flew the cursor in from the top-left corner — + // and every navigation is a fresh document, so a fresh first paint. + it('cuts the marks straight to the target instead of travelling to it', () => { + const target = document.createElement('button') + document.body.append(target) + + watchInPage(document, { aimed: target }, 'aim', 'Clicking Save') + + const { pointer } = drawn() + + expect(pointer.style.transition).toBeFalsy() + expect((marks()[0] as HTMLElement).style.transition).toBeFalsy() + // Mocked rect is 200×40 at (100, 100), so the cursor belongs at its centre. + expect(pointer.style.transform).toBe('translate3d(200px,120px,0)') + }) + + // The inventory pass is the one moment the agent is reading the whole page + // rather than aiming at one thing, so the overlay has to be able to hold many + // marks at once — not just the single ring every other stage draws. + it('sweep lights up every catalogued element and holds them well past the sweep', () => { + vi.useFakeTimers() + + const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')] + document.body.append(...found) + + watchInPage(document, { nodes: found }, 'sweep') + expect(marks()).toHaveLength(found.length) + + // The point of the dwell: the grid is still up long after the pass that drew + // it finished, so there is something to actually look at. jsdom has no Web + // Animations, so this exercises the timed fallback, which holds for the same + // beat as the animated path. + vi.advanceTimersByTime(600) + expect(marks()).toHaveLength(found.length) + + // …and then drains scattered rather than blinking off as one block, which is + // the whole reason each box carries its own offset. Stepping through the + // exit must therefore see the count come down in more than one tick. + const drain = new Set() + + for (let t = 0; t < 7_000; t += 100) { + vi.advanceTimersByTime(100) + drain.add(marks().length) + } + + expect(drain.size).toBeGreaterThan(2) + expect(marks()).toHaveLength(0) + + vi.useRealTimers() + }) + + // jsdom cannot run a Web Animation, but it can be handed one and asked what + // it was given — and this particular option is worth pinning, because getting + // it wrong is invisible rather than broken. `easing` in the options bag is the + // ITERATION easing: it rewrites progress before any keyframe is consulted, so + // a step function there holds progress at 0 for the whole duration. Every box + // was created, animated and removed without painting a single frame. + it('does not step the sweep’s iteration easing, which would stop it painting', () => { + const timings: KeyframeAnimationOptions[] = [] + + Element.prototype.animate = vi.fn(function (this: Element, _frames, options) { + timings.push(options as KeyframeAnimationOptions) + + return { cancel: () => {}, finish: () => {} } as unknown as Animation + }) as unknown as Element['animate'] + + const found = [document.createElement('button'), document.createElement('a')] + document.body.append(...found) + + watchInPage(document, { field: found }, 'sweep') + + // The cursor's idle sway is permanent and starts with the overlay, so it is + // not one of the sweep's own animations. + const sweep = timings.filter(timing => timing.iterations !== Infinity) + + expect(sweep).toHaveLength(found.length) + + for (const timing of sweep) { + expect(String(timing.easing)).not.toMatch(/steps/) + } + }) + + it('names each swept box after the element under it, not its ref', () => { + const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')] + + found[0].id = 'buy' + found[1].className = 'nav wide' + found[2].setAttribute('type', 'search') + document.body.append(...found) + + // Every drawable box gets named, addressable or not — a ref number only ever + // meant something for the subset that had one, and said nothing about what + // the element actually is. + watchInPage(document, { field: found, nodes: found.slice(0, 2) }, 'sweep') + + const { shadow } = drawn() + const cells = [...shadow.children].slice(-found.length) + + expect(cells.map(cell => cell.textContent)).toEqual(['button#buy', 'a.nav', 'input[search]']) + }) + + // The case that made positions necessary: a list of links with no id, no + // class and no type. Naming them by tag alone gives a whole page of boxes all + // reading "a", which is what the ref numbers were replaced for. + it('falls back to position so anonymous siblings are still told apart', () => { + const row = document.createElement('td') + + row.className = 'subtext' + row.innerHTML = '' + document.body.append(row) + + const found = [...row.children] + + watchInPage(document, { field: found }, 'sweep') + + const { shadow } = drawn() + const cells = [...shadow.children].slice(-found.length) + + expect(cells.map(cell => cell.textContent)).toEqual(['a[1]', 'a[2]', 'a[3]']) + }) + + it('keeps a name short enough to sit on the box it names', () => { + const found = [document.createElement('div')] + + found[0].id = 'a-genuinely-enormous-generated-identifier' + document.body.append(...found) + + watchInPage(document, { field: found }, 'sweep') + + const name = marks()[0].textContent as string + + expect(name.length).toBeLessThanOrEqual(16) + expect(name.startsWith('div#a-')).toBe(true) + expect(name.endsWith('…')).toBe(true) + }) + + // A field past a couple of dozen boxes is a texture, not a list. Naming every + // one buries the outlines — which ARE the readout at that size — under a wall + // of tags that collide with each other and overhang their own rectangles. + it('stops naming the boxes once the field is too dense to read', () => { + const few = Array.from({ length: 4 }, () => document.createElement('button')) + document.body.append(...few) + + watchInPage(document, { field: few }, 'sweep') + expect(marks().every(mark => mark.textContent)).toBe(true) + + document.querySelector('hermes-watch')?.remove() + delete (window as unknown as Record).__hermesWatch + + const many = Array.from({ length: 40 }, () => document.createElement('button')) + document.body.append(...many) + + watchInPage(document, { field: many }, 'sweep') + expect(marks().some(mark => mark.textContent)).toBe(false) + }) + + // Every other mark on the page wears its label on the top edge. A box against + // the top of the viewport has nowhere to put one, and a tab drawn off-screen + // is a box that looks unlabelled next to boxes that aren't. + it('flips the label under the box when there is no headroom above it', () => { + const roomy = document.createElement('button') + const flush = document.createElement('button') + const huge = document.createElement('button') + + flush.getBoundingClientRect = () => + ({ bottom: 40, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect + // Taller than the viewport: no room above it and none below it either, so + // the tab has nowhere left to go but inside. + huge.getBoundingClientRect = () => + ({ bottom: 900, height: 900, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect + document.body.append(roomy, flush, huge) + + watchInPage(document, { field: [roomy, flush, huge] }, 'sweep') + + const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement) + + expect(tags[0].style.bottom).toBe('100%') + expect(tags[1].style.top).toBe('100%') + expect(tags[2].style.top).toBe('0px') + }) + + // A label is routinely wider than the box it names, so it is the boxes near the + // right edge that push their tab off screen. Whichever end it lands on, the + // tab's edge sits exactly on the outline's — the two are the same colour and + // meet at a corner, so a pixel of overhang reads as a misprint. + it('hangs the label off the right edge when a left-anchored one would run off', () => { + Object.defineProperty(HTMLElement.prototype, 'offsetWidth', { configurable: true, get: () => 80 }) + + const inboard = document.createElement('button') + const edge = document.createElement('button') + + edge.getBoundingClientRect = () => + ({ bottom: 140, height: 40, left: 990, right: 1020, toJSON: () => ({}), top: 100, width: 30, x: 990, y: 100 }) as DOMRect + document.body.append(inboard, edge) + + watchInPage(document, { field: [inboard, edge] }, 'sweep') + + const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement) + + expect(tags[0].style.left).toBe('0px') + expect(tags[0].style.right).toBe('') + expect(tags[1].style.right).toBe('0px') + expect(tags[1].style.left).toBe('') + }) + + it('sweep draws nothing when the inventory came back empty', () => { + watchInPage(document, { nodes: [] }, 'sweep') + + expect(host()).toBeNull() + }) + + // The field is swept after every action, and a sweep retires the lock. The + // cursor must not go with it — a hand that blinks out a fraction of a second + // after every single step reads as the agent having wandered off. + it('keeps the cursor up when a sweep retires the lock', () => { + const target = document.createElement('button') + const found = [document.createElement('a')] + document.body.append(target, ...found) + + watchInPage(document, { aimed: target }, 'aim') + + const { pointer } = drawn() + const lock = marks()[0] + + expect(pointer.style.opacity).toBe('1') + + watchInPage(document, { field: found }, 'sweep') + + expect(lock.isConnected).toBe(false) + expect(pointer.style.opacity).toBe('1') + }) + + // The chrome is memoized for the life of the page, so a style change to it + // would otherwise be injected, run, and restyle nothing — the layers it + // describes were built by the version before. Reads as "HMR isn't picking my + // CSS up" in dev, and as a long-lived tab stuck on an old build in production. + it('rebuilds its chrome when the injected source changes underneath it', () => { + const target = document.createElement('button') + document.body.append(target) + + const stamp = (value: number) => { + ;(window as unknown as Record).__hermesWatchTag = value + } + + stamp(1) + watchInPage(document, { aimed: target }, 'aim') + + const first = drawn().host + + stamp(1) + watchInPage(document, { aimed: target }, 'aim') + + expect(drawn().host).toBe(first) + + stamp(2) + watchInPage(document, { aimed: target }, 'aim') + + expect(drawn().host).not.toBe(first) + // …and the old one goes with it, rather than stacking a second overlay. + expect(document.querySelectorAll('hermes-watch')).toHaveLength(1) + }) + + it('clear releases the target so the tracking loop can stop', () => { + const target = document.createElement('button') + document.body.append(target) + + const holder: WatchHolder = { aimed: target } + + watchInPage(document, holder, 'aim') + watchInPage(document, holder, 'clear') + + expect(holder.aimed).toBeNull() + }) + + // Same contract as the act engine: the pane injects `watchInPage.toString()` + // into the guest page, where module scope does not exist. One free identifier + // is a ReferenceError that Electron reports only as "Script failed to execute". + it('runs after being stringified and eval’d with no module scope', () => { + const target = document.createElement('button') + document.body.append(target) + + const injected = new Function('return (' + watchInPage.toString() + ')')() as typeof watchInPage + + expect(() => injected(document, { aimed: target }, 'aim', 'Clicking Save')).not.toThrow() + expect(host()).not.toBeNull() + }) +}) diff --git a/apps/desktop/src/lib/preview-act/watch-in-page.ts b/apps/desktop/src/lib/preview-act/watch-in-page.ts new file mode 100644 index 0000000000..0395548d0d --- /dev/null +++ b/apps/desktop/src/lib/preview-act/watch-in-page.ts @@ -0,0 +1,1063 @@ +/** + * WATCH OVERLAY — paints the agent's actions onto the live preview page so a + * person can follow along: the field of things it can reach, a box round the + * one it is about to touch, and a cursor that goes there. + * + * There are exactly two things on screen, and everything the agent does is said + * with one of them: + * + * - THE CURSOR — one arrow, built once, never removed. It is the + * collaborative-editing kind, because that is exactly what this is: a second + * party moving around a document you are also looking at. Arrow geometry is + * Liveblocks' (Apache-2.0). Plain black or white against the guest's own + * colour scheme rather than a brand colour: it stands for a hand, and a hand + * is not a feature of this app. + * - THE MARK — a rectangle glued to an element. A candidate, a lock, a hit + * and a pin are all the same object with a different row in `kinds`: how + * heavy the rule is, how long it lives, how it arrives. Marks are a design + * tool's selection blue, deliberately NOT the + * app's accent — this is chrome drawn on top of someone else's page, and + * every design tool has trained people to read that blue as "a tool has + * selected this", where a branded colour would read as part of the page. + * + * Every mark is two elements: a shell the tracker positions, and a skin that + * carries the border and any animation. They are split because both want + * `transform`, and a mark that animates the property the tracker rewrites every + * frame either stops moving or stops animating. + * + * Injected as source (`watchInPage.toString()`) exactly like the act engine, so + * the same rule holds: no imports, no closure references, no module-scope + * constants — everything the body needs is declared inside it. + * + * Three constraints shape the implementation, each one learned from how + * Playwright, Stagehand and browser-use solve the same problem: + * + * - A page can set `z-index: 2147483647` too, and the browser's top layer + * (``, `popover`, fullscreen) outranks every z-index there is. So + * the host joins the top layer via `popover="manual"` instead of trying to + * outbid the page, and keeps the z-index only as a fallback. + * - A strict CSP rejects `style-src` at CSS PARSE time, which kills + * `cssText`, `setAttribute('style', …)` and any `