diff --git a/agent/agent_init.py b/agent/agent_init.py index 4e097d6dfc..b839b35576 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -546,6 +546,7 @@ def init_agent( clarify_callback: callable = None, read_terminal_callback: callable = None, read_preview_callback: callable = None, + drive_preview_callback: callable = None, read_window_below_callback: callable = None, setup_mcp_callback: callable = None, tour_callback: callable = None, @@ -843,6 +844,7 @@ def init_agent( agent.clarify_callback = clarify_callback agent.read_terminal_callback = read_terminal_callback agent.read_preview_callback = read_preview_callback + agent.drive_preview_callback = drive_preview_callback agent.read_window_below_callback = read_window_below_callback agent.setup_mcp_callback = setup_mcp_callback agent.tour_callback = tour_callback diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 983c95b5d6..2744c1b72d 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -99,7 +99,7 @@ def _ra(): AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset( - {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} + {"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"} ) @@ -3234,6 +3234,37 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i ), next_args, ) + elif function_name == "drive_preview": + def _execute(next_args: dict) -> Any: + from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool + return _finish_agent_tool( + _drive_preview_tool( + action=next_args.get("action", ""), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + text=next_args.get("text"), + key=next_args.get("key"), + submit=next_args.get("submit"), + amount=next_args.get("amount"), + to=next_args.get("to"), + limit=next_args.get("max"), + callback=getattr(agent, "drive_preview_callback", None), + ), + next_args, + ) + elif function_name == "annotate_preview": + def _execute(next_args: dict) -> Any: + from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool + return _finish_agent_tool( + _annotate_preview_tool( + action=next_args.get("action", "add"), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + label=next_args.get("label"), + callback=getattr(agent, "drive_preview_callback", None), + ), + next_args, + ) elif function_name == "read_window_below": def _execute(next_args: dict) -> Any: from tools.read_window_tool import read_window_below_tool as _read_window_below_tool diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 4a828477c3..e7bb9126db 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2202,6 +2202,57 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('read_preview', function_args, tool_duration, result=function_result)}") + elif function_name == "drive_preview": + def _execute(next_args: dict) -> Any: + from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool + return _drive_preview_tool( + action=next_args.get("action", ""), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + text=next_args.get("text"), + key=next_args.get("key"), + submit=next_args.get("submit"), + amount=next_args.get("amount"), + to=next_args.get("to"), + limit=next_args.get("max"), + callback=getattr(agent, "drive_preview_callback", None), + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('drive_preview', function_args, tool_duration, result=function_result)}") + elif function_name == "annotate_preview": + def _execute(next_args: dict) -> Any: + from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool + return _annotate_preview_tool( + action=next_args.get("action", "add"), + ref=next_args.get("ref"), + selector=next_args.get("selector"), + label=next_args.get("label"), + callback=getattr(agent, "drive_preview_callback", None), + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('annotate_preview', function_args, tool_duration, result=function_result)}") elif function_name == "read_window_below": def _execute(next_args: dict) -> Any: from tools.read_window_tool import read_window_below_tool as _read_window_below_tool diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.test.ts b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts new file mode 100644 index 0000000000..2b68b1eb99 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-act.test.ts @@ -0,0 +1,314 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { $rightRailActiveTabId } from '@/store/layout' +import { closeRightRail, openPreview, type PreviewTarget } from '@/store/preview' + +import { actOnActivePreview } from './preview-act' +import { registerPreviewInput } from './preview-input' +import { registerPreviewNav } from './preview-nav' +import { registerPreviewScriptRunner } from './preview-script-runner' + +function urlTarget(url: string): PreviewTarget { + return { kind: 'url', label: 'Browser', source: url, url } +} + +describe('actOnActivePreview (drive_preview tool)', () => { + // URL targets share the singleton Browser tab id, so anything a test + // registers would answer the next one. + let cleanups: Array<() => void> = [] + + const openBrowserTab = () => { + openPreview(urlTarget('https://example.com'), 'tool-result') + + return $rightRailActiveTabId.get()! + } + + const withRunner = (runner: (code: string) => Promise) => + cleanups.push(registerPreviewScriptRunner(openBrowserTab(), runner)) + + beforeEach(() => { + vi.useRealTimers() + + for (const cleanup of cleanups) { + cleanup() + } + + cleanups = [] + closeRightRail() + window.localStorage.clear() + }) + + it('tells the agent to open a page when no live pane is behind the tab', async () => { + const result = await actOnActivePreview({ kind: 'elements' }) + + expect(result.success).toBe(false) + expect(result.error).toContain('open_preview') + }) + + it('injects the engine and returns the page’s answer', async () => { + let injected = '' + + withRunner(async code => { + injected = code + + return JSON.stringify({ acted: 'clicked button "Save"', success: true }) + }) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + expect(result).toMatchObject({ acted: 'clicked button "Save"', success: true }) + // Self-contained payload: the engine source and the action travel together, + // and the holder keeps refs alive across calls on the same page. + expect(injected).toContain('__hermesActHolder') + expect(injected).toContain('"ref":"@e1"') + }) + + it('re-inventories after a mutating action so the next ref is current', async () => { + const actions: string[] = [] + + withRunner(code => { + // Stand in for the guest page: run the script's own settle/rescan shape + // by answering each act() call in order. + actions.push(...(code.match(/"kind":"(\w+)"/g) ?? [])) + + return Promise.resolve( + JSON.stringify({ + acted: 'clicked', + elements: [{ label: 'Log out', ref: '@e1', role: 'button', selector: '#out' }], + success: true, + url: 'https://example.com/app' + }) + ) + }) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + expect(result.elements?.[0].label).toBe('Log out') + expect(result.url).toBe('https://example.com/app') + }) + + it('does not pay the settle delay for a plain inventory', async () => { + let injected = '' + withRunner(async code => { + injected = code + + return JSON.stringify({ elements: [], success: true }) + }) + + await actOnActivePreview({ kind: 'elements' }) + + expect(injected).toContain('0 <= 0') + }) + + it('reports a page that answers with nothing', async () => { + withRunner(async () => '') + + expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).error).toContain('did not answer') + }) + + /** A pane that answers the locate trip with a fixed on-screen point, and the + * read-back trip with an empty inventory. Returns the input spy. */ + const withDrivenPane = () => { + const tabId = openBrowserTab() + const send = vi.fn() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 120, y: 80 }, success: true }) + : // `hit` is the page's witness that the real pointerdown arrived. + JSON.stringify({ elements: [], hit: { tag: 'BUTTON', trusted: true }, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send })) + + return send + } + + const sentTypes = (send: ReturnType) => send.mock.calls.map(([event]) => event.type) + + it('clicks with real input, walking the pointer to where the page said', async () => { + const send = withDrivenPane() + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + const types = sentTypes(send) + + // Stepped rather than teleported: a page learns it is hovered from a stream + // of moves, and one jump to the target skips everything in between. + expect(types.filter(type => type === 'mouseMove').length).toBeGreaterThan(1) + expect(types).toContain('mouseDown') + expect(types).toContain('mouseUp') + expect(send.mock.calls.map(([event]) => event).find(event => event.type === 'mouseDown')).toMatchObject({ + x: 120, + y: 80 + }) + expect(result.acted).toBe('clicked button "Save"') + }) + + it('types by pressing keys, after selecting whatever the field held', async () => { + const send = withDrivenPane() + + await actOnActivePreview({ kind: 'type', ref: '@e1', submit: true, text: 'hi' }) + + const events = send.mock.calls.map(([event]) => event) + const chars = events.filter(event => event.type === 'char').map(event => event.keyCode) + + // One click to focus, then select-all by keyboard — NOT the triple-click + // this used to do. A triple-click is a pointer gesture, so it grabs the + // paragraph under the cursor whenever the target turns out not to be a + // field, and the agent was leaving pages with their body text highlighted. + expect(events.filter(event => event.type === 'mouseDown').map(event => event.clickCount)).toEqual([1]) + expect(events.filter(event => event.type === 'keyDown' && event.keyCode === 'a')[0]).toMatchObject({ + modifiers: ['control', 'meta'] + }) + // The chord must not send a `char` phase, or select-all types a literal 'a'. + expect(chars).toEqual(['h', 'i', 'Enter']) + }) + + it('hovers by walking the pointer over and leaving it there', async () => { + const send = withDrivenPane() + + const result = await actOnActivePreview({ kind: 'hover', ref: '@e1' }) + const types = sentTypes(send) + + expect(types).toContain('mouseMove') + // The whole request is "be on it" — a click here would open the dropdown the + // agent was trying to reveal, or worse, activate it. + expect(types).not.toContain('mouseDown') + expect(result.acted).toBe('hovered over button "Save"') + }) + + // The witness only speaks for verbs that put the button down. Demanding one + // from a key press reported every press as a failure. + it('does not expect a click witness from a verb that never clicks', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at textbox "Search"', point: { x: 40, y: 20 }, success: true }) + : JSON.stringify({ elements: [], hit: null, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + for (const action of [ + { key: 'Escape', kind: 'press', ref: '@e1' }, + { kind: 'hover', ref: '@e1' } + ]) { + expect(await actOnActivePreview(action)).toMatchObject({ success: true }) + } + }) + + it('scrolls by wheeling for real, not by scripting the page', async () => { + const tabId = openBrowserTab() + const send = vi.fn() + let scripted = false + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => { + scripted ||= code.includes('"kind":"scroll"') + + return code.includes('scrollHeight') + ? JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 4_000, success: true }) + : JSON.stringify({ elements: [], success: true }) + }) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send })) + + await actOnActivePreview({ kind: 'scroll' }) + + const wheels = send.mock.calls.map(([event]) => event).filter(event => event.type === 'mouseWheel') + + // A stream of notches, not one delta: scroll-linked headers and lazy loaders + // only react to the events, so a scripted scrollBy leaves them asleep. + expect(wheels.length).toBeGreaterThan(1) + // Electron's wheel delta is wheelDelta-signed, so scrolling DOWN is negative. + expect(wheels.every(event => event.deltaY < 0)).toBe(true) + expect(scripted).toBe(false) + }) + + it('says so plainly when the page has nothing to scroll', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async () => + JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 0, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + expect(await actOnActivePreview({ kind: 'scroll' })).toMatchObject({ note: expect.stringContaining('nothing to scroll') }) + }) + + it('fails loudly when the pointer input never reaches the page', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true }) + : JSON.stringify({ elements: [], hit: null, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + const result = await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + // Everything else about this action travels on the script channel and would + // report success whether or not a single event landed. + expect(result.success).toBe(false) + expect(result.error).toContain('never reached the page') + }) + + it('says so when the overlay itself swallowed the click', async () => { + const tabId = openBrowserTab() + + cleanups.push( + registerPreviewScriptRunner(tabId, async code => + code.includes('"kind":"locate"') + ? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true }) + : JSON.stringify({ elements: [], hit: { tag: 'HERMES-WATCH', trusted: true }, success: true }) + ) + ) + cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() })) + + expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).note).toContain('overlay intercepted') + }) + + it('falls back to scripted events when the pane exposes no input channel', async () => { + let injected = '' + + withRunner(async code => { + injected = code + + return JSON.stringify({ acted: 'clicked', success: true }) + }) + + await actOnActivePreview({ kind: 'click', ref: '@e1' }) + + // The one-trip shape: the engine both acts and re-reads, no locate handshake. + expect(injected).toContain('"kind":"click"') + expect(injected).not.toContain('"kind":"locate"') + }) + + it('routes history verbs to the pane instead of the guest page', async () => { + const back = vi.fn() + const runner = vi.fn() + + const tabId = openBrowserTab() + cleanups.push(registerPreviewNav(tabId, { back, forward: vi.fn(), reload: vi.fn() })) + cleanups.push(registerPreviewScriptRunner(tabId, runner)) + + const result = await actOnActivePreview({ kind: 'back' }) + + expect(back).toHaveBeenCalledOnce() + expect(runner).not.toHaveBeenCalled() + expect(result.success).toBe(true) + expect(result.note).toContain('elements') + }) + + it('reports history verbs with no pane to drive', async () => { + expect((await actOnActivePreview({ kind: 'reload' })).error).toContain('open_preview') + }) +}) diff --git a/apps/desktop/src/app/chat/right-rail/preview-act.ts b/apps/desktop/src/app/chat/right-rail/preview-act.ts new file mode 100644 index 0000000000..ee09246643 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-act.ts @@ -0,0 +1,575 @@ +/** + * PREVIEW ACT — performs the agent's interactions inside the preview pane's + * guest page, so `drive_preview` drives whatever web app is open in the in-app + * browser. + * + * The guest page is out-of-process; nothing here can touch its DOM directly. + * Two separate channels reach it, and the split matters: + * + * - `executeJavaScript` injects the engine SOURCE (see + * lib/preview-act/act-in-page.ts's self-containment contract) to RESOLVE + * and READ — turn 'btn-sign-in' into a node, measure it, inventory the + * page. It is parked on a window global alongside the book of handles so + * they survive between calls, and it vanishes with the page, so a + * navigation retires them and the engine reports that instead of acting on + * the wrong node. + * - `sendInputEvent` (preview-drive.ts) does the ACTING, as real Chromium + * input. Script can only dispatch synthetic events, which the page can tell + * apart and which never move the browser's own hover or focus target. + * + * Panes with no input channel — a remote HTML preview — fall back to the + * engine's synthetic events, which is worse but still works. + * + * A mutating action settles briefly and returns a fresh inventory, so the + * click → re-read → click loop costs one round trip instead of two. + * + * Dynamic-imported by the gateway event handler so the engine payload stays + * out of the boot path. + */ + +import { actEngineSource, type PreviewActAction, type PreviewActResult } from '@/lib/preview-act/act-in-page' +import { watchInPage } from '@/lib/preview-act/watch-in-page' + +import { clickAt, glideTo, pointerPlaced, pressKey, selectAll, typeText, wheelBy } from './preview-drive' +import { activePreviewInput, type PreviewInputHandle } from './preview-input' +import { activePreviewNav, type PreviewNavHandle } from './preview-nav' +import { activePreviewScriptRunner, type PreviewScriptRunner } from './preview-script-runner' + +/** Verbs the pane owns; a guest page cannot drive its own history. */ +const NAV_ACTIONS: readonly (keyof PreviewNavHandle)[] = ['back', 'forward', 'reload'] + +/** How long a click/type is given to land before the page is re-inventoried. + * Long enough for a framework re-render, short enough not to stall the turn. + * A render that misses this window is not lost — the NEXT action re-reads the + * page anyway, so the cost of being early is a slightly stale inventory, not a + * wrong one. That trade is worth several hundred ms on every single step. */ +const SETTLE_MS = 220 + +/** Cap on one round trip into the guest page. A click that starts a navigation + * tears the document down mid-settle, so the injected promise dies with it and + * `executeJavaScript` never settles — without this the tool eats its whole + * 45s bridge deadline for what was actually a successful click. */ +const ACT_TIMEOUT_MS = 8_000 + +/** Verbs that get the look-then-act treatment: locate the target, walk the + * pointer there, then send real input. Actions with no single target (scroll, + * elements) are absent and skip it. */ +const DRIVEN: readonly string[] = ['click', 'hover', 'press', 'type'] + +/** Verbs that put the mouse button down, and so are the only ones the pointerdown + * witness can speak to. `press` and `hover` never click, so demanding a witness + * from them would report every one of them as a failure. */ +const CLICKS: readonly string[] = ['click', 'type'] + +const NOTHING_OPEN = 'No live page is open in the in-app browser — open one with open_preview first.' + +const NAVIGATED = 'The page stopped answering right after — it is probably navigating. Call elements to see where you landed.' + +/** A fingerprint of the overlay's source, so the guest page can tell that the + * code it is running has changed underneath it. + * + * It needs one because the overlay memoizes its own chrome: the cursor, the + * lock frame and the scan line are built ONCE per page and then reused, so an + * edit to any of their styles lands in the injected source, gets injected, and + * changes nothing — the layers it would have styled were built by the previous + * version and are still on the page. In dev that reads as "HMR didn't pick up + * my CSS"; in production it is a long-lived tab pinned to whichever build first + * touched it, across an app upgrade. + * + * Content-addressed rather than a length or a version literal: the change that + * exposed this was one hex colour for another, which is byte-for-byte the same + * length, and a hand-maintained version is a step someone forgets. */ +const WATCH_TAG = (() => { + const source = watchInPage.toString() + let hash = 5381 + + for (let i = 0; i < source.length; i++) { + hash = ((hash << 5) + hash + source.charCodeAt(i)) | 0 + } + + return hash +})() + +/** Injected once per page: the engine, the overlay, and the two helpers every + * script below shares. Idempotent — re-running it on a live page is a no-op. */ +const preamble = () => ` var w = window; + // Reassigned on every trip rather than cached behind a guard: the source is in + // this payload either way, so the guard saved nothing and pinned a long-lived + // tab to whichever build first touched it. The holder is the exception — it + // carries the aimed element from the locate trip to the act trip. + w.__hermesActHolder = w.__hermesActHolder || {}; + w.__hermesAct = ${actEngineSource()}; + w.__hermesWatch_fn = (${watchInPage.toString()}); + w.__hermesWatchTag = ${WATCH_TAG}; + var holder = w.__hermesActHolder; + var act = function (a) { return w.__hermesAct(document, holder, a); }; + var watch = function (stage, label) { + try { w.__hermesWatch_fn(document, holder, stage, label); } catch (err) {} + }; + var wait = function (ms) { return new Promise(function (resolve) { setTimeout(resolve, ms); }); }; + // Sit out a smooth scroll so the pointer is aimed at a target that has stopped + // moving. A scroll that never started fires no scrollend, so only wait on one + // we can actually see happening; capture catches nested scrollers, whose + // scrollend does not bubble. + var restAfterScroll = function () { + return new Promise(function (resolve) { + var sc = document.scrollingElement || document.documentElement; + var x0 = sc ? sc.scrollLeft : 0; + var y0 = sc ? sc.scrollTop : 0; + requestAnimationFrame(function () { + if (!sc || (sc.scrollLeft === x0 && sc.scrollTop === y0)) { return resolve(); } + var done = false; + var finish = function () { + if (done) { return; } + done = true; + document.removeEventListener('scrollend', finish, true); + clearTimeout(timer); + resolve(); + }; + var timer = setTimeout(finish, 350); + document.addEventListener('scrollend', finish, true); + }); + }); + };` + +/** Bring the target on screen, mark it, and report where it came to rest. */ +function buildLocateScript(action: PreviewActAction, focus: boolean): string { + const locate = { focus, kind: 'locate', ref: action.ref, selector: action.selector } + + return `(function () { +${preamble()} + var locate = ${JSON.stringify(locate)}; + var found = act(locate); + if (!found.success) { return Promise.resolve(JSON.stringify(found)); } + watch('aim'); + // Arm a witness for the real input that is about to arrive. Without it a + // click that never reached the page is indistinguishable from one the page + // ignored, and the agent would report success either way. + w.__hermesHit = null; + document.addEventListener('pointerdown', function (e) { + w.__hermesHit = { tag: e.target ? e.target.tagName : '?', trusted: e.isTrusted === true }; + }, { capture: true, once: true }); + // Measure again once the scroll has stopped: real input is aimed at a fixed + // viewport coordinate, so it has to be where the target ENDS UP. + return restAfterScroll().then(function () { + var settled = act(locate); + var best = settled.success ? settled : found; + var at = best.point; + // Last line of defence before the pointer is sent somewhere real. An element + // that is still outside the viewport after we scrolled to it is hidden, not + // placed, and aiming at it would drive the cursor off into a corner and + // click whatever happens to be under that coordinate. + if (at && (at.x < 0 || at.y < 0 || at.x > window.innerWidth || at.y > window.innerHeight)) { + return JSON.stringify({ + error: 'That element is still off-screen after scrolling to it, so it is hidden rather than clickable. Call elements again for what is really on the page.', + success: false + }); + } + return JSON.stringify(best); + }); +})()` +} + +/** Put up a mark that outlives the action that made it. Every other cue on the + * overlay retires on a timer, which is right for narrating a click and no use + * at all for holding a finding on screen while the agent keeps working. */ +function buildPinScript(action: PreviewActAction, label: string): string { + const locate = { kind: 'locate', ref: action.ref, selector: action.selector } + + return `(function () { +${preamble()} + var found = act(${JSON.stringify(locate)}); + if (!found.success) { return JSON.stringify(found); } + watch('pin', ${JSON.stringify(label)}); + return JSON.stringify({ acted: 'pinned ' + String(found.acted || 'it').replace(/^looking at /, ''), success: true }); +})()` +} + +/** Freeze the whole visible field as marks. Needs a fresh inventory first, so + * the overlay has both the field to draw and the refs to number it by. */ +function buildHoldScript(): string { + return `(function () { +${preamble()} + var found = act({ kind: 'elements' }); + if (!found.success) { return JSON.stringify(found); } + watch('hold'); + found.acted = 'held the field'; + return JSON.stringify(found); +})()` +} + +/** Rattle through the field one box at a time. Needs a fresh inventory for the + * overlay to pick from, and nothing else — the page is never touched. */ +function buildStrobeScript(): string { + return `(function () { +${preamble()} + var found = act({ kind: 'elements' }); + if (!found.success) { return JSON.stringify(found); } + watch('strobe'); + found.acted = 'strobed the field'; + return JSON.stringify(found); +})()` +} + +/** Take one mark down, or — with nothing to aim at — all of them. */ +function buildUnpinScript(action: PreviewActAction): string { + const locate = { kind: 'locate', ref: action.ref, selector: action.selector } + const one = !!(action.ref || action.selector) + + return `(function () { +${preamble()} +${ + one + ? ` var found = act(${JSON.stringify(locate)}); + if (!found.success) { return JSON.stringify(found); } + var gone = 'unpinned ' + String(found.acted || 'it').replace(/^looking at /, '');` + : ` holder.aimed = null; + var gone = 'cleared every pin';` +} + watch('unpin'); + return JSON.stringify({ acted: gone, success: true }); +})()` +} + +/** Mark the hit, let the page react, and hand back a fresh inventory. */ +function buildFinishScript(settleMs: number): string { + return `(function () { +${preamble()} + watch('strike'); + return wait(${settleMs}).then(function () { + // A rescan that throws must still answer — an unresolved promise here costs + // the whole bridge deadline. + try { + var out = act({ kind: 'elements' }); + watch('sweep'); + out.hit = w.__hermesHit || null; + return JSON.stringify(out); + } catch (err) { + return JSON.stringify({ note: 'The page changed before it could be re-read: ' + err, success: true }); + } + }); +})()` +} + +/** The no-real-input fallback: the engine both acts and reads, in one trip. + * Both reads mark the field — the explicit `elements` call and the re-read + * every other verb does on its way out — so the page is marked after every + * single action rather than only when the agent asks for an inventory outright. + * They mark it differently, though, and the difference is what each one MEANS: + * an outright `elements` is the agent reading the whole page, which is the + * moment worth showing, so it strobes. The re-read after a click is + * housekeeping, and strobing there would put five seconds of noise between + * every step of a task, so it sweeps. The two never collide: an `elements` call + * settles in 0ms and returns before the re-read. */ +function buildScriptedScript(action: PreviewActAction, settleMs: number): string { + return `(function () { +${preamble()} + var result = act(${JSON.stringify(action)}); + ${ + action.kind === 'elements' + ? "if (result.success) { watch('strobe'); }" + : '' + } + if (!result.success || ${settleMs} <= 0) { return Promise.resolve(JSON.stringify(result)); } + return wait(${settleMs}).then(function () { + try { + var after = act({ kind: 'elements' }); + watch('sweep'); + // One or the other, never both: a re-read answers with the whole + // inventory only when it is the first look at this page. + result.elements = after.elements; + result.delta = after.delta; + result.url = after.url; + result.title = after.title; + } catch (err) { + result.note = 'The page changed before it could be re-read: ' + err; + } + return JSON.stringify(result); + }); +})()` +} + +/** The outcome of one round trip into the page. `silent` is its own case on + * purpose: a page that stops answering mid-action is navigating, whereas one + * that answers with nothing is broken, and the agent needs to hear the + * difference. */ +type Trip = { error: string; kind: 'failed' } | { kind: 'answered'; result: PreviewActResult } | { kind: 'silent' } + +async function runJson(run: PreviewScriptRunner, code: string): Promise { + const raw = await Promise.race([ + run(code).catch((error: unknown) => new Error(String(error))), + new Promise(resolve => setTimeout(resolve, ACT_TIMEOUT_MS)) + ]) + + if (raw === undefined) { + return { kind: 'silent' } + } + + if (raw instanceof Error) { + return { error: 'The page rejected the action: ' + raw.message, kind: 'failed' } + } + + if (typeof raw !== 'string' || !raw) { + return { error: 'The page did not answer the action.', kind: 'failed' } + } + + return { kind: 'answered', result: JSON.parse(raw) as PreviewActResult } +} + +/** Past tense of the verb the agent asked for, against what it actually hit. */ +function describeDone(action: PreviewActAction, target: string): string { + if (action.kind === 'type') { + return 'typed into ' + target + (action.submit ? ' and submitted' : '') + } + + if (action.kind === 'press') { + return 'pressed ' + (action.key || '') + ' on ' + target + } + + if (action.kind === 'hover') { + return 'hovered over ' + target + } + + return 'clicked ' + target +} + +/** Look at the target, walk the pointer over, and act on it for real. */ +async function driveAction( + run: PreviewScriptRunner, + input: PreviewInputHandle, + action: PreviewActAction +): Promise { + // A key press must not be preceded by a click — that would activate the + // control rather than type into it — so the page hands it focus instead. + const trip = await runJson(run, buildLocateScript(action, action.kind === 'press')) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + if (trip.kind === 'silent') { + return { acted: action.kind, note: NAVIGATED, success: true } + } + + const found = trip.result + + if (!found.success) { + return found + } + + if (!found.point) { + return { error: 'Could not work out where that element is on screen.', success: false } + } + + await glideTo(input, found.point) + + if (action.kind === 'click') { + await clickAt(input) + } else if (action.kind === 'type') { + if (found.typable === false) { + return { + error: `${String(found.acted || 'That').replace(/^looking at /, '')} is not a text field, so typing into it would only select the text under the pointer. Click it if it opens one, then type into that.`, + success: false + } + } + + input.focus() + await clickAt(input) + // Select-all inside the now-focused field, so typing replaces what is there + // the way it would for a person. NOT a triple-click: that is a pointer + // gesture and selects the paragraph under the cursor whenever the target + // turns out not to be a field. + await selectAll(input) + await typeText(input, action.text ?? '') + + if (action.submit) { + await pressKey(input, 'Enter') + } + } else if (action.kind === 'press') { + input.focus() + await pressKey(input, action.key || 'Enter') + } + // hover is the glide and nothing else — the pointer is already sitting on the + // target, which is the whole request. + + const target = String(found.acted || '').replace(/^looking at /, '') + const after = await runJson(run, buildFinishScript(SETTLE_MS)) + const acted = describeDone(action, target) + + // The action itself already happened as real input, so a page that will not + // answer the read-back is a page that navigated — never a failed click. + if (after.kind !== 'answered') { + return { acted, note: NAVIGATED, success: true } + } + + const { hit, ...result } = after.result as PreviewActResult & { hit?: { tag: string; trusted: boolean } | null } + + // The witness the locate trip armed. No record means the input never reached + // the document, which the agent must hear about — every other signal here + // travels on the script channel and would report success regardless. + if (!hit && CLICKS.indexOf(action.kind) !== -1) { + return { + ...result, + error: 'The pointer input never reached the page, so nothing was ' + acted.split(' ')[0] + '.', + success: false + } + } + + return { ...result, acted, note: hitNote(hit), success: true } +} + +/** Flag a click the overlay intercepted, which would otherwise look like a page + * that simply ignored it. */ +function hitNote(hit?: { tag: string; trusted: boolean } | null): string | undefined { + return hit && hit.tag === 'HERMES-WATCH' + ? 'The action overlay intercepted the click instead of the page.' + : undefined +} + +/** How far a screenful is, whether there is anywhere to go, and a spot to wheel + * over — the renderer cannot see the guest viewport to work any of it out. */ +function buildScrollAnchorScript(): string { + return `(function () { +${preamble()} + var sc = document.scrollingElement || document.documentElement; + var track = sc.clientHeight || window.innerHeight; + return Promise.resolve(JSON.stringify({ + page: Math.round(window.innerHeight * 0.9), + point: { x: Math.round(window.innerWidth / 2), y: Math.round(track / 2) }, + span: sc.scrollHeight - track, + success: true + })); +})()` +} + +/** Scroll the page the way a hand does — wheel it. The scripted path calls + * `scrollBy`, which moves the page without the page ever seeing an input + * event, so scroll-linked headers, reveal animations and infinite-scroll + * loaders all stay asleep through a scroll that looked like it happened. */ +async function driveScroll( + run: PreviewScriptRunner, + input: PreviewInputHandle, + action: PreviewActAction, +): Promise { + const far = action.amount ?? 0 + + const trip = await runJson( + run, + buildScrollAnchorScript() + ) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + if (trip.kind === 'silent') { + return { acted: 'scrolled', note: NAVIGATED, success: true } + } + + const anchor = trip.result as PreviewActResult & { page?: number; span?: number } + + if (!anchor.span) { + return { ...anchor, acted: 'scrolled the page', note: 'The page has nothing to scroll — it all fits already.' } + } + + // A person does not move the mouse to scroll; the wheel turns wherever their + // hand already is. Only send it somewhere if it has never been anywhere. + if (!pointerPlaced() && anchor.point) { + await glideTo(input, anchor.point) + } + + await wheelBy(input, action.amount ?? anchor.page ?? 600) + + const after = await runJson(run, buildFinishScript(SETTLE_MS)) + + if (after.kind !== 'answered') { + return { acted: 'scrolled the page', note: NAVIGATED, success: true } + } + + return { ...after.result, acted: 'scrolled the page', success: true } +} + +/** Run one action against the ACTIVE preview tab's page. `kind` is a bare + * string: the verb arrives off the wire, and the history ones never reach + * the in-page engine. */ +export async function actOnActivePreview( + action: Omit & { kind: string } +): Promise { + const nav = NAV_ACTIONS.find(verb => verb === action.kind) + + if (nav) { + const handle = activePreviewNav() + + if (!handle) { + return { error: NOTHING_OPEN, success: false } + } + + handle[nav]() + + // Navigation is fire-and-forget through the webview; the new document has + // its own refs, so the agent has to re-inventory either way. + return { acted: nav, note: 'Page is loading — call elements to see what is on it.', success: true } + } + + const run = activePreviewScriptRunner() + + if (!run) { + return { error: NOTHING_OPEN, success: false } + } + + const typed = action as PreviewActAction + + // Annotation, not interaction: nothing is clicked, nothing settles, and the + // page is not re-read, so these skip the whole act-then-inventory path. + if (typed.kind === 'pin' || typed.kind === 'unpin' || typed.kind === 'hold' || typed.kind === 'strobe') { + const mark = + typed.kind === 'hold' + ? buildHoldScript() + : typed.kind === 'strobe' + ? buildStrobeScript() + : typed.kind === 'pin' + ? buildPinScript(typed, typed.text || '') + : buildUnpinScript(typed) + + const trip = await runJson(run, mark) + + if (trip.kind === 'failed') { + return { error: trip.error, success: false } + } + + return trip.kind === 'answered' ? trip.result : { acted: typed.kind, note: NAVIGATED, success: true } + } + + const input = activePreviewInput() + + if (input && DRIVEN.indexOf(typed.kind) !== -1) { + return driveAction(run, input, typed) + } + + // A plain page scroll is a wheel gesture. Jumping to an end is not — no hand + // wheels to the bottom of a long article — so `to` stays a scripted glide. + const plain = typed.kind === 'scroll' && !typed.to && !typed.ref && !typed.selector + + if (plain && input) { + return driveScroll(run, input, typed) + } + + const settle = typed.kind === 'elements' ? 0 : SETTLE_MS + const scripted = await runJson(run, buildScriptedScript(typed, settle)) + + if (scripted.kind === 'failed') { + return { error: scripted.error, success: false } + } + + // The action almost certainly landed — a page that stops answering right + // after a click is one that navigated. Say so instead of failing it. + return scripted.kind === 'silent' ? { acted: typed.kind, note: NAVIGATED, success: true } : scripted.result +} + +// Self-accept so an edit here, or to the in-page sources this module +// stringifies, doesn't reload the whole renderer out from under a live session. +// The bridge re-requests this module per action in dev, so it picks the change +// up without one (see loadPreviewEngine in gateway-event/desktop-bridge.ts). +if (import.meta.hot) { + import.meta.hot.accept() +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-drive.ts b/apps/desktop/src/app/chat/right-rail/preview-drive.ts new file mode 100644 index 0000000000..18b9dbf24e --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-drive.ts @@ -0,0 +1,138 @@ +/** + * PREVIEW DRIVE — the agent's hand on the mouse and keyboard. + * + * Everything here goes out through `sendInputEvent`, so the guest page receives + * ordinary trusted input: the pointer really moves across it, `:hover` really + * matches, focus really lands, and a menu that only exists while hovered is + * actually open by the time the click arrives. That last part is the whole + * point — a synthetic `el.click()` on a dropdown item fires at a node the page + * never rendered, which is why script-driven clicking quietly misses. + * + * The pointer is walked to its target in steps rather than teleported, for the + * same reason: a page learns it is hovered from a stream of moves, and one + * move from nowhere to the target skips every element in between. Fourteen + * steps over ~280ms is what Playwright's action cursor settles on too. + */ + +import type { PreviewInputHandle } from './preview-input' + +export interface DrivePoint { + x: number + y: number +} + +/** The pointer JUMPS. These few steps over a couple of frames exist only so the + * page receives a move stream at all — `:hover`, `pointerenter` and the menus + * that hang off them need more than a single teleport to resolve — not so the + * travel can be watched. A machine does not sweep a mouse across a screen. */ +const GLIDE_STEPS = 3 +const GLIDE_MS = 36 + +/** Gap between keystrokes. Fast enough not to pad the turn, slow enough that a + * page debouncing its input handler still sees them as separate. */ +const KEY_MS = 7 + +/** Notches in a wheel gesture, and how long the gesture takes. Sent as a stream + * rather than one big delta so scroll-linked effects — sticky headers, reveal + * animations, infinite-scroll loaders — get the same series of events a hand + * would produce, and so it looks like momentum rather than a jump cut. */ +const WHEEL_STEPS = 12 +const WHEEL_MS = 240 + +/** Where the pointer was left, so the next glide starts from there instead of + * jumping. Module-level because the pointer is a property of the pane, not of + * any one action. The origin is a placeholder, not a position — see below. */ +let pointer: DrivePoint = { x: 0, y: 0 } +let placed = false + +/** Whether the pointer has ever been sent anywhere. Until it has, its notional + * spot is the top-left corner, which is a real coordinate a caller can wheel + * or click at by accident — so callers that only need it to be SOMEWHERE + * sensible have to know the difference. */ +export function pointerPlaced(): boolean { + return placed +} + +const wait = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)) + +/** Decelerating, like a hand arriving at a target rather than a linear sweep. */ +const easeOut = (t: number) => 1 - Math.pow(1 - t, 3) + +/** Walk the pointer to `to`, letting the page hover everything on the way. */ +export async function glideTo(input: PreviewInputHandle, to: DrivePoint): Promise { + const from = pointer + + for (let step = 1; step <= GLIDE_STEPS; step++) { + const progress = easeOut(step / GLIDE_STEPS) + + input.send({ + type: 'mouseMove', + x: Math.round(from.x + (to.x - from.x) * progress), + y: Math.round(from.y + (to.y - from.y) * progress) + }) + await wait(GLIDE_MS / GLIDE_STEPS) + } + + pointer = to + placed = true +} + +/** Press and release at the pointer's current spot. `clicks` of 3 selects the + * text under it, which is how a field gets cleared without a modifier key. */ +export async function clickAt(input: PreviewInputHandle, clicks = 1): Promise { + for (let click = 1; click <= clicks; click++) { + input.send({ button: 'left', clickCount: click, type: 'mouseDown', x: pointer.x, y: pointer.y }) + input.send({ button: 'left', clickCount: click, type: 'mouseUp', x: pointer.x, y: pointer.y }) + await wait(KEY_MS) + } +} + +/** Wheel `down` pixels' worth of notches at the pointer's current spot; negative + * scrolls up. Electron's wheel delta is the legacy `wheelDelta` sign — positive + * moves the content down, i.e. scrolls UP — so it is the inverse of the number + * a caller asks for, and of DOM `WheelEvent.deltaY`. */ +export async function wheelBy(input: PreviewInputHandle, down: number): Promise { + let sent = 0 + + for (let step = 1; step <= WHEEL_STEPS; step++) { + const so_far = Math.round(down * easeOut(step / WHEEL_STEPS)) + const notch = so_far - sent + + sent = so_far + + if (notch) { + input.send({ deltaX: 0, deltaY: -notch, type: 'mouseWheel', x: pointer.x, y: pointer.y }) + } + + await wait(WHEEL_MS / WHEEL_STEPS) + } +} + +/** Send one key the long way round, so a page watching any of the three sees it. */ +export async function pressKey(input: PreviewInputHandle, key: string): Promise { + input.send({ keyCode: key, type: 'keyDown' }) + input.send({ keyCode: key, type: 'char' }) + input.send({ keyCode: key, type: 'keyUp' }) + await wait(KEY_MS) +} + +/** Select-all inside whatever has focus, which for a focused field is that + * field's own text and nothing else. This replaced a triple-click: a triple + * click is a POINTER gesture, so it selects whatever paragraph sits under the + * cursor whenever the target turns out not to be a field, and the agent was + * leaving pages with their body text highlighted. There is no `char` phase — + * a chord is not text entry, and sending one types a literal 'a'. */ +export async function selectAll(input: PreviewInputHandle): Promise { + const chord = ['control', 'meta'] + + input.send({ keyCode: 'a', modifiers: chord, type: 'keyDown' }) + input.send({ keyCode: 'a', modifiers: chord, type: 'keyUp' }) + await wait(KEY_MS) +} + +/** Type `text` a character at a time into whatever currently has focus. */ +export async function typeText(input: PreviewInputHandle, text: string): Promise { + for (const character of text) { + await pressKey(input, character) + } +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-input.ts b/apps/desktop/src/app/chat/right-rail/preview-input.ts new file mode 100644 index 0000000000..c0f1dbb5a0 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-input.ts @@ -0,0 +1,56 @@ +/** + * PREVIEW INPUT REGISTRY — real input into the preview pane's guest page, the + * difference between the agent DRIVING the browser and merely poking its DOM. + * + * `executeJavaScript` can only ever dispatch synthetic events: `isTrusted` is + * false, the browser's own hover target never moves, `:hover` rules never + * match, and hover-gated menus never open — so a click lands on a dropdown item + * that was never rendered. `sendInputEvent` goes in through Chromium's input + * pipeline instead, producing the same events a hand on the mouse would. + * + * It has to be called on the `` ELEMENT. Sending to the embedder's + * webContents does not reach a guest (electron/electron#20333), which is why + * this is a per-pane registry rather than something main could do. + * + * Coordinates are relative to the webview, and the webview IS the guest + * viewport — so a rect the act engine measured inside the page needs no + * conversion on the way back out. + */ + +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + +/** The subset of Electron's input events the agent needs to drive a page. */ +export type PreviewInputEvent = + | { button: 'left'; clickCount: number; type: 'mouseDown' | 'mouseUp'; x: number; y: number } + | { deltaX: number; deltaY: number; type: 'mouseWheel'; x: number; y: number } + | { keyCode: string; modifiers?: string[]; type: 'char' | 'keyDown' | 'keyUp' } + | { type: 'mouseMove'; x: number; y: number } + +export interface PreviewInputHandle { + /** Give the guest keyboard focus, so key events reach its active element. */ + focus: () => void + send: (event: PreviewInputEvent) => void +} + +const handles = new Map() + +/** Register a live pane's input channel; returns an idempotent unregister. */ +export function registerPreviewInput(tabId: string, handle: PreviewInputHandle): () => void { + handles.set(tabId, handle) + + return () => { + if (handles.get(tabId) === handle) { + handles.delete(tabId) + } + } +} + +/** The ACTIVE preview tab's input channel. Null = nothing real to drive, and + * the caller falls back to synthesizing events inside the page. */ +export function activePreviewInput(): PreviewInputHandle | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && handles.get(tab.id)) || null +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-mind.ts b/apps/desktop/src/app/chat/right-rail/preview-mind.ts new file mode 100644 index 0000000000..3f7a1856c4 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-mind.ts @@ -0,0 +1,29 @@ +/** + * IDLE PULSE — keeps the preview overlay alive while the model reasons. + * + * Everything else the overlay draws is narration of an action, so the pane goes + * dark for the gap between actions — which is most of the time the agent spends + * on a page, and the part where it is deciding what to do with what it just + * read. A page that stops responding the moment the agent is thinking hardest + * reads as the agent having wandered off. + * + * This is deliberately NOT part of the act pipeline: a turn boundary only has + * to poke the overlay the last action left behind. See preview-nudge.ts. + */ + +import { $busy } from '@/store/session' + +import { nudgeOverlay } from './preview-nudge' + +// Module-level, matching how review.ts and coding-status.ts watch this edge. It +// is inert without a live pane, so there is nothing to mount or tear down. +let running = $busy.get() + +$busy.subscribe(busy => { + if (busy === running) { + return + } + + running = busy + nudgeOverlay(busy ? 'think' : 'rest') +}) diff --git a/apps/desktop/src/app/chat/right-rail/preview-nav.ts b/apps/desktop/src/app/chat/right-rail/preview-nav.ts index 954c1a5e9b..5e03db3005 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-nav.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-nav.ts @@ -9,6 +9,9 @@ * sitting in Hermes' own DOM, where `activeElement` is authoritative. */ +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + /** Marks a live browser pane so a gesture can find the one holding focus. */ export const PREVIEW_BROWSER_ATTR = 'data-preview-browser' @@ -31,6 +34,15 @@ export function registerPreviewNav(tabId: string, handle: PreviewNavHandle): () } } +/** The ACTIVE preview tab's commands, for callers with no focus to key off — + * the agent's drive_preview, which runs while focus is in the composer. */ +export function activePreviewNav(): PreviewNavHandle | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && handles.get(tab.id)) || null +} + /** Run `command` on the browser pane holding DOM focus. False = focus is * elsewhere in the app, so the caller falls back to the app-level meaning. */ export function commandFocusedPreview(command: keyof PreviewNavHandle): boolean { diff --git a/apps/desktop/src/app/chat/right-rail/preview-nudge.ts b/apps/desktop/src/app/chat/right-rail/preview-nudge.ts new file mode 100644 index 0000000000..8a8f419987 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-nudge.ts @@ -0,0 +1,38 @@ +/** + * OVERLAY NUDGE — the one way to say a stage to an overlay that is ALREADY on + * the page. + * + * The act pipeline (preview-act.ts) injects the whole engine because it has to: + * it needs to resolve a ref, measure a node and read the page back. The stages + * that only narrate — the idle pulse, the read wipe — need none of that. They + * are one function call against something the last action already left on the + * page, so re-shipping the engine to make it would put the payload on the wire + * on every turn boundary. + * + * A page the agent has never acted on has no overlay, and the nudge is a no-op + * there. That is the right answer rather than a gap: chrome on a page the agent + * never touched would be a lie about what it did. + */ + +import type { WatchStage } from '@/lib/preview-act/watch-in-page' + +import { activePreviewScriptRunner } from './preview-script-runner' + +/** Run one stage against the active pane's overlay, if it has one. */ +export function nudgeOverlay(stage: WatchStage): void { + const run = activePreviewScriptRunner() + + if (!run) { + return + } + + void run(`(function () { + var w = window; + var fn = w.__hermesWatch_fn; + if (!fn) { return 'cold'; } + try { fn(document, w.__hermesActHolder || {}, ${JSON.stringify(stage)}); } catch (err) {} + return 'ok'; +})()`).catch(() => { + // The page navigated out from under us. The next action re-injects. + }) +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx index 7d6150b2f3..817cec638a 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-pane.tsx +++ b/apps/desktop/src/app/chat/right-rail/preview-pane.tsx @@ -1,3 +1,7 @@ +// Side-effect import: watches the turn edge so the overlay keeps a pulse while +// the model reasons. Lives here because the pane is what makes it reachable. +import './preview-mind' + import { useStore } from '@nanostores/react' import type { PointerEvent as ReactPointerEvent } from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' @@ -28,9 +32,10 @@ import { import { type ConsoleEntry } from './preview-console-state' import { previewConsoleState } from './preview-console-store' import { LocalFilePreview, PreviewEmptyState } from './preview-file' +import { type PreviewInputEvent, registerPreviewInput } from './preview-input' import { PREVIEW_BROWSER_ATTR, registerPreviewNav } from './preview-nav' import { registerPreviewPageReader } from './preview-reader' -import { registerPreviewTourRunner } from './preview-tour-runner' +import { registerPreviewScriptRunner } from './preview-script-runner' type PreviewWebview = HTMLElement & { canGoBack?: () => boolean @@ -53,6 +58,7 @@ type PreviewWebview = HTMLElement & { reloadIgnoringCache?: () => void replaceMisspelling?: (word: string) => void selectAll?: () => void + sendInputEvent?: (event: PreviewInputEvent) => void } /** The raw Chromium params riding the webview tag's `context-menu` event. */ @@ -455,15 +461,15 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest = }) }, [isWebPreview, tabId]) - // Publish the TOUR runner for this tab (the tour tool, surface='preview'): - // runs injected driver.js actions inside the guest page so the agent can - // give guided walkthroughs of whatever web app is open here. + // Publish the SCRIPT runner for this tab: the one channel into the guest + // page, shared by the tour tool (injected driver.js walkthroughs) and the + // drive_preview tool (clicking, typing, scrolling the page the user sees). useEffect(() => { if (!isWebPreview || !tabId) { return } - return registerPreviewTourRunner(tabId, async code => { + return registerPreviewScriptRunner(tabId, async code => { const webview = webviewRef.current if (!webview?.executeJavaScript) { @@ -474,6 +480,32 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest = }) }, [isWebPreview, tabId]) + // Publish the INPUT channel for this tab. Same idea as the script runner, but + // it carries real Chromium input rather than script — the agent's clicks and + // keystrokes arrive as trusted events, so the page hovers, focuses and reacts + // exactly as it would under a human hand. + useEffect(() => { + if (!isWebPreview || isRemoteHtml || !tabId) { + return + } + + return registerPreviewInput(tabId, { + focus: () => webviewRef.current?.focus?.(), + send: event => { + const webview = webviewRef.current + + // Never optional-chain this call away: a missing method would make every + // agent click a silent no-op that still reports success, because the + // overlay and the read-back both run on the separate script channel. + if (typeof webview?.sendInputEvent !== 'function') { + throw new Error('preview webview cannot take input events') + } + + webview.sendInputEvent(event) + } + }) + }, [isRemoteHtml, isWebPreview, tabId]) + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) useEffect(() => { if (!consoleOpen) { diff --git a/apps/desktop/src/app/chat/right-rail/preview-reader.ts b/apps/desktop/src/app/chat/right-rail/preview-reader.ts index 7f48c8cbb1..97a372db18 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-reader.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-reader.ts @@ -15,6 +15,8 @@ import { $rightRailActiveTabId } from '@/store/layout' import { $previewTabs } from '@/store/preview' +import { nudgeOverlay } from './preview-nudge' + export interface PreviewReadOptions { /** Characters to return from `start` (capped at PREVIEW_READ_MAX_CHARS). */ count?: number @@ -89,6 +91,13 @@ export async function readActivePreview(opts: PreviewReadOptions = {}): Promise< try { const page = await reader() + // Say it on the page. Reading is by far the cheapest thing the agent + // does — a few hundredths of a second against a model round trip either + // side of it — so a run of reads used to leave the pane dark for the + // twenty seconds it took to page through a document, immediately after + // the one moment that showed anything. + nudgeOverlay('read') + return windowText( { kind: target.kind, path: target.path, title: page.title || target.label, url: page.url || target.url }, page.text, diff --git a/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts b/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts new file mode 100644 index 0000000000..69a0efa2a4 --- /dev/null +++ b/apps/desktop/src/app/chat/right-rail/preview-script-runner.ts @@ -0,0 +1,38 @@ +/** + * PREVIEW SCRIPT RUNNER REGISTRY — the one way anything in the app reaches into + * the preview pane's guest page, the script analog of preview-nav's handle + * registry. + * + * A live browser pane registers its webview's `executeJavaScript` here, keyed + * by tab id; `activePreviewScriptRunner` resolves the ACTIVE tab from the + * store. Both guest-page features ride it — the tour tool (preview-tour.ts) + * and the interaction tool (preview-act.ts) — so their heavy payloads stay out + * of the pane component's static import graph and only load when used. + */ + +import { $rightRailActiveTabId } from '@/store/layout' +import { $previewTabs } from '@/store/preview' + +/** Runs JS source in the pane's guest page, resolving its completion value. */ +export type PreviewScriptRunner = (code: string) => Promise + +const runners = new Map() + +/** Register a live preview's script runner; returns an idempotent unregister. */ +export function registerPreviewScriptRunner(tabId: string, runner: PreviewScriptRunner): () => void { + runners.set(tabId, runner) + + return () => { + if (runners.get(tabId) === runner) { + runners.delete(tabId) + } + } +} + +/** The ACTIVE preview tab's script runner. Null = no live page behind it. */ +export function activePreviewScriptRunner(): PreviewScriptRunner | null { + const tabs = $previewTabs.get() + const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] + + return (tab && runners.get(tab.id)) || null +} diff --git a/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts b/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts deleted file mode 100644 index 84a15c8869..0000000000 --- a/apps/desktop/src/app/chat/right-rail/preview-tour-runner.ts +++ /dev/null @@ -1,38 +0,0 @@ -/** - * PREVIEW TOUR RUNNER REGISTRY — the tour tool's script bridge into the - * preview pane, the tour analog of preview-nav's handle registry. - * - * A live browser pane registers a script runner (its webview's - * `executeJavaScript`) here, keyed by tab id; `activeTourRunner` resolves the - * ACTIVE tab from the store. Kept separate from preview-tour.ts so the pane - * component's static import stays tiny — the engine source + driver.js - * payload only load when a tour actually runs (gateway-event dynamic-imports - * preview-tour.ts). - */ - -import { $rightRailActiveTabId } from '@/store/layout' -import { $previewTabs } from '@/store/preview' - -/** Runs JS source in the pane's guest page, resolving its completion value. */ -export type TourScriptRunner = (code: string) => Promise - -const runners = new Map() - -/** Register a live preview's script runner; returns an idempotent unregister. */ -export function registerPreviewTourRunner(tabId: string, runner: TourScriptRunner): () => void { - runners.set(tabId, runner) - - return () => { - if (runners.get(tabId) === runner) { - runners.delete(tabId) - } - } -} - -/** The ACTIVE preview tab's script runner. Null = no live page to tour. */ -export function activeTourRunner(): TourScriptRunner | null { - const tabs = $previewTabs.get() - const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0] - - return (tab && runners.get(tab.id)) || null -} diff --git a/apps/desktop/src/app/chat/right-rail/preview-tour.ts b/apps/desktop/src/app/chat/right-rail/preview-tour.ts index 12ad67dc71..a033a7c1f8 100644 --- a/apps/desktop/src/app/chat/right-rail/preview-tour.ts +++ b/apps/desktop/src/app/chat/right-rail/preview-tour.ts @@ -21,7 +21,7 @@ import driverIife from 'driver.js/dist/driver.js.iife.js?raw' import { collectTourTargets } from '@/lib/tour/collect-targets' import { runTourEngine, type TourAction, type TourResult } from '@/lib/tour/engine' -import { activeTourRunner } from './preview-tour-runner' +import { activePreviewScriptRunner } from './preview-script-runner' /** Build the idempotent inject-and-run script for one tour action. */ function buildTourScript(action: TourAction): string { @@ -51,7 +51,7 @@ function buildTourScript(action: TourAction): string { /** Run one tour action in the ACTIVE preview tab's page. */ export async function runPreviewTour(action: TourAction): Promise { - const run = activeTourRunner() + const run = activePreviewScriptRunner() if (!run) { return { error: 'No live page is open in the preview pane — open one first.', success: false } diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts index dbb0508510..227a4eb128 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/desktop-bridge.ts @@ -2,6 +2,7 @@ import { readActivePreview } from '@/app/chat/right-rail/preview-reader' import { writeAgentTerminalChunk } from '@/app/right-sidebar/terminal/agent-terminal-stream' import { readActiveTerminal } from '@/app/right-sidebar/terminal/buffer' import { closeAgentTerminalByProc } from '@/app/right-sidebar/terminal/terminals' +import type { PreviewActAction } from '@/lib/preview-act/act-in-page' import type { TourAction, TourStep } from '@/lib/tour' import { $gateway } from '@/store/gateway' import { applyDesktopLayoutPreset, revealDesktopPane } from '@/store/pane-focus' @@ -10,6 +11,34 @@ import { setMessages } from '@/store/session' import type { GatewayEventContext } from './types' +/** The preview engine, loaded on demand so ~25KB of page-injectable source stays + * off the boot path. + * + * In dev that lazy chunk is also a trap. The browser caches a dynamic import by + * URL for the life of the page, so this bridge would hand every action to + * whichever build of the engine loaded first, and no edit to it — or to the + * overlay whose source it stringifies into the page — would reach the guest + * until the whole window reloaded. Asking for a fresh copy is more reliable + * than trusting hot-update propagation to reach a module nothing statically + * imports; the dev server stamps the dependency URLs it has invalidated, so a + * fresh engine pulls a fresh overlay down with it. + * + * The literal path is what a bare specifier can't be here, and it has to track + * this module's real location — hence the fall back to the static import, which + * is also the only branch production keeps, `import.meta.hot` being stripped + * there along with everything it guards. */ +const loadPreviewEngine = () => { + const stable = () => import('@/app/chat/right-rail/preview-act') + + if (!import.meta.hot) { + return stable().then(mod => mod.actOnActivePreview) + } + + return import(/* @vite-ignore */ '/src/app/chat/right-rail/preview-act.ts?hot=' + Date.now()) + .catch(stable) + .then(mod => mod.actOnActivePreview as Awaited>['actOnActivePreview']) +} + /** Desktop-surface bridge events: read-back requests the agent blocks on * (terminal/preview/window), agent terminal streaming, pane reveal, and * message reactions. */ @@ -55,6 +84,50 @@ export function handleDesktopBridgeEvent(ctx: GatewayEventContext): boolean { return true } + if (event.type === 'preview.act.request') { + // drive_preview tool: click/type/scroll/press inside the guest page, or + // drive the pane's history. Dynamic import keeps the injected engine off + // the boot path. Active session only: a background turn must never reach + // into the page the user is working in (desktop AGENTS.md: offer, don't + // hijack). + const requestId = typeof payload?.request_id === 'string' ? payload.request_id : '' + + if (requestId) { + const answer = (result: unknown) => + $gateway.get()?.request('preview.act.respond', { + request_id: requestId, + text: result ? JSON.stringify(result) : '' + }) + + if (isActiveEvent) { + void loadPreviewEngine() + .then(run => + run({ + amount: payload?.amount, + key: payload?.key, + kind: payload?.action ?? '', + max: payload?.max, + ref: payload?.ref, + selector: payload?.selector, + submit: payload?.submit, + text: payload?.text, + to: payload?.to as PreviewActAction['to'] + }) + ) + .then(answer, error => + answer({ error: error instanceof Error ? error.message : String(error), success: false }) + ) + } else { + void answer({ + error: 'The in-app browser only takes actions in the session the user is looking at.', + success: false + }) + } + } + + return true + } + if (event.type === 'window.read.request') { // read_window_below tool: main owns native window enumeration, so ask // it over IPC and answer. Empty text = unavailable (no bridge, or diff --git a/apps/desktop/src/lib/chat-messages/types.ts b/apps/desktop/src/lib/chat-messages/types.ts index 2974b8d56a..0706a73085 100644 --- a/apps/desktop/src/lib/chat-messages/types.ts +++ b/apps/desktop/src/lib/chat-messages/types.ts @@ -121,6 +121,15 @@ export type GatewayEventPayload = { side?: string steps?: unknown step_index?: number + // preview.act.request (drive_preview tool — agent clicking/typing/scrolling in + // the in-app browser). `action` names the verb and `selector` is shared with + // tour above; `ref` addresses an element from the last inventory. + ref?: string + submit?: boolean + key?: string + amount?: number + to?: string + max?: number // message.reaction (agent reacting via the react_to_message tool) — the // durable messages.id, that row's full reaction list after the write, and // the row's role so a live (not-yet-round-tripped) message can be matched. diff --git a/apps/desktop/src/lib/preview-act/act-in-page.test.ts b/apps/desktop/src/lib/preview-act/act-in-page.test.ts new file mode 100644 index 0000000000..f92b1efcc3 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/act-in-page.test.ts @@ -0,0 +1,730 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { actEngineSource, actInPage, type PreviewActHolder } from './act-in-page' + +/** jsdom lays nothing out, so every rect is 0×0 and the engine's visibility + * check would reject the whole page. Give elements a plausible box, honouring + * an explicit width/height so a test can lay out a 1px one, and let + * `display: none` (which jsdom DOES compute) carry the hiding. */ +function layOutTheDocument() { + vi.spyOn(Element.prototype, 'getBoundingClientRect').mockImplementation(function (this: Element) { + const style = getComputedStyle(this) + const width = style.display === 'none' ? 0 : parseFloat(style.width) || 40 + const height = style.display === 'none' ? 0 : parseFloat(style.height) || 40 + const left = parseFloat(style.left) || 0 + + return { bottom: height, height, left, right: left + width, top: 0, width, x: left, y: 0 } as DOMRect + }) +} + +function page(html: string): PreviewActHolder { + document.body.innerHTML = html + + return {} +} + +/** Take the inventory the agent would take before acting. */ +function inventory(holder: PreviewActHolder) { + return actInPage(document, holder, { kind: 'elements' }) +} + +/** Take the inventory and hand back the handle for the first thing on the page. + * Tests address elements the way the agent does — by asking what they are + * called — rather than predicting what the engine will name them. */ +function firstRef(holder: PreviewActHolder) { + return inventory(holder).elements![0].ref +} + +beforeEach(() => { + vi.restoreAllMocks() + layOutTheDocument() + Element.prototype.scrollIntoView = vi.fn() + // jsdom has no hit-testing at all, so the engine skips the occlusion check + // here by default. Tests that stand one in must not leak it into the next. + delete (document as unknown as { elementFromPoint?: unknown }).elementFromPoint +}) + +describe('self-containment', () => { + // The pane injects `actEngineSource()` into the guest page, where module + // scope does not exist. A single free identifier — a module-level constant, + // an imported helper, or a kit factory the assembler forgot to ship — is a + // ReferenceError on every call, and Electron reports it only as "Script + // failed to execute". Evaluating the bundle the way the guest does is the + // only way to catch that here: a plain import of these functions resolves + // the very names the guest would be missing. + // + // It is also the guard on the build. The renderer minifies, so an imported + // binding named inside the stringified core would be renamed to something + // the bundle never declares — green in dev and in this suite if the source + // were assembled any other way, broken only once packaged. + it('runs after being stringified and eval’d with no module scope', () => { + const page = document.createElement('div') + page.innerHTML = '' + document.body.replaceChildren(page) + + const injected = new Function('return ' + actEngineSource())() as typeof actInPage + const holder: PreviewActHolder = {} + + const [save] = injected(document, holder, { kind: 'elements' }).elements! + + expect(save.label).toBe('Save') + + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + expect(injected(document, holder, { kind: 'click', ref: save.ref }).success).toBe(true) + expect(clicked).toHaveBeenCalledOnce() + }) +}) + +describe('elements', () => { + // The handle says what the thing is and which one it is, so a delta line the + // agent reads ten turns later needs no lookup to make sense of. + it('names the interactive nodes after their role and label', () => { + const holder = page(` + + Help + +

Not interactive

+ `) + + const result = inventory(holder) + + expect(result.success).toBe(true) + expect(result.elements?.map(e => [e.ref, e.label])).toEqual([ + ['btn-save', 'Save'], + ['lnk-help', 'Help'], + ['inp-your-name', 'Your name'] + ]) + }) + + it('tells two elements with the same name apart', () => { + const holder = page(` + + + + `) + + expect(inventory(holder).elements?.map(e => e.ref)).toEqual(['btn-edit', 'btn-edit-1', 'btn-edit-2']) + }) + + // Handing a retired name to a different element would silently redirect a + // handle the agent is still holding. The two ids here also disagree, which is + // the page saying outright that these are different buttons — so this must + // not re-bind despite the identical label. + it('never reissues the name of an element that went away', () => { + const holder = page(` +
+ Help + Terms + `) + + expect(firstRef(holder)).toBe('btn-edit') + + document.getElementById('host')!.innerHTML = '' + + const again = inventory(holder) + + expect(again.delta?.removed).toEqual(['btn-edit']) + expect(again.delta?.added?.map(e => e.ref)).toEqual(['btn-edit-1']) + }) + + it('reports role, current value, and disabled state', () => { + const holder = page(` + + + `) + + const [email, go] = inventory(holder).elements! + + expect(email).toMatchObject({ label: 'Email', role: 'input:email', value: 'a@b.co' }) + expect(go).toMatchObject({ disabled: true, label: 'Go' }) + expect(email.disabled).toBeUndefined() + }) + + it('skips hidden controls and unlabelled ones', () => { + const holder = page(` + + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Real']) + }) + + // The screen-reader-only recipe is a 1px box parked at the document origin, + // and it passed a `width >= 1` check. Skip links and CSS-only menu toggles are + // built this way and come FIRST in the document, so they landed on @e1 — the + // agent would aim at one and the pointer would fly to the top-left corner and + // click nothing. + it('skips the visually-hidden controls that sit at the document origin', () => { + const holder = page(` + Jump to content + + Skip navigation + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('skips a control parked off the left edge, which no scroll brings back', () => { + const holder = page(` + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + // Wikipedia and friends are full of these: decorative chrome and collapsed + // menus that are perfectly solid boxes as far as layout is concerned, but + // that the page has already declared are not for anyone to interact with. + it('skips controls the page marked aria-hidden or inert', () => { + const holder = page(` + +
+ + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('skips a control buried under another layer, which would take the click instead', () => { + const holder = page(` + + + `) + + const wall = document.createElement('div') + document.body.append(wall) + + // Stand in for the hit-testing jsdom does not do: every element reports + // itself except the buried one, which reports the sheet lying over it. + document.elementFromPoint = (x: number) => (x === 120 ? wall : document.getElementById('real')) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + }) + + it('keeps a control whose own child is what the hit test lands on', () => { + const holder = page(' Home') + + document.elementFromPoint = () => document.getElementById('icon') + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Home']) + }) + + // The overlay draws far more than the agent is told about, so the two lists + // are collected separately: the field is every visible control, the inventory + // is the describable subset that refs point into. + it('parks the whole interactive field for the overlay, not just the inventory', () => { + const holder = page(` + + + + `) + + expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search']) + expect(holder.nodes).toEqual([document.getElementById('named')]) + expect(holder.field).toEqual([document.getElementById('named'), document.getElementById('mystery')]) + }) + + // The positional `:nth-child` chain this used to fall back to was 74% of a + // real page's inventory and nothing read it. An identity selector is short + // and stable, so it stays; everything else addresses by ref. + it('carries an identity selector only, and omits it when there is none', () => { + const holder = page(` +
+
+
+ `) + + const [byTestId, byId, plain] = inventory(holder).elements! + + expect(byTestId.selector).toBe('[data-testid="submit"]') + expect(byId.selector).toBe('#cancel') + expect(plain.selector).toBeUndefined() + expect(plain.ref).toBe('btn-plain') + }) + + it('honours the cap', () => { + const holder = page(Array.from({ length: 10 }, (_, i) => ``).join('')) + + expect(inventory(holder).elements).toHaveLength(10) + expect(actInPage(document, holder, { full: true, kind: 'elements', max: 3 }).elements).toHaveLength(3) + }) +}) + +// Re-sending the whole inventory after every click is what made a ten-step +// session cost several times what it needed to: the page barely moves between +// steps and the agent was charged for a fresh copy of it each time. +describe('delta', () => { + it('gives the full inventory the first time it looks at a page', () => { + const holder = page('') + const first = inventory(holder) + + expect(first.elements).toHaveLength(1) + expect(first.delta).toBeUndefined() + }) + + it('reports only what moved on every look after that', () => { + const holder = page(` +
+ + + +
+ `) + + inventory(holder) + + document.getElementById('host')!.insertAdjacentHTML('beforeend', '') + + const next = inventory(holder) + + expect(next.elements).toBeUndefined() + expect(next.delta?.added?.map(e => e.ref)).toEqual(['btn-quit']) + expect(next.delta?.same).toBe(3) + }) + + it('says nothing about a page that did not move', () => { + const holder = page('') + inventory(holder) + + const still = inventory(holder) + + expect(still.delta).toEqual({ same: 1 }) + }) + + it('reports a relabelled control as changed, keeping its handle', () => { + const holder = page(` + + Help + Terms + `) + + const ref = firstRef(holder) + + document.getElementById('cart')!.textContent = 'Added' + + const next = inventory(holder) + + expect(next.delta?.changed?.map(e => [e.ref, e.label])).toEqual([[ref, 'Added']]) + expect(actInPage(document, holder, { kind: 'click', ref }).success).toBe(true) + }) + + // `changed` fires on nearly every step of a long task, so it is the one part + // of the payload whose cost compounds. It carries the moved field and the + // handle, and nothing the agent already knows. + it('reports only the field that moved, not the whole element', () => { + const holder = page(` + + Help + Terms + `) + + inventory(holder) + ;(document.getElementById('q') as HTMLInputElement).value = 'shoes' + + const [moved] = inventory(holder).delta!.changed! + + expect(moved).toEqual({ ref: 'inp-search', value: 'shoes' }) + }) + + it('reports a control becoming available on its own', () => { + const holder = page(` + + Help + Terms + `) + + inventory(holder) + ;(document.getElementById('go') as HTMLButtonElement).disabled = false + + expect(inventory(holder).delta?.changed).toEqual([{ disabled: false, ref: 'btn-continue' }]) + }) + + it('falls back to the whole inventory when most of the page is new', () => { + const holder = page('
') + inventory(holder) + + document.getElementById('host')!.innerHTML = 'ABC' + + const next = inventory(holder) + + expect(next.delta).toBeUndefined() + expect(next.elements?.map(e => e.ref)).toEqual(['lnk-a', 'lnk-b', 'lnk-c']) + }) + + // An element that slid past the cap is still clickable, so calling it removed + // would be a lie the agent acts on. + it('does not report an element as removed just because it fell past the cap', () => { + const holder = page(Array.from({ length: 4 }, (_, i) => ``).join('')) + inventory(holder) + + const capped = actInPage(document, holder, { kind: 'elements', max: 2 }) + + expect(capped.delta?.removed).toBeUndefined() + expect(actInPage(document, holder, { kind: 'click', ref: 'btn-b3' }).success).toBe(true) + }) + + // The contract the whole delta exists for. Not a fixed byte count — that + // would break on any wording change — but the relationship between the two + // payloads, which is what has to hold. + it('costs a fraction of the inventory on a page that mostly held still', () => { + const holder = page(` + +
+ ${Array.from({ length: 24 }, (_, i) => ``).join('')} + +
+
+ `) + + const baseline = JSON.stringify(inventory(holder).elements) + + document.getElementById('host')!.innerHTML = '' + document.getElementById('row-3')!.textContent = 'Row action 3 (done)' + + const next = inventory(holder) + + expect(next.delta?.same).toBe(32) + // Measured at roughly 15x on this fixture; the bar is set well below that + // so a wording change does not fail the build, but a regression to + // re-sending the page would. + expect(JSON.stringify(next.delta).length * 5).toBeLessThan(baseline.length) + }) + + it('retires every handle when the page navigates', () => { + const holder = page('') + inventory(holder) + holder.url = 'https://elsewhere.example/other' + + const landed = inventory(holder) + + expect(landed.delta).toBeUndefined() + expect(landed.elements?.map(e => e.ref)).toEqual(['btn-save']) + }) +}) + +// The headline case. A framework re-render destroys the node and builds a new +// one; the agent's handle has to survive that, and it has to hear about it in +// one word rather than as a removal it must react to plus an addition it must +// re-read. +describe('rebind', () => { + /** A page with a stable nav around the part that re-renders, so the re-bind + * has to pick its candidate rather than being handed the only one going. */ + function app(inner: string) { + return page(` + +
${inner}
+ `) + } + + function rerender(inner: string) { + document.getElementById('host')!.innerHTML = inner + } + + it('keeps the handle when a re-render replaces the node', () => { + const holder = app('') + const ref = inventory(holder).elements!.find(e => e.label === 'Sign in')!.ref + const before = document.querySelector('button') + + rerender('') + + const next = inventory(holder) + + expect(document.querySelector('button')).not.toBe(before) + expect(next.delta?.rebound).toEqual([ref]) + expect(next.delta?.added).toBeUndefined() + expect(next.delta?.removed).toBeUndefined() + expect(actInPage(document, holder, { kind: 'click', ref }).success).toBe(true) + }) + + it('follows a label through a count badge appearing on it', () => { + const holder = app('Inbox') + const ref = inventory(holder).elements!.find(e => e.label === 'Inbox')!.ref + + rerender('') + + expect(inventory(holder).delta?.rebound).toEqual([ref]) + }) + + it('will not move a handle across roles', () => { + const holder = app('') + inventory(holder) + + rerender('Continue') + + const next = inventory(holder) + + expect(next.delta?.rebound).toBeUndefined() + expect(next.delta?.removed).toEqual(['btn-continue']) + expect(next.delta?.added?.map(e => e.ref)).toEqual(['lnk-continue']) + }) + + // Two unrelated buttons trading places must not trade handles with them. + it('mints a new handle rather than guess between unrelated candidates', () => { + const holder = app('') + inventory(holder) + + rerender('') + + const next = inventory(holder) + + expect(next.delta?.rebound).toBeUndefined() + expect(next.delta?.removed).toEqual(['btn-delete-account']) + expect(next.delta?.added?.map(e => e.ref)).toEqual(['btn-upload-photo']) + }) + + // Both carry an id and the ids disagree, which is the page saying outright + // that these are two different controls however alike they read. + it('will not move a handle between elements the page marks as distinct', () => { + const holder = app('') + inventory(holder) + + rerender('') + + const next = inventory(holder) + + expect(next.delta?.rebound).toBeUndefined() + expect(next.delta?.removed).toEqual(['btn-save']) + }) +}) + +describe('click', () => { + it('activates the element a ref points at', () => { + const holder = page('') + const ref = firstRef(holder) + + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + const result = actInPage(document, holder, { kind: 'click', ref }) + + expect(result.success).toBe(true) + expect(result.acted).toContain('Save') + expect(clicked).toHaveBeenCalledOnce() + }) + + it('replays the pointer/mouse sequence frameworks bind to', () => { + const holder = page('') + const ref = firstRef(holder) + + const seen: string[] = [] + + for (const type of ['pointerdown', 'mousedown', 'mouseup', 'pointerup', 'click']) { + document.getElementById('save')!.addEventListener(type, () => seen.push(type)) + } + + actInPage(document, holder, { kind: 'click', ref }) + + expect(seen).toContain('mousedown') + expect(seen).toContain('mouseup') + expect(seen.at(-1)).toBe('click') + }) + + it('takes a raw CSS selector when no ref fits', () => { + const holder = page('') + const clicked = vi.fn() + document.getElementById('save')!.addEventListener('click', clicked) + + expect(actInPage(document, holder, { kind: 'click', selector: '#save' }).success).toBe(true) + expect(clicked).toHaveBeenCalledOnce() + }) + + it('refuses a disabled control instead of silently doing nothing', () => { + const holder = page('') + const result = actInPage(document, holder, { kind: 'click', ref: firstRef(holder) }) + + expect(result.success).toBe(false) + expect(result.error).toContain('disabled') + }) + + it('reports the live url so a navigation is visible to the agent', () => { + const holder = page('') + + expect(actInPage(document, holder, { kind: 'click', ref: firstRef(holder) }).url).toBe(document.location.href) + }) +}) + +describe('stale refs', () => { + it('names an unknown ref rather than acting on whatever is nearby', () => { + const holder = page('') + inventory(holder) + + const result = actInPage(document, holder, { kind: 'click', ref: 'btn-imaginary' }) + + expect(result.success).toBe(false) + expect(result.error).toContain('elements') + }) + + it('catches a node that was removed after the snapshot', () => { + const holder = page('') + const ref = firstRef(holder) + document.getElementById('save')!.remove() + + expect(actInPage(document, holder, { kind: 'click', ref }).error).toContain('removed') + }) + + it('invalidates every ref when the page navigated under them', () => { + const holder = page('') + const ref = firstRef(holder) + holder.url = 'https://elsewhere.example/other' + + expect(actInPage(document, holder, { kind: 'click', ref }).error).toContain('navigated') + }) + + it('asks for a target when given neither', () => { + expect(actInPage(document, page(''), { kind: 'click' }).error).toContain('selector') + }) + + it('reports a selector that matches nothing', () => { + expect(actInPage(document, page(''), { kind: 'click', selector: '#nope' }).error).toContain('No element') + }) +}) + +describe('type', () => { + it('enters text and fires the events a controlled input listens for', () => { + const holder = page('') + const ref = firstRef(holder) + + const input = document.getElementById('who') as HTMLInputElement + const events: string[] = [] + input.addEventListener('input', () => events.push('input')) + input.addEventListener('change', () => events.push('change')) + + const result = actInPage(document, holder, { kind: 'type', ref, text: 'Brooklyn' }) + + expect(result.success).toBe(true) + expect(input.value).toBe('Brooklyn') + expect(events).toEqual(['input', 'change']) + }) + + it('bypasses the own-property shadow React installs on tracked inputs', () => { + const holder = page('') + const ref = firstRef(holder) + + // React defines its own `value` accessor on the node to track what it last + // wrote, and ignores an input event that agrees with it. Writing through + // that shadow is exactly how typed text snaps back on the next render. + const input = document.getElementById('who') as HTMLInputElement + const nativeValue = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')! + const shadowWrites: string[] = [] + + Object.defineProperty(input, 'value', { + configurable: true, + get: () => nativeValue.get!.call(input), + set: (next: string) => { + shadowWrites.push(next) + } + }) + + actInPage(document, holder, { kind: 'type', ref, text: 'Brooklyn' }) + + expect(shadowWrites).toEqual([]) + expect(nativeValue.get!.call(input)).toBe('Brooklyn') + }) + + it('writes into a contenteditable host', () => { + const holder = page('
') + + actInPage(document, holder, { kind: 'type', ref: firstRef(holder), text: 'hello' }) + + expect(document.getElementById('editor')!.textContent).toBe('hello') + }) + + it('submits the owning form when asked', () => { + const holder = page('
') + const ref = firstRef(holder) + + const form = document.getElementById('f') as HTMLFormElement + form.requestSubmit = vi.fn() + + const result = actInPage(document, holder, { kind: 'type', ref, submit: true, text: 'cats' }) + + expect(form.requestSubmit).toHaveBeenCalledOnce() + expect(result.acted).toContain('submitted') + }) + + it('refuses a target that has no text to type into', () => { + const holder = page('') + + expect(actInPage(document, holder, { kind: 'type', ref: firstRef(holder), text: 'x' }).error).toContain( + 'not a text field' + ) + }) +}) + +describe('press', () => { + it('sends the key to the target', () => { + const holder = page('') + const ref = firstRef(holder) + + const keys: string[] = [] + document.getElementById('q')!.addEventListener('keydown', e => keys.push((e as KeyboardEvent).key)) + + expect(actInPage(document, holder, { key: 'Enter', kind: 'press', ref }).success).toBe(true) + expect(keys).toEqual(['Enter']) + }) + + it('needs a key', () => { + const holder = page('') + + expect(actInPage(document, holder, { kind: 'press', ref: firstRef(holder) }).error).toContain('key') + }) +}) + +describe('scroll', () => { + it('scrolls the page by about a screen when given no distance', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + const result = actInPage(document, holder, { kind: 'scroll' }) + + expect(result.success).toBe(true) + expect(scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: Math.round(window.innerHeight * 0.9) }) + }) + + it('drops the animation for a reader who asked for reduced motion', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + // Passing 'smooth' explicitly overrides the OS preference, so the engine has + // to opt back out itself rather than leaving it to the browser. + vi.stubGlobal('matchMedia', () => ({ matches: true })) + + actInPage(document, holder, { kind: 'scroll' }) + vi.unstubAllGlobals() + + const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions] + + expect(options.behavior).toBe('auto') + }) + + it('jumps to the bottom', () => { + const holder = page('

long page

') + const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + actInPage(document, holder, { kind: 'scroll', to: 'bottom' }) + + const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions] + + expect(options.top).toBeGreaterThan(window.innerHeight) + }) + + it('scrolls a ref’d container instead of the page', () => { + const holder = page('
') + const ref = firstRef(holder) + + const list = document.getElementById('list') as HTMLElement + list.scrollBy = vi.fn() + const pageScroll = vi.spyOn(window, 'scrollBy').mockImplementation(() => {}) + + const result = actInPage(document, holder, { amount: 200, kind: 'scroll', ref }) + + expect(list.scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: 200 }) + expect(pageScroll).not.toHaveBeenCalled() + expect(result.acted).toContain('Results') + }) +}) diff --git a/apps/desktop/src/lib/preview-act/act-in-page.ts b/apps/desktop/src/lib/preview-act/act-in-page.ts new file mode 100644 index 0000000000..20b3637dd4 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/act-in-page.ts @@ -0,0 +1,634 @@ +/** + * PREVIEW ACT ENGINE — the one function that performs an interaction inside a + * page, so the agent can drive the in-app browser instead of only reading it. + * + * `elements` hands back an inventory of what can be interacted with, each under + * a durable handle that says what it is — `btn-sign-in`, `inp-email` — and + * parks the matching nodes on a holder. Every other verb resolves its target + * from that holder. Once the agent has a baseline for a page, later looks + * answer with a delta instead of the whole inventory (see `survey`). + * + * SELF-CONTAINMENT, and why the helpers arrive as an argument. + * + * The core is injected into the guest page as source, where module scope does + * not exist: a single free identifier is a ReferenceError on every call, which + * Electron reports only as "Script failed to execute". That much has always + * been true. The subtler half is that the renderer build MINIFIES, so an + * imported binding named here would be renamed to something the injected + * prelude never declared — and it would break in packaged builds only, staying + * green in dev and in tests. So the core names nothing it does not receive: + * `kit` carries the naming, visibility, and identity helpers, and + * `actEngineSource` is what assembles them into one injectable string. + * + * `actInPage` below is the ordinary in-process entry point. It is NOT + * stringified and is free to import. + */ + +import { identityKit } from './identity' +import type { IdentityKit } from './identity' +import { namingKit } from './naming' +import type { NamingKit } from './naming' +import type { + PreviewActAction, + PreviewActBinding, + PreviewActDelta, + PreviewActHolder, + PreviewActResult, + PreviewElement, + PreviewElementChange +} from './types' +import { visibilityKit } from './visibility' +import type { VisibilityKit } from './visibility' + +export type { + PreviewActAction, + PreviewActBinding, + PreviewActDelta, + PreviewActHolder, + PreviewActResult, + PreviewElement, + PreviewElementChange +} from './types' + +/** Everything the core needs that it cannot define for itself. */ +export interface ActKit { + identity: IdentityKit + naming: NamingKit + visibility: VisibilityKit +} + +/** Assemble the kits for one document. Used in-process; the guest page gets + * the same three factories through `actEngineSource`. */ +export function buildActKit(doc: Document): ActKit { + const naming = namingKit(doc) + + return { identity: identityKit(naming), naming, visibility: visibilityKit(doc, doc.defaultView) } +} + +/** Run one action against `doc`, resolving refs through `holder`. */ +export function actInPage(doc: Document, holder: PreviewActHolder, action: PreviewActAction): PreviewActResult { + return actInPageCore(doc, holder, action, buildActKit(doc)) +} + +/** The injectable core. Everything it touches is a parameter — see the + * self-containment note above. */ +export function actInPageCore( + doc: Document, + holder: PreviewActHolder, + action: PreviewActAction, + kit: ActKit +): PreviewActResult { + const { anchorOf, labelOf, selectorFor, stableOf, valueOf } = kit.naming + const { affinity, coin, shifted } = kit.identity + const { onScreen, visible } = kit.visibility + + // Declared inside, not at module scope: this body is stringified and eval'd + // in the guest page, where a module-level constant is simply not defined. + const maxElements = 120 + // How many nodes the OVERLAY may draw, as opposed to how many the agent gets + // told about. Far higher, because an extra mark costs one rect read where an + // extra inventory row costs tokens on every single call. + const maxMarks = 600 + // How alike a remembered element and a fresh one have to be before the + // handle moves across. Below it we mint a new handle instead: a re-render + // costing the agent a re-read is a cheap mistake, and a handle silently + // pointing at the wrong button is not. + const rebindBar = 0.6 + const win = doc.defaultView + const here = doc.location ? doc.location.href : '' + + // Passing 'smooth' explicitly OVERRIDES the user's OS-level reduce-motion + // setting, so ask before animating over someone who opted out. + const still = !!(win && win.matchMedia && win.matchMedia('(prefers-reduced-motion: reduce)').matches) + const glide: ScrollBehavior = still ? 'auto' : 'smooth' + + + /** Walk the page and hand back what is interactable, in document order. The + * handles are assigned afterwards, by `survey`. */ + const sight = (max: number) => { + const nodes: Element[] = [] + const field: Element[] = [] + const elements: PreviewElement[] = [] + + const candidates = doc.querySelectorAll( + 'a[href], button, input:not([type="hidden"]), select, textarea, summary, label[for], ' + + '[role="button"], [role="link"], [role="checkbox"], [role="radio"], [role="tab"], ' + + '[role="menuitem"], [role="switch"], [role="option"], [role="combobox"], [role="searchbox"], ' + + '[role="textbox"], [contenteditable=""], [contenteditable="true"], [onclick], ' + + '[tabindex]:not([tabindex="-1"])' + ) + + for (const el of candidates) { + if (elements.length >= max && field.length >= maxMarks) { + break + } + + if (!visible(el)) { + continue + } + + // Drawable from here on. The two lists diverge deliberately: the field is + // what is on screen to be outlined, the inventory is what the agent can + // name and reach — which includes things below the fold it will scroll to. + if (field.length < maxMarks && onScreen(el)) { + field.push(el) + } + + if (elements.length >= max) { + continue + } + + const tag = el.tagName.toLowerCase() + const role = el.getAttribute('role') || (tag === 'input' ? 'input:' + ((el as HTMLInputElement).type || 'text') : tag) + const label = labelOf(el) + const value = valueOf(el) + + // A control with neither a label nor a value is not addressable in prose + // — the agent could not tell it apart from its unlabelled neighbours. + // It is also the one case a durable handle cannot be minted for, since + // there would be nothing to name it after and nothing to re-find it by, + // so dropping it here keeps every handle we DO mint anchorable. + if (!label && !value) { + continue + } + + const entry: PreviewElement = { + label, + // Filled in by `survey`, which is what knows whether this element + // already has a handle. + ref: '', + role + } + + const selector = selectorFor(el) + + if (selector) { + entry.selector = selector + } + + if ((el as HTMLInputElement).disabled) { + entry.disabled = true + } + + if (value) { + entry.value = value + } + + nodes.push(el) + elements.push(entry) + } + + holder.nodes = nodes + holder.field = field + + return { elements, nodes } + } + + + /** Look at the page and say what is there — or, once there is something to + * compare against, only what moved. + * + * Re-sending the whole inventory every action is what took a ten-step + * session from 45k to 85k tokens of context: the page barely changes between + * a scroll and a click, and the agent was being charged for a fresh copy of + * it each time. */ + const survey = (max: number): PreviewActResult => { + // A navigation is a different page. Every handle on the old one is retired + // rather than rebound onto whatever now sits in the same place. + const fresh = holder.url !== here + + if (fresh) { + holder.book = [] + holder.coined = {} + } + + const book = holder.book || (holder.book = []) + const coined = holder.coined || (holder.coined = {}) + const seen = sight(max) + const claimed: Record = {} + const kept: PreviewActBinding[] = [] + const added: PreviewElement[] = [] + const changed: PreviewElementChange[] = [] + const rebound: string[] = [] + const known = new Map() + const waiting: number[] = [] + let same = 0 + + for (const bound of book) { + known.set(bound.el, bound) + } + + // Pass one: the element object itself is still the one we remember. Free, + // and it is what happens on a scroll, a hover, and most clicks. + for (let i = 0; i < seen.elements.length; i++) { + const entry = seen.elements[i] + const bound = known.get(seen.nodes[i]) + + if (!bound) { + waiting.push(i) + + continue + } + + const moved = shifted(bound, entry) + + entry.ref = bound.ref + claimed[bound.ref] = true + kept.push(bound) + + if (!moved) { + same++ + + continue + } + + bound.label = entry.label + bound.name = entry.label || entry.value || '' + bound.off = !!entry.disabled + bound.value = entry.value || '' + changed.push(moved) + } + + // Pass two: whatever is left either replaced something (a framework threw + // the node away and built a new one) or is genuinely new. The pool is only + // the handles whose element is GONE, which is both the correct candidate + // set and a small one. + const pool = book.filter(bound => !claimed[bound.ref] && !doc.contains(bound.el)) + + for (const i of waiting) { + const entry = seen.elements[i] + const el = seen.nodes[i] + + const now: PreviewActBinding = { + el, + label: entry.label, + name: entry.label || entry.value || '', + off: !!entry.disabled, + path: anchorOf(el), + ref: '', + role: entry.role, + stable: stableOf(el), + value: entry.value || '' + } + + let best: PreviewActBinding | undefined + let score = 0 + + for (const bound of pool) { + if (claimed[bound.ref]) { + continue + } + + const rung = affinity(bound, now) + + // Strictly better, so a tie goes to whichever candidate the page put + // first and the same page twice re-binds the same way. + if (rung >= rebindBar && rung > score) { + best = bound + score = rung + } + } + + if (best) { + // Same handle, new node. Reported as one word rather than a removal + // and an addition, because from the agent's side nothing happened — + // its handle still works and it has nothing to re-read. + entry.ref = best.ref + best.el = el + best.label = now.label + best.name = now.name + best.off = now.off + best.path = now.path + best.stable = now.stable + best.value = now.value + claimed[best.ref] = true + kept.push(best) + rebound.push(best.ref) + + continue + } + + now.ref = coin(coined, entry.role, now.name) + entry.ref = now.ref + claimed[now.ref] = true + kept.push(now) + added.push(entry) + } + + // A handle nobody claimed is gone ONLY if its element really left. One that + // is still on the page but fell past `max` keeps working and is simply not + // mentioned — saying "removed" about something the agent can still click + // would be worse than saying nothing. + const removed: string[] = [] + + for (const bound of book) { + if (claimed[bound.ref]) { + continue + } + + if (doc.contains(bound.el) && visible(bound.el)) { + kept.push(bound) + + continue + } + + removed.push(bound.ref) + } + + holder.book = kept + holder.url = here + + // The delta has to actually be cheaper. When half the page is new there is + // nothing to reuse, and a delta is then just the inventory with extra + // framing around it. + const churn = added.length + changed.length + + if (fresh || action.full || churn * 2 >= seen.elements.length) { + return { elements: seen.elements, success: true } + } + + const delta: PreviewActDelta = { same } + + if (added.length) { + delta.added = added + } + + if (changed.length) { + delta.changed = changed + } + + if (removed.length) { + delta.removed = removed + } + + if (rebound.length) { + delta.rebound = rebound + } + + return { delta, success: true } + } + + /** Resolve the action's target: a handle from the book, else a selector. */ + const resolve = (): { el?: Element; error?: string } => { + const ref = (action.ref || '').trim() + + if (ref) { + if (holder.url !== here) { + return { error: 'The page navigated since the last snapshot, so ' + ref + ' no longer points anywhere. Call elements again.' } + } + + const bound = (holder.book || []).filter(entry => entry.ref === ref)[0] + + if (!bound) { + return { error: 'Unknown element ' + ref + '. Call elements to get current refs.' } + } + + if (!doc.contains(bound.el)) { + return { error: ref + ' has been removed from the page since the last snapshot. Call elements again.' } + } + + return { el: bound.el } + } + + const selector = (action.selector || '').trim() + + if (!selector) { + return { error: 'Pass a ref from elements, or a CSS selector.' } + } + + let el: Element | null = null + + try { + el = doc.querySelector(selector) + } catch { + return { error: 'Not a valid CSS selector: ' + selector } + } + + return el ? { el } : { error: 'No element matches ' + selector + '.' } + } + + const describe = (el: Element) => { + const label = labelOf(el) + + return el.tagName.toLowerCase() + (label ? ' "' + label + '"' : '') + } + + const answer = (result: PreviewActResult): PreviewActResult => ({ + ...result, + title: doc.title || '', + url: doc.location ? doc.location.href : '' + }) + + const fail = (error: string) => answer({ error, success: false }) + + /** Center of `el` in viewport coords, for pointer events that read position. */ + const pointAt = (el: Element) => { + const rect = el.getBoundingClientRect() + + return { clientX: Math.round(rect.left + rect.width / 2), clientY: Math.round(rect.top + rect.height / 2) } + } + + if (action.kind === 'elements') { + const looked = survey(Math.max(1, Math.min(action.max || maxElements, maxElements))) + const empty = !looked.delta && !(looked.elements || []).length + + return answer({ + ...looked, + note: empty ? 'No interactive elements found — the page may still be loading.' : undefined + }) + } + + if (action.kind === 'scroll') { + const target = action.ref || action.selector ? resolve() : {} + + if (target.error) { + return fail(target.error) + } + + const scroller = (target.el as HTMLElement | undefined) || null + const page = Math.round((win ? win.innerHeight : 800) * 0.9) + const by = action.to === 'top' ? -1e7 : action.to === 'bottom' ? 1e7 : (action.amount ?? page) + + if (scroller) { + scroller.scrollBy({ behavior: glide, top: by }) + } else if (win) { + win.scrollBy({ behavior: glide, top: by }) + } + + return answer({ acted: scroller ? 'scrolled ' + describe(scroller) : 'scrolled the page', success: true }) + } + + const target = resolve() + + if (target.error || !target.el) { + return fail(target.error || 'No target.') + } + + const el = target.el as HTMLElement + + // Park the target where the watch overlay can find it, before anything can + // fail — a ring around the thing that turned out to be disabled is exactly + // the feedback someone watching wants. + holder.aimed = el + + // 'locate' is the look-before-you-act half of an action: it brings the target + // on screen and names it, so the overlay has something to draw while the real + // verb is still a beat away. + if (action.kind === 'locate') { + if (el.scrollIntoView) { + el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' }) + } + + if (action.focus) { + el.focus() + } + + const spot = pointAt(el) + const tag = el.tagName + + return answer({ + acted: 'looking at ' + describe(el), + point: { x: spot.clientX, y: spot.clientY }, + success: true, + // Real typing starts with a triple-click to clear the field. On anything + // that is not a field that gesture selects the paragraph under it + // instead, which is how the agent ended up highlighting whole pages. + typable: tag === 'TEXTAREA' || tag === 'INPUT' || el.isContentEditable === true + }) + } + + // Instant, unlike the `scroll` verb above. Getting a target on screen is + // plumbing for the click that follows, not something anyone asked to watch — + // and every millisecond of it is time the caller spends waiting for the page + // to stop moving before it can aim real input at a fixed coordinate. + if (el.scrollIntoView) { + el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' }) + } + + // Hovering a disabled control is allowed — a disabled button with a tooltip + // explaining WHY is exactly the thing worth hovering. + if (action.kind === 'hover') { + const init = { bubbles: true, cancelable: true, ...pointAt(el) } + + if (typeof PointerEvent === 'function') { + el.dispatchEvent(new PointerEvent('pointerover', init)) + el.dispatchEvent(new PointerEvent('pointermove', init)) + } + + el.dispatchEvent(new MouseEvent('mouseover', init)) + el.dispatchEvent(new MouseEvent('mousemove', init)) + + return answer({ acted: 'hovered over ' + describe(el), success: true }) + } + + if ((el as HTMLInputElement).disabled) { + return fail(describe(el) + ' is disabled.') + } + + if (action.kind === 'click') { + const init = { bubbles: true, cancelable: true, ...pointAt(el) } + + // Frameworks bind to the pointer/mouse pair as often as to click itself, so + // replay the whole sequence; el.click() then runs the native activation + // (following a link, toggling a checkbox, submitting a form) that a bare + // synthetic MouseEvent would leave to chance. + if (typeof PointerEvent === 'function') { + el.dispatchEvent(new PointerEvent('pointerdown', init)) + el.dispatchEvent(new PointerEvent('pointerup', init)) + } + + el.dispatchEvent(new MouseEvent('mousedown', init)) + el.dispatchEvent(new MouseEvent('mouseup', init)) + el.click() + + return answer({ acted: 'clicked ' + describe(el), success: true }) + } + + if (action.kind === 'type') { + const text = action.text ?? '' + + const editable = el.isContentEditable || (el.getAttribute('contenteditable') ?? 'false') !== 'false' + + el.focus() + + if (editable) { + el.textContent = text + } else if (el.tagName === 'INPUT' || el.tagName === 'TEXTAREA') { + // Assign through the prototype's setter: React (and anything else that + // tracks the DOM value) shadows `value` with its own accessor and ignores + // an input event whose value it thinks it already wrote, so a plain + // `el.value = …` types into a field that snaps back on the next render. + const proto = el.tagName === 'TEXTAREA' ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype + const setter = Object.getOwnPropertyDescriptor(proto, 'value')?.set + + if (setter) { + setter.call(el, text) + } else { + ;(el as HTMLInputElement).value = text + } + } else if (el.tagName === 'SELECT') { + return fail(describe(el) + ' is a dropdown — click it and click the option you want.') + } else { + return fail(describe(el) + ' is not a text field.') + } + + el.dispatchEvent(new Event('input', { bubbles: true })) + el.dispatchEvent(new Event('change', { bubbles: true })) + + if (action.submit) { + const enter = { bubbles: true, cancelable: true, code: 'Enter', key: 'Enter' } + + el.dispatchEvent(new KeyboardEvent('keydown', enter)) + el.dispatchEvent(new KeyboardEvent('keyup', enter)) + + const form = (el as HTMLInputElement).form + + if (form) { + if (form.requestSubmit) { + form.requestSubmit() + } else { + form.submit() + } + } + } + + return answer({ acted: 'typed into ' + describe(el) + (action.submit ? ' and submitted' : ''), success: true }) + } + + if (action.kind === 'press') { + const key = action.key || '' + + if (!key) { + return fail('Pass the key to press, e.g. "Enter" or "Escape".') + } + + const init = { bubbles: true, cancelable: true, code: key.length === 1 ? 'Key' + key.toUpperCase() : key, key } + + el.focus() + el.dispatchEvent(new KeyboardEvent('keydown', init)) + el.dispatchEvent(new KeyboardEvent('keyup', init)) + + return answer({ acted: 'pressed ' + key + ' on ' + describe(el), success: true }) + } + + return fail('Unknown action: ' + String(action.kind)) +} + +/** The engine as one injectable expression: `function (doc, holder, action)`. + * + * The four factories are stringified together so the core's `kit` is built + * inside the guest page, out of sources that travelled with it. This is the + * ONLY supported way to get the engine's source — taking `.toString()` of any + * one piece yields a function whose helpers are not defined, which fails at + * call time rather than at injection. */ +export function actEngineSource(): string { + return `(function (doc, holder, action) { + var naming = (${namingKit.toString()})(doc); + var kit = { + naming: naming, + identity: (${identityKit.toString()})(naming), + visibility: (${visibilityKit.toString()})(doc, doc.defaultView) + }; + return (${actInPageCore.toString()})(doc, holder, action, kit); +})` +} diff --git a/apps/desktop/src/lib/preview-act/identity.ts b/apps/desktop/src/lib/preview-act/identity.ts new file mode 100644 index 0000000000..6c47e89594 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/identity.ts @@ -0,0 +1,129 @@ +/** + * IDENTITY — deciding whether the element in front of us is one we already + * have a handle on, and naming it if it is not. + * + * This is what makes a delta possible at all. If handles churned every time a + * framework re-rendered, every look at the page would be a wall of removals and + * additions and there would be nothing to send but the whole inventory again. + * + * The re-bind ladder is ported from anchortree (Apache-2.0), minus its geometry + * rung. It is deliberately conservative: below the bar we mint a new handle + * instead of guessing, because a re-render costing the agent a re-read is a + * cheap mistake and a handle silently pointing at the wrong button is not. + * + * SELF-CONTAINMENT: see `naming.ts`. Same contract, same reason for the + * factory shape — with one extra consequence. `identityKit` takes the naming + * helpers as an argument rather than importing them, because an import would + * be a cross-module identifier the bundler is free to rename out from under + * the stringified body. + */ + +import type { NamingKit } from './naming' +import type { PreviewActBinding, PreviewElement, PreviewElementChange } from './types' + +export interface IdentityKit { + /** How strongly a remembered element matches one just observed, 0 to 1. */ + affinity(was: PreviewActBinding, now: PreviewActBinding): number + /** Do two labels share at least half their words? */ + alike(a: string, b: string): boolean + /** Mint a handle, disambiguating against the ones already minted. */ + coin(coined: Record, role: string, name: string): string + /** What moved on an element that kept its handle, or null if it held still. */ + shifted(was: PreviewActBinding, entry: PreviewElement): null | PreviewElementChange +} + +/** Build the identity helpers on top of a naming kit. */ +export function identityKit(naming: NamingKit): IdentityKit { + /** What moved on an element that kept its handle, or nothing if it held + * still. Field by field, so a status line ticking over costs the agent one + * short line instead of a re-run of everything already known about it. */ + const shifted = (was: PreviewActBinding, entry: PreviewElement): null | PreviewElementChange => { + const off = !!entry.disabled + const value = entry.value || '' + + if (was.label === entry.label && was.value === value && was.off === off) { + return null + } + + const moved: PreviewElementChange = { ref: was.ref } + + if (was.label !== entry.label) { + moved.label = entry.label + } + + if (was.value !== value) { + moved.value = value + } + + if (was.off !== off) { + moved.disabled = off + } + + return moved + } + + /** Do two labels share at least half their words? Tolerates the count badge + * and the copy edit — "Inbox" against "Inbox (3)". */ + const alike = (a: string, b: string): boolean => { + if (!a || !b) { + return false + } + + const one = a.toLowerCase().split(/\s+/).filter(Boolean) + const two = b.toLowerCase().split(/\s+/).filter(Boolean) + const both = one.filter(word => two.indexOf(word) !== -1).length + const all = one.length + two.filter(word => one.indexOf(word) === -1).length + + return all > 0 && both / all >= 0.5 + } + + /** How strongly a remembered element matches one just observed, 0 to 1. + * + * Ported from anchortree's re-bind ladder (Apache-2.0), minus its geometry + * rung: a centroid is only ever worth 0.1 there, it never reaches the 0.6 + * bar on its own, and carrying coordinates through the book to buy a + * tie-break is not worth the measurement. */ + const affinity = (was: PreviewActBinding, now: PreviewActBinding): number => { + // A button is not a link, however alike the rest of it reads. + if (was.role !== now.role) { + return 0 + } + + // Two elements that BOTH carry a stable attribute and disagree are the page + // telling us outright that they are different things. + if (was.stable && now.stable) { + return was.stable === now.stable ? 1 : 0 + } + + let score = 0 + + if (was.name && was.name === now.name) { + score += 0.6 + } else if (alike(was.name, now.name)) { + score += 0.4 + } + + if (was.path && was.path === now.path) { + score += 0.3 + } + + return score + } + + /** Mint a handle. Legible on purpose: the agent reads `btn-sign-in` in a + * three-line delta on turn nine and knows what it is, where `@e42` would + * send it back to an inventory twenty thousand tokens ago. Suffixes are + * never rewound, so a retired handle's name is not later handed to a + * different element on the same page. */ + const coin = (coined: Record, role: string, name: string): string => { + const named = naming.slug(name) + const stem = naming.stemOf(role) + (named ? '-' + named : '') + const nth = coined[stem] || 0 + + coined[stem] = nth + 1 + + return nth ? stem + '-' + nth : stem + } + + return { affinity, alike, coin, shifted } +} diff --git a/apps/desktop/src/lib/preview-act/naming.test.ts b/apps/desktop/src/lib/preview-act/naming.test.ts new file mode 100644 index 0000000000..3c633ce40a --- /dev/null +++ b/apps/desktop/src/lib/preview-act/naming.test.ts @@ -0,0 +1,249 @@ +import { beforeEach, describe, expect, it } from 'vitest' + +import { identityKit } from './identity' +import { namingKit } from './naming' +import type { PreviewActBinding } from './types' + +const naming = () => namingKit(document) + +function el(html: string): Element { + document.body.innerHTML = html + + return document.body.firstElementChild! +} + +/** A remembered element, with only the fields a given test cares about set. */ +function bound(over: Partial): PreviewActBinding { + return { + el: document.body, + label: '', + name: '', + off: false, + path: 'root', + ref: 'btn-x', + role: 'button', + stable: '', + value: '', + ...over + } +} + +beforeEach(() => { + document.body.innerHTML = '' +}) + +describe('slug', () => { + it('reads as words a person would recognise', () => { + expect(naming().slug('Sign In')).toBe('sign-in') + expect(naming().slug(' Add to cart! ')).toBe('add-to-cart') + }) + + it('collapses runs of punctuation rather than stacking hyphens', () => { + expect(naming().slug('Save — and / continue')).toBe('save-and-continue') + }) + + it('stays short enough to read at a glance, and never ends on a hyphen', () => { + const long = naming().slug('Subscribe to our weekly newsletter for updates') + + expect(long.length).toBeLessThanOrEqual(24) + expect(long.endsWith('-')).toBe(false) + }) + + it('has nothing to say about an unlabelled element', () => { + expect(naming().slug('!!!')).toBe('') + }) +}) + +describe('stemOf', () => { + // The stem is the part of a handle that survives translation: whatever the + // button says, `btn-` tells the agent it is a button. + it('gives each interactive role its own prefix', () => { + const { stemOf } = naming() + + expect(stemOf('button')).toBe('btn') + expect(stemOf('a')).toBe('lnk') + expect(stemOf('input:search')).toBe('srch') + expect(stemOf('input:checkbox')).toBe('chk') + expect(stemOf('select')).toBe('sel') + }) + + it('treats every other kind of text field as one thing', () => { + const { stemOf } = naming() + + expect(stemOf('input:email')).toBe('inp') + expect(stemOf('input:date')).toBe('inp') + expect(stemOf('textbox')).toBe('inp') + }) + + it('falls back rather than inventing a prefix for something unknown', () => { + expect(naming().stemOf('marquee')).toBe('el') + }) +}) + +describe('labelOf', () => { + it('prefers what the page tells assistive clients', () => { + expect(naming().labelOf(el(''))).toBe('Close dialog') + }) + + it('follows aria-labelledby to the element that names it', () => { + document.body.innerHTML = '

Billing

' + + expect(naming().labelOf(document.querySelector('button')!)).toBe('Billing') + }) + + it('falls back through placeholder and title before giving up', () => { + expect(naming().labelOf(el(''))).toBe('Your email') + expect(naming().labelOf(el(''))).toBe('Help') + expect(naming().labelOf(el(''))).toBe('') + }) + + it('flattens the whitespace a formatted template leaves behind', () => { + expect(naming().labelOf(el(''))).toBe('Save changes') + }) +}) + +describe('selectorFor', () => { + // Identity only. A positional chain was 74% of the inventory and wrong the + // moment a sibling appeared. + it('names the element only when the page gave it a durable name', () => { + const { selectorFor } = naming() + + expect(selectorFor(el(''))).toBe('#save') + expect(selectorFor(el(''))).toBe('[data-testid="save"]') + expect(selectorFor(el(''))).toBe('') + }) +}) + +describe('valueOf', () => { + it('reports what a control currently holds', () => { + const input = el('') + + expect(naming().valueOf(input)).toBe('shoes') + }) + + // The DOM hands back "on" for an unset checkbox value, so reading the value + // first made both states identical to the agent. + it('reports a checkbox as its state rather than the value the DOM invents', () => { + const box = el('') as HTMLInputElement + + expect(naming().valueOf(box)).toBe('unchecked') + + box.checked = true + + expect(naming().valueOf(box)).toBe('checked') + }) + + it('still reports a checkbox state when the page did set a value', () => { + const box = el('') as HTMLInputElement + box.checked = true + + expect(naming().valueOf(box)).toBe('checked') + }) + + it('leaves a radio group readable the same way', () => { + const radio = el('') as HTMLInputElement + + expect(naming().valueOf(radio)).toBe('unchecked') + }) +}) + +describe('anchorOf', () => { + // Coarse on purpose: a wrapper div appearing in the chain must not change + // this string, because that churn is exactly what a re-bind sees through. + it('places an element by the landmark it sits in', () => { + document.body.innerHTML = '' + + expect(naming().anchorOf(document.querySelector('button')!)).toBe('nav') + }) + + it('tells two same-tag landmarks apart by their label', () => { + document.body.innerHTML = '' + + expect(naming().anchorOf(document.querySelector('button')!)).toBe('nav#primary-nav') + }) + + it('says so when there is no landmark to place it in', () => { + expect(naming().anchorOf(el(''))).toBe('root') + }) +}) + +describe('alike', () => { + const { alike } = identityKit(naming()) + + it('sees through a count badge and a copy edit', () => { + expect(alike('Inbox', 'Inbox (3)')).toBe(true) + expect(alike('Save changes', 'Save your changes')).toBe(true) + }) + + it('does not call two unrelated labels the same thing', () => { + expect(alike('Sign in', 'Delete account')).toBe(false) + expect(alike('Save', '')).toBe(false) + }) +}) + +describe('affinity', () => { + const { affinity } = identityKit(naming()) + + it('is certain when both sides carry the same stable attribute', () => { + expect(affinity(bound({ stable: 'save' }), bound({ stable: 'save' }))).toBe(1) + }) + + // The page is telling us outright that these are different things. + it('refuses outright when two stable attributes disagree', () => { + expect(affinity(bound({ stable: 'save' }), bound({ stable: 'cancel' }))).toBe(0) + }) + + it('refuses to move a handle across roles, however alike the rest reads', () => { + const was = bound({ name: 'Save', role: 'button' }) + const now = bound({ name: 'Save', role: 'a' }) + + expect(affinity(was, now)).toBe(0) + }) + + it('clears the bar on name and place together, and not on place alone', () => { + const named = affinity(bound({ name: 'Save', path: 'main' }), bound({ name: 'Save', path: 'main' })) + const placed = affinity(bound({ name: 'Save', path: 'main' }), bound({ name: 'Publish', path: 'main' })) + + expect(named).toBeGreaterThanOrEqual(0.6) + expect(placed).toBeLessThan(0.6) + }) +}) + +describe('coin', () => { + const mint = () => { + const { coin } = identityKit(naming()) + const coined: Record = {} + + return (role: string, name: string) => coin(coined, role, name) + } + + it('names a handle after what the thing is and what it says', () => { + const coin = mint() + + expect(coin('button', 'Sign in')).toBe('btn-sign-in') + expect(coin('input:email', 'Email')).toBe('inp-email') + }) + + it('tells repeats apart without renaming the first one', () => { + const coin = mint() + + expect(coin('button', 'Edit')).toBe('btn-edit') + expect(coin('button', 'Edit')).toBe('btn-edit-1') + expect(coin('button', 'Edit')).toBe('btn-edit-2') + }) + + // Never rewound, so a retired handle's name is not later handed to a + // different element on the same page. + it('does not reissue a name after the element holding it goes away', () => { + const coin = mint() + + coin('button', 'Edit') + coin('button', 'Edit') + + expect(coin('button', 'Edit')).toBe('btn-edit-2') + }) + + it('still mints something for a role it has no name for', () => { + expect(mint()('button', '')).toBe('btn') + }) +}) diff --git a/apps/desktop/src/lib/preview-act/naming.ts b/apps/desktop/src/lib/preview-act/naming.ts new file mode 100644 index 0000000000..8c5a018564 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/naming.ts @@ -0,0 +1,214 @@ +/** + * NAMING — what an element is called, and what it is called by. + * + * Everything here turns an element into a short string: the accessible name, + * the durable handle's stem and slug, the landmark it sits under, the identity + * selector. No page state, no holder, no side effects. + * + * SELF-CONTAINMENT. `namingKit` is stringified into the guest page along with + * the engine that uses it (see `act-in-page.ts`'s contract). Module scope does + * not exist there, so the factory closes over nothing but its own argument and + * declares every helper inside itself. A single free identifier is a + * ReferenceError on every call, and Electron reports it only as "Script failed + * to execute". + * + * That constraint is also why this is a factory rather than nine exports: nine + * exports would be nine names for the bundler to mangle independently of the + * call sites inside the stringified engine. One function that stringifies whole + * has no cross-module references to break. + */ + +export interface NamingKit { + /** Where the element sits, coarsely — the nearest landmark. */ + anchorOf(el: Element): string + clamp(text: string, max: number): string + cssEscape(value: string): string + /** The accessible name, by the usual ladder, truncated. */ + labelOf(el: Element): string + /** An `#id` or `[data-testid]`, or empty when the page offers neither. */ + selectorFor(el: Element): string + /** Lowercase, hyphenated, and short enough to read at a glance. */ + slug(name: string): string + /** The strongest identity signal the page offers, if it offers one. */ + stableOf(el: Element): string + /** The handle stem for a role: `btn`, `inp`, `lnk` … */ + stemOf(role: string): string + /** A form control's current value, or its checked state. */ + valueOf(el: Element): string +} + +/** Build the naming helpers against one document. */ +export function namingKit(doc: Document): NamingKit { + const cssEscape = (value: string) => + typeof CSS !== 'undefined' && CSS.escape ? CSS.escape(value) : value.replace(/["\\]/g, '\\$&') + + const clamp = (text: string, max: number) => (text.length > max ? text.slice(0, max - 1) + '…' : text) + + const labelOf = (el: Element): string => { + const aria = el.getAttribute('aria-label') + + if (aria) { + return clamp(aria, 80) + } + + const labelledBy = el.getAttribute('aria-labelledby') + const labelled = labelledBy ? doc.getElementById(labelledBy) : null + const text = ((labelled || el).textContent || '').trim().replace(/\s+/g, ' ') + + if (text) { + return clamp(text, 80) + } + + for (const attr of ['placeholder', 'title', 'alt', 'name', 'value']) { + const value = el.getAttribute(attr) + + if (value) { + return clamp(value, 80) + } + } + + return '' + } + + /** The strongest identity signal the page offers, if it offers one. Nothing a + * re-render does to the surrounding markup disturbs these. */ + const stableOf = (el: Element): string => + el.id || + el.getAttribute('data-testid') || + el.getAttribute('name') || + el.getAttribute('aria-label') || + '' + + /** Handle stems by role, so a handle says what it is before it says which + * one. Anything unrecognised is `el`. */ + const stemOf = (role: string): string => { + if (role === 'button' || role === 'summary') { + return 'btn' + } + + if (role === 'a' || role === 'link') { + return 'lnk' + } + + if (role === 'input:search' || role === 'searchbox') { + return 'srch' + } + + if (role === 'input:checkbox' || role === 'checkbox') { + return 'chk' + } + + if (role === 'input:radio' || role === 'radio') { + return 'rdo' + } + + if (role === 'select' || role === 'combobox') { + return 'sel' + } + + if (role === 'textarea') { + return 'txt' + } + + if (role === 'switch') { + return 'sw' + } + + if (role === 'tab' || role === 'menuitem' || role === 'option') { + return role === 'menuitem' ? 'mi' : role === 'option' ? 'opt' : 'tab' + } + + // Every `input:*` that isn't one of the special cases above, plus the ARIA + // textbox. A date picker and an email field are both places text goes. + return role.indexOf('input') === 0 || role === 'textbox' ? 'inp' : 'el' + } + + /** Lowercase, hyphenated, and short enough to read at a glance. */ + const slug = (name: string): string => { + let out = '' + let dash = false + + for (let i = 0; i < name.length && out.length < 24; i++) { + const ch = name[i] + + if (/[a-zA-Z0-9]/.test(ch)) { + out += ch.toLowerCase() + dash = false + } else if (!dash && out) { + out += '-' + dash = true + } + } + + return out.replace(/-+$/, '') + } + + /** Where the element sits, coarsely: the nearest landmark plus its position + * among same-role elements inside it. Deliberately NOT the CSS selector + * below — a wrapper div appearing anywhere in the chain changes that string + * completely, which is exactly the churn a re-bind has to see through. */ + const anchorOf = (el: Element): string => { + const near = el.closest( + 'main,nav,header,footer,aside,[role="main"],[role="navigation"],[role="banner"],' + + '[role="contentinfo"],[role="complementary"],[role="search"],form[aria-label],section[aria-label]' + ) + + if (!near) { + return 'root' + } + + const named = near.getAttribute('aria-label') || '' + + return near.tagName.toLowerCase() + (named ? '#' + slug(named) : '') + } + + /** The element's own selector, when the page gives it one worth having. + * + * Deliberately identity-only. This used to fall back to a chain of up to + * eight `:nth-child` rungs, and on a real app shell that column was 74% of + * the entire inventory — the single biggest thing the agent was paying for. + * It bought nothing: nothing downstream reads it, a positional chain is + * wrong the moment a sibling appears, and re-finding a node is what the + * durable ref now does properly. An `#id` is short, stable, and the one + * case where naming the node is genuinely useful to the model. */ + const selectorFor = (el: Element): string => { + if (el.id) { + return '#' + cssEscape(el.id) + } + + const testId = el.getAttribute('data-testid') + + return testId ? '[data-testid="' + cssEscape(testId) + '"]' : '' + } + + const valueOf = (el: Element): string => { + const control = el as HTMLInputElement + + // Tickable controls answer with their state, and are asked FIRST: a + // checkbox's `.value` is the string "on" unless the page sets one, so + // reading the value first made a ticked box and an empty one identical to + // the agent — the one thing about a checkbox worth reporting. Gated on the + // input's TYPE, not on `checked` being defined, which it is (as false) on + // every input including text fields. + const kind = (control.type || '').toLowerCase() + + if (kind === 'checkbox' || kind === 'radio') { + return control.checked ? 'checked' : 'unchecked' + } + + // The same control built out of a div, where the state lives in ARIA. + const role = el.getAttribute('role') || '' + + if (role === 'checkbox' || role === 'radio' || role === 'switch') { + return el.getAttribute('aria-checked') === 'true' ? 'checked' : 'unchecked' + } + + if (typeof control.value === 'string' && control.value) { + return clamp(control.value, 60) + } + + return '' + } + + return { anchorOf, clamp, cssEscape, labelOf, selectorFor, slug, stableOf, stemOf, valueOf } +} diff --git a/apps/desktop/src/lib/preview-act/types.ts b/apps/desktop/src/lib/preview-act/types.ts new file mode 100644 index 0000000000..cbbc59f1ec --- /dev/null +++ b/apps/desktop/src/lib/preview-act/types.ts @@ -0,0 +1,157 @@ +/** + * PREVIEW ACT TYPES — the shapes the engine, the overlay, and the Python tool + * all agree on. + * + * Declarations only. They erase at compile time, which is what lets the engine + * modules import them while still stringifying into the guest page as plain JS + * (see `naming.ts` for the self-containment contract they have to satisfy). + * + * Re-exported from `act-in-page.ts`, so every existing import site keeps + * working and nothing outside this directory needs to know they moved. + */ + +/** One interactable node, as the agent sees it. */ +export interface PreviewElement { + /** Present and true only when the control is non-interactive right now. */ + disabled?: boolean + /** Human-readable label (aria-label, text, placeholder, value …). */ + label: string + /** Durable handle for as long as this page is open: 'btn-sign-in'. Legible on + * purpose — see the ref-minting note in `actInPage`. */ + ref: string + /** Explicit ARIA role, else the tag name. */ + role: string + /** The element's `#id` or `[data-testid]`, when it has one. Absent otherwise + * — address the element by its `ref`. */ + selector?: string + /** Current value of a form control, truncated. */ + value?: string +} + +/** An element that is still itself but no longer reads the same. + * + * Only the fields that actually moved are present. Role and selector are + * absent by construction rather than by omission: a change in either would + * mean this is a different element, which the re-bind ladder would have + * refused to match in the first place. */ +export interface PreviewElementChange { + /** Present only when the control's availability flipped. */ + disabled?: boolean + label?: string + ref: string + value?: string +} + +/** What changed on the page since the last look. Sent instead of the whole + * inventory once the agent has a baseline for the page — see `survey`. */ +export interface PreviewActDelta { + /** Elements seen for the first time, in full. */ + added?: PreviewElement[] + /** Same handle, new label/value/disabled state — and nothing else. */ + changed?: PreviewElementChange[] + /** Handles that are gone from the page. */ + removed?: string[] + /** Handles whose element was destroyed and recreated by a re-render. The + * handle still works; nothing about them needs re-reading. */ + rebound?: string[] + /** How many handles were on the page and untouched. */ + same?: number +} + +/** A normalized action. `kind` is the verb; the rest is per-verb payload. */ +export interface PreviewActAction { + /** scroll distance in px. Defaults to ~90% of the viewport height. */ + amount?: number + key?: string + /** `pin`/`unpin`/`hold` never reach the engine — they resolve their targets + * through `locate`/`elements` and then talk to the overlay — but they arrive + * on the same wire. */ + kind: + | 'click' + | 'elements' + | 'hold' + | 'hover' + | 'locate' + | 'pin' + | 'press' + | 'scroll' + | 'strobe' + | 'type' + | 'unpin' + /** locate: also give the target keyboard focus, for a key press that must not + * be preceded by a click (which would activate the control instead). */ + focus?: boolean + /** elements: answer with the whole inventory rather than a delta. */ + full?: boolean + /** Cap on the returned inventory. */ + max?: number + ref?: string + selector?: string + /** type: press Enter (and submit the owning form) after entering text. */ + submit?: boolean + text?: string + to?: 'bottom' | 'top' +} + +export interface PreviewActResult { + /** What the action landed on, for the agent's own log. */ + acted?: string + /** What moved since the last look. Present INSTEAD of `elements` once the + * agent holds a baseline for this page. */ + delta?: PreviewActDelta + /** The full inventory. Sent on the first look at a page, and again whenever + * the page changed too much for a delta to be the cheaper answer. */ + elements?: PreviewElement[] + error?: string + note?: string + /** Viewport centre of a located target, for aiming real pointer input at it. */ + point?: { x: number; y: number } + success: boolean + title?: string + /** locate: whether the target actually takes typed text. */ + typable?: boolean + /** Live document URL after the action — a change means it navigated. */ + url?: string +} + +/** One element the agent has a handle on, remembered across actions. */ +export interface PreviewActBinding { + el: Element + /** What it read as last time. Kept field by field rather than as one hash so + * a change can be reported as only the part that moved. */ + label: string + /** The accessible name at mint time, for re-finding this element after a + * re-render destroys and recreates its node. */ + name: string + /** Whether the control was unavailable last time. */ + off: boolean + /** Nearest-landmark path plus position among same-role siblings. */ + path: string + ref: string + role: string + /** `id` / `name` / `data-testid` / `aria-label`, if the page provides one. + * The strongest re-bind signal there is, and the only one a rewrite of the + * surrounding markup cannot disturb. */ + stable: string + value: string +} + +/** Where the surface keeps what it knows between actions (a window global in + * the preview page), so a handle still means something on the next call. */ +export interface PreviewActHolder { + /** Target of the action in flight, for the watch overlay to draw onto. */ + aimed?: Element | null + /** Every handle minted on this page, live or not yet retired. */ + book?: PreviewActBinding[] + /** Next disambiguating suffix per ref stem, so two "Edit" buttons become + * `btn-edit` and `btn-edit-1`. Never rewound: a retired handle's name is + * not handed to a different element later in the same page. */ + coined?: Record + /** The on-screen subset, for the overlay to outline. Diverges from `nodes` in + * both directions: it drops what is below the fold, and it is not capped at + * the inventory's size. */ + field?: Element[] + nodes?: Element[] + /** URL the snapshot was taken on; a navigation retires every handle. */ + url?: string +} diff --git a/apps/desktop/src/lib/preview-act/visibility.ts b/apps/desktop/src/lib/preview-act/visibility.ts new file mode 100644 index 0000000000..d1a3ffe697 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/visibility.ts @@ -0,0 +1,148 @@ +/** + * VISIBILITY — is this element actually there, and is it on screen. + * + * Two different questions with two different answers. `visible` decides whether + * an element belongs in the inventory at all: painted, not screen-reader-only, + * not buried under something else. `onScreen` decides whether the overlay + * should draw a box around it, which is a narrower question — the inventory + * includes things below the fold the agent will scroll to, and the overlay + * does not. + * + * Most of what is here exists because of a specific class of thing that looks + * interactive and is not: the sr-only recipe in both its spellings, a control + * inside an `aria-hidden` subtree, a 1px box parked at the document origin, an + * element under a cookie wall. Each one of those, admitted, is an action the + * agent takes that visibly does nothing. + * + * SELF-CONTAINMENT: see `naming.ts`. Same contract, same reason for the + * factory shape. + */ + +export interface VisibilityKit { + /** In the viewport right now, and not a page-sized wrapper. */ + onScreen(el: Element): boolean + /** Painted at all, ancestors included. */ + shown(el: Element): boolean + /** Painted, reachable, big enough to aim at, and not buried. */ + visible(el: Element): boolean +} + +/** Build the visibility tests against one document and its view. */ +export function visibilityKit(doc: Document, win: null | Window): VisibilityKit { + /** On screen right now, and worth drawing a box around. The field is strictly + * viewport-bound: a mark past the fold is one nobody can see, it spends the + * budget that should have gone to what IS on screen, and the ones parked far + * off to the side were landing as stray labels in the corner. */ + const onScreen = (el: Element): boolean => { + if (!win) { + return false + } + + const box = el.getBoundingClientRect() + + if (box.right <= 0 || box.bottom <= 0 || box.left >= win.innerWidth || box.top >= win.innerHeight) { + return false + } + + // Page-sized wrappers. Outlining one draws a rectangle around the whole + // screen, which says nothing and frames everything inside it as if it + // mattered. Full-width banners are fine — it takes both dimensions. + return !(box.width >= win.innerWidth * 0.95 && box.height >= win.innerHeight * 0.9) + } + + /** Painted at all, ancestors included. */ + const shown = (el: Element): boolean => { + // Folds in display/visibility/opacity/content-visibility inherited from an + // ANCESTOR, which this element's own computed style does not report: inside + // a parent at opacity 0, every child still reads back opacity 1. + const seen = (el as Element & { checkVisibility?: (opts: object) => boolean }).checkVisibility + + if (typeof seen === 'function' && !seen.call(el, { checkOpacity: true, checkVisibilityCSS: true })) { + return false + } + + const style = win && win.getComputedStyle ? win.getComputedStyle(el) : null + + if (style) { + if (style.display === 'none' || style.visibility === 'hidden' || style.opacity === '0') { + return false + } + + // The screen-reader-only recipe, both spellings: the deprecated `clip` and + // the modern `clip-path: inset(50%)`. Neither exists for any purpose other + // than hiding something visually while keeping it focusable. + if ((style.clip && style.clip !== 'auto') || (style.clipPath || '').indexOf('inset(50%') !== -1) { + return false + } + } + + return true + } + + const visible = (el: Element): boolean => { + if (!shown(el)) { + return false + } + + // Things the page itself declares are not for interacting with, neither of + // which shows up in a computed style or a bounding box: an aria-hidden + // subtree is invisible to every other assistive client, and `inert` cannot + // be clicked at all. Disabled controls are deliberately NOT filtered here — + // they stay in the inventory so the agent is told a button is disabled + // rather than hunting for one that appears not to exist. + if (el.closest('[aria-hidden="true"], [inert]')) { + return false + } + + const rect = el.getBoundingClientRect() + + // Below the fold is fine — we scroll to it. Off to the LEFT or ABOVE is the + // other half of the sr-only trick (`left: -9999px`, `top: -9999px`) and + // scrolling never brings those back. + if (rect.right <= 0 || rect.bottom <= 0) { + return false + } + + // Bigger than the 1px box sr-only collapses to. Nothing a person can aim at + // is 2px across, and admitting those is what put the cursor in the corner: + // they cluster at the document origin, so they sort to the FRONT of the + // inventory and the agent reaches for one as the first thing on the page. + if (rect.width < 3 || rect.height < 3) { + return false + } + + // Occlusion, and the last filter for a reason: everything above is cheap + // and this one forces layout. If the middle of the element is on screen, + // whatever the browser reports at that point has to BE this element — its + // own descendant (an icon inside a link) or the box it paints inside are + // fine, anything else means it is buried under a sticky header, a cookie + // wall, or a transparent layer the page put on top. Those are exactly the + // targets the agent aims at and then misses. Elements below the fold cannot + // be hit-tested from here, so they are taken on trust and land back in this + // function after the scroll. + const midX = rect.left + rect.width / 2 + const midY = rect.top + rect.height / 2 + + const under = (doc as Document & { elementFromPoint?: (x: number, y: number) => Element | null }) + .elementFromPoint + + if ( + typeof under === 'function' && + win && + midX >= 0 && + midY >= 0 && + midX < win.innerWidth && + midY < win.innerHeight + ) { + const over = under.call(doc, midX, midY) + + if (!over || !(over === el || el.contains(over) || over.contains(el))) { + return false + } + } + + return true + } + + return { onScreen, shown, visible } +} diff --git a/apps/desktop/src/lib/preview-act/watch-in-page.test.ts b/apps/desktop/src/lib/preview-act/watch-in-page.test.ts new file mode 100644 index 0000000000..b51c62b7f0 --- /dev/null +++ b/apps/desktop/src/lib/preview-act/watch-in-page.test.ts @@ -0,0 +1,363 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { type WatchHolder, watchInPage } from './watch-in-page' + +/** jsdom has no layout engine, so every rect is zero and nothing would be + * positioned. Give elements a plausible box the way the act-engine tests do. */ +beforeEach(() => { + // Never run the callback: the tracking loop re-arms itself every frame, so a + // stub that invokes synchronously recurses until the stack gives out. The + // first pass happens on the direct call, which is the one worth asserting on. + vi.stubGlobal('requestAnimationFrame', () => 1) + vi.stubGlobal('cancelAnimationFrame', () => {}) + + Element.prototype.getBoundingClientRect = function () { + return { bottom: 140, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 100, width: 200, x: 100, y: 100 } + } + + document.body.replaceChildren() + // The host hangs off documentElement, so clearing body does not reach it. + document.querySelector('hermes-watch')?.remove() + delete (window as unknown as Record).__hermesWatch +}) + +/** The overlay lives in a closed shadow root, so a test can only see the host. */ +const host = () => document.querySelector('hermes-watch') + +/** …except through the parts the engine parks on the window for its own reuse, + * which is the only way to inspect what the overlay actually drew. */ +const drawn = () => + (window as unknown as { __hermesWatch: { parts: Record & { shadow: ShadowRoot } } }) + .__hermesWatch.parts + +/** Everything the overlay drew for this action. The cursor is the only fixed + * layer — every box and pin is a mark that comes and goes. */ +const marks = () => [...drawn().shadow.children].filter(child => child !== drawn().pointer) + +describe('watchInPage', () => { + it('mounts one host on documentElement and reuses it across stages', () => { + const target = document.createElement('button') + document.body.append(target) + + const holder: WatchHolder = { aimed: target } + + watchInPage(document, holder, 'aim') + watchInPage(document, holder, 'strike') + + expect(document.querySelectorAll('hermes-watch')).toHaveLength(1) + expect(host()?.parentElement).toBe(document.documentElement) + }) + + it('stays out of the page, out of the way, and out of the a11y tree', () => { + const target = document.createElement('button') + document.body.append(target) + + watchInPage(document, { aimed: target }, 'aim') + + const node = host() as HTMLElement + + // pointer-events keeps real clicks reaching the page; aria-hidden keeps a + // screen reader from announcing decoration. + expect(node.style.pointerEvents).toBe('none') + expect(node.getAttribute('aria-hidden')).toBe('true') + // The closed shadow root is what keeps the overlay out of the act engine's + // own querySelectorAll inventory — there is nothing to exclude by hand. + expect(node.shadowRoot).toBeNull() + }) + + it('does nothing when there is no target to draw on', () => { + watchInPage(document, {}, 'aim') + watchInPage(document, { aimed: document.createElement('div') }, 'strike') + + expect(host()).toBeNull() + }) + + // The cursor is placed, never interpolated. Beyond reading as a machine, a + // transition here was an outright bug on the first paint: an unset transform + // IS the origin, so the browser flew the cursor in from the top-left corner — + // and every navigation is a fresh document, so a fresh first paint. + it('cuts the marks straight to the target instead of travelling to it', () => { + const target = document.createElement('button') + document.body.append(target) + + watchInPage(document, { aimed: target }, 'aim', 'Clicking Save') + + const { pointer } = drawn() + + expect(pointer.style.transition).toBeFalsy() + expect((marks()[0] as HTMLElement).style.transition).toBeFalsy() + // Mocked rect is 200×40 at (100, 100), so the cursor belongs at its centre. + expect(pointer.style.transform).toBe('translate3d(200px,120px,0)') + }) + + // The inventory pass is the one moment the agent is reading the whole page + // rather than aiming at one thing, so the overlay has to be able to hold many + // marks at once — not just the single ring every other stage draws. + it('sweep lights up every catalogued element and holds them well past the sweep', () => { + vi.useFakeTimers() + + const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')] + document.body.append(...found) + + watchInPage(document, { nodes: found }, 'sweep') + expect(marks()).toHaveLength(found.length) + + // The point of the dwell: the grid is still up long after the pass that drew + // it finished, so there is something to actually look at. jsdom has no Web + // Animations, so this exercises the timed fallback, which holds for the same + // beat as the animated path. + vi.advanceTimersByTime(600) + expect(marks()).toHaveLength(found.length) + + // …and then drains scattered rather than blinking off as one block, which is + // the whole reason each box carries its own offset. Stepping through the + // exit must therefore see the count come down in more than one tick. + const drain = new Set() + + for (let t = 0; t < 7_000; t += 100) { + vi.advanceTimersByTime(100) + drain.add(marks().length) + } + + expect(drain.size).toBeGreaterThan(2) + expect(marks()).toHaveLength(0) + + vi.useRealTimers() + }) + + // jsdom cannot run a Web Animation, but it can be handed one and asked what + // it was given — and this particular option is worth pinning, because getting + // it wrong is invisible rather than broken. `easing` in the options bag is the + // ITERATION easing: it rewrites progress before any keyframe is consulted, so + // a step function there holds progress at 0 for the whole duration. Every box + // was created, animated and removed without painting a single frame. + it('does not step the sweep’s iteration easing, which would stop it painting', () => { + const timings: KeyframeAnimationOptions[] = [] + + Element.prototype.animate = vi.fn(function (this: Element, _frames, options) { + timings.push(options as KeyframeAnimationOptions) + + return { cancel: () => {}, finish: () => {} } as unknown as Animation + }) as unknown as Element['animate'] + + const found = [document.createElement('button'), document.createElement('a')] + document.body.append(...found) + + watchInPage(document, { field: found }, 'sweep') + + // The cursor's idle sway is permanent and starts with the overlay, so it is + // not one of the sweep's own animations. + const sweep = timings.filter(timing => timing.iterations !== Infinity) + + expect(sweep).toHaveLength(found.length) + + for (const timing of sweep) { + expect(String(timing.easing)).not.toMatch(/steps/) + } + }) + + it('names each swept box after the element under it, not its ref', () => { + const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')] + + found[0].id = 'buy' + found[1].className = 'nav wide' + found[2].setAttribute('type', 'search') + document.body.append(...found) + + // Every drawable box gets named, addressable or not — a ref number only ever + // meant something for the subset that had one, and said nothing about what + // the element actually is. + watchInPage(document, { field: found, nodes: found.slice(0, 2) }, 'sweep') + + const { shadow } = drawn() + const cells = [...shadow.children].slice(-found.length) + + expect(cells.map(cell => cell.textContent)).toEqual(['button#buy', 'a.nav', 'input[search]']) + }) + + // The case that made positions necessary: a list of links with no id, no + // class and no type. Naming them by tag alone gives a whole page of boxes all + // reading "a", which is what the ref numbers were replaced for. + it('falls back to position so anonymous siblings are still told apart', () => { + const row = document.createElement('td') + + row.className = 'subtext' + row.innerHTML = '' + document.body.append(row) + + const found = [...row.children] + + watchInPage(document, { field: found }, 'sweep') + + const { shadow } = drawn() + const cells = [...shadow.children].slice(-found.length) + + expect(cells.map(cell => cell.textContent)).toEqual(['a[1]', 'a[2]', 'a[3]']) + }) + + it('keeps a name short enough to sit on the box it names', () => { + const found = [document.createElement('div')] + + found[0].id = 'a-genuinely-enormous-generated-identifier' + document.body.append(...found) + + watchInPage(document, { field: found }, 'sweep') + + const name = marks()[0].textContent as string + + expect(name.length).toBeLessThanOrEqual(16) + expect(name.startsWith('div#a-')).toBe(true) + expect(name.endsWith('…')).toBe(true) + }) + + // A field past a couple of dozen boxes is a texture, not a list. Naming every + // one buries the outlines — which ARE the readout at that size — under a wall + // of tags that collide with each other and overhang their own rectangles. + it('stops naming the boxes once the field is too dense to read', () => { + const few = Array.from({ length: 4 }, () => document.createElement('button')) + document.body.append(...few) + + watchInPage(document, { field: few }, 'sweep') + expect(marks().every(mark => mark.textContent)).toBe(true) + + document.querySelector('hermes-watch')?.remove() + delete (window as unknown as Record).__hermesWatch + + const many = Array.from({ length: 40 }, () => document.createElement('button')) + document.body.append(...many) + + watchInPage(document, { field: many }, 'sweep') + expect(marks().some(mark => mark.textContent)).toBe(false) + }) + + // Every other mark on the page wears its label on the top edge. A box against + // the top of the viewport has nowhere to put one, and a tab drawn off-screen + // is a box that looks unlabelled next to boxes that aren't. + it('flips the label under the box when there is no headroom above it', () => { + const roomy = document.createElement('button') + const flush = document.createElement('button') + const huge = document.createElement('button') + + flush.getBoundingClientRect = () => + ({ bottom: 40, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect + // Taller than the viewport: no room above it and none below it either, so + // the tab has nowhere left to go but inside. + huge.getBoundingClientRect = () => + ({ bottom: 900, height: 900, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect + document.body.append(roomy, flush, huge) + + watchInPage(document, { field: [roomy, flush, huge] }, 'sweep') + + const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement) + + expect(tags[0].style.bottom).toBe('100%') + expect(tags[1].style.top).toBe('100%') + expect(tags[2].style.top).toBe('0px') + }) + + // A label is routinely wider than the box it names, so it is the boxes near the + // right edge that push their tab off screen. Whichever end it lands on, the + // tab's edge sits exactly on the outline's — the two are the same colour and + // meet at a corner, so a pixel of overhang reads as a misprint. + it('hangs the label off the right edge when a left-anchored one would run off', () => { + Object.defineProperty(HTMLElement.prototype, 'offsetWidth', { configurable: true, get: () => 80 }) + + const inboard = document.createElement('button') + const edge = document.createElement('button') + + edge.getBoundingClientRect = () => + ({ bottom: 140, height: 40, left: 990, right: 1020, toJSON: () => ({}), top: 100, width: 30, x: 990, y: 100 }) as DOMRect + document.body.append(inboard, edge) + + watchInPage(document, { field: [inboard, edge] }, 'sweep') + + const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement) + + expect(tags[0].style.left).toBe('0px') + expect(tags[0].style.right).toBe('') + expect(tags[1].style.right).toBe('0px') + expect(tags[1].style.left).toBe('') + }) + + it('sweep draws nothing when the inventory came back empty', () => { + watchInPage(document, { nodes: [] }, 'sweep') + + expect(host()).toBeNull() + }) + + // The field is swept after every action, and a sweep retires the lock. The + // cursor must not go with it — a hand that blinks out a fraction of a second + // after every single step reads as the agent having wandered off. + it('keeps the cursor up when a sweep retires the lock', () => { + const target = document.createElement('button') + const found = [document.createElement('a')] + document.body.append(target, ...found) + + watchInPage(document, { aimed: target }, 'aim') + + const { pointer } = drawn() + const lock = marks()[0] + + expect(pointer.style.opacity).toBe('1') + + watchInPage(document, { field: found }, 'sweep') + + expect(lock.isConnected).toBe(false) + expect(pointer.style.opacity).toBe('1') + }) + + // The chrome is memoized for the life of the page, so a style change to it + // would otherwise be injected, run, and restyle nothing — the layers it + // describes were built by the version before. Reads as "HMR isn't picking my + // CSS up" in dev, and as a long-lived tab stuck on an old build in production. + it('rebuilds its chrome when the injected source changes underneath it', () => { + const target = document.createElement('button') + document.body.append(target) + + const stamp = (value: number) => { + ;(window as unknown as Record).__hermesWatchTag = value + } + + stamp(1) + watchInPage(document, { aimed: target }, 'aim') + + const first = drawn().host + + stamp(1) + watchInPage(document, { aimed: target }, 'aim') + + expect(drawn().host).toBe(first) + + stamp(2) + watchInPage(document, { aimed: target }, 'aim') + + expect(drawn().host).not.toBe(first) + // …and the old one goes with it, rather than stacking a second overlay. + expect(document.querySelectorAll('hermes-watch')).toHaveLength(1) + }) + + it('clear releases the target so the tracking loop can stop', () => { + const target = document.createElement('button') + document.body.append(target) + + const holder: WatchHolder = { aimed: target } + + watchInPage(document, holder, 'aim') + watchInPage(document, holder, 'clear') + + expect(holder.aimed).toBeNull() + }) + + // Same contract as the act engine: the pane injects `watchInPage.toString()` + // into the guest page, where module scope does not exist. One free identifier + // is a ReferenceError that Electron reports only as "Script failed to execute". + it('runs after being stringified and eval’d with no module scope', () => { + const target = document.createElement('button') + document.body.append(target) + + const injected = new Function('return (' + watchInPage.toString() + ')')() as typeof watchInPage + + expect(() => injected(document, { aimed: target }, 'aim', 'Clicking Save')).not.toThrow() + expect(host()).not.toBeNull() + }) +}) diff --git a/apps/desktop/src/lib/preview-act/watch-in-page.ts b/apps/desktop/src/lib/preview-act/watch-in-page.ts new file mode 100644 index 0000000000..0395548d0d --- /dev/null +++ b/apps/desktop/src/lib/preview-act/watch-in-page.ts @@ -0,0 +1,1063 @@ +/** + * WATCH OVERLAY — paints the agent's actions onto the live preview page so a + * person can follow along: the field of things it can reach, a box round the + * one it is about to touch, and a cursor that goes there. + * + * There are exactly two things on screen, and everything the agent does is said + * with one of them: + * + * - THE CURSOR — one arrow, built once, never removed. It is the + * collaborative-editing kind, because that is exactly what this is: a second + * party moving around a document you are also looking at. Arrow geometry is + * Liveblocks' (Apache-2.0). Plain black or white against the guest's own + * colour scheme rather than a brand colour: it stands for a hand, and a hand + * is not a feature of this app. + * - THE MARK — a rectangle glued to an element. A candidate, a lock, a hit + * and a pin are all the same object with a different row in `kinds`: how + * heavy the rule is, how long it lives, how it arrives. Marks are a design + * tool's selection blue, deliberately NOT the + * app's accent — this is chrome drawn on top of someone else's page, and + * every design tool has trained people to read that blue as "a tool has + * selected this", where a branded colour would read as part of the page. + * + * Every mark is two elements: a shell the tracker positions, and a skin that + * carries the border and any animation. They are split because both want + * `transform`, and a mark that animates the property the tracker rewrites every + * frame either stops moving or stops animating. + * + * Injected as source (`watchInPage.toString()`) exactly like the act engine, so + * the same rule holds: no imports, no closure references, no module-scope + * constants — everything the body needs is declared inside it. + * + * Three constraints shape the implementation, each one learned from how + * Playwright, Stagehand and browser-use solve the same problem: + * + * - A page can set `z-index: 2147483647` too, and the browser's top layer + * (``, `popover`, fullscreen) outranks every z-index there is. So + * the host joins the top layer via `popover="manual"` instead of trying to + * outbid the page, and keeps the z-index only as a fallback. + * - A strict CSP rejects `style-src` at CSS PARSE time, which kills + * `cssText`, `setAttribute('style', …)` and any `