Merge pull request #90197 from NousResearch/bb/preview-act

The agent can use the in-app browser, not just look at it
This commit is contained in:
brooklyn!
2026-08-20 05:32:05 -05:00
committed by GitHub
38 changed files with 5746 additions and 49 deletions
+2
View File
@@ -546,6 +546,7 @@ def init_agent(
clarify_callback: callable = None,
read_terminal_callback: callable = None,
read_preview_callback: callable = None,
drive_preview_callback: callable = None,
read_window_below_callback: callable = None,
setup_mcp_callback: callable = None,
tour_callback: callable = None,
@@ -843,6 +844,7 @@ def init_agent(
agent.clarify_callback = clarify_callback
agent.read_terminal_callback = read_terminal_callback
agent.read_preview_callback = read_preview_callback
agent.drive_preview_callback = drive_preview_callback
agent.read_window_below_callback = read_window_below_callback
agent.setup_mcp_callback = setup_mcp_callback
agent.tour_callback = tour_callback
+32 -1
View File
@@ -99,7 +99,7 @@ def _ra():
AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset(
{"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"}
{"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"}
)
@@ -3234,6 +3234,37 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
),
next_args,
)
elif function_name == "drive_preview":
def _execute(next_args: dict) -> Any:
from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool
return _finish_agent_tool(
_drive_preview_tool(
action=next_args.get("action", ""),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
text=next_args.get("text"),
key=next_args.get("key"),
submit=next_args.get("submit"),
amount=next_args.get("amount"),
to=next_args.get("to"),
limit=next_args.get("max"),
callback=getattr(agent, "drive_preview_callback", None),
),
next_args,
)
elif function_name == "annotate_preview":
def _execute(next_args: dict) -> Any:
from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool
return _finish_agent_tool(
_annotate_preview_tool(
action=next_args.get("action", "add"),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
label=next_args.get("label"),
callback=getattr(agent, "drive_preview_callback", None),
),
next_args,
)
elif function_name == "read_window_below":
def _execute(next_args: dict) -> Any:
from tools.read_window_tool import read_window_below_tool as _read_window_below_tool
+51
View File
@@ -2202,6 +2202,57 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('read_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "drive_preview":
def _execute(next_args: dict) -> Any:
from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool
return _drive_preview_tool(
action=next_args.get("action", ""),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
text=next_args.get("text"),
key=next_args.get("key"),
submit=next_args.get("submit"),
amount=next_args.get("amount"),
to=next_args.get("to"),
limit=next_args.get("max"),
callback=getattr(agent, "drive_preview_callback", None),
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
agent,
function_name=function_name,
function_args=function_args,
effective_task_id=effective_task_id,
tool_call_id=getattr(tool_call, "id", "") or "",
execute=_execute,
scope_block=_ts_scope_block,
display_index=i,
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('drive_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "annotate_preview":
def _execute(next_args: dict) -> Any:
from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool
return _annotate_preview_tool(
action=next_args.get("action", "add"),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
label=next_args.get("label"),
callback=getattr(agent, "drive_preview_callback", None),
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
agent,
function_name=function_name,
function_args=function_args,
effective_task_id=effective_task_id,
tool_call_id=getattr(tool_call, "id", "") or "",
execute=_execute,
scope_block=_ts_scope_block,
display_index=i,
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('annotate_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "read_window_below":
def _execute(next_args: dict) -> Any:
from tools.read_window_tool import read_window_below_tool as _read_window_below_tool
@@ -0,0 +1,314 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { $rightRailActiveTabId } from '@/store/layout'
import { closeRightRail, openPreview, type PreviewTarget } from '@/store/preview'
import { actOnActivePreview } from './preview-act'
import { registerPreviewInput } from './preview-input'
import { registerPreviewNav } from './preview-nav'
import { registerPreviewScriptRunner } from './preview-script-runner'
function urlTarget(url: string): PreviewTarget {
return { kind: 'url', label: 'Browser', source: url, url }
}
describe('actOnActivePreview (drive_preview tool)', () => {
// URL targets share the singleton Browser tab id, so anything a test
// registers would answer the next one.
let cleanups: Array<() => void> = []
const openBrowserTab = () => {
openPreview(urlTarget('https://example.com'), 'tool-result')
return $rightRailActiveTabId.get()!
}
const withRunner = (runner: (code: string) => Promise<unknown>) =>
cleanups.push(registerPreviewScriptRunner(openBrowserTab(), runner))
beforeEach(() => {
vi.useRealTimers()
for (const cleanup of cleanups) {
cleanup()
}
cleanups = []
closeRightRail()
window.localStorage.clear()
})
it('tells the agent to open a page when no live pane is behind the tab', async () => {
const result = await actOnActivePreview({ kind: 'elements' })
expect(result.success).toBe(false)
expect(result.error).toContain('open_preview')
})
it('injects the engine and returns the page’s answer', async () => {
let injected = ''
withRunner(async code => {
injected = code
return JSON.stringify({ acted: 'clicked button "Save"', success: true })
})
const result = await actOnActivePreview({ kind: 'click', ref: '@e1' })
expect(result).toMatchObject({ acted: 'clicked button "Save"', success: true })
// Self-contained payload: the engine source and the action travel together,
// and the holder keeps refs alive across calls on the same page.
expect(injected).toContain('__hermesActHolder')
expect(injected).toContain('"ref":"@e1"')
})
it('re-inventories after a mutating action so the next ref is current', async () => {
const actions: string[] = []
withRunner(code => {
// Stand in for the guest page: run the script's own settle/rescan shape
// by answering each act() call in order.
actions.push(...(code.match(/"kind":"(\w+)"/g) ?? []))
return Promise.resolve(
JSON.stringify({
acted: 'clicked',
elements: [{ label: 'Log out', ref: '@e1', role: 'button', selector: '#out' }],
success: true,
url: 'https://example.com/app'
})
)
})
const result = await actOnActivePreview({ kind: 'click', ref: '@e1' })
expect(result.elements?.[0].label).toBe('Log out')
expect(result.url).toBe('https://example.com/app')
})
it('does not pay the settle delay for a plain inventory', async () => {
let injected = ''
withRunner(async code => {
injected = code
return JSON.stringify({ elements: [], success: true })
})
await actOnActivePreview({ kind: 'elements' })
expect(injected).toContain('0 <= 0')
})
it('reports a page that answers with nothing', async () => {
withRunner(async () => '')
expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).error).toContain('did not answer')
})
/** A pane that answers the locate trip with a fixed on-screen point, and the
* read-back trip with an empty inventory. Returns the input spy. */
const withDrivenPane = () => {
const tabId = openBrowserTab()
const send = vi.fn()
cleanups.push(
registerPreviewScriptRunner(tabId, async code =>
code.includes('"kind":"locate"')
? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 120, y: 80 }, success: true })
: // `hit` is the page's witness that the real pointerdown arrived.
JSON.stringify({ elements: [], hit: { tag: 'BUTTON', trusted: true }, success: true })
)
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send }))
return send
}
const sentTypes = (send: ReturnType<typeof vi.fn>) => send.mock.calls.map(([event]) => event.type)
it('clicks with real input, walking the pointer to where the page said', async () => {
const send = withDrivenPane()
const result = await actOnActivePreview({ kind: 'click', ref: '@e1' })
const types = sentTypes(send)
// Stepped rather than teleported: a page learns it is hovered from a stream
// of moves, and one jump to the target skips everything in between.
expect(types.filter(type => type === 'mouseMove').length).toBeGreaterThan(1)
expect(types).toContain('mouseDown')
expect(types).toContain('mouseUp')
expect(send.mock.calls.map(([event]) => event).find(event => event.type === 'mouseDown')).toMatchObject({
x: 120,
y: 80
})
expect(result.acted).toBe('clicked button "Save"')
})
it('types by pressing keys, after selecting whatever the field held', async () => {
const send = withDrivenPane()
await actOnActivePreview({ kind: 'type', ref: '@e1', submit: true, text: 'hi' })
const events = send.mock.calls.map(([event]) => event)
const chars = events.filter(event => event.type === 'char').map(event => event.keyCode)
// One click to focus, then select-all by keyboard — NOT the triple-click
// this used to do. A triple-click is a pointer gesture, so it grabs the
// paragraph under the cursor whenever the target turns out not to be a
// field, and the agent was leaving pages with their body text highlighted.
expect(events.filter(event => event.type === 'mouseDown').map(event => event.clickCount)).toEqual([1])
expect(events.filter(event => event.type === 'keyDown' && event.keyCode === 'a')[0]).toMatchObject({
modifiers: ['control', 'meta']
})
// The chord must not send a `char` phase, or select-all types a literal 'a'.
expect(chars).toEqual(['h', 'i', 'Enter'])
})
it('hovers by walking the pointer over and leaving it there', async () => {
const send = withDrivenPane()
const result = await actOnActivePreview({ kind: 'hover', ref: '@e1' })
const types = sentTypes(send)
expect(types).toContain('mouseMove')
// The whole request is "be on it" — a click here would open the dropdown the
// agent was trying to reveal, or worse, activate it.
expect(types).not.toContain('mouseDown')
expect(result.acted).toBe('hovered over button "Save"')
})
// The witness only speaks for verbs that put the button down. Demanding one
// from a key press reported every press as a failure.
it('does not expect a click witness from a verb that never clicks', async () => {
const tabId = openBrowserTab()
cleanups.push(
registerPreviewScriptRunner(tabId, async code =>
code.includes('"kind":"locate"')
? JSON.stringify({ acted: 'looking at textbox "Search"', point: { x: 40, y: 20 }, success: true })
: JSON.stringify({ elements: [], hit: null, success: true })
)
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() }))
for (const action of [
{ key: 'Escape', kind: 'press', ref: '@e1' },
{ kind: 'hover', ref: '@e1' }
]) {
expect(await actOnActivePreview(action)).toMatchObject({ success: true })
}
})
it('scrolls by wheeling for real, not by scripting the page', async () => {
const tabId = openBrowserTab()
const send = vi.fn()
let scripted = false
cleanups.push(
registerPreviewScriptRunner(tabId, async code => {
scripted ||= code.includes('"kind":"scroll"')
return code.includes('scrollHeight')
? JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 4_000, success: true })
: JSON.stringify({ elements: [], success: true })
})
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send }))
await actOnActivePreview({ kind: 'scroll' })
const wheels = send.mock.calls.map(([event]) => event).filter(event => event.type === 'mouseWheel')
// A stream of notches, not one delta: scroll-linked headers and lazy loaders
// only react to the events, so a scripted scrollBy leaves them asleep.
expect(wheels.length).toBeGreaterThan(1)
// Electron's wheel delta is wheelDelta-signed, so scrolling DOWN is negative.
expect(wheels.every(event => event.deltaY < 0)).toBe(true)
expect(scripted).toBe(false)
})
it('says so plainly when the page has nothing to scroll', async () => {
const tabId = openBrowserTab()
cleanups.push(
registerPreviewScriptRunner(tabId, async () =>
JSON.stringify({ page: 700, point: { x: 500, y: 400 }, span: 0, success: true })
)
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() }))
expect(await actOnActivePreview({ kind: 'scroll' })).toMatchObject({ note: expect.stringContaining('nothing to scroll') })
})
it('fails loudly when the pointer input never reaches the page', async () => {
const tabId = openBrowserTab()
cleanups.push(
registerPreviewScriptRunner(tabId, async code =>
code.includes('"kind":"locate"')
? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true })
: JSON.stringify({ elements: [], hit: null, success: true })
)
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() }))
const result = await actOnActivePreview({ kind: 'click', ref: '@e1' })
// Everything else about this action travels on the script channel and would
// report success whether or not a single event landed.
expect(result.success).toBe(false)
expect(result.error).toContain('never reached the page')
})
it('says so when the overlay itself swallowed the click', async () => {
const tabId = openBrowserTab()
cleanups.push(
registerPreviewScriptRunner(tabId, async code =>
code.includes('"kind":"locate"')
? JSON.stringify({ acted: 'looking at button "Save"', point: { x: 12, y: 8 }, success: true })
: JSON.stringify({ elements: [], hit: { tag: 'HERMES-WATCH', trusted: true }, success: true })
)
)
cleanups.push(registerPreviewInput(tabId, { focus: vi.fn(), send: vi.fn() }))
expect((await actOnActivePreview({ kind: 'click', ref: '@e1' })).note).toContain('overlay intercepted')
})
it('falls back to scripted events when the pane exposes no input channel', async () => {
let injected = ''
withRunner(async code => {
injected = code
return JSON.stringify({ acted: 'clicked', success: true })
})
await actOnActivePreview({ kind: 'click', ref: '@e1' })
// The one-trip shape: the engine both acts and re-reads, no locate handshake.
expect(injected).toContain('"kind":"click"')
expect(injected).not.toContain('"kind":"locate"')
})
it('routes history verbs to the pane instead of the guest page', async () => {
const back = vi.fn()
const runner = vi.fn()
const tabId = openBrowserTab()
cleanups.push(registerPreviewNav(tabId, { back, forward: vi.fn(), reload: vi.fn() }))
cleanups.push(registerPreviewScriptRunner(tabId, runner))
const result = await actOnActivePreview({ kind: 'back' })
expect(back).toHaveBeenCalledOnce()
expect(runner).not.toHaveBeenCalled()
expect(result.success).toBe(true)
expect(result.note).toContain('elements')
})
it('reports history verbs with no pane to drive', async () => {
expect((await actOnActivePreview({ kind: 'reload' })).error).toContain('open_preview')
})
})
@@ -0,0 +1,575 @@
/**
* PREVIEW ACT — performs the agent's interactions inside the preview pane's
* guest page, so `drive_preview` drives whatever web app is open in the in-app
* browser.
*
* The guest page is out-of-process; nothing here can touch its DOM directly.
* Two separate channels reach it, and the split matters:
*
* - `executeJavaScript` injects the engine SOURCE (see
* lib/preview-act/act-in-page.ts's self-containment contract) to RESOLVE
* and READ — turn 'btn-sign-in' into a node, measure it, inventory the
* page. It is parked on a window global alongside the book of handles so
* they survive between calls, and it vanishes with the page, so a
* navigation retires them and the engine reports that instead of acting on
* the wrong node.
* - `sendInputEvent` (preview-drive.ts) does the ACTING, as real Chromium
* input. Script can only dispatch synthetic events, which the page can tell
* apart and which never move the browser's own hover or focus target.
*
* Panes with no input channel — a remote HTML preview — fall back to the
* engine's synthetic events, which is worse but still works.
*
* A mutating action settles briefly and returns a fresh inventory, so the
* click → re-read → click loop costs one round trip instead of two.
*
* Dynamic-imported by the gateway event handler so the engine payload stays
* out of the boot path.
*/
import { actEngineSource, type PreviewActAction, type PreviewActResult } from '@/lib/preview-act/act-in-page'
import { watchInPage } from '@/lib/preview-act/watch-in-page'
import { clickAt, glideTo, pointerPlaced, pressKey, selectAll, typeText, wheelBy } from './preview-drive'
import { activePreviewInput, type PreviewInputHandle } from './preview-input'
import { activePreviewNav, type PreviewNavHandle } from './preview-nav'
import { activePreviewScriptRunner, type PreviewScriptRunner } from './preview-script-runner'
/** Verbs the pane owns; a guest page cannot drive its own history. */
const NAV_ACTIONS: readonly (keyof PreviewNavHandle)[] = ['back', 'forward', 'reload']
/** How long a click/type is given to land before the page is re-inventoried.
* Long enough for a framework re-render, short enough not to stall the turn.
* A render that misses this window is not lost — the NEXT action re-reads the
* page anyway, so the cost of being early is a slightly stale inventory, not a
* wrong one. That trade is worth several hundred ms on every single step. */
const SETTLE_MS = 220
/** Cap on one round trip into the guest page. A click that starts a navigation
* tears the document down mid-settle, so the injected promise dies with it and
* `executeJavaScript` never settles — without this the tool eats its whole
* 45s bridge deadline for what was actually a successful click. */
const ACT_TIMEOUT_MS = 8_000
/** Verbs that get the look-then-act treatment: locate the target, walk the
* pointer there, then send real input. Actions with no single target (scroll,
* elements) are absent and skip it. */
const DRIVEN: readonly string[] = ['click', 'hover', 'press', 'type']
/** Verbs that put the mouse button down, and so are the only ones the pointerdown
* witness can speak to. `press` and `hover` never click, so demanding a witness
* from them would report every one of them as a failure. */
const CLICKS: readonly string[] = ['click', 'type']
const NOTHING_OPEN = 'No live page is open in the in-app browser — open one with open_preview first.'
const NAVIGATED = 'The page stopped answering right after — it is probably navigating. Call elements to see where you landed.'
/** A fingerprint of the overlay's source, so the guest page can tell that the
* code it is running has changed underneath it.
*
* It needs one because the overlay memoizes its own chrome: the cursor, the
* lock frame and the scan line are built ONCE per page and then reused, so an
* edit to any of their styles lands in the injected source, gets injected, and
* changes nothing — the layers it would have styled were built by the previous
* version and are still on the page. In dev that reads as "HMR didn't pick up
* my CSS"; in production it is a long-lived tab pinned to whichever build first
* touched it, across an app upgrade.
*
* Content-addressed rather than a length or a version literal: the change that
* exposed this was one hex colour for another, which is byte-for-byte the same
* length, and a hand-maintained version is a step someone forgets. */
const WATCH_TAG = (() => {
const source = watchInPage.toString()
let hash = 5381
for (let i = 0; i < source.length; i++) {
hash = ((hash << 5) + hash + source.charCodeAt(i)) | 0
}
return hash
})()
/** Injected once per page: the engine, the overlay, and the two helpers every
* script below shares. Idempotent — re-running it on a live page is a no-op. */
const preamble = () => ` var w = window;
// Reassigned on every trip rather than cached behind a guard: the source is in
// this payload either way, so the guard saved nothing and pinned a long-lived
// tab to whichever build first touched it. The holder is the exception — it
// carries the aimed element from the locate trip to the act trip.
w.__hermesActHolder = w.__hermesActHolder || {};
w.__hermesAct = ${actEngineSource()};
w.__hermesWatch_fn = (${watchInPage.toString()});
w.__hermesWatchTag = ${WATCH_TAG};
var holder = w.__hermesActHolder;
var act = function (a) { return w.__hermesAct(document, holder, a); };
var watch = function (stage, label) {
try { w.__hermesWatch_fn(document, holder, stage, label); } catch (err) {}
};
var wait = function (ms) { return new Promise(function (resolve) { setTimeout(resolve, ms); }); };
// Sit out a smooth scroll so the pointer is aimed at a target that has stopped
// moving. A scroll that never started fires no scrollend, so only wait on one
// we can actually see happening; capture catches nested scrollers, whose
// scrollend does not bubble.
var restAfterScroll = function () {
return new Promise(function (resolve) {
var sc = document.scrollingElement || document.documentElement;
var x0 = sc ? sc.scrollLeft : 0;
var y0 = sc ? sc.scrollTop : 0;
requestAnimationFrame(function () {
if (!sc || (sc.scrollLeft === x0 && sc.scrollTop === y0)) { return resolve(); }
var done = false;
var finish = function () {
if (done) { return; }
done = true;
document.removeEventListener('scrollend', finish, true);
clearTimeout(timer);
resolve();
};
var timer = setTimeout(finish, 350);
document.addEventListener('scrollend', finish, true);
});
});
};`
/** Bring the target on screen, mark it, and report where it came to rest. */
function buildLocateScript(action: PreviewActAction, focus: boolean): string {
const locate = { focus, kind: 'locate', ref: action.ref, selector: action.selector }
return `(function () {
${preamble()}
var locate = ${JSON.stringify(locate)};
var found = act(locate);
if (!found.success) { return Promise.resolve(JSON.stringify(found)); }
watch('aim');
// Arm a witness for the real input that is about to arrive. Without it a
// click that never reached the page is indistinguishable from one the page
// ignored, and the agent would report success either way.
w.__hermesHit = null;
document.addEventListener('pointerdown', function (e) {
w.__hermesHit = { tag: e.target ? e.target.tagName : '?', trusted: e.isTrusted === true };
}, { capture: true, once: true });
// Measure again once the scroll has stopped: real input is aimed at a fixed
// viewport coordinate, so it has to be where the target ENDS UP.
return restAfterScroll().then(function () {
var settled = act(locate);
var best = settled.success ? settled : found;
var at = best.point;
// Last line of defence before the pointer is sent somewhere real. An element
// that is still outside the viewport after we scrolled to it is hidden, not
// placed, and aiming at it would drive the cursor off into a corner and
// click whatever happens to be under that coordinate.
if (at && (at.x < 0 || at.y < 0 || at.x > window.innerWidth || at.y > window.innerHeight)) {
return JSON.stringify({
error: 'That element is still off-screen after scrolling to it, so it is hidden rather than clickable. Call elements again for what is really on the page.',
success: false
});
}
return JSON.stringify(best);
});
})()`
}
/** Put up a mark that outlives the action that made it. Every other cue on the
* overlay retires on a timer, which is right for narrating a click and no use
* at all for holding a finding on screen while the agent keeps working. */
function buildPinScript(action: PreviewActAction, label: string): string {
const locate = { kind: 'locate', ref: action.ref, selector: action.selector }
return `(function () {
${preamble()}
var found = act(${JSON.stringify(locate)});
if (!found.success) { return JSON.stringify(found); }
watch('pin', ${JSON.stringify(label)});
return JSON.stringify({ acted: 'pinned ' + String(found.acted || 'it').replace(/^looking at /, ''), success: true });
})()`
}
/** Freeze the whole visible field as marks. Needs a fresh inventory first, so
* the overlay has both the field to draw and the refs to number it by. */
function buildHoldScript(): string {
return `(function () {
${preamble()}
var found = act({ kind: 'elements' });
if (!found.success) { return JSON.stringify(found); }
watch('hold');
found.acted = 'held the field';
return JSON.stringify(found);
})()`
}
/** Rattle through the field one box at a time. Needs a fresh inventory for the
* overlay to pick from, and nothing else — the page is never touched. */
function buildStrobeScript(): string {
return `(function () {
${preamble()}
var found = act({ kind: 'elements' });
if (!found.success) { return JSON.stringify(found); }
watch('strobe');
found.acted = 'strobed the field';
return JSON.stringify(found);
})()`
}
/** Take one mark down, or — with nothing to aim at — all of them. */
function buildUnpinScript(action: PreviewActAction): string {
const locate = { kind: 'locate', ref: action.ref, selector: action.selector }
const one = !!(action.ref || action.selector)
return `(function () {
${preamble()}
${
one
? ` var found = act(${JSON.stringify(locate)});
if (!found.success) { return JSON.stringify(found); }
var gone = 'unpinned ' + String(found.acted || 'it').replace(/^looking at /, '');`
: ` holder.aimed = null;
var gone = 'cleared every pin';`
}
watch('unpin');
return JSON.stringify({ acted: gone, success: true });
})()`
}
/** Mark the hit, let the page react, and hand back a fresh inventory. */
function buildFinishScript(settleMs: number): string {
return `(function () {
${preamble()}
watch('strike');
return wait(${settleMs}).then(function () {
// A rescan that throws must still answer — an unresolved promise here costs
// the whole bridge deadline.
try {
var out = act({ kind: 'elements' });
watch('sweep');
out.hit = w.__hermesHit || null;
return JSON.stringify(out);
} catch (err) {
return JSON.stringify({ note: 'The page changed before it could be re-read: ' + err, success: true });
}
});
})()`
}
/** The no-real-input fallback: the engine both acts and reads, in one trip.
* Both reads mark the field — the explicit `elements` call and the re-read
* every other verb does on its way out — so the page is marked after every
* single action rather than only when the agent asks for an inventory outright.
* They mark it differently, though, and the difference is what each one MEANS:
* an outright `elements` is the agent reading the whole page, which is the
* moment worth showing, so it strobes. The re-read after a click is
* housekeeping, and strobing there would put five seconds of noise between
* every step of a task, so it sweeps. The two never collide: an `elements` call
* settles in 0ms and returns before the re-read. */
function buildScriptedScript(action: PreviewActAction, settleMs: number): string {
return `(function () {
${preamble()}
var result = act(${JSON.stringify(action)});
${
action.kind === 'elements'
? "if (result.success) { watch('strobe'); }"
: ''
}
if (!result.success || ${settleMs} <= 0) { return Promise.resolve(JSON.stringify(result)); }
return wait(${settleMs}).then(function () {
try {
var after = act({ kind: 'elements' });
watch('sweep');
// One or the other, never both: a re-read answers with the whole
// inventory only when it is the first look at this page.
result.elements = after.elements;
result.delta = after.delta;
result.url = after.url;
result.title = after.title;
} catch (err) {
result.note = 'The page changed before it could be re-read: ' + err;
}
return JSON.stringify(result);
});
})()`
}
/** The outcome of one round trip into the page. `silent` is its own case on
* purpose: a page that stops answering mid-action is navigating, whereas one
* that answers with nothing is broken, and the agent needs to hear the
* difference. */
type Trip = { error: string; kind: 'failed' } | { kind: 'answered'; result: PreviewActResult } | { kind: 'silent' }
async function runJson(run: PreviewScriptRunner, code: string): Promise<Trip> {
const raw = await Promise.race([
run(code).catch((error: unknown) => new Error(String(error))),
new Promise<undefined>(resolve => setTimeout(resolve, ACT_TIMEOUT_MS))
])
if (raw === undefined) {
return { kind: 'silent' }
}
if (raw instanceof Error) {
return { error: 'The page rejected the action: ' + raw.message, kind: 'failed' }
}
if (typeof raw !== 'string' || !raw) {
return { error: 'The page did not answer the action.', kind: 'failed' }
}
return { kind: 'answered', result: JSON.parse(raw) as PreviewActResult }
}
/** Past tense of the verb the agent asked for, against what it actually hit. */
function describeDone(action: PreviewActAction, target: string): string {
if (action.kind === 'type') {
return 'typed into ' + target + (action.submit ? ' and submitted' : '')
}
if (action.kind === 'press') {
return 'pressed ' + (action.key || '') + ' on ' + target
}
if (action.kind === 'hover') {
return 'hovered over ' + target
}
return 'clicked ' + target
}
/** Look at the target, walk the pointer over, and act on it for real. */
async function driveAction(
run: PreviewScriptRunner,
input: PreviewInputHandle,
action: PreviewActAction
): Promise<PreviewActResult> {
// A key press must not be preceded by a click — that would activate the
// control rather than type into it — so the page hands it focus instead.
const trip = await runJson(run, buildLocateScript(action, action.kind === 'press'))
if (trip.kind === 'failed') {
return { error: trip.error, success: false }
}
if (trip.kind === 'silent') {
return { acted: action.kind, note: NAVIGATED, success: true }
}
const found = trip.result
if (!found.success) {
return found
}
if (!found.point) {
return { error: 'Could not work out where that element is on screen.', success: false }
}
await glideTo(input, found.point)
if (action.kind === 'click') {
await clickAt(input)
} else if (action.kind === 'type') {
if (found.typable === false) {
return {
error: `${String(found.acted || 'That').replace(/^looking at /, '')} is not a text field, so typing into it would only select the text under the pointer. Click it if it opens one, then type into that.`,
success: false
}
}
input.focus()
await clickAt(input)
// Select-all inside the now-focused field, so typing replaces what is there
// the way it would for a person. NOT a triple-click: that is a pointer
// gesture and selects the paragraph under the cursor whenever the target
// turns out not to be a field.
await selectAll(input)
await typeText(input, action.text ?? '')
if (action.submit) {
await pressKey(input, 'Enter')
}
} else if (action.kind === 'press') {
input.focus()
await pressKey(input, action.key || 'Enter')
}
// hover is the glide and nothing else — the pointer is already sitting on the
// target, which is the whole request.
const target = String(found.acted || '').replace(/^looking at /, '')
const after = await runJson(run, buildFinishScript(SETTLE_MS))
const acted = describeDone(action, target)
// The action itself already happened as real input, so a page that will not
// answer the read-back is a page that navigated — never a failed click.
if (after.kind !== 'answered') {
return { acted, note: NAVIGATED, success: true }
}
const { hit, ...result } = after.result as PreviewActResult & { hit?: { tag: string; trusted: boolean } | null }
// The witness the locate trip armed. No record means the input never reached
// the document, which the agent must hear about — every other signal here
// travels on the script channel and would report success regardless.
if (!hit && CLICKS.indexOf(action.kind) !== -1) {
return {
...result,
error: 'The pointer input never reached the page, so nothing was ' + acted.split(' ')[0] + '.',
success: false
}
}
return { ...result, acted, note: hitNote(hit), success: true }
}
/** Flag a click the overlay intercepted, which would otherwise look like a page
* that simply ignored it. */
function hitNote(hit?: { tag: string; trusted: boolean } | null): string | undefined {
return hit && hit.tag === 'HERMES-WATCH'
? 'The action overlay intercepted the click instead of the page.'
: undefined
}
/** How far a screenful is, whether there is anywhere to go, and a spot to wheel
* over — the renderer cannot see the guest viewport to work any of it out. */
function buildScrollAnchorScript(): string {
return `(function () {
${preamble()}
var sc = document.scrollingElement || document.documentElement;
var track = sc.clientHeight || window.innerHeight;
return Promise.resolve(JSON.stringify({
page: Math.round(window.innerHeight * 0.9),
point: { x: Math.round(window.innerWidth / 2), y: Math.round(track / 2) },
span: sc.scrollHeight - track,
success: true
}));
})()`
}
/** Scroll the page the way a hand does — wheel it. The scripted path calls
* `scrollBy`, which moves the page without the page ever seeing an input
* event, so scroll-linked headers, reveal animations and infinite-scroll
* loaders all stay asleep through a scroll that looked like it happened. */
async function driveScroll(
run: PreviewScriptRunner,
input: PreviewInputHandle,
action: PreviewActAction,
): Promise<PreviewActResult> {
const far = action.amount ?? 0
const trip = await runJson(
run,
buildScrollAnchorScript()
)
if (trip.kind === 'failed') {
return { error: trip.error, success: false }
}
if (trip.kind === 'silent') {
return { acted: 'scrolled', note: NAVIGATED, success: true }
}
const anchor = trip.result as PreviewActResult & { page?: number; span?: number }
if (!anchor.span) {
return { ...anchor, acted: 'scrolled the page', note: 'The page has nothing to scroll — it all fits already.' }
}
// A person does not move the mouse to scroll; the wheel turns wherever their
// hand already is. Only send it somewhere if it has never been anywhere.
if (!pointerPlaced() && anchor.point) {
await glideTo(input, anchor.point)
}
await wheelBy(input, action.amount ?? anchor.page ?? 600)
const after = await runJson(run, buildFinishScript(SETTLE_MS))
if (after.kind !== 'answered') {
return { acted: 'scrolled the page', note: NAVIGATED, success: true }
}
return { ...after.result, acted: 'scrolled the page', success: true }
}
/** Run one action against the ACTIVE preview tab's page. `kind` is a bare
* string: the verb arrives off the wire, and the history ones never reach
* the in-page engine. */
export async function actOnActivePreview(
action: Omit<PreviewActAction, 'kind'> & { kind: string }
): Promise<PreviewActResult> {
const nav = NAV_ACTIONS.find(verb => verb === action.kind)
if (nav) {
const handle = activePreviewNav()
if (!handle) {
return { error: NOTHING_OPEN, success: false }
}
handle[nav]()
// Navigation is fire-and-forget through the webview; the new document has
// its own refs, so the agent has to re-inventory either way.
return { acted: nav, note: 'Page is loading — call elements to see what is on it.', success: true }
}
const run = activePreviewScriptRunner()
if (!run) {
return { error: NOTHING_OPEN, success: false }
}
const typed = action as PreviewActAction
// Annotation, not interaction: nothing is clicked, nothing settles, and the
// page is not re-read, so these skip the whole act-then-inventory path.
if (typed.kind === 'pin' || typed.kind === 'unpin' || typed.kind === 'hold' || typed.kind === 'strobe') {
const mark =
typed.kind === 'hold'
? buildHoldScript()
: typed.kind === 'strobe'
? buildStrobeScript()
: typed.kind === 'pin'
? buildPinScript(typed, typed.text || '')
: buildUnpinScript(typed)
const trip = await runJson(run, mark)
if (trip.kind === 'failed') {
return { error: trip.error, success: false }
}
return trip.kind === 'answered' ? trip.result : { acted: typed.kind, note: NAVIGATED, success: true }
}
const input = activePreviewInput()
if (input && DRIVEN.indexOf(typed.kind) !== -1) {
return driveAction(run, input, typed)
}
// A plain page scroll is a wheel gesture. Jumping to an end is not — no hand
// wheels to the bottom of a long article — so `to` stays a scripted glide.
const plain = typed.kind === 'scroll' && !typed.to && !typed.ref && !typed.selector
if (plain && input) {
return driveScroll(run, input, typed)
}
const settle = typed.kind === 'elements' ? 0 : SETTLE_MS
const scripted = await runJson(run, buildScriptedScript(typed, settle))
if (scripted.kind === 'failed') {
return { error: scripted.error, success: false }
}
// The action almost certainly landed — a page that stops answering right
// after a click is one that navigated. Say so instead of failing it.
return scripted.kind === 'silent' ? { acted: typed.kind, note: NAVIGATED, success: true } : scripted.result
}
// Self-accept so an edit here, or to the in-page sources this module
// stringifies, doesn't reload the whole renderer out from under a live session.
// The bridge re-requests this module per action in dev, so it picks the change
// up without one (see loadPreviewEngine in gateway-event/desktop-bridge.ts).
if (import.meta.hot) {
import.meta.hot.accept()
}
@@ -0,0 +1,138 @@
/**
* PREVIEW DRIVE — the agent's hand on the mouse and keyboard.
*
* Everything here goes out through `sendInputEvent`, so the guest page receives
* ordinary trusted input: the pointer really moves across it, `:hover` really
* matches, focus really lands, and a menu that only exists while hovered is
* actually open by the time the click arrives. That last part is the whole
* point — a synthetic `el.click()` on a dropdown item fires at a node the page
* never rendered, which is why script-driven clicking quietly misses.
*
* The pointer is walked to its target in steps rather than teleported, for the
* same reason: a page learns it is hovered from a stream of moves, and one
* move from nowhere to the target skips every element in between. Fourteen
* steps over ~280ms is what Playwright's action cursor settles on too.
*/
import type { PreviewInputHandle } from './preview-input'
export interface DrivePoint {
x: number
y: number
}
/** The pointer JUMPS. These few steps over a couple of frames exist only so the
* page receives a move stream at all — `:hover`, `pointerenter` and the menus
* that hang off them need more than a single teleport to resolve — not so the
* travel can be watched. A machine does not sweep a mouse across a screen. */
const GLIDE_STEPS = 3
const GLIDE_MS = 36
/** Gap between keystrokes. Fast enough not to pad the turn, slow enough that a
* page debouncing its input handler still sees them as separate. */
const KEY_MS = 7
/** Notches in a wheel gesture, and how long the gesture takes. Sent as a stream
* rather than one big delta so scroll-linked effects — sticky headers, reveal
* animations, infinite-scroll loaders — get the same series of events a hand
* would produce, and so it looks like momentum rather than a jump cut. */
const WHEEL_STEPS = 12
const WHEEL_MS = 240
/** Where the pointer was left, so the next glide starts from there instead of
* jumping. Module-level because the pointer is a property of the pane, not of
* any one action. The origin is a placeholder, not a position — see below. */
let pointer: DrivePoint = { x: 0, y: 0 }
let placed = false
/** Whether the pointer has ever been sent anywhere. Until it has, its notional
* spot is the top-left corner, which is a real coordinate a caller can wheel
* or click at by accident — so callers that only need it to be SOMEWHERE
* sensible have to know the difference. */
export function pointerPlaced(): boolean {
return placed
}
const wait = (ms: number) => new Promise(resolve => setTimeout(resolve, ms))
/** Decelerating, like a hand arriving at a target rather than a linear sweep. */
const easeOut = (t: number) => 1 - Math.pow(1 - t, 3)
/** Walk the pointer to `to`, letting the page hover everything on the way. */
export async function glideTo(input: PreviewInputHandle, to: DrivePoint): Promise<void> {
const from = pointer
for (let step = 1; step <= GLIDE_STEPS; step++) {
const progress = easeOut(step / GLIDE_STEPS)
input.send({
type: 'mouseMove',
x: Math.round(from.x + (to.x - from.x) * progress),
y: Math.round(from.y + (to.y - from.y) * progress)
})
await wait(GLIDE_MS / GLIDE_STEPS)
}
pointer = to
placed = true
}
/** Press and release at the pointer's current spot. `clicks` of 3 selects the
* text under it, which is how a field gets cleared without a modifier key. */
export async function clickAt(input: PreviewInputHandle, clicks = 1): Promise<void> {
for (let click = 1; click <= clicks; click++) {
input.send({ button: 'left', clickCount: click, type: 'mouseDown', x: pointer.x, y: pointer.y })
input.send({ button: 'left', clickCount: click, type: 'mouseUp', x: pointer.x, y: pointer.y })
await wait(KEY_MS)
}
}
/** Wheel `down` pixels' worth of notches at the pointer's current spot; negative
* scrolls up. Electron's wheel delta is the legacy `wheelDelta` sign — positive
* moves the content down, i.e. scrolls UP — so it is the inverse of the number
* a caller asks for, and of DOM `WheelEvent.deltaY`. */
export async function wheelBy(input: PreviewInputHandle, down: number): Promise<void> {
let sent = 0
for (let step = 1; step <= WHEEL_STEPS; step++) {
const so_far = Math.round(down * easeOut(step / WHEEL_STEPS))
const notch = so_far - sent
sent = so_far
if (notch) {
input.send({ deltaX: 0, deltaY: -notch, type: 'mouseWheel', x: pointer.x, y: pointer.y })
}
await wait(WHEEL_MS / WHEEL_STEPS)
}
}
/** Send one key the long way round, so a page watching any of the three sees it. */
export async function pressKey(input: PreviewInputHandle, key: string): Promise<void> {
input.send({ keyCode: key, type: 'keyDown' })
input.send({ keyCode: key, type: 'char' })
input.send({ keyCode: key, type: 'keyUp' })
await wait(KEY_MS)
}
/** Select-all inside whatever has focus, which for a focused field is that
* field's own text and nothing else. This replaced a triple-click: a triple
* click is a POINTER gesture, so it selects whatever paragraph sits under the
* cursor whenever the target turns out not to be a field, and the agent was
* leaving pages with their body text highlighted. There is no `char` phase —
* a chord is not text entry, and sending one types a literal 'a'. */
export async function selectAll(input: PreviewInputHandle): Promise<void> {
const chord = ['control', 'meta']
input.send({ keyCode: 'a', modifiers: chord, type: 'keyDown' })
input.send({ keyCode: 'a', modifiers: chord, type: 'keyUp' })
await wait(KEY_MS)
}
/** Type `text` a character at a time into whatever currently has focus. */
export async function typeText(input: PreviewInputHandle, text: string): Promise<void> {
for (const character of text) {
await pressKey(input, character)
}
}
@@ -0,0 +1,56 @@
/**
* PREVIEW INPUT REGISTRY — real input into the preview pane's guest page, the
* difference between the agent DRIVING the browser and merely poking its DOM.
*
* `executeJavaScript` can only ever dispatch synthetic events: `isTrusted` is
* false, the browser's own hover target never moves, `:hover` rules never
* match, and hover-gated menus never open — so a click lands on a dropdown item
* that was never rendered. `sendInputEvent` goes in through Chromium's input
* pipeline instead, producing the same events a hand on the mouse would.
*
* It has to be called on the `<webview>` ELEMENT. Sending to the embedder's
* webContents does not reach a guest (electron/electron#20333), which is why
* this is a per-pane registry rather than something main could do.
*
* Coordinates are relative to the webview, and the webview IS the guest
* viewport — so a rect the act engine measured inside the page needs no
* conversion on the way back out.
*/
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs } from '@/store/preview'
/** The subset of Electron's input events the agent needs to drive a page. */
export type PreviewInputEvent =
| { button: 'left'; clickCount: number; type: 'mouseDown' | 'mouseUp'; x: number; y: number }
| { deltaX: number; deltaY: number; type: 'mouseWheel'; x: number; y: number }
| { keyCode: string; modifiers?: string[]; type: 'char' | 'keyDown' | 'keyUp' }
| { type: 'mouseMove'; x: number; y: number }
export interface PreviewInputHandle {
/** Give the guest keyboard focus, so key events reach its active element. */
focus: () => void
send: (event: PreviewInputEvent) => void
}
const handles = new Map<string, PreviewInputHandle>()
/** Register a live pane's input channel; returns an idempotent unregister. */
export function registerPreviewInput(tabId: string, handle: PreviewInputHandle): () => void {
handles.set(tabId, handle)
return () => {
if (handles.get(tabId) === handle) {
handles.delete(tabId)
}
}
}
/** The ACTIVE preview tab's input channel. Null = nothing real to drive, and
* the caller falls back to synthesizing events inside the page. */
export function activePreviewInput(): PreviewInputHandle | null {
const tabs = $previewTabs.get()
const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0]
return (tab && handles.get(tab.id)) || null
}
@@ -0,0 +1,29 @@
/**
* IDLE PULSE — keeps the preview overlay alive while the model reasons.
*
* Everything else the overlay draws is narration of an action, so the pane goes
* dark for the gap between actions — which is most of the time the agent spends
* on a page, and the part where it is deciding what to do with what it just
* read. A page that stops responding the moment the agent is thinking hardest
* reads as the agent having wandered off.
*
* This is deliberately NOT part of the act pipeline: a turn boundary only has
* to poke the overlay the last action left behind. See preview-nudge.ts.
*/
import { $busy } from '@/store/session'
import { nudgeOverlay } from './preview-nudge'
// Module-level, matching how review.ts and coding-status.ts watch this edge. It
// is inert without a live pane, so there is nothing to mount or tear down.
let running = $busy.get()
$busy.subscribe(busy => {
if (busy === running) {
return
}
running = busy
nudgeOverlay(busy ? 'think' : 'rest')
})
@@ -9,6 +9,9 @@
* sitting in Hermes' own DOM, where `activeElement` is authoritative.
*/
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs } from '@/store/preview'
/** Marks a live browser pane so a gesture can find the one holding focus. */
export const PREVIEW_BROWSER_ATTR = 'data-preview-browser'
@@ -31,6 +34,15 @@ export function registerPreviewNav(tabId: string, handle: PreviewNavHandle): ()
}
}
/** The ACTIVE preview tab's commands, for callers with no focus to key off —
* the agent's drive_preview, which runs while focus is in the composer. */
export function activePreviewNav(): PreviewNavHandle | null {
const tabs = $previewTabs.get()
const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0]
return (tab && handles.get(tab.id)) || null
}
/** Run `command` on the browser pane holding DOM focus. False = focus is
* elsewhere in the app, so the caller falls back to the app-level meaning. */
export function commandFocusedPreview(command: keyof PreviewNavHandle): boolean {
@@ -0,0 +1,38 @@
/**
* OVERLAY NUDGE — the one way to say a stage to an overlay that is ALREADY on
* the page.
*
* The act pipeline (preview-act.ts) injects the whole engine because it has to:
* it needs to resolve a ref, measure a node and read the page back. The stages
* that only narrate — the idle pulse, the read wipe — need none of that. They
* are one function call against something the last action already left on the
* page, so re-shipping the engine to make it would put the payload on the wire
* on every turn boundary.
*
* A page the agent has never acted on has no overlay, and the nudge is a no-op
* there. That is the right answer rather than a gap: chrome on a page the agent
* never touched would be a lie about what it did.
*/
import type { WatchStage } from '@/lib/preview-act/watch-in-page'
import { activePreviewScriptRunner } from './preview-script-runner'
/** Run one stage against the active pane's overlay, if it has one. */
export function nudgeOverlay(stage: WatchStage): void {
const run = activePreviewScriptRunner()
if (!run) {
return
}
void run(`(function () {
var w = window;
var fn = w.__hermesWatch_fn;
if (!fn) { return 'cold'; }
try { fn(document, w.__hermesActHolder || {}, ${JSON.stringify(stage)}); } catch (err) {}
return 'ok';
})()`).catch(() => {
// The page navigated out from under us. The next action re-injects.
})
}
@@ -1,3 +1,7 @@
// Side-effect import: watches the turn edge so the overlay keeps a pulse while
// the model reasons. Lives here because the pane is what makes it reachable.
import './preview-mind'
import { useStore } from '@nanostores/react'
import type { PointerEvent as ReactPointerEvent } from 'react'
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
@@ -28,9 +32,10 @@ import {
import { type ConsoleEntry } from './preview-console-state'
import { previewConsoleState } from './preview-console-store'
import { LocalFilePreview, PreviewEmptyState } from './preview-file'
import { type PreviewInputEvent, registerPreviewInput } from './preview-input'
import { PREVIEW_BROWSER_ATTR, registerPreviewNav } from './preview-nav'
import { registerPreviewPageReader } from './preview-reader'
import { registerPreviewTourRunner } from './preview-tour-runner'
import { registerPreviewScriptRunner } from './preview-script-runner'
type PreviewWebview = HTMLElement & {
canGoBack?: () => boolean
@@ -53,6 +58,7 @@ type PreviewWebview = HTMLElement & {
reloadIgnoringCache?: () => void
replaceMisspelling?: (word: string) => void
selectAll?: () => void
sendInputEvent?: (event: PreviewInputEvent) => void
}
/** The raw Chromium params riding the webview tag's `context-menu` event. */
@@ -455,15 +461,15 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest =
})
}, [isWebPreview, tabId])
// Publish the TOUR runner for this tab (the tour tool, surface='preview'):
// runs injected driver.js actions inside the guest page so the agent can
// give guided walkthroughs of whatever web app is open here.
// Publish the SCRIPT runner for this tab: the one channel into the guest
// page, shared by the tour tool (injected driver.js walkthroughs) and the
// drive_preview tool (clicking, typing, scrolling the page the user sees).
useEffect(() => {
if (!isWebPreview || !tabId) {
return
}
return registerPreviewTourRunner(tabId, async code => {
return registerPreviewScriptRunner(tabId, async code => {
const webview = webviewRef.current
if (!webview?.executeJavaScript) {
@@ -474,6 +480,32 @@ export function PreviewPane({ embedded = false, onRestartServer, reloadRequest =
})
}, [isWebPreview, tabId])
// Publish the INPUT channel for this tab. Same idea as the script runner, but
// it carries real Chromium input rather than script — the agent's clicks and
// keystrokes arrive as trusted events, so the page hovers, focuses and reacts
// exactly as it would under a human hand.
useEffect(() => {
if (!isWebPreview || isRemoteHtml || !tabId) {
return
}
return registerPreviewInput(tabId, {
focus: () => webviewRef.current?.focus?.(),
send: event => {
const webview = webviewRef.current
// Never optional-chain this call away: a missing method would make every
// agent click a silent no-op that still reports success, because the
// overlay and the read-back both run on the separate script channel.
if (typeof webview?.sendInputEvent !== 'function') {
throw new Error('preview webview cannot take input events')
}
webview.sendInputEvent(event)
}
})
}, [isRemoteHtml, isWebPreview, tabId])
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
if (!consoleOpen) {
@@ -15,6 +15,8 @@
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs } from '@/store/preview'
import { nudgeOverlay } from './preview-nudge'
export interface PreviewReadOptions {
/** Characters to return from `start` (capped at PREVIEW_READ_MAX_CHARS). */
count?: number
@@ -89,6 +91,13 @@ export async function readActivePreview(opts: PreviewReadOptions = {}): Promise<
try {
const page = await reader()
// Say it on the page. Reading is by far the cheapest thing the agent
// does — a few hundredths of a second against a model round trip either
// side of it — so a run of reads used to leave the pane dark for the
// twenty seconds it took to page through a document, immediately after
// the one moment that showed anything.
nudgeOverlay('read')
return windowText(
{ kind: target.kind, path: target.path, title: page.title || target.label, url: page.url || target.url },
page.text,
@@ -0,0 +1,38 @@
/**
* PREVIEW SCRIPT RUNNER REGISTRY — the one way anything in the app reaches into
* the preview pane's guest page, the script analog of preview-nav's handle
* registry.
*
* A live browser pane registers its webview's `executeJavaScript` here, keyed
* by tab id; `activePreviewScriptRunner` resolves the ACTIVE tab from the
* store. Both guest-page features ride it — the tour tool (preview-tour.ts)
* and the interaction tool (preview-act.ts) — so their heavy payloads stay out
* of the pane component's static import graph and only load when used.
*/
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs } from '@/store/preview'
/** Runs JS source in the pane's guest page, resolving its completion value. */
export type PreviewScriptRunner = (code: string) => Promise<unknown>
const runners = new Map<string, PreviewScriptRunner>()
/** Register a live preview's script runner; returns an idempotent unregister. */
export function registerPreviewScriptRunner(tabId: string, runner: PreviewScriptRunner): () => void {
runners.set(tabId, runner)
return () => {
if (runners.get(tabId) === runner) {
runners.delete(tabId)
}
}
}
/** The ACTIVE preview tab's script runner. Null = no live page behind it. */
export function activePreviewScriptRunner(): PreviewScriptRunner | null {
const tabs = $previewTabs.get()
const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0]
return (tab && runners.get(tab.id)) || null
}
@@ -1,38 +0,0 @@
/**
* PREVIEW TOUR RUNNER REGISTRY — the tour tool's script bridge into the
* preview pane, the tour analog of preview-nav's handle registry.
*
* A live browser pane registers a script runner (its webview's
* `executeJavaScript`) here, keyed by tab id; `activeTourRunner` resolves the
* ACTIVE tab from the store. Kept separate from preview-tour.ts so the pane
* component's static import stays tiny — the engine source + driver.js
* payload only load when a tour actually runs (gateway-event dynamic-imports
* preview-tour.ts).
*/
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs } from '@/store/preview'
/** Runs JS source in the pane's guest page, resolving its completion value. */
export type TourScriptRunner = (code: string) => Promise<unknown>
const runners = new Map<string, TourScriptRunner>()
/** Register a live preview's script runner; returns an idempotent unregister. */
export function registerPreviewTourRunner(tabId: string, runner: TourScriptRunner): () => void {
runners.set(tabId, runner)
return () => {
if (runners.get(tabId) === runner) {
runners.delete(tabId)
}
}
}
/** The ACTIVE preview tab's script runner. Null = no live page to tour. */
export function activeTourRunner(): TourScriptRunner | null {
const tabs = $previewTabs.get()
const tab = tabs.find(t => t.id === $rightRailActiveTabId.get()) ?? tabs[0]
return (tab && runners.get(tab.id)) || null
}
@@ -21,7 +21,7 @@ import driverIife from 'driver.js/dist/driver.js.iife.js?raw'
import { collectTourTargets } from '@/lib/tour/collect-targets'
import { runTourEngine, type TourAction, type TourResult } from '@/lib/tour/engine'
import { activeTourRunner } from './preview-tour-runner'
import { activePreviewScriptRunner } from './preview-script-runner'
/** Build the idempotent inject-and-run script for one tour action. */
function buildTourScript(action: TourAction): string {
@@ -51,7 +51,7 @@ function buildTourScript(action: TourAction): string {
/** Run one tour action in the ACTIVE preview tab's page. */
export async function runPreviewTour(action: TourAction): Promise<TourResult> {
const run = activeTourRunner()
const run = activePreviewScriptRunner()
if (!run) {
return { error: 'No live page is open in the preview pane — open one first.', success: false }
@@ -2,6 +2,7 @@ import { readActivePreview } from '@/app/chat/right-rail/preview-reader'
import { writeAgentTerminalChunk } from '@/app/right-sidebar/terminal/agent-terminal-stream'
import { readActiveTerminal } from '@/app/right-sidebar/terminal/buffer'
import { closeAgentTerminalByProc } from '@/app/right-sidebar/terminal/terminals'
import type { PreviewActAction } from '@/lib/preview-act/act-in-page'
import type { TourAction, TourStep } from '@/lib/tour'
import { $gateway } from '@/store/gateway'
import { applyDesktopLayoutPreset, revealDesktopPane } from '@/store/pane-focus'
@@ -10,6 +11,34 @@ import { setMessages } from '@/store/session'
import type { GatewayEventContext } from './types'
/** The preview engine, loaded on demand so ~25KB of page-injectable source stays
* off the boot path.
*
* In dev that lazy chunk is also a trap. The browser caches a dynamic import by
* URL for the life of the page, so this bridge would hand every action to
* whichever build of the engine loaded first, and no edit to it — or to the
* overlay whose source it stringifies into the page — would reach the guest
* until the whole window reloaded. Asking for a fresh copy is more reliable
* than trusting hot-update propagation to reach a module nothing statically
* imports; the dev server stamps the dependency URLs it has invalidated, so a
* fresh engine pulls a fresh overlay down with it.
*
* The literal path is what a bare specifier can't be here, and it has to track
* this module's real location — hence the fall back to the static import, which
* is also the only branch production keeps, `import.meta.hot` being stripped
* there along with everything it guards. */
const loadPreviewEngine = () => {
const stable = () => import('@/app/chat/right-rail/preview-act')
if (!import.meta.hot) {
return stable().then(mod => mod.actOnActivePreview)
}
return import(/* @vite-ignore */ '/src/app/chat/right-rail/preview-act.ts?hot=' + Date.now())
.catch(stable)
.then(mod => mod.actOnActivePreview as Awaited<ReturnType<typeof stable>>['actOnActivePreview'])
}
/** Desktop-surface bridge events: read-back requests the agent blocks on
* (terminal/preview/window), agent terminal streaming, pane reveal, and
* message reactions. */
@@ -55,6 +84,50 @@ export function handleDesktopBridgeEvent(ctx: GatewayEventContext): boolean {
return true
}
if (event.type === 'preview.act.request') {
// drive_preview tool: click/type/scroll/press inside the guest page, or
// drive the pane's history. Dynamic import keeps the injected engine off
// the boot path. Active session only: a background turn must never reach
// into the page the user is working in (desktop AGENTS.md: offer, don't
// hijack).
const requestId = typeof payload?.request_id === 'string' ? payload.request_id : ''
if (requestId) {
const answer = (result: unknown) =>
$gateway.get()?.request('preview.act.respond', {
request_id: requestId,
text: result ? JSON.stringify(result) : ''
})
if (isActiveEvent) {
void loadPreviewEngine()
.then(run =>
run({
amount: payload?.amount,
key: payload?.key,
kind: payload?.action ?? '',
max: payload?.max,
ref: payload?.ref,
selector: payload?.selector,
submit: payload?.submit,
text: payload?.text,
to: payload?.to as PreviewActAction['to']
})
)
.then(answer, error =>
answer({ error: error instanceof Error ? error.message : String(error), success: false })
)
} else {
void answer({
error: 'The in-app browser only takes actions in the session the user is looking at.',
success: false
})
}
}
return true
}
if (event.type === 'window.read.request') {
// read_window_below tool: main owns native window enumeration, so ask
// it over IPC and answer. Empty text = unavailable (no bridge, or
@@ -121,6 +121,15 @@ export type GatewayEventPayload = {
side?: string
steps?: unknown
step_index?: number
// preview.act.request (drive_preview tool — agent clicking/typing/scrolling in
// the in-app browser). `action` names the verb and `selector` is shared with
// tour above; `ref` addresses an element from the last inventory.
ref?: string
submit?: boolean
key?: string
amount?: number
to?: string
max?: number
// message.reaction (agent reacting via the react_to_message tool) — the
// durable messages.id, that row's full reaction list after the write, and
// the row's role so a live (not-yet-round-tripped) message can be matched.
@@ -0,0 +1,730 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { actEngineSource, actInPage, type PreviewActHolder } from './act-in-page'
/** jsdom lays nothing out, so every rect is 0×0 and the engine's visibility
* check would reject the whole page. Give elements a plausible box, honouring
* an explicit width/height so a test can lay out a 1px one, and let
* `display: none` (which jsdom DOES compute) carry the hiding. */
function layOutTheDocument() {
vi.spyOn(Element.prototype, 'getBoundingClientRect').mockImplementation(function (this: Element) {
const style = getComputedStyle(this)
const width = style.display === 'none' ? 0 : parseFloat(style.width) || 40
const height = style.display === 'none' ? 0 : parseFloat(style.height) || 40
const left = parseFloat(style.left) || 0
return { bottom: height, height, left, right: left + width, top: 0, width, x: left, y: 0 } as DOMRect
})
}
function page(html: string): PreviewActHolder {
document.body.innerHTML = html
return {}
}
/** Take the inventory the agent would take before acting. */
function inventory(holder: PreviewActHolder) {
return actInPage(document, holder, { kind: 'elements' })
}
/** Take the inventory and hand back the handle for the first thing on the page.
* Tests address elements the way the agent does — by asking what they are
* called — rather than predicting what the engine will name them. */
function firstRef(holder: PreviewActHolder) {
return inventory(holder).elements![0].ref
}
beforeEach(() => {
vi.restoreAllMocks()
layOutTheDocument()
Element.prototype.scrollIntoView = vi.fn()
// jsdom has no hit-testing at all, so the engine skips the occlusion check
// here by default. Tests that stand one in must not leak it into the next.
delete (document as unknown as { elementFromPoint?: unknown }).elementFromPoint
})
describe('self-containment', () => {
// The pane injects `actEngineSource()` into the guest page, where module
// scope does not exist. A single free identifier — a module-level constant,
// an imported helper, or a kit factory the assembler forgot to ship — is a
// ReferenceError on every call, and Electron reports it only as "Script
// failed to execute". Evaluating the bundle the way the guest does is the
// only way to catch that here: a plain import of these functions resolves
// the very names the guest would be missing.
//
// It is also the guard on the build. The renderer minifies, so an imported
// binding named inside the stringified core would be renamed to something
// the bundle never declares — green in dev and in this suite if the source
// were assembled any other way, broken only once packaged.
it('runs after being stringified and eval’d with no module scope', () => {
const page = document.createElement('div')
page.innerHTML = '<button id="save">Save</button>'
document.body.replaceChildren(page)
const injected = new Function('return ' + actEngineSource())() as typeof actInPage
const holder: PreviewActHolder = {}
const [save] = injected(document, holder, { kind: 'elements' }).elements!
expect(save.label).toBe('Save')
const clicked = vi.fn()
document.getElementById('save')!.addEventListener('click', clicked)
expect(injected(document, holder, { kind: 'click', ref: save.ref }).success).toBe(true)
expect(clicked).toHaveBeenCalledOnce()
})
})
describe('elements', () => {
// The handle says what the thing is and which one it is, so a delta line the
// agent reads ten turns later needs no lookup to make sense of.
it('names the interactive nodes after their role and label', () => {
const holder = page(`
<button id="save">Save</button>
<a href="/help">Help</a>
<input id="who" placeholder="Your name" />
<p>Not interactive</p>
`)
const result = inventory(holder)
expect(result.success).toBe(true)
expect(result.elements?.map(e => [e.ref, e.label])).toEqual([
['btn-save', 'Save'],
['lnk-help', 'Help'],
['inp-your-name', 'Your name']
])
})
it('tells two elements with the same name apart', () => {
const holder = page(`
<button>Edit</button>
<button>Edit</button>
<button>Edit</button>
`)
expect(inventory(holder).elements?.map(e => e.ref)).toEqual(['btn-edit', 'btn-edit-1', 'btn-edit-2'])
})
// Handing a retired name to a different element would silently redirect a
// handle the agent is still holding. The two ids here also disagree, which is
// the page saying outright that these are different buttons — so this must
// not re-bind despite the identical label.
it('never reissues the name of an element that went away', () => {
const holder = page(`
<div id="host"><button id="a">Edit</button></div>
<a href="/help">Help</a>
<a href="/terms">Terms</a>
`)
expect(firstRef(holder)).toBe('btn-edit')
document.getElementById('host')!.innerHTML = '<button id="b">Edit</button>'
const again = inventory(holder)
expect(again.delta?.removed).toEqual(['btn-edit'])
expect(again.delta?.added?.map(e => e.ref)).toEqual(['btn-edit-1'])
})
it('reports role, current value, and disabled state', () => {
const holder = page(`
<input id="email" aria-label="Email" type="email" value="a@b.co" />
<button id="go" disabled>Go</button>
`)
const [email, go] = inventory(holder).elements!
expect(email).toMatchObject({ label: 'Email', role: 'input:email', value: 'a@b.co' })
expect(go).toMatchObject({ disabled: true, label: 'Go' })
expect(email.disabled).toBeUndefined()
})
it('skips hidden controls and unlabelled ones', () => {
const holder = page(`
<button id="real">Real</button>
<button id="gone" style="display: none">Gone</button>
<button id="mystery"></button>
`)
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Real'])
})
// The screen-reader-only recipe is a 1px box parked at the document origin,
// and it passed a `width >= 1` check. Skip links and CSS-only menu toggles are
// built this way and come FIRST in the document, so they landed on @e1 — the
// agent would aim at one and the pointer would fly to the top-left corner and
// click nothing.
it('skips the visually-hidden controls that sit at the document origin', () => {
const holder = page(`
<a href="#content" style="width: 1px; height: 1px; clip: rect(0, 0, 0, 0)">Jump to content</a>
<input aria-label="Toggle sidebar" style="width: 1px; height: 1px" type="checkbox" />
<a href="#main" style="clip-path: inset(50%)">Skip navigation</a>
<button id="real">Search</button>
`)
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search'])
})
it('skips a control parked off the left edge, which no scroll brings back', () => {
const holder = page(`
<button style="position: absolute; left: -9999px">Hidden</button>
<button id="real">Search</button>
`)
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search'])
})
// Wikipedia and friends are full of these: decorative chrome and collapsed
// menus that are perfectly solid boxes as far as layout is concerned, but
// that the page has already declared are not for anyone to interact with.
it('skips controls the page marked aria-hidden or inert', () => {
const holder = page(`
<div aria-hidden="true"><button>Decoration</button></div>
<div inert><button>Collapsed menu</button></div>
<button id="real">Search</button>
`)
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search'])
})
it('skips a control buried under another layer, which would take the click instead', () => {
const holder = page(`
<button style="left: 100px">Accept cookies</button>
<button id="real" style="left: 300px">Search</button>
`)
const wall = document.createElement('div')
document.body.append(wall)
// Stand in for the hit-testing jsdom does not do: every element reports
// itself except the buried one, which reports the sheet lying over it.
document.elementFromPoint = (x: number) => (x === 120 ? wall : document.getElementById('real'))
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search'])
})
it('keeps a control whose own child is what the hit test lands on', () => {
const holder = page('<a href="/home" id="real"><svg id="icon"></svg> Home</a>')
document.elementFromPoint = () => document.getElementById('icon')
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Home'])
})
// The overlay draws far more than the agent is told about, so the two lists
// are collected separately: the field is every visible control, the inventory
// is the describable subset that refs point into.
it('parks the whole interactive field for the overlay, not just the inventory', () => {
const holder = page(`
<button id="named">Search</button>
<button id="mystery"></button>
<button id="gone" style="display: none">Gone</button>
`)
expect(inventory(holder).elements?.map(e => e.label)).toEqual(['Search'])
expect(holder.nodes).toEqual([document.getElementById('named')])
expect(holder.field).toEqual([document.getElementById('named'), document.getElementById('mystery')])
})
// The positional `:nth-child` chain this used to fall back to was 74% of a
// real page's inventory and nothing read it. An identity selector is short
// and stable, so it stays; everything else addresses by ref.
it('carries an identity selector only, and omits it when there is none', () => {
const holder = page(`
<div><button data-testid="submit">Send</button></div>
<div><button id="cancel">Cancel</button></div>
<div><button>Plain</button></div>
`)
const [byTestId, byId, plain] = inventory(holder).elements!
expect(byTestId.selector).toBe('[data-testid="submit"]')
expect(byId.selector).toBe('#cancel')
expect(plain.selector).toBeUndefined()
expect(plain.ref).toBe('btn-plain')
})
it('honours the cap', () => {
const holder = page(Array.from({ length: 10 }, (_, i) => `<button>B${i}</button>`).join(''))
expect(inventory(holder).elements).toHaveLength(10)
expect(actInPage(document, holder, { full: true, kind: 'elements', max: 3 }).elements).toHaveLength(3)
})
})
// Re-sending the whole inventory after every click is what made a ten-step
// session cost several times what it needed to: the page barely moves between
// steps and the agent was charged for a fresh copy of it each time.
describe('delta', () => {
it('gives the full inventory the first time it looks at a page', () => {
const holder = page('<button>Save</button>')
const first = inventory(holder)
expect(first.elements).toHaveLength(1)
expect(first.delta).toBeUndefined()
})
it('reports only what moved on every look after that', () => {
const holder = page(`
<div id="host">
<button id="save">Save</button>
<button id="undo">Undo</button>
<button id="redo">Redo</button>
</div>
`)
inventory(holder)
document.getElementById('host')!.insertAdjacentHTML('beforeend', '<button id="quit">Quit</button>')
const next = inventory(holder)
expect(next.elements).toBeUndefined()
expect(next.delta?.added?.map(e => e.ref)).toEqual(['btn-quit'])
expect(next.delta?.same).toBe(3)
})
it('says nothing about a page that did not move', () => {
const holder = page('<button id="save">Save</button>')
inventory(holder)
const still = inventory(holder)
expect(still.delta).toEqual({ same: 1 })
})
it('reports a relabelled control as changed, keeping its handle', () => {
const holder = page(`
<button id="cart">Add to cart</button>
<a href="/help">Help</a>
<a href="/terms">Terms</a>
`)
const ref = firstRef(holder)
document.getElementById('cart')!.textContent = 'Added'
const next = inventory(holder)
expect(next.delta?.changed?.map(e => [e.ref, e.label])).toEqual([[ref, 'Added']])
expect(actInPage(document, holder, { kind: 'click', ref }).success).toBe(true)
})
// `changed` fires on nearly every step of a long task, so it is the one part
// of the payload whose cost compounds. It carries the moved field and the
// handle, and nothing the agent already knows.
it('reports only the field that moved, not the whole element', () => {
const holder = page(`
<input id="q" aria-label="Search" />
<a href="/help">Help</a>
<a href="/terms">Terms</a>
`)
inventory(holder)
;(document.getElementById('q') as HTMLInputElement).value = 'shoes'
const [moved] = inventory(holder).delta!.changed!
expect(moved).toEqual({ ref: 'inp-search', value: 'shoes' })
})
it('reports a control becoming available on its own', () => {
const holder = page(`
<button id="go" disabled>Continue</button>
<a href="/help">Help</a>
<a href="/terms">Terms</a>
`)
inventory(holder)
;(document.getElementById('go') as HTMLButtonElement).disabled = false
expect(inventory(holder).delta?.changed).toEqual([{ disabled: false, ref: 'btn-continue' }])
})
it('falls back to the whole inventory when most of the page is new', () => {
const holder = page('<div id="host"><button id="save">Save</button></div>')
inventory(holder)
document.getElementById('host')!.innerHTML = '<a href="/a">A</a><a href="/b">B</a><a href="/c">C</a>'
const next = inventory(holder)
expect(next.delta).toBeUndefined()
expect(next.elements?.map(e => e.ref)).toEqual(['lnk-a', 'lnk-b', 'lnk-c'])
})
// An element that slid past the cap is still clickable, so calling it removed
// would be a lie the agent acts on.
it('does not report an element as removed just because it fell past the cap', () => {
const holder = page(Array.from({ length: 4 }, (_, i) => `<button>B${i}</button>`).join(''))
inventory(holder)
const capped = actInPage(document, holder, { kind: 'elements', max: 2 })
expect(capped.delta?.removed).toBeUndefined()
expect(actInPage(document, holder, { kind: 'click', ref: 'btn-b3' }).success).toBe(true)
})
// The contract the whole delta exists for. Not a fixed byte count — that
// would break on any wording change — but the relationship between the two
// payloads, which is what has to hold.
it('costs a fraction of the inventory on a page that mostly held still', () => {
const holder = page(`
<nav>${Array.from({ length: 8 }, (_, i) => `<a href="/n${i}">Section ${i}</a>`).join('')}</nav>
<main>
${Array.from({ length: 24 }, (_, i) => `<button id="row-${i}">Row action ${i}</button>`).join('')}
<input id="q" placeholder="Search everything" />
<div id="host"></div>
</main>
`)
const baseline = JSON.stringify(inventory(holder).elements)
document.getElementById('host')!.innerHTML = '<button id="toast">Dismiss</button>'
document.getElementById('row-3')!.textContent = 'Row action 3 (done)'
const next = inventory(holder)
expect(next.delta?.same).toBe(32)
// Measured at roughly 15x on this fixture; the bar is set well below that
// so a wording change does not fail the build, but a regression to
// re-sending the page would.
expect(JSON.stringify(next.delta).length * 5).toBeLessThan(baseline.length)
})
it('retires every handle when the page navigates', () => {
const holder = page('<button id="save">Save</button>')
inventory(holder)
holder.url = 'https://elsewhere.example/other'
const landed = inventory(holder)
expect(landed.delta).toBeUndefined()
expect(landed.elements?.map(e => e.ref)).toEqual(['btn-save'])
})
})
// The headline case. A framework re-render destroys the node and builds a new
// one; the agent's handle has to survive that, and it has to hear about it in
// one word rather than as a removal it must react to plus an addition it must
// re-read.
describe('rebind', () => {
/** A page with a stable nav around the part that re-renders, so the re-bind
* has to pick its candidate rather than being handed the only one going. */
function app(inner: string) {
return page(`
<nav><a href="/">Home</a><a href="/docs">Docs</a><a href="/pricing">Pricing</a></nav>
<main id="host">${inner}</main>
`)
}
function rerender(inner: string) {
document.getElementById('host')!.innerHTML = inner
}
it('keeps the handle when a re-render replaces the node', () => {
const holder = app('<button>Sign in</button>')
const ref = inventory(holder).elements!.find(e => e.label === 'Sign in')!.ref
const before = document.querySelector('button')
rerender('<span><button>Sign in</button></span>')
const next = inventory(holder)
expect(document.querySelector('button')).not.toBe(before)
expect(next.delta?.rebound).toEqual([ref])
expect(next.delta?.added).toBeUndefined()
expect(next.delta?.removed).toBeUndefined()
expect(actInPage(document, holder, { kind: 'click', ref }).success).toBe(true)
})
it('follows a label through a count badge appearing on it', () => {
const holder = app('<a href="/in">Inbox</a>')
const ref = inventory(holder).elements!.find(e => e.label === 'Inbox')!.ref
rerender('<div><a href="/in">Inbox (3)</a></div>')
expect(inventory(holder).delta?.rebound).toEqual([ref])
})
it('will not move a handle across roles', () => {
const holder = app('<button>Continue</button>')
inventory(holder)
rerender('<a href="/next">Continue</a>')
const next = inventory(holder)
expect(next.delta?.rebound).toBeUndefined()
expect(next.delta?.removed).toEqual(['btn-continue'])
expect(next.delta?.added?.map(e => e.ref)).toEqual(['lnk-continue'])
})
// Two unrelated buttons trading places must not trade handles with them.
it('mints a new handle rather than guess between unrelated candidates', () => {
const holder = app('<button>Delete account</button>')
inventory(holder)
rerender('<button>Upload photo</button>')
const next = inventory(holder)
expect(next.delta?.rebound).toBeUndefined()
expect(next.delta?.removed).toEqual(['btn-delete-account'])
expect(next.delta?.added?.map(e => e.ref)).toEqual(['btn-upload-photo'])
})
// Both carry an id and the ids disagree, which is the page saying outright
// that these are two different controls however alike they read.
it('will not move a handle between elements the page marks as distinct', () => {
const holder = app('<button id="save-draft">Save</button>')
inventory(holder)
rerender('<button id="save-final">Save</button>')
const next = inventory(holder)
expect(next.delta?.rebound).toBeUndefined()
expect(next.delta?.removed).toEqual(['btn-save'])
})
})
describe('click', () => {
it('activates the element a ref points at', () => {
const holder = page('<button id="save">Save</button>')
const ref = firstRef(holder)
const clicked = vi.fn()
document.getElementById('save')!.addEventListener('click', clicked)
const result = actInPage(document, holder, { kind: 'click', ref })
expect(result.success).toBe(true)
expect(result.acted).toContain('Save')
expect(clicked).toHaveBeenCalledOnce()
})
it('replays the pointer/mouse sequence frameworks bind to', () => {
const holder = page('<button id="save">Save</button>')
const ref = firstRef(holder)
const seen: string[] = []
for (const type of ['pointerdown', 'mousedown', 'mouseup', 'pointerup', 'click']) {
document.getElementById('save')!.addEventListener(type, () => seen.push(type))
}
actInPage(document, holder, { kind: 'click', ref })
expect(seen).toContain('mousedown')
expect(seen).toContain('mouseup')
expect(seen.at(-1)).toBe('click')
})
it('takes a raw CSS selector when no ref fits', () => {
const holder = page('<button id="save">Save</button>')
const clicked = vi.fn()
document.getElementById('save')!.addEventListener('click', clicked)
expect(actInPage(document, holder, { kind: 'click', selector: '#save' }).success).toBe(true)
expect(clicked).toHaveBeenCalledOnce()
})
it('refuses a disabled control instead of silently doing nothing', () => {
const holder = page('<button id="save" disabled>Save</button>')
const result = actInPage(document, holder, { kind: 'click', ref: firstRef(holder) })
expect(result.success).toBe(false)
expect(result.error).toContain('disabled')
})
it('reports the live url so a navigation is visible to the agent', () => {
const holder = page('<button id="save">Save</button>')
expect(actInPage(document, holder, { kind: 'click', ref: firstRef(holder) }).url).toBe(document.location.href)
})
})
describe('stale refs', () => {
it('names an unknown ref rather than acting on whatever is nearby', () => {
const holder = page('<button>Only</button>')
inventory(holder)
const result = actInPage(document, holder, { kind: 'click', ref: 'btn-imaginary' })
expect(result.success).toBe(false)
expect(result.error).toContain('elements')
})
it('catches a node that was removed after the snapshot', () => {
const holder = page('<button id="save">Save</button>')
const ref = firstRef(holder)
document.getElementById('save')!.remove()
expect(actInPage(document, holder, { kind: 'click', ref }).error).toContain('removed')
})
it('invalidates every ref when the page navigated under them', () => {
const holder = page('<button>Save</button>')
const ref = firstRef(holder)
holder.url = 'https://elsewhere.example/other'
expect(actInPage(document, holder, { kind: 'click', ref }).error).toContain('navigated')
})
it('asks for a target when given neither', () => {
expect(actInPage(document, page('<button>Save</button>'), { kind: 'click' }).error).toContain('selector')
})
it('reports a selector that matches nothing', () => {
expect(actInPage(document, page(''), { kind: 'click', selector: '#nope' }).error).toContain('No element')
})
})
describe('type', () => {
it('enters text and fires the events a controlled input listens for', () => {
const holder = page('<input id="who" placeholder="Your name" />')
const ref = firstRef(holder)
const input = document.getElementById('who') as HTMLInputElement
const events: string[] = []
input.addEventListener('input', () => events.push('input'))
input.addEventListener('change', () => events.push('change'))
const result = actInPage(document, holder, { kind: 'type', ref, text: 'Brooklyn' })
expect(result.success).toBe(true)
expect(input.value).toBe('Brooklyn')
expect(events).toEqual(['input', 'change'])
})
it('bypasses the own-property shadow React installs on tracked inputs', () => {
const holder = page('<input id="who" placeholder="Your name" />')
const ref = firstRef(holder)
// React defines its own `value` accessor on the node to track what it last
// wrote, and ignores an input event that agrees with it. Writing through
// that shadow is exactly how typed text snaps back on the next render.
const input = document.getElementById('who') as HTMLInputElement
const nativeValue = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')!
const shadowWrites: string[] = []
Object.defineProperty(input, 'value', {
configurable: true,
get: () => nativeValue.get!.call(input),
set: (next: string) => {
shadowWrites.push(next)
}
})
actInPage(document, holder, { kind: 'type', ref, text: 'Brooklyn' })
expect(shadowWrites).toEqual([])
expect(nativeValue.get!.call(input)).toBe('Brooklyn')
})
it('writes into a contenteditable host', () => {
const holder = page('<div id="editor" contenteditable="true" aria-label="Body"></div>')
actInPage(document, holder, { kind: 'type', ref: firstRef(holder), text: 'hello' })
expect(document.getElementById('editor')!.textContent).toBe('hello')
})
it('submits the owning form when asked', () => {
const holder = page('<form id="f"><input id="q" placeholder="Search" /></form>')
const ref = firstRef(holder)
const form = document.getElementById('f') as HTMLFormElement
form.requestSubmit = vi.fn()
const result = actInPage(document, holder, { kind: 'type', ref, submit: true, text: 'cats' })
expect(form.requestSubmit).toHaveBeenCalledOnce()
expect(result.acted).toContain('submitted')
})
it('refuses a target that has no text to type into', () => {
const holder = page('<button id="b">Press</button>')
expect(actInPage(document, holder, { kind: 'type', ref: firstRef(holder), text: 'x' }).error).toContain(
'not a text field'
)
})
})
describe('press', () => {
it('sends the key to the target', () => {
const holder = page('<input id="q" placeholder="Search" />')
const ref = firstRef(holder)
const keys: string[] = []
document.getElementById('q')!.addEventListener('keydown', e => keys.push((e as KeyboardEvent).key))
expect(actInPage(document, holder, { key: 'Enter', kind: 'press', ref }).success).toBe(true)
expect(keys).toEqual(['Enter'])
})
it('needs a key', () => {
const holder = page('<input id="q" placeholder="Search" />')
expect(actInPage(document, holder, { kind: 'press', ref: firstRef(holder) }).error).toContain('key')
})
})
describe('scroll', () => {
it('scrolls the page by about a screen when given no distance', () => {
const holder = page('<p>long page</p>')
const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {})
const result = actInPage(document, holder, { kind: 'scroll' })
expect(result.success).toBe(true)
expect(scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: Math.round(window.innerHeight * 0.9) })
})
it('drops the animation for a reader who asked for reduced motion', () => {
const holder = page('<p>long page</p>')
const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {})
// Passing 'smooth' explicitly overrides the OS preference, so the engine has
// to opt back out itself rather than leaving it to the browser.
vi.stubGlobal('matchMedia', () => ({ matches: true }))
actInPage(document, holder, { kind: 'scroll' })
vi.unstubAllGlobals()
const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions]
expect(options.behavior).toBe('auto')
})
it('jumps to the bottom', () => {
const holder = page('<p>long page</p>')
const scrollBy = vi.spyOn(window, 'scrollBy').mockImplementation(() => {})
actInPage(document, holder, { kind: 'scroll', to: 'bottom' })
const [options] = scrollBy.mock.calls[0] as unknown as [ScrollToOptions]
expect(options.top).toBeGreaterThan(window.innerHeight)
})
it('scrolls a ref’d container instead of the page', () => {
const holder = page('<div id="list" aria-label="Results" tabindex="0" style="overflow: auto"></div>')
const ref = firstRef(holder)
const list = document.getElementById('list') as HTMLElement
list.scrollBy = vi.fn()
const pageScroll = vi.spyOn(window, 'scrollBy').mockImplementation(() => {})
const result = actInPage(document, holder, { amount: 200, kind: 'scroll', ref })
expect(list.scrollBy).toHaveBeenCalledWith({ behavior: 'smooth', top: 200 })
expect(pageScroll).not.toHaveBeenCalled()
expect(result.acted).toContain('Results')
})
})
@@ -0,0 +1,634 @@
/**
* PREVIEW ACT ENGINE — the one function that performs an interaction inside a
* page, so the agent can drive the in-app browser instead of only reading it.
*
* `elements` hands back an inventory of what can be interacted with, each under
* a durable handle that says what it is — `btn-sign-in`, `inp-email` — and
* parks the matching nodes on a holder. Every other verb resolves its target
* from that holder. Once the agent has a baseline for a page, later looks
* answer with a delta instead of the whole inventory (see `survey`).
*
* SELF-CONTAINMENT, and why the helpers arrive as an argument.
*
* The core is injected into the guest page as source, where module scope does
* not exist: a single free identifier is a ReferenceError on every call, which
* Electron reports only as "Script failed to execute". That much has always
* been true. The subtler half is that the renderer build MINIFIES, so an
* imported binding named here would be renamed to something the injected
* prelude never declared — and it would break in packaged builds only, staying
* green in dev and in tests. So the core names nothing it does not receive:
* `kit` carries the naming, visibility, and identity helpers, and
* `actEngineSource` is what assembles them into one injectable string.
*
* `actInPage` below is the ordinary in-process entry point. It is NOT
* stringified and is free to import.
*/
import { identityKit } from './identity'
import type { IdentityKit } from './identity'
import { namingKit } from './naming'
import type { NamingKit } from './naming'
import type {
PreviewActAction,
PreviewActBinding,
PreviewActDelta,
PreviewActHolder,
PreviewActResult,
PreviewElement,
PreviewElementChange
} from './types'
import { visibilityKit } from './visibility'
import type { VisibilityKit } from './visibility'
export type {
PreviewActAction,
PreviewActBinding,
PreviewActDelta,
PreviewActHolder,
PreviewActResult,
PreviewElement,
PreviewElementChange
} from './types'
/** Everything the core needs that it cannot define for itself. */
export interface ActKit {
identity: IdentityKit
naming: NamingKit
visibility: VisibilityKit
}
/** Assemble the kits for one document. Used in-process; the guest page gets
* the same three factories through `actEngineSource`. */
export function buildActKit(doc: Document): ActKit {
const naming = namingKit(doc)
return { identity: identityKit(naming), naming, visibility: visibilityKit(doc, doc.defaultView) }
}
/** Run one action against `doc`, resolving refs through `holder`. */
export function actInPage(doc: Document, holder: PreviewActHolder, action: PreviewActAction): PreviewActResult {
return actInPageCore(doc, holder, action, buildActKit(doc))
}
/** The injectable core. Everything it touches is a parameter — see the
* self-containment note above. */
export function actInPageCore(
doc: Document,
holder: PreviewActHolder,
action: PreviewActAction,
kit: ActKit
): PreviewActResult {
const { anchorOf, labelOf, selectorFor, stableOf, valueOf } = kit.naming
const { affinity, coin, shifted } = kit.identity
const { onScreen, visible } = kit.visibility
// Declared inside, not at module scope: this body is stringified and eval'd
// in the guest page, where a module-level constant is simply not defined.
const maxElements = 120
// How many nodes the OVERLAY may draw, as opposed to how many the agent gets
// told about. Far higher, because an extra mark costs one rect read where an
// extra inventory row costs tokens on every single call.
const maxMarks = 600
// How alike a remembered element and a fresh one have to be before the
// handle moves across. Below it we mint a new handle instead: a re-render
// costing the agent a re-read is a cheap mistake, and a handle silently
// pointing at the wrong button is not.
const rebindBar = 0.6
const win = doc.defaultView
const here = doc.location ? doc.location.href : ''
// Passing 'smooth' explicitly OVERRIDES the user's OS-level reduce-motion
// setting, so ask before animating over someone who opted out.
const still = !!(win && win.matchMedia && win.matchMedia('(prefers-reduced-motion: reduce)').matches)
const glide: ScrollBehavior = still ? 'auto' : 'smooth'
/** Walk the page and hand back what is interactable, in document order. The
* handles are assigned afterwards, by `survey`. */
const sight = (max: number) => {
const nodes: Element[] = []
const field: Element[] = []
const elements: PreviewElement[] = []
const candidates = doc.querySelectorAll(
'a[href], button, input:not([type="hidden"]), select, textarea, summary, label[for], ' +
'[role="button"], [role="link"], [role="checkbox"], [role="radio"], [role="tab"], ' +
'[role="menuitem"], [role="switch"], [role="option"], [role="combobox"], [role="searchbox"], ' +
'[role="textbox"], [contenteditable=""], [contenteditable="true"], [onclick], ' +
'[tabindex]:not([tabindex="-1"])'
)
for (const el of candidates) {
if (elements.length >= max && field.length >= maxMarks) {
break
}
if (!visible(el)) {
continue
}
// Drawable from here on. The two lists diverge deliberately: the field is
// what is on screen to be outlined, the inventory is what the agent can
// name and reach — which includes things below the fold it will scroll to.
if (field.length < maxMarks && onScreen(el)) {
field.push(el)
}
if (elements.length >= max) {
continue
}
const tag = el.tagName.toLowerCase()
const role = el.getAttribute('role') || (tag === 'input' ? 'input:' + ((el as HTMLInputElement).type || 'text') : tag)
const label = labelOf(el)
const value = valueOf(el)
// A control with neither a label nor a value is not addressable in prose
// — the agent could not tell it apart from its unlabelled neighbours.
// It is also the one case a durable handle cannot be minted for, since
// there would be nothing to name it after and nothing to re-find it by,
// so dropping it here keeps every handle we DO mint anchorable.
if (!label && !value) {
continue
}
const entry: PreviewElement = {
label,
// Filled in by `survey`, which is what knows whether this element
// already has a handle.
ref: '',
role
}
const selector = selectorFor(el)
if (selector) {
entry.selector = selector
}
if ((el as HTMLInputElement).disabled) {
entry.disabled = true
}
if (value) {
entry.value = value
}
nodes.push(el)
elements.push(entry)
}
holder.nodes = nodes
holder.field = field
return { elements, nodes }
}
/** Look at the page and say what is there — or, once there is something to
* compare against, only what moved.
*
* Re-sending the whole inventory every action is what took a ten-step
* session from 45k to 85k tokens of context: the page barely changes between
* a scroll and a click, and the agent was being charged for a fresh copy of
* it each time. */
const survey = (max: number): PreviewActResult => {
// A navigation is a different page. Every handle on the old one is retired
// rather than rebound onto whatever now sits in the same place.
const fresh = holder.url !== here
if (fresh) {
holder.book = []
holder.coined = {}
}
const book = holder.book || (holder.book = [])
const coined = holder.coined || (holder.coined = {})
const seen = sight(max)
const claimed: Record<string, boolean> = {}
const kept: PreviewActBinding[] = []
const added: PreviewElement[] = []
const changed: PreviewElementChange[] = []
const rebound: string[] = []
const known = new Map<Element, PreviewActBinding>()
const waiting: number[] = []
let same = 0
for (const bound of book) {
known.set(bound.el, bound)
}
// Pass one: the element object itself is still the one we remember. Free,
// and it is what happens on a scroll, a hover, and most clicks.
for (let i = 0; i < seen.elements.length; i++) {
const entry = seen.elements[i]
const bound = known.get(seen.nodes[i])
if (!bound) {
waiting.push(i)
continue
}
const moved = shifted(bound, entry)
entry.ref = bound.ref
claimed[bound.ref] = true
kept.push(bound)
if (!moved) {
same++
continue
}
bound.label = entry.label
bound.name = entry.label || entry.value || ''
bound.off = !!entry.disabled
bound.value = entry.value || ''
changed.push(moved)
}
// Pass two: whatever is left either replaced something (a framework threw
// the node away and built a new one) or is genuinely new. The pool is only
// the handles whose element is GONE, which is both the correct candidate
// set and a small one.
const pool = book.filter(bound => !claimed[bound.ref] && !doc.contains(bound.el))
for (const i of waiting) {
const entry = seen.elements[i]
const el = seen.nodes[i]
const now: PreviewActBinding = {
el,
label: entry.label,
name: entry.label || entry.value || '',
off: !!entry.disabled,
path: anchorOf(el),
ref: '',
role: entry.role,
stable: stableOf(el),
value: entry.value || ''
}
let best: PreviewActBinding | undefined
let score = 0
for (const bound of pool) {
if (claimed[bound.ref]) {
continue
}
const rung = affinity(bound, now)
// Strictly better, so a tie goes to whichever candidate the page put
// first and the same page twice re-binds the same way.
if (rung >= rebindBar && rung > score) {
best = bound
score = rung
}
}
if (best) {
// Same handle, new node. Reported as one word rather than a removal
// and an addition, because from the agent's side nothing happened —
// its handle still works and it has nothing to re-read.
entry.ref = best.ref
best.el = el
best.label = now.label
best.name = now.name
best.off = now.off
best.path = now.path
best.stable = now.stable
best.value = now.value
claimed[best.ref] = true
kept.push(best)
rebound.push(best.ref)
continue
}
now.ref = coin(coined, entry.role, now.name)
entry.ref = now.ref
claimed[now.ref] = true
kept.push(now)
added.push(entry)
}
// A handle nobody claimed is gone ONLY if its element really left. One that
// is still on the page but fell past `max` keeps working and is simply not
// mentioned — saying "removed" about something the agent can still click
// would be worse than saying nothing.
const removed: string[] = []
for (const bound of book) {
if (claimed[bound.ref]) {
continue
}
if (doc.contains(bound.el) && visible(bound.el)) {
kept.push(bound)
continue
}
removed.push(bound.ref)
}
holder.book = kept
holder.url = here
// The delta has to actually be cheaper. When half the page is new there is
// nothing to reuse, and a delta is then just the inventory with extra
// framing around it.
const churn = added.length + changed.length
if (fresh || action.full || churn * 2 >= seen.elements.length) {
return { elements: seen.elements, success: true }
}
const delta: PreviewActDelta = { same }
if (added.length) {
delta.added = added
}
if (changed.length) {
delta.changed = changed
}
if (removed.length) {
delta.removed = removed
}
if (rebound.length) {
delta.rebound = rebound
}
return { delta, success: true }
}
/** Resolve the action's target: a handle from the book, else a selector. */
const resolve = (): { el?: Element; error?: string } => {
const ref = (action.ref || '').trim()
if (ref) {
if (holder.url !== here) {
return { error: 'The page navigated since the last snapshot, so ' + ref + ' no longer points anywhere. Call elements again.' }
}
const bound = (holder.book || []).filter(entry => entry.ref === ref)[0]
if (!bound) {
return { error: 'Unknown element ' + ref + '. Call elements to get current refs.' }
}
if (!doc.contains(bound.el)) {
return { error: ref + ' has been removed from the page since the last snapshot. Call elements again.' }
}
return { el: bound.el }
}
const selector = (action.selector || '').trim()
if (!selector) {
return { error: 'Pass a ref from elements, or a CSS selector.' }
}
let el: Element | null = null
try {
el = doc.querySelector(selector)
} catch {
return { error: 'Not a valid CSS selector: ' + selector }
}
return el ? { el } : { error: 'No element matches ' + selector + '.' }
}
const describe = (el: Element) => {
const label = labelOf(el)
return el.tagName.toLowerCase() + (label ? ' "' + label + '"' : '')
}
const answer = (result: PreviewActResult): PreviewActResult => ({
...result,
title: doc.title || '',
url: doc.location ? doc.location.href : ''
})
const fail = (error: string) => answer({ error, success: false })
/** Center of `el` in viewport coords, for pointer events that read position. */
const pointAt = (el: Element) => {
const rect = el.getBoundingClientRect()
return { clientX: Math.round(rect.left + rect.width / 2), clientY: Math.round(rect.top + rect.height / 2) }
}
if (action.kind === 'elements') {
const looked = survey(Math.max(1, Math.min(action.max || maxElements, maxElements)))
const empty = !looked.delta && !(looked.elements || []).length
return answer({
...looked,
note: empty ? 'No interactive elements found — the page may still be loading.' : undefined
})
}
if (action.kind === 'scroll') {
const target = action.ref || action.selector ? resolve() : {}
if (target.error) {
return fail(target.error)
}
const scroller = (target.el as HTMLElement | undefined) || null
const page = Math.round((win ? win.innerHeight : 800) * 0.9)
const by = action.to === 'top' ? -1e7 : action.to === 'bottom' ? 1e7 : (action.amount ?? page)
if (scroller) {
scroller.scrollBy({ behavior: glide, top: by })
} else if (win) {
win.scrollBy({ behavior: glide, top: by })
}
return answer({ acted: scroller ? 'scrolled ' + describe(scroller) : 'scrolled the page', success: true })
}
const target = resolve()
if (target.error || !target.el) {
return fail(target.error || 'No target.')
}
const el = target.el as HTMLElement
// Park the target where the watch overlay can find it, before anything can
// fail — a ring around the thing that turned out to be disabled is exactly
// the feedback someone watching wants.
holder.aimed = el
// 'locate' is the look-before-you-act half of an action: it brings the target
// on screen and names it, so the overlay has something to draw while the real
// verb is still a beat away.
if (action.kind === 'locate') {
if (el.scrollIntoView) {
el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' })
}
if (action.focus) {
el.focus()
}
const spot = pointAt(el)
const tag = el.tagName
return answer({
acted: 'looking at ' + describe(el),
point: { x: spot.clientX, y: spot.clientY },
success: true,
// Real typing starts with a triple-click to clear the field. On anything
// that is not a field that gesture selects the paragraph under it
// instead, which is how the agent ended up highlighting whole pages.
typable: tag === 'TEXTAREA' || tag === 'INPUT' || el.isContentEditable === true
})
}
// Instant, unlike the `scroll` verb above. Getting a target on screen is
// plumbing for the click that follows, not something anyone asked to watch —
// and every millisecond of it is time the caller spends waiting for the page
// to stop moving before it can aim real input at a fixed coordinate.
if (el.scrollIntoView) {
el.scrollIntoView({ behavior: 'instant', block: 'center', inline: 'nearest' })
}
// Hovering a disabled control is allowed — a disabled button with a tooltip
// explaining WHY is exactly the thing worth hovering.
if (action.kind === 'hover') {
const init = { bubbles: true, cancelable: true, ...pointAt(el) }
if (typeof PointerEvent === 'function') {
el.dispatchEvent(new PointerEvent('pointerover', init))
el.dispatchEvent(new PointerEvent('pointermove', init))
}
el.dispatchEvent(new MouseEvent('mouseover', init))
el.dispatchEvent(new MouseEvent('mousemove', init))
return answer({ acted: 'hovered over ' + describe(el), success: true })
}
if ((el as HTMLInputElement).disabled) {
return fail(describe(el) + ' is disabled.')
}
if (action.kind === 'click') {
const init = { bubbles: true, cancelable: true, ...pointAt(el) }
// Frameworks bind to the pointer/mouse pair as often as to click itself, so
// replay the whole sequence; el.click() then runs the native activation
// (following a link, toggling a checkbox, submitting a form) that a bare
// synthetic MouseEvent would leave to chance.
if (typeof PointerEvent === 'function') {
el.dispatchEvent(new PointerEvent('pointerdown', init))
el.dispatchEvent(new PointerEvent('pointerup', init))
}
el.dispatchEvent(new MouseEvent('mousedown', init))
el.dispatchEvent(new MouseEvent('mouseup', init))
el.click()
return answer({ acted: 'clicked ' + describe(el), success: true })
}
if (action.kind === 'type') {
const text = action.text ?? ''
const editable = el.isContentEditable || (el.getAttribute('contenteditable') ?? 'false') !== 'false'
el.focus()
if (editable) {
el.textContent = text
} else if (el.tagName === 'INPUT' || el.tagName === 'TEXTAREA') {
// Assign through the prototype's setter: React (and anything else that
// tracks the DOM value) shadows `value` with its own accessor and ignores
// an input event whose value it thinks it already wrote, so a plain
// `el.value = …` types into a field that snaps back on the next render.
const proto = el.tagName === 'TEXTAREA' ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype
const setter = Object.getOwnPropertyDescriptor(proto, 'value')?.set
if (setter) {
setter.call(el, text)
} else {
;(el as HTMLInputElement).value = text
}
} else if (el.tagName === 'SELECT') {
return fail(describe(el) + ' is a dropdown — click it and click the option you want.')
} else {
return fail(describe(el) + ' is not a text field.')
}
el.dispatchEvent(new Event('input', { bubbles: true }))
el.dispatchEvent(new Event('change', { bubbles: true }))
if (action.submit) {
const enter = { bubbles: true, cancelable: true, code: 'Enter', key: 'Enter' }
el.dispatchEvent(new KeyboardEvent('keydown', enter))
el.dispatchEvent(new KeyboardEvent('keyup', enter))
const form = (el as HTMLInputElement).form
if (form) {
if (form.requestSubmit) {
form.requestSubmit()
} else {
form.submit()
}
}
}
return answer({ acted: 'typed into ' + describe(el) + (action.submit ? ' and submitted' : ''), success: true })
}
if (action.kind === 'press') {
const key = action.key || ''
if (!key) {
return fail('Pass the key to press, e.g. "Enter" or "Escape".')
}
const init = { bubbles: true, cancelable: true, code: key.length === 1 ? 'Key' + key.toUpperCase() : key, key }
el.focus()
el.dispatchEvent(new KeyboardEvent('keydown', init))
el.dispatchEvent(new KeyboardEvent('keyup', init))
return answer({ acted: 'pressed ' + key + ' on ' + describe(el), success: true })
}
return fail('Unknown action: ' + String(action.kind))
}
/** The engine as one injectable expression: `function (doc, holder, action)`.
*
* The four factories are stringified together so the core's `kit` is built
* inside the guest page, out of sources that travelled with it. This is the
* ONLY supported way to get the engine's source — taking `.toString()` of any
* one piece yields a function whose helpers are not defined, which fails at
* call time rather than at injection. */
export function actEngineSource(): string {
return `(function (doc, holder, action) {
var naming = (${namingKit.toString()})(doc);
var kit = {
naming: naming,
identity: (${identityKit.toString()})(naming),
visibility: (${visibilityKit.toString()})(doc, doc.defaultView)
};
return (${actInPageCore.toString()})(doc, holder, action, kit);
})`
}
@@ -0,0 +1,129 @@
/**
* IDENTITY — deciding whether the element in front of us is one we already
* have a handle on, and naming it if it is not.
*
* This is what makes a delta possible at all. If handles churned every time a
* framework re-rendered, every look at the page would be a wall of removals and
* additions and there would be nothing to send but the whole inventory again.
*
* The re-bind ladder is ported from anchortree (Apache-2.0), minus its geometry
* rung. It is deliberately conservative: below the bar we mint a new handle
* instead of guessing, because a re-render costing the agent a re-read is a
* cheap mistake and a handle silently pointing at the wrong button is not.
*
* SELF-CONTAINMENT: see `naming.ts`. Same contract, same reason for the
* factory shape — with one extra consequence. `identityKit` takes the naming
* helpers as an argument rather than importing them, because an import would
* be a cross-module identifier the bundler is free to rename out from under
* the stringified body.
*/
import type { NamingKit } from './naming'
import type { PreviewActBinding, PreviewElement, PreviewElementChange } from './types'
export interface IdentityKit {
/** How strongly a remembered element matches one just observed, 0 to 1. */
affinity(was: PreviewActBinding, now: PreviewActBinding): number
/** Do two labels share at least half their words? */
alike(a: string, b: string): boolean
/** Mint a handle, disambiguating against the ones already minted. */
coin(coined: Record<string, number>, role: string, name: string): string
/** What moved on an element that kept its handle, or null if it held still. */
shifted(was: PreviewActBinding, entry: PreviewElement): null | PreviewElementChange
}
/** Build the identity helpers on top of a naming kit. */
export function identityKit(naming: NamingKit): IdentityKit {
/** What moved on an element that kept its handle, or nothing if it held
* still. Field by field, so a status line ticking over costs the agent one
* short line instead of a re-run of everything already known about it. */
const shifted = (was: PreviewActBinding, entry: PreviewElement): null | PreviewElementChange => {
const off = !!entry.disabled
const value = entry.value || ''
if (was.label === entry.label && was.value === value && was.off === off) {
return null
}
const moved: PreviewElementChange = { ref: was.ref }
if (was.label !== entry.label) {
moved.label = entry.label
}
if (was.value !== value) {
moved.value = value
}
if (was.off !== off) {
moved.disabled = off
}
return moved
}
/** Do two labels share at least half their words? Tolerates the count badge
* and the copy edit — "Inbox" against "Inbox (3)". */
const alike = (a: string, b: string): boolean => {
if (!a || !b) {
return false
}
const one = a.toLowerCase().split(/\s+/).filter(Boolean)
const two = b.toLowerCase().split(/\s+/).filter(Boolean)
const both = one.filter(word => two.indexOf(word) !== -1).length
const all = one.length + two.filter(word => one.indexOf(word) === -1).length
return all > 0 && both / all >= 0.5
}
/** How strongly a remembered element matches one just observed, 0 to 1.
*
* Ported from anchortree's re-bind ladder (Apache-2.0), minus its geometry
* rung: a centroid is only ever worth 0.1 there, it never reaches the 0.6
* bar on its own, and carrying coordinates through the book to buy a
* tie-break is not worth the measurement. */
const affinity = (was: PreviewActBinding, now: PreviewActBinding): number => {
// A button is not a link, however alike the rest of it reads.
if (was.role !== now.role) {
return 0
}
// Two elements that BOTH carry a stable attribute and disagree are the page
// telling us outright that they are different things.
if (was.stable && now.stable) {
return was.stable === now.stable ? 1 : 0
}
let score = 0
if (was.name && was.name === now.name) {
score += 0.6
} else if (alike(was.name, now.name)) {
score += 0.4
}
if (was.path && was.path === now.path) {
score += 0.3
}
return score
}
/** Mint a handle. Legible on purpose: the agent reads `btn-sign-in` in a
* three-line delta on turn nine and knows what it is, where `@e42` would
* send it back to an inventory twenty thousand tokens ago. Suffixes are
* never rewound, so a retired handle's name is not later handed to a
* different element on the same page. */
const coin = (coined: Record<string, number>, role: string, name: string): string => {
const named = naming.slug(name)
const stem = naming.stemOf(role) + (named ? '-' + named : '')
const nth = coined[stem] || 0
coined[stem] = nth + 1
return nth ? stem + '-' + nth : stem
}
return { affinity, alike, coin, shifted }
}
@@ -0,0 +1,249 @@
import { beforeEach, describe, expect, it } from 'vitest'
import { identityKit } from './identity'
import { namingKit } from './naming'
import type { PreviewActBinding } from './types'
const naming = () => namingKit(document)
function el(html: string): Element {
document.body.innerHTML = html
return document.body.firstElementChild!
}
/** A remembered element, with only the fields a given test cares about set. */
function bound(over: Partial<PreviewActBinding>): PreviewActBinding {
return {
el: document.body,
label: '',
name: '',
off: false,
path: 'root',
ref: 'btn-x',
role: 'button',
stable: '',
value: '',
...over
}
}
beforeEach(() => {
document.body.innerHTML = ''
})
describe('slug', () => {
it('reads as words a person would recognise', () => {
expect(naming().slug('Sign In')).toBe('sign-in')
expect(naming().slug(' Add to cart! ')).toBe('add-to-cart')
})
it('collapses runs of punctuation rather than stacking hyphens', () => {
expect(naming().slug('Save — and / continue')).toBe('save-and-continue')
})
it('stays short enough to read at a glance, and never ends on a hyphen', () => {
const long = naming().slug('Subscribe to our weekly newsletter for updates')
expect(long.length).toBeLessThanOrEqual(24)
expect(long.endsWith('-')).toBe(false)
})
it('has nothing to say about an unlabelled element', () => {
expect(naming().slug('!!!')).toBe('')
})
})
describe('stemOf', () => {
// The stem is the part of a handle that survives translation: whatever the
// button says, `btn-` tells the agent it is a button.
it('gives each interactive role its own prefix', () => {
const { stemOf } = naming()
expect(stemOf('button')).toBe('btn')
expect(stemOf('a')).toBe('lnk')
expect(stemOf('input:search')).toBe('srch')
expect(stemOf('input:checkbox')).toBe('chk')
expect(stemOf('select')).toBe('sel')
})
it('treats every other kind of text field as one thing', () => {
const { stemOf } = naming()
expect(stemOf('input:email')).toBe('inp')
expect(stemOf('input:date')).toBe('inp')
expect(stemOf('textbox')).toBe('inp')
})
it('falls back rather than inventing a prefix for something unknown', () => {
expect(naming().stemOf('marquee')).toBe('el')
})
})
describe('labelOf', () => {
it('prefers what the page tells assistive clients', () => {
expect(naming().labelOf(el('<button aria-label="Close dialog">×</button>'))).toBe('Close dialog')
})
it('follows aria-labelledby to the element that names it', () => {
document.body.innerHTML = '<h2 id="t">Billing</h2><button aria-labelledby="t">→</button>'
expect(naming().labelOf(document.querySelector('button')!)).toBe('Billing')
})
it('falls back through placeholder and title before giving up', () => {
expect(naming().labelOf(el('<input placeholder="Your email" />'))).toBe('Your email')
expect(naming().labelOf(el('<a href="/x" title="Help"></a>'))).toBe('Help')
expect(naming().labelOf(el('<button></button>'))).toBe('')
})
it('flattens the whitespace a formatted template leaves behind', () => {
expect(naming().labelOf(el('<button>\n Save\n changes\n</button>'))).toBe('Save changes')
})
})
describe('selectorFor', () => {
// Identity only. A positional chain was 74% of the inventory and wrong the
// moment a sibling appeared.
it('names the element only when the page gave it a durable name', () => {
const { selectorFor } = naming()
expect(selectorFor(el('<button id="save">S</button>'))).toBe('#save')
expect(selectorFor(el('<button data-testid="save">S</button>'))).toBe('[data-testid="save"]')
expect(selectorFor(el('<button class="btn primary">S</button>'))).toBe('')
})
})
describe('valueOf', () => {
it('reports what a control currently holds', () => {
const input = el('<input value="shoes" />')
expect(naming().valueOf(input)).toBe('shoes')
})
// The DOM hands back "on" for an unset checkbox value, so reading the value
// first made both states identical to the agent.
it('reports a checkbox as its state rather than the value the DOM invents', () => {
const box = el('<input type="checkbox" />') as HTMLInputElement
expect(naming().valueOf(box)).toBe('unchecked')
box.checked = true
expect(naming().valueOf(box)).toBe('checked')
})
it('still reports a checkbox state when the page did set a value', () => {
const box = el('<input type="checkbox" value="weekly" />') as HTMLInputElement
box.checked = true
expect(naming().valueOf(box)).toBe('checked')
})
it('leaves a radio group readable the same way', () => {
const radio = el('<input type="radio" name="plan" value="pro" />') as HTMLInputElement
expect(naming().valueOf(radio)).toBe('unchecked')
})
})
describe('anchorOf', () => {
// Coarse on purpose: a wrapper div appearing in the chain must not change
// this string, because that churn is exactly what a re-bind sees through.
it('places an element by the landmark it sits in', () => {
document.body.innerHTML = '<nav><div><div><button>Home</button></div></div></nav>'
expect(naming().anchorOf(document.querySelector('button')!)).toBe('nav')
})
it('tells two same-tag landmarks apart by their label', () => {
document.body.innerHTML = '<nav aria-label="Primary nav"><button>Home</button></nav>'
expect(naming().anchorOf(document.querySelector('button')!)).toBe('nav#primary-nav')
})
it('says so when there is no landmark to place it in', () => {
expect(naming().anchorOf(el('<button>Home</button>'))).toBe('root')
})
})
describe('alike', () => {
const { alike } = identityKit(naming())
it('sees through a count badge and a copy edit', () => {
expect(alike('Inbox', 'Inbox (3)')).toBe(true)
expect(alike('Save changes', 'Save your changes')).toBe(true)
})
it('does not call two unrelated labels the same thing', () => {
expect(alike('Sign in', 'Delete account')).toBe(false)
expect(alike('Save', '')).toBe(false)
})
})
describe('affinity', () => {
const { affinity } = identityKit(naming())
it('is certain when both sides carry the same stable attribute', () => {
expect(affinity(bound({ stable: 'save' }), bound({ stable: 'save' }))).toBe(1)
})
// The page is telling us outright that these are different things.
it('refuses outright when two stable attributes disagree', () => {
expect(affinity(bound({ stable: 'save' }), bound({ stable: 'cancel' }))).toBe(0)
})
it('refuses to move a handle across roles, however alike the rest reads', () => {
const was = bound({ name: 'Save', role: 'button' })
const now = bound({ name: 'Save', role: 'a' })
expect(affinity(was, now)).toBe(0)
})
it('clears the bar on name and place together, and not on place alone', () => {
const named = affinity(bound({ name: 'Save', path: 'main' }), bound({ name: 'Save', path: 'main' }))
const placed = affinity(bound({ name: 'Save', path: 'main' }), bound({ name: 'Publish', path: 'main' }))
expect(named).toBeGreaterThanOrEqual(0.6)
expect(placed).toBeLessThan(0.6)
})
})
describe('coin', () => {
const mint = () => {
const { coin } = identityKit(naming())
const coined: Record<string, number> = {}
return (role: string, name: string) => coin(coined, role, name)
}
it('names a handle after what the thing is and what it says', () => {
const coin = mint()
expect(coin('button', 'Sign in')).toBe('btn-sign-in')
expect(coin('input:email', 'Email')).toBe('inp-email')
})
it('tells repeats apart without renaming the first one', () => {
const coin = mint()
expect(coin('button', 'Edit')).toBe('btn-edit')
expect(coin('button', 'Edit')).toBe('btn-edit-1')
expect(coin('button', 'Edit')).toBe('btn-edit-2')
})
// Never rewound, so a retired handle's name is not later handed to a
// different element on the same page.
it('does not reissue a name after the element holding it goes away', () => {
const coin = mint()
coin('button', 'Edit')
coin('button', 'Edit')
expect(coin('button', 'Edit')).toBe('btn-edit-2')
})
it('still mints something for a role it has no name for', () => {
expect(mint()('button', '')).toBe('btn')
})
})
+214
View File
@@ -0,0 +1,214 @@
/**
* NAMING — what an element is called, and what it is called by.
*
* Everything here turns an element into a short string: the accessible name,
* the durable handle's stem and slug, the landmark it sits under, the identity
* selector. No page state, no holder, no side effects.
*
* SELF-CONTAINMENT. `namingKit` is stringified into the guest page along with
* the engine that uses it (see `act-in-page.ts`'s contract). Module scope does
* not exist there, so the factory closes over nothing but its own argument and
* declares every helper inside itself. A single free identifier is a
* ReferenceError on every call, and Electron reports it only as "Script failed
* to execute".
*
* That constraint is also why this is a factory rather than nine exports: nine
* exports would be nine names for the bundler to mangle independently of the
* call sites inside the stringified engine. One function that stringifies whole
* has no cross-module references to break.
*/
export interface NamingKit {
/** Where the element sits, coarsely — the nearest landmark. */
anchorOf(el: Element): string
clamp(text: string, max: number): string
cssEscape(value: string): string
/** The accessible name, by the usual ladder, truncated. */
labelOf(el: Element): string
/** An `#id` or `[data-testid]`, or empty when the page offers neither. */
selectorFor(el: Element): string
/** Lowercase, hyphenated, and short enough to read at a glance. */
slug(name: string): string
/** The strongest identity signal the page offers, if it offers one. */
stableOf(el: Element): string
/** The handle stem for a role: `btn`, `inp`, `lnk` … */
stemOf(role: string): string
/** A form control's current value, or its checked state. */
valueOf(el: Element): string
}
/** Build the naming helpers against one document. */
export function namingKit(doc: Document): NamingKit {
const cssEscape = (value: string) =>
typeof CSS !== 'undefined' && CSS.escape ? CSS.escape(value) : value.replace(/["\\]/g, '\\$&')
const clamp = (text: string, max: number) => (text.length > max ? text.slice(0, max - 1) + '…' : text)
const labelOf = (el: Element): string => {
const aria = el.getAttribute('aria-label')
if (aria) {
return clamp(aria, 80)
}
const labelledBy = el.getAttribute('aria-labelledby')
const labelled = labelledBy ? doc.getElementById(labelledBy) : null
const text = ((labelled || el).textContent || '').trim().replace(/\s+/g, ' ')
if (text) {
return clamp(text, 80)
}
for (const attr of ['placeholder', 'title', 'alt', 'name', 'value']) {
const value = el.getAttribute(attr)
if (value) {
return clamp(value, 80)
}
}
return ''
}
/** The strongest identity signal the page offers, if it offers one. Nothing a
* re-render does to the surrounding markup disturbs these. */
const stableOf = (el: Element): string =>
el.id ||
el.getAttribute('data-testid') ||
el.getAttribute('name') ||
el.getAttribute('aria-label') ||
''
/** Handle stems by role, so a handle says what it is before it says which
* one. Anything unrecognised is `el`. */
const stemOf = (role: string): string => {
if (role === 'button' || role === 'summary') {
return 'btn'
}
if (role === 'a' || role === 'link') {
return 'lnk'
}
if (role === 'input:search' || role === 'searchbox') {
return 'srch'
}
if (role === 'input:checkbox' || role === 'checkbox') {
return 'chk'
}
if (role === 'input:radio' || role === 'radio') {
return 'rdo'
}
if (role === 'select' || role === 'combobox') {
return 'sel'
}
if (role === 'textarea') {
return 'txt'
}
if (role === 'switch') {
return 'sw'
}
if (role === 'tab' || role === 'menuitem' || role === 'option') {
return role === 'menuitem' ? 'mi' : role === 'option' ? 'opt' : 'tab'
}
// Every `input:*` that isn't one of the special cases above, plus the ARIA
// textbox. A date picker and an email field are both places text goes.
return role.indexOf('input') === 0 || role === 'textbox' ? 'inp' : 'el'
}
/** Lowercase, hyphenated, and short enough to read at a glance. */
const slug = (name: string): string => {
let out = ''
let dash = false
for (let i = 0; i < name.length && out.length < 24; i++) {
const ch = name[i]
if (/[a-zA-Z0-9]/.test(ch)) {
out += ch.toLowerCase()
dash = false
} else if (!dash && out) {
out += '-'
dash = true
}
}
return out.replace(/-+$/, '')
}
/** Where the element sits, coarsely: the nearest landmark plus its position
* among same-role elements inside it. Deliberately NOT the CSS selector
* below — a wrapper div appearing anywhere in the chain changes that string
* completely, which is exactly the churn a re-bind has to see through. */
const anchorOf = (el: Element): string => {
const near = el.closest(
'main,nav,header,footer,aside,[role="main"],[role="navigation"],[role="banner"],' +
'[role="contentinfo"],[role="complementary"],[role="search"],form[aria-label],section[aria-label]'
)
if (!near) {
return 'root'
}
const named = near.getAttribute('aria-label') || ''
return near.tagName.toLowerCase() + (named ? '#' + slug(named) : '')
}
/** The element's own selector, when the page gives it one worth having.
*
* Deliberately identity-only. This used to fall back to a chain of up to
* eight `:nth-child` rungs, and on a real app shell that column was 74% of
* the entire inventory — the single biggest thing the agent was paying for.
* It bought nothing: nothing downstream reads it, a positional chain is
* wrong the moment a sibling appears, and re-finding a node is what the
* durable ref now does properly. An `#id` is short, stable, and the one
* case where naming the node is genuinely useful to the model. */
const selectorFor = (el: Element): string => {
if (el.id) {
return '#' + cssEscape(el.id)
}
const testId = el.getAttribute('data-testid')
return testId ? '[data-testid="' + cssEscape(testId) + '"]' : ''
}
const valueOf = (el: Element): string => {
const control = el as HTMLInputElement
// Tickable controls answer with their state, and are asked FIRST: a
// checkbox's `.value` is the string "on" unless the page sets one, so
// reading the value first made a ticked box and an empty one identical to
// the agent — the one thing about a checkbox worth reporting. Gated on the
// input's TYPE, not on `checked` being defined, which it is (as false) on
// every input including text fields.
const kind = (control.type || '').toLowerCase()
if (kind === 'checkbox' || kind === 'radio') {
return control.checked ? 'checked' : 'unchecked'
}
// The same control built out of a div, where the state lives in ARIA.
const role = el.getAttribute('role') || ''
if (role === 'checkbox' || role === 'radio' || role === 'switch') {
return el.getAttribute('aria-checked') === 'true' ? 'checked' : 'unchecked'
}
if (typeof control.value === 'string' && control.value) {
return clamp(control.value, 60)
}
return ''
}
return { anchorOf, clamp, cssEscape, labelOf, selectorFor, slug, stableOf, stemOf, valueOf }
}
+157
View File
@@ -0,0 +1,157 @@
/**
* PREVIEW ACT TYPES — the shapes the engine, the overlay, and the Python tool
* all agree on.
*
* Declarations only. They erase at compile time, which is what lets the engine
* modules import them while still stringifying into the guest page as plain JS
* (see `naming.ts` for the self-containment contract they have to satisfy).
*
* Re-exported from `act-in-page.ts`, so every existing import site keeps
* working and nothing outside this directory needs to know they moved.
*/
/** One interactable node, as the agent sees it. */
export interface PreviewElement {
/** Present and true only when the control is non-interactive right now. */
disabled?: boolean
/** Human-readable label (aria-label, text, placeholder, value …). */
label: string
/** Durable handle for as long as this page is open: 'btn-sign-in'. Legible on
* purpose — see the ref-minting note in `actInPage`. */
ref: string
/** Explicit ARIA role, else the tag name. */
role: string
/** The element's `#id` or `[data-testid]`, when it has one. Absent otherwise
* — address the element by its `ref`. */
selector?: string
/** Current value of a form control, truncated. */
value?: string
}
/** An element that is still itself but no longer reads the same.
*
* Only the fields that actually moved are present. Role and selector are
* absent by construction rather than by omission: a change in either would
* mean this is a different element, which the re-bind ladder would have
* refused to match in the first place. */
export interface PreviewElementChange {
/** Present only when the control's availability flipped. */
disabled?: boolean
label?: string
ref: string
value?: string
}
/** What changed on the page since the last look. Sent instead of the whole
* inventory once the agent has a baseline for the page — see `survey`. */
export interface PreviewActDelta {
/** Elements seen for the first time, in full. */
added?: PreviewElement[]
/** Same handle, new label/value/disabled state — and nothing else. */
changed?: PreviewElementChange[]
/** Handles that are gone from the page. */
removed?: string[]
/** Handles whose element was destroyed and recreated by a re-render. The
* handle still works; nothing about them needs re-reading. */
rebound?: string[]
/** How many handles were on the page and untouched. */
same?: number
}
/** A normalized action. `kind` is the verb; the rest is per-verb payload. */
export interface PreviewActAction {
/** scroll distance in px. Defaults to ~90% of the viewport height. */
amount?: number
key?: string
/** `pin`/`unpin`/`hold` never reach the engine — they resolve their targets
* through `locate`/`elements` and then talk to the overlay — but they arrive
* on the same wire. */
kind:
| 'click'
| 'elements'
| 'hold'
| 'hover'
| 'locate'
| 'pin'
| 'press'
| 'scroll'
| 'strobe'
| 'type'
| 'unpin'
/** locate: also give the target keyboard focus, for a key press that must not
* be preceded by a click (which would activate the control instead). */
focus?: boolean
/** elements: answer with the whole inventory rather than a delta. */
full?: boolean
/** Cap on the returned inventory. */
max?: number
ref?: string
selector?: string
/** type: press Enter (and submit the owning form) after entering text. */
submit?: boolean
text?: string
to?: 'bottom' | 'top'
}
export interface PreviewActResult {
/** What the action landed on, for the agent's own log. */
acted?: string
/** What moved since the last look. Present INSTEAD of `elements` once the
* agent holds a baseline for this page. */
delta?: PreviewActDelta
/** The full inventory. Sent on the first look at a page, and again whenever
* the page changed too much for a delta to be the cheaper answer. */
elements?: PreviewElement[]
error?: string
note?: string
/** Viewport centre of a located target, for aiming real pointer input at it. */
point?: { x: number; y: number }
success: boolean
title?: string
/** locate: whether the target actually takes typed text. */
typable?: boolean
/** Live document URL after the action — a change means it navigated. */
url?: string
}
/** One element the agent has a handle on, remembered across actions. */
export interface PreviewActBinding {
el: Element
/** What it read as last time. Kept field by field rather than as one hash so
* a change can be reported as only the part that moved. */
label: string
/** The accessible name at mint time, for re-finding this element after a
* re-render destroys and recreates its node. */
name: string
/** Whether the control was unavailable last time. */
off: boolean
/** Nearest-landmark path plus position among same-role siblings. */
path: string
ref: string
role: string
/** `id` / `name` / `data-testid` / `aria-label`, if the page provides one.
* The strongest re-bind signal there is, and the only one a rewrite of the
* surrounding markup cannot disturb. */
stable: string
value: string
}
/** Where the surface keeps what it knows between actions (a window global in
* the preview page), so a handle still means something on the next call. */
export interface PreviewActHolder {
/** Target of the action in flight, for the watch overlay to draw onto. */
aimed?: Element | null
/** Every handle minted on this page, live or not yet retired. */
book?: PreviewActBinding[]
/** Next disambiguating suffix per ref stem, so two "Edit" buttons become
* `btn-edit` and `btn-edit-1`. Never rewound: a retired handle's name is
* not handed to a different element later in the same page. */
coined?: Record<string, number>
/** The on-screen subset, for the overlay to outline. Diverges from `nodes` in
* both directions: it drops what is below the fold, and it is not capped at
* the inventory's size. */
field?: Element[]
nodes?: Element[]
/** URL the snapshot was taken on; a navigation retires every handle. */
url?: string
}
@@ -0,0 +1,148 @@
/**
* VISIBILITY — is this element actually there, and is it on screen.
*
* Two different questions with two different answers. `visible` decides whether
* an element belongs in the inventory at all: painted, not screen-reader-only,
* not buried under something else. `onScreen` decides whether the overlay
* should draw a box around it, which is a narrower question — the inventory
* includes things below the fold the agent will scroll to, and the overlay
* does not.
*
* Most of what is here exists because of a specific class of thing that looks
* interactive and is not: the sr-only recipe in both its spellings, a control
* inside an `aria-hidden` subtree, a 1px box parked at the document origin, an
* element under a cookie wall. Each one of those, admitted, is an action the
* agent takes that visibly does nothing.
*
* SELF-CONTAINMENT: see `naming.ts`. Same contract, same reason for the
* factory shape.
*/
export interface VisibilityKit {
/** In the viewport right now, and not a page-sized wrapper. */
onScreen(el: Element): boolean
/** Painted at all, ancestors included. */
shown(el: Element): boolean
/** Painted, reachable, big enough to aim at, and not buried. */
visible(el: Element): boolean
}
/** Build the visibility tests against one document and its view. */
export function visibilityKit(doc: Document, win: null | Window): VisibilityKit {
/** On screen right now, and worth drawing a box around. The field is strictly
* viewport-bound: a mark past the fold is one nobody can see, it spends the
* budget that should have gone to what IS on screen, and the ones parked far
* off to the side were landing as stray labels in the corner. */
const onScreen = (el: Element): boolean => {
if (!win) {
return false
}
const box = el.getBoundingClientRect()
if (box.right <= 0 || box.bottom <= 0 || box.left >= win.innerWidth || box.top >= win.innerHeight) {
return false
}
// Page-sized wrappers. Outlining one draws a rectangle around the whole
// screen, which says nothing and frames everything inside it as if it
// mattered. Full-width banners are fine — it takes both dimensions.
return !(box.width >= win.innerWidth * 0.95 && box.height >= win.innerHeight * 0.9)
}
/** Painted at all, ancestors included. */
const shown = (el: Element): boolean => {
// Folds in display/visibility/opacity/content-visibility inherited from an
// ANCESTOR, which this element's own computed style does not report: inside
// a parent at opacity 0, every child still reads back opacity 1.
const seen = (el as Element & { checkVisibility?: (opts: object) => boolean }).checkVisibility
if (typeof seen === 'function' && !seen.call(el, { checkOpacity: true, checkVisibilityCSS: true })) {
return false
}
const style = win && win.getComputedStyle ? win.getComputedStyle(el) : null
if (style) {
if (style.display === 'none' || style.visibility === 'hidden' || style.opacity === '0') {
return false
}
// The screen-reader-only recipe, both spellings: the deprecated `clip` and
// the modern `clip-path: inset(50%)`. Neither exists for any purpose other
// than hiding something visually while keeping it focusable.
if ((style.clip && style.clip !== 'auto') || (style.clipPath || '').indexOf('inset(50%') !== -1) {
return false
}
}
return true
}
const visible = (el: Element): boolean => {
if (!shown(el)) {
return false
}
// Things the page itself declares are not for interacting with, neither of
// which shows up in a computed style or a bounding box: an aria-hidden
// subtree is invisible to every other assistive client, and `inert` cannot
// be clicked at all. Disabled controls are deliberately NOT filtered here —
// they stay in the inventory so the agent is told a button is disabled
// rather than hunting for one that appears not to exist.
if (el.closest('[aria-hidden="true"], [inert]')) {
return false
}
const rect = el.getBoundingClientRect()
// Below the fold is fine — we scroll to it. Off to the LEFT or ABOVE is the
// other half of the sr-only trick (`left: -9999px`, `top: -9999px`) and
// scrolling never brings those back.
if (rect.right <= 0 || rect.bottom <= 0) {
return false
}
// Bigger than the 1px box sr-only collapses to. Nothing a person can aim at
// is 2px across, and admitting those is what put the cursor in the corner:
// they cluster at the document origin, so they sort to the FRONT of the
// inventory and the agent reaches for one as the first thing on the page.
if (rect.width < 3 || rect.height < 3) {
return false
}
// Occlusion, and the last filter for a reason: everything above is cheap
// and this one forces layout. If the middle of the element is on screen,
// whatever the browser reports at that point has to BE this element — its
// own descendant (an icon inside a link) or the box it paints inside are
// fine, anything else means it is buried under a sticky header, a cookie
// wall, or a transparent layer the page put on top. Those are exactly the
// targets the agent aims at and then misses. Elements below the fold cannot
// be hit-tested from here, so they are taken on trust and land back in this
// function after the scroll.
const midX = rect.left + rect.width / 2
const midY = rect.top + rect.height / 2
const under = (doc as Document & { elementFromPoint?: (x: number, y: number) => Element | null })
.elementFromPoint
if (
typeof under === 'function' &&
win &&
midX >= 0 &&
midY >= 0 &&
midX < win.innerWidth &&
midY < win.innerHeight
) {
const over = under.call(doc, midX, midY)
if (!over || !(over === el || el.contains(over) || over.contains(el))) {
return false
}
}
return true
}
return { onScreen, shown, visible }
}
@@ -0,0 +1,363 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { type WatchHolder, watchInPage } from './watch-in-page'
/** jsdom has no layout engine, so every rect is zero and nothing would be
* positioned. Give elements a plausible box the way the act-engine tests do. */
beforeEach(() => {
// Never run the callback: the tracking loop re-arms itself every frame, so a
// stub that invokes synchronously recurses until the stack gives out. The
// first pass happens on the direct call, which is the one worth asserting on.
vi.stubGlobal('requestAnimationFrame', () => 1)
vi.stubGlobal('cancelAnimationFrame', () => {})
Element.prototype.getBoundingClientRect = function () {
return { bottom: 140, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 100, width: 200, x: 100, y: 100 }
}
document.body.replaceChildren()
// The host hangs off documentElement, so clearing body does not reach it.
document.querySelector('hermes-watch')?.remove()
delete (window as unknown as Record<string, unknown>).__hermesWatch
})
/** The overlay lives in a closed shadow root, so a test can only see the host. */
const host = () => document.querySelector('hermes-watch')
/** …except through the parts the engine parks on the window for its own reuse,
* which is the only way to inspect what the overlay actually drew. */
const drawn = () =>
(window as unknown as { __hermesWatch: { parts: Record<string, HTMLElement> & { shadow: ShadowRoot } } })
.__hermesWatch.parts
/** Everything the overlay drew for this action. The cursor is the only fixed
* layer — every box and pin is a mark that comes and goes. */
const marks = () => [...drawn().shadow.children].filter(child => child !== drawn().pointer)
describe('watchInPage', () => {
it('mounts one host on documentElement and reuses it across stages', () => {
const target = document.createElement('button')
document.body.append(target)
const holder: WatchHolder = { aimed: target }
watchInPage(document, holder, 'aim')
watchInPage(document, holder, 'strike')
expect(document.querySelectorAll('hermes-watch')).toHaveLength(1)
expect(host()?.parentElement).toBe(document.documentElement)
})
it('stays out of the page, out of the way, and out of the a11y tree', () => {
const target = document.createElement('button')
document.body.append(target)
watchInPage(document, { aimed: target }, 'aim')
const node = host() as HTMLElement
// pointer-events keeps real clicks reaching the page; aria-hidden keeps a
// screen reader from announcing decoration.
expect(node.style.pointerEvents).toBe('none')
expect(node.getAttribute('aria-hidden')).toBe('true')
// The closed shadow root is what keeps the overlay out of the act engine's
// own querySelectorAll inventory — there is nothing to exclude by hand.
expect(node.shadowRoot).toBeNull()
})
it('does nothing when there is no target to draw on', () => {
watchInPage(document, {}, 'aim')
watchInPage(document, { aimed: document.createElement('div') }, 'strike')
expect(host()).toBeNull()
})
// The cursor is placed, never interpolated. Beyond reading as a machine, a
// transition here was an outright bug on the first paint: an unset transform
// IS the origin, so the browser flew the cursor in from the top-left corner —
// and every navigation is a fresh document, so a fresh first paint.
it('cuts the marks straight to the target instead of travelling to it', () => {
const target = document.createElement('button')
document.body.append(target)
watchInPage(document, { aimed: target }, 'aim', 'Clicking Save')
const { pointer } = drawn()
expect(pointer.style.transition).toBeFalsy()
expect((marks()[0] as HTMLElement).style.transition).toBeFalsy()
// Mocked rect is 200×40 at (100, 100), so the cursor belongs at its centre.
expect(pointer.style.transform).toBe('translate3d(200px,120px,0)')
})
// The inventory pass is the one moment the agent is reading the whole page
// rather than aiming at one thing, so the overlay has to be able to hold many
// marks at once — not just the single ring every other stage draws.
it('sweep lights up every catalogued element and holds them well past the sweep', () => {
vi.useFakeTimers()
const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')]
document.body.append(...found)
watchInPage(document, { nodes: found }, 'sweep')
expect(marks()).toHaveLength(found.length)
// The point of the dwell: the grid is still up long after the pass that drew
// it finished, so there is something to actually look at. jsdom has no Web
// Animations, so this exercises the timed fallback, which holds for the same
// beat as the animated path.
vi.advanceTimersByTime(600)
expect(marks()).toHaveLength(found.length)
// …and then drains scattered rather than blinking off as one block, which is
// the whole reason each box carries its own offset. Stepping through the
// exit must therefore see the count come down in more than one tick.
const drain = new Set<number>()
for (let t = 0; t < 7_000; t += 100) {
vi.advanceTimersByTime(100)
drain.add(marks().length)
}
expect(drain.size).toBeGreaterThan(2)
expect(marks()).toHaveLength(0)
vi.useRealTimers()
})
// jsdom cannot run a Web Animation, but it can be handed one and asked what
// it was given — and this particular option is worth pinning, because getting
// it wrong is invisible rather than broken. `easing` in the options bag is the
// ITERATION easing: it rewrites progress before any keyframe is consulted, so
// a step function there holds progress at 0 for the whole duration. Every box
// was created, animated and removed without painting a single frame.
it('does not step the sweep’s iteration easing, which would stop it painting', () => {
const timings: KeyframeAnimationOptions[] = []
Element.prototype.animate = vi.fn(function (this: Element, _frames, options) {
timings.push(options as KeyframeAnimationOptions)
return { cancel: () => {}, finish: () => {} } as unknown as Animation
}) as unknown as Element['animate']
const found = [document.createElement('button'), document.createElement('a')]
document.body.append(...found)
watchInPage(document, { field: found }, 'sweep')
// The cursor's idle sway is permanent and starts with the overlay, so it is
// not one of the sweep's own animations.
const sweep = timings.filter(timing => timing.iterations !== Infinity)
expect(sweep).toHaveLength(found.length)
for (const timing of sweep) {
expect(String(timing.easing)).not.toMatch(/steps/)
}
})
it('names each swept box after the element under it, not its ref', () => {
const found = [document.createElement('button'), document.createElement('a'), document.createElement('input')]
found[0].id = 'buy'
found[1].className = 'nav wide'
found[2].setAttribute('type', 'search')
document.body.append(...found)
// Every drawable box gets named, addressable or not — a ref number only ever
// meant something for the subset that had one, and said nothing about what
// the element actually is.
watchInPage(document, { field: found, nodes: found.slice(0, 2) }, 'sweep')
const { shadow } = drawn()
const cells = [...shadow.children].slice(-found.length)
expect(cells.map(cell => cell.textContent)).toEqual(['button#buy', 'a.nav', 'input[search]'])
})
// The case that made positions necessary: a list of links with no id, no
// class and no type. Naming them by tag alone gives a whole page of boxes all
// reading "a", which is what the ref numbers were replaced for.
it('falls back to position so anonymous siblings are still told apart', () => {
const row = document.createElement('td')
row.className = 'subtext'
row.innerHTML = '<a></a><a></a><a></a>'
document.body.append(row)
const found = [...row.children]
watchInPage(document, { field: found }, 'sweep')
const { shadow } = drawn()
const cells = [...shadow.children].slice(-found.length)
expect(cells.map(cell => cell.textContent)).toEqual(['a[1]', 'a[2]', 'a[3]'])
})
it('keeps a name short enough to sit on the box it names', () => {
const found = [document.createElement('div')]
found[0].id = 'a-genuinely-enormous-generated-identifier'
document.body.append(...found)
watchInPage(document, { field: found }, 'sweep')
const name = marks()[0].textContent as string
expect(name.length).toBeLessThanOrEqual(16)
expect(name.startsWith('div#a-')).toBe(true)
expect(name.endsWith('…')).toBe(true)
})
// A field past a couple of dozen boxes is a texture, not a list. Naming every
// one buries the outlines — which ARE the readout at that size — under a wall
// of tags that collide with each other and overhang their own rectangles.
it('stops naming the boxes once the field is too dense to read', () => {
const few = Array.from({ length: 4 }, () => document.createElement('button'))
document.body.append(...few)
watchInPage(document, { field: few }, 'sweep')
expect(marks().every(mark => mark.textContent)).toBe(true)
document.querySelector('hermes-watch')?.remove()
delete (window as unknown as Record<string, unknown>).__hermesWatch
const many = Array.from({ length: 40 }, () => document.createElement('button'))
document.body.append(...many)
watchInPage(document, { field: many }, 'sweep')
expect(marks().some(mark => mark.textContent)).toBe(false)
})
// Every other mark on the page wears its label on the top edge. A box against
// the top of the viewport has nowhere to put one, and a tab drawn off-screen
// is a box that looks unlabelled next to boxes that aren't.
it('flips the label under the box when there is no headroom above it', () => {
const roomy = document.createElement('button')
const flush = document.createElement('button')
const huge = document.createElement('button')
flush.getBoundingClientRect = () =>
({ bottom: 40, height: 40, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect
// Taller than the viewport: no room above it and none below it either, so
// the tab has nowhere left to go but inside.
huge.getBoundingClientRect = () =>
({ bottom: 900, height: 900, left: 100, right: 300, toJSON: () => ({}), top: 0, width: 200, x: 100, y: 0 }) as DOMRect
document.body.append(roomy, flush, huge)
watchInPage(document, { field: [roomy, flush, huge] }, 'sweep')
const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement)
expect(tags[0].style.bottom).toBe('100%')
expect(tags[1].style.top).toBe('100%')
expect(tags[2].style.top).toBe('0px')
})
// A label is routinely wider than the box it names, so it is the boxes near the
// right edge that push their tab off screen. Whichever end it lands on, the
// tab's edge sits exactly on the outline's — the two are the same colour and
// meet at a corner, so a pixel of overhang reads as a misprint.
it('hangs the label off the right edge when a left-anchored one would run off', () => {
Object.defineProperty(HTMLElement.prototype, 'offsetWidth', { configurable: true, get: () => 80 })
const inboard = document.createElement('button')
const edge = document.createElement('button')
edge.getBoundingClientRect = () =>
({ bottom: 140, height: 40, left: 990, right: 1020, toJSON: () => ({}), top: 100, width: 30, x: 990, y: 100 }) as DOMRect
document.body.append(inboard, edge)
watchInPage(document, { field: [inboard, edge] }, 'sweep')
const tags = marks().map(mark => mark.querySelector('div:last-child') as HTMLElement)
expect(tags[0].style.left).toBe('0px')
expect(tags[0].style.right).toBe('')
expect(tags[1].style.right).toBe('0px')
expect(tags[1].style.left).toBe('')
})
it('sweep draws nothing when the inventory came back empty', () => {
watchInPage(document, { nodes: [] }, 'sweep')
expect(host()).toBeNull()
})
// The field is swept after every action, and a sweep retires the lock. The
// cursor must not go with it — a hand that blinks out a fraction of a second
// after every single step reads as the agent having wandered off.
it('keeps the cursor up when a sweep retires the lock', () => {
const target = document.createElement('button')
const found = [document.createElement('a')]
document.body.append(target, ...found)
watchInPage(document, { aimed: target }, 'aim')
const { pointer } = drawn()
const lock = marks()[0]
expect(pointer.style.opacity).toBe('1')
watchInPage(document, { field: found }, 'sweep')
expect(lock.isConnected).toBe(false)
expect(pointer.style.opacity).toBe('1')
})
// The chrome is memoized for the life of the page, so a style change to it
// would otherwise be injected, run, and restyle nothing — the layers it
// describes were built by the version before. Reads as "HMR isn't picking my
// CSS up" in dev, and as a long-lived tab stuck on an old build in production.
it('rebuilds its chrome when the injected source changes underneath it', () => {
const target = document.createElement('button')
document.body.append(target)
const stamp = (value: number) => {
;(window as unknown as Record<string, unknown>).__hermesWatchTag = value
}
stamp(1)
watchInPage(document, { aimed: target }, 'aim')
const first = drawn().host
stamp(1)
watchInPage(document, { aimed: target }, 'aim')
expect(drawn().host).toBe(first)
stamp(2)
watchInPage(document, { aimed: target }, 'aim')
expect(drawn().host).not.toBe(first)
// …and the old one goes with it, rather than stacking a second overlay.
expect(document.querySelectorAll('hermes-watch')).toHaveLength(1)
})
it('clear releases the target so the tracking loop can stop', () => {
const target = document.createElement('button')
document.body.append(target)
const holder: WatchHolder = { aimed: target }
watchInPage(document, holder, 'aim')
watchInPage(document, holder, 'clear')
expect(holder.aimed).toBeNull()
})
// Same contract as the act engine: the pane injects `watchInPage.toString()`
// into the guest page, where module scope does not exist. One free identifier
// is a ReferenceError that Electron reports only as "Script failed to execute".
it('runs after being stringified and eval’d with no module scope', () => {
const target = document.createElement('button')
document.body.append(target)
const injected = new Function('return (' + watchInPage.toString() + ')')() as typeof watchInPage
expect(() => injected(document, { aimed: target }, 'aim', 'Clicking Save')).not.toThrow()
expect(host()).not.toBeNull()
})
})
File diff suppressed because it is too large Load Diff
+2
View File
@@ -470,6 +470,7 @@ class AIAgent:
clarify_callback: callable = None,
read_terminal_callback: callable = None,
read_preview_callback: callable = None,
drive_preview_callback: callable = None,
read_window_below_callback: callable = None,
setup_mcp_callback: callable = None,
tour_callback: callable = None,
@@ -560,6 +561,7 @@ class AIAgent:
clarify_callback=clarify_callback,
read_terminal_callback=read_terminal_callback,
read_preview_callback=read_preview_callback,
drive_preview_callback=drive_preview_callback,
read_window_below_callback=read_window_below_callback,
setup_mcp_callback=setup_mcp_callback,
tour_callback=tour_callback,
+10
View File
@@ -2421,6 +2421,8 @@ class TestAgentRuntimePostHookOwnershipSync:
("clarify", {"question": "Continue?"}),
("read_terminal", {}),
("read_preview", {}),
("drive_preview", {"action": "elements"}),
("annotate_preview", {"action": "clear"}),
("read_window_below", {}),
("setup_mcp", {"server": "linear", "action": "install"}),
("tour", {"action": "stop"}),
@@ -2467,6 +2469,14 @@ class TestAgentRuntimePostHookOwnershipSync:
"tools.read_preview_tool.read_preview_tool",
lambda **kwargs: '{"ok":true}',
)
monkeypatch.setattr(
"tools.drive_preview_tool.drive_preview_tool",
lambda **kwargs: '{"ok":true}',
)
monkeypatch.setattr(
"tools.annotate_preview_tool.annotate_preview_tool",
lambda **kwargs: '{"ok":true}',
)
monkeypatch.setattr(
"tools.read_window_tool.read_window_below_tool",
lambda **kwargs: '{"ok":true}',
+88
View File
@@ -0,0 +1,88 @@
"""Tests for the GUI-surface ``annotate_preview`` tool."""
import json
import pytest
from tools import annotate_preview_tool as an
from tools.registry import registry
def test_lives_in_the_gui_surface_toolset(monkeypatch):
"""Scoped by toolset, not by the backend's env — same as its siblings."""
monkeypatch.delenv("HERMES_DESKTOP", raising=False)
entry = registry.get_entry("annotate_preview")
assert entry is not None
assert entry.toolset == "desktop_ui"
assert entry.check_fn is None
def test_requires_callback():
result = json.loads(an.annotate_preview_tool(ref="@e1", callback=None))
assert "desktop" in result["error"]
def test_rejects_an_unknown_action():
result = json.loads(an.annotate_preview_tool(action="scribble", callback=lambda _p: "{}"))
assert "action must be one of" in result["error"]
@pytest.mark.parametrize("verb", ["add", "remove"])
def test_marking_one_thing_needs_a_target(verb):
"""A mark with nothing to attach to is a mistake worth naming early."""
result = json.loads(an.annotate_preview_tool(action=verb, callback=lambda _p: "{}"))
assert "ref" in result["error"]
def test_speaks_the_renderer_s_verbs():
"""The overlay knows pin/unpin; the model gets words it can guess at."""
sent = []
def cb(payload):
sent.append(payload)
return json.dumps({"acted": "pinned Buy now", "success": True})
an.annotate_preview_tool(action="add", ref="@e4", label="cheapest", callback=cb)
an.annotate_preview_tool(action="remove", ref="@e4", callback=cb)
assert sent[0] == {"action": "pin", "ref": "@e4", "text": "cheapest"}
assert sent[1] == {"action": "unpin", "ref": "@e4"}
def test_clear_sends_no_target_so_the_overlay_drops_them_all():
"""`clear` IS an untargeted unpin — a ref left on it would take down one."""
sent = []
def cb(payload):
sent.append(payload)
return json.dumps({"acted": "cleared every pin", "success": True})
an.annotate_preview_tool(action="clear", ref="@e4", selector="a", callback=cb)
assert sent == [{"action": "unpin"}]
def test_defaults_to_adding():
sent = []
def cb(payload):
sent.append(payload)
return json.dumps({"success": True})
an.annotate_preview_tool(ref="@e2", callback=cb)
assert sent[0]["action"] == "pin"
def test_passes_the_renderer_s_answer_straight_through():
out = json.loads(
an.annotate_preview_tool(ref="@e1", callback=lambda _p: json.dumps({"acted": "pinned Save", "success": True}))
)
assert out == {"acted": "pinned Save", "success": True}
def test_reports_a_silent_bridge():
result = json.loads(an.annotate_preview_tool(ref="@e1", callback=lambda _p: ""))
assert "open_preview" in result["error"]
+128
View File
@@ -0,0 +1,128 @@
"""Tests for the GUI-surface ``drive_preview`` tool."""
import json
from tools import drive_preview_tool as ap
from tools.registry import registry
def test_lives_in_the_gui_surface_toolset(monkeypatch):
"""Mirrors read_preview: scoped by toolset, not by the backend's env."""
monkeypatch.delenv("HERMES_DESKTOP", raising=False)
entry = registry.get_entry("drive_preview")
assert entry is not None
assert entry.toolset == "desktop_ui"
assert entry.check_fn is None
def test_requires_callback():
"""Outside the desktop GUI there is no bridge — a clear error, no crash."""
result = json.loads(ap.drive_preview_tool(action="elements", callback=None))
assert "desktop" in result["error"]
def test_rejects_an_unknown_action():
result = json.loads(ap.drive_preview_tool(action="teleport", callback=lambda _p: "{}"))
assert "action must be one of" in result["error"]
def test_interaction_verbs_need_a_target():
"""A click with nowhere to land is a mistake worth naming before the bridge."""
for verb in ("click", "type", "press"):
result = json.loads(ap.drive_preview_tool(action=verb, text="x", key="Enter", callback=lambda _p: "{}"))
assert "ref" in result["error"], verb
def test_type_needs_text_and_press_needs_a_key():
calls = []
def cb(payload):
calls.append(payload)
return json.dumps({"success": True})
assert "text" in json.loads(ap.drive_preview_tool(action="type", ref="@e1", callback=cb))["error"]
assert "key" in json.loads(ap.drive_preview_tool(action="press", ref="@e1", callback=cb))["error"]
assert calls == []
def test_typing_an_empty_string_is_allowed():
"""Clearing a field is a real intent — only a missing `text` is an error."""
seen = {}
ap.drive_preview_tool(action="type", ref="@e1", text="", callback=lambda p: seen.update(p) or "{}")
assert seen["text"] == ""
def test_scroll_needs_no_target_and_validates_its_destination():
seen = {}
def cb(payload):
seen.clear()
seen.update(payload)
return json.dumps({"success": True})
ap.drive_preview_tool(action="scroll", callback=cb)
assert seen == {"action": "scroll"}
result = json.loads(ap.drive_preview_tool(action="scroll", to="sideways", callback=cb))
assert "to must be one of" in result["error"]
def test_payload_forwards_only_what_was_given():
seen = {}
ap.drive_preview_tool(
action="type",
ref="inp-password",
text="hunter2",
submit=True,
callback=lambda p: seen.update(p) or json.dumps({"success": True}),
)
assert seen == {"action": "type", "ref": "inp-password", "text": "hunter2", "submit": True}
def test_full_asks_the_renderer_for_a_whole_inventory():
seen = {}
ap.drive_preview_tool(
action="elements",
full=True,
callback=lambda p: seen.update(p) or json.dumps({"success": True}),
)
assert seen == {"action": "elements", "full": True}
def test_numeric_arguments_are_validated():
result = json.loads(ap.drive_preview_tool(action="scroll", amount="lots", callback=lambda _p: "{}"))
assert "integers" in result["error"]
def test_empty_answer_means_nothing_open():
result = json.loads(ap.drive_preview_tool(action="elements", callback=lambda _p: ""))
assert "open_preview" in result["error"]
def test_passes_the_renderer_answer_through():
payload = {
"success": True,
"acted": 'clicked button "Sign in"',
"url": "https://example.com/app",
"elements": [{"ref": "@e1", "role": "button", "label": "Log out", "selector": "#out"}],
}
result = json.loads(ap.drive_preview_tool(action="click", ref="@e2", callback=lambda _p: json.dumps(payload)))
assert result == payload
def test_wraps_non_json_text():
result = json.loads(ap.drive_preview_tool(action="elements", callback=lambda _p: "plain words"))
assert result == {"text": "plain words"}
def test_callback_failure_is_reported():
def _boom(_payload):
raise RuntimeError("renderer went away")
result = json.loads(ap.drive_preview_tool(action="elements", callback=_boom))
assert "renderer went away" in result["error"]
@@ -18,7 +18,9 @@ import tui_gateway.server as server
from toolsets import TOOLSETS, resolve_toolset
GUI_TOOLS = {
"annotate_preview",
"close_preview",
"drive_preview",
"close_terminal",
"focus_pane",
"open_preview",
+151
View File
@@ -0,0 +1,151 @@
#!/usr/bin/env python3
"""Leave a mark on the page in the Hermes desktop GUI's in-app browser.
``drive_preview`` already draws every move it makes — the field it can reach,
a box round its target, the cursor going there — but those are transients:
each one stands for a single action and retires itself. That is right for
narrating a click and no use at all for holding a finding on screen.
This is the deliberate one. An annotation outlines an element — or, with
``hold``, the entire visible field at once — and stays until the agent takes it
down, so it can show the user what it found, flag the fields
it is about to fill, or keep its place while it works elsewhere on the page.
Named for TouchDesigner's Annotate — the labelled box you drop around part of a
network to call it out.
Annotations are bound to elements, not coordinates: they ride scrolls and
reflows, and they go when their element does, so a navigation clears them
without the agent having to.
Rides the same ``preview.act`` bridge as ``drive_preview`` rather than opening
a second channel — the renderer already resolves ``@e`` refs and owns the
overlay, so this is one more verb on a wire that exists.
Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for
desktop-sourced sessions.
"""
import json
from typing import Callable, Optional
from tools.registry import registry, tool_error
ACTIONS = ("add", "hold", "remove", "clear")
# Verbs the renderer knows, keyed by ours. `clear` is `unpin` with nothing to
# aim at, which the overlay reads as "all of them".
WIRE = {"add": "pin", "hold": "hold", "remove": "unpin", "clear": "unpin"}
def annotate_preview_tool(
action: str = "add",
ref: Optional[str] = None,
selector: Optional[str] = None,
label: Optional[str] = None,
callback: Optional[Callable] = None,
) -> str:
"""Put one annotation up, take one down, or clear them all."""
if callback is None:
return tool_error("annotate_preview is only available in the Hermes desktop app.")
verb = (action or "add").strip().lower()
if verb not in ACTIONS:
return tool_error(f"action must be one of: {', '.join(ACTIONS)}.")
if verb in ("add", "remove") and not (ref or selector):
return tool_error(
f"{verb} needs a ref from drive_preview action='elements' "
"(e.g. 'btn-sign-in') or a CSS selector."
)
payload = {
name: val
for name, val in (
("action", WIRE[verb]),
("ref", None if verb in ("clear", "hold") else ref),
("selector", None if verb in ("clear", "hold") else selector),
("text", label),
)
if val is not None
}
try:
raw = callback(payload)
except Exception as exc:
return tool_error(f"Failed to annotate the in-app browser: {exc}")
if not raw:
return tool_error(
"The annotation timed out, or no GUI window answered. "
"Open a page with open_preview first."
)
try:
return json.dumps(json.loads(raw), ensure_ascii=False)
except (TypeError, ValueError):
return json.dumps({"text": str(raw)}, ensure_ascii=False)
ANNOTATE_PREVIEW_SCHEMA = {
"name": "annotate_preview",
"description": (
"Draw a lasting mark on the page open in the in-app browser / preview "
"pane of the Hermes desktop GUI. Everything drive_preview draws as it "
"works fades on its own; an annotation STAYS until you remove it, so "
"this is how you point at something. Use it to show the user what you "
"found ('here are the three cheapest'), flag what you are about to "
"change before you change it, or keep your place while you work "
"elsewhere on the page. Address elements by the same refs "
"drive_preview action='elements' hands back. action='add' outlines an "
"element and gives it an optional short label; 'hold' freezes the WHOLE "
"visible field at once — every element the page offers, outlined and "
"named — which is the "
"picture drive_preview flashes as it works, made to stay; 'remove' "
"takes one down; 'clear' takes them all down. Annotations follow their element "
"as the page scrolls and disappear if it does, so a navigation clears "
"them for you. Keep labels to a word or two — they are drawn on the "
"page, not read aloud."
),
"parameters": {
"type": "object",
"properties": {
"action": {
"type": "string",
"enum": list(ACTIONS),
"description": (
"'add' marks one element, 'hold' freezes the whole visible "
"field, 'remove' takes one down, 'clear' takes them all "
"down. Defaults to 'add'."
),
},
"ref": {
"type": "string",
"description": "Element reference from drive_preview action='elements' (e.g. 'btn-sign-in').",
},
"selector": {
"type": "string",
"description": "CSS selector, as a fallback when no ref fits. Prefer ref.",
},
"label": {
"type": "string",
"description": "Short caption drawn on the mark, e.g. 'cheapest'. Optional.",
},
},
"required": [],
},
}
registry.register(
name="annotate_preview",
toolset="desktop_ui",
schema=ANNOTATE_PREVIEW_SCHEMA,
handler=lambda args, **kw: annotate_preview_tool(
action=args.get("action", "add"),
ref=args.get("ref"),
selector=args.get("selector"),
label=args.get("label"),
callback=kw.get("callback"),
),
emoji="🔖",
)
+234
View File
@@ -0,0 +1,234 @@
#!/usr/bin/env python3
"""Interact with the in-app browser / preview pane in the Hermes desktop GUI.
``open_preview`` shows a page and ``read_preview`` reads it; this tool is the
third leg — clicking, typing, scrolling, and history — so the agent can drive
the same page the user is looking at instead of narrating from the outside.
Elements are addressed by refs from ``action="elements"`` that say what they
are: ``btn-sign-in``, ``inp-email``. A ref lasts as long as the page is open,
including across a re-render that destroys and rebuilds the element, and only a
navigation retires it — the renderer says so rather than acting on whatever now
occupies the spot.
Because the refs hold, the renderer answers with a *delta* — what appeared,
what went, what changed, and what was rebound — instead of re-sending the whole
inventory after every click. That is the cheap half of the arrangement, and it
only works because the refs are legible enough to read on their own three turns
later.
Round-trips through the gateway's blocking-prompt bridge like ``read_preview``:
tui_gateway emits ``preview.act.request``, the renderer injects the interaction
engine into the pane's webview and answers ``preview.act.respond`` with the
outcome plus whatever moved. This module is just schema + a thin dispatcher
over the platform-injected callback.
Lives in the ``desktop_ui`` toolset, which the GUI gateway enables only for
desktop-sourced sessions.
"""
import json
from typing import Callable, Optional
from tools.registry import registry, tool_error
ACTIONS = (
"elements",
"click",
"hover",
"type",
"scroll",
"press",
"strobe",
"back",
"forward",
"reload",
)
SCROLL_TO = ("top", "bottom")
# Verbs that need something to act on — a ref from the last inventory, or a
# raw CSS selector. `scroll` is deliberately absent: bare, it scrolls the page.
NEEDS_TARGET = ("click", "hover", "type", "press")
def drive_preview_tool(
action: str = "",
ref: Optional[str] = None,
selector: Optional[str] = None,
text: Optional[str] = None,
key: Optional[str] = None,
submit: Optional[bool] = None,
amount: Optional[int] = None,
to: Optional[str] = None,
limit: Optional[int] = None,
full: Optional[bool] = None,
callback: Optional[Callable] = None,
) -> str:
"""Dispatch one interaction to the desktop renderer and return its outcome."""
if callback is None:
return tool_error("drive_preview is only available in the Hermes desktop app.")
verb = (action or "").strip().lower()
if verb not in ACTIONS:
return tool_error(f"action must be one of: {', '.join(ACTIONS)}.")
if verb in NEEDS_TARGET and not (ref or selector):
return tool_error(
f"{verb} needs a ref from action='elements' (e.g. 'btn-sign-in') or a CSS selector."
)
if verb == "type" and text is None:
return tool_error("type needs the text to enter.")
if verb == "press" and not key:
return tool_error("press needs a key, e.g. 'Enter' or 'Escape'.")
if to is not None and to not in SCROLL_TO:
return tool_error(f"to must be one of: {', '.join(SCROLL_TO)}.")
try:
payload = {
name: val
for name, val in (
("action", verb),
("ref", ref),
("selector", selector),
("text", text),
("key", key),
("submit", submit),
("full", full),
("to", to),
("amount", None if amount is None else int(amount)),
("max", None if limit is None else int(limit)),
)
if val is not None
}
except (TypeError, ValueError):
return tool_error("amount and max must be integers.")
try:
raw = callback(payload)
except Exception as exc:
return tool_error(f"Failed to act on the in-app browser: {exc}")
if not raw:
return tool_error(
"The action timed out, or no GUI window answered. "
"Open a page with open_preview first."
)
# The renderer answers with a JSON object; pass it through, else wrap it.
try:
return json.dumps(json.loads(raw), ensure_ascii=False)
except (TypeError, ValueError):
return json.dumps({"text": str(raw)}, ensure_ascii=False)
ACT_PREVIEW_SCHEMA = {
"name": "drive_preview",
"description": (
"Interact with the page open in the in-app browser / preview pane of "
"the Hermes desktop GUI — the pane open_preview opens beside this "
"chat. This is how you USE a web app the user is looking at: log in, "
"fill a form, click through a flow, page a long document. ALWAYS call "
"action='elements' first to get the current inventory of clickable and "
"typable things — each carries a ref like 'btn-sign-in' or 'inp-email' "
"plus its role, label, and value — then act with that ref instead of "
"guessing a selector. A ref keeps working for as long as the page is "
"open, INCLUDING across a re-render that rebuilds the element, so hold "
"onto the ones you were given. "
"Every action answers with the live url/title plus what moved: the "
"first look at a page returns the full 'elements' inventory, and after "
"that you get a 'delta' instead — 'added' entries in full, 'changed' "
"entries carrying only the ref and whichever of label/value/disabled "
"actually moved, 'removed' and 'rebound' as bare ref lists, and 'same' "
"counting the refs that held. A 'rebound' ref needs NO action from you; it "
"means the page rebuilt that element and your ref already follows it. "
"Anything not mentioned in a delta is unchanged, so do not re-read the "
"page to check. Only a navigation invalidates refs; when told they are "
"stale, call elements again. The mouse "
"and keyboard are real: the pointer travels to its target and the page "
"sees genuine input, so hover menus open and hover-only controls work. "
"Actions: 'elements' (inventory), 'click', 'hover' (move the pointer "
"onto something and leave it there — use it to open a dropdown or "
"reveal a tooltip before clicking inside it), 'type' (set a field's "
"text; submit=true also presses Enter and submits the form), 'scroll' "
"(the page, or a ref'd scrollable), 'press' (a named key), 'strobe' "
"(touch nothing — just rattle the highlight through the page again; "
"'elements' already does this once, so reach for it only when asked to "
"flick, flash, or bounce around the page some more, and note one call "
"runs a whole multi-second burst, so never loop it per element), and "
"'back'/'forward'/'reload' for history. The pane draws every move as "
"it happens so the user can follow along; those marks fade on their "
"own, and annotate_preview is how you leave one up on purpose. Use "
"read_preview when you only "
"need the page's text, and the browser_* tools when the work belongs "
"in a separate automated browser rather than the user's own pane."
),
"parameters": {
"type": "object",
"properties": {
"action": {
"type": "string",
"enum": list(ACTIONS),
"description": "What to do. Start with 'elements'.",
},
"ref": {
"type": "string",
"description": "Element reference from any earlier elements call (e.g. 'btn-sign-in'). Good until the page navigates.",
},
"selector": {
"type": "string",
"description": "CSS selector, as a fallback when no ref fits. Prefer ref.",
},
"text": {"type": "string", "description": "For 'type': the text to enter."},
"submit": {
"type": "boolean",
"description": "For 'type': press Enter and submit the owning form afterwards.",
},
"key": {
"type": "string",
"description": "For 'press': the key name, e.g. 'Enter', 'Escape', 'ArrowDown'.",
},
"amount": {
"type": "integer",
"description": "For 'scroll': pixels to scroll (negative scrolls up). Defaults to about one screen.",
},
"to": {
"type": "string",
"enum": list(SCROLL_TO),
"description": "For 'scroll': jump to the top or bottom instead of a distance.",
},
"max": {
"type": "integer",
"description": "For 'elements': cap the inventory. Defaults to the per-call maximum.",
},
"full": {
"type": "boolean",
"description": "For 'elements': re-read the whole page instead of a delta. Rarely needed.",
},
},
"required": ["action"],
},
}
registry.register(
name="drive_preview",
toolset="desktop_ui",
schema=ACT_PREVIEW_SCHEMA,
handler=lambda args, **kw: drive_preview_tool(
action=args.get("action", ""),
ref=args.get("ref"),
selector=args.get("selector"),
text=args.get("text"),
key=args.get("key"),
submit=args.get("submit"),
amount=args.get("amount"),
to=args.get("to"),
limit=args.get("max"),
full=args.get("full"),
callback=kw.get("callback"),
),
emoji="🖱️",
)
+1 -1
View File
@@ -276,7 +276,7 @@ TOOLSETS = {
"description": "Desktop GUI affordances — in-app terminal/browser panes, pane focus, reactions (GUI sessions only)",
"tools": [
"read_terminal", "close_terminal",
"open_preview", "close_preview", "read_preview",
"open_preview", "close_preview", "read_preview", "drive_preview", "annotate_preview",
"read_window_below",
"focus_pane", "react_to_message",
"setup_mcp", "tour",
+9
View File
@@ -1423,6 +1423,15 @@ def _(rid, params: dict) -> dict:
return _respond(rid, params, "text", allow_expired=True)
@method("preview.act.respond")
def _(rid, params: dict) -> dict:
# `text` is a JSON string with the interaction's outcome (drive_preview
# tool) — what it acted on, the live url/title, and a refreshed element
# inventory. allow_expired=True for the same reason as preview.read: the
# settle-and-rescan can lose the race with the tool's bounded wait.
return _respond(rid, params, "text", allow_expired=True)
@method("window.read.respond")
def _(rid, params: dict) -> dict:
# `text` is a JSON string describing the OS window underneath the Hermes
+15
View File
@@ -3548,6 +3548,7 @@ def _block(
"clarify.request",
"terminal.read.request",
"preview.read.request",
"preview.act.request",
"window.read.request",
"mcp.setup.request",
"tour.request",
@@ -6367,6 +6368,20 @@ def _agent_cbs(sid: str) -> dict:
{k: v for k, v in (("start", start), ("count", count)) if v is not None},
timeout=45,
),
# drive_preview tool (desktop GUI): the renderer injects the interaction
# engine into the preview pane's webview (or drives the pane's history)
# and answers preview.act.respond with the outcome plus a refreshed
# element inventory. Same budget as the preview read, which it ends
# with — a click on a slow page pays for the settle and the re-scan.
# annotate_preview rides this same callback: it resolves a target
# through the same engine and differs only in the verb it sends, so it
# needs a tool of its own but not a channel of its own.
"drive_preview_callback": lambda payload: _block(
"preview.act.request",
sid,
dict(payload),
timeout=45,
),
# read_window_below tool (desktop GUI): the renderer asks its main
# process (which owns native window enumeration) which OS window sits
# directly underneath the Hermes window, and answers
+3 -1
View File
@@ -8,7 +8,7 @@ description: "Authoritative reference for Hermes built-in tools, grouped by tool
This page documents Hermes' built-in tools, grouped by toolset. Availability varies by platform, credentials, and enabled toolsets.
**Quick counts (current registry):** ~84 tools — 10 browser tools (core) + 2 CDP-gated browser tools, 4 file tools, 4 Home Assistant tools, 2 terminal tools (`terminal`, `process`), 9 desktop-GUI tools (`read_terminal`, `close_terminal`, `open_preview`, `close_preview`, `read_preview`, `read_window_below`, `focus_pane`, `react_to_message`, `tour` — desktop-app sessions only), 2 web tools, 5 Feishu tools, 7 Spotify tools (registered by the bundled `spotify` plugin), 5 Yuanbao tools, 12 kanban tools (registered when the kanban dispatcher spawns the agent), 3 project tools (desktop/GUI sessions), 2 Discord tools, 3 video tools (`video_generate`, `xai_video_edit`, `xai_video_extend`), and a handful of standalone tools (`memory`, `clarify`, `delegate_task`, `execute_code`, `cronjob`, `session_search`, `skill_view`/`skill_manage`/`skills_list`, `text_to_speech`, `image_generate`, `vision_analyze`, `video_analyze`, `todo`, `computer_use`, `x_search`).
**Quick counts (current registry):** ~86 tools — 10 browser tools (core) + 2 CDP-gated browser tools, 4 file tools, 4 Home Assistant tools, 2 terminal tools (`terminal`, `process`), 11 desktop-GUI tools (`read_terminal`, `close_terminal`, `open_preview`, `close_preview`, `read_preview`, `drive_preview`, `annotate_preview`, `read_window_below`, `focus_pane`, `react_to_message`, `tour` — desktop-app sessions only), 2 web tools, 5 Feishu tools, 7 Spotify tools (registered by the bundled `spotify` plugin), 5 Yuanbao tools, 12 kanban tools (registered when the kanban dispatcher spawns the agent), 3 project tools (desktop/GUI sessions), 2 Discord tools, 3 video tools (`video_generate`, `xai_video_edit`, `xai_video_extend`), and a handful of standalone tools (`memory`, `clarify`, `delegate_task`, `execute_code`, `cronjob`, `session_search`, `skill_view`/`skill_manage`/`skills_list`, `text_to_speech`, `image_generate`, `vision_analyze`, `video_analyze`, `todo`, `computer_use`, `x_search`).
:::tip MCP Tools
In addition to built-in tools, Hermes can load tools dynamically from MCP servers. MCP tools appear with the prefix `mcp__<server>__` (e.g., `mcp__github__create_issue` for the `github` MCP server). See [MCP Integration](/user-guide/features/mcp) for configuration.
@@ -199,6 +199,8 @@ messaging, and cron sessions.
| `open_preview` | Open a web URL, localhost dev-server URL, or file path in the preview pane beside the chat in the Hermes desktop app. | — |
| `close_preview` | Close the preview pane beside the chat, or one tab inside it. Omit `url` to close the whole pane; pass a URL or file path to close that tab. | — |
| `read_preview` | Read what's currently shown in the preview pane of the Hermes desktop GUI — the in-app Browser's page text (URL + title + rendered text, pageable with `start`/`count`), or a file/artifact tab's identity. | — |
| `drive_preview` | Interact with the page open in the in-app browser: `elements` inventories what's clickable and typable (each with a ref that names it, like `btn-sign-in` or `inp-email`, plus role, label, and value), then `click`, `hover`, `type`, `scroll`, and `press` act on a ref, and `back`/`forward`/`reload` drive the pane's history. The pointer and keyboard are real input, so hover menus open. A ref lasts until the page navigates, including across a re-render that rebuilds the element, so after the first inventory every action answers with just a delta — what was added, removed, changed, or rebound — instead of the whole page again. | — |
| `annotate_preview` | Outline an element in the in-app browser and leave the mark up until it's removed — the deliberate counterpart to the transient cues `drive_preview` draws as it works. `add` marks a ref with an optional short label, `remove` takes one down, `clear` takes them all. Marks follow their element and vanish with it, so a navigation clears them. | — |
| `read_window_below` | Identify the OS window directly underneath the Hermes desktop window — app name, title, bounds (metadata only, never pixels). On macOS, other apps' titles appear only when Screen Recording is already granted; the tool never prompts for it. | — |
| `focus_pane` | Reveal and focus a pane in the Hermes desktop app (chat, files, terminal, review, sessions). | — |
| `react_to_message` | React to a message with a single emoji, iMessage-tapback style. Opt-in via Settings → Appearance (`display.message_reactions`). | — |
+1 -1
View File
@@ -71,7 +71,7 @@ Or in-session:
| `video_gen` | `video_generate`, `xai_video_edit`, `xai_video_extend` | Text-to-video and image-to-video via plugin-registered backends (xAI Grok-Imagine, FAL.ai Veo 3.1 / Pixverse v6 / Kling O3). Pass `image_url` to animate an image; omit it for text-to-video. `xai_video_edit` / `xai_video_extend` are provider-specific edit/extend tools, gated on xAI Imagine credentials. |
| `kanban` | `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_block`, `kanban_comment`, `kanban_complete`, `kanban_create`, `kanban_heartbeat`, `kanban_link`, `kanban_list`, `kanban_request_changes`, `kanban_request_review`, `kanban_show`, `kanban_unblock` | Multi-agent coordination tools. Registered for dispatcher-spawned task workers (`HERMES_KANBAN_TASK`) and for profiles that explicitly list the `kanban` toolset by name (the `all`/`*` wildcard does **not** enable it). Workers mark tasks done, request first-class review, block, heartbeat, comment, and create/link follow-up tasks; orchestrator profiles additionally get board-routing tools like list/unblock. `delegate_task` children are not Kanban run owners: their schema strips/disables this toolset and runtime guards reject direct board mutations, even if parent `HERMES_KANBAN_*` env vars are present. |
| `memory` | `memory` | Persistent cross-session memory management. |
| `desktop_ui` | `close_preview`, `close_terminal`, `focus_pane`, `open_preview`, `react_to_message`, `read_preview`, `read_terminal`, `read_window_below`, `tour` | Affordances that act on the Hermes desktop app itself — read/close the embedded terminal pane, open/read/close the in-app browser, identify the OS window behind the app, reveal a pane, react to a message, run a guided tour (highlight + narrate UI elements in the app or the preview pane). Enabled for sessions whose source is the desktop app, whichever backend it's connected to (local, SSH, URL, or Hermes Cloud). Never present on CLI, TUI, messaging, or cron sessions. |
| `desktop_ui` | `annotate_preview`, `close_preview`, `close_terminal`, `drive_preview`, `focus_pane`, `open_preview`, `react_to_message`, `read_preview`, `read_terminal`, `read_window_below`, `tour` | Affordances that act on the Hermes desktop app itself — read/close the embedded terminal pane, open, read, close, interact with, and annotate the in-app browser, identify the OS window behind the app, reveal a pane, react to a message, run a guided tour (highlight + narrate UI elements in the app or the preview pane). Enabled for sessions whose source is the desktop app, whichever backend it's connected to (local, SSH, URL, or Hermes Cloud). Never present on CLI, TUI, messaging, or cron sessions. |
| `project` | `project_create`, `project_list`, `project_switch` | Create and switch desktop [Projects](../user-guide/cli.md) (named, multi-folder workspaces). GUI / desktop sessions only. |
| `safe` | `image_generate`, `vision_analyze`, `web_extract`, `web_search` (via `includes`) | Read-only research + media generation. No file writes, no terminal, no code execution. |
| `search` | `web_search` | Web search only (without extract). |