Merge branch 'simp/r2-tools-a' into simp/integration2

# Conflicts:
#	tests/test_managed_runtime_resolution.py
This commit is contained in:
Teknium
2026-09-02 17:07:24 -07:00
21 changed files with 3492 additions and 3823 deletions
+1 -1
View File
@@ -95,7 +95,7 @@ _ALLOWED: dict[tuple[str, str], str] = {
("hermes_cli/main_install_repair.py", "npm"): (
"_resolve_node_runtime_npm()'s WSL re-scan: PATH minus /mnt/* IS the question."
),
("tools/browser_tool.py", "npx"): (
("tools/browser_tool_install.py", "npx"): (
"agent-browser runs via `npx`, resolved against the extended browser "
"PATH that _merge_browser_path() already seeds with the managed dirs."
),
-1
View File
@@ -296,7 +296,6 @@ class TestBrowserSupervisorRedaction:
closed_by="agent",
),),
frame_tree={"top": {"frame_id": "f1", "url": "about:blank", "origin": "null", "is_oopif": False}},
console_errors=(),
active=True,
cdp_url="ws://example.invalid/devtools/browser/mock",
task_id="test",
+50 -122
View File
@@ -32,7 +32,6 @@ _CDP_PRIVATE_PAGE_ALLOWED_METHODS = {
"Page.stopLoading",
}
# method → result paths that are ALWAYS opaque base64 (protocol-declared binary).
# redact_sensitive_text's Fernet pattern ("gAAAA" + base64 alphabet) can match
# arbitrary spans inside such payloads and corrupt the decoded bytes; the payload
@@ -55,49 +54,41 @@ _CDP_FLAGGED_BINARY_PATHS: Dict[str, tuple] = {
}
def _redact_cdp_output(
value: Any,
*,
always_paths: tuple = (),
flagged_paths: tuple = (),
) -> Any:
def _redact_cdp_output(value: Any, *, always_paths: tuple = (), flagged_paths: tuple = ()) -> Any:
"""Redact browser-originated CDP result text; opaque bytes stay byte-identical.
Exemptions come ONLY from the calling method's spec as exact result paths.
Path suffixes propagate only into the matching subtree, so ``base64Encoded``
is honored solely as a sibling on the trusted carrier object — never as
ambient trust a ``Runtime.evaluate`` by-value object could spoof (#94142).
ambient trust a ``Runtime.evaluate`` by-value object could spoof.
"""
from agent.redact import redact_sensitive_text
if isinstance(value, str):
return redact_sensitive_text(value, force=True)
if isinstance(value, list):
return [_redact_cdp_output(item) for item in value]
if isinstance(value, tuple):
return tuple(_redact_cdp_output(item) for item in value)
if isinstance(value, dict):
base64_flagged = value.get("base64Encoded") is True
if isinstance(value, (list, tuple)):
return type(value)(_redact_cdp_output(item) for item in value)
if not isinstance(value, dict):
return value
base64_flagged = value.get("base64Encoded") is True
def leaf(paths: tuple, key: str) -> bool:
return any(len(p) == 1 and p[0] == key for p in paths)
def leaf(paths: tuple, key: str) -> bool:
return any(len(p) == 1 and p[0] == key for p in paths)
def descend(paths: tuple, key: str) -> tuple:
return tuple(p[1:] for p in paths if len(p) > 1 and p[0] == key)
def descend(paths: tuple, key: str) -> tuple:
return tuple(p[1:] for p in paths if len(p) > 1 and p[0] == key)
redacted: Dict[str, Any] = {}
for key, item in value.items():
opaque = leaf(always_paths, key) or (leaf(flagged_paths, key) and base64_flagged)
if isinstance(item, str) and opaque:
redacted[key] = item
else:
redacted[key] = _redact_cdp_output(
item, always_paths=descend(always_paths, key), flagged_paths=descend(flagged_paths, key),
)
return redacted
redacted: Dict[str, Any] = {}
for key, item in value.items():
opaque = leaf(always_paths, key) or (leaf(flagged_paths, key) and base64_flagged)
if isinstance(item, str) and opaque:
redacted[key] = item
else:
redacted[key] = _redact_cdp_output(
item,
always_paths=descend(always_paths, key),
flagged_paths=descend(flagged_paths, key),
)
return redacted
return value
# ``websockets`` is a direct dependency; wrap so a stale env yields a clean error.
try:
@@ -117,13 +108,11 @@ def _run_async(coro):
loop = asyncio.get_running_loop()
except RuntimeError:
loop = None
if loop and loop.is_running():
import concurrent.futures
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
future = pool.submit(asyncio.run, coro)
return future.result()
return pool.submit(asyncio.run, coro).result()
return asyncio.run(coro)
@@ -163,30 +152,25 @@ _METHOD_PARAM_GUARDS = {
}
def _browser_cdp_private_guard(
*,
task_id: str,
method: str,
params: Dict[str, Any],
) -> Optional[str]:
def _browser_cdp_private_guard(*, task_id: str, method: str, params: Dict[str, Any]) -> Optional[str]:
"""Apply the browser SSRF/private-page guard to raw CDP calls.
Raw CDP shares the cloud/private-network boundary of ``browser_snapshot`` /
``browser_console`` / ``browser_eval`` and must not become their bypass.
Guard probes are best-effort; a probe failure never breaks local/custom CDP
workflows.
"""
try:
from tools import browser_tool as bt # type: ignore[import-not-found]
if not bt._eval_ssrf_guard_active(task_id): # type: ignore[attr-defined]
return None
guard = _METHOD_PARAM_GUARDS.get(method)
if guard is not None:
probe, template = guard
literal = probe(bt, params or {})
if literal:
return _blocked(template.format(literal), method)
if method not in _CDP_PRIVATE_PAGE_ALLOWED_METHODS:
blocked_url = bt._current_page_private_url(task_id) # type: ignore[attr-defined]
if blocked_url:
@@ -197,17 +181,12 @@ def _browser_cdp_private_guard(
method,
)
except Exception as exc: # noqa: BLE001
# Guard probes are best-effort; never break local/custom CDP workflows.
logger.debug("browser_cdp: private-page guard probe failed: %s", exc)
return None
async def _cdp_call(
ws_url: str,
method: str,
params: Dict[str, Any],
target_id: Optional[str],
timeout: float,
ws_url: str, method: str, params: Dict[str, Any], target_id: Optional[str], timeout: float,
) -> Dict[str, Any]:
"""Make a single CDP call, optionally attaching to a target first.
@@ -232,22 +211,18 @@ async def _cdp_call(
next_id += 1
await ws.send(json.dumps({"id": call_id, **req}))
deadline = asyncio.get_running_loop().time() + timeout
while True:
while True: # ignore events / out-of-order responses
remaining = deadline - asyncio.get_running_loop().time()
if remaining <= 0:
raise TimeoutError(f"Timed out {what}")
msg = json.loads(await asyncio.wait_for(ws.recv(), timeout=remaining))
if msg.get("id") == call_id:
return msg
# Ignore events / out-of-order responses
session_id: Optional[str] = None
if target_id:
msg = await _send(
{
"method": "Target.attachToTarget",
"params": {"targetId": target_id, "flatten": True},
},
{"method": "Target.attachToTarget", "params": {"targetId": target_id, "flatten": True}},
f"attaching to target {target_id}",
)
if "error" in msg:
@@ -266,11 +241,7 @@ async def _cdp_call(
def _browser_cdp_via_supervisor(
task_id: str,
frame_id: str,
method: str,
params: Optional[Dict[str, Any]],
timeout: float,
task_id: str, frame_id: str, method: str, params: Optional[Dict[str, Any]], timeout: float,
) -> str:
"""Route a CDP call through the live supervisor session for an OOPIF frame."""
try:
@@ -291,11 +262,9 @@ def _browser_cdp_via_supervisor(
f"frame_tree with frame_ids you can pass here."
)
snap = supervisor.snapshot()
tree = snap.frame_tree
tree = supervisor.snapshot().frame_tree
frame_info: Optional[Dict[str, Any]] = next(
(f for f in [tree.get("top"), *(tree.get("children") or [])]
if f and f.get("frame_id") == frame_id),
(f for f in [tree.get("top"), *(tree.get("children") or [])] if f and f.get("frame_id") == frame_id),
None,
)
if frame_info is None:
@@ -304,7 +273,6 @@ def _browser_cdp_via_supervisor(
raw = supervisor._frames.get(frame_id) # type: ignore[attr-defined]
if raw is not None:
frame_info = raw.to_dict()
if frame_info is None:
return tool_error(
f"frame_id {frame_id!r} not found in supervisor state. "
@@ -325,10 +293,7 @@ def _browser_cdp_via_supervisor(
loop = supervisor._loop # type: ignore[attr-defined]
if loop is None or not loop.is_running():
return tool_error(
"CDP supervisor loop is not running. Try reconnecting with "
"/browser connect."
)
return tool_error("CDP supervisor loop is not running. Try reconnecting with /browser connect.")
try:
from agent.async_utils import safe_schedule_threadsafe
@@ -337,25 +302,20 @@ def _browser_cdp_via_supervisor(
loop,
)
if fut is None:
return tool_error(
"CDP call via supervisor failed: loop unavailable",
cdp_docs=CDP_DOCS_URL,
)
return tool_error("CDP call via supervisor failed: loop unavailable", cdp_docs=CDP_DOCS_URL)
result_msg = fut.result(timeout=timeout + 2)
except Exception as exc:
return tool_error(
f"CDP call via supervisor failed: {type(exc).__name__}: {exc}",
cdp_docs=CDP_DOCS_URL,
f"CDP call via supervisor failed: {type(exc).__name__}: {exc}", cdp_docs=CDP_DOCS_URL,
)
payload: Dict[str, Any] = {
return json.dumps({
"success": True,
"method": method,
"frame_id": frame_id,
"session_id": child_sid,
"result": result_msg.get("result", {}),
}
return json.dumps(payload, ensure_ascii=False)
}, ensure_ascii=False)
def browser_cdp(
@@ -372,40 +332,26 @@ def browser_cdp(
(OOPIF from ``browser_snapshot.frame_tree``) routes through the supervisor's
live WebSocket instead — the only reliable way to evaluate inside an iframe
on backends where fresh per-call connections hit signed-URL expiry
(Browserbase). Returns JSON ``{"success": True, "method", "result"}`` or
``{"error": ...}``.
(Browserbase). Both paths share the same private-page/SSRF guard. Returns
JSON ``{"success": True, "method", "result"}`` or ``{"error": ...}``.
"""
effective_task_id = task_id or "default"
if frame_id:
# Same private-page/SSRF boundary as the stateless path below.
blocked = _browser_cdp_private_guard(
task_id=effective_task_id,
method=method,
params=params or {},
)
blocked = _browser_cdp_private_guard(task_id=effective_task_id, method=method, params=params or {})
if blocked:
return blocked
return _browser_cdp_via_supervisor(
task_id=effective_task_id,
frame_id=frame_id,
method=method,
params=params,
timeout=timeout,
task_id=effective_task_id, frame_id=frame_id, method=method, params=params, timeout=timeout,
)
if not method or not isinstance(method, str):
return tool_error(
"'method' is required (e.g. 'Target.getTargets')",
cdp_docs=CDP_DOCS_URL,
)
return tool_error("'method' is required (e.g. 'Target.getTargets')", cdp_docs=CDP_DOCS_URL)
if not _WS_AVAILABLE:
return tool_error(
"The 'websockets' Python package is required but not installed. "
"Install it with: pip install websockets"
)
endpoint = _resolve_cdp_endpoint()
if not endpoint:
return tool_error(
@@ -415,7 +361,6 @@ def browser_cdp(
"and does not expose CDP.",
cdp_docs=CDP_DOCS_URL,
)
if not endpoint.startswith(("ws://", "wss://")):
return tool_error(
f"CDP endpoint is not a WebSocket URL: {endpoint!r}. "
@@ -423,18 +368,11 @@ def browser_cdp(
"resolver should have rewritten this. Check that a Chromium-family "
"browser is actually listening on the debug port."
)
call_params: Dict[str, Any] = params or {}
if not isinstance(call_params, dict):
return tool_error(
f"'params' must be an object/dict, got {type(call_params).__name__}"
)
return tool_error(f"'params' must be an object/dict, got {type(call_params).__name__}")
blocked = _browser_cdp_private_guard(
task_id=effective_task_id,
method=method,
params=call_params,
)
blocked = _browser_cdp_private_guard(task_id=effective_task_id, method=method, params=call_params)
if blocked:
return blocked
@@ -445,14 +383,9 @@ def browser_cdp(
safe_timeout = max(1.0, min(safe_timeout, 300.0))
try:
result = _run_async(
_cdp_call(endpoint, method, call_params, target_id, safe_timeout)
)
result = _run_async(_cdp_call(endpoint, method, call_params, target_id, safe_timeout))
except asyncio.TimeoutError as exc:
return tool_error(
f"CDP call timed out after {safe_timeout}s: {exc}",
method=method,
)
return tool_error(f"CDP call timed out after {safe_timeout}s: {exc}", method=method)
except (TimeoutError, RuntimeError) as exc:
return tool_error(str(exc), method=method)
except WebSocketException as exc:
@@ -463,10 +396,7 @@ def browser_cdp(
)
except Exception as exc: # pragma: no cover — unexpected
logger.exception("browser_cdp unexpected error")
return tool_error(
f"Unexpected error: {type(exc).__name__}: {exc}",
method=method,
)
return tool_error(f"Unexpected error: {type(exc).__name__}: {exc}", method=method)
payload: Dict[str, Any] = {
"success": True,
@@ -586,6 +516,8 @@ def _browser_cdp_check() -> bool:
Camofox (REST-only), default local agent-browser (hidden CDP port) and cloud
providers whose per-session ``cdp_url`` isn't surfaced are gated out. Thin
wrapper so ``registry.register`` stays a top-level statement (AST scan).
Raw (no-I/O) gate: check_fns run at every startup; resolving the endpoint
over HTTP here would block launch on a stale endpoint.
"""
try:
from tools.browser_tool import ( # type: ignore[import-not-found]
@@ -595,11 +527,7 @@ def _browser_cdp_check() -> bool:
except ImportError as exc: # pragma: no cover — defensive
logger.debug("browser_cdp check: browser_tool import failed: %s", exc)
return False
if not check_browser_requirements():
return False
# Raw (no-I/O) gate: check_fns run at every startup; resolving the
# endpoint over HTTP here would block launch on a stale endpoint.
return bool(_get_cdp_override_raw())
return bool(check_browser_requirements() and _get_cdp_override_raw())
registry.register(
+11 -24
View File
@@ -80,31 +80,18 @@ def browser_dialog(
"""Respond to a pending dialog on the active task's CDP supervisor."""
supervisor = SUPERVISOR_REGISTRY.get(task_id or "default")
if supervisor is None:
return json.dumps(
{
"success": False,
"error": (
"No CDP supervisor is attached to this task. Either the "
"browser backend doesn't expose CDP (Camofox, default "
"Playwright) or no browser session has been started yet. "
"Call browser_navigate or /browser connect first."
),
}
)
result = supervisor.respond_to_dialog(
action=action,
prompt_text=prompt_text,
dialog_id=dialog_id,
)
return json.dumps({
"success": False,
"error": (
"No CDP supervisor is attached to this task. Either the "
"browser backend doesn't expose CDP (Camofox, default "
"Playwright) or no browser session has been started yet. "
"Call browser_navigate or /browser connect first."
),
})
result = supervisor.respond_to_dialog(action=action, prompt_text=prompt_text, dialog_id=dialog_id)
if result.get("ok"):
return json.dumps(
{
"success": True,
"action": action,
"dialog": result.get("dialog", {}),
}
)
return json.dumps({"success": True, "action": action, "dialog": result.get("dialog", {})})
return json.dumps({"success": False, "error": result.get("error", "unknown error")})
+13 -47
View File
@@ -51,10 +51,7 @@ def extension_controller_available(action: str) -> bool:
consults the process-local broker directly and fails closed on any gap.
"""
try:
from gateway.browser_control_broker import (
browser_control_enabled,
get_browser_control_broker,
)
from gateway.browser_control_broker import browser_control_enabled, get_browser_control_broker
if not browser_control_enabled():
return False
@@ -63,17 +60,11 @@ def extension_controller_available(action: str) -> bool:
return False
broker = get_browser_control_broker()
scope = broker.scope_for_session(
session_id=session_id,
principal_id=principal_id,
transport_family=transport_family,
session_id=session_id, principal_id=principal_id, transport_family=transport_family,
)
return scope is not None and broker.select(scope, action) is not None
except Exception:
logger.debug(
"browser extension availability check failed for %s",
action,
exc_info=True,
)
logger.debug("browser extension availability check failed for %s", action, exc_info=True)
return False
@@ -97,17 +88,11 @@ def route_browser_tool(
called exactly once when the feature is off or no server-bound identity
exists; once a controller is selected its result/exception is final.
"""
if not enabled:
return fallback()
if not str(principal_id or "").strip() or not str(transport_family or "").strip():
if not enabled or not str(principal_id or "").strip() or not str(transport_family or "").strip():
return fallback()
identity = dict(
session_id=session_id,
task_id=task_id,
principal_id=principal_id,
transport_family=transport_family,
session_id=session_id, task_id=task_id, principal_id=principal_id, transport_family=transport_family,
)
scope = broker.scope_for_session(**identity)
if scope is None:
@@ -117,25 +102,16 @@ def route_browser_tool(
lane_bound = getattr(broker, "lane_registered", None)
if callable(lane_bound) and not lane_bound(**identity):
return fallback()
raise _controller_unavailable(
f"bound browser controller unavailable for {action}"
)
raise _controller_unavailable(f"bound browser controller unavailable for {action}")
controller = broker.select(scope, action)
if controller is None:
raise _controller_unavailable(
f"bound browser controller cannot execute {action}"
)
if broker.select(scope, action) is None:
raise _controller_unavailable(f"bound browser controller cannot execute {action}")
# Controller is authoritative: never retry the legacy backend. Registry
# handlers must return a string; keep string results byte-identical and
# serialize decoded JSON values at this boundary.
result = broker.dispatch(
scope, action=action, arguments=args, tool_call_id=tool_call_id
)
if isinstance(result, str):
return result
return json.dumps(result, ensure_ascii=False)
result = broker.dispatch(scope, action=action, arguments=args, tool_call_id=tool_call_id)
return result if isinstance(result, str) else json.dumps(result, ensure_ascii=False)
def current_tool_call_id() -> str:
@@ -164,23 +140,13 @@ def routed_browser_handler(
Feature off (or gateway unimportable) ⇒ the legacy handler runs unchanged.
"""
try:
from gateway.browser_control_broker import (
browser_control_enabled,
get_browser_control_broker,
)
from gateway.browser_control_broker import browser_control_enabled, get_browser_control_broker
except Exception as exc: # pragma: no cover - defensive, gateway always present
logger.debug(
"browser extension router unavailable (%s); using legacy backend",
exc,
)
logger.debug("browser extension router unavailable (%s); using legacy backend", exc)
return fallback()
if not browser_control_enabled():
return fallback()
if tool_call_id is None:
tool_call_id = current_tool_call_id()
try:
env_session, env_principal, env_transport = _bound_identity()
except Exception:
@@ -196,5 +162,5 @@ def routed_browser_handler(
task_id=task_id,
principal_id=principal_id or env_principal,
transport_family=transport_family or env_transport,
tool_call_id=tool_call_id,
tool_call_id=current_tool_call_id() if tool_call_id is None else tool_call_id,
)
+3 -6
View File
@@ -25,8 +25,7 @@ logger = logging.getLogger(__name__)
LIGHTPANDA_INSTALL_URL = "https://lightpanda.io/docs/run-locally/installation/one-liner"
LIGHTPANDA_INSTALL_HINT = (
f"Install Lightpanda from {LIGHTPANDA_INSTALL_URL} and make sure "
"`lightpanda` is on PATH"
f"Install Lightpanda from {LIGHTPANDA_INSTALL_URL} and make sure " "`lightpanda` is on PATH"
)
_READY_TIMEOUT_S = 10.0
@@ -57,8 +56,7 @@ class LightpandaServer:
def _home_candidates() -> list:
home = Path.home()
candidates = [
home / ".lightpanda" / "lightpanda",
home / ".local" / "bin" / "lightpanda",
home / ".lightpanda" / "lightpanda", home / ".local" / "bin" / "lightpanda"
]
try:
from hermes_constants import get_hermes_home
@@ -285,8 +283,7 @@ def launch_lightpanda(
with _servers_lock:
_servers[session_name] = server
logger.info(
"Started lightpanda serve (pid %s, port %s) for session %s",
proc.pid, port, session_name,
"Started lightpanda serve (pid %s, port %s) for session %s", proc.pid, port, session_name
)
return server, None
+56 -134
View File
@@ -57,12 +57,9 @@ if TYPE_CHECKING:
logger = logging.getLogger(__name__)
# Ring buffer of recent console-level events.
CONSOLE_HISTORY_MAX = 50
def _redact_cdp_error_text(exc: object) -> str:
"""Redact CDP endpoint credentials from an exception's string form.
"""Redact CDP endpoint credentials from an exception's (or URL's) string form.
``websockets`` bakes the raw target URL (``?token=`` / ``user:pass@``) into
its exception messages. Every egress point that turns such an exception into
@@ -91,17 +88,8 @@ def _schedule(coro, loop, *, timeout: float):
return fut.result(timeout=timeout)
# ── Data model ────────────────────────────────────────────────────────────────
@dataclass
class ConsoleEvent:
"""Ring buffer entry for console + exception traffic."""
ts: float
level: str # "log" | "error" | "warning" | "exception"
text: str
url: Optional[str] = None
def _err(exc: BaseException) -> Dict[str, Any]:
return {"ok": False, "error": f"{type(exc).__name__}: {exc}"}
@dataclass(frozen=True)
@@ -111,7 +99,6 @@ class SupervisorSnapshot:
pending_dialogs: Tuple[PendingDialog, ...]
recent_dialogs: Tuple[DialogRecord, ...]
frame_tree: Dict[str, Any]
console_errors: Tuple[ConsoleEvent, ...]
active: bool # False if supervisor is detached/stopped
cdp_url: str
task_id: str
@@ -163,7 +150,6 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
self._pending_dialogs: Dict[str, PendingDialog] = {}
self._recent_dialogs: List[DialogRecord] = []
self._frames: Dict[str, FrameInfo] = {}
self._console_events: List[ConsoleEvent] = []
self._active = False
# Supervisor loop machinery — populated in start().
@@ -197,21 +183,14 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
self._start_error = None
self._stop_requested = False
self._thread = threading.Thread(
target=self._thread_main,
name=f"cdp-supervisor-{self.task_id}",
daemon=True,
target=self._thread_main, name=f"cdp-supervisor-{self.task_id}", daemon=True,
)
self._thread.start()
if not self._ready_event.wait(timeout=timeout):
self.stop()
try:
from agent.redact import redact_cdp_url
_safe_url = redact_cdp_url(self.cdp_url)
except Exception:
_safe_url = "<cdp_url redacted>"
raise TimeoutError(
f"CDP supervisor did not attach within {timeout}s "
f"(cdp_url={_safe_url[:80]}...)"
f"(cdp_url={_redact_cdp_error_text(self.cdp_url)[:80]}...)"
)
if self._start_error is not None:
err = self._start_error
@@ -247,7 +226,6 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
pending_dialogs=tuple(self._pending_dialogs.values()),
recent_dialogs=tuple(self._recent_dialogs[-RECENT_DIALOGS_MAX:]),
frame_tree=self._build_frame_tree_locked(),
console_errors=tuple(self._console_events[-CONSOLE_HISTORY_MAX:]),
active=self._active,
cdp_url=self.cdp_url,
task_id=self.task_id,
@@ -309,7 +287,7 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
except _LoopUnavailable as e:
return {"ok": False, "error": str(e)}
except Exception as e:
return {"ok": False, "error": f"{type(e).__name__}: {e}"}
return _err(e)
return {"ok": True, "dialog": dialog.to_dict()}
def evaluate_runtime(
@@ -362,13 +340,12 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
# CDP's recursion guard with the protocol-level error ``Object
# reference chain is too long``. Retry once with returnByValue=False
# so Chrome returns the description string instead of failing.
if return_by_value and "reference chain is too long" in str(exc).lower():
try:
response = _run_eval(False)
except Exception as exc2:
return {"ok": False, "error": f"{type(exc2).__name__}: {exc2}"}
else:
return {"ok": False, "error": f"{type(exc).__name__}: {exc}"}
if not (return_by_value and "reference chain is too long" in str(exc).lower()):
return _err(exc)
try:
response = _run_eval(False)
except Exception as exc2:
return _err(exc2)
# Response: {"result": {"result": {"type", "value", ...}, "exceptionDetails"?}}
result_payload = response.get("result", {}) if isinstance(response, dict) else {}
@@ -382,7 +359,6 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
result_obj = result_payload.get("result", {})
result_type = result_obj.get("type", "undefined")
if "value" in result_obj:
value = result_obj["value"]
elif result_type == "undefined":
@@ -391,7 +367,6 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
# Non-serializable (functions, DOM nodes…) — give the model the
# browser's description so it gets *something*.
value = result_obj.get("description") or result_obj.get("unserializableValue")
return {"ok": True, "result": value, "result_type": result_type}
# ── Supervisor loop internals ────────────────────────────────────────────
@@ -404,13 +379,10 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
asyncio.set_event_loop(loop)
loop.run_until_complete(self._run())
except BaseException as e: # noqa: BLE001 — propagate via _start_error
if not self._ready_event.is_set():
self._start_error = e
self._ready_event.set()
else:
if not self._fail_start(e):
logger.warning("CDP supervisor %s crashed: %s", self.task_id, e)
finally:
# Flush remaining tasks before closing the loop to avoid
# Cancel + flush remaining tasks before closing the loop to avoid
# "Task was destroyed but it is pending" warnings.
try:
pending = [t for t in asyncio.all_tasks(loop) if not t.done()]
@@ -418,19 +390,23 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
t.cancel()
if pending:
loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
except Exception:
pass
try:
loop.close()
except Exception:
pass
with self._state_lock:
self._active = False
def _fail_start(self, e: BaseException) -> bool:
"""Propagate ``e`` to ``start()`` if we never got ready; True if it was consumed."""
if self._ready_event.is_set():
return False
self._start_error = e
self._ready_event.set()
return True
async def _close_ws(self) -> None:
"""Detach and close the current WebSocket, swallowing close errors."""
ws = self._ws
self._ws = None
ws, self._ws = self._ws, None
if ws is not None:
try:
await ws.close()
@@ -442,7 +418,8 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
Browserbase tears down the CDP socket every time a short-lived client
(e.g. agent-browser's per-command CDP client) disconnects, so on drop we
reset per-session ids, re-attach, and keep going.
reset per-session ids, re-attach, and keep going. A failure before the
first successful attach is fatal for ``start()``.
"""
attempt = 0
last_success_at = 0.0
@@ -451,15 +428,11 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
while not self._stop_requested:
try:
self._ws = await asyncio.wait_for(
websockets.connect(self.cdp_url, max_size=50 * 1024 * 1024),
timeout=10.0,
websockets.connect(self.cdp_url, max_size=50 * 1024 * 1024), timeout=10.0,
)
except Exception as e:
attempt += 1
if not self._ready_event.is_set():
# Never connected once — fatal for start().
self._start_error = e
self._ready_event.set()
if self._fail_start(e):
return
logger.warning(
"CDP supervisor %s: connect failed (attempt %s): %s",
@@ -481,20 +454,14 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
self._active = True
last_success_at = time.time()
backoff = 0.5 # reset after a successful attach
if not self._ready_event.is_set():
self._ready_event.set()
self._ready_event.set()
await reader_task
except BaseException as e:
if not self._ready_event.is_set():
# Never got to ready — propagate to start().
self._start_error = e
self._ready_event.set()
if self._fail_start(e):
raise
logger.warning(
"CDP supervisor %s: session dropped after %.1fs: %s",
self.task_id,
time.time() - last_success_at,
_redact_cdp_error_text(e),
self.task_id, time.time() - last_success_at, _redact_cdp_error_text(e),
)
finally:
with self._state_lock:
@@ -505,22 +472,19 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
await reader_task
except (asyncio.CancelledError, Exception):
pass
for handle in list(self._dialog_watchdogs.values()):
for handle in self._dialog_watchdogs.values():
handle.cancel()
self._dialog_watchdogs.clear()
await self._close_ws()
if self._stop_requested:
return
logger.debug(
"CDP supervisor %s: reconnecting in %.1fs...", self.task_id, backoff,
)
logger.debug("CDP supervisor %s: reconnecting in %.1fs...", self.task_id, backoff)
await asyncio.sleep(backoff)
backoff = min(backoff * 2, 10.0)
async def _attach_initial_page(self) -> None:
"""Find a page target, attach flattened session, enable domains, install dialog bridge."""
"""Find (or create) a page target, attach flattened, enable domains, install dialog bridge."""
resp = await self._cdp("Target.getTargets")
targets = resp.get("result", {}).get("targetInfos", [])
page_target = next((t for t in targets if t.get("type") == "page"), None)
@@ -529,11 +493,7 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
target_id = created["result"]["targetId"]
else:
target_id = page_target["targetId"]
attach = await self._cdp(
"Target.attachToTarget",
{"targetId": target_id, "flatten": True},
)
attach = await self._cdp("Target.attachToTarget", {"targetId": target_id, "flatten": True})
self._page_session_id = attach["result"]["sessionId"]
await self._enable_page_domains(self._page_session_id, timeout=10.0)
await self._install_dialog_bridge(self._page_session_id)
@@ -565,7 +525,7 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
self._pending_calls.pop(call_id, None)
async def _read_loop(self) -> None:
"""Continuously dispatch incoming CDP frames."""
"""Continuously dispatch incoming CDP frames (responses → futures, events → handlers)."""
assert self._ws is not None
try:
async for raw in self._ws:
@@ -580,58 +540,22 @@ class CDPSupervisor(DialogSupervisionMixin, FrameTrackingMixin):
fut = self._pending_calls.pop(msg["id"], None)
if fut is not None and not fut.done():
if "error" in msg:
fut.set_exception(
RuntimeError(f"CDP error on id={msg['id']}: {msg['error']}")
)
fut.set_exception(RuntimeError(f"CDP error on id={msg['id']}: {msg['error']}"))
else:
fut.set_result(msg)
elif "method" in msg:
await self._on_event(msg["method"], msg.get("params", {}), msg.get("sessionId"))
handler = self._EVENT_HANDLERS.get(msg["method"])
if handler is not None:
result = handler(self, msg.get("params", {}), msg.get("sessionId"))
if result is not None:
await result
except Exception as e:
logger.debug("CDP read loop exited: %s", e)
# ── Event dispatch ──────────────────────────────────────────────────────
async def _on_event(
self, method: str, params: Dict[str, Any], session_id: Optional[str]
) -> None:
handler = self._EVENT_HANDLERS.get(method)
if handler is None:
return
result = handler(self, params, session_id)
if result is not None:
await result
# ── Console / exception ring buffer ─────────────────────────────────────
def _on_console(self, params: Dict[str, Any], *, level_from: str) -> None:
if level_from == "exception":
details = params.get("exceptionDetails") or {}
text = str(details.get("text") or "")
url = details.get("url")
event = ConsoleEvent(ts=time.time(), level="exception", text=text, url=url)
else:
raw_level = str(params.get("type") or "log")
level = "error" if raw_level in {"error", "assert"} else (
"warning" if raw_level == "warning" else "log"
)
args = params.get("args") or []
parts: List[str] = []
for a in args[:4]:
if isinstance(a, dict):
parts.append(str(a.get("value") or a.get("description") or ""))
event = ConsoleEvent(ts=time.time(), level=level, text=" ".join(parts))
with self._state_lock:
self._console_events.append(event)
self._console_events = _trim_ring(self._console_events, CONSOLE_HISTORY_MAX)
# CDP event → handler(self, params, session_id). Async handlers return an
# awaitable that ``_on_event`` awaits; sync handlers return None.
# awaitable that ``_read_loop`` awaits; sync handlers return None.
_EVENT_HANDLERS: Dict[str, Callable[..., Any]] = {
**DialogSupervisionMixin.EVENT_HANDLERS,
**FrameTrackingMixin.EVENT_HANDLERS,
"Runtime.consoleAPICalled": lambda self, p, _sid: self._on_console(p, level_from="api"),
"Runtime.exceptionThrown": lambda self, p, _sid: self._on_console(p, level_from="exception"),
**DialogSupervisionMixin.EVENT_HANDLERS, **FrameTrackingMixin.EVENT_HANDLERS
}
@@ -664,27 +588,26 @@ class _SupervisorRegistry:
) -> CDPSupervisor:
"""Idempotently ensure a supervisor is running for ``(task_id, cdp_url)``.
An existing supervisor bound to a different ``cdp_url`` (or unhealthy)
is stopped and replaced.
An existing supervisor bound to a different ``cdp_url`` (or unhealthy:
dead thread / stopped loop) is stopped and replaced.
"""
with self._lock:
existing = self._by_task.get(task_id)
if existing is not None:
if existing.cdp_url == cdp_url:
thread_ok = existing._thread is not None and existing._thread.is_alive()
loop_ok = existing._loop is not None and existing._loop.is_running()
if thread_ok and loop_ok:
return existing
# URL changed or unhealthy — tear down, fall through to re-create.
thread, loop = existing._thread, existing._loop
if (
existing.cdp_url == cdp_url
and thread is not None and thread.is_alive()
and loop is not None and loop.is_running()
):
return existing
self._by_task.pop(task_id, None)
if existing is not None:
existing.stop()
supervisor = CDPSupervisor(
task_id=task_id,
cdp_url=cdp_url,
dialog_policy=dialog_policy,
dialog_timeout_s=dialog_timeout_s,
task_id=task_id, cdp_url=cdp_url,
dialog_policy=dialog_policy, dialog_timeout_s=dialog_timeout_s,
)
supervisor.start(timeout=start_timeout)
with self._lock:
@@ -706,9 +629,9 @@ class _SupervisorRegistry:
def stop_all(self) -> None:
"""Stop every running supervisor. For shutdown / test teardown."""
with self._lock:
items = list(self._by_task.items())
items = list(self._by_task.values())
self._by_task.clear()
for _, supervisor in items:
for supervisor in items:
supervisor.stop()
@@ -717,7 +640,6 @@ SUPERVISOR_REGISTRY = _SupervisorRegistry()
__all__ = [
"CDPSupervisor",
"ConsoleEvent",
"DEFAULT_DIALOG_POLICY",
"DEFAULT_DIALOG_TIMEOUT_S",
"DIALOG_POLICY_AUTO_ACCEPT",
+88 -182
View File
@@ -22,7 +22,7 @@ import json
import logging
import time
from dataclasses import dataclass
from typing import Any, Callable, Coroutine, Dict, Optional
from typing import Any, Callable, Dict, Optional
from urllib.parse import parse_qs, urlparse
# Logger-name parity with the origin module (records must look unchanged).
@@ -46,11 +46,9 @@ def _trim_ring(events: list, keep: int) -> list:
DIALOG_POLICY_MUST_RESPOND = "must_respond"
DIALOG_POLICY_AUTO_DISMISS = "auto_dismiss"
DIALOG_POLICY_AUTO_ACCEPT = "auto_accept"
_VALID_POLICIES = frozenset(
{DIALOG_POLICY_MUST_RESPOND, DIALOG_POLICY_AUTO_DISMISS, DIALOG_POLICY_AUTO_ACCEPT}
)
DEFAULT_DIALOG_POLICY = DIALOG_POLICY_MUST_RESPOND
DEFAULT_DIALOG_TIMEOUT_S = 300.0
@@ -64,7 +62,12 @@ RECENT_DIALOGS_MAX = 20
DIALOG_BRIDGE_HOST = "hermes-dialog-bridge.invalid"
DIALOG_BRIDGE_URL_PATTERN = f"http://{DIALOG_BRIDGE_HOST}/*"
# Injected into every frame via Page.addScriptToEvaluateOnNewDocument.
# Injected into every frame via Page.addScriptToEvaluateOnNewDocument. Uses a
# sync GET with query params so the Fetch interceptor never parses a body; if
# the bridge is unreachable it returns null so the page still sees *some*
# behavior (the backend auto-dismisses). onbeforeunload is left native — it
# can't be prompted synchronously without racing navigation; the native-dialog
# fallback path still surfaces it in recent_dialogs.
_DIALOG_BRIDGE_SCRIPT = r"""
(() => {
if (window.__hermesDialogBridgeInstalled) return;
@@ -73,8 +76,6 @@ _DIALOG_BRIDGE_SCRIPT = r"""
function ask(kind, message, defaultPrompt) {
try {
const xhr = new XMLHttpRequest();
// Use GET with query params so we don't need to worry about request
// body encoding in the Fetch interceptor.
const params = new URLSearchParams({
kind: String(kind || ""),
message: String(message == null ? "" : message),
@@ -83,9 +84,8 @@ _DIALOG_BRIDGE_SCRIPT = r"""
xhr.open("GET", ENDPOINT + "?" + params.toString(), false); // sync
xhr.send(null);
if (xhr.status !== 200) return null;
const body = xhr.responseText || "";
let parsed;
try { parsed = JSON.parse(body); } catch (e) { return null; }
try { parsed = JSON.parse(xhr.responseText || ""); } catch (e) { return null; }
if (kind === "alert") return undefined;
if (kind === "confirm") return Boolean(parsed && parsed.accept);
if (kind === "prompt") {
@@ -94,14 +94,9 @@ _DIALOG_BRIDGE_SCRIPT = r"""
}
return null;
} catch (e) {
// If the bridge is unreachable, fall back to the native call so the
// page still sees *some* behavior (the backend will auto-dismiss).
return null;
}
}
const realAlert = window.alert;
const realConfirm = window.confirm;
const realPrompt = window.prompt;
window.alert = function(message) { ask("alert", message, ""); };
window.confirm = function(message) {
const r = ask("confirm", message, "");
@@ -111,10 +106,6 @@ _DIALOG_BRIDGE_SCRIPT = r"""
const r = ask("prompt", message, def == null ? "" : def);
return r === null ? null : String(r);
};
// onbeforeunload — we can't really synchronously prompt the user from this
// event without racing navigation. Leave native behavior for now; the
// supervisor's native-dialog fallback path still surfaces them in
// recent_dialogs.
})();
"""
@@ -178,6 +169,14 @@ class DialogRecord:
class DialogSupervisionMixin:
"""Dialog event handling for ``CDPSupervisor`` (all methods run on its loop)."""
async def _cdp_quiet(self, method: str, params: Dict[str, Any], *, session_id: Optional[str],
timeout: float, what: str) -> None:
"""Best-effort CDP call: failures are logged at debug and swallowed."""
try:
await self._cdp(method, params, session_id=session_id, timeout=timeout)
except Exception as e:
logger.debug("%s failed (%s): %s", method, what, e)
async def _install_dialog_bridge(self, session_id: str) -> None:
"""Install the dialog-bridge init script + Fetch interceptor on a session.
@@ -185,110 +184,65 @@ class DialogSupervisionMixin:
scoped to the bridge URL catches the XHRs, which surface as pending
dialogs and are fulfilled when the agent responds. Idempotent at the CDP
level (Chromium de-dupes identical add-script calls; Fetch.enable
replaces prior patterns).
replaces prior patterns). The final Runtime.evaluate injects into the
already-loaded document so existing pages pick up the override on reconnect.
"""
try:
await self._cdp(
"Page.addScriptToEvaluateOnNewDocument",
{"source": _DIALOG_BRIDGE_SCRIPT, "runImmediately": True},
session_id=session_id,
timeout=5.0,
)
except Exception as e:
logger.debug(
"dialog bridge: addScriptToEvaluateOnNewDocument failed on sid=%s: %s",
(session_id or "")[:16], e,
)
try:
await self._cdp(
"Fetch.enable",
{
"patterns": [
{
"urlPattern": DIALOG_BRIDGE_URL_PATTERN,
"requestStage": "Request",
}
],
"handleAuthRequests": False,
},
session_id=session_id,
timeout=5.0,
)
except Exception as e:
logger.debug(
"dialog bridge: Fetch.enable failed on sid=%s: %s",
(session_id or "")[:16], e,
)
# Best-effort inject into the already-loaded document so existing pages
# pick up the override on reconnect.
try:
await self._cdp(
"Runtime.evaluate",
{"expression": _DIALOG_BRIDGE_SCRIPT, "returnByValue": True},
session_id=session_id,
timeout=3.0,
)
except Exception:
pass
sid = (session_id or "")[:16]
await self._cdp_quiet(
"Page.addScriptToEvaluateOnNewDocument",
{"source": _DIALOG_BRIDGE_SCRIPT, "runImmediately": True},
session_id=session_id, timeout=5.0, what=f"dialog bridge sid={sid}",
)
await self._cdp_quiet(
"Fetch.enable",
{"patterns": [{"urlPattern": DIALOG_BRIDGE_URL_PATTERN, "requestStage": "Request"}],
"handleAuthRequests": False},
session_id=session_id, timeout=5.0, what=f"dialog bridge sid={sid}",
)
await self._cdp_quiet(
"Runtime.evaluate",
{"expression": _DIALOG_BRIDGE_SCRIPT, "returnByValue": True},
session_id=session_id, timeout=3.0, what=f"dialog bridge inject sid={sid}",
)
# ── Capture ──────────────────────────────────────────────────────────────
async def _on_dialog_opening(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
dialog = self._new_dialog(
async def _on_dialog_opening(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
self._admit_dialog(self._new_dialog(
type=str(params.get("type") or ""),
message=str(params.get("message") or ""),
default_prompt=str(params.get("defaultPrompt") or ""),
session_id=session_id,
frame_id=params.get("frameId"),
)
self._admit_dialog(dialog, self._auto_handle_dialog)
))
async def _on_fetch_paused(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
async def _on_fetch_paused(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
"""Bridge XHR captured mid-flight — materialize as a pending dialog.
The page's JS thread is blocked on the XHR until we Fetch.fulfillRequest
(from ``respond_to_dialog`` or the watchdog).
(from ``respond_to_dialog`` or the watchdog). Requests for other hosts
are forwarded unchanged so the page sees its own request.
"""
url = str(params.get("request", {}).get("url") or "")
request_id = params.get("requestId")
if not request_id:
return
if DIALOG_BRIDGE_HOST not in url:
# Not ours — forward unchanged so the page sees its own request.
try:
await self._cdp(
"Fetch.continueRequest", {"requestId": request_id},
session_id=session_id, timeout=3.0,
)
except Exception:
pass
await self._cdp_quiet("Fetch.continueRequest", {"requestId": request_id},
session_id=session_id, timeout=3.0, what="passthrough")
return
q = parse_qs(urlparse(url).query)
dialog = self._new_dialog(
self._admit_dialog(self._new_dialog(
type=q.get("kind", [""])[0] or "alert",
message=q.get("message", [""])[0],
default_prompt=q.get("default_prompt", [""])[0],
session_id=session_id,
frame_id=params.get("frameId"),
bridge_request_id=str(request_id),
)
self._admit_dialog(dialog, self._fulfill_bridge_request)
))
def _new_dialog(
self,
*,
type: str,
message: str,
default_prompt: str,
session_id: Optional[str],
frame_id: Optional[str],
bridge_request_id: Optional[str] = None,
) -> PendingDialog:
def _new_dialog(self, *, type: str, message: str, default_prompt: str, session_id: Optional[str],
frame_id: Optional[str], bridge_request_id: Optional[str] = None) -> PendingDialog:
self._dialog_seq += 1
return PendingDialog(
id=f"d-{self._dialog_seq}",
@@ -301,12 +255,8 @@ class DialogSupervisionMixin:
bridge_request_id=bridge_request_id,
)
def _admit_dialog(
self,
dialog: PendingDialog,
responder: Callable[..., Coroutine[Any, Any, None]],
) -> None:
"""Apply the dialog policy: auto-respond via ``responder`` or queue + arm watchdog.
def _admit_dialog(self, dialog: PendingDialog) -> None:
"""Apply the dialog policy: auto-respond, or queue + arm the watchdog.
Auto policies archive FIRST (tagged ``auto_policy``) so the ``closed``
event that follows our own response isn't re-archived as ``remote``.
@@ -316,61 +266,33 @@ class DialogSupervisionMixin:
DIALOG_POLICY_AUTO_ACCEPT: (True, dialog.default_prompt),
}.get(self.dialog_policy)
if auto is not None:
accept, prompt_text = auto
with self._state_lock:
self._archive_dialog_locked(dialog, "auto_policy")
asyncio.create_task(responder(dialog, accept=accept, prompt_text=prompt_text))
asyncio.create_task(self._respond_quiet(dialog, accept=auto[0], prompt_text=auto[1]))
return
# must_respond → add to pending and arm watchdog.
with self._state_lock:
self._pending_dialogs[dialog.id] = dialog
loop = asyncio.get_running_loop()
handle = loop.call_later(
self._dialog_watchdogs[dialog.id] = asyncio.get_running_loop().call_later(
self.dialog_timeout_s,
lambda: asyncio.create_task(self._dialog_timeout_expired(dialog.id)),
)
self._dialog_watchdogs[dialog.id] = handle
# ── Responding ───────────────────────────────────────────────────────────
async def _respond(
self, dialog: PendingDialog, *, accept: bool, prompt_text: Optional[str]
) -> None:
"""Bridge-fulfill for XHR-captured dialogs, else native CDP. Raises on native CDP failure."""
async def _respond(self, dialog: PendingDialog, *, accept: bool, prompt_text: Optional[str]) -> None:
"""Bridge-fulfill for XHR-captured dialogs, else native CDP.
Native path sends ``promptText`` only for prompt dialogs when
``prompt_text`` is given, and raises on CDP failure; the bridge path
(Fetch.fulfillRequest so the page unblocks) swallows failures.
"""
if dialog.bridge_request_id:
await self._fulfill_bridge_request(dialog, accept=accept, prompt_text=prompt_text or "")
else:
await self._native_handle_dialog(dialog, accept=accept, prompt_text=prompt_text)
async def _native_handle_dialog(
self, dialog: PendingDialog, *, accept: bool, prompt_text: Optional[str]
) -> None:
"""Page.handleJavaScriptDialog; ``promptText`` sent only for prompt dialogs
when ``prompt_text`` is given. Raises on CDP failure."""
params: Dict[str, Any] = {"accept": accept}
if prompt_text is not None and dialog.type == "prompt":
params["promptText"] = prompt_text
await self._cdp(
"Page.handleJavaScriptDialog",
params,
session_id=dialog.cdp_session_id or None,
timeout=5.0,
)
async def _fulfill_bridge_request(
self, dialog: PendingDialog, *, accept: bool, prompt_text: str
) -> None:
"""Resolve a bridge XHR via Fetch.fulfillRequest so the page unblocks."""
if not dialog.bridge_request_id:
return
payload = {
"accept": bool(accept),
"prompt_text": prompt_text if dialog.type == "prompt" else "",
"dialog_id": dialog.id,
}
body = json.dumps(payload).encode()
try:
await self._cdp(
body = json.dumps({
"accept": bool(accept),
"prompt_text": (prompt_text or "") if dialog.type == "prompt" else "",
"dialog_id": dialog.id,
}).encode()
await self._cdp_quiet(
"Fetch.fulfillRequest",
{
"requestId": dialog.bridge_request_id,
@@ -381,24 +303,23 @@ class DialogSupervisionMixin:
],
"body": base64.b64encode(body).decode(),
},
session_id=dialog.cdp_session_id or None,
timeout=5.0,
session_id=dialog.cdp_session_id or None, timeout=5.0, what=f"bridge fulfill {dialog.id}",
)
except Exception as e:
logger.debug("bridge fulfill failed for %s: %s", dialog.id, e)
return
params: Dict[str, Any] = {"accept": accept}
if prompt_text is not None and dialog.type == "prompt":
params["promptText"] = prompt_text
await self._cdp("Page.handleJavaScriptDialog", params,
session_id=dialog.cdp_session_id or None, timeout=5.0)
async def _auto_handle_dialog(
self, dialog: PendingDialog, *, accept: bool, prompt_text: str
) -> None:
"""Auto-policy response for a native dialog (already archived by the caller)."""
async def _respond_quiet(self, dialog: PendingDialog, *, accept: bool, prompt_text: Optional[str]) -> None:
"""Auto-policy / watchdog response (already archived by the caller); failures logged only."""
try:
await self._native_handle_dialog(dialog, accept=accept, prompt_text=prompt_text)
await self._respond(dialog, accept=accept, prompt_text=prompt_text)
except Exception as e:
logger.debug("auto-handle CDP call failed for %s: %s", dialog.id, e)
logger.debug("auto response failed for %s: %s", dialog.id, e)
async def _handle_dialog_cdp(
self, dialog: PendingDialog, *, accept: bool, prompt_text: str
) -> None:
async def _handle_dialog_cdp(self, dialog: PendingDialog, *, accept: bool, prompt_text: str) -> None:
"""Agent response path.
The dialog is retired regardless of outcome — a CDP error usually means
@@ -416,19 +337,13 @@ class DialogSupervisionMixin:
return
logger.warning(
"CDP supervisor %s: dialog %s (%s) auto-dismissed after %ss timeout",
self.task_id,
dialog_id,
dialog.type,
self.dialog_timeout_s,
self.task_id, dialog_id, dialog.type, self.dialog_timeout_s,
)
try:
# Archive with watchdog tag BEFORE unblocking the page.
with self._state_lock:
if self._pending_dialogs.pop(dialog_id, None) is not None:
self._archive_dialog_locked(dialog, "watchdog")
await self._respond(dialog, accept=False, prompt_text=None)
except Exception as e:
logger.debug("auto-dismiss failed for %s: %s", dialog_id, e)
# Archive with watchdog tag BEFORE unblocking the page.
with self._state_lock:
if self._pending_dialogs.pop(dialog_id, None) is not None:
self._archive_dialog_locked(dialog, "watchdog")
await self._respond_quiet(dialog, accept=False, prompt_text=None)
# ── Bookkeeping ──────────────────────────────────────────────────────────
@@ -444,29 +359,20 @@ class DialogSupervisionMixin:
def _archive_dialog_locked(self, dialog: PendingDialog, closed_by: str) -> None:
"""Move a pending dialog to the recent_dialogs ring buffer. Must hold state_lock."""
record = DialogRecord(
id=dialog.id,
type=dialog.type,
message=dialog.message,
opened_at=dialog.opened_at,
closed_at=time.time(),
closed_by=closed_by,
frame_id=dialog.frame_id,
)
self._recent_dialogs.append(record)
self._recent_dialogs.append(DialogRecord(
id=dialog.id, type=dialog.type, message=dialog.message, opened_at=dialog.opened_at,
closed_at=time.time(), closed_by=closed_by, frame_id=dialog.frame_id,
))
self._recent_dialogs = _trim_ring(self._recent_dialogs, RECENT_DIALOGS_MAX)
async def _on_dialog_closed(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
async def _on_dialog_closed(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
# ``Page.javascriptDialogClosed`` carries only ``result``/``userInput``, not
# the message. Match by session id and clear the oldest native dialog on
# it — the JS thread blocks while a dialog is up, so at most one is in
# flight per session. Bridge dialogs resolve via Fetch.fulfillRequest.
with self._state_lock:
candidate_ids = [
d.id
for d in self._pending_dialogs.values()
d.id for d in self._pending_dialogs.values()
if d.cdp_session_id == session_id and d.bridge_request_id is None
]
if candidate_ids:
+32 -70
View File
@@ -41,12 +41,7 @@ class FrameInfo:
name: str = ""
def to_dict(self) -> Dict[str, Any]:
d = {
"frame_id": self.frame_id,
"url": self.url,
"origin": self.origin,
"is_oopif": self.is_oopif,
}
d = {"frame_id": self.frame_id, "url": self.url, "origin": self.origin, "is_oopif": self.is_oopif}
if self.cdp_session_id:
d["session_id"] = self.cdp_session_id
if self.parent_frame_id:
@@ -63,49 +58,36 @@ class FrameTrackingMixin:
"""Page.enable + Runtime.enable + nested auto-attach on one session."""
await self._cdp("Page.enable", session_id=session_id, timeout=timeout)
await self._cdp("Runtime.enable", session_id=session_id, timeout=timeout)
await self._cdp(
"Target.setAutoAttach", _AUTO_ATTACH_PARAMS,
session_id=session_id, timeout=timeout,
)
await self._cdp("Target.setAutoAttach", _AUTO_ATTACH_PARAMS, session_id=session_id, timeout=timeout)
def _on_frame_attached(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
def _on_frame_attached(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
frame_id = params.get("frameId")
if not frame_id:
return
with self._state_lock:
self._frames[frame_id] = FrameInfo(
frame_id=frame_id,
url="",
origin="",
parent_frame_id=params.get("parentFrameId"),
is_oopif=False,
cdp_session_id=session_id,
frame_id=frame_id, url="", origin="", parent_frame_id=params.get("parentFrameId"),
is_oopif=False, cdp_session_id=session_id,
)
def _on_frame_navigated(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
def _on_frame_navigated(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
frame = params.get("frame") or {}
frame_id = frame.get("id")
if not frame_id:
return
with self._state_lock:
existing = self._frames.get(frame_id)
old = self._frames.get(frame_id)
self._frames[frame_id] = FrameInfo(
frame_id=frame_id,
url=str(frame.get("url") or ""),
origin=str(frame.get("securityOrigin") or frame.get("origin") or ""),
parent_frame_id=frame.get("parentId") or (existing.parent_frame_id if existing else None),
is_oopif=bool(existing.is_oopif if existing else False),
cdp_session_id=existing.cdp_session_id if existing else session_id,
name=str(frame.get("name") or (existing.name if existing else "")),
parent_frame_id=frame.get("parentId") or (old.parent_frame_id if old else None),
is_oopif=bool(old.is_oopif if old else False),
cdp_session_id=old.cdp_session_id if old else session_id,
name=str(frame.get("name") or (old.name if old else "")),
)
def _on_frame_detached(
self, params: Dict[str, Any], session_id: Optional[str]
) -> None:
def _on_frame_detached(self, params: Dict[str, Any], session_id: Optional[str]) -> None:
"""Drop a frame only when it's truly gone.
``reason="swap"`` means the frame is migrating processes (e.g. promoted
@@ -115,14 +97,11 @@ class FrameTrackingMixin:
alive, so keep it until Target.detached + a later frameDetached clear it.
"""
frame_id = params.get("frameId")
if not frame_id:
return
reason = str(params.get("reason") or "remove").lower()
if reason == "swap":
if not frame_id or str(params.get("reason") or "remove").lower() == "swap":
return
with self._state_lock:
existing = self._frames.get(frame_id)
if existing and existing.is_oopif and existing.cdp_session_id:
old = self._frames.get(frame_id)
if old and old.is_oopif and old.cdp_session_id:
return
self._frames.pop(frame_id, None)
@@ -132,22 +111,17 @@ class FrameTrackingMixin:
target_type = info.get("type")
if not sid or target_type not in {"iframe", "worker"}:
return
# Record the frame with its OOPIF session id for interaction routing.
if target_type == "iframe":
# Record the frame with its OOPIF session id for interaction routing;
# origin is filled by frameNavigated on the child session.
target_id = info.get("targetId")
with self._state_lock:
existing = self._frames.get(target_id)
old = self._frames.get(target_id)
self._frames[target_id] = FrameInfo(
frame_id=target_id,
url=str(info.get("url") or ""),
origin="", # filled by frameNavigated on the child session
parent_frame_id=(existing.parent_frame_id if existing else None),
is_oopif=True,
cdp_session_id=sid,
name=str(info.get("title") or (existing.name if existing else "")),
frame_id=target_id, url=str(info.get("url") or ""), origin="",
parent_frame_id=(old.parent_frame_id if old else None), is_oopif=True,
cdp_session_id=sid, name=str(info.get("title") or (old.name if old else "")),
)
# Enable child domains off-loop: awaiting the replies here would deadlock
# because only the reader can resolve those Futures.
asyncio.create_task(self._enable_child_domains(sid))
@@ -178,25 +152,21 @@ class FrameTrackingMixin:
self._frames[fid] = replace(frame, cdp_session_id=None)
def _build_frame_tree_locked(self) -> Dict[str, Any]:
"""Build the capped frame_tree payload. Must be called under state lock."""
frames = self._frames
empty = {"top": None, "children": [], "truncated": False}
if not frames:
return empty
"""Build the capped frame_tree payload. Must be called under state lock.
# Top frame: one with no parent, preferring oopif=False.
Top frame = one with no parent, preferring oopif=False. BFS from it,
capped by FRAME_TREE_MAX_ENTRIES and FRAME_TREE_MAX_OOPIF_DEPTH for
OOPIF branches.
"""
frames = self._frames
tops = [f for f in frames.values() if not f.parent_frame_id]
top = next((f for f in tops if not f.is_oopif), tops[0] if tops else None)
if top is None:
return empty
return {"top": None, "children": [], "truncated": False}
# BFS from top, capped by FRAME_TREE_MAX_ENTRIES and
# FRAME_TREE_MAX_OOPIF_DEPTH for OOPIF branches.
children: List[Dict[str, Any]] = []
truncated = False
queue: List[Tuple[FrameInfo, int]] = [
(f, 1) for f in frames.values() if f.parent_frame_id == top.frame_id
]
queue: List[Tuple[FrameInfo, int]] = [(f, 1) for f in frames.values() if f.parent_frame_id == top.frame_id]
visited: set[str] = {top.frame_id}
while queue and len(children) < FRAME_TREE_MAX_ENTRIES:
frame, depth = queue.pop(0)
@@ -207,17 +177,9 @@ class FrameTrackingMixin:
truncated = True
continue
children.append(frame.to_dict())
for f in frames.values():
if f.parent_frame_id == frame.frame_id and f.frame_id not in visited:
queue.append((f, depth + 1))
if queue:
truncated = True
return {
"top": top.to_dict(),
"children": children,
"truncated": truncated,
}
queue.extend((f, depth + 1) for f in frames.values()
if f.parent_frame_id == frame.frame_id and f.frame_id not in visited)
return {"top": top.to_dict(), "children": children, "truncated": truncated or bool(queue)}
# CDP event → handler(self, params, session_id); merged into CDPSupervisor._EVENT_HANDLERS.
EVENT_HANDLERS: Dict[str, Callable[..., Any]] = {
+241 -3097
View File
File diff suppressed because it is too large Load Diff
+188
View File
@@ -0,0 +1,188 @@
"""User-supplied CDP endpoint resolution (browser.cdp_url / real-profile), dialog-policy config and the per-task CDP supervisor lifecycle.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
import os
from typing import Tuple
from tools.browser_tool_origin import origin_module as _origin
def _resolve_cdp_override(cdp_url: str) -> str:
"""Normalize a user-supplied CDP endpoint into a concrete websocket URL.
Full ``ws://.../devtools/browser/...`` endpoints pass through; HTTP
discovery roots and bare ``ws://host:port`` are resolved via
``/json/version`` → ``webSocketDebuggerUrl`` (falls back to the raw value
with a warning if discovery fails).
"""
_bt = _origin()
raw = (cdp_url or "").strip()
if not raw:
return ""
lowered = raw.lower()
if "/devtools/browser/" in lowered:
return raw
discovery_url = raw
if lowered.startswith(("ws://", "wss://")):
if raw.count(":") == 2 and raw.rstrip("/").rsplit(":", 1)[-1].isdigit() and "/" not in raw.split(":", 2)[-1]:
discovery_url = ("http://" if lowered.startswith("ws://") else "https://") + raw.split("://", 1)[1]
else:
return raw
if discovery_url.lower().endswith("/json/version"):
version_url = discovery_url
else:
version_url = discovery_url.rstrip("/") + "/json/version"
try:
import requests # lazy — shared module object, test patches still apply
response = requests.get(version_url, timeout=10)
response.raise_for_status()
payload = response.json()
except Exception as exc:
_bt.logger.warning(
"Failed to resolve CDP endpoint %s via %s: %s",
_bt._sanitize_url_for_logs(raw),
_bt._sanitize_url_for_logs(version_url),
_bt._sanitize_url_for_logs(exc),
)
return raw
ws_url = str(payload.get("webSocketDebuggerUrl") or "").strip()
if ws_url:
_bt.logger.info(
"Resolved CDP endpoint %s -> %s",
_bt._sanitize_url_for_logs(raw),
_bt._sanitize_url_for_logs(ws_url),
)
return ws_url
_bt.logger.warning(
"CDP discovery at %s did not return webSocketDebuggerUrl; using raw endpoint",
_bt._sanitize_url_for_logs(version_url),
)
return raw
def _get_cdp_override_raw() -> str:
"""Return the *configured* CDP override without any network I/O.
Precedence: ``BROWSER_CDP_URL`` env (live ``/browser connect`` override),
then ``browser.cdp_url`` in config.yaml. Callers that only need to know
*whether* an override exists (check_fn gates, ``_is_local_mode`` /
``_is_local_backend``, ``hermes doctor``) MUST use this, not
:func:`_get_cdp_override`: that one does a 10s HTTP discovery, and a stale
``cdp_url`` pointing at a dead Chrome would stall every startup's schema
build with no error — no side effects during schema build.
"""
_bt = _origin()
env_override = os.environ.get("BROWSER_CDP_URL", "").strip()
if env_override:
return env_override
return _bt._browser_cfg(
"cdp_url", "", lambda v: str(v or "").strip(), "browser.cdp_url from config"
)
def _get_cdp_override() -> str:
"""Return the resolved CDP URL override, or "" (skips cloud AND local launch).
May perform an HTTP ``/json/version`` discovery request — only call on
paths about to *connect* (session creation, supervisor attach); pure
is-it-configured gates must use :func:`_get_cdp_override_raw`.
"""
_bt = _origin()
raw = _bt._get_cdp_override_raw()
if not raw:
return ""
return _bt._resolve_cdp_override(raw)
def _get_dialog_policy_config() -> Tuple[str, float]:
"""Read ``browser.dialog_policy`` + ``browser.dialog_timeout_s`` from config.
Returns a ``(policy, timeout_s)`` tuple, falling back to the supervisor's
defaults when keys are absent or invalid.
"""
# Defer imports so browser_tool can be imported in minimal environments.
_bt = _origin()
from tools.browser_supervisor import (
DEFAULT_DIALOG_POLICY, DEFAULT_DIALOG_TIMEOUT_S, _VALID_POLICIES
)
try:
from hermes_cli.config import read_raw_config
cfg = read_raw_config()
browser_cfg = cfg.get("browser", {}) if isinstance(cfg, dict) else {}
if not isinstance(browser_cfg, dict):
return DEFAULT_DIALOG_POLICY, DEFAULT_DIALOG_TIMEOUT_S
policy = str(browser_cfg.get("dialog_policy") or DEFAULT_DIALOG_POLICY)
if policy not in _VALID_POLICIES:
_bt.logger.debug("Invalid browser.dialog_policy=%r; using default", policy)
policy = DEFAULT_DIALOG_POLICY
timeout_raw = browser_cfg.get("dialog_timeout_s")
try:
timeout_s = float(timeout_raw) if timeout_raw is not None else DEFAULT_DIALOG_TIMEOUT_S
if timeout_s <= 0:
timeout_s = DEFAULT_DIALOG_TIMEOUT_S
except (TypeError, ValueError):
timeout_s = DEFAULT_DIALOG_TIMEOUT_S
return policy, timeout_s
except Exception:
return DEFAULT_DIALOG_POLICY, DEFAULT_DIALOG_TIMEOUT_S
def _ensure_cdp_supervisor(task_id: str) -> None:
"""Start a CDP supervisor for ``task_id`` if an endpoint is reachable.
Idempotent (``SupervisorRegistry.get_or_start`` skips an existing
``(task_id, cdp_url)`` and restarts on URL change), so safe on every
navigate / ``/browser connect``. URL precedence: the CDP override, then the
session's own ``cdp_url`` (cloud providers). Swallows all errors — a failed
attach must not break the session; snapshots just lack
``pending_dialogs`` / ``frame_tree``.
"""
_bt = _origin()
cdp_url = _bt._get_cdp_override()
if not cdp_url:
# Fallback: active session may carry a per-session CDP URL from a
# cloud provider (Browserbase sets this).
with _bt._cleanup_lock:
session_info = _bt._active_sessions.get(task_id, {})
maybe = str(session_info.get("cdp_url") or "")
if maybe:
cdp_url = _bt._resolve_cdp_override(maybe)
if not cdp_url:
return
try:
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
policy, timeout_s = _bt._get_dialog_policy_config()
SUPERVISOR_REGISTRY.get_or_start(
task_id=task_id, cdp_url=cdp_url, dialog_policy=policy, dialog_timeout_s=timeout_s
)
except Exception as exc:
_bt.logger.debug(
"CDP supervisor attach for task=%s failed (non-fatal): %s", task_id, exc
)
def _stop_cdp_supervisor(task_id: str) -> None:
"""Stop the CDP supervisor for ``task_id`` if one exists. No-op otherwise."""
_bt = _origin()
try:
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
SUPERVISOR_REGISTRY.stop(task_id)
except Exception as exc:
_bt.logger.debug("CDP supervisor stop for task=%s failed (non-fatal): %s", task_id, exc)
+368
View File
@@ -0,0 +1,368 @@
"""Cloud browser provider resolution (explicit browser.cloud_provider, auto-detect, per-profile cache), backend/engine selection and headed-mode flags.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
from __future__ import annotations
import os
from typing import Optional
from agent.browser_provider import BrowserProvider as CloudBrowserProvider
from tools.browser_tool_origin import origin_module as _origin
def _is_legacy_provider_registry_overridden() -> bool:
"""True when a test has patched ``_PROVIDER_REGISTRY`` to a custom value.
Each registered value is compared by identity against the canonical class
in ``_DEFAULT_PROVIDER_REGISTRY`` (extra keys count too); adding a built-in
provider only requires extending that default dict.
"""
_bt = _origin()
try:
for key, default_cls in _bt._DEFAULT_PROVIDER_REGISTRY.items():
if _bt._PROVIDER_REGISTRY.get(key) is not default_cls:
return True
# Extra keys not in the default registry → also an override.
return len(_bt._PROVIDER_REGISTRY) != len(_bt._DEFAULT_PROVIDER_REGISTRY)
except Exception:
return False
def _ensure_browser_plugins_loaded() -> None:
"""Idempotently trigger plugin discovery so the browser registry is populated.
``model_tools`` normally does this as an import side effect, but
``_get_cloud_provider`` is also reached from standalone scripts and test
harnesses that never import it; cheap on repeat calls.
"""
_bt = _origin()
try:
from hermes_cli.plugins import _ensure_plugins_discovered
_ensure_plugins_discovered()
except Exception as exc:
_bt.logger.debug("Browser plugin discovery failed (non-fatal): %s", exc)
def _get_cloud_provider() -> Optional[CloudBrowserProvider]:
"""Return the provider cached for the active Hermes profile."""
_bt = _origin()
scope = _bt.hermes_home_key()
with _bt._cloud_provider_cache_lock:
# Tests and legacy reset paths clear the boolean. Treat that as a full
# reset even if a previous scoped resolution remains mirrored here.
if not _bt._cloud_provider_resolved:
_bt._cached_cloud_provider_scope = None
_bt._cached_cloud_providers.clear()
while True:
before_generation = _bt._browser_registry_generation(scope=scope)
cache_key = (scope, before_generation)
if cache_key in _bt._cached_cloud_providers:
_bt._cached_cloud_provider = _bt._cached_cloud_providers[cache_key]
_bt._cloud_provider_resolved = True
_bt._cached_cloud_provider_scope = scope
return _bt._cached_cloud_provider
_bt._cached_cloud_provider = None
_bt._cloud_provider_resolved = False
resolved = _bt._resolve_cloud_provider_uncached()
after_generation = _bt._browser_registry_generation(scope=scope)
if before_generation != after_generation:
# A force reload replaced/unloaded this profile's provider
# while resolution was in progress. Discard the stale result
# and resolve against the new registry generation.
continue
if _bt._cloud_provider_resolved:
_bt._cached_cloud_provider_scope = scope
for stale_key in [
key for key in _bt._cached_cloud_providers if key[0] == scope
]:
_bt._cached_cloud_providers.pop(stale_key, None)
_bt._cached_cloud_providers[cache_key] = resolved
return resolved
def _instantiate_explicit_cloud_provider(provider_key: str) -> Optional[CloudBrowserProvider]:
"""Build the provider named by ``browser.cloud_provider``.
Test fixtures that patch ``_PROVIDER_REGISTRY`` drive the legacy dict;
otherwise the plugin registry is consulted (after idempotent discovery).
Strict selection: a stored-but-unregistered name raises ``ValueError``
(never a silent reroute to auto-detect). Any other instantiation error is
logged and yields None so the next call retries.
"""
_bt = _origin()
try:
if _bt._is_legacy_provider_registry_overridden():
factory = _bt._PROVIDER_REGISTRY.get(provider_key)
resolved = factory() if factory is not None else None
else:
_bt._ensure_browser_plugins_loaded()
resolved = _bt._registry_get_browser_provider(provider_key)
if resolved is None:
from tools.tool_backend_helpers import selection_error
raise ValueError(selection_error(
"browser",
f"'{provider_key}'",
"no registered browser plugin has that name (install "
"the corresponding plugin or fix the config key "
"spelling)",
))
return resolved
except ValueError:
raise
except Exception:
_bt.logger.warning(
"Failed to instantiate explicit cloud_provider %r; will retry on next call",
provider_key,
exc_info=True,
)
return None
def _autodetect_cloud_provider() -> Optional[CloudBrowserProvider]:
"""Auto-detect: Browser Use (managed Nous gateway or API key), then Browserbase.
Uses the legacy class names bound on this module so tests that
``monkeypatch.setattr(browser_tool, "BrowserUseProvider", ...)`` keep
driving this branch. Third-party plugins are intentionally NOT reachable
from auto-detect — only via explicit ``browser.cloud_provider: <name>``.
Never raises (a failure must not poison the cache).
"""
_bt = _origin()
try:
for cls in (_bt.BrowserUseProvider, _bt.BrowserbaseProvider):
fallback_provider = cls()
if fallback_provider.is_configured():
return fallback_provider
except Exception: # pragma: no cover - defensive: never poison cache
_bt.logger.debug("Cloud provider auto-detect failed", exc_info=True)
return None
def _resolve_cloud_provider_uncached() -> Optional[CloudBrowserProvider]:
"""Return the configured cloud browser provider, or None for local mode.
Reads ``browser.cloud_provider`` and pins the result in the cache only when
it is definitive (explicit ``local``/``camofox``, or a resolved provider).
Explicit selection routes through :mod:`agent.browser_registry` so
third-party plugins participate; auto-detect (only when no selection was
ever written) walks Browser Use then Browserbase. A transient None
(unreadable config, missing credentials) is NOT cached so it can self-heal.
"""
_bt = _origin()
resolved: Optional[CloudBrowserProvider] = None
provider_key = None
try:
from hermes_cli.config import read_raw_config
browser_cfg = read_raw_config().get("browser", {})
if isinstance(browser_cfg, dict) and "cloud_provider" in browser_cfg:
provider_key = _bt.normalize_browser_cloud_provider(browser_cfg.get("cloud_provider"))
if provider_key in ("local", "camofox"):
# Camofox runs through the built-in browser tools, not a cloud provider.
_bt._cached_cloud_provider = None
_bt._cloud_provider_resolved = True
return None
if provider_key == "nous":
# Managed "Nous Subscription" is serviced by the Browser Use provider.
provider_key = "browser-use"
if provider_key:
resolved = _bt._instantiate_explicit_cloud_provider(provider_key)
if resolved is None:
return None
except ValueError:
raise
except Exception as e:
# Config may be temporarily unreadable; still try auto-detect so
# env-based / managed-gateway credentials can resolve. Don't pin cache.
_bt.logger.debug("Could not read cloud_provider from config: %s", e)
if resolved is None and provider_key is None:
resolved = _bt._autodetect_cloud_provider()
if resolved is None:
return None
_bt._cached_cloud_provider = resolved
_bt._cloud_provider_resolved = True
return _bt._cached_cloud_provider
def _is_local_mode() -> bool:
"""Return True when the browser tool will use a local browser backend."""
_bt = _origin()
if _bt._get_cdp_override_raw():
return False
return _bt._get_cloud_provider() is None
def _is_local_backend() -> bool:
"""Return True when the browser runs locally AND the terminal is also local.
SSRF protection only matters when the browser can reach networks the user's
terminal cannot: cloud backends, and a local browser paired with a
containerized terminal (docker/modal/daytona/ssh/singularity). A CDP
override is never trusted as local (that Chrome may live off-host) and MUST
be checked before the Camofox short-circuit so Camofox + override still
fails the local check; ``_is_local_mode`` treats overrides the same way —
keep the two in agreement.
"""
_bt = _origin()
if _bt._get_cdp_override_raw():
return False
if _bt._is_camofox_mode():
return True
if _bt._get_cloud_provider() is not None:
return False
# Scope-aware: under gateway multiplexing the routed profile's terminal
# backend lives in the per-turn terminal scope, not the process env.
from tools.terminal_scope import terminal_env
terminal_backend = terminal_env("TERMINAL_ENV", "local").strip().lower()
return terminal_backend in ("local", "")
def _get_browser_engine() -> str:
"""Return the browser engine: ``auto`` (no ``--engine`` flag), ``lightpanda`` or ``chrome``.
``browser.engine`` first, then ``AGENT_BROWSER_ENGINE``, then ``auto``;
cached. Lightpanda is much faster on navigation but has no graphical
renderer (no screenshots).
"""
_bt = _origin()
if _bt._browser_engine_resolved:
return _bt._cached_browser_engine
_bt._browser_engine_resolved = True
# Config file takes priority; env var only if config didn't set a value.
_bt._cached_browser_engine = _bt._browser_cfg(
"engine", "auto",
lambda v: str(v).strip().lower() if v and str(v).strip() else "auto",
"browser.engine from config",
)
if _bt._cached_browser_engine == "auto":
env_val = os.environ.get("AGENT_BROWSER_ENGINE", "").strip().lower()
if env_val:
_bt._cached_browser_engine = env_val
# Validate: agent-browser only accepts "chrome" and "lightpanda".
_VALID_ENGINES = {"auto", "lightpanda", "chrome"}
if _bt._cached_browser_engine not in _VALID_ENGINES:
_bt.logger.warning(
"Unknown browser engine %r (valid: %s), falling back to 'auto'",
_bt._cached_browser_engine, ", ".join(sorted(_VALID_ENGINES)),
)
_bt._cached_browser_engine = "auto"
return _bt._cached_browser_engine
def _is_headed_mode() -> bool:
"""Return True when the browser should launch in headed (visible) mode.
Reads ``config["browser"]["headed"]`` with ``AGENT_BROWSER_HEADED`` env
var as fallback. Result is cached after the first call.
"""
_bt = _origin()
if _bt._headed_mode_resolved:
return _bt._cached_headed_mode # type: ignore[return-value]
_bt._headed_mode_resolved = True
_bt._cached_headed_mode = _bt._browser_cfg(
"headed", False,
lambda v: False if v is None else str(v).strip().lower() in ("true", "1", "yes"),
"browser.headed from config",
)
if not _bt._cached_headed_mode:
env_val = os.environ.get("AGENT_BROWSER_HEADED", "").strip()
if env_val and env_val.lower() in ("true", "1", "yes"):
_bt._cached_headed_mode = True
return _bt._cached_headed_mode
def _should_inject_engine(engine: str) -> bool:
"""Return True when the engine flag should be added to agent-browser commands.
Only inject ``--engine`` for non-cloud, non-camofox local sessions where
the engine is explicitly set (not ``auto``).
"""
_bt = _origin()
if engine == "auto":
return False
if _bt._is_camofox_mode():
return False
return _bt._is_local_mode()
def _auto_local_for_private_urls() -> bool:
"""``browser.auto_local_for_private_urls`` (default True), cached for the process.
When on, ``browser_navigate`` routes private/loopback/LAN URLs to a local
Chromium sidecar even with a cloud provider configured; public URLs keep
using the cloud provider in the same conversation.
"""
_bt = _origin()
if _bt._auto_local_for_private_urls_resolved:
return _bt._cached_auto_local_for_private_urls
_bt._auto_local_for_private_urls_resolved = True
_bt._cached_auto_local_for_private_urls = _bt._browser_cfg(
"auto_local_for_private_urls", _bt._cached_auto_local_for_private_urls, bool,
"auto_local_for_private_urls from config",
)
return _bt._cached_auto_local_for_private_urls
def _use_real_profile() -> bool:
"""Return whether the user consented to real-profile local browsing.
Reads ``browser.use_real_profile`` (default False) on EVERY call — it is a
consent switch, so flipping it off must take effect without a restart, and
in a multiplexed gateway each profile's config must decide for itself.
The read is one YAML load per local session creation (not per command),
so there is no hot-path cost to keeping it uncached.
"""
_bt = _origin()
return _bt._browser_cfg("use_real_profile", False, bool, "use_real_profile from config")
def _allow_private_urls() -> bool:
"""Return whether the browser is allowed to navigate to private/internal addresses.
Reads ``config["browser"]["allow_private_urls"]``. Single-profile calls
cache the result for the process lifetime; multiplexed profile turns resolve
their context-local config on each call. Defaults to ``False`` (SSRF
protection active).
"""
_bt = _origin()
# The profile multiplexer scopes config with a ContextVar while sharing
# this module. Never reuse another profile's private-network opt-out.
if _bt.get_hermes_home_override() is not None:
return _bt._resolve_allow_private_urls()
if _bt._allow_private_urls_resolved:
return _bt._cached_allow_private_urls
_bt._allow_private_urls_resolved = True
_bt._cached_allow_private_urls = _bt._resolve_allow_private_urls()
return _bt._cached_allow_private_urls
def _resolve_allow_private_urls() -> bool:
"""Read the browser private-URL toggle from the active config scope."""
_bt = _origin()
return _bt._browser_cfg(
"allow_private_urls", False,
lambda v: _bt.is_truthy_value(v, default=False),
"allow_private_urls from config",
)
+3 -6
View File
@@ -65,13 +65,11 @@ def _current_page_private_url(effective_task_id: str) -> Optional[str]:
_bt = _origin()
try:
url_result = _bt._run_browser_command(
effective_task_id, "eval", ["window.location.href"],
timeout=5, _engine_override="auto",
effective_task_id, "eval", ["window.location.href"], timeout=5, _engine_override="auto"
)
if url_result.get("success"):
current_url = (
url_result.get("data", {}).get("result", "")
.strip().strip('"').strip("'")
url_result.get("data", {}).get("result", "") .strip().strip('"').strip("'")
)
if current_url and (
_bt._is_always_blocked_url(current_url) or not _bt._is_safe_url(current_url)
@@ -96,8 +94,7 @@ _RISKY_BROWSER_EVAL_PATTERNS: tuple[tuple[re.Pattern[str], str], ...] = (
_JS_STRING_LITERAL_RE = re.compile(
r"""'(?:\\.|[^'\\])*'|\"(?:\\.|[^\"\\])*\"|`(?:\\.|[^`\\])*`""",
re.S,
r"""'(?:\\.|[^'\\])*'|\"(?:\\.|[^\"\\])*\"|`(?:\\.|[^`\\])*`""", re.S
)
+490
View File
@@ -0,0 +1,490 @@
"""agent-browser / Chromium discovery and install: PATH merging, npx resolution, candidate binaries, Chromium detection + auto-install, requirement checks.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
import functools
import os
import shutil
import subprocess
import sys
from pathlib import Path
from typing import List, Optional
from tools.browser_tool_origin import origin_module as _origin
@functools.lru_cache(maxsize=1)
def _discover_homebrew_node_dirs() -> tuple[str, ...]:
"""Find Homebrew versioned Node.js bin directories (e.g. node@20, node@24).
When Node is installed via ``brew install node@24`` and NOT linked into
/opt/homebrew/bin, agent-browser isn't discoverable on the default PATH.
This function finds those directories so they can be prepended.
"""
dirs: list[str] = []
homebrew_opt = "/opt/homebrew/opt"
if not os.path.isdir(homebrew_opt):
return tuple(dirs)
try:
for entry in os.listdir(homebrew_opt):
if entry.startswith("node") and entry != "node":
bin_dir = os.path.join(homebrew_opt, entry, "bin")
if os.path.isdir(bin_dir):
dirs.append(bin_dir)
except OSError:
pass
return tuple(dirs)
def _browser_candidate_path_dirs() -> list[str]:
"""Return ordered browser CLI PATH candidates shared by discovery and execution."""
_bt = _origin()
hermes_home = _bt.get_hermes_home()
hermes_node_bin = str(hermes_home / "node" / "bin")
hermes_node_root = str(hermes_home / "node")
hermes_nm_bin = str(hermes_home / "node_modules" / ".bin")
return [hermes_node_bin, hermes_node_root, hermes_nm_bin, *list(_bt._discover_homebrew_node_dirs()), *_bt._SANE_PATH_DIRS]
def _merge_browser_path(existing_path: str = "") -> str:
"""Prepend browser-specific PATH fallbacks without reordering existing entries."""
_bt = _origin()
path_parts = [p for p in (existing_path or "").split(os.pathsep) if p]
existing_parts = set(path_parts)
prefix_parts: list[str] = []
for part in _bt._browser_candidate_path_dirs():
if not part or part in existing_parts or part in prefix_parts:
continue
if os.path.isdir(part):
prefix_parts.append(part)
return os.pathsep.join(prefix_parts + path_parts)
def _browser_install_hint() -> str:
_bt = _origin()
if _bt._is_termux_environment():
return "npm install -g agent-browser && agent-browser install"
return "npm install -g agent-browser && agent-browser install --with-deps"
def _is_npx_agent_browser_sentinel(browser_cmd: str) -> bool:
_bt = _origin()
return browser_cmd.strip() == _bt.NPX_AGENT_BROWSER_SENTINEL
def _requires_real_termux_browser_install(browser_cmd: str) -> bool:
_bt = _origin()
return _bt._is_termux_environment() and _bt._is_local_mode() and _bt._is_npx_agent_browser_sentinel(browser_cmd)
def _termux_browser_install_error() -> str:
_bt = _origin()
return (
"Local browser automation on Termux cannot rely on the bare npx fallback. "
f"Install agent-browser explicitly first: {_bt._browser_install_hint()}"
)
def _agent_browser_candidate_present(path: str | None) -> bool:
if not path:
return False
if " " in path and path.split()[0].endswith("npx"):
return True
return os.path.exists(path) and (os.name == "nt" or os.access(path, os.X_OK))
def _resolve_npx_bin() -> Optional[str]:
"""Resolve a runnable npx, preferring the Hermes-managed/Homebrew extended PATH.
Bare PATH first would let a broken system npx shadow a healthy managed one
with no recovery, so every candidate is validated with ``node_tool_runnable``
before being trusted.
"""
_bt = _origin()
extended_path = _bt._merge_browser_path("")
if extended_path:
extended_npx = shutil.which("npx", path=extended_path)
if extended_npx and _bt.node_tool_runnable(extended_npx):
return extended_npx
npx_path = shutil.which("npx")
if npx_path and _bt.node_tool_runnable(npx_path):
return npx_path
return None
def _agent_browser_candidates(extended_path: str):
"""Yield agent-browser lookup candidates in resolution order (lazily — each is a filesystem probe).
Order: ambient PATH (global install) → extended PATH (Hermes-managed Node,
macOS versioned Homebrew, Termux/system dirs) → repo-local node_modules/.bin.
The local lookup goes through ``shutil.which`` with an explicit path so
Windows resolves the ``.cmd`` shim (CreateProcess cannot run npm's
extensionless POSIX shim — WinError 193) while POSIX keeps the plain one.
"""
yield shutil.which("agent-browser")
if extended_path:
yield shutil.which("agent-browser", path=extended_path)
local_bin_dir = Path(__file__).parent.parent / "node_modules" / ".bin"
if local_bin_dir.is_dir():
yield shutil.which("agent-browser", path=str(local_bin_dir))
def _find_agent_browser(*, validate: bool = True) -> str:
"""
Find the agent-browser CLI executable.
Checks in order: current PATH, Homebrew/common bin dirs, Hermes-managed
node, local node_modules/.bin/, npx fallback, then a lazy install.
Every candidate is validated with ``agent_browser_runnable`` before it is
cached. A bare ``shutil.which`` hit is NOT trusted: agent-browser's npm
postinstall re-points a global symlink at our local node_modules binary,
which disappears on the next ``hermes update`` and leaves a dangling link
that ``which`` still reports but exec fails on (exit 127). Validating lets a
dead candidate fall through instead of being cached and killing every
browser tool. ``validate=False`` (schema-time check_fn) only tests presence
and never caches.
Raises:
FileNotFoundError: If agent-browser is not installed
"""
_bt = _origin()
if _bt._agent_browser_resolved:
if _bt._cached_agent_browser is None:
raise FileNotFoundError(
"agent-browser CLI not found (cached). Install it with: "
f"{_bt._browser_install_hint()}\n"
"Or ensure npx is available in your PATH."
)
return _bt._cached_agent_browser
def _accept(candidate: str) -> str:
# _agent_browser_resolved is set at each accept site (not before the
# search) so a concurrent reader never sees resolved=True with a None cache.
if validate:
_bt._cached_agent_browser = candidate
_bt._agent_browser_resolved = True
return candidate
ok = _bt.agent_browser_runnable if validate else _bt._agent_browser_candidate_present
extended_path = _bt._merge_browser_path("")
for candidate in _bt._agent_browser_candidates(extended_path):
if candidate and ok(candidate):
return _accept(candidate)
# npx fallback (also searches the extended PATH)
if _bt._resolve_npx_bin():
return _accept(_bt.NPX_AGENT_BROWSER_SENTINEL)
if not validate:
raise FileNotFoundError("agent-browser CLI not found")
# Nothing found — try lazy installation before giving up.
try:
from hermes_cli.dep_ensure import ensure_dependency
if ensure_dependency("browser"):
candidates = [
shutil.which("agent-browser"),
shutil.which("agent-browser", path=extended_path) if extended_path else None,
shutil.which("agent-browser", path=str(_bt.get_hermes_home() / "node_modules" / ".bin")),
shutil.which("agent-browser", path=str(_bt.get_hermes_home() / "node" / "bin")),
shutil.which("agent-browser", path=str(_bt.get_hermes_home() / "node")),
]
for recheck in candidates:
if recheck and _bt.agent_browser_runnable(recheck):
return _accept(recheck)
except Exception:
pass
_bt._agent_browser_resolved = True
raise FileNotFoundError(
"agent-browser CLI not found. Install it with: "
f"{_bt._browser_install_hint()}\n"
"Or ensure npx is available in your PATH."
)
def warm_agent_browser_npx_cache(timeout: float = 60.0) -> bool:
"""Best-effort pre-fetch of the agent-browser npm package via npx.
agent-browser resolves lazily via ``npx agent-browser`` (not a root
package.json dependency), so the first invocation in a session would pay
npx's registry fetch; ``hermes update`` / ``hermes doctor --fix`` call this
to warm the cache first. Runs with the credential-scrubbed env every other
agent-browser spawn uses (registry-fetched npm code must never see the
operator keyring), in its own process group, and tree-kills on timeout so
a surviving descendant cannot hold the capture pipe open.
Never raises; True only when npx actually exited 0.
"""
_bt = _origin()
npx_bin = _bt._resolve_npx_bin()
if not npx_bin:
return False
env = _bt._build_browser_env()
env["PATH"] = _bt._merge_browser_path(env.get("PATH", ""))
popen_kwargs: dict = {
"stdout": subprocess.PIPE,
"stderr": subprocess.PIPE,
"text": True,
"env": env,
"creationflags": _bt.windows_hide_flags(),
}
if os.name == "posix":
popen_kwargs["start_new_session"] = True
else:
popen_kwargs["creationflags"] |= getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
cmd = [
npx_bin,
# --ignore-scripts: AGENT_BROWSER_NPX_SPEC is a floating ^0.26.0
# range, not an exact pin — a compromised future 0.26.x patch must
# not get to run its own install-time lifecycle scripts here.
"--ignore-scripts",
# --prefer-offline: once cached, repeat `hermes update`/`doctor
# --fix` runs shouldn't hit the registry just to re-confirm
# "latest" is still latest — that would defeat the point of
# warming the cache in the first place.
"--prefer-offline",
"-y",
_bt.AGENT_BROWSER_NPX_SPEC,
"--version",
]
try:
proc = subprocess.Popen(cmd, stdin=subprocess.DEVNULL, **popen_kwargs)
except Exception:
return False
try:
proc.communicate(timeout=timeout)
return proc.returncode == 0
except subprocess.TimeoutExpired:
_bt._kill_process_tree(proc)
try:
proc.communicate(timeout=5)
except Exception:
pass
return False
except Exception:
_bt._kill_process_tree(proc)
return False
def _chromium_search_roots() -> List[str]:
"""Directories to scan for a Chromium / headless-shell build, in the order
agent-browser and Playwright probe them: ``PLAYWRIGHT_BROWSERS_PATH``, then
Playwright's per-OS default cache."""
roots: List[str] = []
env_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH", "").strip()
if env_path and env_path != "0":
roots.append(env_path)
home = os.path.expanduser("~")
roots.append(os.path.join(home, ".cache", "ms-playwright"))
if sys.platform == "darwin":
roots.append(os.path.join(home, "Library", "Caches", "ms-playwright"))
if sys.platform == "win32":
local = os.environ.get("LOCALAPPDATA") or os.path.join(
home, "AppData", "Local"
)
roots.append(os.path.join(local, "ms-playwright"))
return roots
def _chromium_installed() -> bool:
"""Return True when a usable Chromium (or headless-shell) build is on disk.
Checks ``AGENT_BROWSER_EXECUTABLE_PATH``, then system Chrome/Chromium on
PATH, then Playwright's cache (``chromium-*`` / ``chromium_headless_shell-*``
dirs). Without a binary the CLI hangs on first use until the command
timeout fires, so the tool must not be advertised.
"""
_bt = _origin()
if _bt._cached_chromium_installed is not None:
return _bt._cached_chromium_installed
# 1. AGENT_BROWSER_EXECUTABLE_PATH — explicit user-configured browser
ab_path = os.environ.get("AGENT_BROWSER_EXECUTABLE_PATH", "").strip()
if ab_path and (os.path.isfile(ab_path) or shutil.which(ab_path)):
_bt._cached_chromium_installed = True
return True
# 2. System Chrome/Chromium in PATH (common names)
system_chrome = (
shutil.which("google-chrome")
or shutil.which("chromium")
or shutil.which("chromium-browser")
or shutil.which("chrome")
)
if system_chrome:
_bt._cached_chromium_installed = True
return True
# 3. Playwright browser cache (legacy — chromium-* / chromium_headless_shell-* dirs)
for root in _bt._chromium_search_roots():
if not root or not os.path.isdir(root):
continue
try:
entries = os.listdir(root)
except OSError:
continue
# Playwright names them ``chromium-<build>`` and
# ``chromium_headless_shell-<build>``; agent-browser accepts either.
for entry in entries:
if entry.startswith("chromium-") or entry.startswith(
"chromium_headless_shell-"
):
_bt._cached_chromium_installed = True
return True
_bt._cached_chromium_installed = False
return False
def _maybe_autoinstall_chromium() -> bool:
"""Best-effort, gated download of the Chromium *binary* on local cold start.
Binary only (``agent-browser install``), never ``--with-deps`` — that shells
``apt`` and needs root, so missing system libraries stay a user action.
Gated by ``security.allow_lazy_installs``, skipped in Docker (Chromium ships
in the image), attempted once per process. True only when Chromium is
present afterwards.
"""
_bt = _origin()
if _bt._chromium_autoinstall_attempted:
return _bt._chromium_installed()
_bt._chromium_autoinstall_attempted = True
if _bt._running_in_docker():
return False
from tools.lazy_deps import _allow_lazy_installs
if not _allow_lazy_installs():
return False
try:
browser_cmd = _bt._find_agent_browser()
except FileNotFoundError:
return False
if _bt._is_npx_agent_browser_sentinel(browser_cmd):
install_cmd = [
_bt._resolve_npx_bin() or "npx", "--ignore-scripts", "-y", _bt.AGENT_BROWSER_NPX_SPEC, "install",
]
else:
install_cmd = [browser_cmd, "install"]
_bt.logger.info(
"browser: Chromium missing — auto-installing the browser binary "
"(one-time ~170MB; disable via security.allow_lazy_installs)"
)
try:
proc = subprocess.run(
install_cmd,
capture_output=True,
text=True, encoding='utf-8', errors='replace',
timeout=600,
env=_bt._build_browser_env(),
)
except (OSError, subprocess.SubprocessError) as e:
_bt.logger.warning("browser: Chromium auto-install failed to start: %s", e)
return False
if proc.returncode != 0:
tail = (proc.stderr or proc.stdout or "").strip()[-300:]
_bt.logger.warning(
"browser: Chromium auto-install exited %s: %s", proc.returncode, tail
)
return False
_bt._cached_chromium_installed = None
return _bt._chromium_installed()
def _running_in_docker() -> bool:
"""Best-effort detection of whether we're inside a Docker container."""
if os.path.exists("/.dockerenv"):
return True
try:
with open("/proc/1/cgroup", "rt", encoding="utf-8") as fp:
return "docker" in fp.read()
except OSError:
return False
def check_browser_requirements() -> bool:
"""Whether the browser tools should be advertised.
Local mode needs the ``agent-browser`` CLI plus a Chromium build (except
Lightpanda-only text workflows); cloud mode needs the CLI plus provider
credentials (the provider hosts its own Chromium).
"""
# Browser Use CLI backend — browser_exec replaces the whole browser_*
# surface (including browser_cdp/browser_dialog, whose check_fns funnel
# through here), so hide these tools from the model.
_bt = _origin()
if _bt._is_browser_use_cli_mode():
return False
# Camofox backend — only needs the server URL, no agent-browser CLI
if _bt._is_camofox_mode():
return True
# CDP override mode can connect to an existing remote/local browser endpoint
# without requiring the local agent-browser binary on PATH.
# Raw (no-I/O) check: this runs during tool-schema assembly at startup,
# where a stale endpoint must not cost a blocking HTTP probe.
if _bt._get_cdp_override_raw():
return True
# The agent-browser CLI is required for local launch and cloud-provider flows.
# Tool-schema assembly runs during Desktop startup; do not execute
# ``agent-browser --version`` here, because Windows .cmd shims route through
# cmd.exe and can flash a console before the user invokes any browser tool.
# Actual browser execution paths still validate the candidate before use.
try:
browser_cmd = _bt._find_agent_browser(validate=False)
except FileNotFoundError:
return False
# On Termux, the bare npx fallback is too fragile to treat as a satisfied
# local browser dependency. Require a real install (global or local) so the
# browser tool is not advertised as available when it will likely fail on
# first use.
if _bt._requires_real_termux_browser_install(browser_cmd):
return False
# In cloud mode, also require provider credentials. Cloud browsers
# don't need a local Chromium binary.
provider = _bt._get_cloud_provider()
if provider is not None:
return provider.is_configured()
# Local mode with Lightpanda can provide text/navigation tools without a
# local Chromium install. Chrome fallback, screenshots, and browser_vision
# will still return actionable Chromium install errors if invoked.
if _bt._using_lightpanda_engine():
return True
# Local Chrome mode: agent-browser needs a Chromium build on disk. Without
# it the CLI hangs on first use until the command timeout fires.
return _bt._chromium_installed()
def check_browser_vision_requirements() -> bool:
"""Advertise ``browser_vision`` only with BOTH a working browser AND a vision
backend — otherwise it fails at call time with a cryptic provider error."""
_bt = _origin()
if not _bt.check_browser_requirements():
return False
try:
from tools.vision_tools import check_vision_requirements
except ImportError:
return False
return check_vision_requirements()
+864
View File
@@ -0,0 +1,864 @@
"""Browser session lifecycle: inactivity janitor, orphan reaper, per-session teardown, atexit emergency cleanup.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
import contextlib
import os
import shutil
import signal
import subprocess
import threading
import time
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, Optional, Tuple
from hermes_constants import get_hermes_home, reset_hermes_home_override, set_hermes_home_override
from tools.browser_tool_origin import origin_module as _origin
def _session_expiry_timestamp(session_info: Dict[str, Any]) -> Optional[float]:
"""Return a provider-authoritative session expiry as epoch seconds.
Cloud providers may omit ``expires_at``. Unknown or malformed values are
therefore treated as having no known expiry, preserving the existing
lifecycle for local browsers and providers without an expiry contract.
"""
_bt = _origin()
value = session_info.get("expires_at")
if isinstance(value, (int, float)) and not isinstance(value, bool):
return float(value)
if not isinstance(value, str) or not value.strip():
return None
normalized = value.strip()
if normalized.endswith(("Z", "z")):
normalized = f"{normalized[:-1]}+00:00"
try:
parsed = datetime.fromisoformat(normalized)
except ValueError:
_bt.logger.warning("Ignoring invalid cloud browser session expiry timestamp")
return None
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.timestamp()
def _session_has_expired(
session_info: Dict[str, Any], *, now: Optional[float] = None
) -> bool:
"""Return whether a cached browser session crossed its provider deadline."""
_bt = _origin()
expires_at = _bt._session_expiry_timestamp(session_info)
if expires_at is None:
return False
return (time.time() if now is None else now) >= expires_at
def _emergency_cleanup_all_sessions():
"""
Emergency cleanup of all active browser sessions.
Called on process exit or interrupt to prevent orphaned sessions.
Also runs the orphan reaper to clean up daemons left behind by previously
crashed hermes processes — this way every clean hermes exit sweeps
accumulated orphans, not just ones that actively used the browser tool.
"""
_bt = _origin()
if _bt._cleanup_done:
return
_bt._cleanup_done = True
# Clean up this process's own sessions first, so their owner_pid files
# are removed before the reaper scans.
# Real-profile Chrome processes are launched directly (not by
# agent-browser), so the session cleanup below never reaps them.
try:
_bt._terminate_real_profile_chrome()
except Exception as e:
_bt.logger.debug("Real-profile chrome cleanup on exit failed: %s", e)
if _bt._active_sessions:
_bt.logger.info("Emergency cleanup: closing %s active session(s)...",
len(_bt._active_sessions))
try:
_bt.cleanup_all_browsers()
except Exception as e:
_bt.logger.error("Emergency cleanup error: %s", e)
finally:
with _bt._cleanup_lock:
_bt._active_sessions.clear()
_bt._session_last_activity.clear()
_bt._session_owner_homes.clear()
_bt._cleanup_failures.clear()
_bt._recording_sessions.clear()
# Lightpanda servers (Browser Use mode) are processes we spawned; the
# session cleanup above stops the tracked ones, this catches any that
# fell out of ``_active_sessions``.
try:
from tools.browser_lightpanda import stop_all_lightpanda
stop_all_lightpanda()
except Exception as e:
_bt.logger.debug("Lightpanda cleanup on exit failed: %s", e)
# Sweep orphans from other crashed hermes processes. Safe even if we
# never used the browser — uses owner_pid liveness to avoid reaping
# daemons owned by other live hermes processes.
try:
_bt._reap_orphaned_browser_sessions()
except Exception as e:
_bt.logger.debug("Orphan reap on exit failed: %s", e)
@contextlib.contextmanager
def _session_owner_scope(task_id: str):
"""Run under the Hermes home + secret scope owning ``task_id``'s session (no-op if unrecorded).
The janitor thread is process-global, so each teardown must re-enter its
OWN profile's scope rather than inherit the spawning profile's; never falls
through to ``os.environ``.
"""
_bt = _origin()
owner_home = _bt._session_owner_homes.get(task_id)
if owner_home is None:
yield
return
from agent.secret_scope import (
build_profile_secret_scope, reset_secret_scope, set_secret_scope
)
from hermes_cli.env_loader import hydrate_profile_secret_sources
home_token = set_hermes_home_override(owner_home)
try:
hydrate_profile_secret_sources(Path(owner_home))
secret_token = set_secret_scope(build_profile_secret_scope(Path(owner_home)))
try:
yield
finally:
reset_secret_scope(secret_token)
finally:
reset_hermes_home_override(home_token)
def _cleanup_inactive_browser_sessions():
"""Close sessions inactive longer than the timeout (called by the cleanup thread).
Each session is torn down under its owner profile's scope. A session whose
cleanup keeps failing is force-reaped after MAX_INACTIVITY_CLEANUP_FAILURES
attempts instead of retrying forever; only a successful cleanup clears its
failure count.
"""
_bt = _origin()
current_time = time.time()
sessions_to_cleanup = []
with _bt._cleanup_lock:
for task_id, last_time in list(_bt._session_last_activity.items()):
if current_time - last_time > _bt.BROWSER_SESSION_INACTIVITY_TIMEOUT:
sessions_to_cleanup.append(task_id)
for task_id in sessions_to_cleanup:
elapsed = int(current_time - _bt._session_last_activity.get(task_id, current_time))
_bt.logger.info("Cleaning up inactive session for task: %s (inactive for %ss)", task_id, elapsed)
try:
with _bt._session_owner_scope(task_id):
_bt.cleanup_browser(task_id)
with _bt._cleanup_lock:
_bt._session_last_activity.pop(task_id, None)
_bt._session_owner_homes.pop(task_id, None)
_bt._cleanup_failures.pop(task_id, None)
except Exception as e:
with _bt._cleanup_lock:
failures = _bt._cleanup_failures[task_id] = _bt._cleanup_failures.get(task_id, 0) + 1
if failures < _bt.MAX_INACTIVITY_CLEANUP_FAILURES:
_bt.logger.warning("Error cleaning up inactive session %s (attempt %d/%d): %s",
task_id, failures, _bt.MAX_INACTIVITY_CLEANUP_FAILURES, e)
continue
_bt.logger.error("Browser cleanup failed %d times for inactive session %s; "
"force-reaping: %s", failures, task_id, e)
try:
with _bt._session_owner_scope(task_id):
_bt._force_reap_browser_session(task_id)
except Exception as reap_exc:
_bt.logger.error("Force-reap of browser session %s failed: %s", task_id, reap_exc)
finally:
with _bt._cleanup_lock:
_bt._session_owner_homes.pop(task_id, None)
_bt._cleanup_failures.pop(task_id, None)
def _write_owner_pid(socket_dir: str, session_name: str) -> None:
"""Record the current hermes PID as the owner of a browser socket dir.
Written atomically to ``<socket_dir>/<session_name>.owner_pid`` so the
orphan reaper can distinguish daemons owned by a live hermes process
(don't reap) from daemons whose owner crashed (reap). Best-effort —
an OSError here just falls back to the legacy ``tracked_names``
heuristic in the reaper.
"""
_bt = _origin()
try:
path = os.path.join(socket_dir, f"{session_name}.owner_pid")
with open(path, "w", encoding="utf-8") as f:
f.write(str(os.getpid()))
except OSError as exc:
_bt.logger.debug("Could not write owner_pid file for %s: %s",
session_name, exc)
def _verify_reapable_browser_daemon(daemon_pid: int, socket_dir: str,
session_name: str) -> bool:
"""Confirm a live PID is genuinely *this* session's agent-browser daemon.
The ``.pid`` file lives in a world-writable, predictably-named temp dir and
is written by the daemon, not us: a same-user actor can plant one pointing
at a victim PID, or a recycled PID can land on an unrelated process — and
reaping is a *tree* kill, i.e. an arbitrary-process DoS. Two psutil checks
must both pass: (1) identity — ``agent-browser`` in the name or cmdline;
(2) binding — the socket dir path/basename in the cmdline, or
``AGENT_BROWSER_SOCKET_DIR`` in its environ. (2) is the real spoof defense:
an attacker would need a real daemon embedding this exact path, which they
could already signal. Fail-closed on any ambiguity (unreadable cmdline, no
match): refuse to reap and leave process and socket dir alone.
"""
_bt = _origin()
try:
import psutil
except ImportError: # psutil is a hard dep; defensive only
_bt.logger.warning(
"Refusing to reap browser daemon PID %d (session %s): "
"psutil unavailable for identity verification",
daemon_pid, session_name)
return False
try:
proc = psutil.Process(daemon_pid)
name = (proc.name() or "").lower()
cmdline = " ".join(proc.cmdline() or []).lower()
except psutil.NoSuchProcess:
# Vanished between the liveness check and now — nothing to reap.
return False
except (psutil.AccessDenied, OSError) as exc:
_bt.logger.warning(
"Refusing to reap browser daemon PID %d (session %s): "
"could not read process identity (%s)",
daemon_pid, session_name, exc)
return False
looks_like_browser = "agent-browser" in name or "agent-browser" in cmdline
if not looks_like_browser:
_bt.logger.warning(
"Refusing to reap PID %d (session %s): not an agent-browser "
"process (name=%r)", daemon_pid, session_name, name)
return False
# Binding check: the live process must reference *this* socket dir.
socket_dir_l = socket_dir.lower()
socket_base_l = os.path.basename(socket_dir).lower()
bound = socket_dir_l in cmdline or (
socket_base_l and socket_base_l in cmdline)
if not bound:
try:
env_dir = (proc.environ() or {}).get(
"AGENT_BROWSER_SOCKET_DIR", "")
bound = bool(env_dir) and os.path.normpath(env_dir) == \
os.path.normpath(socket_dir)
except (psutil.AccessDenied, psutil.NoSuchProcess, OSError):
# environ() can be denied even same-user on some platforms.
# cmdline already failed to bind — fail closed.
bound = False
if not bound:
_bt.logger.warning(
"Refusing to reap agent-browser PID %d: not bound to session "
"socket dir %s (possible recycled PID or planted pid file)",
daemon_pid, socket_dir)
return False
return True
def _socket_dir_idle_seconds(socket_dir: str) -> Optional[float]:
"""Seconds since anything in ``socket_dir`` was last written; None if unknown (fail safe).
Every command writes ``_stdout_<cmd>`` / ``_stderr_<cmd>`` there, so the
newest mtime is a last-activity marker that survives hermes restarts and
lost in-memory bookkeeping. The dir's own mtime is not enough — rewriting
an existing ``_stdout_click`` doesn't touch it — so entries are scanned too.
"""
try:
latest = os.path.getmtime(socket_dir)
except OSError:
return None
try:
with os.scandir(socket_dir) as entries:
for entry in entries:
try:
latest = max(latest, entry.stat().st_mtime)
except OSError:
continue
except OSError:
pass # dir mtime alone is still a usable lower bound
return max(0.0, time.time() - latest)
def _owner_pid_alive(socket_dir: str, session_name: str) -> Tuple[Optional[int], Optional[bool]]:
"""Read ``<session>.owner_pid`` and report ``(pid, alive)``; ``(None, None)`` when missing/corrupt."""
owner_pid_file = os.path.join(socket_dir, f"{session_name}.owner_pid")
if not os.path.isfile(owner_pid_file):
return None, None
try:
owner_pid = int(Path(owner_pid_file).read_text(encoding="utf-8").strip())
# ``os.kill(pid, 0)`` is NOT a no-op on Windows; use the cross-platform check.
from gateway.status import _pid_exists
return owner_pid, _pid_exists(owner_pid)
except (ValueError, OSError):
return None, None # corrupt file — fall through to legacy handling
def _reap_socket_dir(socket_dir: str, session_name: str, tracked_names: set) -> bool:
"""Reap one ``agent-browser-<session>`` dir if orphaned; return True when a daemon was killed.
Ownership priority: (1) a live ``owner_pid`` means another hermes process
owns it — leave it alone UNLESS it is untracked here and idle past
``BROWSER_ORPHAN_GRACE_SECONDS`` (owner-alive alone made leaked daemons
immortal: in-memory tracking is lost on any exception path and the daemon's
own idle timeout doesn't fire when it is wedged); (2) no owner_pid (legacy)
falls back to this process's tracking. A pidless dir is only stale after the
grace period — deleting it immediately races the creator's first stdout open.
The daemon PID is verified as ours before a tree-kill (world-writable dir,
recycled PIDs), and refused without a start-time fingerprint.
"""
_bt = _origin()
owner_pid, owner_alive = _bt._owner_pid_alive(socket_dir, session_name)
if owner_alive is True:
if session_name in tracked_names:
return False
idle_s = _bt._socket_dir_idle_seconds(socket_dir)
if idle_s is None or idle_s < _bt.BROWSER_ORPHAN_GRACE_SECONDS:
return False # unknown age or within grace — fail safe
_bt.logger.warning(
"Browser session %s has a live owner (PID %s) but is untracked "
"and idle for %ds (grace %ds) — treating as leaked and reaping",
session_name, owner_pid, int(idle_s),
_bt.BROWSER_ORPHAN_GRACE_SECONDS)
elif owner_alive is None and session_name in tracked_names:
return False
pid_file = os.path.join(socket_dir, f"{session_name}.pid")
if not os.path.isfile(pid_file):
idle_s = _bt._socket_dir_idle_seconds(socket_dir)
if idle_s is None or idle_s < _bt.BROWSER_ORPHAN_GRACE_SECONDS:
return False
shutil.rmtree(socket_dir, ignore_errors=True)
return False
try:
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
except (ValueError, OSError):
shutil.rmtree(socket_dir, ignore_errors=True)
return False
from gateway.status import _pid_exists
if not _pid_exists(daemon_pid):
shutil.rmtree(socket_dir, ignore_errors=True)
return False
if not _bt._verify_reapable_browser_daemon(daemon_pid, socket_dir, session_name):
return False # leave process and dir for a later sweep once the imposter PID is gone
# Tree-kill so Chromium children (renderer, GPU, ...) go too, not just the daemon.
reaped = False
try:
from gateway.status import get_process_start_time
from tools.process_registry import ProcessRegistry
daemon_start = get_process_start_time(daemon_pid)
if daemon_start is None:
_bt.logger.warning(
"Refusing to reap browser daemon PID %d (session %s): "
"no start-time fingerprint available", daemon_pid, session_name)
return False
ProcessRegistry._terminate_host_pid(daemon_pid, daemon_start)
_bt.logger.info("Reaped orphaned browser daemon PID %d (session %s)",
daemon_pid, session_name)
reaped = True
except (ProcessLookupError, PermissionError, OSError):
pass
shutil.rmtree(socket_dir, ignore_errors=True)
return reaped
def _reap_orphaned_browser_sessions():
"""Kill agent-browser daemons whose owning hermes process is gone.
When the process that created a session exits uncleanly (SIGKILL, crash,
gateway restart) the in-memory ``_active_sessions`` tracking is lost but the
node + Chromium processes keep running. Scans the tmp dir for
``agent-browser-*`` socket dirs and applies ``_reap_socket_dir``'s ownership
rules (owner_pid file first — cross-process safe, two hermes instances never
reap each other — then in-process tracking for legacy daemons).
Safe to call from any context — atexit, cleanup thread, or on demand.
"""
_bt = _origin()
import glob
# Lightpanda servers (Browser Use mode) keep their own records (no
# agent-browser socket dir); sweep them with the same owner-liveness rule
# BEFORE the daemon scan, which may return early.
try:
from tools.browser_lightpanda import reap_orphaned_lightpanda
reap_orphaned_lightpanda()
except Exception as e:
_bt.logger.debug("Lightpanda orphan reap failed: %s", e)
tmpdir = _bt._socket_safe_tmpdir()
socket_dirs = []
for prefix in ("agent-browser-h_*", "agent-browser-cdp_*", "agent-browser-hermes_*"):
socket_dirs += glob.glob(os.path.join(tmpdir, prefix))
if not socket_dirs:
return
with _bt._cleanup_lock:
tracked_names = {
info.get("session_name")
for info in _bt._active_sessions.values()
if info.get("session_name")
}
reaped = 0
for socket_dir in socket_dirs:
session_name = os.path.basename(socket_dir).removeprefix("agent-browser-")
if session_name and _bt._reap_socket_dir(socket_dir, session_name, tracked_names):
reaped += 1
if reaped:
_bt.logger.info("Reaped %d orphaned browser session(s) from previous run(s)", reaped)
def _browser_cleanup_thread_worker():
"""Every 30s: close sessions idle past BROWSER_SESSION_INACTIVITY_TIMEOUT.
Also reaps orphaned daemons on startup AND every BROWSER_ORPHAN_REAP_INTERVAL
seconds — a daemon can fall out of in-memory tracking at any point in a
long-lived process, and a startup-only reap could never recover from that.
"""
_bt = _origin()
reap_every_cycles = max(1, round(_bt.BROWSER_ORPHAN_REAP_INTERVAL / 30))
cycle = 0
while _bt._cleanup_running:
# cycle 0 is the startup reap; then every reap_every_cycles.
if cycle % reap_every_cycles == 0:
try:
_bt._reap_orphaned_browser_sessions()
except Exception as e:
_bt.logger.warning("Orphan reap error: %s", e)
cycle += 1
try:
_bt._cleanup_inactive_browser_sessions()
except Exception as e:
_bt.logger.warning("Cleanup thread error: %s", e)
# Sleep in 1-second intervals so we can stop quickly if needed
for _ in range(30):
if not _bt._cleanup_running:
break
time.sleep(1)
def _start_browser_cleanup_thread():
"""Start the background cleanup thread if not already running."""
_bt = _origin()
with _bt._cleanup_lock:
if _bt._cleanup_thread is None or not _bt._cleanup_thread.is_alive():
_bt._cleanup_running = True
_bt._cleanup_thread = threading.Thread(
target=_bt._browser_cleanup_thread_worker, daemon=True, name="browser-cleanup"
)
_bt._cleanup_thread.start()
_bt.logger.info("Started inactivity cleanup thread (timeout: %ss)", _bt.BROWSER_SESSION_INACTIVITY_TIMEOUT)
def _stop_browser_cleanup_thread():
"""Stop the background cleanup thread."""
_bt = _origin()
_bt._cleanup_running = False
if _bt._cleanup_thread is not None:
_bt._cleanup_thread.join(timeout=5)
def _update_session_activity(task_id: str):
"""Update the last activity timestamp for a session.
Also records the owning Hermes home on first sight so the process-global
janitor can tear the session down under its owner's scope. An
activity touch deliberately does NOT reset ``_cleanup_failures`` — only a
successful cleanup does.
"""
_bt = _origin()
with _bt._cleanup_lock:
_bt._session_last_activity[task_id] = time.time()
_bt._session_owner_homes.setdefault(task_id, str(get_hermes_home()))
def _kill_process_tree(proc: "subprocess.Popen") -> None:
"""Best-effort kill of *proc* and every descendant it spawned; never raises.
``Popen.kill()`` only signals the direct child. npm/npx fork helpers and
agent-browser's detached daemon grandchild, which survive a plain kill and
keep a capture pipe open so ``communicate()`` never sees EOF — on Windows
there is no non-blocking read to poll around that, so the whole tree must
go. No grace period: the caller already burned its full timeout waiting.
Delegates to :func:`agent.deadline.kill_process_tree` (taskkill /T /F,
killpg, plus a psutil sweep that reaches ``setsid``'d descendants) and
falls back to :func:`_legacy_kill_process_tree` on any failure.
"""
_bt = _origin()
try:
from agent.deadline import kill_process_tree as _deadline_kill_tree
_deadline_kill_tree(proc.pid)
except Exception:
_bt._legacy_kill_process_tree(proc)
def _legacy_kill_process_tree(proc: "subprocess.Popen") -> None:
"""Local tree-kill — fallback when agent.deadline is unavailable."""
if os.name == "nt":
try:
subprocess.run(
["taskkill", "/PID", str(proc.pid), "/T", "/F"],
check=False,
capture_output=True,
stdin=subprocess.DEVNULL,
)
except Exception:
pass
return
# os.killpg/signal.SIGKILL don't exist on Windows; this branch is
# POSIX-only (the `os.name == "nt"` check above already returns first
# on Windows), but resolve them defensively via getattr anyway so an
# accidental future refactor that drops that guard degrades to a plain
# kill() instead of AttributeError — same discipline as
# tools/mcp_stdio_watchdog.py's _terminate_process_group.
killpg = getattr(os, "killpg", None)
if killpg is None: # windows-footgun: ok - non-POSIX fallback
try:
proc.kill()
except Exception:
pass
return
try:
pgid = os.getpgid(proc.pid)
except (ProcessLookupError, OSError):
return
sigkill = getattr(signal, "SIGKILL", signal.SIGTERM)
for sig in (signal.SIGTERM, sigkill):
try:
killpg(pgid, sig)
except (ProcessLookupError, PermissionError, OSError):
return
def _pid_exists(pid: int) -> bool:
"""Best-effort 'is this PID alive' check (signal 0 / psutil on Windows)."""
if pid <= 0:
return False
if os.name == "nt":
try:
import psutil
return psutil.pid_exists(pid)
except Exception:
return False
try:
os.kill(pid, 0) # windows-footgun: ok — psutil.pid_exists above handles Windows
except ProcessLookupError:
return False
except PermissionError:
return True
except OSError:
return False
return True
def _cleanup_old_screenshots(screenshots_dir, max_age_hours=24):
"""Remove browser screenshots older than max_age_hours to prevent disk bloat.
Throttled to run at most once per hour per directory to avoid repeated
scans on screenshot-heavy workflows.
"""
_bt = _origin()
key = str(screenshots_dir)
now = time.time()
if now - _bt._last_screenshot_cleanup_by_dir.get(key, 0.0) < 3600:
return
_bt._last_screenshot_cleanup_by_dir[key] = now
try:
cutoff = time.time() - (max_age_hours * 3600)
for f in screenshots_dir.glob("browser_screenshot_*.png"):
try:
if f.stat().st_mtime < cutoff:
f.unlink()
except Exception as e:
_bt.logger.debug("Failed to clean old screenshot %s: %s", f, e)
except Exception as e:
_bt.logger.debug("Screenshot cleanup error (non-critical): %s", e)
def _cleanup_old_recordings(max_age_hours=72):
"""Remove browser recordings older than max_age_hours to prevent disk bloat."""
_bt = _origin()
try:
hermes_home = get_hermes_home()
recordings_dir = hermes_home / "browser_recordings"
if not recordings_dir.exists():
return
cutoff = time.time() - (max_age_hours * 3600)
for f in recordings_dir.glob("session_*.webm"):
try:
if f.stat().st_mtime < cutoff:
f.unlink()
except Exception as e:
_bt.logger.debug("Failed to clean old recording %s: %s", f, e)
except Exception as e:
_bt.logger.debug("Recording cleanup error (non-critical): %s", e)
def _drop_last_active_binding(task_id: str) -> None:
"""Drop stale last-active ownership after cleaning ``task_id``.
Cleaning a bare task drops its binding; cleaning a sidecar drops the binding
only if that sidecar was still the recorded owner — so a later
click/snapshot can't resurrect a cleaned sidecar on about:blank while a
primary-session binding is preserved.
"""
_bt = _origin()
if _bt._is_local_sidecar_key(task_id):
bare_task_id = _bt._bare_task_id_for_session_key(task_id)
if _bt._last_active_session_key.get(bare_task_id) == task_id:
_bt._last_active_session_key.pop(bare_task_id, None)
else:
_bt._last_active_session_key.pop(task_id, None)
def cleanup_browser(task_id: Optional[str] = None) -> None:
"""Clean up browser session(s) for a task (task completion / inactivity timeout).
A bare task id reaps BOTH the primary session and any hybrid local sidecar
spawned for it; a key already carrying ``::local`` (inactivity loop) reaps
only that one.
"""
_bt = _origin()
if task_id is None:
task_id = "default"
session_keys = [task_id]
if not _bt._is_local_sidecar_key(task_id):
sidecar_key = f"{task_id}{_bt._LOCAL_SUFFIX}"
with _bt._cleanup_lock:
if sidecar_key in _bt._active_sessions:
session_keys.append(sidecar_key)
for session_key in session_keys:
_bt._cleanup_single_browser_session(session_key)
_bt._drop_last_active_binding(task_id)
def _kill_verified_daemon(socket_dir: str, session_name: str) -> bool:
"""Tree-kill the daemon recorded in ``<socket_dir>/<session>.pid`` if it is verifiably ours.
The .pid file lives in a world-writable temp dir and PIDs recycle: the
process must pass ``_verify_reapable_browser_daemon`` and have a start-time
fingerprint (so the kill refuses if the PID is swapped between check and
kill). Returns True when a kill was issued. Never raises.
"""
_bt = _origin()
pid_file = os.path.join(socket_dir, f"{session_name}.pid")
if not os.path.isfile(pid_file):
return False
try:
from tools.process_registry import ProcessRegistry
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
if not _bt._verify_reapable_browser_daemon(daemon_pid, socket_dir, session_name):
_bt.logger.debug(
"Skipped daemon kill for %s: pid %s failed identity "
"verification", session_name, daemon_pid)
return False
from gateway.status import get_process_start_time
daemon_start = get_process_start_time(daemon_pid)
if daemon_start is None:
_bt.logger.debug(
"Skipped daemon kill for %s: no start-time "
"fingerprint for pid %s", session_name, daemon_pid)
return False
ProcessRegistry._terminate_host_pid(daemon_pid, daemon_start)
_bt.logger.debug("Killed daemon pid %s for %s", daemon_pid, session_name)
return True
except (ProcessLookupError, ValueError, PermissionError, OSError):
_bt.logger.debug("Could not kill daemon pid for %s (already dead or inaccessible)", session_name)
return False
def _release_session_resources(task_id: str, session_info: Dict[str, Any]) -> None:
"""Untrack ``task_id``, close its cloud provider session, kill its daemon.
The unconditional tail of ``_cleanup_single_browser_session``; also the
whole of the janitor's force-reap path, which skips the polite
agent-browser/Camofox ``close`` that kept failing but must still release
the cloud session and the local Chromium.
"""
_bt = _origin()
bb_session_id = session_info.get("bb_session_id", "unknown")
with _bt._cleanup_lock:
_bt._active_sessions.pop(task_id, None)
_bt._session_last_activity.pop(task_id, None)
_bt._session_owner_homes.pop(task_id, None)
_bt._cleanup_failures.pop(task_id, None)
# Cloud mode only — local sidecars have bb_session_id=None.
if bb_session_id:
provider = _bt._get_cloud_provider()
if provider is not None:
try:
provider.close_session(bb_session_id)
except Exception as e:
_bt.logger.warning("Could not close cloud browser session: %s", e)
session_name = session_info.get("session_name", "")
if session_name:
socket_dir = os.path.join(_bt._socket_safe_tmpdir(), f"agent-browser-{session_name}")
if os.path.exists(socket_dir):
_bt._kill_verified_daemon(socket_dir, session_name)
shutil.rmtree(socket_dir, ignore_errors=True)
def _force_reap_browser_session(task_id: str) -> None:
"""Janitor last resort after repeated cleanup failures.
Skips the ``close`` round-trips that keep failing and goes straight to
``_release_session_resources`` (cloud close + daemon kill + untrack).
"""
_bt = _origin()
_bt._stop_cdp_supervisor(task_id)
with _bt._cleanup_lock:
session_info = _bt._active_sessions.get(task_id)
_bt._session_last_activity.pop(task_id, None)
_bt._recording_sessions.discard(task_id)
if session_info:
_bt._release_session_resources(task_id, session_info)
_bt._drop_last_active_binding(task_id)
def _cleanup_single_browser_session(task_id: str) -> None:
"""Internal: reap a single browser session by its exact session key."""
# Stop the CDP supervisor for this task FIRST so we close our WebSocket
# before the backend tears down the underlying CDP endpoint.
_bt = _origin()
_bt._stop_cdp_supervisor(task_id)
# Also clean up Camofox session if running in Camofox mode.
# Skip full close when managed persistence is enabled — the browser
# profile (and its session cookies) must survive across agent tasks.
# The inactivity reaper still frees idle resources.
if _bt._is_camofox_mode():
try:
from tools.browser_camofox import camofox_close, camofox_soft_cleanup
if not camofox_soft_cleanup(task_id):
camofox_close(task_id)
except Exception as e:
_bt.logger.debug("Camofox cleanup for task %s: %s", task_id, e)
_bt.logger.debug("cleanup_browser called for task_id: %s", task_id)
_bt.logger.debug("Active sessions: %s", list(_bt._active_sessions.keys()))
# Check if session exists (under lock), but don't remove yet -
# _run_browser_command needs it to build the close command.
with _bt._cleanup_lock:
session_info = _bt._active_sessions.get(task_id)
if session_info:
bb_session_id = session_info.get("bb_session_id", "unknown")
_bt.logger.debug("Found session for task %s: bb_session_id=%s", task_id, bb_session_id)
# Stop auto-recording before closing (saves the file)
_bt._maybe_stop_recording(task_id)
# A Lightpanda session is a process Hermes spawned itself (Browser
# Use mode); there is no agent-browser daemon to send ``close`` to.
# An expired cloud CDP URL cannot accept an agent-browser close command.
# Avoid feeding it back through _get_session_info(), which would try to
# renew the session recursively while cleanup is still in progress.
if (session_info.get("features") or {}).get("lightpanda"):
try:
from tools.browser_lightpanda import stop_lightpanda
stop_lightpanda(session_info.get("session_name", ""))
except Exception as e:
_bt.logger.warning("lightpanda stop failed for task %s: %s", task_id, e)
elif _bt._session_has_expired(session_info):
_bt.logger.debug(
"Skipping agent-browser close for expired session %s", task_id
)
else:
try:
_bt._run_browser_command(task_id, "close", [], timeout=10)
_bt.logger.debug(
"agent-browser close command completed for task %s", task_id
)
except Exception as e:
_bt.logger.warning("agent-browser close failed for task %s: %s", task_id, e)
_bt._release_session_resources(task_id, session_info)
_bt.logger.debug("Removed task %s from active sessions", task_id)
else:
_bt.logger.debug("No active session found for task_id: %s", task_id)
def cleanup_all_browsers() -> None:
"""
Clean up all active browser sessions.
Useful for cleanup on shutdown.
"""
_bt = _origin()
with _bt._cleanup_lock:
task_ids = list(_bt._active_sessions.keys())
for task_id in task_ids:
_bt.cleanup_browser(task_id)
# Tear down CDP supervisors for all tasks so background threads exit.
try:
from tools.browser_supervisor import SUPERVISOR_REGISTRY # type: ignore[import-not-found]
SUPERVISOR_REGISTRY.stop_all()
except Exception:
pass
# Reset cached lookups so they are re-evaluated on next use.
_bt._cached_agent_browser = None
_bt._agent_browser_resolved = False
_bt._discover_homebrew_node_dirs.cache_clear()
# Flip the resolved flag BEFORE nulling the cache so a concurrent
# reader never sees ``resolved=True`` with ``cache=None``.
_bt._command_timeout_resolved = False
_bt._cached_command_timeout = None
_bt._snapshot_threshold_resolved = False
_bt._cached_snapshot_threshold = None
_bt._cached_chromium_installed = None
_bt._chromium_autoinstall_attempted = False
_bt._cached_browser_engine = None
_bt._browser_engine_resolved = False
+6 -16
View File
@@ -58,8 +58,7 @@ def lightpanda_engine_status() -> Tuple[bool, str]:
if bu_mode:
try:
from tools.browser_use_cli import (
_read_browser_cfg,
is_legacy_browser_use_cloud_config,
_read_browser_cfg, is_legacy_browser_use_cloud_config
)
if is_legacy_browser_use_cloud_config(_read_browser_cfg()):
@@ -131,16 +130,13 @@ def _needs_lightpanda_fallback(engine: str, command: str, result: Dict[str, Any]
def _annotate_lightpanda_fallback(result: Dict[str, Any], reason: str) -> Dict[str, Any]:
"""Add a user-visible Chrome fallback warning to a browser command result."""
warning = (
"⚠ Lightpanda fallback: Chrome was used for this browser action. "
f"{reason}"
"⚠ Lightpanda fallback: Chrome was used for this browser action. " f"{reason}"
)
annotated = dict(result)
annotated["fallback_warning"] = warning
annotated["browser_engine"] = "chrome"
annotated["browser_engine_fallback"] = {
"from": "lightpanda",
"to": "chrome",
"reason": reason,
"from": "lightpanda", "to": "chrome", "reason": reason
}
data = annotated.get("data")
if isinstance(data, dict):
@@ -148,8 +144,7 @@ def _annotate_lightpanda_fallback(result: Dict[str, Any], reason: str) -> Dict[s
data.setdefault("fallback_warning", warning)
data.setdefault("browser_engine", "chrome")
data.setdefault(
"browser_engine_fallback",
{"from": "lightpanda", "to": "chrome", "reason": reason},
"browser_engine_fallback", {"from": "lightpanda", "to": "chrome", "reason": reason}
)
annotated["data"] = data
return annotated
@@ -165,10 +160,7 @@ def _copy_fallback_warning(target: Dict[str, Any], result: Dict[str, Any]) -> Di
def _run_chrome_fallback_command(
task_id: str,
command: str,
args: List[str],
timeout: int,
task_id: str, command: str, args: List[str], timeout: int
) -> Dict[str, Any]:
"""Run a browser command in a temporary Chrome session at the current URL.
@@ -259,9 +251,7 @@ def _run_chrome_fallback_command(
def _chrome_fallback_screenshot(
task_id: str,
args: List[str],
timeout: int,
task_id: str, args: List[str], timeout: int
) -> Dict[str, Any]:
"""Take a screenshot using a temporary Chrome session."""
_bt = _origin()
+4 -1
View File
@@ -60,7 +60,10 @@ def origin_module():
(``from tools import browser_tool`` in a test that calls the helper directly);
3. ``sys.modules`` / the ``tools`` package attribute / a fresh import.
"""
start = sys._getframe(2)
try:
start = sys._getframe(2)
except ValueError: # called directly by the interpreter (atexit callback)
start = None
frame = start
while frame is not None:
if frame.f_globals.get("__name__") == _ORIGIN_NAME:
+1 -4
View File
@@ -332,10 +332,7 @@ def _real_profile_cdp() -> tuple:
)
from hermes_cli.browser_connect import (
chromium_executable,
detect_default_chromium,
real_profile_copy_dir,
snapshot_real_profile,
chromium_executable, detect_default_chromium, real_profile_copy_dir, snapshot_real_profile
)
with _bt._real_profile_cdp_lock:
+809
View File
@@ -0,0 +1,809 @@
"""agent-browser session management: daemon spawn, per-backend session creation (local/lightpanda/cdp/cloud), cached session lookup, command execution with timeout handling and output interpretation.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
import json
import logging
import os
import shutil
import subprocess
from pathlib import Path
from typing import Any, Dict, List, Optional
from tools.browser_tool_origin import origin_module as _origin
def _needs_chromium_sandbox_bypass() -> bool:
"""Return True when Chromium needs --no-sandbox to start reliably."""
_bt = _origin()
if hasattr(os, "geteuid") and os.geteuid() == 0:
return True
if _bt._running_in_docker():
return True
userns_restrict = "/proc/sys/kernel/apparmor_restrict_unprivileged_userns"
try:
with open(userns_restrict, encoding="utf-8") as f:
if f.read().strip() == "1":
return True
except OSError:
pass
return False
def _apply_chromium_sandbox_args(browser_env: Dict[str, str]) -> None:
"""Add required Chromium sandbox flags without overriding user settings."""
_bt = _origin()
if (
"AGENT_BROWSER_ARGS" not in browser_env
and "AGENT_BROWSER_CHROME_FLAGS" not in browser_env
and _bt._needs_chromium_sandbox_bypass()
):
_bt.logger.debug(
"browser: sandbox bypass needed (root/docker/AppArmor userns) — "
"injecting --no-sandbox"
)
browser_env["AGENT_BROWSER_ARGS"] = "--no-sandbox,--disable-dev-shm-usage"
def _read_command_output_files(stdout_path: str, stderr_path: str) -> tuple[str, str]:
"""Best-effort read of agent-browser stdout/stderr temp files."""
stdout = stderr = ""
for path, slot in ((stdout_path, "stdout"), (stderr_path, "stderr")):
try:
with open(path, "r", encoding="utf-8") as f:
text = f.read().strip()
except OSError:
continue
if slot == "stdout":
stdout = text
else:
stderr = text
return stdout, stderr
def _unlink_command_output_files(*paths: str) -> None:
for path in paths:
try:
os.unlink(path)
except OSError:
pass
def _format_browser_timeout_error(
command: str, timeout: int, stdout: str, stderr: str
) -> str:
"""Build an actionable timeout message from captured daemon output."""
_bt = _origin()
parts = [f"Command timed out after {timeout} seconds"]
detail = (stderr or stdout or "").strip()
if detail:
parts.append(detail[:1500])
combined = f"{stderr}\n{stdout}".lower()
hints: list[str] = []
if "sandbox" in combined:
hints.append(
"Chromium sandbox launch failed. Set AGENT_BROWSER_ARGS="
"'--no-sandbox,--disable-dev-shm-usage' in your environment, "
"or run: npx agent-browser install --with-deps"
)
elif command == "open" and _bt._is_local_mode():
if _bt._running_in_docker():
hints.append(
"The browser daemon may still be starting or Chromium may be "
"missing. Pull the latest image: "
"docker pull ghcr.io/nousresearch/hermes-agent:latest"
)
else:
hints.append(
"The browser daemon may still be starting, or Chromium may be "
"missing system libraries. Install/repair with: "
"npx agent-browser install --with-deps "
"(or: npx playwright install --with-deps chromium)"
)
if hints:
parts.extend(hints)
return "\n".join(parts)
def _agent_browser_argv(browser_cmd: str) -> list:
"""Command prefix to invoke agent-browser (concrete binary or npx sentinel).
Concrete executable paths stay a single argv item (spaces intact); only the
synthetic npx sentinel expands. npx is resolved through the same
PATH + extended-PATH cascade ``_find_agent_browser`` uses — a bare
``shutil.which("npx")`` would let a broken system npx shadow a healthy
Hermes-managed one. If npx isn't found at all (Termux, bare container) the
bare name is used so Popen raises a readable ``FileNotFoundError: 'npx'``.
``--ignore-scripts``: AGENT_BROWSER_NPX_SPEC is a floating range, not an
exact pin — a compromised future patch must not run install-time scripts.
"""
_bt = _origin()
if _bt._is_npx_agent_browser_sentinel(browser_cmd):
_npx_bin = _bt._resolve_npx_bin() or "npx"
return [_npx_bin, "--ignore-scripts", "--prefer-offline", "-y", _bt.AGENT_BROWSER_NPX_SPEC]
return [browser_cmd]
def _prepare_session_socket_dir(session_name: str) -> str:
"""Create the per-session agent-browser socket dir and claim it with our PID.
Each session gets its own dir so parallel workers don't fight over the
default socket path ("Failed to create socket directory: Permission
denied"). The owner_pid file is written BEFORE first use: another hermes
process's orphan reaper rmtree's any agent-browser-* dir in the shared
tmpdir that carries no live owner, which would delete this one mid-command.
"""
_bt = _origin()
socket_dir = os.path.join(_bt._socket_safe_tmpdir(), f"agent-browser-{session_name}")
os.makedirs(socket_dir, mode=0o700, exist_ok=True)
_bt._write_owner_pid(socket_dir, session_name)
return socket_dir
def _agent_browser_command_env(socket_dir: str) -> Dict[str, str]:
"""Credential-scrubbed env for one agent-browser command.
Adds the discovery-time PATH fallbacks, the session socket dir, and the
daemon-side idle self-termination (``AGENT_BROWSER_IDLE_TIMEOUT_MS``,
agent-browser 0.24+) mirroring the Python-side inactivity janitor —
unless the user set the idle timeout explicitly.
"""
_bt = _origin()
env = _bt._build_browser_env()
env["PATH"] = _bt._merge_browser_path(env.get("PATH", ""))
env["AGENT_BROWSER_SOCKET_DIR"] = socket_dir
if "AGENT_BROWSER_IDLE_TIMEOUT_MS" not in env:
env["AGENT_BROWSER_IDLE_TIMEOUT_MS"] = str(_bt.BROWSER_SESSION_INACTIVITY_TIMEOUT * 1000)
return env
def _popen_agent_browser(argv: List[str], env: Dict[str, str], socket_dir: str, tag: str) -> "subprocess.Popen":
"""Spawn agent-browser with stdout/stderr redirected to ``socket_dir/_stdout_<tag>``.
Temp files instead of pipes: the CLI forks a background daemon that inherits
its fds, so with pipes ``communicate()`` never sees EOF until the timeout.
Windows: CREATE_NO_WINDOW only (NOT CREATE_NEW_PROCESS_GROUP, which on
Python 3.11 cancels asyncio's running loop task and surfaces as
KeyboardInterrupt in the CLI), STARTF_USESTDHANDLES so CreateProcess hands
the child ONLY our three handles (leaked parent console handles make the
Rust binary's daemon grandchild die silently), close_fds=True for the rest.
Returns the Popen; the caller reads/unlinks the two files.
"""
_bt = _origin()
stdout_path = os.path.join(socket_dir, f"_stdout_{tag}")
stderr_path = os.path.join(socket_dir, f"_stderr_{tag}")
stdout_fd = os.open(stdout_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
stderr_fd = os.open(stderr_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
try:
_popen_extra: dict = {}
if os.name == "nt":
_popen_extra["creationflags"] = _bt.windows_hide_flags()
_popen_extra["close_fds"] = True
_si = subprocess.STARTUPINFO()
_si.dwFlags |= subprocess.STARTF_USESTDHANDLES
_popen_extra["startupinfo"] = _si
return subprocess.Popen(
argv, stdout=stdout_fd, stderr=stderr_fd,
stdin=subprocess.DEVNULL, env=env, **_popen_extra,
)
finally:
os.close(stdout_fd)
os.close(stderr_fd)
def _create_local_session(task_id: str, allow_real_profile: bool = True) -> Dict[str, str]:
_bt = _origin()
import uuid
# Real-profile consent: attach this local session (via CDP) to the user's
# browser running on a hermes-owned SNAPSHOT of their real profile, logins
# included. Fail closed on resolver/launch errors — a consented user must
# never be silently downgraded to a throwaway. The hybrid private-URL
# sidecar passes allow_real_profile=False: handing the user's cookie jar to
# an arbitrary internal host the model chose is a larger, unconsented
# exposure than the routing rule protects against (and a real-profile
# failure must not break private-URL routing).
if allow_real_profile:
cdp_url, err = _bt._real_profile_cdp()
if err:
raise RuntimeError(err)
if cdp_url:
session_name = f"rp_{uuid.uuid4().hex[:10]}"
_bt.logger.info(
"Created real-profile local session %s for task %s", session_name, task_id
)
return {
"session_name": session_name,
"bb_session_id": None,
"cdp_url": _bt._resolve_cdp_override(cdp_url),
"features": {"local": True, "real_profile": True},
}
# Browser Use mode drives whatever CDP endpoint it is handed; with
# ``browser.engine: lightpanda`` that endpoint is a Hermes-spawned
# ``lightpanda serve``. The built-in tools never reach this branch —
# they are hidden in Browser Use mode — and keep driving Lightpanda via
# ``agent-browser --engine lightpanda`` on the plain local session below.
if _bt._is_browser_use_cli_mode() and _bt._using_lightpanda_engine():
return _bt._create_lightpanda_session(task_id)
session_name = f"h_{uuid.uuid4().hex[:10]}"
_bt.logger.info("Created local browser session %s for task %s",
session_name, task_id)
return {
"session_name": session_name,
"bb_session_id": None,
"cdp_url": None,
"features": {"local": True},
}
def _create_lightpanda_session(task_id: str) -> Dict[str, Any]:
"""Spawn ``lightpanda serve`` for this session key (Browser Use mode)."""
_bt = _origin()
import uuid
from tools.browser_lightpanda import launch_lightpanda
session_name = f"lp_{uuid.uuid4().hex[:10]}"
server, err = launch_lightpanda(
session_name, block_private_networks=not _bt._is_local_backend()
)
if err:
raise RuntimeError(err)
_bt.logger.info(
"Created Lightpanda session %s (port %s) for task %s", session_name, server.port, task_id
)
return {
"session_name": session_name,
"bb_session_id": None,
"cdp_url": server.cdp_url,
"features": {"local": True, "lightpanda": True},
}
def _local_backend_process_dead(session_info: Dict[str, Any]) -> bool:
"""True for a Lightpanda session whose ``lightpanda serve`` is gone."""
if not (session_info.get("features") or {}).get("lightpanda"):
return False
from tools.browser_lightpanda import get_server
server = get_server(session_info.get("session_name", ""))
return server is None or not server.is_alive()
def _create_cdp_session(task_id: str, cdp_url: str) -> Dict[str, str]:
"""Create a session that connects to a user-supplied CDP endpoint."""
_bt = _origin()
import uuid
session_name = f"cdp_{uuid.uuid4().hex[:10]}"
_bt.logger.info("Created CDP browser session %s → %s for task %s",
session_name, _bt._sanitize_url_for_logs(cdp_url), task_id)
return {
"session_name": session_name,
"bb_session_id": None,
"cdp_url": cdp_url,
"features": {"cdp_override": True},
}
def _create_cloud_session_or_fallback(task_id: str, provider) -> Dict[str, Any]:
"""Create a cloud session; fall back to local Chromium (marked degraded) on failure.
Some cloud providers (Browser-Use v3) return an HTTP CDP discovery URL
instead of a raw websocket endpoint, so ``cdp_url`` is resolved here.
"""
_bt = _origin()
try:
session_info = provider.create_session(task_id)
if not session_info or not isinstance(session_info, dict):
raise ValueError(f"Cloud provider returned invalid session: {session_info!r}")
if session_info.get("cdp_url"):
session_info = dict(session_info)
session_info["cdp_url"] = _bt._resolve_cdp_override(str(session_info["cdp_url"]))
return session_info
except Exception as e:
provider_name = type(provider).__name__
_bt.logger.warning(
"Cloud provider %s failed (%s); attempting fallback to local "
"Chromium for task %s",
provider_name, e, task_id,
exc_info=True,
)
try:
session_info = _bt._create_local_session(task_id)
except Exception as local_error:
raise RuntimeError(
f"Cloud provider {provider_name} failed ({e}) and local "
f"fallback also failed ({local_error})"
) from e
# Mark session as degraded for observability
if isinstance(session_info, dict):
session_info = dict(session_info)
session_info["fallback_from_cloud"] = True
session_info["fallback_reason"] = str(e)
session_info["fallback_provider"] = provider_name
return session_info
def _create_session_for_key(task_id: str, force_local: bool) -> Dict[str, Any]:
"""Create a fresh session for ``task_id`` (runs OUTSIDE the lock: cloud mode makes a network call).
Precedence: CDP override > hybrid local sidecar > cloud provider > local.
The hybrid private-URL sidecar NEVER gets the real profile — presenting real
cookies to an arbitrary LAN host the model routed there is unconsented
exposure (see ``_create_local_session``).
"""
_bt = _origin()
cdp_override = _bt._get_cdp_override()
if cdp_override and not force_local:
return _bt._create_cdp_session(task_id, cdp_override)
if force_local:
return _bt._create_local_session(task_id, allow_real_profile=False)
provider = _bt._get_cloud_provider()
if provider is None:
return _bt._create_local_session(task_id)
return _bt._create_cloud_session_or_fallback(task_id, provider)
def _get_session_info(task_id: Optional[str] = None) -> Dict[str, Any]:
"""Get or create session info for a session key (thread-safe).
``task_id`` may carry the ``::local`` suffix (hybrid local sidecar), which
forces a local Chromium even when a cloud provider is configured. Also
starts the inactivity cleanup thread and touches activity tracking.
Returns a dict with ``session_name`` (always) plus ``bb_session_id`` /
``cdp_url`` for cloud sessions.
"""
_bt = _origin()
if task_id is None:
task_id = "default"
# Start the cleanup thread if not running (handles inactivity timeouts)
_bt._start_browser_cleanup_thread()
# Update activity timestamp for this session
_bt._update_session_activity(task_id)
with _bt._cleanup_lock:
# Check if we already have a session for this task
existing_session = _bt._active_sessions.get(task_id)
# Suspect-session recycle: a previous command
# timeout marked this cached session suspect via the SuspectableBackend
# adapter. ensure_healthy() tears it down here, at next use, and we fall
# through to create a fresh session — the expensive recycle lives on this
# path, not on the timeout path (mark must stay cheap).
if existing_session is not None and not _bt._browser_session_backend(task_id).ensure_healthy():
# Teardown removes the activity entry; the replacement must be
# tracked by the inactivity reaper like an initial session.
_bt._update_session_activity(task_id)
with _bt._cleanup_lock:
replacement = _bt._active_sessions.get(task_id)
if replacement is not None and replacement is not existing_session:
# Another thread already recycled and re-created it.
return replacement
existing_session = None
if existing_session is not None:
if (
not _bt._session_has_expired(existing_session)
and not _bt._local_backend_process_dead(existing_session)
):
return existing_session
_bt.logger.info(
"Replacing expired or dead browser session for task %s", task_id
)
_bt._cleanup_single_browser_session(task_id)
# Cleanup removes the activity entry. The replacement session must be
# tracked by the inactivity reaper just like an initial session.
_bt._update_session_activity(task_id)
# Guard against a concurrent replacement: another thread may have
# already cleaned up the expired session and created a fresh one
# while we were waiting. If so, return the live replacement instead
# of falling through to create yet another session.
with _bt._cleanup_lock:
replacement = _bt._active_sessions.get(task_id)
if replacement is not None and replacement is not existing_session:
return replacement
# Hybrid routing: session keys ending with ``::local`` force a local
# Chromium regardless of the globally-configured cloud provider. Public
# URLs in the same conversation continue to use the cloud session under
# the bare task_id key.
force_local = _bt._is_local_sidecar_key(task_id)
session_info = _bt._create_session_for_key(task_id, force_local)
with _bt._cleanup_lock:
# Double-check: another thread may have created a session while we
# were doing the network call. Use the existing one to avoid leaking
# orphan cloud sessions.
if task_id in _bt._active_sessions:
return _bt._active_sessions[task_id]
session_info = dict(session_info)
session_info.setdefault("session_key", task_id)
session_info.setdefault("owner_task_id", _bt._bare_task_id_for_session_key(task_id))
_bt._active_sessions[task_id] = session_info
# A brand-new session is healthy by definition — drop any stale
# suspect flag left by a wedged-path eviction of its predecessor.
_bt._suspect_browser_sessions.pop(task_id, None)
# Lazy-start the CDP supervisor now that the session exists (if the
# backend surfaces a CDP URL via override or session_info["cdp_url"]).
# Idempotent; swallows errors. See _ensure_cdp_supervisor for details.
# Skip for local sidecars — they have no CDP URL — and for Lightpanda
# sessions: those only exist in Browser Use mode, where the browser_*
# tools that consume supervisor state are hidden, so the supervisor
# would just hold an idle second CDP connection to the process.
if not force_local and not (session_info.get("features") or {}).get("lightpanda"):
_bt._ensure_cdp_supervisor(task_id)
return session_info
def _discard_timed_out_browser_session(
task_id: str, session_info: Dict[str, Any], task_socket_dir: str
) -> None:
"""Drop a stuck client generation without losing cloud cleanup state."""
_bt = _origin()
with _bt._cleanup_lock:
if _bt._active_sessions.get(task_id) is not session_info:
return
_bt._stop_cdp_supervisor(task_id)
if session_info.get("bb_session_id") or session_info.get("cdp_url"):
import uuid
replacement = dict(session_info)
replacement["session_name"] = f"h_{uuid.uuid4().hex[:10]}"
replacement.pop("_first_nav", None)
_bt._active_sessions[task_id] = replacement
else:
_bt._active_sessions.pop(task_id, None)
_bt._session_last_activity.pop(task_id, None)
bare_task_id = _bt._bare_task_id_for_session_key(task_id)
if _bt._last_active_session_key.get(bare_task_id) == task_id:
_bt._last_active_session_key.pop(bare_task_id, None)
session_name = str(session_info.get("session_name") or "")
if session_name:
pid_file = os.path.join(task_socket_dir, f"{session_name}.pid")
if os.path.isfile(pid_file):
try:
daemon_pid = int(Path(pid_file).read_text(encoding="utf-8").strip())
if not _bt._verify_reapable_browser_daemon(daemon_pid, task_socket_dir, session_name):
return
# Tree-kill: the daemon spawns Chromium
# children; terminating only the daemon PID leaks the whole
# Chromium tree. agent.deadline.kill_process_tree escalates
# SIGTERM → SIGKILL across the tree.
from agent import deadline as _deadline
_deadline.kill_process_tree(daemon_pid)
except (ProcessLookupError, ValueError, PermissionError, OSError):
_bt.logger.debug("Could not kill timed-out browser daemon for %s", session_name)
return
shutil.rmtree(task_socket_dir, ignore_errors=True)
def _read_browser_daemon_pid(task_socket_dir: str, session_name: str) -> Optional[int]:
"""Read the agent-browser daemon PID for a session (best-effort)."""
pid_file = os.path.join(task_socket_dir, f"{session_name}.pid")
try:
return int(Path(pid_file).read_text(encoding="utf-8").strip())
except (OSError, ValueError):
return None
def _browser_daemon_responsive(task_socket_dir: str, probe_timeout_s: float = 1.0) -> bool:
"""Cheap liveness probe: connect to the daemon's unix control socket.
A successful connect proves the accept loop is alive (the command wedged on
the page/CDP side, not the daemon). Windows uses named pipes — no probe is
possible, so report unresponsive (tree-kill + respawn is the safe recovery).
"""
if os.name == "nt":
return False
import socket as socket_mod
if not hasattr(socket_mod, "AF_UNIX"):
return False
try:
entries = os.listdir(task_socket_dir)
except OSError:
return False
sock_paths = [
os.path.join(task_socket_dir, e) for e in entries if e.endswith(".sock")
]
for sock_path in sock_paths:
try:
with socket_mod.socket(socket_mod.AF_UNIX, socket_mod.SOCK_STREAM) as s:
s.settimeout(probe_timeout_s)
s.connect(sock_path)
return True
except OSError:
continue
return False
def _handle_browser_command_timeout(
task_id: str, session_info: Dict[str, Any], task_socket_dir: str
) -> None:
"""Recover session state after a browser command timeout.
* Cloud / CDP sessions: no local daemon to probe — replace the stuck client
generation now (fresh ``session_name``, same ``bb_session_id`` so cloud
cleanup still works).
* Local daemon alive (PID live, identity-verified, control socket accepts):
only the *command* wedged; mark the session suspect and let the next use
recycle it via ``ensure_healthy`` → clean ``close`` → fresh session.
* Local daemon wedged/dead: it cannot service a clean close and its Chromium
children would leak — tree-kill and evict now; the next call respawns.
Both local branches ``mark_suspect`` first (cheap, lock-free) so the
poisoned-cache invariant holds even if eviction races another thread's
replacement (the flag then costs one harmless no-op teardown).
"""
_bt = _origin()
if session_info.get("bb_session_id") or session_info.get("cdp_url"):
_bt._discard_timed_out_browser_session(task_id, session_info, task_socket_dir)
return
_bt._browser_session_backend(task_id).mark_suspect(
"browser command timed out; session may be poisoned"
)
session_name = str(session_info.get("session_name") or "")
daemon_pid = _bt._read_browser_daemon_pid(task_socket_dir, session_name) if session_name else None
daemon_alive = (
daemon_pid is not None
and _bt._pid_exists(daemon_pid)
and _bt._verify_reapable_browser_daemon(daemon_pid, task_socket_dir, session_name)
and _bt._browser_daemon_responsive(task_socket_dir)
)
if daemon_alive:
_bt.logger.warning(
"browser daemon for %s is alive after command timeout; session "
"marked suspect and will be recycled at next use", task_id,
)
return
_bt.logger.warning(
"browser daemon for %s is wedged or dead after command timeout; "
"tree-killing and evicting the session", task_id,
)
_bt._discard_timed_out_browser_session(task_id, session_info, task_socket_dir)
# The poisoned entry is gone (evicted, or superseded by a concurrent
# replacement discard refused to touch) — either way the cache no longer
# holds the timed-out session, so drop the flag: it must not poison a
# session created later under the same key.
_bt._suspect_browser_sessions.pop(task_id, None)
def _interpret_browser_command_output(command: str, stdout: str, stderr: str, returncode: int) -> Dict[str, Any]:
"""Turn a finished agent-browser process's output into a result dict.
Empty stdout with rc=0 is a broken state (stale daemon) and is reported as
failure rather than a silent ``{"success": True, "data": {}}`` — except for
commands in ``_EMPTY_OK_COMMANDS``. Non-JSON output is an error, except
for ``screenshot`` where the saved path is recovered from the prose.
"""
_bt = _origin()
if stderr and stderr.strip():
level = logging.WARNING if returncode != 0 else logging.DEBUG
_bt.logger.log(level, "browser '%s' stderr: %s", command, stderr.strip()[:500])
stdout_text = stdout.strip()
if not stdout_text and returncode == 0 and command not in _bt._EMPTY_OK_COMMANDS:
_bt.logger.warning("browser '%s' returned empty output (rc=0)", command)
return {"success": False, "error": f"Browser command '{command}' returned no output"}
if not stdout_text:
if returncode != 0:
error_msg = stderr.strip() if stderr else f"Command failed with code {returncode}"
_bt.logger.warning("browser '%s' failed (rc=%s): %s", command, returncode, error_msg[:300])
return {"success": False, "error": error_msg}
return {"success": True, "data": {}}
try:
parsed = json.loads(stdout_text)
except json.JSONDecodeError:
raw = stdout_text[:2000]
_bt.logger.warning("browser '%s' returned non-JSON output (rc=%s): %s",
command, returncode, raw[:500])
if command == "screenshot":
stderr_text = (stderr or "").strip()
combined_text = "\n".join(part for part in [stdout_text, stderr_text] if part)
recovered_path = _bt._extract_screenshot_path_from_text(combined_text)
if recovered_path and Path(recovered_path).exists():
_bt.logger.info(
"browser 'screenshot' recovered file from non-JSON output: %s", recovered_path
)
return {"success": True, "data": {"path": recovered_path, "raw": raw}}
return {"success": False, "error": f"Non-JSON output from agent-browser for '{command}': {raw}"}
# Empty snapshot content is a common sign of daemon/CDP issues.
if command == "snapshot" and parsed.get("success"):
snap_data = parsed.get("data", {})
if not snap_data.get("snapshot") and not snap_data.get("refs"):
_bt.logger.warning("snapshot returned empty content. "
"Possible stale daemon or CDP connection issue. "
"returncode=%s", returncode)
return parsed
def _run_browser_command(
task_id: str,
command: str,
args: List[str] = None,
timeout: Optional[int] = None,
_engine_override: Optional[str] = None,
) -> Dict[str, Any]:
"""Run one agent-browser CLI command against the task's session; returns its parsed JSON.
``timeout=None`` reads ``browser.command_timeout`` (default 30s).
``_engine_override`` forces an engine for this call only (the Lightpanda
fallback uses it to retry with Chrome without touching global state).
"""
_bt = _origin()
if timeout is None:
timeout = _bt._safe_command_timeout()
args = args or []
# Build the command
try:
browser_cmd = _bt._find_agent_browser()
except FileNotFoundError as e:
_bt.logger.warning("agent-browser CLI not found: %s", e)
return {"success": False, "error": str(e)}
if _bt._requires_real_termux_browser_install(browser_cmd):
error = _bt._termux_browser_install_error()
_bt.logger.warning("browser command blocked on Termux: %s", error)
return {"success": False, "error": error}
# Local mode with no Chromium on disk: fail fast with an actionable
# message instead of hanging for _command_timeout seconds per call.
# Skip when engine=lightpanda — LP doesn't need Chromium for navigation.
if (
_bt._is_local_mode()
and not _bt._chromium_installed()
and _bt._get_browser_engine() != "lightpanda"
and not _bt._maybe_autoinstall_chromium()
):
if _bt._running_in_docker():
hint = (
"Chromium browser is missing. You're running in Docker — pull "
"the latest image to get the bundled Chromium: "
"docker pull ghcr.io/nousresearch/hermes-agent:latest"
)
else:
hint = (
"Chromium browser is missing. Install it with: "
"npx agent-browser install --with-deps "
"(or: npx playwright install --with-deps chromium)"
)
_bt.logger.warning("browser command blocked: %s", hint)
return {"success": False, "error": hint}
from tools.interrupt import is_interrupted
if is_interrupted():
return {"success": False, "error": "Interrupted"}
# Get session info (creates Browserbase session with proxies if needed)
try:
session_info = _bt._get_session_info(task_id)
except Exception as e:
_bt.logger.warning("Failed to create browser session for task=%s: %s", task_id, e)
return {"success": False, "error": f"Failed to create browser session: {str(e)}"}
# Cleanup stops the supervisor before closing the backend; keep it stopped.
if command != "close" and session_info.get("cdp_url"):
_bt._ensure_cdp_supervisor(task_id)
# Build the command with the appropriate backend flag.
# Cloud mode: --cdp <websocket_url> connects to Browserbase.
# Local mode: --session <name> launches a local headless Chromium.
# The rest of the command (--json, command, args) is identical.
if session_info.get("cdp_url"):
# Cloud mode — connect to remote Browserbase browser via CDP
# IMPORTANT: Do NOT use --session with --cdp. In agent-browser >=0.13,
# --session creates a local browser instance and silently ignores --cdp.
backend_args = ["--cdp", session_info["cdp_url"]]
else:
# Local mode — launch Chromium (headless by default, headed when configured)
backend_args = ["--session", session_info["session_name"]]
if _bt._is_headed_mode():
backend_args.append("--headed")
# Lightpanda engine injection (local mode only, agent-browser v0.25.3+).
# Use the resolved session backend rather than global cloud-provider state:
# hybrid private-URL routing can create a local sidecar while a cloud
# provider remains configured for public URLs.
engine = _engine_override or _bt._get_browser_engine()
if engine != "auto" and not _bt._is_camofox_mode() and not session_info.get("cdp_url"):
backend_args += ["--engine", engine]
cmd_parts = _bt._agent_browser_argv(browser_cmd) + backend_args + ["--json", command] + args
try:
task_socket_dir = _bt._prepare_session_socket_dir(session_info["session_name"])
_bt.logger.debug("browser cmd=%s task=%s socket_dir=%s (%d chars)",
command, task_id, task_socket_dir, len(task_socket_dir))
browser_env = _bt._agent_browser_command_env(task_socket_dir)
# Chromium-only launch flags are rejected by Lightpanda. Strip both
# the current and legacy variables for Lightpanda commands; explicit
# Chrome commands and fallback use the shared Chromium policy.
if engine == "lightpanda":
_stripped_args = browser_env.pop("AGENT_BROWSER_ARGS", None)
_stripped_flags = browser_env.pop("AGENT_BROWSER_CHROME_FLAGS", None)
if _stripped_args is not None or _stripped_flags is not None:
_bt.logger.debug(
"browser: stripped Chromium-only AGENT_BROWSER_ARGS/"
"AGENT_BROWSER_CHROME_FLAGS for Lightpanda command %s "
"(agent-browser rejects them with --engine lightpanda)",
command,
)
else:
_bt._apply_chromium_sandbox_args(browser_env)
stdout_path = os.path.join(task_socket_dir, f"_stdout_{command}")
stderr_path = os.path.join(task_socket_dir, f"_stderr_{command}")
proc = _bt._popen_agent_browser(cmd_parts, browser_env, task_socket_dir, command)
try:
proc.wait(timeout=timeout)
except subprocess.TimeoutExpired:
proc.kill()
proc.wait()
stdout, stderr = _bt._read_command_output_files(stdout_path, stderr_path)
_bt._unlink_command_output_files(stdout_path, stderr_path)
_bt._handle_browser_command_timeout(task_id, session_info, task_socket_dir)
if stderr and stderr.strip():
_bt.logger.warning(
"browser '%s' stderr after timeout: %s", command, stderr.strip()[:500]
)
_bt.logger.warning("browser '%s' timed out after %ds (task=%s, socket_dir=%s)",
command, timeout, task_id, task_socket_dir)
result = {
"success": False,
"error": _bt._format_browser_timeout_error(command, timeout, stdout, stderr),
}
# Fall through to fallback check below
else:
with open(stdout_path, "r", encoding="utf-8") as f:
stdout = f.read()
with open(stderr_path, "r", encoding="utf-8") as f:
stderr = f.read()
_bt._unlink_command_output_files(stdout_path, stderr_path)
result = _bt._interpret_browser_command_output(command, stdout, stderr, proc.returncode)
except Exception as e:
_bt.logger.warning("browser '%s' exception: %s", command, e, exc_info=True)
result = {"success": False, "error": str(e)}
# --- Lightpanda automatic Chrome fallback ---
# If engine is lightpanda and the result looks broken, retry with Chrome.
# This runs for ALL exit paths (timeout, empty, non-JSON, nonzero rc, parsed).
fallback_reason = _bt._lightpanda_fallback_reason(engine, command, result)
if fallback_reason:
_bt.logger.info(
"Lightpanda fallback: retrying '%s' with Chrome (task=%s): %s",
command,
task_id,
fallback_reason,
)
# For screenshots, use the dedicated Chrome fallback helper
# (spins up a separate Chrome session to the same URL).
if command == "screenshot":
fallback_result = _bt._chrome_fallback_screenshot(task_id, args or [], timeout)
else:
fallback_result = _bt._run_chrome_fallback_command(task_id, command, args, timeout)
return _bt._annotate_lightpanda_fallback(fallback_result, fallback_reason)
return result
+180
View File
@@ -0,0 +1,180 @@
"""browser_vision helpers: Lightpanda pre-route, native provider vision, auxiliary-LLM screenshot analysis.
Split out of ``tools/browser_tool.py``; every name is re-imported there so
``tools.browser_tool.<name>`` keeps resolving (and monkeypatching). Origin
symbols and module state are read/written through ``_bt`` (the origin module,
resolved per call by :func:`tools.browser_tool_origin.origin_module`) so
``patch("tools.browser_tool.X")`` is honoured and no import cycle exists.
"""
import os
import shutil
from pathlib import Path
from typing import Any, Dict, Optional, Tuple
from tools.browser_tool_origin import origin_module as _origin
def _vision_mode_label() -> str:
_bt = _origin()
_cp = _bt._get_cloud_provider()
return "local" if _cp is None else f"cloud ({_cp.provider_name()})"
def _lightpanda_vision_preroute(
effective_task_id: str, annotate: bool, screenshot_path: Path,
) -> Tuple[bool, Optional[str], Path]:
"""Capture the vision screenshot through the Chrome fallback when Lightpanda is the engine.
Lightpanda has no graphical renderer, so the normal path would fail with a
CDP error or return a placeholder PNG. Returns ``(prerouted, fallback_warning,
screenshot_path)``; on fallback failure ``prerouted`` is False and the caller
takes the normal screenshot path (forcing Chrome) so ``_run_browser_command``
still produces the standard fallback metadata/error.
"""
_bt = _origin()
engine = _bt._get_browser_engine()
if engine != "lightpanda" or not _bt._should_inject_engine(engine):
return False, None, screenshot_path
_bt.logger.debug("browser_vision: pre-routing screenshot to Chrome (engine=lightpanda)")
screenshot_args = ["--annotate"] if annotate else []
fb_result = _bt._chrome_fallback_screenshot(effective_task_id, screenshot_args, _bt._get_command_timeout())
fb_result = _bt._annotate_lightpanda_fallback(fb_result, _bt._LP_VISION_FALLBACK_REASON)
if not fb_result.get("success"):
_bt.logger.warning("Lightpanda Chrome fallback vision screenshot failed: %s", fb_result.get("error"))
return False, None, screenshot_path
fb_path = fb_result.get("data", {}).get("path", "")
if fb_path and os.path.exists(fb_path):
import uuid as uuid_mod
from hermes_constants import get_hermes_dir
screenshots_dir = get_hermes_dir("cache/screenshots", "browser_screenshots")
screenshots_dir.mkdir(parents=True, exist_ok=True)
persistent_path = screenshots_dir / f"browser_screenshot_{uuid_mod.uuid4().hex}.png"
shutil.copy2(fb_path, persistent_path)
screenshot_path = persistent_path
return True, fb_result.get("fallback_warning"), screenshot_path
def _native_vision_result(
screenshot_path: Path, question: str, annotate: bool,
result: Dict[str, Any], lp_fallback_warning: Optional[str],
) -> Dict[str, Any]:
"""Multimodal tool-result envelope: the main model inspects the pixels itself.
History-reuse cap: this embed is baked into the tool result and re-sent on
every later turn, exactly like vision_analyze's native path — apply the same
proactive resize so full-res screenshots can't enter immutable history
uncapped. The helper's stat/dimension quick-estimate skips the resize when
already under both caps; without Pillow it fails open to the raw bytes.
"""
from tools.vision_tools import (
_EMBED_MAX_DIMENSION,
_EMBED_TARGET_BYTES,
_build_native_vision_tool_result,
_resize_image_for_vision,
)
data_url = _resize_image_for_vision(
screenshot_path,
mime_type="image/png",
max_base64_bytes=_EMBED_TARGET_BYTES,
max_dimension=_EMBED_MAX_DIMENSION,
force_jpeg=True,
)
native_result = _build_native_vision_tool_result(
image_url=str(screenshot_path),
question=question,
image_data_url=data_url,
image_size_bytes=screenshot_path.stat().st_size,
)
meta = native_result.setdefault("meta", {})
meta["screenshot_path"] = str(screenshot_path)
if lp_fallback_warning:
meta["fallback_warning"] = lp_fallback_warning
if annotate and result.get("data", {}).get("annotations"):
meta["annotations"] = result["data"]["annotations"]
native_result["text_summary"] = (
f"{native_result.get('text_summary', '')} " f"Screenshot path: {screenshot_path}"
).strip()
return native_result
def _analyze_screenshot_with_aux_llm(screenshot_path: Path, question: str) -> str:
"""One-shot aux vision-LLM analysis (not baked into history), secret-redacted.
Encodes at full resolution; on a size-related provider rejection the image
is downscaled once and retried. Timeout/temperature come from
``auxiliary.vision.*`` — local vision models (llama.cpp, ollama) can take
well over 30s, so the default timeout is generous.
"""
_bt = _origin()
import base64
vision_prompt = (
f"You are analyzing a screenshot of a web browser.\n\n"
f"User's question: {question}\n\n"
f"Provide a detailed and helpful answer based on what you see in the screenshot. "
f"If there are interactive elements, describe them. If there are verification challenges "
f"or CAPTCHAs, describe what type they are and what action might be needed. "
f"Focus on answering the user's specific question."
)
_screenshot_bytes = screenshot_path.read_bytes()
_screenshot_b64 = base64.b64encode(_screenshot_bytes).decode("ascii")
data_url = f"data:image/png;base64,{_screenshot_b64}"
vision_model = _bt._get_vision_model()
_bt.logger.debug("browser_vision: analysing screenshot (%d bytes)",
len(_screenshot_bytes))
vision_timeout = 120.0
vision_temperature = 0.1
try:
from hermes_cli.config import load_config
_vision_cfg = _bt.cfg_get(load_config(), "auxiliary", "vision", default={})
_vt = _vision_cfg.get("timeout")
if _vt is not None:
vision_timeout = float(_vt)
_vtemp = _vision_cfg.get("temperature")
if _vtemp is not None:
vision_temperature = float(_vtemp)
except Exception:
pass
call_kwargs = {
"task": "vision",
"messages": [
{
"role": "user",
"content": [
{"type": "text", "text": vision_prompt},
{"type": "image_url", "image_url": {"url": data_url}},
],
}
],
"temperature": vision_temperature,
"timeout": vision_timeout,
}
if vision_model:
call_kwargs["model"] = vision_model
try:
response = _bt._lazy_call_llm(**call_kwargs)
except Exception as _api_err:
from tools.vision_tools import (
_is_image_size_error, _resize_image_for_vision, _RESIZE_TARGET_BYTES,
)
if not (_is_image_size_error(_api_err) and len(data_url) > _RESIZE_TARGET_BYTES):
raise
_bt.logger.info(
"Vision API rejected screenshot (%.1f MB); "
"auto-resizing to ~%.0f MB and retrying...",
len(data_url) / (1024 * 1024),
_RESIZE_TARGET_BYTES / (1024 * 1024),
)
data_url = _resize_image_for_vision(screenshot_path, mime_type="image/png")
call_kwargs["messages"][0]["content"][1]["image_url"]["url"] = data_url
response = _bt._lazy_call_llm(**call_kwargs)
analysis = (response.choices[0].message.content or "").strip()
# Redact secrets the vision LLM may have read from the screenshot.
from agent.redact import redact_sensitive_text
return redact_sensitive_text(analysis)
+84 -112
View File
@@ -4,6 +4,7 @@ When browser.backend is "browser-use", the model gets ``browser_exec`` tool
instead of default browser tools
"""
import contextlib
import importlib
import json
import logging
@@ -91,14 +92,20 @@ _URL_RE = re.compile(r"https?://[^\s'\"\\)]+", re.IGNORECASE)
_FHS_BIN_DIRS = ("/usr/local/sbin", "/usr/local/bin", "/usr/sbin", "/usr/bin", "/sbin", "/bin")
def _quiet(fn: Callable[[], Any], default: Any, log_prefix: str = "") -> Any:
"""``fn()``, or ``default`` on any exception (debug-logged when ``log_prefix`` is set)."""
try:
return fn()
except Exception as e:
if log_prefix:
logger.debug("%s: %s", log_prefix, e)
return default
def _lazy_call(module: str, name: str, default: Any, log_prefix: str) -> Any:
"""Call ``module.name()`` resolved at call time (so tests can stub the module or
patch the attribute); on any failure log ``log_prefix`` and return ``default``."""
try:
return getattr(importlib.import_module(module), name)()
except Exception as e:
logger.debug("%s: %s", log_prefix, e)
return default
return _quiet(lambda: getattr(importlib.import_module(module), name)(), default, log_prefix)
def _camofox_active(context: str = "") -> bool:
@@ -106,6 +113,16 @@ def _camofox_active(context: str = "") -> bool:
return _lazy_call("tools.browser_camofox", "is_camofox_mode", False, f"Camofox activity check failed{context}")
def _real_profile_consented() -> bool:
"""Whether the user opted in to real-profile local browsing (config read)."""
return _lazy_call("tools.browser_tool", "_use_real_profile", False, "real-profile consent lookup failed")
def _lightpanda_engine_in_use() -> bool:
return _lazy_call("tools.browser_tool", "lightpanda_engine_status", (False, ""),
"lightpanda engine status unavailable")[0]
def _set_cdp_env(env: dict, cdp: str) -> None:
"""Export a CDP endpoint under the BU_CDP_* contract (http(s) → URL, else WS)."""
env["BU_CDP_URL" if cdp.startswith(("http://", "https://")) else "BU_CDP_WS"] = cdp
@@ -142,9 +159,9 @@ def _blocked_url_in_code(code: str) -> Optional[str]:
def _base_subprocess_env() -> dict:
from tools.browser_tool import _build_browser_env
env = _build_browser_env()
# The CLI runs under its own Python (uv tool / uvx); inherited
# PYTHONPATH/PYTHONHOME (Hermes's venv) win over its site-packages → wrong-ABI
# C-extensions (pydantic_core) and a crash. Strip both.
# The CLI runs under its own Python (uv tool / uvx); inherited PYTHONPATH/PYTHONHOME
# (Hermes's venv) win over its site-packages → wrong-ABI C-extensions (pydantic_core)
# and a crash. Strip both.
env.pop("PYTHONPATH", None)
env.pop("PYTHONHOME", None)
env["PATH"] = _floor_subprocess_path(env.get("PATH", ""))
@@ -153,24 +170,18 @@ def _base_subprocess_env() -> dict:
def _floor_subprocess_path(path: str) -> str:
"""Guarantee core system dirs survive onto the CLI subprocess PATH.
Profile workers (kanban bots, cron) can inherit a PATH of only version-manager
dirs; the uv browser-use binary's POSIX sh trampoline resolves
``dirname``/``realpath`` through PATH and dies (exit 127) without /usr/bin.
Reuses browser_tool's ``_merge_browser_path`` floor, else appends FHS bin
dirs. Windows .cmd shims don't trampoline through PATH, so no-op there.
"""
"""Guarantee core system dirs survive onto the CLI subprocess PATH: profile workers
(kanban bots, cron) can inherit a PATH of only version-manager dirs, and the uv
browser-use binary's POSIX sh trampoline resolves ``dirname``/``realpath`` through
PATH and dies (exit 127) without /usr/bin. Reuses browser_tool's ``_merge_browser_path``
floor, else appends FHS bin dirs. Windows .cmd shims don't trampoline: no-op there."""
if os.name == "nt":
return path
try:
with contextlib.suppress(Exception):
from tools.browser_tool import _merge_browser_path
return _merge_browser_path(path or "")
except Exception:
pass
parts = [p for p in (path or "").split(os.pathsep) if p]
existing = set(parts)
parts.extend(d for d in _FHS_BIN_DIRS if d not in existing and os.path.isdir(d))
parts.extend(d for d in _FHS_BIN_DIRS if d not in set(parts) and os.path.isdir(d))
return os.pathsep.join(parts)
@@ -185,6 +196,10 @@ def _read_browser_cfg() -> dict:
return {}
def _use_gateway(browser_cfg: dict) -> bool:
return is_truthy_value(browser_cfg.get("use_gateway"), default=False)
def get_browser_backend() -> str:
"""Return the configured browser backend key ("" = unset → default).
@@ -198,16 +213,14 @@ def get_browser_backend() -> str:
def is_legacy_browser_use_cloud_config(browser_cfg: dict) -> bool:
"""True for pre-CLI direct-API Browser Use cloud configs"""
"""True for pre-CLI direct-API Browser Use cloud configs. An explicit backend or
a non-Browser-Use cloud_provider wins; Camofox is selected via env var, not
cloud_provider, so a Camofox user with a stray BROWSER_USE_API_KEY keeps it."""
if not isinstance(browser_cfg, dict) or browser_cfg.get("backend"):
return False # an explicit backend choice wins
provider = str(browser_cfg.get("cloud_provider") or "").strip().lower()
if provider not in {"browser-use", ""}:
return False # explicit local/Browserbase/… choices win
if is_truthy_value(browser_cfg.get("use_gateway"), default=False):
return False
# Camofox is selected via env var, not cloud_provider — a Camofox user
# with a stray BROWSER_USE_API_KEY must keep their explicit choice.
provider = str(browser_cfg.get("cloud_provider") or "").strip().lower()
if provider not in {"browser-use", ""} or _use_gateway(browser_cfg):
return False
if _camofox_active(" during migration"):
return False
return bool(os.getenv("BROWSER_USE_API_KEY"))
@@ -217,21 +230,17 @@ def is_browser_use_cli_mode() -> bool:
"""True when the Browser Use CLI replaces the built-in browser stack.
Browser Use mode is the DEFAULT: unset ``browser.backend`` ("") enables it
whenever the CLI is runnable (installed binary or uvx); ``browser.backend:
off`` (or ``/browser use off``) keeps the built-in browser_* tools. Camofox
always falls back to the built-in tools regardless — Firefox-based with a
custom HTTP API and no CDP surface, so the CDP-only harness cannot drive it.
whenever the CLI is runnable (installed binary or uvx), so browsing never
silently breaks; ``browser.backend: off`` (or ``/browser use off``) keeps the
built-in browser_* tools. Camofox always falls back to the built-in tools —
Firefox-based with a custom HTTP API and no CDP surface for the harness.
"""
if _camofox_active():
return False
backend = get_browser_backend()
if backend:
return backend == _BACKEND_KEY
if is_legacy_browser_use_cloud_config(_read_browser_cfg()):
return True
# Default (backend unset): Browser Use mode when the CLI can run at all;
# otherwise keep the built-in tools so browsing never silently breaks.
return _find_cli() is not None
return is_legacy_browser_use_cloud_config(_read_browser_cfg()) or _find_cli() is not None
_NOTICE_STAMP_NAME = ".browser_use_default_notice"
@@ -246,16 +255,12 @@ def default_downgrade_notice() -> Optional[str]:
if get_browser_backend() or _camofox_active() or _find_cli() is not None:
return None # explicit choice / Camofox / CLI present — nothing downgraded
stamp = Path(get_hermes_home()) / "cache" / _NOTICE_STAMP_NAME
try:
with contextlib.suppress(OSError):
if 0 <= time.time() - stamp.stat().st_mtime < _NOTICE_INTERVAL_S:
return None
except OSError:
pass
try:
with contextlib.suppress(OSError):
stamp.parent.mkdir(parents=True, exist_ok=True)
stamp.touch()
except OSError:
pass
return ("Browser Use CLI not found — using the built-in browser tools. "
"Run `hermes tools` (Browser Automation → Browser Use) to install it, "
"or `browser.backend: off` in config.yaml to silence this.")
@@ -296,27 +301,25 @@ def _find_cli() -> Optional[List[str]]:
def install_cli(timeout_s: int = 600) -> Tuple[bool, str]:
"""Install the browser-use CLI persistently via ``uv tool install``
(managed uv via ``ensure_uv`` → uv on PATH), linking the binary into
``$HERMES_HOME/bin`` (``UV_TOOL_BIN_DIR``) so ``_find_cli()`` resolves it for
every profile without touching the user's PATH. Returns ``(ok, message)``;
never raises.
"""Install the browser-use CLI persistently via ``uv tool install`` (managed uv
via ``ensure_uv`` → uv on PATH), linking the binary into ``$HERMES_HOME/bin``
(``UV_TOOL_BIN_DIR``) so ``_find_cli()`` resolves it for every profile without
touching the user's PATH. Returns ``(ok, message)``; never raises.
MANAGED-FIRST: only the managed copy short-circuits. A browser-use on PATH is a
user-level side install and must NOT prevent provisioning the canonical
Hermes-managed copy (version drift, no updates via hermes tools).
"""
# MANAGED-FIRST: only the managed copy short-circuits. A browser-use on PATH
# is a user-level side install and must NOT prevent provisioning the
# canonical Hermes-managed copy (version drift, no updates via hermes tools).
bin_dir = _managed_bin_dir()
managed = shutil.which("browser-use", path=bin_dir)
if managed:
return True, f"browser-use CLI already installed ({managed})"
uv_bin: Optional[str] = None
try:
def _managed_uv() -> Optional[str]:
from hermes_cli.managed_uv import ensure_uv
uv_bin = str(ensure_uv() or "") or None
except Exception as e:
logger.debug("Managed uv bootstrap unavailable: %s", e)
uv_bin = uv_bin or shutil.which("uv")
return str(ensure_uv() or "") or None
uv_bin = _quiet(_managed_uv, None, "Managed uv bootstrap unavailable") or shutil.which("uv")
if not uv_bin:
return False, ("uv is not available and could not be bootstrapped. Install uv "
"(https://docs.astral.sh/uv/) and run `uv tool install browser-use`.")
@@ -385,9 +388,9 @@ def _native_screenshot_result(result: Dict[str, Any], path: str) -> Optional[Dic
if not _should_use_native_vision_fast_path():
return None
# History-reuse cap: this data URL bakes into the tool result and is
# re-sent every later turn — same policy as the vision_analyze /
# browser_vision native embeds (256 KB / 1568 px, JPEG quality ladder).
# History-reuse cap: this data URL bakes into the tool result and is re-sent
# every later turn — same policy as the vision_analyze / browser_vision native
# embeds (256 KB / 1568 px, JPEG quality ladder).
data_url = _resize_image_for_vision(
Path(path), mime_type="image/png", max_base64_bytes=_EMBED_TARGET_BYTES,
max_dimension=_EMBED_MAX_DIMENSION, force_jpeg=True,
@@ -418,13 +421,9 @@ def _resolve_lightpanda_cdp(env: dict, task_id: Optional[str], session_name: str
private to this session and the own-tab preamble is skipped."""
try:
from tools.browser_tool import _get_session_info, _using_lightpanda_engine
except Exception as e: # pragma: no cover — stubbed browser_tool in tests
logger.debug("browser_tool lightpanda resolution unavailable: %s", e)
return None
try:
if not _using_lightpanda_engine():
return None
except Exception as e:
except Exception as e: # stubbed browser_tool in tests / engine lookup failure
logger.debug("browser engine lookup failed: %s", e)
return None
err = _export_session_cdp(
@@ -460,29 +459,22 @@ def _resolve_backend_cdp(env: dict, task_id: Optional[str], session_name: str =
logger.debug("browser_tool backend resolution unavailable: %s", e)
return None
try:
override = _get_cdp_override()
except Exception:
override = ""
override = _quiet(_get_cdp_override, "")
if override:
_set_cdp_env(env, override)
return None
try:
provider = _get_cloud_provider()
except Exception as e:
logger.debug("Cloud provider lookup failed: %s", e)
provider = None
provider = _quiet(_get_cloud_provider, None, "Cloud provider lookup failed")
if provider is None:
return _resolve_lightpanda_cdp(env, task_id, session_name)
# Browser Use direct-API configs: the CLI talks to BU cloud natively
# (BU_AUTOSPAWN / auth login) — the legacy provider would create a second,
# redundant session. The Nous-gateway variant (use_gateway: true) DOES resolve
# through the provider: the gateway provisions the browser server-side and
# returns its CDP URL, giving subscribers CLI mode with no raw key.
# Browser Use direct-API configs: the CLI talks to BU cloud natively (BU_AUTOSPAWN /
# auth login) — the legacy provider would create a second, redundant session. The
# Nous-gateway variant (use_gateway: true) DOES resolve through the provider: the
# gateway provisions the browser server-side and returns its CDP URL, giving
# subscribers CLI mode with no raw key.
provider_key = str(getattr(provider, "name", "") or "").strip().lower()
if provider_key == _BACKEND_KEY and not is_truthy_value(_read_browser_cfg().get("use_gateway"), default=False):
if provider_key == _BACKEND_KEY and not _use_gateway(_read_browser_cfg()):
env[_PRIVATE_BROWSER_SENTINEL] = "1" # named BU cloud browsers are exclusive to their daemon
return None
@@ -501,11 +493,6 @@ def _resolve_backend_cdp(env: dict, task_id: Optional[str], session_name: str =
return err
def _real_profile_consented() -> bool:
"""Whether the user opted in to real-profile local browsing (config read)."""
return _lazy_call("tools.browser_tool", "_use_real_profile", False, "real-profile consent lookup failed")
def _resolve_real_profile_cdp(env: dict, force_local: bool) -> Optional[str]:
"""Point the harness at the user's real-profile copy-browser when consented.
@@ -527,19 +514,14 @@ def _resolve_real_profile_cdp(env: dict, force_local: bool) -> Optional[str]:
logger.debug("real-profile backend resolution unavailable: %s", e)
return None
try:
if _get_cdp_override_raw():
return None
except Exception:
pass
if _quiet(_get_cdp_override_raw, ""):
return None
if not force_local:
# Only auto-upgrade genuinely-local attaches; any cloud path (provider or
# legacy BU cloud config) stays on its backend unless the model passes local=true.
try:
if _get_cloud_provider() is not None:
return None
except Exception:
# Only auto-upgrade genuinely-local attaches; any cloud path (provider, provider
# lookup failure, or legacy BU cloud config) stays on its backend unless the model
# passes local=true.
if _quiet(_get_cloud_provider, object()) is not None:
return None
if is_legacy_browser_use_cloud_config(_read_browser_cfg()):
return None
@@ -576,14 +558,13 @@ def _windows_popen_kwargs() -> dict:
"""Hide the console the .cmd shim would flash on Windows (as browser_tool does)."""
if os.name != "nt":
return {}
try:
def _flags() -> dict:
from hermes_cli._subprocess_compat import windows_hide_flags
si = subprocess.STARTUPINFO()
si.dwFlags |= subprocess.STARTF_USESHOWWINDOW
return {"creationflags": windows_hide_flags(), "startupinfo": si}
except Exception as e:
logger.debug("Windows hide-flags unavailable: %s", e)
return {}
return _quiet(_flags, {}, "Windows hide-flags unavailable")
def _clamp_timeout(timeout_s: Any) -> int:
@@ -746,21 +727,12 @@ _HELPERS_DIGEST = (
# lives with the session, not in the process-wide TTL-cached check_fn.
def _lightpanda_engine_in_use() -> bool:
return _lazy_call("tools.browser_tool", "lightpanda_engine_status", (False, ""),
"lightpanda engine status unavailable")[0]
def _description_header() -> str:
"""Header tailored to whether the active model can see images natively"""
if _lightpanda_engine_in_use(): # no screenshots at all, whatever the model can see
return _HEADER_BASE + _HEADER_TEXT_ONLY + _HEADER_LIGHTPANDA
try:
from tools.vision_tools import _should_use_native_vision_fast_path
if _should_use_native_vision_fast_path():
return _HEADER_BASE + _HEADER_VISION
except Exception:
pass
if _lazy_call("tools.vision_tools", "_should_use_native_vision_fast_path", False, ""):
return _HEADER_BASE + _HEADER_VISION
return _HEADER_BASE + _HEADER_TEXT_ONLY