Files
hermes-agent/plugins/google_meet/meet_bot.py
T

606 lines
23 KiB
Python

"""Headless Google Meet bot — Playwright + live-caption scraping.
Standalone subprocess spawned by ``process_manager.py``. Config comes from env
vars; status + transcript are written under ``$HERMES_MEET_OUT_DIR`` and read
by the ``meet_*`` tools — no IPC beyond the filesystem.
We don't parse WebRTC audio: we enable Meet's built-in live captions and watch
the caption container via a MutationObserver. Lossy and English-biased, but
deterministic (no STT billing) and stable across Meet UI rewrites thanks to
the container's ARIA role. Only ``https://meet.google.com/`` URLs are accepted.
Debug run: ``HERMES_MEET_URL=... HERMES_MEET_OUT_DIR=/tmp/meet-debug
HERMES_MEET_HEADED=1 python -m plugins.google_meet.meet_bot``
"""
from __future__ import annotations
import os
import re
import shutil
import signal
import subprocess
import sys
import threading
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Optional
from plugins.google_meet._jsonfile import write_json_atomic
# Short three-segment code, a lookup URL, or /new. Anything else is rejected.
MEET_URL_RE = re.compile(
r"^https://meet\.google\.com/("
r"[a-z0-9]{3,}-[a-z0-9]{3,}-[a-z0-9]{3,}"
r"|lookup/[^/?#]+"
r"|new"
r")(?:[/?#].*)?$"
)
# Filenames the bot reads/writes in ``HERMES_MEET_OUT_DIR``.
SAY_QUEUE_FILENAME = "say_queue.jsonl"
SAY_PCM_FILENAME = "speaker.pcm"
_FFMPEG_MISSING = "ffmpeg not found — install via `brew install ffmpeg` for realtime on macOS"
def _is_safe_meet_url(url: str) -> bool:
"""True if *url* is a Google Meet URL we're willing to navigate to."""
return isinstance(url, str) and bool(MEET_URL_RE.match(url.strip()))
def _meeting_id_from_url(url: str) -> str:
"""3-segment meeting code, or a timestamped id for ``/lookup/...`` and ``/new``."""
m = re.search(r"meet\.google\.com/([a-z0-9]{3,}-[a-z0-9]{3,}-[a-z0-9]{3,})", url or "")
return m.group(1) if m else f"meet-{int(time.time())}"
def _quiet(fn, *args, **kwargs):
"""Call *fn*, swallowing any exception (best-effort teardown steps)."""
try:
return fn(*args, **kwargs)
except Exception:
return None
# status.json keys in file order → _BotState attribute + initial value.
_STATUS_FIELDS = (
("meetingId", "meeting_id", None), ("url", "url", None),
("inCall", "in_call", False), ("captioning", "captioning", False),
("captionsEnabledAttempted", "captions_enabled_attempted", False),
("lobbyWaiting", "lobby_waiting", False),
("joinAttemptedAt", "join_attempted_at", None), ("joinedAt", "joined_at", None),
("lastCaptionAt", "last_caption_at", None), ("transcriptLines", "transcript_lines", 0),
("transcriptPath", "transcript_path", None),
("error", "error", None), ("exited", "exited", False), ("pid", None, None),
# v2 realtime telemetry.
("realtime", "realtime", False), ("realtimeReady", "realtime_ready", False),
("realtimeDevice", "realtime_device", None), ("audioBytesOut", "audio_bytes_out", 0),
("lastAudioOutAt", "last_audio_out_at", None), ("lastBargeInAt", "last_barge_in_at", None),
("leaveReason", "leave_reason", None),
)
class _BotState:
"""Single-process mutable state, flushed to ``status.json`` on each change."""
def __init__(self, out_dir: Path, meeting_id: str, url: str):
for _, attr, default in _STATUS_FIELDS:
if attr:
setattr(self, attr, default)
self.out_dir = out_dir
self.meeting_id = meeting_id
self.url = url
self._seen: set = set() # "speaker|text" keys already written
out_dir.mkdir(parents=True, exist_ok=True)
self.transcript_path = out_dir / "transcript.txt"
self.status_path = out_dir / "status.json"
self._flush()
def record_caption(self, speaker: str, text: str) -> None:
"""Append a caption line unless this exact (speaker, text) was already seen."""
speaker = (speaker or "").strip() or "Unknown"
text = (text or "").strip()
key = f"{speaker}|{text}"
if not text or key in self._seen:
return
self._seen.add(key)
self.transcript_lines += 1
self.last_caption_at = time.time()
ts = time.strftime("%H:%M:%S", time.localtime(self.last_caption_at))
with self.transcript_path.open("a", encoding="utf-8") as f:
f.write(f"[{ts}] {speaker}: {text}\n")
self._flush()
def _flush(self) -> None:
data = {key: getattr(self, attr) if attr else None for key, attr, _ in _STATUS_FIELDS}
data["transcriptPath"] = str(self.transcript_path)
data["pid"] = os.getpid() # overrides keep the table's key order
write_json_atomic(self.status_path, data)
def set(self, **kwargs) -> None:
for k, v in kwargs.items():
setattr(self, k, v)
self._flush()
# JS injected into the Meet tab: MutationObserver on the caption container
# collects {speaker, text}; ``window.__hermesMeetDrain()`` pulls new entries.
_CAPTION_OBSERVER_JS = r"""
(() => {
if (window.__hermesMeetInstalled) return;
window.__hermesMeetInstalled = true;
window.__hermesMeetQueue = [];
const captionSelector = '[role="region"][aria-label*="aption" i], ' +
'div[jsname="YSxPC"], ' + // legacy
'div[jsname="tgaKEf"]'; // current (Apr 2026)
function pushEntry(speaker, text) {
if (!text || !text.trim()) return;
window.__hermesMeetQueue.push({
ts: Date.now(),
speaker: (speaker || '').trim(),
text: text.trim(),
});
}
function scan(root) {
// Meet captions render as rows of speaker label + text block. Selectors
// vary across Meet rewrites; try a few shapes and fall back to raw text.
const rows = root.querySelectorAll('div[jsname="dsyhDe"], div.CNusmb, div.TBMuR');
if (rows.length) {
rows.forEach((row) => {
const spkEl = row.querySelector('div.KcIKyf, div.zs7s8d, span[jsname="YSxPC"]');
const txtEl = row.querySelector('div.bh44bd, span[jsname="tgaKEf"], div.iTTPOb');
pushEntry(spkEl ? spkEl.innerText : '', txtEl ? txtEl.innerText : row.innerText);
});
return;
}
// Fallback: treat the whole region's innerText as one anonymous line.
pushEntry('', (root.innerText || '').split('\n').filter(Boolean).pop());
}
function attach() {
const el = document.querySelector(captionSelector);
if (!el) return false;
new MutationObserver(() => scan(el)).observe(el, { childList: true, subtree: true, characterData: true });
scan(el);
return true;
}
// Retry on interval — the caption region only appears after captions are
// enabled and someone speaks.
if (!attach()) {
const iv = setInterval(() => { if (attach()) clearInterval(iv); }, 1500);
}
window.__hermesMeetDrain = () => {
const out = window.__hermesMeetQueue.slice();
window.__hermesMeetQueue = [];
return out;
};
})();
"""
# Best-effort caption toggle: Meet binds it to the ``c`` key; click targeting
# is too brittle to rely on.
_ENABLE_CAPTIONS_JS = (
"(() => { document.body.dispatchEvent(new KeyboardEvent('keydown', "
"{ key: 'c', code: 'KeyC', keyCode: 67, which: 67, bubbles: true })); return true; })();"
)
_LEAVE_CALL_JS = (
"() => { const b = document.querySelector('button[aria-label*=\"eave call\"]');"
" if (b) b.click(); }"
)
# True once we're clearly past the lobby: leave button, caption region
# (only once our observer is installed) or participant list visible.
_ADMISSION_PROBE_JS = r"""
(() => {
if (document.querySelector('button[aria-label*="eave call" i]')) return true;
if (window.__hermesMeetInstalled && document.querySelector(
'[role="region"][aria-label*="aption" i], div[jsname="YSxPC"], div[jsname="tgaKEf"]')) return true;
return !!document.querySelector('[aria-label*="articipants" i]');
})();
"""
# English only — what Meet shows when the host denies or removes a guest.
_DENIED_PROBE_JS = r"""
(() => {
const text = document.body ? document.body.innerText || '' : '';
return /You can't join this video call|You were removed from the meeting|No one responded to your request to join/i.test(text);
})();
"""
def _probe(page, js: str) -> bool:
"""Evaluate a boolean JS probe; conservative — False on any error."""
try:
return bool(page.evaluate(js))
except Exception:
return False
def _visible(locator):
"""``locator.first`` if it exists and is visible, else None (swallows Playwright errors)."""
try:
first = locator.first
return first if first.count() and first.is_visible() else None
except Exception:
return None
def _start_pcm_pump(rt: dict, bridge_info: dict, pcm_path: Path, state: "_BotState") -> None:
"""Stream the growing ``speaker.pcm`` (24kHz s16le mono) into the OS device Chrome's fake mic reads."""
bridge_info = bridge_info or {}
platform_tag = bridge_info.get("platform")
if platform_tag == "linux":
sink = bridge_info.get("write_target") or "hermes_meet_sink"
cmd = ["paplay", "--raw", "--rate=24000", "--format=s16le", "--channels=1",
f"--device={sink}", str(pcm_path)]
missing = "paplay not found — install pulseaudio-utils for realtime on Linux"
elif platform_tag == "darwin":
# The user must have BlackHole selected as default input for Chrome
# to pick it up; ffmpeg targets the device by audiotoolbox index.
if not shutil.which("ffmpeg"):
state.set(error=_FFMPEG_MISSING)
return
device_name = bridge_info.get("write_target") or "BlackHole 2ch"
cmd = ["ffmpeg", "-nostdin", "-hide_banner", "-loglevel", "error", "-re",
"-f", "s16le", "-ar", "24000", "-ac", "1", "-i", str(pcm_path),
"-f", "audiotoolbox", "-audio_device_index", _mac_audio_device_index(device_name), "-"]
missing = _FFMPEG_MISSING
else:
return
try:
rt["pcm_pump"] = subprocess.Popen(
cmd, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
)
except FileNotFoundError:
state.set(error=missing)
except Exception as e:
if platform_tag != "darwin":
raise
state.set(error=f"macOS pcm pump failed to start: {e}")
def _start_realtime_speaker(rt: dict, cfg: "_BotConfig", stop_flag: dict, state: "_BotState") -> None:
"""Wire up the OpenAI Realtime session, the say-queue speaker thread and the PCM pump."""
try:
from plugins.google_meet.realtime.openai_client import RealtimeSession, RealtimeSpeaker
except Exception as e:
state.set(error=f"realtime import failed: {e}")
return
pcm_path = cfg.out_dir / SAY_PCM_FILENAME
queue_path = cfg.out_dir / SAY_QUEUE_FILENAME
pcm_path.write_bytes(b"") # start each session with a clean sink file
queue_path.touch() # so the speaker poller doesn't error on first iteration
try:
session = RealtimeSession(
api_key=cfg.realtime_api_key, model=cfg.realtime_model, voice=cfg.realtime_voice,
instructions=cfg.realtime_instructions, audio_sink_path=pcm_path, sample_rate=24000,
)
session.connect()
except Exception as e:
state.set(error=f"realtime connect failed: {e}")
return
rt["session"] = session
speaker = RealtimeSpeaker(
session=session, queue_path=queue_path, processed_path=cfg.out_dir / "say_processed.jsonl",
)
def _speaker_loop():
try:
speaker.run_until_stopped(lambda: stop_flag.get("stop", False))
except Exception as e:
state.set(error=f"realtime speaker crashed: {e}")
rt["speaker_thread"] = threading.Thread(target=_speaker_loop, name="meet-speaker", daemon=True)
rt["speaker_thread"].start()
_start_pcm_pump(rt, rt["bridge_info"], pcm_path, state)
state.set(realtime_ready=True)
def _mac_audio_device_index(device_name: str) -> str:
"""ffmpeg ``-audio_device_index`` for *device_name* (case-insensitive), ``"0"`` if not found.
ffmpeg prints the avfoundation device table on stderr as ``[N] Name``.
"""
try:
out = subprocess.run(
["ffmpeg", "-f", "avfoundation", "-list_devices", "true", "-i", ""],
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=10,
)
except Exception:
return "0"
needle = device_name.strip().lower()
for line in (out.stderr or "").splitlines():
m = re.search(r"\[(\d+)\]\s+(.+)$", line)
if m and m.group(2).strip().lower() == needle:
return m.group(1)
return "0"
def _setup_realtime(rt: dict, api_key: str, state: _BotState) -> None:
"""Provision the virtual audio bridge; on any failure fall back to transcribe mode."""
if not api_key:
state.set(error="realtime mode requested but no API key in HERMES_MEET_REALTIME_KEY/OPENAI_API_KEY — falling back to transcribe")
rt["enabled"] = False
return
try:
from plugins.google_meet.audio_bridge import AudioBridge
bridge = AudioBridge()
rt["bridge_info"] = bridge.setup()
rt["bridge"] = bridge
state.set(realtime=True, realtime_device=rt["bridge_info"].get("device_name"))
except Exception as e:
state.set(error=f"audio bridge setup failed: {e} — falling back to transcribe")
rt["enabled"] = False
def _teardown_realtime(rt: dict) -> None:
if rt.get("pcm_pump"):
_quiet(rt["pcm_pump"].terminate)
_quiet(rt["pcm_pump"].wait, timeout=3)
if rt["speaker_thread"] is not None:
_quiet(rt["speaker_thread"].join, timeout=5.0)
if rt["session"]:
_quiet(rt["session"].close)
if rt["bridge"]:
_quiet(rt["bridge"].teardown)
@dataclass
class _BotConfig:
"""Everything the bot reads from ``HERMES_MEET_*`` env vars."""
url: str
out_dir: Optional[Path]
headed: bool
auth_state: str
guest_name: str
duration_s: Optional[float]
realtime: bool
realtime_api_key: str
realtime_model: str
realtime_voice: str
realtime_instructions: str
lobby_timeout: float
def _config_from_env() -> _BotConfig:
env = os.environ.get
out_raw = env("HERMES_MEET_OUT_DIR", "").strip()
return _BotConfig(
url=env("HERMES_MEET_URL", "").strip(),
out_dir=Path(out_raw) if out_raw else None,
headed=env("HERMES_MEET_HEADED", "").lower() in {"1", "true", "yes"},
auth_state=env("HERMES_MEET_AUTH_STATE", "").strip(),
guest_name=env("HERMES_MEET_GUEST_NAME", "Hermes Agent"),
duration_s=_parse_duration(env("HERMES_MEET_DURATION", "")),
realtime=env("HERMES_MEET_MODE", "transcribe").strip().lower() == "realtime",
# HERMES_MEET_REALTIME_KEY is resolved by process_manager.start() through the
# parent's profile secret scope; the OPENAI_API_KEY fallback only serves
# standalone `python -m plugins.google_meet.meet_bot` runs.
realtime_api_key=env("HERMES_MEET_REALTIME_KEY") or env("OPENAI_API_KEY", ""),
realtime_model=env("HERMES_MEET_REALTIME_MODEL", "gpt-realtime"),
realtime_voice=env("HERMES_MEET_REALTIME_VOICE", "alloy"),
realtime_instructions=env("HERMES_MEET_REALTIME_INSTRUCTIONS", ""),
lobby_timeout=float(env("HERMES_MEET_LOBBY_TIMEOUT", "300")),
)
def _join(page, cfg: _BotConfig, state: _BotState) -> None:
"""Fill the guest-name field (guest mode) and click 'Join now' / 'Ask to join'.
'Ask to join' means we're in the lobby → ``lobby_waiting``.
"""
name_box = _visible(page.locator('input[aria-label*="name" i]'))
if name_box is not None:
_quiet(name_box.fill, cfg.guest_name, timeout=2_000)
for label in ("Join now", "Ask to join"):
btn = _visible(page.get_by_role("button", name=label, exact=False))
if btn is None:
continue
try:
btn.click(timeout=3_000)
if label == "Ask to join":
state.set(lobby_waiting=True)
break
except Exception:
continue
def _drain_loop(page, cfg: _BotConfig, state: _BotState, rt: dict, stop_flag: dict) -> None:
"""Admission + caption drain loop; runs until SIGTERM, duration expiry, lobby timeout/denial or page loss.
Sets ``state.leave_reason`` for every exit but SIGTERM. Also triggers
barge-in and mirrors realtime counters into status.json.
"""
deadline = (time.time() + cfg.duration_s) if cfg.duration_s else None
lobby_deadline = time.time() + cfg.lobby_timeout
last_admission_check = 0.0
while not stop_flag["stop"]:
now = time.time()
if deadline and now > deadline:
state.set(leave_reason="duration_expired")
return
if not state.in_call and (now - last_admission_check) > 3.0:
last_admission_check = now
if _probe(page, _ADMISSION_PROBE_JS):
state.set(in_call=True, lobby_waiting=False, joined_at=now)
elif now > lobby_deadline:
waited = int(lobby_deadline - state.join_attempted_at) if state.join_attempted_at else 0
state.set(
error=f"lobby timeout — host never admitted the bot within {waited}s",
leave_reason="lobby_timeout",
)
return
elif _probe(page, _DENIED_PROBE_JS):
state.set(error="host denied admission", leave_reason="denied")
return
try:
queued = page.evaluate("window.__hermesMeetDrain && window.__hermesMeetDrain()")
for entry in queued if isinstance(queued, list) else ():
if not isinstance(entry, dict):
continue
speaker = str(entry.get("speaker", ""))
state.record_caption(speaker=speaker, text=str(entry.get("text", "")))
# Barge-in: a real human spoke while we may be generating
# audio — cancel the in-flight response.
if (rt["session"] is not None
and _looks_like_human_speaker(speaker, cfg.guest_name)
and _quiet(rt["session"].cancel_response)):
state.set(last_barge_in_at=now)
except Exception:
# Meet reloaded or we got booted — exit rather than spin.
if page.is_closed():
state.set(leave_reason="page_closed")
return
if rt["session"] is not None:
state.set(
audio_bytes_out=rt["session"].audio_bytes_out,
last_audio_out_at=rt["session"].last_audio_out_at,
)
time.sleep(1.0)
def run_bot() -> int:
cfg = _config_from_env()
if not _is_safe_meet_url(cfg.url):
sys.stderr.write(
"google_meet bot: refusing to launch — HERMES_MEET_URL must be a "
"meet.google.com URL. got: %r\n" % cfg.url
)
return 2
if cfg.out_dir is None:
sys.stderr.write("google_meet bot: HERMES_MEET_OUT_DIR is required\n")
return 2
state = _BotState(out_dir=cfg.out_dir, meeting_id=_meeting_id_from_url(cfg.url), url=cfg.url)
# SIGTERM sets a flag (not an exception) so the Playwright teardown below
# still runs and ``meet_leave`` gets a finalized transcript.
stop_flag = {"stop": False}
def _on_signal(_sig, _frame):
stop_flag["stop"] = True
signal.signal(signal.SIGTERM, _on_signal)
signal.signal(signal.SIGINT, _on_signal)
# Realtime resources tracked in one dict so teardown works however we exit.
rt = {"enabled": cfg.realtime, "bridge": None, "bridge_info": None,
"session": None, "speaker_thread": None}
if rt["enabled"]:
_setup_realtime(rt, cfg.realtime_api_key, state)
try:
from playwright.sync_api import sync_playwright
except ImportError as e:
state.set(error=f"playwright not installed: {e}", exited=True)
sys.stderr.write(
"google_meet bot: playwright is not installed. Run "
"`pip install playwright && python -m playwright install chromium`\n"
)
if rt["bridge"]:
rt["bridge"].teardown()
return 3
chrome_args = ["--use-fake-ui-for-media-stream", "--disable-blink-features=AutomationControlled"]
if not rt["enabled"]:
# Silent fake device — mic content is irrelevant when we're not speaking.
chrome_args.insert(1, "--use-fake-device-for-media-stream")
elif rt["bridge_info"] and rt["bridge_info"].get("platform") == "linux":
# Playwright's launch() takes no env: set PULSE_SOURCE on our own
# process so the child Chrome inherits the virtual source.
os.environ["PULSE_SOURCE"] = rt["bridge_info"].get("device_name", "")
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(headless=not cfg.headed, args=chrome_args)
context_args = {
"viewport": {"width": 1280, "height": 800},
"user_agent": (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
),
"permissions": ["microphone", "camera"],
}
if cfg.auth_state and Path(cfg.auth_state).is_file():
context_args["storage_state"] = cfg.auth_state
context = browser.new_context(**context_args)
page = context.new_page()
try:
page.goto(cfg.url, wait_until="domcontentloaded", timeout=30_000)
except Exception as e:
state.set(error=f"navigate failed: {e}", exited=True)
return 4
_join(page, cfg, state)
if _quiet(page.evaluate, _ENABLE_CAPTIONS_JS):
state.set(captions_enabled_attempted=True)
try:
page.evaluate(_CAPTION_OBSERVER_JS)
except Exception as e:
state.set(error=f"caption observer install failed: {e}")
# in_call stays False until admission is confirmed by the drain loop.
state.set(captioning=True, join_attempted_at=time.time())
if rt["enabled"]:
_start_realtime_speaker(rt, cfg, stop_flag, state)
_drain_loop(page, cfg, state, rt, stop_flag)
_quiet(page.evaluate, _LEAVE_CALL_JS)
context.close()
browser.close()
_teardown_realtime(rt)
state.set(in_call=False, captioning=False, exited=True)
return 0
except Exception as e:
state.set(error=f"unhandled: {e}", exited=True)
return 1
def _looks_like_human_speaker(speaker: str, bot_guest_name: str) -> bool:
"""Whether a caption's speaker is probably a human rather than our own echo.
Meet attributes our fake-mic audio to the bot's own name; blank/unknown
speakers (raw-text fallback) are ambiguous, so neither triggers barge-in.
"""
if not speaker or not speaker.strip():
return False
return speaker.strip().lower() not in {"unknown", "you", bot_guest_name.strip().lower()}
_DURATION_UNITS = {"h": 3600.0, "m": 60.0, "s": 1.0}
def _parse_duration(raw: str) -> Optional[float]:
"""Parse ``30m`` / ``2h`` / ``90`` (seconds) → float seconds, or None."""
if not raw:
return None
raw = raw.strip().lower()
mult = _DURATION_UNITS.get(raw[-1:])
try:
return float(raw[:-1]) * mult if mult else float(raw)
except ValueError:
return None
if __name__ == "__main__": # pragma: no cover — subprocess entry point
sys.exit(run_bot())