395 lines
17 KiB
Python
395 lines
17 KiB
Python
"""
|
|
Transport-agnostic WhatsApp behavior shared by the Baileys bridge adapter and the
|
|
Cloud API adapter: allow-list / DM / group gating, mention detection, quoted-reply-
|
|
to-bot detection, broadcast filtering, WhatsApp markdown conversion, chunk budgeting.
|
|
|
|
Mixin contract — the adapter must set these on ``self`` before any of the
|
|
mixin's methods are called (typically in ``__init__``):
|
|
|
|
self.config # gateway.config.PlatformConfig
|
|
self.name # str — adapter name (used in log lines)
|
|
self._dm_policy # str: "open" | "allowlist" | "disabled"
|
|
self._allow_from # set[str]
|
|
self._group_policy # str: "open" | "allowlist" | "disabled"
|
|
self._group_allow_from # set[str]
|
|
self._mention_patterns # list[re.Pattern]
|
|
self._reply_prefix # Optional[str]
|
|
|
|
Class attributes ``MAX_MESSAGE_LENGTH`` and ``DEFAULT_REPLY_PREFIX`` are
|
|
defined on the mixin and may be overridden per-adapter if needed.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
from typing import Any, Dict, Optional
|
|
|
|
from gateway.platforms._shared import get_scoped_secret as _get_wsecret
|
|
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_TRUTHY = {"true", "1", "yes", "on"}
|
|
|
|
|
|
class WhatsAppBehaviorMixin:
|
|
"""Shared behavior for all WhatsApp adapters (Baileys + Cloud API); owns no state
|
|
of its own — see the module docstring for the host adapter's attribute contract."""
|
|
|
|
# Practical UX limit, not the ~65K protocol max (long messages are unreadable on mobile).
|
|
MAX_MESSAGE_LENGTH: int = 4096
|
|
supports_code_blocks = True # WhatsApp renders fenced code blocks (monospace)
|
|
|
|
DEFAULT_REPLY_PREFIX: str = "⚕ *Hermes Agent*\n────────────\n"
|
|
|
|
_OUTBOUND_INVISIBLE_CHARS_RE = re.compile(r"[\u200b\u2060\u2063\ufeff]")
|
|
_OUTBOUND_ODD_SPACE_RE = re.compile(r"[\u00a0\u1680\u180e\u2000-\u200a\u202f\u205f\u3000]")
|
|
|
|
@classmethod
|
|
def _sanitize_outbound_text(cls, content: str) -> str:
|
|
"""Strip zero-width format chars (WORD JOINER etc.) and normalize odd unicode
|
|
spaces — WhatsApp renders them as mojibake prefixes. Emoji joiners are kept."""
|
|
if not content:
|
|
return content
|
|
content = cls._OUTBOUND_INVISIBLE_CHARS_RE.sub("", content)
|
|
return cls._OUTBOUND_ODD_SPACE_RE.sub(" ", content)
|
|
|
|
@property
|
|
def enforces_own_access_policy(self) -> bool:
|
|
"""WhatsApp gates DM/group access at intake via dm_policy/group_policy."""
|
|
return True
|
|
|
|
def _effective_reply_prefix(self) -> str:
|
|
"""Prefix for outgoing replies in self-chat mode (Cloud API overrides to ``""``)."""
|
|
whatsapp_mode = _get_wsecret("WHATSAPP_MODE", default="self-chat") or "self-chat"
|
|
if whatsapp_mode != "self-chat":
|
|
return ""
|
|
if self._reply_prefix is not None:
|
|
return self._reply_prefix.replace("\\n", "\n")
|
|
env_prefix = _get_wsecret("WHATSAPP_REPLY_PREFIX")
|
|
if env_prefix is not None:
|
|
return env_prefix.replace("\\n", "\n")
|
|
return self.DEFAULT_REPLY_PREFIX
|
|
|
|
def _outgoing_chunk_limit(self) -> int:
|
|
"""Reserve room for the reply prefix; floor keeps space for pagination/fence repair."""
|
|
return max(1024, self.MAX_MESSAGE_LENGTH - len(self._effective_reply_prefix()))
|
|
|
|
def _whatsapp_require_mention(self) -> bool:
|
|
configured = self.config.extra.get("require_mention")
|
|
if configured is not None:
|
|
if isinstance(configured, str):
|
|
return configured.lower() in _TRUTHY
|
|
return bool(configured)
|
|
return (_get_wsecret("WHATSAPP_REQUIRE_MENTION", default="false") or "false").lower() in _TRUTHY
|
|
|
|
def _whatsapp_free_response_chats(self) -> set[str]:
|
|
raw = self.config.extra.get("free_response_chats")
|
|
if raw is None:
|
|
raw = _get_wsecret("WHATSAPP_FREE_RESPONSE_CHATS", default="") or ""
|
|
return self._coerce_allow_list(raw)
|
|
|
|
@staticmethod
|
|
def _coerce_allow_list(raw) -> set[str]:
|
|
"""Parse allow_from / group_allow_from from config (list) or env var (CSV)."""
|
|
if raw is None:
|
|
return set()
|
|
if isinstance(raw, list):
|
|
return {str(part).strip() for part in raw if str(part).strip()}
|
|
return {part.strip() for part in str(raw).split(",") if part.strip()}
|
|
|
|
def _select_dm_allowlist(self, extra: Dict[str, Any], env_keys, read_env) -> Any:
|
|
"""Pick the raw DM allowlist by key *presence*: ``allow_from``/``allowFrom`` in
|
|
config (an explicit empty list stays authoritative), then the first truthy
|
|
env carrier. Records the winning source in ``_dm_allowlist_source`` so live
|
|
DM checks keep the same precedence."""
|
|
for key in ("allow_from", "allowFrom"):
|
|
if key in extra:
|
|
self._dm_allowlist_source = "config"
|
|
return extra.get(key)
|
|
for env in env_keys:
|
|
if read_env(env):
|
|
self._dm_allowlist_source = env
|
|
return read_env(env)
|
|
self._dm_allowlist_source = None
|
|
return None
|
|
|
|
def _live_dm_allow_from(self) -> set[str]:
|
|
"""Allowlist currently enforced for DM intake / strict DM auth.
|
|
|
|
Env-seeded adapters re-read the same key so pairing approve/revoke takes
|
|
effect without restart; a removed key (sole-entry revoke) means empty, not
|
|
the construction snapshot. Config-seeded adapters keep the in-memory set
|
|
(pairing revoke purges it in place) — a stale env value must not broaden access.
|
|
"""
|
|
source = getattr(self, "_dm_allowlist_source", None)
|
|
if isinstance(source, str) and source != "config":
|
|
if source in os.environ:
|
|
return self._coerce_allow_list(os.environ.get(source, ""))
|
|
return set()
|
|
return set(self._allow_from or ())
|
|
|
|
# ------------------------------------------------------------------ JID helpers
|
|
@staticmethod
|
|
def _normalize_whatsapp_id(value: Optional[str]) -> str:
|
|
if not value:
|
|
return ""
|
|
normalized = str(value).strip()
|
|
if ":" in normalized and "@" in normalized:
|
|
normalized = normalized.replace(":", "@", 1)
|
|
return normalized
|
|
|
|
@staticmethod
|
|
def _is_broadcast_chat(chat_id: str) -> bool:
|
|
"""Status updates (Stories) and Channel/Newsletter broadcasts — never reply
|
|
(answering a Story spams the status feed; Channel posts aren't addressable)."""
|
|
if not chat_id:
|
|
return False
|
|
cid = chat_id.strip().lower()
|
|
return cid == "status@broadcast" or cid.endswith("@broadcast") or cid.endswith("@newsletter")
|
|
|
|
# ------------------------------------------------------------------ gating
|
|
def _open_dm_opted_in(self) -> bool:
|
|
if os.getenv("GATEWAY_ALLOW_ALL_USERS", "").lower() in {"true", "1", "yes"}:
|
|
return True
|
|
return (_get_wsecret("WHATSAPP_ALLOW_ALL_USERS", default="") or "").lower() in {"true", "1", "yes"}
|
|
|
|
@staticmethod
|
|
def _matches_whatsapp_allowlist(candidate: str, allow_from) -> bool:
|
|
"""Match a WhatsApp identifier against an allowlist across phone/LID forms.
|
|
|
|
Inbound senders arrive as ``<id>@lid`` while allowlists hold phone numbers
|
|
(or vice versa), so resolve both sides through the bridge's lid-mapping
|
|
files via ``gateway.whatsapp_identity``.
|
|
"""
|
|
if not allow_from:
|
|
return False
|
|
if candidate in allow_from:
|
|
return True
|
|
from gateway.whatsapp_identity import (
|
|
expand_whatsapp_aliases,
|
|
normalize_whatsapp_identifier,
|
|
)
|
|
candidate_aliases = expand_whatsapp_aliases(candidate)
|
|
if not candidate_aliases:
|
|
return False
|
|
for entry in allow_from:
|
|
if entry == "*":
|
|
return True
|
|
if normalize_whatsapp_identifier(entry) in candidate_aliases:
|
|
return True
|
|
if expand_whatsapp_aliases(entry) & candidate_aliases:
|
|
return True
|
|
return False
|
|
|
|
def _is_dm_allowed(self, sender_id: str) -> bool:
|
|
"""Strict DM authorization — pairing does not imply access."""
|
|
if self._dm_policy == "allowlist":
|
|
return self._matches_whatsapp_allowlist(sender_id, self._live_dm_allow_from())
|
|
if self._dm_policy == "open":
|
|
return self._open_dm_opted_in()
|
|
return False
|
|
|
|
def _is_dm_intake_allowed(self, sender_id: str) -> bool:
|
|
"""Whether a DM may reach the gateway intake (pairing handshake path)."""
|
|
principal = str(sender_id or "").strip()
|
|
if not principal:
|
|
return False
|
|
if self._dm_policy == "allowlist":
|
|
return self._matches_whatsapp_allowlist(principal, self._live_dm_allow_from())
|
|
if self._dm_policy == "pairing":
|
|
return True
|
|
if self._dm_policy == "open":
|
|
return self._open_dm_opted_in()
|
|
return False
|
|
|
|
def _is_group_allowed(self, chat_id: str) -> bool:
|
|
"""Check whether a group chat should be processed."""
|
|
if self._group_policy == "allowlist":
|
|
return self._matches_whatsapp_allowlist(chat_id, self._group_allow_from)
|
|
return self._group_policy == "open"
|
|
|
|
def _compile_mention_patterns(self):
|
|
patterns = self.config.extra.get("mention_patterns")
|
|
if patterns is None:
|
|
raw = (_get_wsecret("WHATSAPP_MENTION_PATTERNS", default="") or "").strip()
|
|
if raw:
|
|
try:
|
|
patterns = json.loads(raw)
|
|
except Exception:
|
|
patterns = [part.strip() for part in raw.splitlines() if part.strip()]
|
|
if not patterns:
|
|
patterns = [part.strip() for part in raw.split(",") if part.strip()]
|
|
if patterns is None:
|
|
return []
|
|
if isinstance(patterns, str):
|
|
patterns = [patterns]
|
|
if not isinstance(patterns, list):
|
|
logger.warning(
|
|
"[%s] whatsapp mention_patterns must be a list or string; got %s",
|
|
self.name, type(patterns).__name__,
|
|
)
|
|
return []
|
|
compiled = []
|
|
for pattern in patterns:
|
|
if not isinstance(pattern, str) or not pattern.strip():
|
|
continue
|
|
try:
|
|
compiled.append(re.compile(pattern, re.IGNORECASE))
|
|
except re.error as exc:
|
|
logger.warning("[%s] Invalid WhatsApp mention pattern %r: %s", self.name, pattern, exc)
|
|
if compiled:
|
|
logger.info("[%s] Loaded %d WhatsApp mention pattern(s)", self.name, len(compiled))
|
|
return compiled
|
|
|
|
def _bot_ids_from_message(self, data: Dict[str, Any]) -> set[str]:
|
|
bot_ids = set()
|
|
for candidate in data.get("botIds") or []:
|
|
normalized = self._normalize_whatsapp_id(candidate)
|
|
if normalized:
|
|
bot_ids.add(normalized)
|
|
return bot_ids
|
|
|
|
def _message_is_reply_to_bot(self, data: Dict[str, Any]) -> bool:
|
|
quoted_participant = self._normalize_whatsapp_id(data.get("quotedParticipant"))
|
|
if not quoted_participant:
|
|
return False
|
|
return quoted_participant in self._bot_ids_from_message(data)
|
|
|
|
def _message_mentions_bot(self, data: Dict[str, Any]) -> bool:
|
|
bot_ids = self._bot_ids_from_message(data)
|
|
if not bot_ids:
|
|
return False
|
|
mentioned_ids = {
|
|
nid
|
|
for candidate in (data.get("mentionedIds") or [])
|
|
if (nid := self._normalize_whatsapp_id(candidate))
|
|
}
|
|
if mentioned_ids & bot_ids:
|
|
return True
|
|
lower_body = str(data.get("body") or "").lower()
|
|
for bot_id in bot_ids:
|
|
bare_id = bot_id.split("@", 1)[0].lower()
|
|
if bare_id and (f"@{bare_id}" in lower_body or bare_id in lower_body):
|
|
return True
|
|
return False
|
|
|
|
def _message_matches_mention_patterns(self, data: Dict[str, Any]) -> bool:
|
|
if not self._mention_patterns:
|
|
return False
|
|
body = str(data.get("body") or "")
|
|
return any(pattern.search(body) for pattern in self._mention_patterns)
|
|
|
|
def _clean_bot_mention_text(self, text: str, data: Dict[str, Any]) -> str:
|
|
if not text:
|
|
return text
|
|
cleaned = text
|
|
for bot_id in self._bot_ids_from_message(data):
|
|
bare_id = bot_id.split("@", 1)[0]
|
|
if bare_id:
|
|
cleaned = re.sub(rf"@{re.escape(bare_id)}\b[,:\-]*\s*", "", cleaned)
|
|
return cleaned.strip() or text
|
|
|
|
def _should_process_message(self, data: Dict[str, Any]) -> bool:
|
|
chat_id = str(data.get("chatId") or "")
|
|
# Broadcast pseudo-chats are filtered even in self-chat mode (fromMe events).
|
|
if self._is_broadcast_chat(chat_id):
|
|
return False
|
|
if not data.get("isGroup", False):
|
|
# DMs that pass the policy gate are always processed
|
|
sender_id = str(data.get("senderId") or data.get("from") or "")
|
|
return self._is_dm_intake_allowed(sender_id)
|
|
if not self._is_group_allowed(chat_id):
|
|
return False
|
|
# Group messages: check mention / free-response settings
|
|
if chat_id in self._whatsapp_free_response_chats():
|
|
return True
|
|
if not self._whatsapp_require_mention():
|
|
return True
|
|
body = str(data.get("body") or "").strip()
|
|
if body.startswith("/"):
|
|
return True
|
|
if self._message_is_reply_to_bot(data):
|
|
return True
|
|
if self._message_mentions_bot(data):
|
|
return True
|
|
return self._message_matches_mention_patterns(data)
|
|
|
|
# ------------------------------------------------------------------ formatting
|
|
def format_message(self, content: str) -> str:
|
|
"""Convert markdown to WhatsApp syntax (*bold*, _italic_, ~strike~); fenced and
|
|
inline code are protected via placeholder substitution."""
|
|
if not content:
|
|
return content
|
|
content = self._sanitize_outbound_text(content)
|
|
_FENCE_PH = "\x00FENCE"
|
|
fences: list[str] = []
|
|
|
|
def _save_fence(m: re.Match) -> str:
|
|
fences.append(m.group(0))
|
|
return f"{_FENCE_PH}{len(fences) - 1}\x00"
|
|
|
|
result = re.sub(r"```[\s\S]*?```", _save_fence, content)
|
|
_CODE_PH = "\x00CODE"
|
|
codes: list[str] = []
|
|
|
|
def _save_code(m: re.Match) -> str:
|
|
codes.append(m.group(0))
|
|
return f"{_CODE_PH}{len(codes) - 1}\x00"
|
|
|
|
result = re.sub(r"`[^`\n]+`", _save_code, result)
|
|
|
|
# Italic *text* → _text_ BEFORE bold so **bold** doesn't become italic;
|
|
# lookarounds skip list bullets and bold delimiters.
|
|
result = re.sub(r"(?<!\*)\*(?!\s|\*)([^*\n]*?\S[^*\n]*?)\*(?!\*)", r"_\1_", result)
|
|
result = re.sub(r"\*\*(.+?)\*\*", r"*\1*", result)
|
|
result = re.sub(r"__(.+?)__", r"*\1*", result)
|
|
result = re.sub(r"~~(.+?)~~", r"~\1~", result)
|
|
|
|
# "# Header" → *Header*, stripping *...* already produced above so
|
|
# "# **Title**" doesn't render with literal asterisks.
|
|
def _header_to_bold(m: re.Match) -> str:
|
|
inner = m.group(1).strip()
|
|
while len(inner) > 1 and inner.startswith("*") and inner.endswith("*"):
|
|
inner = inner[1:-1].strip()
|
|
return f"*{inner}*"
|
|
|
|
result = re.sub(r"^#{1,6}\s+(.+)$", _header_to_bold, result, flags=re.MULTILINE)
|
|
# [text](url) → text (url)
|
|
result = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r"\1 (\2)", result)
|
|
for i, fence in enumerate(fences):
|
|
result = result.replace(f"{_FENCE_PH}{i}\x00", fence)
|
|
for i, code in enumerate(codes):
|
|
result = result.replace(f"{_CODE_PH}{i}\x00", code)
|
|
return result
|
|
|
|
|
|
def resolve_whatsapp_bridge_dir() -> Path:
|
|
"""Bridge directory for CLI and adapter. A read-only install tree (e.g. Docker
|
|
/opt/hermes) is mirrored to HERMES_HOME so npm install works."""
|
|
import shutil
|
|
from pathlib import Path as _Path
|
|
from hermes_constants import get_hermes_home
|
|
install_bridge = _Path(__file__).resolve().parents[2] / "scripts" / "whatsapp-bridge"
|
|
hermes_home_bridge = get_hermes_home() / "scripts" / "whatsapp-bridge"
|
|
try:
|
|
test_file = install_bridge / ".write_test"
|
|
test_file.touch()
|
|
test_file.unlink()
|
|
install_writable = True
|
|
except (OSError, PermissionError):
|
|
install_writable = False
|
|
if install_writable:
|
|
return install_bridge
|
|
if hermes_home_bridge.exists():
|
|
return hermes_home_bridge
|
|
try:
|
|
hermes_home_bridge.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.copytree(install_bridge, hermes_home_bridge, dirs_exist_ok=False)
|
|
return hermes_home_bridge
|
|
except Exception:
|
|
return install_bridge
|