"""Shared threat-pattern library for context window security scanning. Single source of truth for prompt-injection / promptware / exfiltration patterns used by ``agent/prompt_builder.py``, ``tools/memory_tool.py`` and ``agent/tool_dispatch_helpers.py``. Each pattern is a ``(regex, pattern_id, scope)`` tuple. Scope controls which scanners use it: ``"all"`` everywhere; ``"context"`` adds promptware / C2 / role hijack for context files, memory and tool results (warn-level, since tool results contain content the user did not author); ``"strict"`` adds aggressive checks only for user-mediated writes (memory, skill installs) where blocking can be resolved interactively. New patterns must anchor on C2-specific vocabulary or unambiguous attack behavior, NOT bossy English ("you must", "you are obligated to" are common in legitimate AGENTS.md / CLAUDE.md content). Filler between key tokens is the bounded ``_FILLER`` — unbounded ``(?:\\w+\\s+)*`` backtracks badly. """ from __future__ import annotations import re import unicodedata from typing import List, Optional, Tuple # Hard cap on scanned text: scanners are advisory guards, and bounding input # keeps worst-case runtime predictable while catching injection near the start. MAX_SCAN_CHARS = 65_536 # Bounded filler between key attack words (up to eight words of obfuscation). _FILLER = r"(?:\w+\s+){0,8}" # Env var reference ending in a secret-ish suffix (see exfil comment below). _SECRET_VAR = r"\$\{?\w*(?:KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL)S?\b" # Verb prefix for "modify agent config" patterns. _MODIFY = r"(update|modify|edit|write|change|append|add\s+to)\s+[^\n]{0,2048}" # Each entry: (regex, pattern_id, scope) # scope ∈ {"all", "context", "strict"} _PATTERNS: List[Tuple[str, str, str]] = [ # ── Classic prompt injection (applies everywhere) ──────────────── (rf'ignore\s+{_FILLER}(previous|all|above|prior)\s+{_FILLER}instructions', "prompt_injection", "all"), (r'system\s+prompt\s+override', "sys_prompt_override", "all"), (rf'disregard\s+{_FILLER}(your|all|any)\s+{_FILLER}(instructions|rules|guidelines)', "disregard_rules", "all"), (rf'act\s+as\s+(if|though)\s+{_FILLER}you\s+{_FILLER}(have\s+no|don\'t\s+have)\s+{_FILLER}(restrictions|limits|rules)', "bypass_restrictions", "all"), (r'', "html_comment_injection", "all"), (r'<\s*div\s+style\s*=\s*["\'][^>]{0,2048}display\s*:\s*none', "hidden_div", "all"), (r'translate\s+[^\n]{0,512}\s+into\s+[^\n]{0,512}\s+and\s+(execute|run|eval)', "translate_execute", "all"), (rf'do\s+not\s+{_FILLER}tell\s+{_FILLER}the\s+user', "deception_hide", "all"), # ── Role-play / identity hijack (context + strict; common attack # surface in scraped web content and poisoned context files) ── (rf'you\s+are\s+{_FILLER}now\s+(?:a|an|the)\s+', "role_hijack", "context"), (rf'pretend\s+{_FILLER}(you\s+are|to\s+be)\s+', "role_pretend", "context"), (rf'output\s+{_FILLER}(system|initial)\s+prompt', "leak_system_prompt", "context"), (rf'(respond|answer|reply)\s+without\s+{_FILLER}(restrictions|limitations|filters|safety)', "remove_filters", "context"), (rf'you\s+have\s+been\s+{_FILLER}(updated|upgraded|patched)\s+to', "fake_update", "context"), # "name yourself X" is a Brainworm-specific tell — identity override # via spec instead of jailbreak. Anchored on the verb pair so it # doesn't match "name your variables" etc. (r'\bname\s+yourself\s+\w+', "identity_override", "context"), # ── C2 / Brainworm-style promptware (context scope) ────────────── # These anchor on C2-specific vocabulary. "register as a node" appears # in legitimate distributed-systems docs, but in combination with the # other patterns the signal is strong; we WARN, not block, so a security # researcher reading the Brainworm post in a webpage doesn't break their # session. (r'register\s+(as\s+)?a?\s*node', "c2_node_registration", "context"), (r'(heartbeat|beacon|check[\s\-]?in)\s+(to|with)\s+', "c2_heartbeat", "context"), (r'pull\s+(down\s+)?(?:new\s+)?task(?:ing|s)?\b', "c2_task_pull", "context"), (r'connect\s+to\s+the\s+network\b', "c2_network_connect", "context"), # Verb-anchored "you must register/connect/report/beacon" — the verbs # are C2-specific so this avoids the broader "you must X" false positive. (r'you\s+must\s+(?:\w+\s+){0,3}(register|connect|report|beacon)\b', "forced_action", "context"), # Anti-forensic instructions ("never write to disk", "one-liners only") # — extremely unusual in legitimate content; near-zero false positive. (r'only\s+use\s+one[\s\-]?liners?\b', "anti_forensic_oneliner", "context"), (rf'never\s+{_FILLER}(?:create|write)\s+{_FILLER}(?:script|file)\s+{_FILLER}disk', "anti_forensic_disk", "context"), # Environment-variable unsetting targeting known agent runtimes — # this is pure attack behavior (Brainworm sub-session bypass). (r'unset\s+\w*(?:CLAUDE|CODEX|HERMES|AGENT|OPENAI|ANTHROPIC)\w*', "env_var_unset_agent", "context"), # ── Known C2 / red-team framework names (near-zero false positive # outside security research; warn-only by default) ───────────── # NOTE: do not add common English words here. Every token must be a # distinctive offensive-security tool brand, otherwise legitimate # AGENTS.md / SOUL.md content false-positives and the whole file is # blocked. "praxis" was removed for exactly this reason — it's a common # word and a legitimate agent name (Greek for practice/action), not a # C2-specific tell like the brands below. (r'\b(?:cobalt\s*strike|sliver|havoc|mythic|metasploit|brainworm)\b', "known_c2_framework", "context"), (r'\bc2\s+(?:server|channel|infrastructure|beacon)\b', "c2_explicit", "context"), (r'\bcommand\s+and\s+control\b', "c2_explicit_long", "context"), # ── Exfiltration via curl/wget/cat with secrets (applies everywhere) ── # Anchor env var name end with \b to avoid false positives on legitimate # env vars like $TRILLIUM_ETAPI_URL that contain KEY/TOKEN/API as # substrings. API is dropped from the alternation outright: mid-name API # is ubiquitous in benign var names, and every real secret shape it # caught ($OPENAI_API_KEY) already ends in KEY/TOKEN. (rf'curl\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_curl", "all"), (rf'wget\s+[^\n]{{0,2048}}{_SECRET_VAR}', "exfil_wget", "all"), (r'cat\s+[^\n]{0,2048}(\.env|credentials|\.netrc|\.pgpass|\.npmrc|\.pypirc)', "read_secrets", "all"), (r'(send|post|upload|transmit)\s+[^\n]{0,2048}\s+(to|at)\s+https?://', "send_to_url", "strict"), (rf'(include|output|print|share)\s+{_FILLER}(conversation|chat\s+history|previous\s+messages|full\s+context|entire\s+context)', "context_exfil", "strict"), # ── Persistence / SSH backdoor (strict scope — memory + skills) ── (r'authorized_keys', "ssh_backdoor", "strict"), (r'\$HOME/\.ssh|\~/\.ssh', "ssh_access", "strict"), (r'\$HOME/\.hermes/\.env|\~/\.hermes/\.env', "hermes_env", "strict"), (rf'{_MODIFY}(?:AGENTS\.md|CLAUDE\.md|\.cursorrules|\.clinerules)', "agent_config_mod", "strict"), (rf'{_MODIFY}\.hermes/(config\.yaml|SOUL\.md)', "hermes_config_mod", "strict"), # ── Hardcoded secrets ──────────────────────────────────────────── (r'(?:api[_-]?key|token|secret|password)\s*[=:]\s*["\'][A-Za-z0-9+/=_-]{20,}', "hardcoded_secret", "strict"), ] # Invisible / bidirectional unicode characters used in injection attacks. # Aligned with skills_guard.py INVISIBLE_CHARS — directional isolates # (U+2066-U+2069) and invisible math operators (U+2062-U+2064) are real # attack tools. INVISIBLE_CHARS = frozenset({ '\u200b', # zero-width space '\u200c', # zero-width non-joiner '\u200d', # zero-width joiner '\u2060', # word joiner '\u2062', # invisible times '\u2063', # invisible separator '\u2064', # invisible plus '\ufeff', # zero-width no-break space (BOM) '\u202a', # left-to-right embedding '\u202b', # right-to-left embedding '\u202c', # pop directional formatting '\u202d', # left-to-right override '\u202e', # right-to-left override '\u2066', # left-to-right isolate '\u2067', # right-to-left isolate '\u2068', # first strong isolate '\u2069', # pop directional isolate }) # Compiled pattern sets by scope, built once at import. Scope inclusion is # cumulative: "all" patterns land in every set, "context" in context + strict, # "strict" in strict only. _SCOPE_SETS = {"all": ("all", "context", "strict"), "context": ("context", "strict"), "strict": ("strict",)} def _compile() -> dict[str, List[Tuple[re.Pattern, str]]]: compiled: dict[str, List[Tuple[re.Pattern, str]]] = {"all": [], "context": [], "strict": []} for pattern, pid, scope in _PATTERNS: if scope not in _SCOPE_SETS: raise ValueError(f"threat_patterns: unknown scope {scope!r} for pattern {pid!r}") entry = (re.compile(pattern, re.IGNORECASE), pid) for s in _SCOPE_SETS[scope]: compiled[s].append(entry) return compiled _COMPILED = _compile() def scan_for_threats(content: str, scope: str = "context") -> List[str]: """Return matched pattern IDs in ``content`` for ``scope`` (see module docstring). Invisible unicode characters are reported as ``"invisible_unicode_U+XXXX"`` so callers can surface the offending codepoint. """ if not content: return [] content = content[:MAX_SCAN_CHARS] # Invisible unicode is checked on the RAW content: NFKC normalisation below # can strip some of these codepoints. findings: List[str] = [f"invisible_unicode_U+{ord(ch):04X}" for ch in set(content) & INVISIBLE_CHARS] # NFKC folds full-width / compatibility variants (cat → cat) so homograph # substitution can't bypass keyword checks. It does NOT fold cross-script # confusables (Cyrillic ``а`` U+0430) — that would need a TR#39 database. normalised = unicodedata.normalize("NFKC", content) patterns = _COMPILED.get(scope) if patterns is None: raise ValueError(f"scan_for_threats: unknown scope {scope!r}") findings.extend(pid for compiled, pid in patterns if compiled.search(normalised)) return findings def first_threat_message(content: str, scope: str = "strict") -> Optional[str]: """Return a user-facing error for the first threat found, or None (block-on-first-hit paths).""" findings = scan_for_threats(content, scope=scope) if not findings: return None pid = findings[0] if pid.startswith("invisible_unicode_"): codepoint = pid.replace("invisible_unicode_", "") return f"Blocked: content contains invisible unicode character {codepoint} (possible injection)." return ( f"Blocked: content matches threat pattern '{pid}'. " f"Content is injected into the system prompt and must not contain " f"injection or exfiltration payloads." ) __all__ = [ "INVISIBLE_CHARS", "MAX_SCAN_CHARS", "scan_for_threats", "first_threat_message", ]