diff --git a/tools/approval.py b/tools/approval.py index 87d233a0cb..a91a54aeaa 100644 --- a/tools/approval.py +++ b/tools/approval.py @@ -73,8 +73,7 @@ logger = logging.getLogger(__name__) _YOLO_MODE_FROZEN: bool = is_truthy_value(os.getenv("HERMES_YOLO_MODE", "")) -# ========================================================================= -# Per-session approval state (thread-safe) +# ========================================================================= Per-session approval state (thread-safe) # ========================================================================= _lock = threading.Lock() @@ -83,15 +82,13 @@ _session_approved: dict[str, set] = {} _session_yolo: set[str] = set() _permanent_approved: set = set() -# ========================================================================= -# Consecutive-denial circuit breaker for smart approvals -# ========================================================================= -# Each retry of a smart-denied command burns another guardian LLM call. After -# ``approvals.denial_breaker_threshold`` consecutive guardian DENY verdicts in one session -# (default 3; 0 disables) the deny message escalates to a hard-stop instruction; any approval -# resets the tally. Only TOOL RESULT text changes — no history surgery, no interrupts — so it -# is prompt-cache-invariant. Capped so short-lived session keys cannot grow it without bound; -# oldest (least recently denied) entries are evicted. +# ========================================================================= Consecutive-denial circuit breaker for +# smart approvals ========================================================================= Each retry of a +# smart-denied command burns another guardian LLM call. After ``approvals.denial_breaker_threshold`` consecutive +# guardian DENY verdicts in one session (default 3; 0 disables) the deny message escalates to a hard-stop instruction; +# any approval resets the tally. Only TOOL RESULT text changes — no history surgery, no interrupts — so it is +# prompt-cache-invariant. Capped so short-lived session keys cannot grow it without bound; oldest (least recently +# denied) entries are evicted. _denial_tally: dict[str, int] = {} _DENIAL_TALLY_MAX_SESSIONS = 256 @@ -139,9 +136,8 @@ def _denial_breaker_addendum(session_key: str) -> str: "operation. Report the blocked operation to the user and either ask them to run it manually or use /approve." ) -# ========================================================================= -# Gateway approval queue (the blocking wait loop lives in approval_gateway_wait) -# ========================================================================= +# ========================================================================= Gateway approval queue (the blocking wait +# loop lives in approval_gateway_wait) ========================================================================= _gateway_queues: dict[str, list] = {} # session_key → [_ApprovalEntry, …] @@ -286,14 +282,12 @@ def clear_session(session_key: str) -> None: _pending.pop(session_key, None) entries = _gateway_queues.pop(session_key, []) for entry in entries: - # Cancel blocked waits now so the old run unwinds instead of idling - # until timeout. + # Cancel blocked waits now so the old run unwinds instead of idling until timeout. entry.result = "deny" entry.event.set() _release_permission_mode_dependents(session_key) - # Session-persistent code kernels (local and remote) share this owner key - # and die at the same boundary so a finished conversation cannot leak a - # live interpreter. + # Session-persistent code kernels (local and remote) share this owner key and die at the same boundary so a + # finished conversation cannot leak a live interpreter. for module, shutdown in (("tools.code_kernel", "shutdown_kernels_for_owner"), ("tools.code_kernel_remote", "shutdown_remote_kernels_for_owner")): try: @@ -355,8 +349,7 @@ def _persist_choice(session_key: str, choice: str, warnings: list[tuple]) -> Non save_permanent_allowlist(_permanent_approved) -# ========================================================================= -# Config persistence for permanent allowlist +# ========================================================================= Config persistence for permanent allowlist # ========================================================================= def load_permanent_allowlist() -> set: @@ -385,8 +378,7 @@ def save_permanent_allowlist(patterns: set): logger.warning("Could not save allowlist: %s", e) -# ========================================================================= -# Bypass check (yolo / mode=off) +# ========================================================================= Bypass check (yolo / mode=off) # ========================================================================= def is_approval_bypass_active_for_session(session_key: str) -> bool: @@ -401,8 +393,7 @@ def is_approval_bypass_active() -> bool: return is_approval_bypass_active_for_session(get_current_session_key(default="")) -# ========================================================================= -# Result builders shared by the gates +# ========================================================================= Result builders shared by the gates # ========================================================================= def _approved() -> dict: @@ -468,9 +459,8 @@ def _pending_result(spec, session_key: str, *, command: str, description: str, return result -# ========================================================================= -# Unattended contexts (nobody present to answer a prompt) -# ========================================================================= +# ========================================================================= Unattended contexts (nobody present to +# answer a prompt) ========================================================================= @dataclass(frozen=True) class _Unattended: @@ -569,12 +559,10 @@ def _unattended_deny(command: str, ctx: _Unattended) -> dict | None: return None -# ========================================================================= -# Human-decision engine shared by the three gates -# ========================================================================= -# Every flagged action reaches a human the same way — selected plugin transport → gateway -# round-trip → pending fallback → CLI prompt → persist — so the consent contract (silence is -# not consent, deny is a hard halt, a smart-DENY override is one operation) cannot drift +# ========================================================================= Human-decision engine shared by the three +# gates ========================================================================= Every flagged action reaches a human +# the same way — selected plugin transport → gateway round-trip → pending fallback → CLI prompt → persist — so the +# consent contract (silence is not consent, deny is a hard halt, a smart-DENY override is one operation) cannot drift # between gates. Only wording and a few policy knobs differ per flavor; they live in _GateSpec. @dataclass(frozen=True) @@ -723,8 +711,7 @@ def _human_decision(spec: _GateSpec, *, command: str, description: str, outcome=outcome, **extra) def grant(choice: str) -> dict: - # A smart-DENY owner override is always one operation, even if an - # older client returns "session" or "always". + # A smart-DENY owner override is always one operation, even if an older client returns "session" or "always". if not smart_denied: _persist_choice(session_key, choice, warnings) if spec.user_approved: @@ -746,13 +733,11 @@ def _human_decision(spec: _GateSpec, *, command: str, description: str, return deny(spec.transport_denied, "denied") return grant(choice) - # Gateway/async approval: block the agent thread until /approve or /deny, - # mirroring the CLI's synchronous input() flow. The agent never sees - # "approval_required" here — it gets output or a definitive BLOCKED. + # Gateway/async approval: block the agent thread until /approve or /deny, mirroring the CLI's synchronous input() + # flow. The agent never sees "approval_required" here — it gets output or a definitive BLOCKED. if is_gateway or is_ask: - # Redacted copies for user-visible rendering only (the gateway paints - # them into Discord/Slack); the raw command still executes after - # approval and persistence keys off pattern_key. + # Redacted copies for user-visible rendering only (the gateway paints them into Discord/Slack); the raw + # command still executes after approval and persistence keys off pattern_key. display_command = redact_sensitive_text(command) display_description = redact_sensitive_text(description) notify_cb = _gateway_notify_cb(session_key) @@ -786,10 +771,9 @@ def _human_decision(spec: _GateSpec, *, command: str, description: str, timeout_addendum="", deny_reason=deny_reason) return grant(choice) - # No gateway callback (cron, batch, or ask-mode leaked into an - # interactive CLI, historically via `import gateway.run`): paint the - # local panel when possible instead of a pending_approval that makes - # the agent look "auto-blocked". + # No gateway callback (cron, batch, or ask-mode leaked into an interactive CLI, historically via `import + # gateway.run`): paint the local panel when possible instead of a pending_approval that makes the agent look + # "auto-blocked". if not _should_fall_through_to_cli_approval( is_cli=is_cli, approval_callback=approval_callback, notify_cb=notify_cb, ): @@ -845,8 +829,7 @@ def _run_approval_gate( caller's job. ``fail_closed_when_no_human``: a non-interactive, non-gateway, non-cron context BLOCKS instead of auto-approving, so a plugin-flagged action never runs ungated. """ - # Hardline blocks are the caller's job BEFORE this gate, so yolo here only skips the - # recoverable approval layer. + # Hardline blocks are the caller's job BEFORE this gate, so yolo here only skips the recoverable approval layer. if _yolo_active(): return _approved() session_key = get_current_session_key() @@ -992,9 +975,8 @@ def request_tool_approval(tool_name: str, reason: str, *, rule_key: str = "", ap ) -# ========================================================================= -# Combined pre-exec guard (tirith + dangerous command detection) -# ========================================================================= +# ========================================================================= Combined pre-exec guard (tirith + +# dangerous command detection) ========================================================================= def _format_tirith_description(tirith_result: dict) -> str: """Human-readable severity/title/description summary of tirith findings.""" @@ -1060,9 +1042,8 @@ def check_all_command_guards(command: str, env_type: str, return result return _approved() - # Gather findings: warnings = [(pattern_key, description, is_tirith)]. - # Tirith block AND warn both go through the approval flow (block used to - # be a hard stop) so users can inspect the findings and approve. + # Gather findings: warnings = [(pattern_key, description, is_tirith)]. Tirith block AND warn both go through the + # approval flow (block used to be a hard stop) so users can inspect the findings and approve. tirith_result = _tirith_scan(command) is_dangerous, pattern_key, description = detect_dangerous_command(command) warnings = [] @@ -1082,11 +1063,9 @@ def check_all_command_guards(command: str, env_type: str, primary_key = warnings[0][0] all_keys = [key for key, _, _ in warnings] - # "Always" is offered when at least one warning is a dangerous-pattern key - # the persistence layer would actually allowlist permanently. Pure-tirith - # findings are session-max by design, so a tirith-only prompt hides Always; - # mixed prompts offer it (the pattern key persists, tirith downgrades to - # session — see _persist_choice). + # "Always" is offered when at least one warning is a dangerous-pattern key the persistence layer would actually + # allowlist permanently. Pure-tirith findings are session-max by design, so a tirith-only prompt hides Always; + # mixed prompts offer it (the pattern key persists, tirith downgrades to session — see _persist_choice). return _human_decision( _COMMAND_GATE, command=command, description=combined_desc, pattern_key=primary_key, pattern_keys=all_keys, warnings=warnings, @@ -1115,8 +1094,7 @@ def check_execute_code_guard(code: str, env_type: str, has_host_access: bool = F pattern_key = "execute_code" description = _EXECUTE_CODE_DESCRIPTION - # Isolated backends already sandbox the child. vercel_sandbox has no - # host-bind concept so it stays always-skipped. + # Isolated backends already sandbox the child. vercel_sandbox has no host-bind concept so it stays always-skipped. if env_type == "vercel_sandbox": return _approved() if _should_skip_container_guards(env_type, has_host_access=has_host_access): @@ -1138,20 +1116,16 @@ def check_execute_code_guard(code: str, env_type: str, has_host_access: bool = F ) return _approved() - # Only gateway/ask contexts get the one-shot whole-script approval. In an - # interactive CLI the script's terminal() calls are guarded per-call - # (context propagates into the RPC thread, #33057), so a whole-script - # prompt would fire on every execute_code call. Ask-mode still takes this - # path even with INTERACTIVE set (how gateway/smart tests and messaging - # ask-mode drive whole-script approval); when that leaks into a CLI with no - # notify callback, the engine falls through to the CLI Dangerous Command - # panel instead of a silent pending_approval. + # Only gateway/ask contexts get the one-shot whole-script approval. In an interactive CLI the script's terminal() + # calls are guarded per-call (context propagates into the RPC thread, #33057), so a whole-script prompt would fire + # on every execute_code call. Ask-mode still takes this path even with INTERACTIVE set (how gateway/smart tests + # and messaging ask-mode drive whole-script approval); when that leaks into a CLI with no notify callback, the + # engine falls through to the CLI Dangerous Command panel instead of a silent pending_approval. if not is_gateway and not is_ask: return _approved() session_key = get_current_session_key() - # Built only past the early-return gates so common paths don't copy a - # potentially-large script into this string. + # Built only past the early-return gates so common paths don't copy a potentially-large script into this string. command = f"execute_code <<'PY'\n{code}\nPY" # Without this, "Approve session" / "Always" choices are stored but never @@ -1159,10 +1133,9 @@ def check_execute_code_guard(code: str, env_type: str, has_host_access: bool = F if is_approved(session_key, pattern_key): return _approved() - # Smart mode: an APPROVE only suppresses the redundant whole-script prompt; - # the per-call terminal() guards still run independently. The gateway - # renders the pending payload to Discord/Slack, so the script body is - # redacted for display; the raw code is what gets assessed and run. + # Smart mode: an APPROVE only suppresses the redundant whole-script prompt; the per-call terminal() guards still + # run independently. The gateway renders the pending payload to Discord/Slack, so the script body is redacted for + # display; the raw code is what gets assessed and run. from agent.redact import redact_sensitive_text return _human_decision( _EXECUTE_CODE_GATE, command=command, description=description, pattern_key=pattern_key, diff --git a/tools/approval_context.py b/tools/approval_context.py index 50abad0b94..a11c24b488 100644 --- a/tools/approval_context.py +++ b/tools/approval_context.py @@ -18,22 +18,19 @@ def _ctx(name: str, default: "str | None" = "") -> contextvars.ContextVar: return contextvars.ContextVar(name, default=default) -# Per-thread/per-task gateway session identity: gateway runs agent turns -# concurrently in executor threads, so a process-global env var is racy (the -# env fallback stays for legacy single-threaded callers). +# Per-thread/per-task gateway session identity: gateway runs agent turns concurrently in executor threads, so a +# process-global env var is racy (the env fallback stays for legacy single-threaded callers). _approval_session_key: contextvars.ContextVar[str] = _ctx("approval_session_key") _approval_turn_id: contextvars.ContextVar[str] = _ctx("approval_turn_id") _approval_tool_call_id: contextvars.ContextVar[str] = _ctx("approval_tool_call_id") -# Hermes session id (observability identity, distinct from the gateway routing -# session_key), forwarded to approval hooks so observer plugins attach marks to -# the REAL session scope — otherwise they fall back to a synthetic "default" +# Hermes session id (observability identity, distinct from the gateway routing session_key), forwarded to approval +# hooks so observer plugins attach marks to the REAL session scope — otherwise they fall back to a synthetic "default" # session whose scope never closes, so close-time exporters never ship them. _approval_session_id: contextvars.ContextVar[str] = _ctx("approval_session_id") -# Interactive-CLI flag. Concurrent ACP sessions share a ThreadPoolExecutor, so -# mutating os.environ["HERMES_INTERACTIVE"] races: one session's `finally` -# restore can clobber another's set mid-run, dropping it onto the -# non-interactive auto-approve path so a dangerous command runs without the -# approval callback firing (GHSA-96vc-wcxf-jjff). None = unset → env fallback. +# Interactive-CLI flag. Concurrent ACP sessions share a ThreadPoolExecutor, so mutating +# os.environ["HERMES_INTERACTIVE"] races: one session's `finally` restore can clobber another's set mid-run, dropping +# it onto the non-interactive auto-approve path so a dangerous command runs without the approval callback firing +# (GHSA-96vc-wcxf-jjff). None = unset → env fallback. _hermes_interactive_ctx: contextvars.ContextVar[str | None] = _ctx("hermes_interactive", None) @@ -132,11 +129,9 @@ def _is_cron_approval_context() -> bool: return is_truthy_value(_session_env("HERMES_CRON_SESSION")) -#: Programmatic/unattended platforms: no human can answer a prompt and the -#: adapter has no ``send_exec_approval`` / ``/approve`` surface. Governed by -#: ``approvals.unattended_mode`` (default deny), mirroring ``cron_mode`` — -#: never an interactive round-trip that blocks for the full timeout with -#: nobody to answer. +# : Programmatic/unattended platforms: no human can answer a prompt and the : adapter has no ``send_exec_approval`` / +# ``/approve`` surface. Governed by : ``approvals.unattended_mode`` (default deny), mirroring ``cron_mode`` — : never +# an interactive round-trip that blocks for the full timeout with : nobody to answer. _UNATTENDED_APPROVAL_PLATFORMS = frozenset({"webhook", "msgraph_webhook", "api_server"}) diff --git a/tools/approval_detection.py b/tools/approval_detection.py index 8ed08a8d8b..d219118f66 100644 --- a/tools/approval_detection.py +++ b/tools/approval_detection.py @@ -22,9 +22,8 @@ _SSH_SENSITIVE_PATH = r'(?:~|\$home|\$\{home\})/\.ssh(?:/|$)' _HERMES_ENV_PATH = ( r'(?:~\/\.hermes/|(?:\$home|\$\{home\})/\.hermes/|(?:\$hermes_home|\$\{hermes_home\})/)' r'\.env\b' ) -# ~/.hermes/config.yaml IS the security policy (approvals.mode, yolo, allowlist) and the config -# cache is mtime-keyed, so a write takes effect mid-session. Terminal-side coverage (sed -i, tee, -# >, cp) pairs the file_tools deny. +# ~/.hermes/config.yaml IS the security policy (approvals.mode, yolo, allowlist) and the config cache is mtime-keyed, +# so a write takes effect mid-session. Terminal-side coverage (sed -i, tee, >, cp) pairs the file_tools deny. _HERMES_CONFIG_PATH = ( r'(?:~\/\.hermes/|(?:\$home|\$\{home\})/\.hermes/|(?:\$hermes_home|\$\{hermes_home\})/)' r'config\.yaml\b' ) @@ -58,11 +57,10 @@ _WRITE_TARGET_BOUNDARY = r'(?=[\s;&|<>"\']|$)' # shutdown, DoS). Recoverable operations (git reset --hard, chmod -R 777, curl|sh) stay in # DANGEROUS_PATTERNS. -# Start-of-command position: start of string, newline, subshell opener ($( or backtick), optionally -# consuming sudo/env/exec/nohup/setsid/time wrappers. Keeps shutdown/reboot rules from firing on -# "echo reboot" / "grep 'shutdown' log". Real ;/&/| separators are converted to newlines by the -# quote-aware _mark_command_starts pass; keeping them here mistakes quoted data -# (grep '(safe|rm -rf /)') for commands. +# Start-of-command position: start of string, newline, subshell opener ($( or backtick), optionally consuming +# sudo/env/exec/nohup/setsid/time wrappers. Keeps shutdown/reboot rules from firing on "echo reboot" / "grep +# 'shutdown' log". Real ;/&/| separators are converted to newlines by the quote-aware _mark_command_starts pass; +# keeping them here mistakes quoted data (grep '(safe|rm -rf /)') for commands. _CMDPOS = ( r'(?:^|[\n`]|\$\()' r'\s*' # start position, optional whitespace r'(?:sudo\s+(?:-[^\s]+\s+)*)?' r'(?:env\s+(?:\w+=\S*\s+)*)?' # optional sudo with flags, env VAR=VAL pairs @@ -286,12 +284,11 @@ DANGEROUS_PATTERNS = [ # between `hermes` and `gateway` (`hermes -p ade gateway restart`) are allowed so a profile flag can't slip past. (r'\bhermes\s+(?:-{1,2}\S+(?:\s+\S+)?\s+)*gateway\s+(stop|restart)\b', "stop/restart hermes gateway (kills running agents)"), (r'\bhermes\s+update\b', "hermes update (restarts gateway, kills running agents)"), - # Docker/Podman daemon redirect — global flags or env that point the CLI at a DIFFERENT (often - # remote) daemon: `docker -H ssh://prod stop app` looks local but operates on remote infra, so - # any redirect requires approval regardless of subcommand. The flag must be in global position - # (before the subcommand) and -H/--host/--context must carry a value, keeping `docker -h` and - # `docker run -h ` out. Listed BEFORE the lifecycle rules so a redirected lifecycle - # command surfaces the more specific reason. + # Docker/Podman daemon redirect — global flags or env that point the CLI at a DIFFERENT (often remote) daemon: + # `docker -H ssh://prod stop app` looks local but operates on remote infra, so any redirect requires approval + # regardless of subcommand. The flag must be in global position (before the subcommand) and -H/--host/--context + # must carry a value, keeping `docker -h` and `docker run -h ` out. Listed BEFORE the lifecycle rules so + # a redirected lifecycle command surfaces the more specific reason. (r'\bdocker\s+(?:-{1,2}\S+(?:[=\s]\S+)?\s+)*(?:-h|--host)[=\s]+\S+', "docker with remote daemon redirect (-H/--host)"), (r'\bdocker\s+(?:-{1,2}\S+(?:[=\s]\S+)?\s+)*(?:-c|--context)[=\s]+\S+', "docker with daemon redirect (--context: alternate daemon)"), (r'\bdocker\s+context\s+use\b', "docker context use (switches default daemon for future commands)"), @@ -312,11 +309,10 @@ DANGEROUS_PATTERNS = [ # pattern above, so catch the structural form. (r'\bkill\b.*\$\(\s*(pgrep|pidof)\b', "kill process via pgrep/pidof expansion (self-termination)"), (r'\bkill\b.*`\s*(pgrep|pidof)\b', "kill process via backtick pgrep/pidof expansion (self-termination)"), - # launchctl-driven gateway stop/restart on macOS (label `ai.hermes.gateway`). Two independent - # lookaheads, NOT a sequential match: a for-loop building the label from a list defined EARLIER - # (`for item in 'ai.hermes...'; do launchctl bootout "$label"`) never has "hermes" after the - # verb, and that slipped past and restarted 4 gateways with zero approval. Erring broad is - # correct for an approval gate: an extra prompt is cheap. + # launchctl-driven gateway stop/restart on macOS (label `ai.hermes.gateway`). Two independent lookaheads, NOT a + # sequential match: a for-loop building the label from a list defined EARLIER (`for item in 'ai.hermes...'; do + # launchctl bootout "$label"`) never has "hermes" after the verb, and that slipped past and restarted 4 gateways + # with zero approval. Erring broad is correct for an approval gate: an extra prompt is cheap. (r'(?=[\s\S]*\blaunchctl\s+(?:stop|kickstart|bootout|unload|kill|disable|remove)\b)(?=[\s\S]*\b(?:hermes|ai\.hermes)\b)', "stop/restart hermes launchd service (kills running agents)"), (rf'\b(cp|mv|install)\b.*\s{_SYSTEM_CONFIG_PATH}', "copy/move file into system config path"), (rf'\b(cp|mv|install)\b.*\s["\']?{_PROJECT_SENSITIVE_WRITE_TARGET}["\']?{_COMMAND_TAIL}', "overwrite project env/config file"), @@ -336,9 +332,8 @@ DANGEROUS_PATTERNS = [ # write_file/patch deny so the terminal side is not an open door. (rf'\bsed\s+-[^\s]*i.*(?:{_HERMES_CONFIG_PATH}|{_HERMES_ENV_PATH})', "in-place edit of Hermes config/env"), (rf'\bsed\s+--in-place\b.*(?:{_HERMES_CONFIG_PATH}|{_HERMES_ENV_PATH})', "in-place edit of Hermes config/env (long flag)"), - # perl/ruby -i: the flag may be its own token after other flags (`-p -i -e`), combined (`-pi`), - # or carry a backup suffix (`-i.bak`), so match any flag token containing `i` anywhere; - # `perl -e '...'` (no -i) does not trip. + # perl/ruby -i: the flag may be its own token after other flags (`-p -i -e`), combined (`-pi`), or carry a backup + # suffix (`-i.bak`), so match any flag token containing `i` anywhere; `perl -e '...'` (no -i) does not trip. (rf'\b(?:perl|ruby)\b.*(?:^|\s)-[^\s]*i\b.*(?:{_HERMES_CONFIG_PATH}|{_HERMES_ENV_PATH})', "in-place edit of Hermes config/env (perl/ruby)"), # Interpreter heredocs are handled by _execution_flag_findings(); only shell heredocs stay # regex-based. `bash <<'EOF'` runs arbitrary commands without triggering the `bash -c` path. @@ -351,19 +346,17 @@ DANGEROUS_PATTERNS = [ (r'\bgit\s+push\b.*-f\b', "git force push short flag (rewrites remote history)"), (r'\bgit\s+clean\s+-[^\s]*f', "git clean with force (deletes untracked files)"), (r'\bgit\s+branch\s+-D\b', "git branch force delete"), - # `-D` = `-d --force`; the long spellings are different tokens, so match delete+force in either - # order, bounded to one command segment (no `;`/`|`/`&`/newline) so an unrelated later command - # isn't contaminated. + # `-D` = `-d --force`; the long spellings are different tokens, so match delete+force in either order, bounded to + # one command segment (no `;`/`|`/`&`/newline) so an unrelated later command isn't contaminated. (r'\bgit\s+branch\b[^;|&\n]*?(?:-d\b|--delete\b)[^;|&\n]*?(?:-f\b|--force\b)', "git branch force delete (long flags)"), (r'\bgit\s+branch\b[^;|&\n]*?(?:-f\b|--force\b)[^;|&\n]*?(?:-d\b|--delete\b)', "git branch force delete (long flags, force-first)"), # chmod +x then immediate run: the script content may hold dangerous commands individual patterns miss. (r'\bchmod\s+\+x\b.*[;&|]+\s*\./', "chmod +x followed by immediate execution"), - # Sudo stdin/askpass/shell/list-privs flags. The agent has no TTY, so sudo invocations that - # succeed non-interactively read the password from stdin (-S) or askpass (-A); -s (shell) and - # -a (list) are gated as privilege chains (read SUDO_PASSWORD from .env -> sudo -S -s). Plain - # `sudo cmd` is TTY-bound and excluded. Input is lowercased, so S/s and A/a collapse. Lazy - # `[^;|&\n]*?` allows flag args without spanning separators. sudo resolves unambiguous - # long-flag prefixes: `--stdin` is the only long option starting with "st", `--askpass` the + # Sudo stdin/askpass/shell/list-privs flags. The agent has no TTY, so sudo invocations that succeed + # non-interactively read the password from stdin (-S) or askpass (-A); -s (shell) and -a (list) are gated as + # privilege chains (read SUDO_PASSWORD from .env -> sudo -S -s). Plain `sudo cmd` is TTY-bound and excluded. Input + # is lowercased, so S/s and A/a collapse. Lazy `[^;|&\n]*?` allows flag args without spanning separators. sudo + # resolves unambiguous long-flag prefixes: `--stdin` is the only long option starting with "st", `--askpass` the # only one starting with "a". (r'\bsudo\b[^;|&\n]*?\s+(?:-s\b|--st[a-z]*\b|-a\b|--a[a-z]*\b)', "sudo with privilege flag (stdin/askpass/shell/list)"), # Combined short-flag form (-nS, -sa, -las). @@ -404,19 +397,17 @@ def _normalize_command_for_detection(command: str) -> str: # precede the generic escape strip below, whose [^\n] class skips newlines and would leave the # backslash wedged between tokens, defeating the structured rm/mkfs/dd patterns incl. the HARDLINE floor. command = re.sub(r'\\\r?\n', '', command) - # Fold absolute user/Hermes home prefixes to ~/ and ~/.hermes/ so the static patterns catch - # /home/alice/.bashrc and C:\Users\alice\.bashrc. Resolved at detection time (not import time) - # so it tracks HOME/HERMES_HOME set later. MUST run before the backslash strip (which would - # dissolve C:\Users\alice to C:Usersalice). Hermes home first: on Windows it nests under the - # user home, and folding the user home first would eat the prefix it needs. + # Fold absolute user/Hermes home prefixes to ~/ and ~/.hermes/ so the static patterns catch /home/alice/.bashrc + # and C:\Users\alice\.bashrc. Resolved at detection time (not import time) so it tracks HOME/HERMES_HOME set + # later. MUST run before the backslash strip (which would dissolve C:\Users\alice to C:Usersalice). Hermes home + # first: on Windows it nests under the user home, and folding the user home first would eat the prefix it needs. command = _rewrite_resolved_hermes_home(command) command = _rewrite_resolved_user_home(command) # Strip backslash-escapes (r\m -> rm) and empty-string literals (r''m -> rm). command = re.sub(r'\\([^\n])', r'\1', command) command = re.sub(r"''|\"\"", '', command) - # Collapse $IFS / ${IFS...} (incl. `${IFS:0:1}`) to a space: IFS defaults to whitespace, so - # `rm${IFS}-rf${IFS}/` runs as `rm -rf /`, and every pattern — incl. the hardline floor — - # anchors on literal \s between tokens. + # Collapse $IFS / ${IFS...} (incl. `${IFS:0:1}`) to a space: IFS defaults to whitespace, so `rm${IFS}-rf${IFS}/` + # runs as `rm -rf /`, and every pattern — incl. the hardline floor — anchors on literal \s between tokens. return re.sub(r'\$\{IFS\b[^}]*\}|\$IFS\b', ' ', command) @@ -501,9 +492,8 @@ _READ_TOOL_EXEC_FLAGS = { "sort": {"--compress-program"}, "rg": {"--pre", "--hostname-bin"}, "ag": {"--pager"}, "man": {"--pager", "--html", "-P", "-H"}, } -# Required-argument options are ownership boundaries: an option-looking next token is data, not -# another option. These sets mirror the invocation grammar of the supported binaries (ripgrep 14, -# GNU sort, man-db, and ag 2.2). +# Required-argument options are ownership boundaries: an option-looking next token is data, not another option. These +# sets mirror the invocation grammar of the supported binaries (ripgrep 14, GNU sort, man-db, and ag 2.2). _READ_TOOL_LONG_OPTIONS_WITH_ARG = { "rg": { "--after-context", "--before-context", "--color", "--colors", "--context", "--context-separator", @@ -715,9 +705,8 @@ def _interpreter_exec_flag(family: str, args: list[str]) -> str | None: comparable = option.lower() if powershell else option if comparable in flags: return comparable - # `-Wonce` and `ruby -rjson` attach an option value; they are not short-option bundles - # containing an execution flag. PowerShell's normal long options also use one dash, so - # bundle parsing never applies to that family. + # `-Wonce` and `ruby -rjson` attach an option value; they are not short-option bundles containing an execution + # flag. PowerShell's normal long options also use one dash, so bundle parsing never applies to that family. has_attached_option_value = any( option.startswith(short) and len(option) > len(short) for short in with_arg if short.startswith("-") and not short.startswith("--") @@ -1049,18 +1038,16 @@ def _command_detection_variants(command: str): seen.add(variant) return True - # Windows-path variant: normalization strips backslashes as shell escapes, so `del - # C:\Users\me\.ssh\id_rsa` reaches the patterns as `del C:Usersme.sshid_rsa`. When the RAW - # command has a drive-letter or UNC backslash path, also yield a variant with backslashes - # flattened to `/` BEFORE normalization. Gated on a real path shape so POSIX escape semantics - # (`echo a\"b`) are untouched elsewhere. + # Windows-path variant: normalization strips backslashes as shell escapes, so `del C:\Users\me\.ssh\id_rsa` + # reaches the patterns as `del C:Usersme.sshid_rsa`. When the RAW command has a drive-letter or UNC backslash + # path, also yield a variant with backslashes flattened to `/` BEFORE normalization. Gated on a real path shape so + # POSIX escape semantics (`echo a\"b`) are untouched elsewhere. if re.search(r"(?:[A-Za-z]:|\\\\)[\\\\]", command) or re.search(r"[A-Za-z]:\\", command): win_variant = _normalize_command_for_detection(_mask_quoted_newlines(command.replace("\\", "/"))) if fresh(win_variant): yield win_variant - # Program-bearing options are parsed in their owning command's context; surfacing only the - # payload lets the hardline floor inspect what will actually run without promoting similar - # flags or quoted prose. + # Program-bearing options are parsed in their owning command's context; surfacing only the payload lets the + # hardline floor inspect what will actually run without promoting similar flags or quoted prose. pending = [normalized] while pending: for _, payload in _execution_flag_findings(pending.pop()): @@ -1072,10 +1059,9 @@ def _command_detection_variants(command: str): if marked_payload != payload and fresh(marked_payload): yield marked_payload pending.append(payload) - # Subshell `(cmd)` / brace-group `{ cmd; }` openers put `cmd` at a real command position the - # flat `_CMDPOS` patterns can't see (adding `(`/`{` there would match quoted prose like - # `--title "(reboot)"`). Insert a newline at each start the QUOTE-AWARE tokenizer found - # instead; this covers every `_CMDPOS` rule in one place. + # Subshell `(cmd)` / brace-group `{ cmd; }` openers put `cmd` at a real command position the flat `_CMDPOS` + # patterns can't see (adding `(`/`{` there would match quoted prose like `--title "(reboot)"`). Insert a newline + # at each start the QUOTE-AWARE tokenizer found instead; this covers every `_CMDPOS` rule in one place. marked = _mark_command_starts(grep_safe) if marked != grep_safe and fresh(marked): yield marked diff --git a/tools/approval_floors.py b/tools/approval_floors.py index b99565df4e..684df0790f 100644 --- a/tools/approval_floors.py +++ b/tools/approval_floors.py @@ -102,9 +102,8 @@ def _hardline_block_result(description: str, command: str = "") -> dict: "need to run it, run it yourself in a terminal outside the " "agent." ) - # The parser-limit block is almost always a giant inline payload, not a - # forbidden operation, and is typically followed by blind rephrase retries — - # point at the saved script (or the write_file recipe). + # The parser-limit block is almost always a giant inline payload, not a forbidden operation, and is typically + # followed by blind rephrase retries — point at the saved script (or the write_file recipe). if description in (_PARSER_LIMIT_DESCRIPTION, _MALFORMED_EXEC_DESCRIPTION): saved = _a._save_blocked_payload(command) if command else None if saved: @@ -132,11 +131,10 @@ def _sudo_stdin_block_result(description: str) -> dict: "manually in your own terminal.")} -# Shell control characters that make a command compound when they appear OUTSIDE -# quotes. Inside quotes they are literal to the outer shell — but they become -# executable again if an option like `-c`/`-e`/`--eval` (or a git `-c alias.x=!...`) -# hands the quoted argument to another interpreter, so quoted control chars only -# disqualify a command when such an option is present. +# Shell control characters that make a command compound when they appear OUTSIDE quotes. Inside quotes they are +# literal to the outer shell — but they become executable again if an option like `-c`/`-e`/`--eval` (or a git `-c +# alias.x=!...`) hands the quoted argument to another interpreter, so quoted control chars only disqualify a command +# when such an option is present. _SHELL_CONTROL_CHARS = frozenset("\n\r;&|<>`$()") _REINTERPRETED_ARGUMENT_RE = re.compile(r"(?:^|[ \t])(?:-[^-\s]*[ce]|--(?:command|eval))(?:[= \t]|$)") diff --git a/tools/approval_human_wait.py b/tools/approval_human_wait.py index 5526ce3742..d4573370e4 100644 --- a/tools/approval_human_wait.py +++ b/tools/approval_human_wait.py @@ -29,10 +29,9 @@ class _HumanWaitState: _human_wait_lock = threading.Lock() _human_wait_states: dict[str, _HumanWaitState] = {} _HUMAN_WAIT_MAX_SESSIONS = 256 -# Margin added on top of approvals.timeout when clamping a window's contribution -# (read-side AND close-side) and when bounding the authorization gate's -# serialization-lock acquire in agent/tool_executor.py. One constant so the -# clamps can't drift apart. +# Margin added on top of approvals.timeout when clamping a window's contribution (read-side AND close-side) and when +# bounding the authorization gate's serialization-lock acquire in agent/tool_executor.py. One constant so the clamps +# can't drift apart. HUMAN_WAIT_MARGIN_S = 60.0 @@ -139,8 +138,7 @@ def human_wait_seconds(session_key: str | None = None) -> float: wedged-window hang).""" key = _resolve_key(session_key) now = time.monotonic() - # Resolve the clamp outside the lock: it reads the config cache, which must - # never nest under _human_wait_lock. + # Resolve the clamp outside the lock: it reads the config cache, which must never nest under _human_wait_lock. ceiling = human_wait_ceiling() with _human_wait_lock: state = _human_wait_states.get(key) diff --git a/tools/approval_prompt.py b/tools/approval_prompt.py index 8ed6f62389..184eeedd00 100644 --- a/tools/approval_prompt.py +++ b/tools/approval_prompt.py @@ -38,9 +38,8 @@ def prompt_dangerous_approval(command: str, description: str, timeout_seconds: i from tools import approval as _a if timeout_seconds is None: timeout_seconds = _a._get_approval_timeout() - # Everything below is a human prompt (callback panel or input() fallback, both - # bounded by the approval deadline): record it as human-wait time so the - # concurrent batch deadline excludes it. + # Everything below is a human prompt (callback panel or input() fallback, both bounded by the approval deadline): + # record it as human-wait time so the concurrent batch deadline excludes it. with human_wait_window(): return _ask_human(command, description, timeout_seconds, allow_permanent, approval_callback, allow_session, smart_denied) @@ -78,9 +77,8 @@ def _read_choice(prompt: str, timeout_seconds: int) -> str | None: def _ask_human(command: str, description: str, timeout_seconds: int, allow_permanent: bool, approval_callback, allow_session: bool, smart_denied: bool) -> str: - # Redact before any user-visible rendering; the original `command` still - # executes after approval. Same redactor as memory/log sanitization so - # tokens mask consistently across surfaces. + # Redact before any user-visible rendering; the original `command` still executes after approval. Same redactor as + # memory/log sanitization so tokens mask consistently across surfaces. from agent.redact import redact_sensitive_text display_command = redact_sensitive_text(command) display_description = redact_sensitive_text(description) @@ -98,11 +96,10 @@ def _ask_human(command: str, description: str, timeout_seconds: int, allow_perma logger.error("Approval callback failed: %s", e, exc_info=True) return "deny" - # Fail-closed guard: when prompt_toolkit owns the terminal and no callback is - # registered on this thread, the input() fallback would spawn a daemon thread - # whose read never sees Enter (keystrokes go to prompt_toolkit) — an invisible - # deadlock. Deny loudly instead; threads needing interactive approval must - # install a callback via tools.terminal_tool.set_approval_callback() first. + # Fail-closed guard: when prompt_toolkit owns the terminal and no callback is registered on this thread, the + # input() fallback would spawn a daemon thread whose read never sees Enter (keystrokes go to prompt_toolkit) — an + # invisible deadlock. Deny loudly instead; threads needing interactive approval must install a callback via + # tools.terminal_tool.set_approval_callback() first. try: from prompt_toolkit.application.current import get_app_or_none if get_app_or_none() is not None: @@ -148,9 +145,8 @@ def _ask_human(command: str, description: str, timeout_seconds: int, allow_perma def get_plugin_manager(): """Lazy plugin-manager seam used by tests and early tool-only imports.""" from hermes_cli.plugins import discover_plugins, get_plugin_manager as _get_manager - # Approval can be imported before model_tools (which triggers discovery); make - # an explicitly selected transport available on the first approval instead of - # treating the undiscovered registry as unavailable. + # Approval can be imported before model_tools (which triggers discovery); make an explicitly selected transport + # available on the first approval instead of treating the undiscovered registry as unavailable. discover_plugins() return _get_manager() diff --git a/tools/approval_smart.py b/tools/approval_smart.py index dc27e6c816..9d10dcd932 100644 --- a/tools/approval_smart.py +++ b/tools/approval_smart.py @@ -79,9 +79,8 @@ def _smart_approve(command: str, description: str) -> str: try: from agent.auxiliary_client import _get_task_timeout, call_llm - # Pass the timeout explicitly AND log call + duration: this synchronous call - # gates EVERY flagged command, and a stalled provider once froze turns for - # tens of minutes with zero log output. + # Pass the timeout explicitly AND log call + duration: this synchronous call gates EVERY flagged command, and + # a stalled provider once froze turns for tens of minutes with zero log output. smart_timeout = _get_task_timeout("approval") logger.debug("Smart approvals: assessing risk for command (timeout=%ss)", smart_timeout) system_prompt = _SYSTEM_PROMPT